diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index 5dd3d4f0..371895cc 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -4,7 +4,8 @@ on: push: branches: [main, "spike/**"] pull_request: - branches: [main] + # feat/** too: fix PRs stacked on a feature branch (e.g. PRs into #189's branch) need CI for the red/green cycle. + branches: [main, "feat/**"] # CI runs three sequential jobs: lint → fast tests → integration tests. # diff --git a/.mcp.json b/.mcp.json index 8737c844..2c07df3d 100644 --- a/.mcp.json +++ b/.mcp.json @@ -5,8 +5,7 @@ "args": ["run", "tallyman", "mcp"], "cwd": "/Users/paddy/tallyman", "env": { - "TALLYMAN_PROJECT": "spike", - "TALLYMAN_COMPANION_URL": "http://127.0.0.1:7860" + "TALLYMAN_PROJECT": "spike" } } } diff --git a/LICENSE b/LICENSE new file mode 100644 index 00000000..be3f7b28 --- /dev/null +++ b/LICENSE @@ -0,0 +1,661 @@ + GNU AFFERO GENERAL PUBLIC LICENSE + Version 3, 19 November 2007 + + Copyright (C) 2007 Free Software Foundation, Inc. + Everyone is permitted to copy and distribute verbatim copies + of this license document, but changing it is not allowed. + + Preamble + + The GNU Affero General Public License is a free, copyleft license for +software and other kinds of works, specifically designed to ensure +cooperation with the community in the case of network server software. + + The licenses for most software and other practical works are designed +to take away your freedom to share and change the works. By contrast, +our General Public Licenses are intended to guarantee your freedom to +share and change all versions of a program--to make sure it remains free +software for all its users. + + When we speak of free software, we are referring to freedom, not +price. Our General Public Licenses are designed to make sure that you +have the freedom to distribute copies of free software (and charge for +them if you wish), that you receive source code or can get it if you +want it, that you can change the software or use pieces of it in new +free programs, and that you know you can do these things. + + Developers that use our General Public Licenses protect your rights +with two steps: (1) assert copyright on the software, and (2) offer +you this License which gives you legal permission to copy, distribute +and/or modify the software. + + A secondary benefit of defending all users' freedom is that +improvements made in alternate versions of the program, if they +receive widespread use, become available for other developers to +incorporate. Many developers of free software are heartened and +encouraged by the resulting cooperation. However, in the case of +software used on network servers, this result may fail to come about. +The GNU General Public License permits making a modified version and +letting the public access it on a server without ever releasing its +source code to the public. + + The GNU Affero General Public License is designed specifically to +ensure that, in such cases, the modified source code becomes available +to the community. It requires the operator of a network server to +provide the source code of the modified version running there to the +users of that server. Therefore, public use of a modified version, on +a publicly accessible server, gives the public access to the source +code of the modified version. + + An older license, called the Affero General Public License and +published by Affero, was designed to accomplish similar goals. This is +a different license, not a version of the Affero GPL, but Affero has +released a new version of the Affero GPL which permits relicensing under +this license. + + The precise terms and conditions for copying, distribution and +modification follow. + + TERMS AND CONDITIONS + + 0. Definitions. + + "This License" refers to version 3 of the GNU Affero General Public License. + + "Copyright" also means copyright-like laws that apply to other kinds of +works, such as semiconductor masks. + + "The Program" refers to any copyrightable work licensed under this +License. Each licensee is addressed as "you". "Licensees" and +"recipients" may be individuals or organizations. + + To "modify" a work means to copy from or adapt all or part of the work +in a fashion requiring copyright permission, other than the making of an +exact copy. The resulting work is called a "modified version" of the +earlier work or a work "based on" the earlier work. + + A "covered work" means either the unmodified Program or a work based +on the Program. + + To "propagate" a work means to do anything with it that, without +permission, would make you directly or secondarily liable for +infringement under applicable copyright law, except executing it on a +computer or modifying a private copy. Propagation includes copying, +distribution (with or without modification), making available to the +public, and in some countries other activities as well. + + To "convey" a work means any kind of propagation that enables other +parties to make or receive copies. Mere interaction with a user through +a computer network, with no transfer of a copy, is not conveying. + + An interactive user interface displays "Appropriate Legal Notices" +to the extent that it includes a convenient and prominently visible +feature that (1) displays an appropriate copyright notice, and (2) +tells the user that there is no warranty for the work (except to the +extent that warranties are provided), that licensees may convey the +work under this License, and how to view a copy of this License. If +the interface presents a list of user commands or options, such as a +menu, a prominent item in the list meets this criterion. + + 1. Source Code. + + The "source code" for a work means the preferred form of the work +for making modifications to it. "Object code" means any non-source +form of a work. + + A "Standard Interface" means an interface that either is an official +standard defined by a recognized standards body, or, in the case of +interfaces specified for a particular programming language, one that +is widely used among developers working in that language. + + The "System Libraries" of an executable work include anything, other +than the work as a whole, that (a) is included in the normal form of +packaging a Major Component, but which is not part of that Major +Component, and (b) serves only to enable use of the work with that +Major Component, or to implement a Standard Interface for which an +implementation is available to the public in source code form. A +"Major Component", in this context, means a major essential component +(kernel, window system, and so on) of the specific operating system +(if any) on which the executable work runs, or a compiler used to +produce the work, or an object code interpreter used to run it. + + The "Corresponding Source" for a work in object code form means all +the source code needed to generate, install, and (for an executable +work) run the object code and to modify the work, including scripts to +control those activities. However, it does not include the work's +System Libraries, or general-purpose tools or generally available free +programs which are used unmodified in performing those activities but +which are not part of the work. For example, Corresponding Source +includes interface definition files associated with source files for +the work, and the source code for shared libraries and dynamically +linked subprograms that the work is specifically designed to require, +such as by intimate data communication or control flow between those +subprograms and other parts of the work. + + The Corresponding Source need not include anything that users +can regenerate automatically from other parts of the Corresponding +Source. + + The Corresponding Source for a work in source code form is that +same work. + + 2. Basic Permissions. + + All rights granted under this License are granted for the term of +copyright on the Program, and are irrevocable provided the stated +conditions are met. This License explicitly affirms your unlimited +permission to run the unmodified Program. The output from running a +covered work is covered by this License only if the output, given its +content, constitutes a covered work. This License acknowledges your +rights of fair use or other equivalent, as provided by copyright law. + + You may make, run and propagate covered works that you do not +convey, without conditions so long as your license otherwise remains +in force. You may convey covered works to others for the sole purpose +of having them make modifications exclusively for you, or provide you +with facilities for running those works, provided that you comply with +the terms of this License in conveying all material for which you do +not control copyright. Those thus making or running the covered works +for you must do so exclusively on your behalf, under your direction +and control, on terms that prohibit them from making any copies of +your copyrighted material outside their relationship with you. + + Conveying under any other circumstances is permitted solely under +the conditions stated below. Sublicensing is not allowed; section 10 +makes it unnecessary. + + 3. Protecting Users' Legal Rights From Anti-Circumvention Law. + + No covered work shall be deemed part of an effective technological +measure under any applicable law fulfilling obligations under article +11 of the WIPO copyright treaty adopted on 20 December 1996, or +similar laws prohibiting or restricting circumvention of such +measures. + + When you convey a covered work, you waive any legal power to forbid +circumvention of technological measures to the extent such circumvention +is effected by exercising rights under this License with respect to +the covered work, and you disclaim any intention to limit operation or +modification of the work as a means of enforcing, against the work's +users, your or third parties' legal rights to forbid circumvention of +technological measures. + + 4. Conveying Verbatim Copies. + + You may convey verbatim copies of the Program's source code as you +receive it, in any medium, provided that you conspicuously and +appropriately publish on each copy an appropriate copyright notice; +keep intact all notices stating that this License and any +non-permissive terms added in accord with section 7 apply to the code; +keep intact all notices of the absence of any warranty; and give all +recipients a copy of this License along with the Program. + + You may charge any price or no price for each copy that you convey, +and you may offer support or warranty protection for a fee. + + 5. Conveying Modified Source Versions. + + You may convey a work based on the Program, or the modifications to +produce it from the Program, in the form of source code under the +terms of section 4, provided that you also meet all of these conditions: + + a) The work must carry prominent notices stating that you modified + it, and giving a relevant date. + + b) The work must carry prominent notices stating that it is + released under this License and any conditions added under section + 7. This requirement modifies the requirement in section 4 to + "keep intact all notices". + + c) You must license the entire work, as a whole, under this + License to anyone who comes into possession of a copy. This + License will therefore apply, along with any applicable section 7 + additional terms, to the whole of the work, and all its parts, + regardless of how they are packaged. This License gives no + permission to license the work in any other way, but it does not + invalidate such permission if you have separately received it. + + d) If the work has interactive user interfaces, each must display + Appropriate Legal Notices; however, if the Program has interactive + interfaces that do not display Appropriate Legal Notices, your + work need not make them do so. + + A compilation of a covered work with other separate and independent +works, which are not by their nature extensions of the covered work, +and which are not combined with it such as to form a larger program, +in or on a volume of a storage or distribution medium, is called an +"aggregate" if the compilation and its resulting copyright are not +used to limit the access or legal rights of the compilation's users +beyond what the individual works permit. Inclusion of a covered work +in an aggregate does not cause this License to apply to the other +parts of the aggregate. + + 6. Conveying Non-Source Forms. + + You may convey a covered work in object code form under the terms +of sections 4 and 5, provided that you also convey the +machine-readable Corresponding Source under the terms of this License, +in one of these ways: + + a) Convey the object code in, or embodied in, a physical product + (including a physical distribution medium), accompanied by the + Corresponding Source fixed on a durable physical medium + customarily used for software interchange. + + b) Convey the object code in, or embodied in, a physical product + (including a physical distribution medium), accompanied by a + written offer, valid for at least three years and valid for as + long as you offer spare parts or customer support for that product + model, to give anyone who possesses the object code either (1) a + copy of the Corresponding Source for all the software in the + product that is covered by this License, on a durable physical + medium customarily used for software interchange, for a price no + more than your reasonable cost of physically performing this + conveying of source, or (2) access to copy the + Corresponding Source from a network server at no charge. + + c) Convey individual copies of the object code with a copy of the + written offer to provide the Corresponding Source. This + alternative is allowed only occasionally and noncommercially, and + only if you received the object code with such an offer, in accord + with subsection 6b. + + d) Convey the object code by offering access from a designated + place (gratis or for a charge), and offer equivalent access to the + Corresponding Source in the same way through the same place at no + further charge. You need not require recipients to copy the + Corresponding Source along with the object code. If the place to + copy the object code is a network server, the Corresponding Source + may be on a different server (operated by you or a third party) + that supports equivalent copying facilities, provided you maintain + clear directions next to the object code saying where to find the + Corresponding Source. Regardless of what server hosts the + Corresponding Source, you remain obligated to ensure that it is + available for as long as needed to satisfy these requirements. + + e) Convey the object code using peer-to-peer transmission, provided + you inform other peers where the object code and Corresponding + Source of the work are being offered to the general public at no + charge under subsection 6d. + + A separable portion of the object code, whose source code is excluded +from the Corresponding Source as a System Library, need not be +included in conveying the object code work. + + A "User Product" is either (1) a "consumer product", which means any +tangible personal property which is normally used for personal, family, +or household purposes, or (2) anything designed or sold for incorporation +into a dwelling. In determining whether a product is a consumer product, +doubtful cases shall be resolved in favor of coverage. For a particular +product received by a particular user, "normally used" refers to a +typical or common use of that class of product, regardless of the status +of the particular user or of the way in which the particular user +actually uses, or expects or is expected to use, the product. A product +is a consumer product regardless of whether the product has substantial +commercial, industrial or non-consumer uses, unless such uses represent +the only significant mode of use of the product. + + "Installation Information" for a User Product means any methods, +procedures, authorization keys, or other information required to install +and execute modified versions of a covered work in that User Product from +a modified version of its Corresponding Source. The information must +suffice to ensure that the continued functioning of the modified object +code is in no case prevented or interfered with solely because +modification has been made. + + If you convey an object code work under this section in, or with, or +specifically for use in, a User Product, and the conveying occurs as +part of a transaction in which the right of possession and use of the +User Product is transferred to the recipient in perpetuity or for a +fixed term (regardless of how the transaction is characterized), the +Corresponding Source conveyed under this section must be accompanied +by the Installation Information. But this requirement does not apply +if neither you nor any third party retains the ability to install +modified object code on the User Product (for example, the work has +been installed in ROM). + + The requirement to provide Installation Information does not include a +requirement to continue to provide support service, warranty, or updates +for a work that has been modified or installed by the recipient, or for +the User Product in which it has been modified or installed. Access to a +network may be denied when the modification itself materially and +adversely affects the operation of the network or violates the rules and +protocols for communication across the network. + + Corresponding Source conveyed, and Installation Information provided, +in accord with this section must be in a format that is publicly +documented (and with an implementation available to the public in +source code form), and must require no special password or key for +unpacking, reading or copying. + + 7. Additional Terms. + + "Additional permissions" are terms that supplement the terms of this +License by making exceptions from one or more of its conditions. +Additional permissions that are applicable to the entire Program shall +be treated as though they were included in this License, to the extent +that they are valid under applicable law. If additional permissions +apply only to part of the Program, that part may be used separately +under those permissions, but the entire Program remains governed by +this License without regard to the additional permissions. + + When you convey a copy of a covered work, you may at your option +remove any additional permissions from that copy, or from any part of +it. (Additional permissions may be written to require their own +removal in certain cases when you modify the work.) You may place +additional permissions on material, added by you to a covered work, +for which you have or can give appropriate copyright permission. + + Notwithstanding any other provision of this License, for material you +add to a covered work, you may (if authorized by the copyright holders of +that material) supplement the terms of this License with terms: + + a) Disclaiming warranty or limiting liability differently from the + terms of sections 15 and 16 of this License; or + + b) Requiring preservation of specified reasonable legal notices or + author attributions in that material or in the Appropriate Legal + Notices displayed by works containing it; or + + c) Prohibiting misrepresentation of the origin of that material, or + requiring that modified versions of such material be marked in + reasonable ways as different from the original version; or + + d) Limiting the use for publicity purposes of names of licensors or + authors of the material; or + + e) Declining to grant rights under trademark law for use of some + trade names, trademarks, or service marks; or + + f) Requiring indemnification of licensors and authors of that + material by anyone who conveys the material (or modified versions of + it) with contractual assumptions of liability to the recipient, for + any liability that these contractual assumptions directly impose on + those licensors and authors. + + All other non-permissive additional terms are considered "further +restrictions" within the meaning of section 10. If the Program as you +received it, or any part of it, contains a notice stating that it is +governed by this License along with a term that is a further +restriction, you may remove that term. If a license document contains +a further restriction but permits relicensing or conveying under this +License, you may add to a covered work material governed by the terms +of that license document, provided that the further restriction does +not survive such relicensing or conveying. + + If you add terms to a covered work in accord with this section, you +must place, in the relevant source files, a statement of the +additional terms that apply to those files, or a notice indicating +where to find the applicable terms. + + Additional terms, permissive or non-permissive, may be stated in the +form of a separately written license, or stated as exceptions; +the above requirements apply either way. + + 8. Termination. + + You may not propagate or modify a covered work except as expressly +provided under this License. Any attempt otherwise to propagate or +modify it is void, and will automatically terminate your rights under +this License (including any patent licenses granted under the third +paragraph of section 11). + + However, if you cease all violation of this License, then your +license from a particular copyright holder is reinstated (a) +provisionally, unless and until the copyright holder explicitly and +finally terminates your license, and (b) permanently, if the copyright +holder fails to notify you of the violation by some reasonable means +prior to 60 days after the cessation. + + Moreover, your license from a particular copyright holder is +reinstated permanently if the copyright holder notifies you of the +violation by some reasonable means, this is the first time you have +received notice of violation of this License (for any work) from that +copyright holder, and you cure the violation prior to 30 days after +your receipt of the notice. + + Termination of your rights under this section does not terminate the +licenses of parties who have received copies or rights from you under +this License. If your rights have been terminated and not permanently +reinstated, you do not qualify to receive new licenses for the same +material under section 10. + + 9. Acceptance Not Required for Having Copies. + + You are not required to accept this License in order to receive or +run a copy of the Program. Ancillary propagation of a covered work +occurring solely as a consequence of using peer-to-peer transmission +to receive a copy likewise does not require acceptance. However, +nothing other than this License grants you permission to propagate or +modify any covered work. These actions infringe copyright if you do +not accept this License. Therefore, by modifying or propagating a +covered work, you indicate your acceptance of this License to do so. + + 10. Automatic Licensing of Downstream Recipients. + + Each time you convey a covered work, the recipient automatically +receives a license from the original licensors, to run, modify and +propagate that work, subject to this License. You are not responsible +for enforcing compliance by third parties with this License. + + An "entity transaction" is a transaction transferring control of an +organization, or substantially all assets of one, or subdividing an +organization, or merging organizations. If propagation of a covered +work results from an entity transaction, each party to that +transaction who receives a copy of the work also receives whatever +licenses to the work the party's predecessor in interest had or could +give under the previous paragraph, plus a right to possession of the +Corresponding Source of the work from the predecessor in interest, if +the predecessor has it or can get it with reasonable efforts. + + You may not impose any further restrictions on the exercise of the +rights granted or affirmed under this License. For example, you may +not impose a license fee, royalty, or other charge for exercise of +rights granted under this License, and you may not initiate litigation +(including a cross-claim or counterclaim in a lawsuit) alleging that +any patent claim is infringed by making, using, selling, offering for +sale, or importing the Program or any portion of it. + + 11. Patents. + + A "contributor" is a copyright holder who authorizes use under this +License of the Program or a work on which the Program is based. The +work thus licensed is called the contributor's "contributor version". + + A contributor's "essential patent claims" are all patent claims +owned or controlled by the contributor, whether already acquired or +hereafter acquired, that would be infringed by some manner, permitted +by this License, of making, using, or selling its contributor version, +but do not include claims that would be infringed only as a +consequence of further modification of the contributor version. For +purposes of this definition, "control" includes the right to grant +patent sublicenses in a manner consistent with the requirements of +this License. + + Each contributor grants you a non-exclusive, worldwide, royalty-free +patent license under the contributor's essential patent claims, to +make, use, sell, offer for sale, import and otherwise run, modify and +propagate the contents of its contributor version. + + In the following three paragraphs, a "patent license" is any express +agreement or commitment, however denominated, not to enforce a patent +(such as an express permission to practice a patent or covenant not to +sue for patent infringement). To "grant" such a patent license to a +party means to make such an agreement or commitment not to enforce a +patent against the party. + + If you convey a covered work, knowingly relying on a patent license, +and the Corresponding Source of the work is not available for anyone +to copy, free of charge and under the terms of this License, through a +publicly available network server or other readily accessible means, +then you must either (1) cause the Corresponding Source to be so +available, or (2) arrange to deprive yourself of the benefit of the +patent license for this particular work, or (3) arrange, in a manner +consistent with the requirements of this License, to extend the patent +license to downstream recipients. "Knowingly relying" means you have +actual knowledge that, but for the patent license, your conveying the +covered work in a country, or your recipient's use of the covered work +in a country, would infringe one or more identifiable patents in that +country that you have reason to believe are valid. + + If, pursuant to or in connection with a single transaction or +arrangement, you convey, or propagate by procuring conveyance of, a +covered work, and grant a patent license to some of the parties +receiving the covered work authorizing them to use, propagate, modify +or convey a specific copy of the covered work, then the patent license +you grant is automatically extended to all recipients of the covered +work and works based on it. + + A patent license is "discriminatory" if it does not include within +the scope of its coverage, prohibits the exercise of, or is +conditioned on the non-exercise of one or more of the rights that are +specifically granted under this License. You may not convey a covered +work if you are a party to an arrangement with a third party that is +in the business of distributing software, under which you make payment +to the third party based on the extent of your activity of conveying +the work, and under which the third party grants, to any of the +parties who would receive the covered work from you, a discriminatory +patent license (a) in connection with copies of the covered work +conveyed by you (or copies made from those copies), or (b) primarily +for and in connection with specific products or compilations that +contain the covered work, unless you entered into that arrangement, +or that patent license was granted, prior to 28 March 2007. + + Nothing in this License shall be construed as excluding or limiting +any implied license or other defenses to infringement that may +otherwise be available to you under applicable patent law. + + 12. No Surrender of Others' Freedom. + + If conditions are imposed on you (whether by court order, agreement or +otherwise) that contradict the conditions of this License, they do not +excuse you from the conditions of this License. If you cannot convey a +covered work so as to satisfy simultaneously your obligations under this +License and any other pertinent obligations, then as a consequence you may +not convey it at all. For example, if you agree to terms that obligate you +to collect a royalty for further conveying from those to whom you convey +the Program, the only way you could satisfy both those terms and this +License would be to refrain entirely from conveying the Program. + + 13. Remote Network Interaction; Use with the GNU General Public License. + + Notwithstanding any other provision of this License, if you modify the +Program, your modified version must prominently offer all users +interacting with it remotely through a computer network (if your version +supports such interaction) an opportunity to receive the Corresponding +Source of your version by providing access to the Corresponding Source +from a network server at no charge, through some standard or customary +means of facilitating copying of software. This Corresponding Source +shall include the Corresponding Source for any work covered by version 3 +of the GNU General Public License that is incorporated pursuant to the +following paragraph. + + Notwithstanding any other provision of this License, you have +permission to link or combine any covered work with a work licensed +under version 3 of the GNU General Public License into a single +combined work, and to convey the resulting work. The terms of this +License will continue to apply to the part which is the covered work, +but the work with which it is combined will remain governed by version +3 of the GNU General Public License. + + 14. Revised Versions of this License. + + The Free Software Foundation may publish revised and/or new versions of +the GNU Affero General Public License from time to time. Such new versions +will be similar in spirit to the present version, but may differ in detail to +address new problems or concerns. + + Each version is given a distinguishing version number. If the +Program specifies that a certain numbered version of the GNU Affero General +Public License "or any later version" applies to it, you have the +option of following the terms and conditions either of that numbered +version or of any later version published by the Free Software +Foundation. If the Program does not specify a version number of the +GNU Affero General Public License, you may choose any version ever published +by the Free Software Foundation. + + If the Program specifies that a proxy can decide which future +versions of the GNU Affero General Public License can be used, that proxy's +public statement of acceptance of a version permanently authorizes you +to choose that version for the Program. + + Later license versions may give you additional or different +permissions. However, no additional obligations are imposed on any +author or copyright holder as a result of your choosing to follow a +later version. + + 15. Disclaimer of Warranty. + + THERE IS NO WARRANTY FOR THE PROGRAM, TO THE EXTENT PERMITTED BY +APPLICABLE LAW. EXCEPT WHEN OTHERWISE STATED IN WRITING THE COPYRIGHT +HOLDERS AND/OR OTHER PARTIES PROVIDE THE PROGRAM "AS IS" WITHOUT WARRANTY +OF ANY KIND, EITHER EXPRESSED OR IMPLIED, INCLUDING, BUT NOT LIMITED TO, +THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR +PURPOSE. THE ENTIRE RISK AS TO THE QUALITY AND PERFORMANCE OF THE PROGRAM +IS WITH YOU. SHOULD THE PROGRAM PROVE DEFECTIVE, YOU ASSUME THE COST OF +ALL NECESSARY SERVICING, REPAIR OR CORRECTION. + + 16. Limitation of Liability. + + IN NO EVENT UNLESS REQUIRED BY APPLICABLE LAW OR AGREED TO IN WRITING +WILL ANY COPYRIGHT HOLDER, OR ANY OTHER PARTY WHO MODIFIES AND/OR CONVEYS +THE PROGRAM AS PERMITTED ABOVE, BE LIABLE TO YOU FOR DAMAGES, INCLUDING ANY +GENERAL, SPECIAL, INCIDENTAL OR CONSEQUENTIAL DAMAGES ARISING OUT OF THE +USE OR INABILITY TO USE THE PROGRAM (INCLUDING BUT NOT LIMITED TO LOSS OF +DATA OR DATA BEING RENDERED INACCURATE OR LOSSES SUSTAINED BY YOU OR THIRD +PARTIES OR A FAILURE OF THE PROGRAM TO OPERATE WITH ANY OTHER PROGRAMS), +EVEN IF SUCH HOLDER OR OTHER PARTY HAS BEEN ADVISED OF THE POSSIBILITY OF +SUCH DAMAGES. + + 17. Interpretation of Sections 15 and 16. + + If the disclaimer of warranty and limitation of liability provided +above cannot be given local legal effect according to their terms, +reviewing courts shall apply local law that most closely approximates +an absolute waiver of all civil liability in connection with the +Program, unless a warranty or assumption of liability accompanies a +copy of the Program in return for a fee. + + END OF TERMS AND CONDITIONS + + How to Apply These Terms to Your New Programs + + If you develop a new program, and you want it to be of the greatest +possible use to the public, the best way to achieve this is to make it +free software which everyone can redistribute and change under these terms. + + To do so, attach the following notices to the program. It is safest +to attach them to the start of each source file to most effectively +state the exclusion of warranty; and each file should have at least +the "copyright" line and a pointer to where the full notice is found. + + + Copyright (C) + + This program is free software: you can redistribute it and/or modify + it under the terms of the GNU Affero General Public License as published by + the Free Software Foundation, either version 3 of the License, or + (at your option) any later version. + + This program is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU Affero General Public License for more details. + + You should have received a copy of the GNU Affero General Public License + along with this program. If not, see . + +Also add information on how to contact you by electronic and paper mail. + + If your software can interact with users remotely through a computer +network, you should also make sure that it provides a way for users to +get its source. For example, if your program is a web application, its +interface could display a "Source" link that leads users to an archive +of the code. There are many ways you could offer source, and different +solutions will be better for different programs; see section 13 for the +specific requirements. + + You should also get your employer (if you work as a programmer) or school, +if any, to sign a "copyright disclaimer" for the program, if necessary. +For more information on this, and how to apply and follow the GNU AGPL, see +. diff --git a/README.md b/README.md index a35d0512..21f06f7c 100644 --- a/README.md +++ b/README.md @@ -4,7 +4,7 @@ Tallyman — a data science environment designed for coding agents. Stop squinti Two windows: Claude Code in one, a browser in the other. You work by defining a set of named results that can depend on each other, then making sure those results live up to their name. You describe them; the agent writes the queries. You tell it what `likely_customers` should mean, and a moment later the answer is a table in the browser at full size, sortable and searchable, with statistics over every column. Your prompt sits above the query the agent wrote, so you read your intent, the code that came out of it, and the result together. You notice it's catching people who already churned, so you sharpen the sentence and the agent revises the query. Everything built on top of that name updates to match, fast enough that you don't lose your place. How big the data is, what has already been computed, and what needs recomputing never enter into it. -Those queries are expressions: self-contained programs that declare the raw files and other results they read. Writing the expression is the agent's entire job. An expression is declarative and can be introspected, so tallyman reads its dependencies straight off it and builds a directed graph of the project's aliased expressions. When a parent updates, its children are recomputed; a raw file changing on disk counts as an update too. Tallyman executes each expression once and writes the result and its summary statistics to disk. Reading it back is out of core: scrolling, sorting, and searching pull only the pieces of data needed to fill the screen, so nothing has to fit in memory and four million rows opens like four thousand. +Those queries are expressions: self-contained programs that name, by alias, the imported data and the other results they read. Writing the expression is the agent's entire job. An expression is declarative and can be introspected, so tallyman reads its dependencies straight off it and builds a directed graph of the project's aliased expressions. When a parent updates, its children are recomputed; importing a new version of a data file counts as an update too. Tallyman executes each expression once and writes the result and its summary statistics to disk. Reading it back is out of core: scrolling, sorting, and searching pull only the pieces of data needed to fill the screen, so nothing has to fit in memory and four million rows opens like four thousand. You aren't typing `df.sort_values('lifetime_sales')` just to see which customers are at the top, you aren't waiting for the LLM to print a 10 row table like it's coming out of a 1200 baud modem. You aren't running out of memory in the middle of a session. You aren't building the ad-hoc cache that every long notebook grows, the pickle in /tmp behind an if not exists guard that you never quite trust. You aren't nursing a kernel along for days because one cell takes five minutes to rerun, or bracing yourself before you close the window. None of that is in your head while you work. What's in your head is the data and what it means. @@ -50,58 +50,77 @@ index of all the docs — start with [docs/architecture.md](docs/architecture.md ## V0 scope -End-to-end: a Claude Code MCP tool that compiles a xorq expression, materializes -a result to a content-hashed catalog entry on disk, and pushes a live update -to a browser companion via SSE. +End-to-end: a Claude Code MCP tool that compiles a xorq expression into a +content-hashed catalog entry on disk (running it, and writing its result to a +file when the entry does expensive work), and pushes a live update to a browser +companion via SSE. What's working: -- **MCP tools** (FastMCP over stdio): - - Catalog: `catalog_run`, `catalog_load_parquet`, `catalog_create`, +- **MCP tools** (FastMCP over stdio, 31 tools): + - Catalog: `catalog_import_source`, `catalog_run`, `catalog_create`, `catalog_revise`, `catalog_alias`, `catalog_rename`, `catalog_unalias`, - `catalog_list`, `catalog_diff`, `catalog_chart`, `catalog_recalc`, plus the - summary-stat / post-processing / display-klass authoring tools. + `catalog_list`, `catalog_diff`, `catalog_promote_diff`, `catalog_chart`, + `catalog_chart_errors`, `catalog_scan_staleness`, `catalog_recalc`, + `catalog_export_marimo`, plus the summary-stat / post-processing / + display-klass authoring tools. - Notebook: `notebook_reorder`, `notebook_remove`, `notebook_edit_markdown`. - Project: `project_list`, `project_new`, `project_switch`. - See [docs/architecture.md](docs/architecture.md) for the full tool surface. + See [docs/mcp-server.md](docs/mcp-server.md) for every tool and its side + effects. - **Companion** (FastAPI on `:7860`) — serves the React SPA (`packages/app/dist`) as a catch-all and exposes a JSON API + SSE under `/{project}/api/*`: - - SPA tabs: **Catalog** (entry list + detail with V_n chips and forensic - history), **Notebook** (curated narrative anchored on aliases, drag-reorder, - inline markdown editor, × remove), **Diff** (code diff, schema diff, - per-column stats, key-joined side-by-side, head() side-by-side), **Cache** - (per-entry cache footprint), and **Log** (linear, filterable activity view). - - JSON: `/{project}/api/{entries,entry/,aliases,notebook,errors,log, - data/,diff_data/...,disk_usage,result_cache,staleness}`, plus the - mutation routes (`PATCH notebook`, `PUT code/`, + - SPA pages: **Catalog** (entry list + detail with V_n chips, forensic + history, and a metadata tab with the entry's disk footprint, sources, + parents and children), **Notebook** (curated narrative anchored on aliases, + drag-reorder, inline markdown editor, × remove), **Diff** (code diff, schema + diff, per-column stats, key-joined side-by-side, head() side-by-side, and a + promote button), **Cache** (the result snapshots on disk, with a delete + button; pinned snapshots cannot be deleted, and a snapshot whose entry a + reset retired is labelled as such), **Log** (linear, filterable + activity view), and the project list. + - JSON: `/{project}/api/{entries,entry/,entry_cache/, + session/,aliases,notebook,notebook_full,errors,error/,log, + data/,diff_data/...,disk_usage,result_cache,staleness,telemetry}`, + plus the mutation routes (`PATCH notebook`, `PUT code/`, `PUT markdown/`, `POST reset`, `POST recalc`, - `POST promote_diff/...`). - - `/{project}/api/sse` — live updates (`new_entry`, `build_failed`, - `alias_changed`, `notebook_changed`, `recalc`, `summary_stat_changed`). + `POST promote_diff/...`, `DELETE result_cache/`, `DELETE errors`). + - `/{project}/api/sse` — live updates. The SPA listens for `new_entry`, + `build_failed`, `notebook_changed`, `chart_attached`, + `post_processing_changed`, `summary_stat_changed`, `recalc` and + `project_switched`. - `/internal/notify` — the MCP server's notification hook; fans out to SSE. - **Buckaroo subprocess** — `tallyman run` spawns `python -m buckaroo.server` on `:8700` (falls back to a random port if busy), watches for the - `BUCKAROO_PORT=...` handshake, and lazily creates per-entry sessions on - first view by POSTing the entry's `xorq_build/` dir to Buckaroo's - `/load_expr` endpoint (PR 776) — sort/search push down to the xorq - backend rather than paging over a materialised parquet. The build dir is + `BUCKAROO_PORT=...` handshake, and opens a session for an entry by POSTing a + build dir to Buckaroo's `/load_expr` endpoint (PR 776), after making sure + every file the entry reads exists. It does this when an entry's catalog page + opens, and for every cell each time the notebook page loads. A *worthy* + entry (one whose query tallyman materialized to a result file when the entry + was created) is handed a view build, a build that is one read of that file + (`.xorq_view_build/`). A *cheap* entry (a filter, selection or computed column + over one file, which keeps no file of its own) is handed its own build, expanded into a stable per-entry path (`.xorq_build_expanded/`, gated by a `.complete` marker) so `${TALLYMAN_PROJECT_ROOT}` placeholders are resolved - before xorq's loader sees them. Sessions are persisted in a global - `~/.tallyman-notebooks/buckaroo_sessions.json` (keyed by content hash, shared - across projects) and invalidated by start-time when Buckaroo restarts. - Tear-down rides along with the companion. Disable with `--no-buckaroo`. + before xorq's loader sees them. Buckaroo's sorting, search and summary stats + run as queries over that build. A session's id is derived from the project and + the content hash, so tallyman keeps no session file. Tear-down rides along with + the companion. Disable with `--no-buckaroo`. - **Build artifacts are portable.** xorq's absolute filesystem paths are - rewritten to `${TALLYMAN_PROJECT_ROOT}` on write and expanded back on load. + rewritten to `${TALLYMAN_PROJECT_ROOT}` on write and expanded back on load + (with one known gap for copied projects, #209). - **`tallyman serve `** — read-only companion against a project - directory that may live anywhere on disk. Mutation routes return 403. + directory that may live anywhere on disk. Mutation routes return 403, and no + Buckaroo subprocess runs, so entry grids do not load. What's NOT yet implemented: 1. Column-level lineage (xorq has the data; there is no lineage view today). -2. ML training pipeline (storyboard beats 7-8). +2. A dedicated ML training tool, `catalog_train` (storyboard beats 7-8, #2). + Models can already be fitted as catalog entries with `xorq.ml`, as + `catalog_run`'s tool description shows. ## Running the spike @@ -125,7 +144,8 @@ In another terminal, launch Claude Code from this directory; it picks up Recommended prompts: -> Use catalog_load_parquet to load `orders.parquet`. +> Use catalog_import_source to import +> `~/.tallyman-notebooks/projects/spike/data/orders.parquet` as `orders`. > > Now use catalog_create to make a named entry `shoe_sales` that groups orders > by region and totals the price. @@ -147,22 +167,36 @@ tar xzf my-project.tgz -C ~/projects/ uv run tallyman serve ~/projects/spike ``` -The companion runs read-only: same catalog, same forensic history, no edit -affordances. Mutation routes return 403. +The companion runs read-only: same catalog, same forensic history, same charts. +The edit controls still show, but mutation routes return 403, and there is no +Buckaroo grid. +The archive includes `compute_cache/`, and a known defect (#209) makes a copied +project's cheap entries read from the original location; see +[docs/installing.md](docs/installing.md#sharing-a-project). ## Conventions worth knowing -- xorq 0.3.x reads use `xo.deferred_read_parquet` (NOT `xo.read_parquet` — that - resolves through ibis's backend loader and fails). Use - `import xorq.api as xo` and `import xorq.vendor.ibis as ibis`. Do NOT - `import ibis` directly. -- Prefer `from tallyman_xorq.io import read_project_file; t = read_project_file("name.parquet")` - over absolute paths — the catalog records project-relative intent and the - build is portable across machines/users. +- A file enters the catalog only through `catalog_import_source(path, alias)`, + which copies its bytes into the project and makes each version of it an entry + under a **source alias**; the path can be anywhere and is never read again. + Import it again to bring in new data: different bytes mint the next version, + and the entries downstream are recalculated. A CSV takes its `schema` and + reader options in the import call, and they are fixed there. +- Recipes read entries by alias, never files: + `tracked_expr_from_alias("orders")` follows an alias and + `pinned_expr_from_alias("orders-v2")` pins one version. `read_project_file`, + `tallyman_read_csv` and `xo.deferred_read_csv` are build errors that name the + import to use, and so is `xo.deferred_read_parquet` of any file outside the + project's `compute_cache/`. `xo.read_parquet` resolves through ibis's backend + loader and fails. Use `import xorq.api as xo` and `import xorq.vendor.ibis as + ibis`. Do NOT `import ibis` directly. - Content hash is xorq's build hash — same code + same inputs → same hash → same - entry dir (idempotent). -- All catalog state lives on disk. The MCP server holds no in-memory state; the - companion only holds the SSE subscriber list. + entry dir (idempotent). A source entry's hash is instead an md5 of the + imported bytes and the reader options. +- All catalog state lives on disk. The MCP server and the companion keep only + in-memory caches of things that never change (loaded builds and reads, keyed + by content hash), plus the MCP session's active project and the companion's + SSE subscribers and diff sessions. - `TALLYMAN_PROJECT_PATH` overrides project_dir() resolution for the active project. Used by `tallyman serve` to point at a project directory anywhere on disk. @@ -172,3 +206,7 @@ affordances. Mutation routes return 403. uv run pytest # full suite uv run pytest tests/test_pack.py # the pack / portability proof ``` + +## License + +Tallyman is licensed under the GNU Affero General Public License, version 3 only (`AGPL-3.0-only`). The full text is in [LICENSE](LICENSE). diff --git a/demo/datasets.md b/demo/datasets.md index b43e9fba..ae6bcbf3 100644 --- a/demo/datasets.md +++ b/demo/datasets.md @@ -97,19 +97,19 @@ df['tip_pct'] = (df['tip_amount'] / df['fare_amount'].replace(0, float('nan'))) Type these into Claude Code with the tallyman MCP server running: ``` -Use catalog_load_parquet to load /tmp/nyc_taxi/yellow_tripdata_2024-01.parquet, -name it yellow_jan_2024. Prompt: "NYC yellow cab trips January 2024". +Use catalog_import_source to import /tmp/nyc_taxi/yellow_tripdata_2024-01.parquet +under the alias yellow_jan_2024. Prompt: "NYC yellow cab trips January 2024". ``` ``` -Create a named entry trips_by_zone that joins yellow_jan_2024 with the taxi zone -lookup at /tmp/nyc_taxi/taxi_zone_lookup.csv on PULocationID, groups by Zone, -and counts trips with mean fare. Name it trips_by_zone. +Import /tmp/nyc_taxi/taxi_zone_lookup.csv under the alias taxi_zones, then create +a named entry trips_by_zone that joins yellow_jan_2024 with taxi_zones on +PULocationID, groups by Zone, and counts trips with mean fare. ``` ``` -Use catalog_load_parquet to load /tmp/citibike/citibike_h1_2024.parquet, -name it citibike_h1. Prompt: "Citibike trips January–June 2024". +Use catalog_import_source to import /tmp/citibike/citibike_h1_2024.parquet +under the alias citibike_h1. Prompt: "Citibike trips January–June 2024". ``` ``` diff --git a/demo/storyboard.json b/demo/storyboard.json index 256712e0..a8083638 100644 --- a/demo/storyboard.json +++ b/demo/storyboard.json @@ -2,9 +2,13 @@ "project": "spike", "steps": [ { - "tool": "catalog_load_parquet", - "args": {"rel_path": "orders.parquet", "name": "orders", "prompt": "raw orders dataset"}, - "narration": "beat 1: load the orders parquet and name it" + "tool": "catalog_import_source", + "args": { + "outside_path": "${TALLYMAN_PROJECT_ROOT}/data/orders.parquet", + "alias": "orders", + "prompt": "raw orders dataset" + }, + "narration": "beat 1: import the orders parquet under a source alias" }, { "tool": "catalog_create", @@ -21,7 +25,7 @@ "code": "from tallyman_xorq.io import tracked_expr_from_alias\nt = tracked_expr_from_alias('shoe_sales')\nexpr = t.order_by('total')\n", "prompt": "verify shoe_sales totals are reasonable" }, - "narration": "beat 4: scratch verification — sort the result" + "narration": "beat 4: scratch verification \u2014 sort the result" }, { "tool": "catalog_revise", @@ -39,8 +43,14 @@ "vega_spec": { "mark": "bar", "encoding": { - "x": {"field": "region", "type": "nominal"}, - "y": {"field": "total", "type": "quantitative"} + "x": { + "field": "region", + "type": "nominal" + }, + "y": { + "field": "total", + "type": "quantitative" + } } } }, @@ -52,7 +62,7 @@ "cell_id": "PLACEHOLDER", "markdown": "## Shoe sales by region\n\nBroader bucket: boots, sneakers, sandals." }, - "narration": "beat 10: rewrite the markdown for the audience (skipped — needs cell_id resolution)", + "narration": "beat 10: rewrite the markdown for the audience (skipped \u2014 needs cell_id resolution)", "skip": true } ] diff --git a/docs/architecture-new.md b/docs/architecture-new.md new file mode 100644 index 00000000..1405b410 --- /dev/null +++ b/docs/architecture-new.md @@ -0,0 +1,859 @@ +# Tallyman architecture + +This document describes tallyman as the code in `src/` builds it. Where it and +the code disagree, the code is right. The rules the system promises are stated +normatively in [system-contract.md](system-contract.md), and +[caching.md](caching.md), [expression-lifecycle.md](expression-lifecycle.md), +[reactive-recalc.md](reactive-recalc.md) and [mcp-server.md](mcp-server.md) each +go deeper on one subject. The reasons for the design are in +[ADR-007](../plans/ADR-007-tallyman-owned-materialization.md) (tallyman writes +its own result files), [ADR-008](../plans/ADR-008-row-order-of-reads.md) (every +file carries a row-order column), [ADR-009](../plans/ADR-009-digest-stability.md) +(what the result digest covers) and +[ADR-011](../plans/ADR-011-sources-are-aliases.md) (a raw input is an alias). + +## 1. What tallyman is + +Tallyman is a notebook without cells. Its unit of work is an **entry**: one +dataframe result, stored on disk under a content hash and usually given a name. +An agent, Claude Code, writes each entry as a short Python **recipe** that +builds a xorq expression (a deferred dataframe computation, which runs only when +executed) over other entries, which it reads by name. Tallyman freezes the +expression, runs it, stores the result and records which entries it read, so +that revising one entry can recompute the entries built on it. Data files come +in through an explicit import that makes each version of a file an entry too. A +React app shows the catalog in the browser, and Buckaroo, a separate data-grid +server, draws each entry's rows. + +## 2. The processes + +| Process | Started by | Owns | +|---|---|---| +| MCP server, `src/tallyman_mcp/server.py` | Claude Code runs `tallyman mcp`, one per session, over stdio | the session's active project; the builds and imports the agent asks for | +| Companion, `src/tallyman_companion/app.py` | `tallyman run`, a FastAPI app on port 7860 unless `--port` names another | the API and event stream the browser uses; the builds, recalcs and resets the browser asks for; the Cache page's delete; the Buckaroo subprocess | +| Buckaroo | the companion, through `BuckarooManager` (`buckaroo_lifecycle.py`) | grid sessions and their queries: paging, sorting, search and summary statistics | + +**The MCP server** has 31 tools and one prompt. It remembers the session's +active project, read from `~/.tallyman-notebooks/active_project` on the first +call and changed only by `project_switch` or `project_new`, which go through the +companion. `_with_checkpoint` commits a checkpoint (one git commit of the +catalog, section 10) after every tool that returns without an error, except +those in `_NO_CHECKPOINT`: most read-only tools, the project tools, and +`catalog_recalc` and `catalog_promote_diff`, which commit their own. After a +change the server posts to the companion's `/internal/notify`, best effort, and +it posts nothing when no companion serves its projects (below). + +**The companion** serves the React app, a JSON API under `/{project}/api/`, and +a stream of Server-Sent Events (SSE, named messages on an HTTP response the +browser keeps open) at `/{project}/api/sse`. It builds through +`PUT /{project}/api/code/{alias}` (the code editor) and +`POST /{project}/api/promote_diff/{alias}/{va}/{vb}`, and its middleware +`_checkpoint_after_mutation` commits a checkpoint after every successful non-GET +request, except on routes that commit their own or change no authored state. +`tallyman serve ` runs a read-only companion with no Buckaroo. + +**One server per data dir.** The **data dir** is the directory that holds every +project, `~/.tallyman-notebooks` unless `TALLYMAN_HOME` names another. +`tallyman run` claims it before it starts anything else +(`tallyman_core/server_lock.py`): it takes an exclusive `flock` on +`/server.lock` and holds it on a descriptor it keeps open while it +serves, so the kernel drops the lock however the process exits. The file also +holds the **owner record**, JSON naming the server's pid, host, port, bind +address, start time, data dir and command line. A second `tallyman run` on the +same data dir is refused with a message that names the holder and says how to +run a second tallyman on its own data dir and port. Each server holds state no +other process sees (its SSE subscribers, its Buckaroo sessions, its memos), so +two serving one project would each show a view the other's writes never reach. +The MCP server and `tallyman reset-to` find their companion on each call with +`server_lock.companion_url()`, from the owner record's port, and believe the +record only while its lock is held. Nothing else names the companion: no +default port and no environment variable. With no server on the data dir there +is no companion, so notifies are skipped, tool replies carry `url: null`, and +`project_switch` and `project_new` return an error. Each notify, project switch +and project creation carries its data dir as `home`, and a companion serving +another data dir answers 409 and changes nothing; the browser sends no `home` +and is not checked. `tallyman mcp`, one per Claude Code session, and +`tallyman serve` claim nothing. + +**Buckaroo** runs as `python -m buckaroo.server --port 8700 --no-browser +--stdio-control` and exits when its stdin closes. It displays, and queries only +for statistics, sorting, search and paging; tallyman finishes an entry's +computation before handing it over, so a computation never fails inside a grid +query. Buckaroo keeps each grid's state as a **session** in memory, and drops a +session that has had no browser attached for an hour. + +**xorq** is a library. Tallyman uses its expression API (`xorq.vendor.ibis`, an +ibis fork), `build_expr` to freeze an expression to a directory named by its +12-character hash, `load_expr` to load one back, and its embedded DataFusion +engine, the only backend. xorq's hash covers a graph's structure and the paths +of the files it reads, not their contents. Tallyman does not use xorq's cache: +no build contains a xorq cache node, a recipe that calls `.cache()` fails to +build, and tallyman writes, names, verifies and re-creates every result file +itself (`tallyman_xorq/materialize.py`). The catalog store is tallyman's own +(`tallyman_core/catalog.py`, `catalog_state.py`). + +Both the MCP server and the companion write the catalog. The **project lock** +(`catalog_state.project_lock`, a `flock` on `artifacts/catalog/.checkpoint.lock`) +makes their builds, imports, heals (re-creations of a missing result file, +section 7), checkpoints and resets happen one at a time. It is re-entrant within +a thread and blocks with no timeout. Smaller writes (an alias, a notebook cell, a +chart, a display config, `config.json`) take no lock and replace their whole +file atomically. + +**The execution lock.** Each process executes reads on one DataFusion session, +xorq's default backend (`xorq.config.default_backend()`), and two threads +executing on it at once fail with `RuntimeError: Already borrowed`. Both +processes run work on thread pools (FastAPI's for requests, FastMCP's for tool +calls), so every execution on that backend holds `execution.execution_lock`, one +re-entrant lock per process: `/api/data` pages, post-processing runs, the +primary-key probes and `full_diff`'s Buckaroo helpers. Executions in one process +therefore run one at a time, and a long one, such as a diff's summaries, makes +that process's page reads wait. Another process has its own backend and its own +lock. The project lock comes first: anything that can heal +(`cached_result_expr`, `ensure_materialized`) runs before the execution lock is +taken, and `project_lock` raises `RuntimeError` in a thread that holds the +execution lock and would take a new `flock`. Two executions run on connections +of their own and take no execution lock, so a build or heal does not hold up +page reads: a materialization's stream (`materialize._stream_to_parquet`) and a +cheap entry's row count at build (`result_cache.stream_row_count`). +`tests/test_execution_lock.py` checks the rule in the source: every execution in +`src/` sits inside `with execution_lock():` or is one of those two, and nothing +that can take a project lock is called inside that block. + +## 3. The project on disk + +A project is `~/.tallyman-notebooks/projects//`; `TALLYMAN_HOME` moves +the root. A file is **cache** when tallyman can make it again from files that +are not cache and check what it made, so anything may delete it (ADR-007's +rule). A file is **record** when nothing can re-create it. A **log** only +explains what happened. Paths are under `artifacts/catalog/`, the catalog's git +repository, unless they start with `artifacts/` or `data/`. + +| Path | Holds | Kind | +|---|---|---| +| `aliases.jsonl` | one line per alias, `{alias, latest, history, kind}` | record, tracked | +| `entries.jsonl` | the complete entry directories the last checkpoint saw | record, tracked | +| `entries/.zip` | the **recipe zip**: a deterministic archive of the entry's `expr.py`, `schema.json`, `manifest.json` and `xorq_build/`, written by the first checkpoint after the entry is made; nothing reads it back | record, tracked | +| `config.json` | project settings, one key: `auto_recalc` | record, tracked | +| `notebook.jsonl`, `chart_specs/`, `display_configs/`, `prompts/`, `post_processing/`, `stats/` | notebook cells, per-entry charts and grid settings, the prompts each entry was built from, the project's statistic and post-processing functions | record, tracked | +| `entries//` | the entry directory (section 4) | record, untracked: every read uses it, and nothing re-creates it | +| `entries//.xorq_build_expanded/`, `.xorq_view_build/`, `.buckaroo_stat_cache/`, `primary_key.json` | per-entry derived files | cache | +| `compute_cache/result_cache/.parquet` | snapshots, the result files of worthy entries (section 4) | cache, unless pinned because it cannot be made again faithfully (section 8) | +| `bullpen/entries//`, `bullpen/cas/` | the **bullpen**: entry directories and clones a reset retired, kept so a reset forward can bring them back (section 10) | record, parked | +| `diff_stat_cache/-/` | Buckaroo statistics per diffed pair | cache | +| `data/.cas/` | clones: the bytes of every imported file | record | +| `artifacts/display/` | display classes, outside the repository | record, untracked | +| `artifacts/errors.jsonl`, `events.jsonl`, `telemetry.jsonl` | failures, the activity log, grid-load timings | log | + +The catalog's `.gitignore` keeps the untracked paths out of `git add -A`, and +`catalog.assert_catalog_consistent`, run after every reset, rejects any tracked +path outside `TRACKED_SURFACE`. Since nothing reads a recipe zip, cloning the +catalog repository does not recreate entry directories. The logs sit outside the +repository, so a reset does not rewind them, and `errors.jsonl` holds no state: +dismissing the error banner deletes it. `tallyman init` writes a fixture at +`data/orders.parquet`, which no build reads until it is imported. + +Two files sit at the top of the data dir, outside every project: +`active_project`, the name of the active project, and `server.lock`, the claim +and owner record of the `tallyman run` serving the data dir (section 2). A +server that exits leaves its record in the file, and nothing believes it once +the lock is gone. + +## 4. Core objects + +### Entries + +An **entry** is one computation, built once over inputs fixed forever, and +stored in `entries//`. Its **content hash** is its name: for a +computed entry, xorq's hash of its expression after tallyman's rewrite +(section 6). Every file that expression reads is another entry's snapshot, named +by that entry's hash, so a child's hash is a function of its parents'. The chain +ends at source entries, the entries an import makes, one per version of a file, +whose hash comes from the imported bytes (section 5). The absolute project path +is part of those read paths, so the same recipe in a project at another path has +another hash. An entry directory holds: + +- `expr.py`, the **recipe**, with the project path replaced by + `${TALLYMAN_PROJECT_ROOT}`. It names inputs by alias, so running it again can + mean something else; tallyman re-runs one only to mint a new entry in a + recalc, and in a diagnostic after an **unfaithful heal** (a re-created result + whose rows differ from the ones recorded, section 7). +- `xorq_build/`, the **build**: the expression as `build_expr` froze it, with + every input fixed and paths made portable (`portable.make_portable_inplace`). + A worthy parent appears as the path of its snapshot, a cheap parent as its + graph inlined. Once an entry exists, its build is what it means and its recipe + is documentation. +- `schema.json`: columns, types and the row count. +- `manifest.json`, the **manifest**: what the build does not record. It is the + directory's last write, made atomically by `manifest.write_manifest`, and a + directory without one is not an entry. After create, only an unfaithful heal + rewrites it. + +| Manifest field | Meaning | +|---|---| +| `content_hash`, `project`, `created_at`, `prompt` | identity and authorship | +| `parents` | `[{hash, ref, follow}]`, the parent edges | +| `cache_worthy`, `cache_worthy_why` | the worthy-or-cheap verdict and its reason | +| `row_count`, `execute_seconds`, `compile_seconds`, `cache_bytes` | measurements | +| `result_digest` | the content digest of a worthy entry's snapshot | +| `reproducible`, `nonreproducible_columns` | whether two runs at create gave the same digest | +| `unfaithful_heal_digest` | the digest an unfaithful heal wrote; it pins the snapshot | +| `snapshot_format`, `engine_versions` | the snapshot format version and the xorq, xorq-datafusion and pyarrow versions | +| `provenance` | a source entry only: where its data came from, `{alias, version, path, digest, suffix, reader, imported_at}` | + +A worthy entry's **snapshot** is its result as one parquet file, +`compute_cache/result_cache/.parquet` (`materialize.snapshot_path`), +a path that depends on the content hash alone. + +### Worthy and cheap entries + +`worthiness.classify_expr` decides once, at build, on the author's expression, +whether an entry is **worthy** or **cheap**. Tallyman **materializes** a worthy +entry: it runs it to completion and writes its snapshot when the entry is +created. A cheap entry has no file; its small plan runs again on every read, +over files that exist. + +An entry is cheap only if every relation in it is a file read, filter, +projection (renames, casts and computed columns), column drop, null drop or null +fill; it reads exactly one file; and no value in it multiplies rows (`Unnest`), +depends on row arrival order (a window function) or is impure (`random()`, +`uuid()`, `now()`, `today()`, any UDF). Anything else is worthy, including every +aggregate, join, sort, limit and union. The test is an allow-list because a cheap +entry pages by the `__row_order` column of the one file it reads (every +snapshot ends in one, numbering its rows `0..N-1`), so a wrong "cheap" +gives unstable paging where a wrong "worthy" costs a copy. A source entry is +worthy by definition, and a filter over a worthy entry is cheap, since it reads +a snapshot. + +### Aliases and versions + +An **alias** is a mutable name for a line of entries, a record +`{alias, latest, history, kind}` in `aliases.jsonl` (`tallyman_core/aliases.py`). +`latest` is the head, `history` holds versions V1 to Vn, and `-v` names +version N; a name matching that pattern is refused. `set_alias` moves the head +and appends to the history. + +A **catalog alias** names computed entries and moves by create, revise, recalc +and promote. A **source alias** names the versions of an imported dataset and +moves only by an import. A name is one kind or the other, and `set_alias` keeps +an alias's kind equal to its entries' kind on every route (`AliasKindMismatch`). +A source alias can be renamed or removed; the entry keeps the name it was +imported as, so a message naming a version asks the alias store which alias +holds it (`source_import.current_source_version`). An entry no alias has held, +built by `catalog_run`, is a **scratch entry**, and must be named with +`catalog_alias` before a recipe can read it. + +### Parent edges + +While a recipe runs, the readers in `tallyman_xorq/io.py` record each entry it +reads as a **parent edge** in `manifest.parents` (through `parent_capture.py`). +`tracked_expr_from_alias("sales")` records a **followed edge**, +`{hash: , ref: "sales", follow: true}`: the child goes stale when the alias +moves. `pinned_expr_from_alias("sales-v2")` records a **pinned edge**, with +`ref: "sales-v2"` and `follow: false`: the child stays on that version. Both +return the parent's result through `cached_result_expr` (section 7). The edges +are the whole record of what an entry depends on. `follow` decides what a recalc +does and has no effect on reads, which go through the build. + +### What a recipe may read + +A recipe names aliases, and a build reads only the **arena**, the files tallyman +owns: the clones under `data/.cas/` and the snapshots under +`compute_cache/result_cache/`. Everything else is refused at build, and the +message says what to write instead. + +| A recipe that | Is refused by | +|---|---| +| calls `read_project_file` or `tallyman_read_csv` | `io._source_entry_read` | +| calls `xo.deferred_read_csv` | `build._csv_direct_read_check` | +| calls `xo.deferred_read_parquet` on a file outside `compute_cache/` | `build._raw_parquet_read_check` | +| passes a content hash to either reader | `io` (ADR-011 D5, a bare content hash is refused in a recipe) | +| passes a bare alias to `pinned_expr_from_alias` | `io`: it would pin whatever the head happened to be | +| names an alias or version that does not exist | `io` (`ProjectDataNotFound`) | +| reads in-memory data (`ibis.memtable`, `read_in_memory`) | `source_cache.rewrite_for_build` | +| calls `.cache()` | `rewrite_for_build` | +| assigns to `__row_order` | `row_order.assert_not_assigned` | +| is cheap and drops `__row_order` | `row_order.require_on_cheap`, showing the corrected select | +| joins three entries in one chain, first and two right-hand sides carrying `__row_order` | `row_order.assert_joinable`, showing the `.drop("__row_order")` fix | +| sorts, then loses a sort key in a later order-keeping step | `row_order.canonical_sorted` | +| is a revision that follows its own alias | `catalog_revise` and `PUT /code`, after the build | + +Two generated recipes break the naming rule on purpose. A source entry's recipe +calls `read_project_file`, which a context variable +(`source_import._SOURCE_ENTRY`) resolves to that entry's own snapshot; the path +in the call is provenance. A promoted diff's recipe names two content hashes +(section 11). Operations are refused too: a revise or diff promotion aimed at a +source alias is refused before anything is built, with the text of +`aliases.source_alias_refusal` (the companion answers 409), and `catalog_alias` +refuses to give a source entry a catalog name. + +## 5. Data in: importing a file + +This picture covers sections 5 to 7. Solid arrows happen when an entry is +created, dotted ones when a missing snapshot is healed. + +```mermaid +flowchart LR + F["outside file"] -->|"catalog_import_source"| CL["clone in data/.cas"] + CL --> SS["source entry snapshot"] + SS -->|"tracked_expr_from_alias"| B["build_and_persist"] + B -->|"worthy: materialize"| WS["worthy entry snapshot"] + B -->|"cheap: no file"| FB["frozen build"] + CL -.->|"heal"| SS + FB -.->|"heal"| WS + WS --> R["cached_result_expr"] + FB --> R + WS -->|"view build"| G["Buckaroo grid"] + FB -->|"expanded build"| G +``` + +A file enters the catalog only by an **import**, the MCP tool +`catalog_import_source(outside_path, alias, pinned_version=None, prompt="", +schema=None, reader_options=None)`, which calls `source_import.update_and_depend`. +It copies the bytes into the arena and points a source alias at a new **source +entry**, an ordinary entry whose rows are the file's rows. It refuses a +directory, a missing file, an alias name in version syntax, a name that is a +catalog alias, and a suffix other than `.parquet`, `.pq`, `.csv`, `.tsv` or +`.txt`. It fixes the reader options, digests the file (md5) and computes the +entry hash, then, under the project lock, picks the case (`_plan_version`), +writes the entry if it does not exist (`_mint`) and moves the alias. "Bytes" +below means the bytes read with the given reader options. + +| State | Result | +|---|---| +| the alias does not exist | mint v1 | +| no `pinned_version`, bytes differ from the head | mint the next version | +| no `pinned_version`, bytes equal the head | nothing changes; the head is returned | +| bytes equal an older version | refused, naming `reset_to`, and `pinned_expr_from_alias("-v")` to read that version | +| `pinned_version=N` exists and matches | nothing changes; vN is returned and the head stays | +| `pinned_version=N` exists and does not match | refused: the file, or its reader options, are not that version's | +| `pinned_version` is head + 1, bytes are new | mint it | +| `pinned_version` beyond head + 1, or below 1 | refused: versions cannot be skipped | +| bytes are a version of another alias | refused, naming that alias and version | + +**The entry hash.** `source_entry_hash` is `md5("source||")` cut to 12 hex characters, the shape of xorq's hash. It cannot come +from xorq, because the entry's recipe reads the snapshot the hash names. It +covers the bytes and the reader options and nothing else, so two projects that +import one file each hold their own entry, snapshot and clone under one hash. + +**Reader options.** A parquet file takes none. A CSV takes a `schema` (a dict by +column name, or ADR-005's positional list with `"&rest"`) and `polars.scan_csv` +options such as `separator`; `infer_schema_length` and `schema_overrides` are +refused. The options must survive a round trip through JSON, so a callable is +refused (`_reader_for`). They are recorded on the entry and hashed with the +bytes (`ordered_copy._reader_signature`), so a file is read one way for good, +and a CSV read two ways is two imports under two aliases. A message that advises +an import prints them (`source_import.import_call`), since a call without them +names another entry. + +**The clone.** `source_identity.ensure_cas_path` copies the file to +`data/.cas/`, copy-on-write where the filesystem offers it, +digests the copy, and refuses it (`CloneDigestMismatch`) if the file changed +during the copy. The **clone** is the bytes as imported; it re-creates the +source entry's snapshot, and nothing deletes it (section 8). + +**The snapshot: pyarrow and polars.** The import writes the file's rows in file +order, plus a last `__row_order` column numbered `0..N-1`, to the entry's +snapshot path. Both readers feed one pyarrow writer, +`materialize.write_pinned_parquet`, with every snapshot's settings (zstd level 3, +parquet format 2.6, statistics, a page index) in row groups of 122,880 rows +(`ordered_copy.ORDERED_COPY_ROW_GROUP_ROWS`). A parquet file needs no parser: +pyarrow reads it with `ParquetFile.iter_batches`, so the file's types survive +exactly. A CSV is parsed by polars (`io._materialize_ordered`), which keeps row +order and runs ADR-005's schema language and inference ladder (100 rows, then +10,000, then the whole file, unless every column is pinned), and whose batches +reach the writer through `collect_batches` without the frame being held whole. + +**Zoned timestamps in a CSV.** A column whose schema type is a timestamp with a +zone, such as `timestamp('America/New_York')`, is read as text and parsed after +the scan (`io._parse_zoned`), because polars' CSV reader would parse text with no +UTC offset as UTC and convert it. Text with an offset is an instant and is +converted into the zone. Text without one is a wall-clock time in the zone and +keeps it, so `09:30` stays `09:30` New York time. This is ADR-005's rule for a +zoned schema on offset-less text (its D9(a): attach the zone, do not convert). +The import raises, naming the column, the row and the value, for offset-less +text that names a time the zone skips or repeats at a daylight-saving change, +for a column that mixes text with and without an offset (polars infers one +format per column from its first non-null value), and for text that is not a +timestamp. The first non-null value is checked before the read, since polars +1.40.1 writes nulls instead of raising when that value matches no format. The +declared precision is kept, `ns` included. + +**The entry.** `_mint` writes a generated `expr.py` whose header records the +path, digest and reader options, a real `xorq_build/`, `schema.json`, and last +the manifest, which marks the entry worthy and records the snapshot's +`result_digest` and the version's **provenance**, where it came from. The +presence of `provenance` is what makes an entry a source entry. Its `path` is +never read again, so moving, editing or deleting the original changes no build, +and its `alias` and `version` are the name the version was imported as. + +**One set of bytes, one alias.** Bytes that are already a version of another +alias in the project are refused (`_refuse_bytes_held_elsewhere`), and the error +suggests a catalog entry whose recipe is `tracked_expr_from_alias("")` as a second name. The rule keys on the entry hash, so one CSV under two +sets of reader options can be two aliases. + +**An existing version.** When the entry exists, the import rewrites none of its +files. If its snapshot is gone, the import restores the clone from the given +file (when the clone is gone too) and heals the snapshot through +`ensure_materialized`, verified against the recorded digest, so a repair never +changes a version's rows. A directory a crash left without a manifest is not an +entry, and is written again. + +**As a catalog operation.** A minted version is recorded as an `alias_set` +event, gets a notebook cell at v1 (or the previous version's chart and display +config), is announced with `new_entry`, and runs auto-recalc for the alias's +followers (section 9), all committed by one checkpoint. An import that changes +nothing records and announces nothing, and its checkpoint is an empty commit. + +## 6. The write path + +`catalog_run` builds a scratch entry. `catalog_create(name, code)` refuses a +source-alias name, an existing alias or a version-shaped name, then builds and +sets the alias. `catalog_revise(name, code)` and `PUT /code` refuse a source or +unknown alias, build, refuse a recipe that follows its own alias, and move the +alias. All call `build.build_and_persist`, which holds the project lock for the +whole build, so a child's build waits for its parent's materialization: + +1. **Run the recipe** (`_import_script`) with the parent collector armed. This is + the one moment names resolve; refused reads fail here. +2. **Lint.** `_nondeterminism_warnings` warns about `now()`, `today()`, + `random()`, `uuid()` and an unseeded `sample()`. +3. **Refuse raw reads** (`_csv_direct_read_check`, `_raw_parquet_read_check`). +4. **Classify and rewrite.** `classify_expr` gives the verdict, and + `source_cache.rewrite_for_build` refuses what section 4 lists, gives a worthy + entry the **canonical sort** (the author's keys, then `__row_order`, then every + other sortable column, a total order), and moves a cheap entry's + `__row_order` last. +5. **Freeze.** `build_expr` builds into a temporary directory named by the + content hash. If that entry exists with a manifest, the build appends the + prompt and returns it. +6. **Lay down** the entry directory: `xorq_build/`, made portable, and `expr.py`. +7. **Execute, with the reproducibility check.** A worthy entry is materialized by + `materialize(project, hash, check_reproducible=True, publish=False)`: two runs + of the frozen build, their digests compared, the first file left complete at + a temporary name. A cheap entry is streamed once in full (`stream_row_count`) + and nothing is kept, so a failing cast surfaces here, in tallyman. +8. **Record** `schema.json`, then the manifest (`write_manifest`). +9. **Publish.** `materialize.publish_snapshot` moves the staged file over the + snapshot path with one `os.replace`, once the entry is complete. +10. **Prompt log.** `_append_prompt` appends to `prompts/.jsonl`. + +**Materialization.** `materialize` loads the frozen build +(`result_cache.load_entry_expr`) and rebinds it onto a fresh single-partition +connection (`single_partition_backend`, batches of 8,192), so a float aggregate +merges in one order on any machine. `_stream_to_parquet` drops any inherited +`__row_order` and ibis's `__row_order_right`, numbers the rows in a new last +`__row_order`, and writes row groups of 1,048,576 rows with the settings of +section 5, to a unique temporary name under the project lock. It returns the +file's digest, read back (`digest.file_digests`). When step 7's runs differ, the +entry builds anyway: the manifest records `reproducible: false` and the columns +that differed, the snapshot is pinned, and the reply warns the author. A heal +runs once and replaces the file at once. + +**After the build**, the tool sets the alias, records an event in `events.jsonl`, +appends a notebook cell for a new alias, carries chart and display config +forward on a revise (`carry_forward_entry_config`), notifies the companion with +`new_entry`, and runs auto-recalc after a revise. The checkpoint comes last, as +the tool returns, and is taken even when nothing changed. + +**What a failure leaves behind.** + +- Before step 6, nothing under `entries/`. Parent snapshots the recipe's reads + re-created stay, as correct cache. +- From step 6 on, the build removes the directory it created and its temporary + file; the file already at the snapshot path is untouched. +- A killed process can leave a directory with no manifest, which every writer + treats as absent and every read refuses (section 7), and a `...tmp` + file in `result_cache/` that nothing removes. Killed between steps 8 and 9, it + leaves a complete entry over whatever file was at the path, or none, which the + next read heals. +- The MCP tools record the failure in `errors.jsonl`, add a `build_error` event, + notify `build_failed` and skip the checkpoint; `PUT /code` answers 400. +- A revision refused for following its own alias is already built, so its entry + stays without an alias and the next checkpoint records it. +- A failed checkpoint is logged; the next one's `git add -A` takes in the change. +- A failed import removes the directory it created; a clone or snapshot it wrote + stays, named by content, for the next import of the same bytes. + +## 7. The read path + +**Loading a build.** `result_cache.load_entry_expr` fills the project path back +into a stable copy of the build, the **expanded build** in +`.xorq_build_expanded/` (`portable.ensure_expanded_build`, gated by a +`.complete` marker), and loads it with `load_expr`. A missing or unloadable build +is an error naming the entry. Nothing falls back to `expr.py`, which would read +the aliases' current heads instead of the entry's parents. + +**The canonical read.** Every consumer inside tallyman's processes (`/api/data` +pages, charts, diffs, post-processing, a recipe's readers) reads a result through +`result_cache.cached_result_expr(project, hash)`. It calls `ensure_materialized`, +then returns, for a worthy entry, one bare read of its snapshot +(`deferred_read_parquet`, memoized in `_snapshot_read`), without loading the +build; this read is also how a worthy parent enters a child's build. For a cheap +entry it returns the loaded graph rebound onto the default backend +(`_resolve_result_plan`, memoized, through `rebind_onto`). Both are +single-backend expressions, so entries compose in joins, unions and diffs. + +**Making files exist.** `materialize.ensure_materialized` puts every file an +entry reads, and its own snapshot, on disk before anything runs: + +1. A worthy entry whose snapshot exists is done, with nothing loaded. +2. A source entry whose snapshot is missing is healed from its clone + (`_heal_a_source`). +3. Otherwise the build is loaded, and each missing file it reads, always another + entry's snapshot, is made by recursing on the hash in its name (`_recreate`). +4. A worthy entry whose snapshot is missing is healed (`_heal`). + +**A directory with no manifest is refused.** The manifest holds what a read +needs: the worthy-or-cheap verdict, the digest a heal is checked against, the +pin, and a source entry's provenance. `result_cache.entry_manifest` reads it and, +when the file is missing, raises `BuildError` with the message +`entry in '' has no manifest.json: `. A hash with no +directory at all gets the same error. `cache_worthy` reads only the manifest, so +a file at the snapshot path says nothing about the verdict. +`cached_result_expr` and `ensure_materialized` raise before they load or write +anything, and a child that reads such an entry raises the parent's error as it +is. Nothing answers around the error: `/api/data`, `/api/entry`, +`/api/entry_cache` and `/api/notebook_full` answer 500, and `/api/session` +answers status `error` with the message. (`/api/data`, `/api/entry` and +`/api/entry_cache` answer 404 for a hash with no directory, which they check +first.) Running the recipe again, or importing the file again for a source +entry, writes the entry again under the same hash. + +A **heal** re-creates a missing snapshot and checks it against `result_digest`. +A computed entry heals by running `materialize` once. A source entry heals from +its clone, which `source_import.rewrite_source_snapshot` parses again with the +reader options in `provenance`. Both take the project lock and look again for the +file first. If a source entry's clone is gone too, the heal fails with an error +naming both files and the `catalog_import_source` call, reader options and +`pinned_version` included, that restores the version. + +**Verification.** `result_cache._verify_self_heal` compares a healed file's +digest with the recorded one. A mismatch is an unfaithful heal: the rows are +served, as the honest output of the frozen build, and the heal logs a warning +that blames an engine upgrade (`_engine_change`), a recipe whose graph hashes +differently when re-run (`recipe_is_structurally_nondeterministic`) or a graph +that runs differently each time; writes the new digest into +`unfaithful_heal_digest`, which pins the file; records an `unfaithful_heal` +error for the page's banner; deletes the entry's Buckaroo statistics; and calls +`UNFAITHFUL_HEAL_HOOKS`, where the companion registers a forced grid reload +(section 11). `catalog_scan_staleness(verify_results=True)` runs the same +comparison over every snapshot on disk (`staleness.verify_sweep`) and writes +nothing. + +**Row order.** Every snapshot ends in `__row_order`, an int64 column holding +`0..N-1` in physical order. `row_order.page` orders a page by any sort keys it is +given and then `__row_order`, which has no ties, so a request returns the same +rows in any process; `/api/data` gives no keys. The build rules of section 4 keep +the column meaningful, and the canonical sort decides the order a worthy entry's +rows are numbered in. ADR-008 gives the reasons. + +**Digest stability.** `result_digest` is `arrow-sha256:` +(`digest.content_digest`), a SHA-256 over the file's Arrow data read back, with +separate streams per column for validity, lengths and values. It ignores the +row-group size, codec, writer version and string encoding, and changes with any +value, null, name, type or row order. Layout matters in one place: an ungrouped +float total depends on the batch boundaries of the file it reads, so the +row-group sizes (1,048,576 computed, 122,880 source) and the batch size are fixed +by `SNAPSHOT_FORMAT_VERSION`, which the manifest records (ADR-009). + +**Memos.** Each process keeps `_resolve_result_plan` (256 loaded builds) and +`_snapshot_read` (1,024 snapshot reads), cleared by +`cached_result_expr.cache_clear()`. File existence is checked on every call, +outside the memos, so they change latency and never rows. + +## 8. Pins and deletion + +A **pinned** snapshot is one the Cache page refuses to delete because it cannot +be made again faithfully. `materialize.pinned_reason` decides from the manifest +that speaks for the file (`snapshot_manifest`: the live entry's, or the copy a +reset parked) and, for a source entry, from whether its clone exists. + +| Pin reason | Read from | +|---|---| +| a source version whose clone is gone (for a retired one, from `bullpen/cas/` too) | `provenance` and the clone's path | +| a recipe whose two runs at create gave different digests | `reproducible: false` | +| a heal that produced different rows than were built | `unfaithful_heal_digest` | + +The pin travels with the manifest through a reset, and dismissing the error +banner does not lift it. It protects the file from the Cache page and nothing +else. + +**The Cache page is the one deleter.** Nothing in tallyman deletes a snapshot on +its own, and nothing writes one speculatively: a snapshot is written when its +entry is created or when a read needs it. +`GET /{project}/api/result_cache` lists every file in `result_cache/` with two +labels. A **retired** file has no live entry, but a reset parked one in the +bullpen, and that manifest decides its pin. An **orphan** has no entry, live or +parked, and is never pinned. `DELETE /{project}/api/result_cache/{hash}` unlinks +a file, answering 400 for a malformed hash, 403 in serve mode, 404 for no +snapshot and 409 with the reason for a pinned one. The next read heals the file; +a grid already open on it fails until the entry is opened again. + +**Clones.** A clone is record, the only copy of the imported bytes, and no read, +heal or import deletes one. A source snapshot is cache while its clone exists +and pinned once it is gone. Importing the same bytes again restores the clone +only when the snapshot is gone as well. `source_identity.gc_cas` is the only code +that removes a clone from `data/.cas/`, and only `reset_to` calls it, moving each +clone that no source entry in `entries.jsonl` names +(`catalog_state._live_source_digests`) to `bullpen/cas/`. If the bullpen already +holds that file, the live copy is removed; if any surviving manifest is +unreadable, the sweep is skipped. + +## 9. Staleness and recalc + +**One axis.** An entry is stale when the alias of one of its followed edges +points at a different hash than the edge recorded, and for no other reason +(`staleness.entry_staleness`). A pinned edge never makes an entry stale, and a +changed data file matters only once it is imported, which moves a source alias. +The scan reads `aliases.jsonl` and the manifests and opens no data file. An edge +whose alias is missing is reported under `unknown_axes`. + +**The scan.** `staleness.scan` judges every entry. Only an alias head can be +**directly stale** (`live: true`); a superseded version reports `live: false`, +since rebuilding it would move no alias. A live entry that is not directly stale +but sits downstream of one that is, is **transitively stale**. +`catalog_scan_staleness` and `GET /{project}/api/staleness` expose the scan and, +with auto-recalc on, add `orphan_stale`: each directly stale entry, classified by +`recalc.classify_orphans` as self-following, explained by a recorded error, or +unexplained. + +**Recalc.** `recalc.recalc(project, roots, dry_run)` rebuilds a **cone**: the +roots and every entry reachable from them through edges, parents first +(`dependents.descendant_cone`, which reports a cycle), with superseded versions +dropped unless named as roots. A dry run, the default, reports planned actions. +A real run (`_replay_cone`) replays each member's recipe through +`build_and_persist`, and before its children replay, moves every alias whose +head was that member to the new hash, so a child's `tracked_expr_from_alias` +reads the new head. A member whose inputs did not move, such as a pinned child, +replays to the same hash and is left alone. The walk stops at the first failure, +keeps the rebuilt prefix, records the failure and the stale entries it skipped in +`errors.jsonl`, and reports `status: "failed"`. A real run that changed something +takes one checkpoint. With no roots, `catalog_recalc` and +`POST /{project}/api/recalc` use every directly stale entry. + +**Auto-recalc.** When a project enables it (`config.auto_recalc_enabled`: the +`TALLYMAN_AUTO_RECALC` variable, then `auto_recalc` in `config.json`, then on), +an operation that moves an existing alias recomputes that alias's followers +within the operation: a revise from either surface, an import that mints a +version, and a diff promotion that re-points an existing alias. +`recalc.auto_recalc` takes as roots the entries this alias made directly stale +(`followers_of`), replays their cone without a checkpoint, and logs every other +directly stale entry, reporting it in `orphan_stale`. The operation's checkpoint +commits the head move and the cascade together, so one reset undoes both, and a +`recalc` event carries the `{old hash: new hash}` remap. + +## 10. Checkpoints and reset + +A **checkpoint** (`catalog_state.checkpoint_catalog`) is one git commit of the +catalog, tagged `step-NNN`, its **step**. Under the project lock it writes +`entries.jsonl` from the entry directories that have a manifest, zips any +without a tracked zip (`catalog.zip_pending_entries`), runs `git add -A`, commits +and tags. `genesis` records `step-000` when a project is created. + +**Reset.** `catalog_state.reset_to(project, ref)` returns the catalog to a step +or label, under the project lock: + +1. `git reset --hard` restores every tracked file, so source aliases rewind with + the rest. +2. `prune_entries` moves each entry directory `entries.jsonl` does not list into + the **bullpen**, `bullpen/entries//`, where a reset parks what it + retires so a later reset forward can bring it back. +3. `restore_from_bullpen` copies back each listed directory that is missing, and + the clones restored source entries name. It copies, so the bullpen keeps its + set and a project can go back and forth repeatedly. +4. `_retire_cas_clones` parks clones no surviving source entry names (section 8). +5. `catalog.assert_catalog_consistent` checks the allow-list, that the recipe + zips match `entries.jsonl`, and that every hash an alias, chart or display + config names has a zip. + +A reset leaves `compute_cache/` alone. A snapshot is named by its entry's hash, +so one left by a retired entry can never be served for another; the Cache page +shows it as retired, and a snapshot missing after a reset is healed as usual. + +**A parked directory replaced by the live one.** When a reset retires an entry +the bullpen already holds, `_retire` replaces the parked copy with the live one. +They can differ (`created_at`, `prompt`, and the `result_digest` of a recipe that +is not reproducible), and the live one matches the snapshot on disk, since a +create always rewrites the snapshot. A live directory without a manifest is +dropped instead. **A reset forward** therefore brings back the manifest that +matches the file, and while an entry is retired the Cache page reads its parked +manifest, counting a parked clone as present. + +After a reset the process clears its memos. The companion's +`POST /{project}/api/reset` also clears `diff_stat_cache/`, reloads the grids and +publishes `project_reset`; `tallyman reset-to ` asks a running companion to +do the same through `/internal/notify`. + +## 11. The viewer + +**Opening an entry.** When an entry's data tab opens, the browser asks +`GET /{project}/api/session/{hash}`, and `BuckarooManager.load_session` +restarts a dead Buckaroo (at most every 30 seconds), calls +`ensure_materialized` so any heal is done and verified first, and posts +`/load_expr`. A worthy entry is handed a **view build**, a build whose whole +graph is one read of its snapshot, written once to `.xorq_view_build/` with a +marker recording the snapshot path (`ensure_view_build`); a cheap entry is handed +its expanded build. The post carries the session id `entry--`, +`project_root` (`artifacts/`), `cache_storage_path` (`.buckaroo_stat_cache/`), +`row_order_column` and, for an entry with a display config, its +`column_config_overrides`. A failure before the post is answered as status +`error` with the reason. + +Tallyman keeps no map of sessions: the id is a function of project and hash, and +every open posts again. +Buckaroo skips the work when it holds that session with the same build directory +and the post carries no configuration, and re-creates a session it dropped. The +notebook page's route, `GET /{project}/api/notebook_full`, opens one for every +cell. + +**Events.** The companion publishes SSE events from its own routes and +republishes what the MCP server and the command line post to +`/internal/notify`. On `recalc`, `project_reset` and every **klass** change (a +statistic, post-processing function or display class written for the project), +it reloads grids with `reload_project_sessions`, which posts +`/reload_expr/` for every entry and treats 404 or 400 as "not open". +The browser refetches on `new_entry`, `build_failed`, `notebook_changed`, +`chart_attached`, `post_processing_changed`, `summary_stat_changed`, `recalc` +and `project_switched`, and follows a `recalc` remap in a background tab. It +ignores `entry_added`, `alias_changed`, `alias_renamed`, `display_changed`, +`project_reset` and `unfaithful_heal`. + +**The forced reload after an unfaithful heal.** The companion's hook calls +`BuckarooManager.force_reload_session`, which posts `/load_expr` for the entry's +session with `force_reload: true`, so Buckaroo recomputes the statistics the heal +deleted instead of showing ones from the old rows. The hook also publishes +`unfaithful_heal`. + +**Diffs.** `GET /{project}/api/diff_data/{alias}/{va}/{vb}` reads two versions +through `cached_result_expr`, finds a join key (`primary_key.diff_keys`, cached in +`primary_key.json`; a search whose queries run longer than one second in total +answers 504, and time spent waiting for the execution lock does not count), and +computes the summaries (`tallyman_xorq.diff.full_diff`). For the grid, +`tallyman_companion.diff.build_compare_expr` builds an outer join on the key, +with `__row_order` dropped from both sides and `membership`, `_eq`, `_pct_delta` +and `_abs_delta` columns, posted as session `diff--`. Diff sessions, +unlike entry sessions, are remembered until Buckaroo restarts. The join is not +materialized, so Buckaroo runs it for every grid query. + +**Promoted diffs.** `catalog_promote_diff` and the companion's promote route +save a diff as an entry whose generated recipe is: + +```python +from tallyman_companion.diff import build_diff_expr +expr = build_diff_expr(a_hash="", b_hash="", keys=[...]) +``` + +It names two content hashes and records no parent edge, because +`build_diff_expr` reads both sides through `cached_result_expr`, not through the +readers that record edges; a promoted diff never goes stale or takes part in a +recalc as a follower. It contains a join, so it is worthy. Its alias is +`diff__v_v` unless the MCP tool is given another; re-pointing an +existing one runs auto-recalc. Both surfaces commit their own checkpoint and +announce `entry_added`. + +## 12. Invariants + +Where the code breaks one of these, section 13 names the issue. + +- **A content hash names one result**: every read, warm or cold, in either + process, returns the rows the entry's build fixed, unless the recipe is itself + nondeterministic, which the digest detects. +- **A recipe names aliases**, never a file or a bare content hash. +- **A build reads only the arena**: every file it reads is a snapshot named by + its entry's content hash. +- **Names resolve once**, when an entry is minted, and no read, heal or grid + hand-off resolves a name to decide rows. +- **The manifest completes an entry**: it is the last write, atomic, and a + directory without one is not an entry. Writers build over it, and every read + refuses it; nothing stands in for the manifest's verdict. +- **A snapshot changes only by an atomic replace of a complete file**, under the + project lock. +- **Two routines write result bytes**, `materialize` and the import's + `_write_snapshot`, with the same parquet settings. +- **Every snapshot ends in `__row_order`**, and every `/api/data` page is ordered + by it. +- **A re-created snapshot is verified before it is served**, and a mismatch pins + it. +- **Only the user deletes a snapshot**, through the Cache page, which refuses a + pinned one. +- **No clone's bytes are deleted**: a reset parks a clone no surviving entry + names in the bullpen. +- **An alias's kind matches its entries' kind**, and a source alias moves only by + an import. +- **A source alias's history only grows**, and no two of its versions hold the + same bytes read the same way. +- **An import never shifts a zoned CSV time**: offset-less text keeps its + wall-clock time in the column's zone, and a wall-clock time the zone skips or + repeats is an error. +- **Staleness has one axis**: a followed alias that points somewhere new. +- **One operation, one checkpoint**: a head move and its cascade commit together. +- **One write at a time per project**: builds, imports, heals, checkpoints and + resets hold the project lock. +- **One execution at a time per process** on its shared default backend, under + the execution lock, and a thread that holds it takes no new project lock. +- **One server per data dir**: `tallyman run` holds `server.lock` while it + serves, and a client of a data dir reaches only that data dir's companion. + +## 13. Known defects + +The open issues that touch this design, as of `63bcdd6`. The full list, including +the viewer, export and hint issues, with a priority for each, is +[`plans/open-bugs-2026-09-24.md`](../plans/open-bugs-2026-09-24.md), made at +`1f8cb02`. Since then four were fixed on this branch, and the sections above +describe the fixed behaviour: #118 by #242 (the execution lock, section 2), #183 +by #241 (one server per data dir, section 2), #204 by #245 (a directory with no +manifest is refused, section 7) and #231 by #244 (zoned CSV times, section 5). + +Wrong rows, pins or edges without an error: + +- #229: a pinned child and a tracked child with the same query compile to one content hash, so they are one entry and the first build's edge wins; a pinned alias can then move when its parent is revised. +- #228: `build._raw_parquet_read_check` allows `xo.deferred_read_parquet` of any file under `compute_cache/`, so a recipe can read a snapshot by its path: a bare content hash, with no parent edge and no `ensure_materialized` first. +- #232: an unfaithful heal of a source entry is attributed to a graph that runs differently each time ("or source drift under off"), though the likely cause is a reader change; a corrupt clone is healed from and pinned. +- #205: the canonical sort leaves nested columns out of its tie-break, so rows tied on every sortable column can come out in either order. +- #206: the snapshot writer drops any column named `__row_order_right`, including one the author made. +- #208: an unfaithful heal of a worthy parent changes its cheap children's rows under their hashes, and only the parent is flagged and reloaded. +- #209: an expanded build's marker does not record the project path, so a copied project keeps reading the old location; the marker is also trusted when the expanded folder is gone. +- #200: `full_diff` keeps `__row_order` as a data column, so one inserted row shows every later row as changed. +- #142: the CSV reader null-fills a short row instead of raising. + +Import: + +- #224: a parquet file with a `fixed_size_binary` or UUID column fails at import with a long traceback that names no column. +- #225: a failed import leaves its clone in `data/.cas/` and its snapshot in `result_cache/`, and nothing deletes the clone. +- #227: `catalog_import_source` lets a CSV parse error, or a clone that fails its digest check, escape as a plain `ValueError`; nothing reaches `errors.jsonl`, and the message names the clone and `tallyman_read_csv`. +- #234: the error for an older version's bytes names reset commands that do not exist. +- #239: importing a version's bytes again restores a lost clone only when the snapshot is gone too, so a version whose clone alone is lost stays pinned. +- #237: `catalog_list` does not report an alias's kind. + +Writes, the lock and processes: + +- #240: alias, notebook and `config.json` writes load, change and replace the file without the project lock, so two overlapping writers lose one change. +- #233: a recipe's alias reads take the project from the `active_project` file, while the MCP tool builds into its session's project and a companion route into the project in its URL. +- #186: the project lock blocks with no timeout, so a page read that needs a heal waits behind any build in the other process. +- #190: `PUT /code` and `POST /promote_diff` build on the companion's event loop, so the whole UI stops answering while they wait for the lock or build. +- #226: a writer killed mid-write leaves a `.tmp` in `result_cache/` or `data/.cas/` that nothing lists or deletes. +- #230: every failed build leaves its temporary `tallyman_expr_.py` in the OS temp directory. + +Recalc and row order: + +- #238: `catalog_recalc` with a source entry as a root fails, because the replay runs the generated `read_project_file` outside the import. +- #199: the three-way join check also refuses chains of semi and anti joins, which add no right-hand columns and cannot collide. +- #185: whether a recipe is pure is neither recorded nor passed on, so an entry built on a non-reproducible parent is recorded as reproducible, and `today()` passes the check at create. +- #187: an ungrouped float `SUM` or `AVG` depends on the row-group layout of the file it reads, which only the snapshot format version holds fixed. + +The viewer: + +- #170: Buckaroo is pointed at `artifacts/`, so it never finds the project's statistics and post-processing functions under `artifacts/catalog/`. +- #172: diff sessions are keyed by the two hashes with no project, so two projects holding the same source versions can share one. +- #188: the live diff grid hands Buckaroo an unmaterialized join, which Buckaroo runs for every query. +- #201: a klass reload posts `/reload_expr` once per catalog entry, one after another, from the event loop. +- #202: every grid open posts `/load_expr`, so two opens at once both load, and a promoted diff runs Buckaroo's statistics again on every open. +- #203: an unfaithful heal runs its checks and the forced Buckaroo reload while holding the project lock, and the reload opens a session nobody has open. +- #210: the plan memo keeps every loaded build and its backend objects alive, up to 256 of them. +- #235: the SPA has no listener for `project_reset`, `unfaithful_heal`, `entry_added` or `alias_changed`, so an open page goes stale. +- buckaroo-data/buckaroo#974: Buckaroo ignores `row_order_column`, so grid pages are not ordered by `__row_order`. + +Code and text left behind by ADR-011: #236. diff --git a/docs/architecture.md b/docs/architecture.md index d4093026..6187134a 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -1,438 +1,1013 @@ # Tallyman architecture overview -This is the top-level map of the tallyman codebase. Read it first, then follow -the cross-references into the per-subsystem docs. For a document index with a -currency note on each file, see [Related documentation](#related-documentation) -at the end. +This is the top-level map of tallyman: how the system works from end to end, +and where each part is documented in more depth. Read it first. The +[Related documentation](#related-documentation) section at the end lists every +doc, with a note on how current each one is. ## What tallyman is -Tallyman is a deconstructed notebook platform. Instead of a notebook file with -inline cells and outputs, the unit of work is a **catalog entry**: a single -xorq expression, compiled and stored on disk under a content hash. Claude Code -is the author — it drives tallyman through an MCP server, creating and revising -entries as xorq expressions. The entries accumulate in an on-disk, -content-addressed, git-backed catalog. A FastAPI companion server plus a React -SPA visualize that catalog in a browser, and a Buckaroo subprocess provides -interactive data grids over each entry's result. +Tallyman is a notebook without cells. Its unit of work is a **catalog entry**: +one xorq expression (a deferred dataframe computation, which runs only when +asked), compiled and stored on disk under its **content hash**. Claude Code is +the author. It drives tallyman through an MCP server, sending Python that +builds an expression, and each successful call becomes an entry in a +git-backed catalog on disk. A FastAPI companion server and a React single-page +app (SPA) show the catalog in a browser, and a Buckaroo subprocess draws the +interactive data grid for each entry. -Nothing in tallyman is a long-lived application server holding state in memory. -The catalog on disk is the source of truth. The running processes (companion, -Buckaroo, MCP server) are views and editors over it; the MCP server holds no -in-memory catalog state, and the companion holds only its SSE subscriber list. +The catalog on disk is the source of truth, and the running processes are +views of it and editors of it. Each process keeps in-memory caches keyed by +content hash, which change how fast an answer comes back and never what the +answer is. Beyond those caches, the MCP server remembers which project its +session is working on, and the companion holds its open SSE streams and the +diff sessions it has opened in Buckaroo. So one server runs per **data dir**, +the directory that holds every project (`~/.tallyman-notebooks` unless +`TALLYMAN_HOME` names another): `tallyman run` holds an exclusive lock on +`/server.lock` while it serves, and a second `tallyman run` on the same +data dir is refused with a message naming the one that holds it. See +[One server per data dir](#one-server-per-data-dir). + +### Terms + +These are the project's own terms. The other docs use them with the same +meaning. + +- **Catalog:** a project's entries, aliases and notebook, kept in a git + repository at `/artifacts/catalog/`. +- **Entry:** one catalog computation, stored in `entries//` and + committed as `entries/.zip`. +- **Content hash:** an entry's identity, xorq's 12-character hash of the + entry's expression after tallyman's rewrite. Every file the expression reads + is a snapshot named by the content hash of the entry it holds, so the hash + covers the inputs as well as the structure. A source entry (below) is the + exception that makes this work: its hash is an md5 of the imported bytes and + the reader options, truncated to the same 12 characters. +- **Recipe:** the Python the author submitted, kept as the entry's `expr.py`. + It names its inputs by alias, so run again later it can mean something else. +- **Build:** the entry's `xorq_build/` directory, the expression frozen to disk + with every input fixed. A read of a cheap entry and every heal of a computed + entry load the build (a source entry heals from its clone); + a worthy entry whose snapshot exists is read from the snapshot alone. The + recipe is run again only to make a new entry (a revise or a recalc), and by + one diagnostic after a heal that went wrong. +- **Manifest:** the entry's `manifest.json`, which records what the build does + not say: parent hashes, the cheap-or-worthy verdict, the result digest, and + for a source entry where its data came from. +- **Alias:** a mutable name, such as `sales`, that points at the latest content + hash of a logical entry and keeps every hash it has pointed at, as versions + V1, V2 and so on. An alias has a kind: a **catalog alias** names + computations, and a **source alias** names versions of an imported file. +- **Import:** `catalog_import_source`, the one way a file enters the catalog. + It copies the file's bytes into the project, writes one snapshot of them and + points a source alias at the new entry. A recipe never opens a file. +- **Source entry:** one version of a source alias. It is an ordinary entry + whose manifest also records `provenance`: the path the file was imported + from, its digest, the reader options, and the name it was imported as. The + path is never read again. +- **Clone:** the imported bytes, kept under `data/.cas/`, + named by their md5 digest and checked against it when written. A source + entry's snapshot is made again from its clone. +- **`__row_order`:** an int64 column holding `0..N-1` in a file's physical row + order. It is the last column of every file tallyman writes, and pages of an + entry are sorted by it. +- **Worthy and cheap entries:** a **worthy** entry is one tallyman + materializes, because its plan does work that is expensive or that cannot + keep its input's row order (an aggregate, a join, a sort, a window function, + among others). A **cheap** entry is row-preserving over one file (filters, + column selections, computed columns) and has no file of its own; its small + plan re-runs on every read. +- **Snapshot:** the parquet file that holds a worthy entry's result, + `compute_cache/result_cache/.parquet`. A source entry is worthy, + and its snapshot holds the imported rows in the file's order. +- **Materialize:** run an entry's build to completion and write the result to + its snapshot. +- **Heal:** make a missing snapshot again (a source entry's from its clone, + any other by running the entry's build) and check it against what was + recorded. +- **Result digest:** a content digest of a snapshot, recorded in the manifest + when the snapshot is first written; every heal is checked against it. +- **Pinned snapshot:** one the Cache page refuses to delete, because it cannot + be made again faithfully. +- **Checkpoint:** one git commit of the catalog repository, tagged `step-NNN`. + Each operation that changes the catalog lands as one checkpoint. +- **Reset:** `reset_to`, which returns the catalog to an earlier checkpoint. +- **Bullpen:** the directory a reset moves retired entry directories and + clones into, so that a later reset forward can bring them back. +- **Project lock:** a file lock on the project that builds, materializations, + checkpoints and resets take, so that they happen one at a time. +- **Session:** one grid's state inside the Buckaroo process. +- **View build:** a build whose whole graph is one read of a worthy entry's + snapshot. It is what Buckaroo is handed for a worthy entry. +- **Klass:** a summary statistic, post-processing function or display class + written for the project, for Buckaroo's grids to use. ### Big-picture flow ``` Claude Code - │ (MCP tool calls over stdio) + │ MCP tool calls over stdio ▼ - tallyman MCP server ──────────────► on-disk catalog - (compiles xorq exprs, ~/.tallyman-notebooks/projects// - checkpoints to git) (content-addressed entries, aliases, + tallyman MCP server ──────────────► on-disk project + (builds entries, ~/.tallyman-notebooks/projects// + checkpoints to git) (entries, snapshots, aliases, │ notebook, git history) │ best-effort HTTP notify ▲ - ▼ │ reads/writes + ▼ │ reads and writes tallyman companion (FastAPI :7860) ────────┘ │ REST + SSE ▼ React SPA (packages/app) in the browser │ embeds grids ▼ - Buckaroo subprocess (:8700) ◄─── companion POSTs xorq_build/ dirs - (interactive dataframe grids, sort/search push down to xorq) + Buckaroo subprocess (:8700) ◄─── companion posts a build to display + (grids: paging, sorting, search, summary statistics) ``` -The typical loop: Claude Code calls an MCP tool to create or revise an entry, -the MCP server compiles the xorq expression and writes a content-addressed -entry plus a git checkpoint, then notifies the companion. The companion emits a -Server-Sent Event (SSE, a push message over an HTTP stream the browser holds -open; see [Live updates over SSE](#live-updates-over-sse)), the SPA refetches -the affected data, and when the user opens -an entry the companion warms a Buckaroo session so the grid loads. See -[expression-lifecycle.md](expression-lifecycle.md) for the full -create-to-view path. +The typical loop: Claude Code calls an MCP tool to create or revise an entry. +The MCP server imports the code and builds the entry. It checks the +expression, decides whether the entry is worthy, freezes it, runs its query to +completion (writing the snapshot of a worthy entry), and writes the entry +directory. It then notifies the companion, best effort, and commits a +checkpoint as the tool returns. The companion publishes a Server-Sent Event +(SSE, a message on an HTTP stream the browser keeps open; see +[Live updates over SSE](#live-updates-over-sse)), and the SPA refetches what +changed. When the user opens the entry, the companion first makes sure every +file the entry reads exists, then asks Buckaroo to open a session for it, and +the grid connects to that session over a WebSocket. +[expression-lifecycle.md](expression-lifecycle.md) follows one entry through +all of this. -Three processes run on one machine: `tallyman_mcp` (spawned by Claude Code over -stdio), `tallyman_companion` (the FastAPI app, started with `tallyman run`), and -a `buckaroo` server subprocess that the CLI/companion supervises. +Three processes run on one machine: `tallyman_mcp`, which Claude Code spawns +over stdio (one per Claude Code session); `tallyman_companion`, the FastAPI app +that `tallyman run` starts; and the Buckaroo server, which `tallyman run` +spawns and stops. The MCP server and the companion both write to the catalog. +The project lock makes their builds, materializations, checkpoints and resets +happen one at a time; smaller writes (an alias, a notebook cell, a chart) are +atomic file replacements that take no lock. ## Component map -Tallyman is six subsystems: five Python packages under `src/` and one frontend -workspace under `packages/`. The dependency direction runs core ← xorq ← {mcp, -companion} ← cli; nothing in `tallyman_core` imports the web, MCP, or compute -layers. - -**tallyman_core** (`src/tallyman_core/`) is the native catalog model. It owns -the on-disk representation: versioned entries keyed by content hash, mutable -aliases with version history, the notebook cell list, chart specs, display -configs, project-global post-processing and summary-stat functions, and the -prompt/error/event logs. It manages the git-backed checkpoint transaction -(capture pointers, zip recipes, stage, commit, tag) and the reset-to-revision -operation that rewinds the catalog and reconciles untracked build artifacts -through a holding area called the bullpen. A key invariant lives here: -`assert_catalog_consistent` enforces an allow-listed tracked surface and -verifies that every hash referenced by an alias, chart, or display config has a -durable recipe zip. The call direction is one-way: `catalog_state` calls -`catalog`, never the reverse. Design: [native-catalog-store.md](../plans/native-catalog-store.md). - -**tallyman_xorq** (`src/tallyman_xorq/`) is the xorq integration layer. It -compiles a xorq expression into a content-addressed entry, computes the content -hash from the expression structure (and, in some source-identity modes, the -source file digests), decides whether the entry is expensive enough to bake a -result snapshot, and writes a portable build directory whose absolute paths are -rewritten to `${TALLYMAN_PROJECT_ROOT}` placeholders. It also implements -reactive staleness detection (comparing recorded manifest fields against the -current world) and the recalc cone that recomputes dependents in dependency -order. Reconstruction of an entry's expression from its persisted `expr.py` -happens here, using context variables that pin the entry's recorded source -digests. See [caching.md](caching.md) and [reactive-recalc.md](reactive-recalc.md). - -**tallyman_companion** (`src/tallyman_companion/`) is the FastAPI web server on -port 7860. It surfaces catalog views over REST, pushes live updates over SSE, -manages the Buckaroo subprocess lifecycle, and handles browser-initiated -mutations (code revision, diff promotion, resets). There are no server-side -HTML templates — it serves the compiled React SPA (`packages/app/dist`) as a -catch-all and mounts `/assets` and `/static`. A checkpoint middleware wraps -mutating routes so each authored change lands as one git revision, with an -opt-out denylist for routes that self-checkpoint or should not checkpoint at -all. It builds diff expressions (outer join with membership and per-column -delta/equality sentinel columns) and computes the Buckaroo column-config -overrides that color them. No dedicated subsystem doc yet — the route table is -in `app.py`; the create-to-view path is in [expression-lifecycle.md](expression-lifecycle.md). - -**tallyman_mcp** (`src/tallyman_mcp/`) is the FastMCP server that Claude Code -talks to over stdio. It exposes the catalog, notebook, and project tools -(`catalog_run`, `catalog_create`, `catalog_revise`, `catalog_alias`, -`catalog_diff`, `catalog_recalc`, the chart/display/stat/post-processing tools, -`notebook_*`, and `project_*`). Checkpointing is opt-out: every tool -auto-checkpoints at the dispatch boundary unless it is on the no-checkpoint -list. The active project is session-sticky (seeded on first tool call, surviving -disk changes within the session), and project lifecycle changes are POSTed to -the companion rather than written directly so SSE stays honest. Notifications to -the companion are best-effort and never raise. Every tool, its parameters, and -its side effects (checkpoint, SSE notify, auto-recalc) are documented in -[mcp-server.md](mcp-server.md). - -**tallyman_cli** (`src/tallyman_cli/`) is the Click command-line interface -(entry point `tallyman`). It initializes projects (with synthetic fixture data), -runs the companion and the MCP service, and owns the Buckaroo subprocess via -`BuckarooManager`, which spawns `python -m buckaroo.server` on port 8700 (or a -random free port) and exits when its stdin closes. It also provides `serve` -(read-only companion against a project directory anywhere on disk), `pack` -(portable tarball excluding cache and session state), `reset-to` / `revisions`, -and storyboard `replay` for deterministic rehearsal. See [installing.md](installing.md). - -**Frontend** (`packages/app/`) is the React 18 + Vite SPA. It builds to `dist/` -and is served by FastAPI as a catch-all; it drives refetches off an SSE version -counter rather than polling, defers grid loads until scroll via -`LazyBuckarooEmbed`, and remaps views to new hashes when a recalc event arrives. -Its `BuckarooEmbed` component mounts `BuckarooServerView` from `buckaroo-js-core` -directly and connects over WebSocket. No dedicated frontend doc yet. - -## On-disk catalog layout - -A project lives at `~/.tallyman-notebooks/projects//`. The home root is -`~/.tallyman-notebooks/` by default and is overridable with the `TALLYMAN_HOME` -environment variable (`paths.py:tallyman_home`). The single active project name -is recorded in `~/.tallyman-notebooks/active_project` (one line). The catalog -itself is a git repository at `/artifacts/catalog/`. +Tallyman is six parts: five Python packages under `src/` and one frontend +workspace under `packages/`. Imports mostly run one way, core ← xorq ← +{companion, mcp} ← cli. The exceptions: `tallyman_core` imports `tallyman_xorq` +lazily inside three functions (`reset_to` clears the read memo and retires +clones, and `run_post_processing` reads an entry's result), the MCP server +imports the companion's diff helpers for `catalog_promote_diff`, and the +companion imports the CLI's fixture writer for `/api/projects/new`. + +**tallyman_core** (`src/tallyman_core/`) is the catalog model and store. It +owns the on-disk representation: entries keyed by content hash, aliases and +their version history, the notebook's cell list, chart specs, display configs, +the project's post-processing and summary-stat functions, per-project settings +(`config.json`), and the prompt, error and event logs. It runs the checkpoint +(record the entry pointers, zip new recipes, `git add -A`, commit once, tag the +step) and `reset_to`, which rewinds the catalog to a step and reconciles the +files git does not track through the bullpen. It holds the project lock +(`catalog_state.project_lock`). `assert_catalog_consistent` enforces an +allow-listed set of tracked paths and checks that every hash an alias, chart or +display config names has a committed recipe zip. The call direction is one +way: `catalog_state` calls `catalog`, never the reverse. Design: +[native-catalog-store.md](../plans/native-catalog-store.md). + +**tallyman_xorq** (`src/tallyman_xorq/`) turns recipes into entries and serves +their results. `build.py` imports a recipe and builds the entry. `io.py` holds +what recipes read other entries with (`tracked_expr_from_alias`, +`pinned_expr_from_alias`) and the refusals of a raw file read. +`source_import.py` imports a file as a source entry: `source_identity.py` keeps +the clones, and `ordered_copy.py` holds the reader options a source entry +records and the row-group size of its snapshot. +`worthiness.py` decides cheap or worthy. `source_cache.py` and `row_order.py` +check and rewrite the expression before it is frozen. `materialize.py` writes +snapshots and makes sure the files an entry reads exist. `digest.py` computes +content digests. `result_cache.py` is the one read of an entry's result. +`portable.py` makes builds relocatable. `staleness.py`, `dependents.py` and +`recalc.py` are the reactive system, and `primary_key.py` and `diff.py` serve +version diffs. See [caching.md](caching.md) and +[reactive-recalc.md](reactive-recalc.md). + +**tallyman_companion** (`src/tallyman_companion/`) is the FastAPI web server +on port 7860. It serves the catalog over REST, pushes live updates over SSE, +handles edits made in the browser (a code revision, a diff promotion, notebook +edits, a recalc, deleting a snapshot, project switching), and talks to Buckaroo +through `BuckarooManager` (`buckaroo_lifecycle.py`), which spawns the +subprocess, opens sessions and reloads them when klasses change. It has no +server-side HTML templates: it serves the built SPA (`packages/app/dist`) for +every GET that no route matches (an unmatched `/api` path gets a JSON 404), and +mounts the SPA's `/assets`. A checkpoint middleware commits one git revision +after each successful mutating request, except on an explicit list of exempt +routes (`/internal/*` and the project routes, cache and log clears, telemetry, +reset, and the routes that checkpoint themselves). `diff.py` builds the diff +comparison, an outer join with a membership column and per-column delta and +equality columns, and the Buckaroo column settings that colour it. No +dedicated doc covers the route table yet; it is in `app.py`, and the +create-to-view path is in [expression-lifecycle.md](expression-lifecycle.md). + +**tallyman_mcp** (`src/tallyman_mcp/server.py`) is the FastMCP server that +Claude Code talks to over stdio: 31 tools and one prompt. Every tool +checkpoints after it succeeds unless it is on the `_NO_CHECKPOINT` list. The +active project is sticky for the session: seeded on the first call, then +changed only by `project_switch` or `project_new`, which go through the +companion so that its SSE stream stays honest. It finds the companion on each +call from the port recorded in the data dir's `server.lock`; with no server on +the data dir it sends no notifications, its replies carry no entry links, and +`project_switch` and `project_new` return an error. Notifications to the +companion are best effort and never raise. [mcp-server.md](mcp-server.md) +documents every tool, its parameters and its side effects. + +**tallyman_cli** (`src/tallyman_cli/main.py`) is the Click command line, +`tallyman`. `init` creates a project, with a synthetic `data/orders.parquet` +unless given `--no-fixture` (written, not imported: a recipe reads it only after +`catalog_import_source`), and records its step-000 checkpoint. `run` claims the +data dir, refusing to start when another server holds it, then starts the +companion and, unless given `--no-buckaroo`, the Buckaroo subprocess +(`python -m buckaroo.server --stdio-control`, which exits when its stdin +closes) on port 8700, or on a random port if 8700 is taken. `mcp` starts the +MCP server. `serve` runs a read-only companion, without Buckaroo, against a +project directory anywhere on disk. `pack` tars a project directory. `reset-to`, +`revisions` and `revisions label` move through and name checkpoints. `replay` +runs a storyboard of MCP tool calls. See [installing.md](installing.md). + +**Frontend** (`packages/app/`) is a React 18 and Vite SPA with pages for the +catalog, the notebook, diffs, the Cache page, the activity log and the project +list. It refetches when an SSE event arrives instead of polling. The catalog +page's data tab asks for the entry's Buckaroo session as soon as the entry +opens, and shows a spinner, then the grid, or the reason it failed with a retry +button (#133). The notebook page's data route opens a Buckaroo session for every +cell each time the page loads (#202), and each cell's grid connects only when +the cell scrolls near the viewport (`LazyBuckarooEmbed`). The grid itself is +`BuckarooServerView` from `buckaroo-js-core`, connected to the Buckaroo process +over a WebSocket. No dedicated frontend doc yet. + +## On-disk layout + +A project lives at `~/.tallyman-notebooks/projects//`. The home root +is `~/.tallyman-notebooks/` by default and can be moved with the +`TALLYMAN_HOME` environment variable (`paths.tallyman_home`). The active +project's name is the one line of `~/.tallyman-notebooks/active_project`. +`server.lock` beside it is the claim of the `tallyman run` serving the data dir, +and its owner record ([One server per data dir](#one-server-per-data-dir)). ``` ~/.tallyman-notebooks/ - active_project # one line: the active project name - buckaroo_sessions.json # global session map {hash: {session_id, project, started_at}} + active_project # one line: the active project's name + server.lock # held by the running tallyman run; its pid, port and start time projects// artifacts/ - catalog/ # git repo — the tracked catalog - entries/.zip # tracked recipe zip (expr.py, schema.json, xorq_build/) - entries// # untracked build dir (gitignored, reconciled on reset) - entries.jsonl # pointer list, one {hash} per line - aliases.jsonl # one {alias, latest, history:[V1,V2,...]} per line - notebook.jsonl # one {cell_id, alias, markdown} per cell - compute_cache.jsonl # pointer list of warm cache files - config.json # project settings, e.g. {auto_recalc: bool} - chart_specs/.vl.json # Vega-Lite specs, keyed by content hash - display_configs/.json # {column_config_overrides, diff_provenance} - post_processing/.py # process(expr) functions (_disabled/ = soft-deleted) - stats/.py # compute(col) functions (_disabled/ = soft-deleted) - prompts/.jsonl # append-only per-entry prompt history - bullpen/ # untracked holding area for evicted entries/caches - compute_cache/ # untracked baked result snapshots (xorq) - diff_stat_cache/ # untracked Buckaroo diff-stat caches per entry pair - .gitignore # deny-by-default (entries/*/, bullpen/, caches, *.tmp) - display/.py # display klass files (ColAnalysis subclasses) - errors.jsonl # append-only error log (outside catalog repo) - events.jsonl # append-only activity log (outside catalog repo) - exports/ # marimo .py, screenshots, CSVs - data/ # user input parquets (fixtures) - data/.cas/ # content-addressed source clones (CoW), cas mode only + catalog/ # git repo: the catalog + entries/.zip # tracked recipe zip (expr.py, schema.json, manifest.json, xorq_build/) + entries// # untracked entry directory (below) + entries.jsonl # the entries a checkpoint recorded, one {hash} per line + aliases.jsonl # one {alias, latest, history, kind} per line + notebook.jsonl # one {cell_id, alias, markdown} per cell + config.json # project settings, e.g. {"auto_recalc": true} + chart_specs/.vl.json # Vega-Lite specs, by content hash + display_configs/.json # {column_config_overrides, diff_provenance} + post_processing/.py # process(expr) functions (_disabled/ holds removed ones) + stats/.py # compute(col) functions (_disabled/ holds removed ones) + prompts/.jsonl # the prompts each entry was built from + .gitignore # keeps the untracked paths below out of git add -A + .checkpoint.lock # the project lock (untracked) + compute_cache/ # untracked; files tallyman can make again + result_cache/.parquet # snapshots of worthy entries (source entries too) + bullpen/ # untracked; entries/ and cas/ that a reset retired + diff_stat_cache/-/ # untracked; Buckaroo statistics per diffed pair + display/.py # display klasses (outside the catalog repo) + errors.jsonl # error log (outside the catalog repo) + events.jsonl # activity log (outside the catalog repo) + telemetry.jsonl # Buckaroo grid-load timings (outside the catalog repo) + exports/ + data/ # tallyman init's fixture (an import takes any path) + data/.cas/ # clones: the bytes of every imported file + buckaroo.log # the Buckaroo subprocess's stderr (tallyman run) + notebook_marimo.py # written by catalog_export_marimo ``` -Key formats and what's tracked vs untracked: - -- **Recipe zip** (`entries/.zip`) is the durable, deterministic, - content-addressed entry. It contains `expr.py` (the author's literal source, - with portable placeholders), `schema.json` (field names and types), and - `xorq_build/` (the portable expression directory with `expr.yaml` plus deps). - The zip writer only runs inside a checkpoint. -- **Entry build dir** (`entries//`) is ephemeral and gitignored. It holds - `manifest.json` (the build-completeness sentinel), the load-time-expanded - `.xorq_build_expanded/`, and per-entry caches (`.buckaroo_stat_cache/`). A - directory without a `manifest.json` is treated as partial/crashed and is not - zipped. -- **Manifest** (`manifest.json`) carries entry metadata: `content_hash`, - `result_digest`, `sources` (`{rel_path: digest}`), `parents` (DAG edges), - `row_count`, `compile_seconds`, `execute_seconds`, `cache_worthy`, - `cache_bytes`. Written atomically (temp file + `os.replace`). -- **JSONL pointer files** (`entries.jsonl`, `compute_cache.jsonl`) and the - structured logs (`prompts/`, `errors.jsonl`, `events.jsonl`) are - append-oriented and line-delimited. They replaced the single `catalog.yaml` - that older docs reference. -- **Activity logs** (`events.jsonl`, `errors.jsonl`) live in `artifacts/`, - outside the catalog git repo, so they survive reset-to-revision and have no - size cap. The UI filters them on read. -- **No per-entry `result.parquet`.** That layer was removed (#104); an expensive - entry's rows live in the baked `.cache()` snapshot under `compute_cache/`, a - cheap entry keeps no copy and recomputes on read. - -All catalog writers use atomic writes, so a crash mid-write or a checkpoint -firing during a write window leaves a whole file, not a torn one. +An entry directory, `entries//`, holds the four recipe members +(`expr.py`, `schema.json`, `manifest.json`, `xorq_build/`) and the per-entry +caches, which are made again on demand: `.xorq_build_expanded/` (the build with +the project path filled back in), `.xorq_view_build/` (a worthy entry's view +build), `.buckaroo_stat_cache/` (Buckaroo's summary statistics) and +`primary_key.json` (the key a diff joins on). `.xorq_build_expanded/` and +`.xorq_view_build/` each have a sibling `.complete` marker, written last. + +Key formats, and what is tracked: + +- **Recipe zip** (`entries/.zip`) is the committed record of an entry. + Nothing reads it back, so a clone of the catalog repository does not recreate + entry directories from it. The checkpoint writes it, deterministically, once + per entry, and nothing else does, so it keeps the manifest as it was at + create. +- **Manifest** (`manifest.json`) records `content_hash`, `project`, + `created_at`, `prompt`, `row_count`, `execute_seconds`, `compile_seconds`, + `cache_worthy` and `cache_worthy_why` (the cheap-or-worthy verdict and its + reason), `cache_bytes` (the snapshot's size), `result_digest`, `reproducible` + and `nonreproducible_columns`, `snapshot_format` and `engine_versions`, + `parents` (`[{hash, ref, follow}]`), `unfaithful_heal_digest` (set by an + unfaithful heal, and a pin) and, on a source entry only, `provenance` + (`{alias, version, path, digest, suffix, reader, imported_at}`, where `alias` + and `version` are the name it was imported as). It is the entry directory's + last write, atomic, and its presence means the entry is complete: the entry + list, the checkpoint, recalc and the build skip or rebuild a directory that + has none. Every read refuses such a directory: `result_cache.entry_manifest` + raises a `BuildError` naming the missing `manifest.json` before anything is + loaded or written, the companion's entry routes answer 500, and running the + recipe again writes the entry again. After create only an unfaithful heal + rewrites the manifest, to record `unfaithful_heal_digest`, atomically, under + the project lock. +- **`compute_cache/`** holds the files tallyman writes and can make again: + snapshots, a source entry's included. It is untracked, a reset leaves it + alone, and anything may delete it: the next read makes what it needs again. + The exceptions are the pinned snapshots, which cannot be made again; one of + them is a source entry's snapshot once its clone is gone. +- **Logs** (`errors.jsonl`, `events.jsonl`, `telemetry.jsonl`) live in + `artifacts/`, outside the catalog repository, so a reset does not rewind + them and a recorded failure survives it. Dismissing the error banner deletes + `errors.jsonl`, which holds no pin (pins are decided from the manifest), and the + readers of `errors.jsonl` skip a line that is not a JSON object, so one torn + append does not break the banner or the error page. +- **Display klasses** (`artifacts/display/`) are also outside the catalog + repository, so no checkpoint commits them and no reset rewinds them. + Summary stats and post-processing functions are inside it. +- **No per-entry `result.parquet`.** That layer was removed in #104; the + companion still sweeps any old one away once per project at startup. + +Tracked catalog files are written to a temporary file and renamed into place, +so a checkpoint that fires during a write commits a whole file. +`entries.jsonl` is written only inside the checkpoint, under the project lock. ## Core domain concepts -**Content hash (identity).** Every entry is identified by a hash derived from -its expression structure (xorq's tokenization) and, depending on the -source-identity mode, its source file digests. Identity is structural: two -entries with the same expression and inputs collapse to the same hash, which is -what makes builds idempotent. Because hashes are content-addressed and globally -unique by construction, Buckaroo sessions are keyed globally by content hash, -not per project. The source-identity mode (`off` / `cas` / `salt`, default -`cas`) is decided in [ADR-002-source-identity-content-hash.md](../plans/ADR-002-source-identity-content-hash.md). - -**Result digest.** A second identity axis, recorded for *worthy* (snapshot-baking) -entries only. The content hash keys the expression graph; the `result_digest` -keys the executed *bytes* as a row **multiset**, not a sequence. It is the -SHA-256 of the entry's baked snapshot parquet (`snapshot_file_digest`), which the -bake writes after sorting on a synthetic `original_row_order` and pinning the -parquet write settings, so the bytes are reproducible run-to-run and the file -hash is order-insensitive. Cheap, row-preserving entries record no digest — they -have no snapshot to hash and recompute live. A mismatch when an evicted snapshot -self-heals points at execution nondeterminism (sampling, `now()`, an impure UDF, -source drift), not the unordered-scan row reshuffling the canonical ordering now -absorbs. Design: [ADR-004-result-digest-canonical-ordering.md](../plans/ADR-004-result-digest-canonical-ordering.md). - -**Alias and V_n versions.** An alias is a named, mutable pointer (for example -`sales`) to the latest content hash of a logical entry. Each alias carries an -ordered `history` list of every hash it has pointed at (V1, V2, …, oldest -first). Revising an alias mints a new content hash, advances `latest`, and -appends to `history`; the old versions remain as forensic lineage. -`catalog_diff` resolves version indices (-1 latest, -2 previous) through this -history. - -**Parent edges and following.** At build time each entry records its direct DAG -parents as `{hash, ref, follow}`. `tracked_expr_from_alias('name')` records -`follow=True`: the edge names an alias, and the child goes stale as that alias -advances. `pinned_expr_from_alias('hash')` records `follow=False`: the edge pins -an exact hash and is never disturbed by recalc. Aliases are resolved to hashes -at read time, not baked into the edge. - -**Staleness.** Read-only and side-effect-free. An entry is stale on the alias -axis when a `follow=True` parent's alias head no longer equals the recorded -hash, or on the source axis when a recorded source digest no longer matches the -file's current digest. Computing staleness never executes anything; it only -compares the manifest against the current world and returns reasons. - -**Recalc cone.** When an alias head advances, its dependents form a cone of -entries that may now be stale. The cone is recomputed in topological order -(Kahn's algorithm over intra-cone parent edges) so parents rebuild before -children. Auto-recalc only recomputes the followers of the alias that just -moved; pre-existing ("orphan") staleness is left in place, logged, and -classified against the recorded error store rather than treated as an -unexplained invariant break. See [reactive-recalc.md](reactive-recalc.md) and -[recalc-mechanism.md](../plans/recalc-mechanism.md). - -**Result cache vs source cache (two-axis caching).** Tallyman caches along two -axes. The source axis caches reads of input files. The result axis bakes a -result snapshot for entries judged expensive (those whose expression contains -aggregates, joins, sorts, windows, or UDFs); cheap, row-preserving entries -recompute on every read even when an ancestor is expensive. Baked snapshots -self-heal on read, so a cold read transparently rematerializes. See -[caching.md](caching.md). - -**Portability.** A build directory embeds absolute filesystem paths in -`expr.yaml`. On write these are rewritten to `${TALLYMAN_PROJECT_ROOT}` -placeholders; on load they are expanded into a stable per-entry directory marked -complete by a sentinel, so a project can be copied or packed and run from -anywhere on disk, and the expanded path stays consistent so Buckaroo's -snapshot-cache keys match across restarts. - -**Checkpoint and reset-to-revision.** A checkpoint is an atomic git transaction -under a per-project file lock: it captures pointers, zips pending recipes, -stages all tracked files, commits once, and tags the step. Reset-to-revision -does a hard git reset to a commit and then reconciles untracked build artifacts -back to the recorded pointers, evicting to or restoring from the bullpen without -recompute (a forward reset copies back from the bullpen). Live operations never -read the bullpen. - -**Live updates over SSE.** The companion pushes changes to the browser with -Server-Sent Events (SSE): the SPA opens one long-lived HTTP stream to -`GET /{project}/api/sse` through the browser's native `EventSource`, and the -server writes named event messages down it. This is the inverse of polling. The -browser never asks "anything new?" on a timer; it holds the stream open and the -server speaks when something changes. The SPA registers listeners for these -kinds: `new_entry`, `build_failed`, `notebook_changed`, `chart_attached`, -`post_processing_changed`, `summary_stat_changed`, `recalc`, and -`project_switched` (plus `hello` and `ping`, which open and keep the connection -alive). An event is a signal, not a data payload: every one increments a -monotonic `version` counter held in a React context (`SSEContext.tsx`), and -components key their effects on that counter, so a bump triggers exactly one -refetch of the affected REST resource. That is what "refetches off an SSE -version counter rather than polling" means. Two events also carry state the -refetch cannot derive: `recalc` ships the `{oldHash: newHash}` remap so an open -entry view can follow its entry to the new hash, and `project_switched` ships -the project name to navigate to. If the stream drops, the context flips to -`offline`. Sources: `SSEContext.tsx` on the browser side, the `/{project}/api/sse` -route and the `/internal/notify` fan-out in `app.py` on the server side. +### Identity: the content hash + +An entry's content hash is the name xorq gives the build directory of the +entry's rewritten expression. xorq hashes a file read by its path alone, so +tallyman puts the content in the path: every file a recipe's expression reads is +a worthy entry's snapshot, named by that entry's content hash, and a child's +hash is therefore a function of its parent's. The chain bottoms out at source +entries, whose hash is `md5("source||")` truncated to +12 hex characters (`source_import.source_entry_hash`): the md5 digest of the +imported bytes and the reader options they were parsed with, and nothing else. +It cannot come from xorq, because a source entry's generated recipe reads the +snapshot that the hash names. So the hash of any entry covers the bytes of every +file under it. Two entries with the same expression over the same inputs +collapse to one hash, which makes building idempotent. The absolute path of the +project is part of a computed entry's hash, so the same recipe in a project at +another path gets another hash; a source entry's is not, so two projects that +import one file each hold their own entry under the same hash. The clone store +underneath is [ADR-002](../plans/ADR-002-source-identity-content-hash.md)'s, +narrowed by [ADR-011](../plans/ADR-011-sources-are-aliases.md) to one mode: +every import digests and clones. + +### Aliases and versions + +An alias is a line `{alias, latest, history, kind}` in `aliases.jsonl`. +Revising an alias builds a new entry, moves `latest` to its +hash and appends that hash to `history`; the old versions stay as they were. +The kind is `catalog` or `source`, and a name is one or the other, never both. +`set_alias` keeps an alias's kind matching its entries: a catalog alias never +points at a source entry and a source alias never points at a computed one, +whichever route sets it. A source alias moves only by an import: revising it, +promoting a diff onto it and `catalog_alias` onto a source entry are refused, +in the MCP tools and in the companion's code-edit and promote-diff routes +alike, before anything is built. Renaming and removing a source alias work as +for a catalog alias. +`catalog_diff` and the diff page pick versions from this history (`-1` is the +latest, `-2` the one before). A Buckaroo session id is derived from the project +and the content hash, `entry--`, so tallyman keeps no record of +sessions. + +### Parent edges + +At build time each entry records the entries its recipe read, as +`{hash, ref, follow}` in `manifest.parents`. +`tracked_expr_from_alias("sales")` records `follow=True`: the child follows the +alias and goes stale when the alias moves. `pinned_expr_from_alias` takes only a +version reference such as `"sales-v2"` and records `follow=False`: the child +stays on that entry. A bare alias is refused (#166), because it would pin +whatever the head happened to be, and so is a bare content hash (ADR-011 D5, no +bare hashes in recipes), so every edge an authored recipe records names an +alias (a promoted diff's generated recipe is the exception: it names its two +entries by hash and records no edge); an entry with no alias, +one built by `catalog_run`, has to be named before anything can build on it. A +source alias is read the same way as any other. The edge stores the hash the +alias pointed at when the child was built, and the staleness scan looks the +alias up again to compare. + +### Worthy and cheap entries + +Tallyman decides once, when an entry is built, whether it is worthy, and +records the verdict in the manifest as `cache_worthy`, with a short reason in +`cache_worthy_why` +(`worthiness.classify_expr`). An entry is cheap only if every relation +operation in it is a file read, a filter, a column selection or computed +column, a column drop, a drop of null rows or a fill of nulls; it reads exactly +one file; and no value in it multiplies rows (`unnest`), depends on the order +rows arrive in (a window function) or is not pure (`random()`, `uuid()`, +`now()`, `today()`, any UDF). Everything else is worthy. The test is an +allow-list, so an operation nobody has classified costs a copy instead of +unstable paging. A cheap entry must keep `__row_order`, because it pages by the +column of the file it reads: a select that drops the column fails the build, +and the error shows the corrected select. + +### Materialization + +A worthy entry is materialized when it is created, by one routine, +`materialize`, which every heal of a computed entry also uses. (A source entry's +snapshot is written by the import, and made again from its clone, with the same +writer settings.) `materialize` runs the entry's build on a single-partition +connection (so a float total is merged in one order), streams the rows through a +writer with a pinned layout (zstd, row groups of 1,048,576 rows, a page index), +numbers them in a last `__row_order` column, writes a temporary file and renames +it over the snapshot, all under the project lock, and returns the content digest +of the file it wrote. A heal renames at once; a create leaves the file at its +temporary name and renames it after the manifest is written, so a failed build +never touches the file already at the path. At create it runs the query twice +and compares the two digests. If they differ, the entry still builds, the +manifest records `reproducible: false` with the columns that differed, and the +snapshot is pinned. A cheap entry writes nothing: its plan is streamed once in +full at create, so an error in it fails the tool call. Every file that holds a +result is tallyman's; no build contains a xorq cache node, and xorq's own cache +is not used. [caching.md](caching.md) has the details. + +### Reads: `cached_result_expr` and `ensure_materialized` + +Every consumer reads an entry's result through `cached_result_expr`: `/api/data` +pages, charts, diffs, post-processing, and a child recipe chaining off the +entry. It first calls `ensure_materialized`, which makes every file the entry's +plan reads exist before anything runs: a missing snapshot is made again by +running its entry's build, or, for a source entry, by parsing its clone again +with the reader options it recorded (`_heal_a_source`). No plan reads a clone, +and nothing in a read makes one again. With the clone gone the source entry's +snapshot is the last copy of its rows, so it is pinned; if it is gone too, the +read fails with an error naming the missing clone and the +`catalog_import_source` call, reader options included, that repairs the version. +The Buckaroo hand-off calls `ensure_materialized` too. A worthy entry then reads +as one bare read of its snapshot, without loading its build when the file +exists, and a cheap entry as its frozen plan, re-run over files that exist. A +healed snapshot is checked against the recorded `result_digest`. A mismatch is +still served, since the rows are the honest output of the frozen build, but the +heal records the digest it wrote in the manifest's `unfaithful_heal_digest`, +which pins the file, logs an `unfaithful_heal` error for the error banner, and +wipes the entry's Buckaroo statistics; when the heal runs in the companion, +Buckaroo is also told to reload the entry's grid. The pin is part of the +manifest, so it moves with the entry through a reset and survives dismissing the +banner, which deletes the error log. + +### Sources: imports and source aliases + +A recipe names aliases, and a build reads only files tallyman owns +([ADR-011](../plans/ADR-011-sources-are-aliases.md)). A file enters the catalog +by one call, `catalog_import_source(outside_path, alias, pinned_version=None, +schema=..., reader_options=...)` (`source_import.update_and_depend`). It +digests the file, clones the bytes to `data/.cas/` and checks +the clone against the digest, writes the source entry's snapshot, writes the +entry itself (a generated recipe, a frozen build, a schema and a manifest with +`provenance`), and appends the entry to the source alias's history. The path +can be anywhere, and after the import it is never read again, so editing, +moving or deleting the original changes no build. New data arrives by importing +again under the same alias: different bytes mint the next version, and every +entry that follows the alias goes stale exactly as it would after a revise. +The import emits the same events as a revise, runs auto-recalc on the same +switch, and lands with its cascade as one checkpoint. + +The call's outcome depends on the alias's history (ADR-011 D3, the case table). +Identical bytes are a no-op returning the current version. `pinned_version=N` +claims the file is version N, or the next version to mint, and is refused if it +is neither. Versions cannot be skipped, and history only grows: bytes equal to +an older version are refused, with `reset_to` named as the way back and +`pinned_expr_from_alias("-v")` as the way to read that version. Bytes +another alias of the project already holds are refused too, naming that alias: +one set of bytes, read one way, is one version under one alias, and a second +name for it is a catalog entry whose recipe is `tracked_expr_from_alias("")`. A re-import of a version whose snapshot is gone rewrites nothing of +the entry: it restores the clone from the given file if needed and heals the +snapshot the way a read would, verified against the recorded digest. + +Reader options are fixed at import (ADR-011 D12). A parquet file takes none: its +snapshot is written by pyarrow in the file's order, so it keeps the file's +types. A CSV is parsed by polars under the schema and `scan_csv` options named in +the call, through ADR-005's schema language and inference ladder, and its rows +go to the same pyarrow writer in batches. A timestamp column whose schema type +has a zone is read as text and parsed in that zone (`io._parse_zoned`): text with +a UTC offset is converted into the zone, and text without one keeps its +wall-clock time. Offset-less text that names a time the zone skips or repeats at +a daylight-saving change, and a column that mixes text with and without an +offset, are errors naming the column, row and value. The options are recorded +on the entry and are part of its hash, so a CSV read two ways is two imports +under two aliases. A source entry's snapshot is written with the same pyarrow settings as +any other snapshot, in row groups of 122,880 rows rather than 1,048,576. + +The build refuses every other way of reading a file: `read_project_file`, +`tallyman_read_csv`, `xo.deferred_read_csv`, and `xo.deferred_read_parquet` of a +file outside `compute_cache/`, each with an error naming the import to use. +`read_project_file` survives only inside the recipe the importer generates, where +a context variable resolves it to the entry's own snapshot. + +### Row order + +Every file tallyman writes ends in `__row_order`, so a page served by `/api/data` is +sorted by `__row_order`, and the same request returns the same rows in any +process. (The paging helper also takes user sort keys and puts `__row_order` +after them; no route passes any yet, and Buckaroo's grid does not use the +column yet, buckaroo-data/buckaroo#974.) Every `order_by` in a recipe also gets +`__row_order` and then the remaining columns appended as tie-breakers, and a +sort followed only by steps that keep row order (filters, selections, limits) +still decides the order that is written. +[system-contract.md](system-contract.md) states the rules. + +### Result digest + +`result_digest` is `arrow-sha256:`, a SHA-256 over the Arrow data of a +snapshot as read back (`digest.py`). It does not change with how +the rows were batched, the codec, the row-group size or the writer's version, +and it does change with any value, any null, the order of the rows, and the +column names and types. Worthy entries record it; a cheap entry has no snapshot +and records none. Its one job is to show whether a re-created snapshot +reproduced the original. A mismatch is attributed to an engine change (the +manifest records the xorq, xorq-datafusion and pyarrow versions and the +snapshot format version), to a recipe that bakes a changing value into its +graph (#88), or to a graph that runs differently each time (#83). Running each +new worthy entry twice finds most such recipes at create; what it cannot find, +such as `today()` or a non-pure parent, is #185. Design: +[ADR-004](../plans/ADR-004-result-digest-canonical-ordering.md) and +[ADR-009](../plans/ADR-009-digest-stability.md). + +### Staleness + +Staleness is a judgment that runs no query, opens no data file and changes no +catalog state; it reads `aliases.jsonl` and the manifests. An entry is stale +when a `follow=True` parent's alias now points at a different hash than the one +recorded, and for no other reason (ADR-011 D6, one staleness axis). A pinned +parent never makes its child stale. A changed input file is not a reason until +it is imported: the import moves the source alias, and the entries following +it go stale through the same rule. Only an entry that is the current head of an +alias counts as directly stale (#154); a superseded version is reported with +`live=False`. A parent alias that no longer exists is reported under +`unknown_axes`. + +### Recalc cone + +When an alias head advances, the entries that followed it become directly +stale. They and every current head built on them form the +cone. Recalc replays each member's recipe in topological order (Kahn's +algorithm over the edges inside the cone), so a parent rebuilds and its alias +moves before its children replay. A member whose inputs did not move, such as +a child that pins a version of its parent, replays to the same hash and is left +alone. Auto-recalc, on by default for each project, runs this for the followers +of the alias a revise or an import just moved; staleness from any other cause +is left in place, logged, and classified against the recorded errors. See +[reactive-recalc.md](reactive-recalc.md). + +### Portability + +A build contains absolute paths in its `expr.yaml`. The build step rewrites the +project's path to a `${TALLYMAN_PROJECT_ROOT}` placeholder, +and a read fills it back in, into the stable per-entry directory +`.xorq_build_expanded/`. A project can therefore be packed, copied or cloned to +another path, and the expanded build keeps one path across restarts. The +expanded directory's marker +does not record which project path it was filled in with, so a copy that +carries the expanded directories along keeps reading the old location (#209). + +### Checkpoint and reset + +A checkpoint takes the project lock, records the complete entry directories in +`entries.jsonl`, zips any entry that has no recipe zip yet, runs `git add -A`, +commits once and tags `step-NNN`. The MCP server checkpoints after each tool, +and the companion after each mutating request; a recalc and a diff promotion +checkpoint themselves, once each. `reset_to` takes the lock, runs +`git reset --hard` to the step, and reconciles the files git does not track. Entry +directories the step does not list move to the bullpen; listed ones that are +missing are copied back from it. Source clones that no surviving entry refers to +move to `bullpen/cas/`, never deleted, and clones a restored entry needs are +copied back. A reset leaves `compute_cache/` alone: its files are named by +content hash, a leftover cannot be served for another entry, and a file that is +missing afterwards is healed like any other. An entry directory retired a second +time replaces the copy already parked under its name, since the live one is the +one that agrees with the snapshot on disk; a live directory with no manifest (an +interrupted build's) is dropped instead. The one live reader of the bullpen is +the Cache page: it lists a retired entry's snapshot as `retired` and decides its +pin from the manifest parked in the bullpen, counting a retired source version's +clone as present when it is parked in `bullpen/cas/`, since a reset forward +brings both back. + +### The project lock + +One re-entrant file lock, `catalog_state.project_lock` (a `flock` on +`artifacts/catalog/.checkpoint.lock`), is taken by a build (for its whole +length, recipe import included), an import, a materialization or heal, a +checkpoint and a reset. It holds between the MCP server and the companion, and +it is re-entrant within a thread, since a build materializes, and an import +heals, while it holds the lock. It blocks with no timeout, so a +page request whose entry needs a heal waits behind any build in the other +process (#186), and the two companion routes that build on the event loop, +`PUT /code` and `POST /promote_diff`, freeze the whole UI while they wait or +build (#190). Smaller writes to tracked files (aliases, notebook cells, charts, +display configs, `config.json`) take no lock: each replaces its whole file +atomically, so two processes editing the same file at the same moment can lose +one of the edits. The lock covers no reads either; they have the execution lock, +below. + +### The execution lock + +Each process executes reads on one DataFusion session, xorq's default backend, +and two threads executing on it at once fail with +`RuntimeError: Already borrowed`. The companion serves requests from FastAPI's +thread pool and the MCP server runs tool calls on FastMCP's, so +`execution.execution_lock`, one re-entrant lock per process, is held around +every execution on that backend: a `/api/data` page, a post-processing run, a +primary-key probe and `full_diff`'s Buckaroo helpers. Executions in one process +run one at a time, and another +process has its own backend and lock. The project lock comes first: anything +that can heal (`cached_result_expr`, `ensure_materialized`) runs before the +execution lock is taken, and `project_lock` raises in a thread that holds the +execution lock and would take a new file lock. A materialization's stream and a +cheap entry's row count at build run on connections of their own and take no +execution lock, so a build does not hold up page reads. +`tests/test_execution_lock.py` checks in the source that every other execution +sits inside `with execution_lock():`. + +### One server per data dir + +`tallyman run` claims its data dir before it starts anything else +(`tallyman_core/server_lock.py`): an exclusive `flock` on +`/server.lock`, held on a descriptor the server keeps open, so the +kernel drops it however the process exits. The file also holds an owner record +(pid, host, port, bind address, start time, data dir, command line), which names +the holder when a second `tallyman run` on the same data dir is refused. The MCP +server and `tallyman reset-to` find their companion from that record's port +(`companion_url()`), and believe it only while the lock is held; no default port +or environment variable stands in. Each notify, project switch and project +creation names its data dir as `home`, and a companion serving another data dir +answers 409. `tallyman mcp` and `tallyman serve` claim nothing. A second tallyman +runs on its own data dir and port: +`TALLYMAN_HOME= tallyman run --port `. + +### Live updates over SSE + +The companion pushes changes to the browser with Server-Sent Events: the SPA +opens one long-lived HTTP stream to +`GET /{project}/api/sse` through the browser's `EventSource`, and the server +writes named events down it. The browser never polls. The SPA listens for +`new_entry`, `build_failed`, `notebook_changed`, `chart_attached`, +`post_processing_changed`, `summary_stat_changed`, `recalc` and +`project_switched`, plus `hello` and `ping`, which open and keep the stream +alive. Each listened event except `project_switched` increments a `version` +counter in a React context (`SSEContext.tsx`), and components refetch their +REST resource when it changes. Two events carry state a refetch cannot derive: +`recalc` carries the `{oldHash: newHash}` remap, so that an entry view open in a +background tab moves to the entry's new hash (a focused tab stays put), and +`project_switched` carries the project to navigate to. The server also +publishes `entry_added`, `alias_changed`, `alias_renamed`, `display_changed`, +`project_reset` and `unfaithful_heal`, which the SPA has no listener for, so +they cause no refetch. If the stream drops, the context reports `offline`. +Sources: `SSEContext.tsx` in the browser; the `/{project}/api/sse` route and the +`/internal/notify` fan-out in `app.py` on the server. ## Request and data-flow paths -### catalog_run / catalog_create — author a new entry - -1. Claude Code calls the MCP tool with xorq source. -2. `tallyman_xorq` compiles the expression, computes the content hash, and runs - the build, writing the entry build dir: `expr.py`, `xorq_build/`, - `schema.json`, and finally `manifest.json` (written last, atomically, as the - completeness sentinel). `cache_worthy` entries bake a result snapshot into - `compute_cache/`. -3. The MCP dispatch boundary fires a checkpoint: `tallyman_core` zips the recipe, - stages the tracked surface, commits one git revision, and tags it. -4. The MCP server best-effort POSTs a notification to the companion. -5. The companion emits an SSE `new_entry` event; the SPA bumps its version - counter and refetches the entry list. See [expression-lifecycle.md](expression-lifecycle.md). - -### catalog_revise + auto-recalc — revise an entry and cascade - -1. A revision arrives from `catalog_revise` (MCP) or `PUT /code` (companion). It - mints a new content hash, advances the alias `latest`, and appends the old - hash to `history`. Charts and display configs carry forward from the old hash - to the new one only where the new hash does not already define them. - Self-alias references are rejected — an entry following its own alias would be - permanently stale by design. -2. Auto-recalc, if enabled for the project, walks the recalc cone of the alias's - followers in topological order and rebuilds each, re-pointing aliases before - replaying children. This walk is checkpoint-free. -3. The whole thing lands as one checkpoint, so head advance plus cascade is a - single git revision that reset-to-revision undoes atomically. -4. The companion emits an SSE `recalc` event carrying `{oldHash: newHash}` for - each remapped entry; backgrounded SPA views navigate to the new hash, focused - views stay put. See [reactive-recalc.md](reactive-recalc.md) and - [auto-recalc-on-revise.md](../plans/auto-recalc-on-revise.md). +### catalog_run / catalog_create: author a new entry + +1. Claude Code calls the tool with Python that binds `expr`. +2. `build_and_persist` takes the project lock and imports the code. While it + runs, `tracked_expr_from_alias` and `pinned_expr_from_alias` record each + parent edge and make the parent's files exist. +3. The build refuses what cannot become a sound entry: a raw file read, a bare + content hash or bare alias passed to `pinned_expr_from_alias`, an in-memory + table, a `.cache()` call, an assignment to `__row_order`, a cheap entry that + drops it, and a join chain over three entries that all carry it. + It classifies the entry, adds the canonical sort to a worthy entry, and + freezes the expression with xorq's `build_expr`, whose directory name is the + content hash. If a complete entry with that hash is already on disk, the + build stops and returns it. +4. It writes the entry directory (`xorq_build/` with portable paths, + `expr.py`) and runs the query: a worthy entry is materialized, twice, and its + snapshot written at a temporary name; a cheap entry is streamed once and + nothing is kept. +5. It writes `schema.json` (read from the snapshot for a worthy entry) and then + `manifest.json`, atomically. Only then is a worthy entry's snapshot moved + into place, as the last write to the entry's result. +6. `catalog_create` sets the alias and adds a notebook cell. The tool notifies + the companion, and as it returns the dispatch wrapper commits a checkpoint, + which zips the recipe. +7. The companion publishes `new_entry`, and the SPA refetches the entry list. + +If the build fails after it has created the entry directory, the directory is +removed, along with the snapshot's temporary file. The file at the snapshot's +path is untouched, so a snapshot that was already there, such as one a reset +left behind, survives a failed re-add. + +### catalog_import_source: bring a file in + +1. Claude Code calls the tool with a path, a source alias and, for a CSV, a + schema and reader options. +2. `update_and_depend` refuses a directory, a name that is already a catalog + alias, and a file that is neither parquet nor CSV. It decides the reader, + digests the file and computes the entry hash, then takes the project lock + and consults the alias's history (the case table under + [Sources](#sources-imports-and-source-aliases)). +3. To mint a version it clones the bytes to `data/.cas/`, writes the snapshot, + and writes the entry directory: a generated `expr.py` that records the path, + digest and reader options in its header, a frozen `xorq_build/`, + `schema.json`, and `manifest.json` with `provenance`, last. A failure removes + the entry directory it created. It then appends the entry to the alias. +4. The tool records an `alias_set` event, adds a notebook cell for a new alias + (or carries charts and display configs forward from the previous version), + notifies the companion with `new_entry`, and runs auto-recalc for the + alias's followers. The checkpoint as the tool returns takes in the import + and its cascade. A no-op import records and notifies nothing. + +### catalog_revise + auto-recalc: revise an entry and cascade + +1. A revision arrives from `catalog_revise` (MCP) or `PUT /code` (companion); + both refuse a source alias, before building anything, with an error that + names the import. It builds a new entry, moves the alias's `latest` to the + new hash and appends it to `history`. Charts and display configs carry + forward from the old hash to the new one where the new hash has none of its + own. A revision that reads its own alias by name is rejected: it would follow + its own head and be stale forever. +2. If auto-recalc is on for the project (it is by default), the followers the + revise made stale, and every current head built on them, are rebuilt in + topological order, each alias re-pointed before its children replay. The + walk takes no checkpoint of its own. +3. The revise and the cascade land as one checkpoint, so a reset to the step + before undoes both. +4. A cascade that changed anything publishes a `recalc` SSE event with the + remap, and the companion reloads the project's Buckaroo sessions. See + [reactive-recalc.md](reactive-recalc.md). ### Viewing an entry grid -1. The SPA entry-detail pane mounts `LazyBuckarooEmbed`, which waits for the grid - to scroll near the viewport (IntersectionObserver) and polls for session - startup. -2. The companion POSTs the entry's `xorq_build/` directory to the Buckaroo - subprocess's `/load_expr`. First the build dir is expanded into a stable - per-entry path with `${TALLYMAN_PROJECT_ROOT}` resolved, so Buckaroo's - snapshot-cache key matches and a cold read of an expensive entry self-heals - its baked snapshot rather than recomputing. -3. Buckaroo creates a session keyed by content hash and streams the grid over - WebSocket; sort and search push down to the xorq backend rather than paging a - materialized parquet. -4. The data tab stays mounted (hidden) across tab switches to keep the WS session - alive. Sessions live only in Buckaroo's RAM; a Buckaroo restart (detected via - a `started_at` timestamp on `/health`) clears the session maps and the next - view re-POSTs. +1. The catalog page's data tab requests `GET /{project}/api/session/{hash}` as + soon as it opens. On the notebook page, the data route (`/api/notebook_full`) + opens a session for every cell each time the page loads or refetches (on any + SSE event), and each cell's grid asks `/api/session` for its WebSocket URL + once the cell scrolls near the viewport (#202). +2. The companion calls `load_session`, which runs `ensure_materialized` first, + so a missing file is made again and checked before Buckaroo is involved. A + failure there shows in the page with its reason and a retry button. +3. It posts `/load_expr` to Buckaroo with the session id + `entry--`. A worthy entry is handed its view build + (`/.xorq_view_build/`, written once); a cheap entry its own build, + expanded into `/.xorq_build_expanded/`. The body also names the entry's + statistics cache directory, `__row_order` as the `row_order_column`, and the + `project_root` Buckaroo searches for klasses. That root is `artifacts/`, so + Buckaroo finds the display klasses there but not the stats and + post-processing functions, which tallyman writes under `artifacts/catalog/` + (#170). +4. Buckaroo creates the session, or answers from the one it holds, and the grid + connects over a WebSocket. Paging, sorting, search and summary statistics are + Buckaroo's queries over the build it was handed, which for a worthy entry is a + read of the snapshot. Buckaroo 0.15.6, the pinned version, ignores + `row_order_column`, so the grid's pages are not yet ordered by `__row_order` + (buckaroo-data/buckaroo#974); `/api/data` pages are. +5. Tallyman keeps no record of sessions. Every open posts `/load_expr` again + with the same id. Buckaroo skips the work while it holds that session with + the same build directory, and creates it again after dropping it (it drops a + session idle for an hour, and loses them all on a restart). Two opens at once + both post, and a promoted diff entry, which sends its column colouring, + re-runs Buckaroo's statistics on every open (#202). ### Diffing versions -1. The SPA diff page resolves the version pair (defaulting to V_{n-1} vs V_n via - the alias history) and requests the diff from the companion. -2. The companion builds a compare expression: an outer join of the two versions - with a membership column (a-only / b-only / both) plus per-column `{col}_eq`, - `{col}_pct_delta`, and `{col}_abs_delta` sentinel columns. The compare - expression is memoized for the process lifetime because the two entry hashes - are immutable. -3. Buckaroo column-config overrides color the diff: categorical coloring for - key/equality columns, numeric coloring for the delta columns. -4. The diff grid loads through the same `/load_expr` path as a normal entry view, - with diff stats cached per entry pair under `diff_stat_cache/`. +1. The diff page resolves the version pair (by default V_{n-1} against V_n) and + requests `/{project}/api/diff_data/{alias}/{va}/{vb}`. +2. The companion reads both sides through `cached_result_expr`, so both sides' + files exist first, and computes the code, schema, statistics, head and keyed + diffs (`full_diff`). Those summaries still include `__row_order` as a data + column (#200). +3. For the grid it builds a compare expression: an outer join of the two + versions on the diff key, with `__row_order` dropped from both sides, a + membership column (a only, b only, both), and per-column `{col}_eq`, + `{col}_pct_delta` and `{col}_abs_delta` columns. The expression is memoized + per pair and key (an LRU of 128, cleared on a reset or recalc). +4. The compare build is posted to Buckaroo as session `diff--` (the first + 12 characters of each hash), with statistics cached per pair under + `diff_stat_cache/`. The join is not materialized first, so Buckaroo runs it + for every query of the diff grid (#188). + +`catalog_promote_diff` and the diff page's promote button turn a diff into an +entry of its own, whose recipe calls `build_diff_expr(a_hash, b_hash, keys)`. +It contains a join, so it is worthy and materialized like any other entry. + +### Resetting to a revision + +`tallyman reset-to ` (CLI) and `POST /{project}/api/reset` (companion) +call `reset_to`, described under [Checkpoint and reset](#checkpoint-and-reset). +The companion then clears its in-memory result and compare memos and the +`diff_stat_cache/` directory, reloads the project's Buckaroo sessions, and +publishes `project_reset`. The CLI posts `project_reset` to the companion's +`/internal/notify`, which does the same clean-up. `tallyman revisions` lists the +steps, and `tallyman revisions label ` names one. + +## Known defects + +These open issues describe places where the system does not yet do what the +rest of the docs say it should. The docs describe the current behaviour and cite +the issue where it matters. + +Writes, the lock and processes: + +- #186: the project lock is one blocking lock with no timeout, so slow work in + one process blocks page reads in the other. +- #190: `PUT /code` and `POST /promote_diff` build on the companion's event + loop, which freezes the UI while they wait for the lock or build. + +Row order and diffs: + +- #199: the three-way join check also refuses chains of semi and anti joins. +- #200: `full_diff` keeps `__row_order` as a data column. +- #205: the canonical sort leaves nested columns out of its tie-break. +- #206: the snapshot writer drops any column named `__row_order_right`, + including one the author made. +- #188: the live diff grid hands Buckaroo an unmaterialized join. + +Heals and Buckaroo: + +- #202: every grid open posts `/load_expr`; concurrent opens load twice, and a + promoted diff re-runs Buckaroo's statistics on every open. +- #201: a klass reload posts `/reload_expr` once per catalog entry, one after + another, on the event loop. +- #203: an unfaithful heal runs its checks and the forced Buckaroo reload while + holding the project lock, and the reload opens a session for an entry nobody + has open. +- #208: an unfaithful heal of a worthy parent changes its cheap children's rows + under their hashes; only the parent is flagged. +- #209: a copied project keeps reading the old path through its expanded + builds. +- #210: the plan memo keeps each loaded build and its backend objects alive. + +Design questions still open: #185 (a non-pure recipe's verdict is not recorded +or passed on to entries built on it) and #187 (an ungrouped float `SUM` depends +on the layout of the file it reads, which the snapshot format version pins). +Found while checking these docs (#233): a recipe's +`tracked_expr_from_alias` and `pinned_expr_from_alias` resolve the project from +the `active_project` file, not the MCP session's own project, so the two can +disagree after another session switches projects and a recipe then looks its +aliases up in the other project (related to #39). +Older open issues in the same areas: #170 (Buckaroo is not pointed at the +project's stats and post-processing functions) and #157 (Buckaroo's on-disk +statistics cache has not been seen to give a first-load hit). + +Filed on 2026-09-24 against the same code, and described one by one in +[architecture-new.md](architecture-new.md#13-known-defects) and +[plans/open-bugs-2026-09-24.md](../plans/open-bugs-2026-09-24.md): wrong rows or +edges without an error (#228, #229, #232), the import (#224, #225, #227, +#234, #237, #239), writes and leftovers (#226, #230, #240), recalc of a source +entry (#238), the SPA's missing SSE listeners (#235) and ADR-011 leftovers in +the code (#236). + +These no longer apply since ADR-011 (a raw input is a source alias, PRs #217 to +#219), because the mechanism each was about is gone: #197 (a parquet source's +copy changed column types; pyarrow writes a source snapshot now), #198 (a CSV's +first read got JSON-rewritten reader options; the options are fixed at import +and must be plain values), #211 and #207 (the ordered copy's `.digest` file and +the copies left behind by source edits; ordered copies no longer exist), #191 +(staleness could not resolve a CSV outside `data/`; staleness no longer reads +files), and the two staleness defects the first version of this list named (the +scan wiping the source-digest memo, and a hash-pinned child stale for good on +the source axis). + +Fixed on this branch: + +- #193 by #222: a failed build deleted the snapshot already on disk for its + hash. +- #194, #195 and #196 by #223: a reset could pair a non-reproducible entry's + older manifest with its newer snapshot, a retired entry's snapshot lost its + pin, and the pin from an unfaithful heal lived in `errors.jsonl`, so + dismissing the error banner lifted it. +- #118 by #242: concurrent executions on a process's shared backend failed with + `Already borrowed`. See [The execution lock](#the-execution-lock). +- #183 by #241: two servers on one project went undetected. See + [One server per data dir](#one-server-per-data-dir). +- #204 by #245: with its manifest gone, an entry's worthiness was guessed from + whether a snapshot existed, so a worthy entry that had lost both was served + as cheap. Every read now refuses such a directory, which reverses #95, where + `/api/data` served it with a total of 0. +- #231 by #244: a zoned timestamp in a CSV schema read offset-less text as UTC + and converted it, so `09:30` in New York was stored as `04:30-05:00`. ## Related documentation -Currency notes below reflect a docs-vs-code audit on 2026-06-25. They will drift; -when in doubt, the code wins. - -### Architecture docs (`docs/`) — describe the current system - -- [system-contract.md](system-contract.md) — **normative**: the invariants and - binding rules the system guarantees (identity, read/write/cache contracts). - Where the descriptive docs and this contract disagree, the difference is a - bug. The read path implements it as of the #163 fix (PR #167). **Current.** -- [expression-lifecycle.md](expression-lifecycle.md) — one expression from MCP - ingest to rendered rows, naming every artifact and cache write. **Current.** -- [reactive-recalc.md](reactive-recalc.md) — revise an alias, recompute its - dependents; the cone and the dependency graph. **Current.** -- [caching.md](caching.md) — the caches across the stack and their invalidation. - **Mostly current.** -- [installing.md](installing.md) — install and run the spike. **Mostly current.** -- [mcp-server.md](mcp-server.md) — every MCP tool and prompt Claude Code drives, - with parameters, return shapes, and per-tool side effects. **Current.** - -### Design records / ADRs (`plans/`) - -- [ADR-002-source-identity-content-hash.md](../plans/ADR-002-source-identity-content-hash.md) - — content-addressed source reads so `content_hash` tracks source data. - **Mostly current.** -- [ADR-001-git-subprocess-threading.md](../plans/ADR-001-git-subprocess-threading.md) — - calling git from the multithreaded server (fork-safe `posix_spawn`). - **Mostly current.** -- [ADR-003-result-cache-cost-rubric.md](../plans/ADR-003-result-cache-cost-rubric.md) — - a *proposed* cost-vs-size cache rubric. **Partially stale / not adopted:** the - structural `cache_worthy` admission test it proposes to remove is still the - live gatekeeper, and `ensure_result` it names was removed (#73). -- [ADR-004-result-digest-canonical-ordering.md](../plans/ADR-004-result-digest-canonical-ordering.md) - — `result_digest` as a row multiset via a canonically-ordered snapshot hash, - replacing the per-row Python digest (#137). **Current** (implemented: the - digest is now `snapshot_file_digest`, and `tallyman_read_csv` injects - `original_row_order`). +Currency notes below reflect a docs-against-code check on 2026-09-22 against +the branch of #189 (ADR-007, ADR-008 and ADR-009 implemented), redone on +2026-09-24 after ADR-011 was merged into it. They will drift; when in doubt, the +code wins. + +### Architecture docs (`docs/`): the current system + +- [system-contract.md](system-contract.md): **normative**. The invariants and + rules the system guarantees (identity, reads, writes, materialization, row + order). Where the code or a descriptive doc disagrees with it, the difference + is a bug; its last section lists the known ones. **Current.** +- [expression-lifecycle.md](expression-lifecycle.md): one expression from MCP + ingest to rendered rows, naming every file written and when. **Current.** +- [caching.md](caching.md): every cache in xorq, tallyman and Buckaroo, what it + saves, what keys it, and what invalidates it. **Current.** +- [reactive-recalc.md](reactive-recalc.md): revise an alias, recompute its + dependents; the dependency graph, staleness and the cone. **Current.** +- [mcp-server.md](mcp-server.md): every MCP tool and the prompt, with + parameters, return shapes and side effects. **Current.** +- [installing.md](installing.md): install and run tallyman. **Current.** + +### Design records (`plans/ADR-*.md`) + +- [ADR-001-git-subprocess-threading.md](../plans/ADR-001-git-subprocess-threading.md): + calling git from the multithreaded server (fork-free `posix_spawn`). + **Accepted; mostly current.** +- [ADR-002-source-identity-content-hash.md](../plans/ADR-002-source-identity-content-hash.md): + content-addressed source clones, so `content_hash` tracks source data. + **Narrowed by ADR-011:** the clone store stands, and a reset moves + unreferenced clones to the bullpen instead of deleting them (ADR-007); the + identity modes, `manifest.sources` and the source-digest memo are gone, and a + clone is written only by an import. +- [ADR-003-result-cache-cost-rubric.md](../plans/ADR-003-result-cache-cost-rubric.md): + a cost-against-size cache rubric. **Proposed, not adopted.** The structural + cheap-or-worthy test it would remove still decides, now as ADR-008's + allow-list recorded in the manifest; `classify_build` and `ensure_result`, + which it names, are gone; ADR-007's bare-read chaining addressed its + motivating case; its budget and eviction half is still open. +- [ADR-004-result-digest-canonical-ordering.md](../plans/ADR-004-result-digest-canonical-ordering.md): + a canonically ordered snapshot and a digest of it, replacing the per-row + Python digest (#137). **Partly superseded:** the digest is a content digest of + the file read back (ADR-009), and the row-index column is `__row_order`, + written by every writer and appended to every sort (ADR-008). +- [ADR-005-intelligent-csv-import.md](../plans/ADR-005-intelligent-csv-import.md): + the CSV reader's schema and error contract. **Partly superseded** by ADR-008 + and ADR-011: the column is `__row_order`, the trailing `order_by` is gone, and + the reader runs inside `catalog_import_source` with its options fixed there; + the parsed rows are the source entry's snapshot, made again from the clone + when missing. The schema language, inference ladder and error contract are + unchanged. +- [ADR-006-read-path-loads-builds.md](../plans/ADR-006-read-path-loads-builds.md): + reads load the frozen build (#163). **Partly superseded** by ADR-007: ADR-006 + D4 (chaining inlines the parent's cache node) and ADR-006 D8 (the manifest + records a snapshot key that reads check) are retired, and ADR-006 D5 (the + canonical sort) is amended by ADR-008 and ADR-009. +- [ADR-007-tallyman-owned-materialization.md](../plans/ADR-007-tallyman-owned-materialization.md), + [ADR-008-row-order-of-reads.md](../plans/ADR-008-row-order-of-reads.md) and + [ADR-009-digest-stability.md](../plans/ADR-009-digest-stability.md): the cache + redesign that this doc and [caching.md](caching.md) describe. Tallyman writes + its own result files, every file carries `__row_order`, and a re-created file + is flagged only when the result changed. **Accepted (2026-09-22), implemented + in #189.** Each has an "Implementation notes" section saying where the code + differs from its text. +- [ADR-010-immutable-store-one-owner.md](../plans/ADR-010-immutable-store-one-owner.md): + an immutable result store, every entry materialized, one owning process. + **Rejected (2026-09-22)**; kept for the record. ADR-007, ADR-008 and ADR-009 + stand. +- [ADR-011-sources-are-aliases.md](../plans/ADR-011-sources-are-aliases.md): + a raw input is a source alias whose versions are entries, a file enters only + by `catalog_import_source`, recipes name aliases and never bare hashes, and + staleness has one axis. **Accepted (2026-09-22), implemented** in #217, #218 + and #219, merged into #189's branch. Its implementation notes record where the + code differs from the decisions. ### Plans (`plans/`) -- [native-catalog-store.md](../plans/native-catalog-store.md) — the native - `tallyman_core.catalog` that replaced xorq's catalog package. **Mostly current.** -- [recalc-mechanism.md](../plans/recalc-mechanism.md) — how reactive recalc - works. **Mostly current.** -- [auto-recalc-on-revise.md](../plans/auto-recalc-on-revise.md) — atomic - auto-recalc on revise. **Partially stale:** line numbers and a couple of - function names drifted (`_recalc_walk` → `_replay_cone`), and the "future - Stage C" frontend SSE listener already shipped. -- [remove-ondemand-result-parquet.md](../plans/remove-ondemand-result-parquet.md) - — removing the on-demand `result.parquet` layer (#104). **Current.** -- [project_switcher.md](../plans/project_switcher.md) — the project switcher. - **Mostly current.** -- [89-determinism-prereqs-execution.md](../plans/89-determinism-prereqs-execution.md) - — clearing #89's determinism prerequisites. **Mostly current.** -- [catalog-xorq-integration-tests.md](../plans/catalog-xorq-integration-tests.md) - — coexistence/reset integration tests. **Partially stale:** references the old - `catalog.yaml` / `aliases.json` formats since replaced by JSONL. -- [llm-summary-stats.md](../plans/llm-summary-stats.md) — LLM-authored summary - stats. **Partially stale:** the Buckaroo-side klass pattern it describes - (`_Generated_*` classes) is now a `@stat()` decorator; a few signatures and - the notify `kind` differ. - -### Research notes / experiment logs (`plans/`, `demo/`) — point-in-time records - -The digest investigation behind #137 (current): +- [native-catalog-store.md](../plans/native-catalog-store.md): the native + catalog store that replaced xorq's catalog package. **Mostly current:** + `compute_cache.jsonl` and the reset's compute-cache prune are gone (ADR-007). +- [recalc-mechanism.md](../plans/recalc-mechanism.md): how reactive recalc + works. **Partly stale:** auto-recalc on revise now exists, and new data + arrives by an import, which advances a source alias like a revise; there is + no source axis. +- [auto-recalc-on-revise.md](../plans/auto-recalc-on-revise.md): atomic + auto-recalc on revise. **Implemented; partly stale:** function names drifted + (`_recalc_walk` is `_replay_cone`), and its "future Stage C" SSE listener + shipped. +- [remove-ondemand-result-parquet.md](../plans/remove-ondemand-result-parquet.md): + removing the on-demand `result.parquet` layer (#104). **Partly stale:** the + single materialized copy is now tallyman's snapshot, not xorq's cache + (ADR-007). +- [project_switcher.md](../plans/project_switcher.md): the project switcher. + **Mostly current:** the home root is `~/.tallyman-notebooks/`, and there is + no Buckaroo session file (ADR-007). +- [89-determinism-prereqs-execution.md](../plans/89-determinism-prereqs-execution.md): + clearing #89's determinism prerequisites. **Point in time;** the digest and + heal verification it describes were replaced by ADR-007 and ADR-009. +- [cache-soundness-audit.md](../plans/cache-soundness-audit.md): an inventory + of the #163 bug class, 2026-07-30. **Point in time;** #167 and #189 changed + several findings. +- [catalog-xorq-integration-tests.md](../plans/catalog-xorq-integration-tests.md): + coexistence and reset integration tests. **Partly stale:** it refers to the + old `catalog.yaml` and `aliases.json` formats. +- [llm-summary-stats.md](../plans/llm-summary-stats.md): LLM-authored summary + stats. **Partly stale:** Buckaroo's `_Generated_*` classes are now a `@stat()` + decorator, a few signatures and the notify `kind` differ, and the stats are + not found by Buckaroo (#170). + +### Research notes and experiment logs (`plans/`, `demo/`, `docs/research/`) + +Point-in-time records. The digest investigation behind #137: [datafusion-scan-order-findings.md](../plans/datafusion-scan-order-findings.md) -— why a parallel datafusion scan emits rows in a different order each run, and -the polars-ingest decision — and +(why a parallel DataFusion scan emits rows in a different order each run, and +the decision to ingest with polars, which stands for CSV: an import parses a CSV +with polars, and pyarrow copies a parquet file in file order) and [result-digest-vs-xorq-staleness.md](../plans/result-digest-vs-xorq-staleness.md) -— why `result_digest` can't reuse xorq's content-aware cache staleness. +(why `result_digest` cannot reuse xorq's cache staleness). Both have a status +note for what #189 changed. Older, historical: [eda-prompt-research.md](../plans/eda-prompt-research.md), @@ -442,23 +1017,25 @@ Older, historical: [plotting-testcases.md](../plans/plotting-testcases.md), [xorq-sklearn-assessment.md](../plans/xorq-sklearn-assessment.md), [ds-demo-scripts.md](../plans/ds-demo-scripts.md), -[demo/datasets.md](../demo/datasets.md). These are historical; staleness mostly -doesn't apply, except where they assert current system behavior (a few reference -the removed `result.parquet` and the old `~/.tallyman/` path). +[demo/datasets.md](../demo/datasets.md), and the CSV importer research under +[docs/research/csv-importers/](research/csv-importers/). Some of them mention +the removed `result.parquet`, xorq's result cache and the old `~/.tallyman/` +path; `ds-demo-scripts.md` has a status note saying so. -### Root & meta +### Root and meta -- [README.md](../README.md) — V0 spike overview and run instructions. **Current.** -- [proposal.md](../proposal.md) — the talk pitch. **Current** (it's a pitch, not - a spec). +- [README.md](../README.md): what tallyman is, and how to run it. **Current.** +- [tallyman_explanation.md](../tallyman_explanation.md): the owner's framing of + tallyman, a feature list and a walkthrough. **Current** in the sections + written from the code; the owner's own sections are his and are not checked. +- [proposal.md](../proposal.md): the talk pitch. **Current** (a pitch, not a + spec). -The original `plan.md` (V0.6 plan) and `TICKETS.md` (V0 punchlist) were removed -as stale cruft — they predated the React SPA migration, the `result.parquet` -removal, and the JSONL catalog format. This doc supersedes them as the -architecture reference. +The original `plan.md` (V0.6 plan) and `TICKETS.md` (V0 punch list) were removed +as stale: they predated the React SPA, the `result.parquet` removal and the +JSONL catalog format. This doc replaces them as the architecture reference. > Gaps: there is no dedicated reference for the REST API, the CLI, or the -> frontend SPA architecture. (The MCP tool surface and the authoring extension -> points it exposes — display klasses, summary stats, post-processing — are now -> covered in [mcp-server.md](mcp-server.md).) See the project's gap tracking for -> the current list. +> frontend. The MCP tool surface, including the extension points it exposes +> (display klasses, summary stats, post-processing), is covered in +> [mcp-server.md](mcp-server.md). diff --git a/docs/caching.md b/docs/caching.md index 5ccb7498..3bb34527 100644 --- a/docs/caching.md +++ b/docs/caching.md @@ -3,14 +3,29 @@ Tallyman sits on two other projects that each cache aggressively: xorq (deferred expression execution) and Buckaroo (the dataframe viewer). This doc maps every cache in the three layers — what work each one saves, what -its key is, and when it is invalidated. Bottom-up: xorq first, since -tallyman's persistence is built on it. +its key is, and when it is invalidated. Bottom-up: xorq first, then what +tallyman does with it, then Buckaroo. + +Tallyman does not use xorq's cache for results. No build contains a xorq +cache node (ADR-007 D1, builds carry no cache nodes): xorq stays the layer +that builds, hashes, loads and executes expressions, and the files that hold +results are tallyman's own. The xorq section is here because Buckaroo's stat +cache (the summary statistics Buckaroo computes for each column of a grid) is +built on it, and because it explains what the design replaced. + +Terms follow [architecture.md](architecture.md#terms). An **entry** is one +catalog computation, stored under its **content hash**, xorq's hash of the +entry's expression. Every file the expression reads is a snapshot named by the +content hash of the entry it holds, and a **source entry** (one version of an +imported file) is hashed from the md5 of its bytes and its reader options, so +the hash also covers the bytes of every input and the identity of every parent. The dominant pattern everywhere is content-addressing: keys are derived from immutable inputs (an expression's structure, an entry's content -hash), so entries never go stale and "invalidation" is usually a space -decision, not a correctness one. The exceptions are called out as they -appear and collected at the end. +hash, an imported file's digest), so entries never go stale and "invalidation" +is usually a space decision, not a correctness one. The exceptions are +called out as they appear and collected at the end. Where an open issue says +tallyman does not yet behave as described, the paragraph says so and cites it. ## xorq: the expression cache @@ -45,8 +60,9 @@ Snapshot's blindness to data changes is deliberate (reproducible build artifacts keyed on what the expression *is*, not on what the source file happened to contain), but it is the one real footgun in the stack: `ParquetSnapshotCache`'s own docstring notes it does not re-key when -source files change. Tallyman uses snapshot everywhere and gets away with -it because catalog entries are immutable by construction; see below. +source files change. Tallyman names every file a recipe reads by a content +hash, so xorq's path-only keys are content-honest without tallyman using +either strategy; see below. ### Storage backends @@ -74,218 +90,513 @@ would mask filesystem changes within one long-lived process. ## tallyman -Tallyman uses xorq's cache machinery wholesale — strategy, key -computation, hit/miss logic, atomic writes are all xorq's. The only -override is *where* the files land: every cache constructor is passed a -`base_path`/`cache_dir` inside the project's catalog directory instead of -the global `~/.cache/xorq`. Per-project placement makes the cache travel -with the project and lets `reset-to` manage it; a shared global directory -would break both the isolation and the reset semantics. - -### Result cache (`src/tallyman_xorq/result_cache.py`, `source_cache.py`) - -Tallyman decides what to materialize at *build* time and bakes the decision -into the entry's recipe, so every later read takes the same path. Before a -build, `rewrite_for_build` (`source_cache.py:89`) rewrites the submitted -expression in three steps: reject in-memory reads, inject a source-read cache -after each non-parquet file read, and — for an expensive expression — wrap the -whole thing in a top-level result cache. Both kinds of cache node are -`ParquetSnapshotCache`s whose storage resolves to the per-project compute -cache at load time (xorq's `load_expr(cache_dir=…)` rewrites every node's base -path), so there is no separate `result_cache/` directory; both land under -`/artifacts/catalog/compute_cache/`. - -- **Source-read cache** — every non-parquet file read (`read_csv`, - `read_json`) gets a `.cache()` node injected immediately downstream - (`source_cache.py:142`). The snapshot key is the read's path only, so every - entry reading the same source hits the same cached parquet. Always on, - independent of result-cache worthiness. `read_parquet` / `read_delta` are - exempt (`_EXEMPT_READS`): re-reading a columnar source is already a pushdown. -- **Baked result snapshot** — an expression is *worthy* when it contains an - Aggregate / Join / Sort / window / UDF (`_EXPENSIVE_OPS`, - `result_cache.py:49`). A worthy expression is first put in a canonical total - order (`_canonical_sorted`: the author's own `order_by` keys, then - `original_row_order`, then the remaining sortable columns — ADR D5, amended) - so the baked bytes are deterministic across the build and every later heal, - then wrapped in a result cache with `relative_path="result_cache"`, so its - snapshot lands under `compute_cache/result_cache/`; executing the entry at - build time materializes it once. A non-parquet read is *not* by itself - worthy (`classify_build`, `result_cache.py:62`) — its parse is already - handled by the source-read cache. Worthiness is decided by two predicates that - must agree: `_is_worthy_expr` gates the bake (`source_cache.py:62`) and - `classify_build` records `cache_worthy` in the manifest. `_is_worthy_expr` - matches a UDF by class ancestry (`type(node).__mro__`, `source_cache.py:84`) - and `classify_build` by the serialized op name containing `UDF` - (`result_cache.py:85`; a scalar UDF serializes as `op: ScalarUDF`); the two now - agree for a scalar UDF, so a scalar-UDF-only entry is worthy under both and - bakes its snapshot instead of recomputing the whole graph on every read (#81). - -A cheap entry — a source read plus projections / renames / row-wise scalar -math — bakes no result snapshot. Recompute costs about the same as reading a -copy (a pushdown over a columnar source, or over its already-cached parse), so -an extra copy would burn storage for no savings. No `result.parquet` is -written for any entry, cheap or expensive, at build time or on demand (#73 and -follow-up). - -Every consumer reads an entry's result through one function, -`cached_result_expr`, whose internals are the canonical read of -`docs/system-contract.md` (#163): expand the entry's frozen `xorq_build/`, -`load_expr(expanded, cache_dir=compute_cache)`, then the deep cache-dir -rewrite (`portable.rewrite_cache_dirs`). A missing or unloadable build is a -hard error naming the entry — reads never re-import `expr.py` (ADR D6): - -- Expensive entry → `deferred_read_parquet` of the baked snapshot, on the - default backend. The snapshot was written when the entry built, so reading it - skips re-running the graph; the read asserts its derived snapshot key matches - the one the manifest recorded (ADR D8). If the file was evicted since, the - *frozen build* is executed once and the read proceeds — a self-heal - re-checked on every call, single-flighted per `(project, content_hash)` so - concurrent cold readers don't both run the shared op; the healed bytes are - verified against the recorded `result_digest` before they are served, and a - mismatch wipes the entry's Buckaroo stat cache, records a durable - `unfaithful_heal` error, and evicts its Buckaroo session (`_verify_self_heal`; - ADR D7/D10/D12, #79, #83). -- Cheap entry → the loaded build's graph, rebound onto the default backend and - recomputed on read from its content-pinned sources. - -Either shape is a single-backend expression, so two entries compose (`union`, +Tallyman writes its own result files. It decides at build time whether an entry +is worth materializing, writes the file once when the entry is created, makes +sure every file an entry reads exists before anything executes, and lets a data +file in only by an import, as a source entry whose snapshot carries a row-order +column. xorq's own cache is not in that path. ADR-007 +(`plans/ADR-007-tallyman-owned-materialization.md`) records the problems that +led there: cache files landing under `~/.cache/xorq` outside the project, a +writer that races on a fixed temporary file, and a build that re-ran an +expensive parent whenever a child was built. + +### Worthy and cheap entries (`src/tallyman_xorq/worthiness.py`) + +A **worthy** entry is one tallyman materializes: it writes the result to a +parquet file when the entry is created, and every read afterward reads that +file. A **cheap** entry writes nothing: its small plan re-runs on every read, +over files that exist. The verdict is computed once, at build, on the +expression the author wrote (`classify_expr`) and recorded in the manifest as +`cache_worthy`, with a short `cache_worthy_why` such as `ops:Aggregate` or +`values:Unnest`. `result_cache.cache_worthy` reads it back from the manifest; +nothing re-derives it from the serialized build. + +An entry is cheap only if all of these hold: + +- every relation operation in it is a file read, a filter, a column selection + or computed column (which covers rename and cast), a column drop, a drop of + null rows or a fill of nulls; +- it reads exactly one file; +- no value operation in it multiplies rows (`unnest`), depends on the order + rows arrive in (a window function, which covers `row_number` and `lag`), or + is not pure (`random()`, `uuid()`, `now()`, `today()` and any UDF). + +Everything else is worthy: an aggregate, join, sort, limit, union, distinct, +sample, and any operation nobody has considered yet. The test is an allow-list +because a cheap entry pages by the `__row_order` of the one file it reads (see +Row order below), so a wrong "cheap" gives unstable paging and a wrong "worthy" +costs a copy. A UDF is matched by its base class, so a scalar UDF makes an entry +worthy without any other expensive operation (#81). + +### Snapshots (`src/tallyman_xorq/materialize.py`) + +A worthy entry's **snapshot** is +`/result_cache/.parquet` (`snapshot_path`), so its +name is a function of the content hash and nothing else. `materialize` writes +the snapshot of every computed entry, used by the build and by every heal of one +(a **heal** re-creates a snapshot that is missing from disk); a source entry's +snapshot is written by the import's writer (`source_import._write_snapshot`), at +import and when it is healed from its clone (see "Source entries" below). +`materialize`: + +- it loads the entry's frozen build and rebinds every backend in it onto a + fresh single-partition connection (`single_partition_backend`, with + `target_partitions = 1` and an explicit `batch_size`), so a float aggregate + merges its partial sums in one order on any machine; +- it streams the result as record batches, drops `__row_order` and ibis's + `__row_order_right` if present, and appends a new `__row_order` counting + `0..N-1` as the last column; +- it regroups the stream into row groups of 1,048,576 rows, each combined into + contiguous arrays, and writes zstd level 3 in parquet format 2.6, with + statistics and a page index; +- it writes to a unique temporary name in the destination directory and then + replaces the destination with `os.replace`, under the project's write lock. + +A create always runs the query, and what it writes replaces whatever file is at +the path. A create of a worthy entry runs it twice through the writer and +compares the content digests of the two files (`check_reproducible`), since at +create time nothing is recorded to compare against. If they differ the build +still succeeds, and the manifest records `reproducible: false` with the columns +that differed. The file is then **pinned**: the Cache page will not delete it, +since it cannot be re-created faithfully. A heal runs the query once, replaces +the file at once, and checks the result against the recorded digest. + +A create does not replace the file straight away. It calls +`materialize(..., publish=False)`, which leaves the finished file at its +temporary name, and the build moves it into place (`publish_snapshot`) as its +last step, after the manifest is written. A build that fails before then removes +only its temporary file, so a file already at the path is kept, such as one a +reset left on disk, which for an entry that is not reproducible is the only copy +of its rows (#193, fixed in #222). A source entry needs no such staging: the +import keeps a snapshot already at its path, which holds the same rows, since +the path is named by the bytes and the reader options. + +**Pins.** `pinned_reason` decides from the entry's manifest and, for a source +entry, whether its clone is on disk. A snapshot is pinned when the manifest says +`reproducible: false`, when it holds `unfaithful_heal_digest` (the digest an +unfaithful heal wrote, below), or when it is a source entry whose clone is gone +(for a retired one, from both `data/.cas/` and the bullpen). Because the first +two are part of the manifest, the pin moves with the entry through a reset and +survives the error banner's dismiss, which deletes `errors.jsonl` (#196, fixed +in #223). For a file whose entry a reset retired, the manifest parked in the +bullpen speaks for it (`snapshot_manifest`), so the pin holds while the entry is +retired (#195). + +The row-group size and the batch size decide the batch boundaries that an entry +built on the file sees, and an ungrouped float total depends on them (#187), so +they are part of the reproducibility contract. `SNAPSHOT_FORMAT_VERSION` stands +for both, and for the row groups of a source entry's snapshot; the manifest +records it as `snapshot_format`, next to the xorq, xorq-datafusion and pyarrow +versions in `engine_versions`. Changing any of them is a corpus rebuild. + +`result_digest` in the manifest is `arrow-sha256:`, a SHA-256 over the +file's Arrow data read back (`src/tallyman_xorq/digest.py`). It does not depend +on how the rows were batched, on the codec, the row-group size, the writer's +version, or whether a text column is `string` or `large_string`. It does depend +on every value, on which slots are null, on the order of the rows, and on the +column names and types. Per-column digests name the columns that differ +between two runs. + +### Source entries (`src/tallyman_xorq/source_import.py`) + +A recipe never reads a data file. A file enters the catalog only through +`catalog_import_source` (`update_and_depend`), which makes it a **source +entry**, a version of a **source alias** (ADR-011, +`plans/ADR-011-sources-are-aliases.md`). The import: + +- digests the file (md5) and clones its bytes, copy-on-write where the + filesystem supports it and a plain copy elsewhere, to + `/data/.cas/`, the **clone**. The copy is digested + again after it is written and refused if it does not match its name; +- writes the entry's snapshot, `/result_cache/.parquet`, + holding the file's rows in file order plus a last column `__row_order`. A + parquet file is copied by pyarrow, so its column types survive; a CSV is parsed + by polars under the schema and `scan_csv` options named in the import call, + and its batches go to the same pyarrow writer. Either way the file has the + pinned layout of every snapshot, in row groups of 122,880 rows; +- writes the entry: a generated recipe, a frozen build, a schema, and a manifest + whose `provenance` records the outside path, the digest, the reader options and + the name it was imported as. + +The entry's content hash is `md5("source||")`, cut to +12 hex characters, so it is a function of the bytes and the reader options and +nothing else. The reader options are fixed at import and must be plain values: a +callable, whose `repr` would carry a memory address, is refused. The outside path +is provenance and is never read again, so editing or deleting the original file +changes nothing. New data arrives by importing again under the same alias, which +mints the next version under a new hash; the old versions keep their rows. + +A source entry is worthy, and its snapshot is cache in the sense of ADR-007 D13 +(a file is cache only if it can be made again): the clone holds the bytes and the +manifest holds the reader options, so a deleted snapshot is written again from +the clone (`_heal_a_source`) and checked against the recorded `result_digest` +like any other heal. Nothing in a read writes a clone. With the clone gone, the +snapshot is the last copy of those rows, so `pinned_reason` pins it; if the +snapshot is gone as well, the read fails with an error naming the missing clone +and the `catalog_import_source` call, reader options included, that restores +it. Importing the same bytes again under the alias that holds them rewrites +nothing of the entry. When the snapshot is gone, it writes the clone back from +the given file, verified against the digest, and heals the snapshot. When the +snapshot is still there, it leaves a lost clone lost, so the version stays +pinned (#239). + +Every other way of reading a file is a build error: `read_project_file`, +`tallyman_read_csv`, `xo.deferred_read_csv`, and `xo.deferred_read_parquet` of a +file outside `compute_cache/`. `read_project_file` is still called by the recipe +the importer generates, where a context variable (`_SOURCE_ENTRY`) resolves it to +the entry's own snapshot. + +### Reads (`cached_result_expr` and `ensure_materialized`) + +Every in-process consumer (page reads, charts, diffs, post-processing, a child +recipe) reads an entry's result through one function, `cached_result_expr`, +whose internals are the canonical read of `docs/system-contract.md` (#163): the +entry's frozen `xorq_build/` is expanded and loaded, and a read never re-imports +`expr.py`. (Only the attribution of an unfaithful heal re-imports it, as a +diagnostic.) Buckaroo's grid is handed a build instead; see "Buckaroo +integration caches" below. A missing or unloadable build is a hard error naming +the entry. `cached_result_expr` calls `ensure_materialized(project, +content_hash)` and then returns: + +- for a worthy entry, one bare `deferred_read_parquet` of the snapshot on the + default backend, memoized per `(project, content_hash)`. When the file exists + the entry's build is not loaded at all; +- for a cheap entry, the loaded build's graph rebound onto the default backend. + +`ensure_materialized` guarantees that every file the entry's plan reads, and +its own snapshot, is on disk before anything executes: + +1. A worthy entry whose snapshot exists is done, and no build is loaded. +2. A source entry whose snapshot is missing is healed from its clone, and + nothing else is needed: its own build reads the very file that is missing. +3. Otherwise it loads the frozen build and collects the file each `Read` node + points at. Every one is another entry's snapshot, and a missing one is made + again by recursing on the hash in its name. +4. For a worthy entry it then heals the entry's own snapshot, under the + project lock and after re-checking that the file is still missing, and + verifies it. + +Whether an entry is worthy comes from its manifest (`cache_worthy`), and nothing +stands in for it: a file at the snapshot path says nothing about the verdict. An +entry directory with no manifest, which a build or import killed before its last +write leaves, is refused. `result_cache.entry_manifest` raises a `BuildError` +naming the missing `manifest.json` before anything is loaded or written, a child +whose build reads that entry's snapshot raises the same error, and running the +recipe again (or importing the file again, for a source entry) writes the entry +again under the same hash. + +| File | Written by | If it is missing | +|---|---|---| +| Snapshot of a computed entry, `compute_cache/result_cache/.parquet` | `materialize` | re-run the entry's build, and verify the digest | +| Snapshot of a source entry, same directory | the import | parse the clone again with the recorded reader options, and verify the digest; with the clone gone too, raise an error naming the clone and the import that repairs it | +| Clone, `data/.cas/` | the import (`ensure_cas_path`) | nothing in a read makes it again; the snapshot is pinned while it exists; a re-import of the same bytes writes the clone back only when the snapshot is gone too (#239) | + +A healed snapshot is checked against the recorded `result_digest`. A mismatch is +still served, since the rows are the honest output of the frozen build, but +never silently (`_verify_self_heal`). It logs a warning that attributes the +change: an engine version that differs from the manifest's `engine_versions`, a +recipe that re-derives a different graph hash (#88), or a fixed graph that runs +differently each time (#83). It records the digest it wrote in the manifest's +`unfaithful_heal_digest`, which pins the file, records a durable +`unfaithful_heal` error for the error banner, wipes the entry's Buckaroo stat +cache, and fires the registered hooks. That field is the only one written after +create: an unfaithful heal is the one thing that rewrites a manifest, +atomically, under the heal's lock. In the companion the hook posts a forced +reload of the entry's grid to Buckaroo and publishes an `unfaithful_heal` SSE +event. The SPA has no listener for that event; the error appears in the catalog +page's error banner the next time the page refetches. All of this runs while the +heal still holds the project lock, and the forced reload is posted whether or +not a grid is open, without a promoted diff's colouring (#203). Only the healed +entry is flagged: a cheap child of it reads the same snapshot, so the child's +rows change under its hash with no record and no reload (#208). The MCP server +registers no hook, so a heal that runs there records the pin and the error and +wipes the stat cache only. A source entry's snapshot made again from its clone +goes through the same check, so if a reader now parses the bytes differently, +the difference is recorded as an unfaithful heal and the manifest keeps the +digest of the rows that were imported. + +Both shapes are single-backend expressions, so two entries compose (`union`, `join`, a diff) without tripping xorq's "multiple backends" guard. Chaining -(`tracked_expr_from_alias`) uses the sibling `entry_graph_expr` instead: the -parent's cache node stays in the graph, so a child's build is self-contained -and self-healing (ADR D4). The viewer's paginated reads, diffs, and -post-processing all go through `cached_result_expr`; nothing reads a -pre-existing `result.parquet`, because none is written. - -Snapshot strategy is the right one here because an entry is an immutable, -content-addressed artifact — its result must not invalidate just because an -upstream file's mtime drifts. No TTL: entries are permanent history, not -expiring scratch. - -How a source file's content reaches the key is set by `TALLYMAN_SOURCE_IDENTITY` -(`source_identity.py:52`). The default is `cas`: `read_project_file` reads through a -content-addressed clone at `/data/.cas/` (`io.py:58`), so the -path xorq tokenizes embeds the content digest and every xorq-level key — build -hash and snapshot keys alike — is content-honest, and a rebuild over an edited -source forks the hash instead of deduping to the stale entry. `off` is the -historical path-only mode: no digest, no clone, an edit collides with the prior -entry. `salt` is the remaining exception to baking: it folds source digests into -the entry hash but leaves xorq's own keys path-only, so a baked snapshot's -path-only key would collide across entries with identical paths but different -content. Under salt, `rewrite_for_build` bakes no cache at all -(`source_cache.py:130`) and every read recomputes. +(`tracked_expr_from_alias`, `pinned_expr_from_alias`) returns +`cached_result_expr`. A child of a worthy parent therefore reads the parent's +snapshot by its literal path, which contains the parent's content hash: the +child's identity is a function of the parent's, and does not change when the +file's bytes do. The parent's snapshot is made to exist before the child is +built, since a bare read cannot be composed over a missing file. A cheap +parent's graph is inlined instead. The child's manifest records only its parent +edges: a source version is an entry, so the parent edges are the whole record of +what the child depends on (ADR-011 D6 deleted `manifest.sources`, the per-file +digest map children used to inherit). The viewer's paginated reads, diffs, and +post-processing all go through +`cached_result_expr`. + +### Row order (`src/tallyman_xorq/row_order.py`) + +Every file tallyman writes ends in an int64 `__row_order` column holding +`0..N-1` in the file's physical row order, and every page request sorts by it: +`ORDER BY __row_order`, or the user's sort keys and then `__row_order`, so the +same request returns the same rows in any process and any cache state +(`row_order.page`, which `/api/data` uses; that route takes no user sort yet). +Buckaroo pages the grid in its own process. Tallyman tells it the column's name +with `row_order_column` in the `/load_expr` payload, but Buckaroo 0.15.6, the +pinned version, ignores the hint, so the grid's pages are not yet ordered by it +(buckaroo-data/buckaroo#974). The build enforces what makes the rule safe: + +- A cheap entry has no file of its own and pages by its parent's column, so it + must keep it. A select that omits it fails the build with the corrected + select in the message, and tallyman moves the column to the last position at + the top of the expression. A worthy entry may drop it, since the writer + numbers its rows. +- Assigning to `__row_order` is an error. A copy under another name, such as + `__row_order_v1`, is ordinary data. +- Every `order_by` in a recipe gets `__row_order`, and then the remaining + sortable columns, appended as tie-breakers, so a sort that feeds a `limit` + decides the same rows on any connection. A sort that is not the last step is + hoisted: the top-level sort leads with its keys. If a key was dropped or + overwritten, the build fails and names it. Nested columns (lists, structs, + maps) are left out of the tie-break, so two rows that tie on every other + column can still come out in either order (#205). +- Every entry carries the column, and ibis names a join's right-hand copy + `__row_order_right`, so joining three entries in one recipe needs + `.drop("__row_order")` on the right-hand inputs. The check counts right-hand + inputs without looking at the join kind, so a chain of semi or anti joins, + which adds no right-hand columns, is refused too (#199). The snapshot writer + drops `__row_order_right`, so a join entry can be joined again; it drops any + column of that name, including one the author made, and a join with a + non-default `rname` leaves its own collision column behind (#206). +- The diff grid's compare expression drops the column from both sides, but + `full_diff`, which computes the diff page's summaries and `catalog_diff`'s + reply, keeps it as a data column (#200). The primary-key search skips it. ### Compute cache (`compute_cache_dir` in `src/tallyman_core/paths.py`) -`/artifacts/catalog/compute_cache/` is where every cache node's -storage resolves at load time, so it holds three things: the source-read -parses, the baked result snapshots (under `result_cache/`), and any -sub-expression results xorq caches along the way. Build and view both load -against it (`build.py` and `load_entry` pass `cache_dir=compute_cache_dir`), so -a freshly added expression computes there cold and warms it for later reads. - -Its distinguishing feature is reset-awareness. The warm set is recorded in -`compute_cache.jsonl` (one `{"path": relpath}` per line), a git-tracked pointer -file captured at each checkpoint; `reset_to` -(`src/tallyman_core/catalog_state.py:327`) prunes or restores the cache to -match the target step. Evicted files move to `bullpen/` rather than being -deleted, so a forward reset restores them instead of recomputing, while a -re-added entry still computes honestly cold. `reset_to` also reclaims orphaned -`cas` clones: `/data/.cas` lives outside the catalog git repo, so -`git reset` can't touch it, and `_gc_cas_clones` (`catalog_state.py:349`, -`source_identity.gc_cas`) deletes any `.cas` file no surviving entry's -`manifest.sources` still references (#86). +`/artifacts/catalog/compute_cache/` holds one directory of files that +tallyman writes: `result_cache/`, the snapshots of worthy entries, source +entries included. By rule everything in it is cache: a file lives here only if +`ensure_materialized` can re-create it and check what it made, so the cold state +is an empty `compute_cache/`, and reading any entry then re-creates every file +it needs. Three kinds of snapshot are exceptions, and are pinned (see "Pins" +above): the snapshot of an entry recorded as not reproducible, which cannot be +made again faithfully; one whose heal already produced different rows than were +built (`unfaithful_heal_digest`); and the snapshot of a source entry whose clone +is gone, which is the last copy of the imported rows. A pin protects the file +from the Cache page and from nothing else (#185). + +Files are deleted only by an explicit user action, and written only because +something is about to read them. The startup warm-up writes nothing here, the +verify sweep (`catalog_scan_staleness(verify_results=True)`) reads and never +writes, and a reset leaves the directory alone. The Cache page's delete is the +one deleter. The page answers 409 with the reason for a pinned snapshot. It +lists a snapshot whose entry a reset retired as a `retired` row, whose pin comes +from the manifest parked in the bullpen, and a snapshot that no entry names, +live or retired, as an `orphan` row, so that the user can delete either. A +source entry's snapshot is listed like any other. + +The project's write lock (`catalog_state.project_lock`) is taken by a build +(for its whole length), an import, a materialization or heal, a checkpoint and +a reset. Other writes take no lock: alias and notebook updates, +chart and display configs and `config.json` each replace a whole file +atomically, the logs append a line, and the Cache page's delete unlinks the +file. The lock is a file +lock on `.checkpoint.lock`, so it holds between the MCP server and the +companion, and it is re-entrant within a thread. It blocks with no timeout: a +page read whose entry needs a heal waits behind any build in the other process +(#186), and the companion's `PUT /code` and `POST /promote_diff` routes build on +the event loop, so the whole UI stops answering while they wait or build (#190). +It covers no reads. Executions have a lock of their own, +`execution.execution_lock`, one re-entrant lock per process, held around every +execution on the process's shared default backend, which fails with +`RuntimeError: Already borrowed` when two threads execute on it at once. A heal +runs before that lock is taken, since the project lock comes first, and +`project_lock` raises in a thread that holds the execution lock and would take a +new file lock. A materialization's stream and a cheap entry's row count at build +run on connections of their own and take no execution lock. + +`reset_to` (`src/tallyman_core/catalog_state.py`) returns the catalog with +`git reset --hard` and reconciles entry directories through the bullpen (the +directory a reset moves retired files into, so that a later reset forward can +bring them back), but it does not manage `compute_cache/` (ADR-007 D14, a +reset leaves `compute_cache/` alone). Snapshots are named by content hash, so +one left behind by a retired entry cannot be served for another entry: it is +unreferenced disk until that entry comes back or the user deletes it. A clone +is data, the only copy tallyman has of bytes it imported (the outside file is +never read again), and `/data/.cas` lives outside the catalog git repo. +So `reset_to` moves the clones that no surviving source entry's +`manifest.provenance` names into `/bullpen/cas/` and never deletes one +(`source_identity.gc_cas` with a bullpen), and a reset forward copies back the +clones a restored entry names. A reset that cannot read every surviving +manifest skips the sweep. `compute_cache.jsonl` no longer exists. + +When a reset retires an entry whose directory the bullpen already holds (an +entry retired once, restored, and retired again), the live directory replaces +the parked one, since it is the one that agrees with the snapshot on disk: a +create always rewrites the snapshot, and an entry that is not reproducible +records another `result_digest` each time (#194, fixed in #223). A live +directory with no manifest, which an interrupted build leaves, never replaces a +parked one and is dropped instead. The bullpen has one live reader besides +`reset_to`: the Cache page reads a retired entry's parked manifest to decide its +snapshot's pin, and for a retired source version it counts a clone parked in +`bullpen/cas/` as present, since a reset forward brings the entry and the clone +back together (#195). ### In-memory caches in the companion -All bounded LRUs over immutable keys, so eviction means a cheap rebuild -and staleness is impossible: - -- `cached_result_expr` — its build-loading work is memoized on - `_resolve_result_plan`, `lru_cache(256)` keyed `(project, content_hash)`; - `cached_result_expr` re-exposes that memo's `cache_clear` / `cache_info`. - The key finally determines the value — the plan is a pure function of the - immutable build — and the memo saves expanding + loading the build and - re-deriving its baked-snapshot path. `cached_result_expr` itself is a thin - wrapper that re-checks the snapshot's on-disk presence on every call, so an - evicted snapshot self-heals; the heal runs under a per-entry lock - (`_heal_lock`) so only the first of several concurrent cold readers executes - it, and a peer process that wins xorq's fixed-`.tmp` rename is caught and - retried once if the snapshot landed, else re-raised (#79). +The first two are bounded LRUs over immutable keys, so eviction means a cheap +rebuild and staleness is impossible; the third is a short TTL: + +- `cached_result_expr` — two memos. `_resolve_result_plan` is + `lru_cache(256)` keyed `(project, content_hash)`: it holds a loaded build, + its graph rebound onto the default backend, and the list of files the build + reads. `_snapshot_read` is `lru_cache(1024)` and holds the bare read of a + snapshot, so one read has one table name in the shared backend. + `cached_result_expr.cache_clear()` clears both. Whether each file exists is + checked on every call, inside `ensure_materialized`, since existence is the + one input that stays mutable. The plan memo also keeps the build as it was + loaded, which nothing reads again, so up to 256 unused DataFusion backends + stay in memory (#210). The MCP server has the same two memos. - `_build_compare_expr` — `lru_cache(128)` keyed `(project, a_hash, b_hash, keys)`; saves rebuilding diff outer-join - expressions (`src/tallyman_companion/app.py:189`). Build dirs land under - `$TMPDIR/tallyman_diff_builds/`. + expressions. Build dirs land under `$TMPDIR/tallyman_diff_builds/`. - Disk-usage payload — per-project, 3-second TTL (`_DISK_USAGE_TTL` in `src/tallyman_companion/app.py`). The only time-based cache in tallyman; it coalesces filesystem walks during SSE bursts. -On companion startup, a 3-second warmup budget pre-populates -`cached_result_expr` for as many entries as fit; the rest warm lazily. +On companion startup a 3-second warmup budget loads the frozen builds of the +active project's cheap entries into the plan memo, so the first page request +does not pay for the load. A worthy entry is skipped, since reading it does not +load its build. The warmup writes nothing under `compute_cache/`; loading a +cheap entry's build can write its `.xorq_build_expanded/`. ### Buckaroo integration caches (`src/tallyman_companion/buckaroo_lifecycle.py`) -- **Session map** — `buckaroo_sessions.json` in the tallyman home - (`TALLYMAN_HOME`, default `~/.tallyman-notebooks/`) maps content hash to - Buckaroo session id, saving a `/load_expr` POST per entry per restart. - Invalidated when Buckaroo's `/health` reports a new `started` timestamp - (restart detected, map cleared) or when an entry's parquet is evicted. -- **Per-entry stat cache** — `/.buckaroo_stat_cache/parquet/`, a +Tallyman keeps no record of Buckaroo's sessions (ADR-007 D6, Buckaroo is handed +something that already exists). A **session** is one grid's state inside the +Buckaroo process, and its id is `entry--` +(`BuckarooManager.session_id_for`), a function of the two and nothing else, so +it is never stale. There is no session file. + +- **Opening a grid**: `load_session` calls `ensure_materialized` first, so a + failure of the computation surfaces in tallyman and never inside a grid + query, then POSTs `/load_expr` with the derived id every time. Buckaroo + skips the work when it already holds a session with that id and the same + build directory and the post carries none of `component_config`, + `column_config_overrides`, `extra_grid_config`, `init_sd` or + `skip_stat_columns`. It creates the session again if it has dropped it, as it + does after a session has been idle for an hour. Nothing coalesces two opens + of one entry, so opens that overlap both post and both run Buckaroo's + pipeline; a promoted diff entry sends `column_config_overrides`, so it + re-runs the pipeline on every open; and the notebook page's data route posts + for every cell on every page load (#202). +- **What Buckaroo is handed**: for a worthy entry, a **view build**, a build + whose whole graph is one read of the entry's snapshot, written once to + `/.xorq_view_build/` and rebuilt if the snapshot's path changes (a + sibling `.xorq_view_build.complete` file records the path it was made for). + Buckaroo never executes an aggregate, join or sort on tallyman's behalf and + never writes a snapshot. For a cheap entry, the entry's own expanded build, + a small plan over files that exist. Both directories are stable per-entry + paths, so Buckaroo is handed the same build after a restart, which is what its + on-disk stat cache relies on. The body also + carries `row_order_column`, the name `__row_order`, which Buckaroo 0.15.6 + ignores, and `project_root`, where Buckaroo looks for the project's `stats/`, + `post_processing/` and `display/` klasses. Tallyman sends `artifacts/`, + which holds `display/`, but it writes stats and post-processing functions + under `artifacts/catalog/`, so Buckaroo does not find them (#170). +- **After an unfaithful heal**: `_verify_self_heal` wipes the entry's stat + cache, and the companion's hook POSTs `/load_expr` for the entry's id with + `force_reload: true`, so an open grid does not keep stats computed from the + old rows. The post goes out whether or not a grid is open, which opens a + session nobody asked for, and it carries no column colouring (#203). +- **After a klass change**: a klass is a summary stat, post-processing or + display class written for the project. `reload_project_sessions` POSTs + `/reload_expr/` for every entry of the project, treats the 404 (or 400) + that Buckaroo answers for an unknown session as "not open", and clears the + stat cache of each grid it reloaded. That is one request per entry per + change, sent one after another from the companion's event loop (#201), and + the stat-cache wipe after each reload is more than a klass change needs + (#177). A reset or a recalc that moved an alias reloads sessions the same way. +- **Per-entry stat cache**: `/.buckaroo_stat_cache/parquet/`, a `ParquetSnapshotCache` Buckaroo writes its summary stats into (the companion passes the path at `/load_expr` time). Deleted wholesale on - stat reload so Buckaroo recomputes. Always on, orthogonal to the - result-cache worthiness rubric. -- **Diff stat cache** — `/artifacts/catalog/diff_stat_cache/` + stat reload so Buckaroo recomputes. Always on, orthogonal to whether an entry + is worthy or cheap. In practice a first load after a Buckaroo restart has not + been seen to hit it (#157). +- **Diff stat cache**: `/artifacts/catalog/diff_stat_cache/` `{a_hash[:12]}-{b_hash[:12]}/`, the same idea for a comparison session, keyed by the entry pair. Re-opening the same diff reuses the per-column - stats instead of recomputing over the full join. + stats instead of recomputing over the full join. A reset or a recalc that + moved an alias deletes the whole directory. The live diff still posts an + unmaterialized join to Buckaroo (ADR-007 D10, which would have built every + diff as an entry first, was moved to #188). Its session ids, + `diff--`, are remembered in the companion's memory and + forgotten when the companion starts a Buckaroo whose `/health` reports a new + `started` timestamp. ### Per-entry immutable records -Not caches in the eviction sense — immutable build outputs, valid forever -because the entry they describe never changes. `paths.py:99` draws the line -explicitly: `ENTRY_ARTIFACT_NAMES` (`xorq_build/`, `manifest.json`, -`schema.json`) are the immutable artifacts; the write-isolated perf overlay -symlinks them read-only — safe because nothing ever rewrites them — and omits -`ENTRY_CACHE_NAMES` (`.buckaroo_stat_cache`, `.xorq_build_expanded`) so a -benchmark starts honestly cold. - -The entry directory itself is gitignored. Its durable, git-tracked form is the -recipe zip `entries/.zip` — a deterministic archive of `expr.py`, -`xorq_build/`, `manifest.json`, and `schema.json`, written by the checkpoint -(`tallyman_core/catalog.py`). So the build dir is untracked-but-durable: the -recipe zip carries it across a clone, and `entries.jsonl` records which dirs -should exist so `reset_to` can reconcile them from the bullpen. (See the native -catalog store, `catalog.py` / `catalog_state.py`, for the full tracked surface.) +Not caches in the eviction sense — build outputs, valid forever because the +entry they describe never changes. `paths.py` draws the line explicitly: +`ENTRY_ARTIFACT_NAMES` (`xorq_build/`, `manifest.json`, `schema.json`) are the +artifacts; the write-isolated perf overlay symlinks them read-only — safe +because the only later write, an unfaithful heal recording +`unfaithful_heal_digest` in `manifest.json`, is an atomic replace that swaps +the overlay's link for a file and leaves the original alone — and omits +`ENTRY_CACHE_NAMES` (`.buckaroo_stat_cache`, `.xorq_build_expanded`, +`.xorq_view_build`) so a benchmark starts honestly cold. + +The entry directory itself is gitignored. What the catalog repository tracks for +it is the recipe zip `entries/.zip`, a deterministic archive of `expr.py`, +`xorq_build/`, `manifest.json`, and `schema.json`, written once, by the first +checkpoint after create (`tallyman_core/catalog.py`), so its manifest does not +carry an `unfaithful_heal_digest` recorded later. Nothing reads the zip back +(`catalog.py`): it records what a checkpoint committed, and a clone of the +catalog repository does not recreate entry directories from it. `entries.jsonl` +records which dirs should exist so `reset_to` can reconcile them from the +bullpen. (See the native catalog store, `catalog.py` / `catalog_state.py`, for +the full tracked surface.) - **Primary key** (`src/tallyman_xorq/primary_key.py`) — `/primary_key.json` saves a full-table cardinality scan. A cheap row-preserving entry with no cached key inherits its parent version's - key (if the columns still exist) without scanning. + key (if the columns still exist) without scanning. The search skips + `__row_order`, which is unique in every table. - **Portable build expansion** (`src/tallyman_xorq/portable.py`) — `/.xorq_build_expanded/` plus a `.complete` marker, written last, gates reuse; a crashed expansion redoes on next access. The - expansion must live at a stable path: xorq's snapshot key includes the - read path, so a random tmp dir would change the expression hash and - bust the result cache on every process restart. (Content-addressed but - regenerable, so it is an `ENTRY_CACHE_NAME`, not an artifact — the overlay - omits it rather than symlinking it.) + expansion lives at a stable path so that Buckaroo is handed the same build + directory after every process restart, which its on-disk stat cache was + meant to rely on (#157). (Content-addressed but regenerable, so it is an + `ENTRY_CACHE_NAME`, not an artifact — the overlay omits it rather than + symlinking it.) The marker does not record which project path the build was + expanded with, so a project copied together with these directories keeps + reading the old location, and fails once that is gone (#209). A worthy + entry's build is first expanded at create, when `materialize` loads it; a + cheap entry's the first time something loads it (a read, the grid, a child's + build, the companion's startup warm-up). - **Manifest / schema** — `/manifest.json`, `/schema.json`: row counts, timings (including #87's cache-admission fields: - `compile_seconds`, `cache_worthy`, `cache_bytes`), the #83 `result_digest` - (a content hash of the executed result bytes), and the schema, all - derived from the expression at build so later reads don't re-walk it. + `compile_seconds`, `cache_worthy`, `cache_worthy_why`, `cache_bytes`), the + `result_digest` (a content digest of the snapshot, worthy entries only), + `reproducible` and `nonreproducible_columns`, `snapshot_format` and + `engine_versions`, `parents`, `unfaithful_heal_digest` once an unfaithful + heal has pinned the file, and on a source entry `provenance` (where the file + came from, its digest, the reader options and the name it was imported as). + A worthy entry's schema is read from the file `materialize` wrote, which is + why a `timestamp[s]` column is recorded as `timestamp[ms]`. Every entry's + schema ends in `__row_order`. All of it is fixed at build so later reads + don't re-walk the expression. - **Alias history** — `/aliases.jsonl`, one line per alias holding - its current head and the append-only version log (`aliases.py`). It is a + its current head, the append-only version log and its kind, `catalog` or + `source` (`aliases.py`). It is a git-tracked file in the catalog repo, not a per-entry artifact, so it rolls back with `git reset` on a `reset_to`. - **Per-hash config** — `/chart_specs/.vl.json` and `/display_configs/.json` hold an entry's chart and display config, keyed by content hash. These are mutable (set or cleared from the UI), not build outputs. Because the key is the hash, a revise mints a new hash and - would orphan them, so `carry_forward_entry_config` (`entry_config.py:21`) + would orphan them, so `carry_forward_entry_config` (`entry_config.py`) copies each from the prior version unless the new version already defines its - own — called from `catalog_revise` (`server.py:572`) and the companion's + own — called from `catalog_revise` (`server.py`) and the companion's `put_code`. Post-processing and summary stats are not carried: they are project-global (keyed by name, not hash) and already apply to every version (#109). @@ -342,43 +653,60 @@ catalog store, `catalog.py` / `catalog_state.py`, for the full tracked surface.) ## Invalidation, in one view -Most of the stack never invalidates because it never can be stale: xorq -snapshot keys, tallyman entry hashes, primary-key files, and manifests -all rely on "same key, same bytes, forever". Cleanup for those is a -space concern (manual delete, bullpen moves on reset), and a deleted file -self-heals by recomputing. +Most of the stack never invalidates because it never can be stale: tallyman +entry hashes (a source entry's named by its bytes and reader options), snapshot +paths (named by content hash), primary-key files, and manifests all rely on +"same key, same rows, forever". Cleanup for those is a space concern. +Only the user deletes a file, and a deleted file is made again and verified +the next time something reads it. One documented hole breaks "never stale": execution nondeterminism. An entry whose recipe calls `now()` / `random()` / an unseeded `sample()` or an impure -UDF produces different bytes each run under one content hash, so a cold -recompute can disagree with what was built (#88). The build flags these as -advisory lint warnings (`_nondeterminism_warnings`, `build.py`); it does not -block them. The runtime backstop is `result_digest`: every self-heal verifies -the repopulated snapshot against it before serving (`_verify_self_heal`), and -an unfaithful heal wipes the entry's stat cache, records a durable error, and -evicts its Buckaroo session (ADR D7/D10/D12). The second hole this section -used to document — cold reads re-running `expr.py` and re-digesting live -sources, serving edited bytes under the original hash (#115/#163) — is closed: -reads load the frozen build, whose leaves are content-pinned `.cas` clones and -inlined parent graphs, so a cold read cannot see a post-build edit at all. +UDF produces different rows each run under one content hash, so a cold recompute +can disagree with what was built (#88). The build flags `now()`, `today()`, +`random()`, `uuid()` and an unseeded `sample()` as advisory lint warnings +(`_nondeterminism_warnings`, `build.py`); an impure UDF is not flagged, since +purity cannot be read off the graph. A worthy entry is also run twice when it is +created, so a recipe that is not reproducible is known from the start: the +manifest records `reproducible: false`, the build result names the columns that +differed, and the snapshot is pinned. That check cannot see `today()`, since +both runs agree (the lint does), or what an entry inherits from a +non-reproducible parent, since both runs read the same parent file (#185). The +runtime backstop is `result_digest`: every heal verifies the repopulated +snapshot against it before serving (`_verify_self_heal`), and an unfaithful heal +pins the file in its manifest, wipes the entry's stat cache, records a durable +error, and, in the companion, forces Buckaroo to reload the entry's grid. It +does not reach the cheap entries that read the healed snapshot, whose rows +change with it (#208). An engine or writer upgrade that changes results is +attributed to the versions in `engine_versions`, and the remedy is a corpus +rebuild. The second hole this section used to document — cold reads re-running +`expr.py` and re-digesting live sources, serving edited bytes under the original +hash (#115/#163) — is closed: reads load the frozen build, whose leaves are bare +reads of snapshots named by content hash and inlined cheap parent graphs, and a +data file enters only by an import, so a cold read cannot see a post-build edit +at all. Where staleness is actually possible, it is handled explicitly: -- **mtime tracking** — xorq's modification-time strategy and Buckaroo's - file cache invalidate automatically when the source file changes. -- **TTL** — xorq's `ParquetTTLStorage` (default one day) and the - companion's 3-second disk-usage cache are the only clocks in the - system. +- **mtime tracking** — Buckaroo's file cache invalidates automatically when the + source file changes. Tallyman keeps no stat-keyed memo: a file is digested + once, when it is imported, and the staleness scan reads no file at all. +- **TTL** — xorq's `ParquetTTLStorage` (default one day), the + companion's 3-second disk-usage cache, and Buckaroo's eviction of a session + idle for an hour are the only clocks in the system. - **State-change purges** — Buckaroo's op-chain key change, the JS - `purgeInfiniteCache` on sort/ops change, the session map clearing on - Buckaroo restart, and the stat-cache deletion on reload. - -The one rule to remember: snapshot-strategy caches do not notice upstream data -changes. The `cas` default makes that safe end-to-end: an edited source forks -the entry hash at build rather than deduping to the stale one, and reads — -warm, cold, or self-healing — resolve through the frozen build to the -content-addressed clone the entry was built from, never the live file. `off` -reverts to xorq's path-only identity (an edit collides silently, and a loaded -build reads whatever bytes sit at the recorded path); `salt` mixes digests -into the entry hash but bakes no snapshot cache at all (path-only keys would -collide across salted entries) and recomputes on every read. + `purgeInfiniteCache` on sort/ops change, the diff-session bookkeeping + clearing on Buckaroo restart, the stat-cache deletion on a klass reload + or an unfaithful heal, and the deletion of `diff_stat_cache/` on a reset or + a recalc. + +The one rule to remember: a cache keyed on a path does not notice upstream data +changes, so tallyman puts the content in the path. Every file a recipe reads is +a snapshot named by a content hash, and a source entry's hash is a function of +the bytes it imported and the reader options. New data arrives only as an +import, which mints a new source version under a new hash and makes the entries +following that alias stale, and reads — warm, cold, or healing — resolve through +the frozen build to the snapshot the entry was built from, never the outside +file. The clone is what lets `ensure_materialized` make a deleted source +snapshot again; with the clone gone the snapshot is pinned, and once both are +gone the error names the clone and the import that restores it. diff --git a/docs/expression-lifecycle.md b/docs/expression-lifecycle.md index 86c23d13..b946c8af 100644 --- a/docs/expression-lifecycle.md +++ b/docs/expression-lifecycle.md @@ -7,26 +7,36 @@ Two walkthroughs: a brand-new expression (everything cold), then the same expression on a later visit (warm). It is a companion to `caching.md`, which maps the caches statically; this doc puts them on a timeline. -The one fact that makes the timeline make sense: an entry is written to disk -at *build* time, and so is the only materialized copy of its result — an -expensive entry's `.cache()` snapshot is baked during the build, not on first -read. Just one cache is populated *lazily*, at first view: Buckaroo's stat -cache on first `/load_expr`. So "build" and "first view" each write different -things, and a restart re-runs the view-time work against caches that survived -on disk. +The one fact that makes the timeline make sense: an **entry** (one catalog +computation, stored under its content hash) is written to disk at *build* +time, and so is the only materialized copy of its result. A **worthy** entry, +one whose plan does expensive work such as an aggregate or a join, has a +**snapshot** (a parquet file of its rows, which tallyman writes itself), and +the snapshot is written during the build, not on first read. A **cheap** entry, +a filter or selection over one file, keeps no copy of its rows. The rest is +written lazily, when the entry is first read or viewed: Buckaroo's stat cache +on the first `/load_expr`, the view build a worthy entry's grid is handed, a +cheap entry's expanded build, and the primary key a diff joins on. So "build" +and "first view" each write different things, and a restart re-runs the +view-time work against caches that survived on disk. Other terms follow +[architecture.md](architecture.md#terms). ## Cast | Layer | Code | Role | |---|---|---| | MCP server | `src/tallyman_mcp/server.py` | receives code + alias, calls the builder | -| Builder | `src/tallyman_xorq/build.py` | compiles, executes, persists the entry | +| Builder | `src/tallyman_xorq/build.py` | imports the recipe, materializes a worthy entry, persists the entry | | xorq compiler | `build_expr` / `load_expr` | tokenizes to a content hash, runs the graph | -| Result cache | `src/tallyman_xorq/result_cache.py`, `source_cache.py` | bakes an expensive entry's snapshot at build; `cached_result_expr` resolves reads | +| Import | `src/tallyman_xorq/source_import.py`, `source_identity.py` | `catalog_import_source`: clones a file's bytes and writes a source entry, whose snapshot ends in `__row_order` | +| Reads in a recipe | `src/tallyman_xorq/io.py` | `tracked_expr_from_alias`, `pinned_expr_from_alias`, and the refusals of a raw file read | +| Classifier and rewrite | `worthiness.py`, `source_cache.py`, `row_order.py` | worthy or cheap, the canonical sort, the `__row_order` rules | +| Materialization | `src/tallyman_xorq/materialize.py` | `materialize` writes snapshots; `ensure_materialized` makes every file an entry reads exist | +| Reads | `src/tallyman_xorq/result_cache.py` | `cached_result_expr`, the one canonical read | | State + git | `src/tallyman_core/catalog_state.py`, `catalog.py` | per-concern tracked files in tallyman's native catalog git repo; the checkpoint commits | | Companion | `src/tallyman_companion/app.py` | HTTP/SSE API the React SPA talks to | -| Buckaroo mgr | `src/tallyman_companion/buckaroo_lifecycle.py` | owns the Buckaroo subprocess + session map | -| Paths | `src/tallyman_core/paths.py` | single source of truth for every on-disk location | +| Buckaroo mgr | `src/tallyman_companion/buckaroo_lifecycle.py` | owns the Buckaroo subprocess and opens its sessions | +| Paths | `src/tallyman_core/paths.py` | the project's directory layout (`compute_cache/result_cache/` is named in `materialize.py`, and `data/.cas/` in `source_identity.py`) | Throughout, `` is `entry_dir(project, content_hash)` = `/artifacts/catalog/entries//`, and `` is @@ -40,82 +50,124 @@ Throughout, `` is `entry_dir(project, content_hash)` = The model calls one of three tools (`server.py`): -- `catalog_run(code, prompt)` — execute and persist an *anonymous* entry - (`server.py:203`). No alias. -- `catalog_create(name, code, prompt)` — persist a **named** entry - (`server.py:507`). Errors if the alias already exists (`get_alias` check at - `server.py:526`). +- `catalog_run(code, prompt)` — execute and persist an *anonymous* entry. No + alias. +- `catalog_create(name, code, prompt)` — persist a **named** entry. Errors if + the alias already exists (a `get_alias` check). - `catalog_revise(name, code, ...)` — update an existing alias to a new version. The new version's per-hash chart and display config are carried forward from the prior version unless it defines its own - (`carry_forward_entry_config`, `entry_config.py:21`; §3); the same applies to + (`carry_forward_entry_config`, `entry_config.py`; §3); the same applies to the companion's `put_code` code-edit path. The code is a Python string that must bind a variable `expr` to an -ibis/xorq expression, typically built from `tracked_expr_from_alias("alias")` (another -entry) or `read_project_file("file")` (a raw source). All three converge on -`_run_and_record` → `build_and_persist(project, code, prompt)` (`build.py:328`). - -## 2. Build + execute (`build_and_persist`, `build.py:328`) - -In order: - -1. **Import the user code** in a fresh module scope (`_import_script`, - `build.py:308`). During the import, `read_project_file` records each source file - it touches and `tracked_expr_from_alias` records each parent edge; - `source_identity` / `parent_capture` collect over the import - (`build.py:355-362`) for the `cas` (default) / `salt` identity modes and the - parent graph. -2. **Rewrite, then compile.** `rewrite_for_build` (`build.py:387`, - `source_cache.py:89`) rewrites the author's expression before it is built: - reject in-memory reads, inject a source-read cache after each non-parquet - read, and — when the expression is expensive — wrap it in a top-level result - cache. `build_expr(rewritten, builds_dir=)` (`build.py:398`) then - compiles into a throwaway temp dir whose name *is* the content hash — xorq's - dask-style token over the (rewritten) expression structure - (`content_hash = build_path.name`, `build.py:403`). The hash is - content-sensitive in two of the three identity modes. Under the `cas` default - each `read_project_file` read already went through a `data/.cas/` clone - (`io.py:58`), so the digest is baked into the path xorq tokenizes; under - `salt` the digests are folded in afterward (`si.salted_hash`, `build.py:407`) - and no cache is baked; `off` leaves the hash path-only. `expr.py` keeps the - author's literal source; `xorq_build/` carries the cache nodes. -3. **Idempotency check** (`build.py:414`): if `` already exists for this - hash, nothing is rebuilt — the prompt is appended as a re-run event and the +ibis/xorq expression built from other entries: `tracked_expr_from_alias("alias")` +follows an alias, and `pinned_expr_from_alias("alias-v2")` pins one version of +it. A recipe never opens a file. All three tools converge on `_run_and_record` → +`build_and_persist(project, code, prompt)`. The companion's code editor +(`PUT /{project}/api/code/{alias}`) calls `build_and_persist` directly, after +refusing a source alias. + +### Before this: the data is imported + +The data at the bottom of every chain entered the catalog earlier, through +`catalog_import_source(outside_path, alias, ...)` (`source_import.update_and_depend`, +ADR-011). That call is not a build of a recipe. It digests the file, clones its +bytes to `/data/.cas/` (checked against the digest after +the copy), writes the rows in file order with a last `__row_order` column to +`compute_cache/result_cache/.parquet`, and writes an entry +directory of its own: a generated `expr.py` whose header records the path, the +digest and the reader options, a frozen `xorq_build/`, `schema.json`, and a +`manifest.json` whose `provenance` says where the data came from. The content +hash is an md5 of the bytes' digest and the reader options. It then appends the +entry to the source alias, so the recipe above can read it with +`tracked_expr_from_alias("orders")`. The outside file is never read again. + +## 2. Build + execute (`build_and_persist`) + +The whole build holds the project's write lock (`catalog_state.project_lock`), +so builds, materializations, checkpoints and resets in either process happen one +at a time. In order: + +1. **Import the user code** in a fresh module scope (`_import_script`). During + the import, `tracked_expr_from_alias` and `pinned_expr_from_alias` resolve + each alias, record the parent edge (`parent_capture` collects them for the + manifest), and make the parent's files exist first (`ensure_materialized`), + because a child cannot be built over a snapshot that is missing. + `read_project_file` and `tallyman_read_csv` raise here, naming the import to + use, and so does a bare content hash or bare alias passed to + `pinned_expr_from_alias`. +2. **Check, classify, rewrite, compile.** The build refuses a raw + `xo.deferred_read_csv` and a raw `xo.deferred_read_parquet` of a file that + is not under `compute_cache/`, since such a read gets no digest and no + clone. `worthiness.classify_expr` then decides, once and on the + expression the author wrote, whether the entry is **worthy** (materialized) + or **cheap** (row-preserving over one file, no file of its own); the + verdict goes in the manifest. `rewrite_for_build` rejects in-memory reads, a + recipe that already contains a xorq cache node, an assignment to + `__row_order`, a join chain over three entries that all carry that column + (the error shows the `.drop("__row_order")` fix; chains of semi or anti + joins are refused too, #199), and a cheap entry that drops the column. It + gives a worthy + entry the canonical sort (the author's sort keys, then `__row_order`, then + the remaining sortable columns), and moves `__row_order` to the last + position of a cheap entry. `build_expr(rewritten, builds_dir=)` then + compiles into a throwaway temp dir whose name *is* the content hash: xorq's + dask-style token over the rewritten expression structure + (`content_hash = build_path.name`). The hash follows the content of every + input, because every path xorq tokenizes is a snapshot named by a content + hash, down to the source entries, whose hash is a function of their bytes. + `expr.py` keeps the author's literal source. +3. **Idempotency check**: if `/manifest.json` already exists for this + hash, nothing is rebuilt: the prompt is appended as a re-run event and the existing `BuildResult` is returned. (This is the cheap path that Part 2 - leans on.) -4. **Lay down the entry** (`build.py:430-450`): create `/`, copy xorq's - build output to `/xorq_build/`, rewrite project-root paths to the + leans on.) A directory with no manifest, left by a build that crashed, is + built again. +4. **Lay down the entry**: create `/`, copy xorq's build output to + `/xorq_build/`, rewrite project-root paths to the `${TALLYMAN_PROJECT_ROOT}` placeholder (`make_portable_inplace`), and write the author's source to `/expr.py`. -5. **Execute** (`build.py:459-496`): `load_expr(build_path, - cache_dir=compute_cache_dir(project))` binds the expression to the - **per-project compute cache**, then the entry is executed to force row-level - evaluation and record a content digest of the result (`result_digest`, #83). - This is the one place the expression is actually computed, and what it writes - depends on worthiness: an **expensive** entry runs `loaded.count().execute()` - (`build.py:491`), which materializes its baked result-cache snapshot under - `compute_cache/result_cache/`, then hashes the result (`result_digest(loaded)`, - `build.py:492`); a **cheap** entry streams the result once through - `count_and_result_digest(loaded)` (`build.py:496`), getting its row count and - digest in a single pass (an honest full evaluation that fails fast on a bad - cast / UDF, but caches no rows). No `result.parquet` is written either way. -6. **Derive metadata** (`build.py:497-561`): row count and `result_digest` come - from the execute above and schema from the expression - (`loaded.schema().to_pyarrow()`) — no parquet is read back. Write - `/schema.json` and `/manifest.json` (the manifest carries the - #87 admission fields `compile_seconds`, `cache_worthy`, `cache_bytes` — the - baked snapshot's size, or `None` for a cheap entry — plus the #83 - `result_digest`, a content hash of the executed bytes), and append the prompt - to `/prompts/.jsonl`. +5. **Execute.** This is the one place the entry's query runs, and what it writes + depends on the verdict. A **worthy** entry is materialized: `materialize` + loads the entry's frozen build (which writes the expanded build, + `/.xorq_build_expanded/`), rebinds it onto a single-partition + connection, and streams the rows through the snapshot writer, numbering + them in a last `__row_order` column. The file stays at a temporary name + beside `compute_cache/result_cache/.parquet` until step 6 + moves it into place. It runs the query twice and compares the content + digests of the two files, so a recipe that is not reproducible is known from + the start (`reproducible: false`, and the file is pinned). A **cheap** entry + streams its frozen plan once, through `stream_row_count`, which forces + row-level evaluation (a bad cast fails here, in tallyman's process; any UDF + makes an entry worthy, so a UDF fails in `materialize`) and yields the exact + row count, but keeps no rows. No `result.parquet` is written either way. If + anything from step 4 on fails, the build removes the entry directory it + created and its own temporary file, and nothing else: a file already at the + snapshot's path, such as one a reset left behind, stays as it was (#193, + fixed in #222). +6. **Derive metadata**: the row count and the `result_digest` (an + `arrow-sha256:` content digest of the snapshot, worthy entries only) come + from the execute above. A worthy entry's schema is read from the file + `materialize` wrote, since parquet changes some types (a `timestamp[s]` + column comes back `timestamp[ms]`) and the writer adds `__row_order`; a cheap + entry's schema comes from the expression, which already ends in that column. + Write `/schema.json` and `/manifest.json`. The manifest carries the verdict + (`cache_worthy`, `cache_worthy_why`), `compile_seconds` and `cache_bytes` + (the snapshot's size, or `None` for a cheap entry), `result_digest`, + `reproducible`, `snapshot_format`, `engine_versions` and `parents`, and is + the entry directory's last write, atomic, so its presence means the entry + is complete. (Only a source entry's manifest has `provenance`, written by + the import.) Then, for a worthy entry, `publish_snapshot` moves the staged + file into place with one atomic replace. That is the last write to the + entry's result, so the snapshot's path changes only once the entry is + complete. Last, the prompt is appended to `/prompts/.jsonl`. 7. **Mark persisted; the checkpoint commits.** `build_and_persist` sets - `catalog_registered = True` (`build.py:585`), meaning the entry dir is fully - on disk — it no longer writes git itself. Durability is the *checkpoint's* - job: when the MCP tool returns, the `_with_checkpoint` decorator that wraps - every tool at registration (`server.py:163`, via `_checkpointing_tool` at - `server.py:174`) calls `checkpoint_catalog` once, which under the per-project - lock zips any complete entry into `entries/.zip` and makes - a single git commit + step tag (`catalog_state.py:277`). There is no + `catalog_registered = True`, meaning the entry dir is fully on disk — it does + not write git itself. Durability is the *checkpoint's* job: when the MCP tool + returns, the `_with_checkpoint` decorator that wraps every tool at + registration (via `_checkpointing_tool`) calls `checkpoint_catalog` once, + which under the per-project lock zips any complete entry into + `entries/.zip` and makes a single git commit + step tag. There is no `_xcat_add`, no xorq catalog subprocess, and no out-of-lock writer — which is how #48 is resolved by construction. @@ -125,137 +177,184 @@ In order: |---|---|---| | Build recipe | `/xorq_build/` (portable) | step 4 | | User source | `/expr.py` | step 4 | -| **Baked result snapshot** (expensive only) | `/compute_cache/result_cache/.parquet` | the execute in step 5 | -| **Compute cache** (source parses + baked snapshots) | `/compute_cache/` | xorq, during `load_expr` / execute, step 5 | +| **Snapshot** (worthy only) | `/compute_cache/result_cache/.parquet` | `materialize` at a temporary name, step 5; `publish_snapshot`, step 6 | +| Expanded build (worthy only) | `/.xorq_build_expanded/` and its `.complete` marker | `materialize` loading the build, step 5 | | Schema / manifest | `/schema.json`, `/manifest.json` | step 6 | | Prompt history | `/prompts/.jsonl` | step 6 | | Chart / display config (if attached or carried forward) | `/chart_specs/.vl.json`, `/display_configs/.json` | UI, or `carry_forward_entry_config` on revise (§3) | -| Source digests (cas/salt only) | `/artifacts/source_digests.json` (stat-keyed memo) + `/manifest.json` `sources` map | `source_identity`, step 1 | -| Content source clones (cas only) | `/data/.cas/` (outside the catalog repo) | `read_project_file` → `ensure_cas_path`, step 1 | | Alias | `/aliases.jsonl` | `set_alias` (§3 below) | -| Recipe zip + pointers + commit | `/entries/.zip`, `entries.jsonl`, `compute_cache.jsonl` + a git commit | the checkpoint, step 7 | +| Recipe zip + pointers + commit | `/entries/.zip`, `entries.jsonl` + a git commit | the checkpoint, step 7 | -`xorq_build/`, `manifest.json`, and `schema.json` are `ENTRY_ARTIFACT_NAMES` -(`paths.py:99`) — immutable build outputs, partitioned from the regenerable -`ENTRY_CACHE_NAMES` (`.buckaroo_stat_cache`, `.xorq_build_expanded`). The perf +`xorq_build/`, `manifest.json`, and `schema.json` are `ENTRY_ARTIFACT_NAMES` in +`paths.py` — build outputs, partitioned from the regenerable `ENTRY_CACHE_NAMES` +(`.buckaroo_stat_cache`, `.xorq_build_expanded`, `.xorq_view_build`). The perf overlay symlinks the artifacts read-only and omits the caches, which is the -operational proof of the split: nothing ever rewrites an artifact. There is no -`result.parquet` in this list — none is written. The single materialized copy -of an expensive entry's rows is the baked `.cache()` snapshot under -`compute_cache/result_cache/` (§6), and a cheap entry keeps no copy at all. - -The entry directory is gitignored; its git-tracked durable form is the recipe -zip `entries/.zip` that the checkpoint commits (step 7). So the build dir -is untracked-but-durable — the recipe zip carries it across a clone, and -`entries.jsonl` records which dirs should exist so `reset_to` reconciles them -from the bullpen, never silently rewriting them. +operational proof of the split. One artifact is written again after create: an +unfaithful heal records `unfaithful_heal_digest` in `manifest.json`, by an +atomic replace that swaps an overlay's link for a file and leaves the original +alone. There is no `result.parquet` in this list — none is written. The single +materialized copy of a worthy entry's rows is its snapshot (§6), and a cheap +entry keeps no copy at all. + +The entry directory is gitignored; what the catalog repository tracks for it is +the recipe zip `entries/.zip` that the checkpoint commits (step 7). +Nothing reads the zip back (`catalog.py`): it records what a checkpoint +committed, and a clone of the catalog repository does not recreate entry +directories from it. `entries.jsonl` records which dirs should exist so +`reset_to` can reconcile them from the bullpen. What is **not** written yet — these are lazy, and that is the whole point of the timeline: - **`/.buckaroo_stat_cache/parquet/`** — Buckaroo's summary stats. Populated on the first `/load_expr`, by Buckaroo, at view time (§5). +- **`/.xorq_view_build/`** — for a worthy entry, the build Buckaroo is + handed (§5). Written on the first view. +- **`/.xorq_build_expanded/`** — for a cheap entry, its build with the + project's path filled back in. Written the first time something loads the + build: a page read, the grid, a child's build, or the companion's startup + warm-up. - **`/primary_key.json`** — written on the first primary-key scan. -The result-cache snapshot is deliberately *not* on this list: it bakes at -build, in step 5, not on first read (§6). +The snapshot is deliberately *not* on this list: it is written at build, in +step 5, not on first read (§6). ## 3. Alias registration + notify For a named entry, `catalog_create` calls `set_alias(project, name, hash, -expect_exists=False)` (`server.py:531`), which writes a line to the tracked -`/aliases.jsonl` — `{"alias", "latest", "history": [...]}` — and -appends a notebook cell. Aliases live in their own tracked file now, not in a -`catalog.yaml`; #103 made that possible by giving tallyman its own native -catalog repo (the old xorq catalog package rejected extra tracked files, which -is why aliases used to be smuggled into `catalog.yaml`). Then -`_notify("new_entry", content_hash=…, alias=…, version=…)` (`server.py:536`) -POSTs the companion's `/internal/notify`, which publishes an SSE event to any -connected SPA (`app.py`). The viewer's catalog list refreshes; nothing touches -Buckaroo yet. The git commit for all of this lands when the tool returns, at -the checkpoint (§2 step 7). +expect_exists=False)`, which rewrites the tracked `/aliases.jsonl` (one +`{"alias", "latest", "history": [...], "kind"}` line per alias) with the new +alias in it, and then appends a notebook cell. The kind is `catalog` here; +`set_alias` refuses a name that is already a source alias, and refuses to point +a catalog alias at a source entry. Aliases live in their own tracked file now, +not in a `catalog.yaml`; #103 made that possible by giving tallyman its own +native catalog repo (the old xorq catalog package rejected extra tracked files, +which is why aliases used to be smuggled into `catalog.yaml`). Then +`_notify("new_entry", content_hash=…, alias=…, version=…)` POSTs the companion's +`/internal/notify`, which publishes an SSE event (a message on the HTTP stream +each open browser tab holds) to any connected SPA (`app.py`). The viewer's +catalog list refreshes; nothing touches Buckaroo yet. The git commit for all of +this lands when the tool returns, at the checkpoint (§2 step 7), so the +notification goes out a moment before the commit. ## 4. The user clicks the new expression -The SPA issues `GET /{project}/api/entry/{hash-or-alias}` -(`api_entry_detail`, `app.py:534`). The handler: +The SPA issues `GET /{project}/api/entry/{hash-or-alias}` (`api_entry_detail`). +The handler: -1. Resolves an alias to its hash (`app.py:539-541`). +1. Resolves an alias to its hash. 2. Reads `manifest.json`, `schema.json`, `expr.py`, forensic history, chart - spec, and display config off disk (`app.py:548-585`) — all cheap. -3. Calls `buckaroo.ensure_session(content_hash, project, - column_config_overrides=…)` (`app.py:589`) — this is where Buckaroo - enters. -4. Returns the entry payload plus `buckaroo_session` (the session id) and - `buckaroo_ws_base` (the websocket base URL) (`app.py:599-614`). - -## 5. `ensure_session` → Buckaroo `/load_expr` (`buckaroo_lifecycle.py:479`) - -1. `_maybe_restart()` (revives a crashed subprocess, throttled); bail to - `None` if Buckaroo isn't running — the SPA then falls back to a - `pandas.to_html` preview. -2. **No pre-heal** (ADR D4): entry builds are self-contained — a - `tracked_expr_from_alias` chain keeps the parent's cache node in the child's - build — so Buckaroo's replay of the posted build regenerates any evicted - snapshot itself through ordinary cache mechanics on first query. (Before - #163's fix, chaining stripped the parent's cache node to a bare snapshot - read, and `ensure_session` had to pre-execute `cached_result_expr` here so a - cold grid wasn't empty; both the stripping and the pre-heal are retired.) -3. **Session-map check** (`buckaroo_lifecycle.py:535`+): for a new entry this - misses, so a session is created. -4. **Stable build expansion**: `ensure_expanded_build` materializes - `${TALLYMAN_PROJECT_ROOT}` in the entry's `xorq_build/` into - `entry_expanded_build_dir(project, hash)` — a **stable per-entry path**, - never a random tmp dir, gated by a `.complete` marker. The path matters - because xorq embeds the build-dir path in the expression hash that keys the - stat cache; a random path would miss the stat cache on every restart. -5. **Stat-cache dir**: `stat_cache = /.buckaroo_stat_cache`, + spec, and display config off disk — all cheap. +3. Returns that metadata plus `buckaroo_ws_base` (the websocket base URL). The + request never touches Buckaroo, and `buckaroo_session` is always `None`: + loading a grid can mean re-creating a missing file, which must not block the + page (#133). + +The grid loads separately. On the catalog page the data tab +(`BuckarooDataPane`) requests `GET /{project}/api/session/{hash}` +(`get_session`) as soon as the entry opens, shows a spinner with the elapsed +time while the request blocks, and then shows the grid, or the reason it failed +with a retry button (#133). The tab stays mounted, hidden, when another tab is +shown, so its websocket stays open. On the notebook page, `LazyBuckarooEmbed` +waits until a cell scrolls near the viewport, then polls the same route until it +returns a websocket URL. The notebook's data route (`/api/notebook_full`) has +also already posted `/load_expr` for every cell (#202). + +## 5. `load_session` → Buckaroo `/load_expr` (`buckaroo_lifecycle.py`) + +`get_session` calls `BuckarooManager.load_session`, which returns a typed status +(`ok`, `unavailable`, `no_build`, `timeout` or `error`) so the SPA can show a +precise message and a retry. + +1. `_maybe_restart()` (revives a crashed subprocess, throttled); if Buckaroo isn't + running the status is `unavailable`. +2. **Tallyman finishes its own work first.** `ensure_materialized` makes every + file the entry's plan reads, and its own snapshot, exist. Here everything is + already on disk, so for a worthy entry this is a manifest read and one `stat` + call and no build is loaded. A file that had to be re-created is verified + before it is served (§6). A failure here becomes status `error` with the + reason, in tallyman's process, and never surfaces as a failed grid query + inside Buckaroo. +3. **Pick what Buckaroo is handed.** For a **worthy** entry, `ensure_view_build` + writes a *view build* once to the stable directory + `/.xorq_view_build/`: a build whose whole graph is one read of the + snapshot. Buckaroo never executes the entry's aggregate, join or sort, and + never writes a snapshot. For a **cheap** entry, `ensure_expanded_build` + materializes `${TALLYMAN_PROJECT_ROOT}` in the entry's `xorq_build/` into + `entry_expanded_build_dir(project, hash)`, a **stable per-entry path**, never + a random tmp dir, gated by a `.complete` marker. Both paths are stable so + that Buckaroo is handed the same build directory after a restart, which is + what its on-disk stat cache is meant to rely on. +4. **Stat-cache dir**: `stat_cache = /.buckaroo_stat_cache`, `mkdir(exist_ok=True)` — preserves an existing cache, never wipes it. -6. **POST `/load_expr`** (`buckaroo_lifecycle.py:573`) with - `build_dir=` (the entry's own recipe build — the #102 viewer-read - build dir was removed in #104), `project_root=` (where - Buckaroo scans for project-authored stat/post-processing klasses), and - `cache_storage_path=`. Buckaroo loads the xorq expression, - creates a session, and returns its `session` id. -7. The session id is cached in `_sessions` (keyed by content hash, shared - across projects) and persisted to `buckaroo_sessions.json` in - `TALLYMAN_HOME` (default `~/.tallyman-notebooks/`). +5. **POST `/load_expr`** with `session=entry--`, + `build_dir` (from step 3), `project_root=`, + `cache_storage_path=`, and `row_order_column="__row_order"`. + Buckaroo looks for the project's klasses (its summary stats, + post-processing functions and display classes) in `stats/`, + `post_processing/` and `display/` under `project_root`; tallyman keeps + display classes in `artifacts/display/` but writes the other two under + `artifacts/catalog/`, so Buckaroo does not find them (#170). A promoted diff + entry adds `column_config_overrides`, and the companion adds `telemetry_url` + when it knows its own address. Buckaroo loads the xorq expression, creates + the session, and returns its `session` id. +6. Nothing is remembered. The session id is a function of the project and the + hash, so tallyman keeps no session map and no session file. Because the stat cache is empty for a new entry, Buckaroo computes summary -stats from scratch and **writes `/.buckaroo_stat_cache/parquet/`** — -this is the cold population. With buckaroo 0.15.0 + `BUCKAROO_PERF=1` this -shows up in `~/.buckaroo/logs/server.log` as `misses=N`, `snapshots -written>0`, and a `firstpull.summary_stats secs=…` span. +stats from scratch and **writes `/.buckaroo_stat_cache/parquet/`**; +this is the cold population. Buckaroo logs one line per run for it +(`xorq stat cache […]: N hit(s), M miss(es), K snapshot(s) written …`) to its +own `~/.buckaroo/logs/server.log`, and posts its per-load timings to the +companion as `firstpull.*` spans (kept in `artifacts/telemetry.jsonl`, shown when +a `buckaroo` event is expanded in the Log tab). The subprocess's stderr goes to +`/buckaroo.log`, in the project `tallyman run` started with. ## 6. The websocket and first data pull -The SPA opens `ws://…/ws/{session}` against `buckaroo_ws_base` (also served -by `GET /{project}/api/session/{hash}`, `app.py:617`, which is just -`ensure_session` + the ws URL). Over the socket Buckaroo streams the first -row window and the summary-stats payload. The JS `SmartRowCache` holds row -segments client-side from here on (see `caching.md`). +The SPA opens `ws://…/ws/{session}` from the URL the session route returned. +Over the socket Buckaroo streams the first row window and the summary-stats +payload. The JS `SmartRowCache` holds row segments client-side from here on (see +`caching.md`). Buckaroo pages in its own process. Tallyman named `__row_order` +as the `row_order_column`, but Buckaroo 0.15.6, the pinned version, ignores that +hint, so the grid's pages are not yet ordered by it +(buckaroo-data/buckaroo#974). -The **result cache** is read through one function, `cached_result_expr`, whose +The **result** is read through one function, `cached_result_expr`, whose internals load the entry's frozen build (the canonical read of -`docs/system-contract.md`) — and, unlike the stat cache, it was already -populated at build (step 5), so a read never writes it except to self-heal an -eviction. For an expensive entry (`classify_build` worthy — Aggregate / Join / -Sort / window / UDF) it returns a `deferred_read_parquet` of the baked snapshot -under `compute_cache/result_cache/`, on the default backend, after asserting -the derived snapshot key matches the one the manifest recorded (ADR D8); if -the snapshot was evicted it executes the frozen build once to repopulate, then -reads. That heal is single-flighted per `(project, content_hash)` under -`_heal_lock`, so concurrent cold readers don't both execute the shared op and -trip DataFusion's "Already borrowed", and a peer process losing xorq's -fixed-`.tmp` rename retries once if the snapshot landed (#79). The healed -bytes are verified against the `result_digest` recorded at build before they -are served; a mismatch records a durable `unfaithful_heal` error, wipes the -entry's stat cache, and evicts its Buckaroo session (`_verify_self_heal`; ADR -D7/D10/D12). A cheap entry returns the loaded build's graph rebound onto the -default backend, recomputed on read from content-pinned sources. Every reader -takes this path: the paginated viewer (`api_data` calls -`cached_result_expr(...).limit(...).execute()`), both sides of a diff, and -post-processing. +`docs/system-contract.md`). Unlike the stat cache, the snapshot was already +written at build (step 5), so a read never writes one except to re-create a +snapshot the user deleted. `cached_result_expr` calls `ensure_materialized` +first, and then: + +- for a worthy entry, returns one `deferred_read_parquet` of the snapshot under + `compute_cache/result_cache/`, on the default backend, memoized per + `(project, content_hash)`. When the file exists the entry's build is not + loaded at all; +- for a cheap entry, returns the loaded build's graph rebound onto the default + backend, recomputed on read over files that exist. + +If a snapshot is missing, `ensure_materialized` re-creates it under the project +lock, after re-checking that it is still missing, by running the frozen build +once through the same writer, and verifies the result against the +`result_digest` recorded at build. Missing files it reads are made first, each a +parent's snapshot, by recursing on the hash in its name. A source entry's +snapshot is made again from its clone with the reader options in its manifest, +and checked the same way; if the clone is gone too, the error names it and the +import call that repairs the version. A mismatch is still served, but never +silently: `_verify_self_heal` pins the file by recording the digest it wrote in +the manifest's `unfaithful_heal_digest`, records a durable `unfaithful_heal` +error for the error banner, wipes the entry's stat cache, and fires the hooks. +In the companion the hook posts a forced reload of the entry's grid to Buckaroo +and publishes an `unfaithful_heal` SSE event, which the SPA has no listener for; +the error shows in the catalog page's error banner. The checks and the hook run +while the heal holds the project lock (#203), and cheap entries that read the +healed snapshot are not flagged (#208). + +Every reader takes this path: the paginated viewer (`api_data` pages the entry +with `row_order.page`, `ORDER BY __row_order` and then `LIMIT/OFFSET`, so the same +request returns the same rows), charts (which fetch up to 100,000 rows through +`api_data`), both sides of a diff, chaining a child recipe, and post-processing. --- @@ -266,76 +365,97 @@ Three distinct "warm" scenarios, because they hit different caches. ## A. Re-submitting identical code (build idempotency) `build_and_persist` runs `build_expr` again, gets the **same content hash**, -finds `` already present (`build.py:414`), appends the prompt, and -returns the existing `BuildResult`. No execution, no bake, no new compute-cache -writes. The expression is never recomputed when its structure (and, under salt -mode, its source content) is unchanged. +finds `` and its manifest already present, appends the prompt, and +returns the existing `BuildResult`. No execution, no materialization, and no new +file. The expression is never recomputed when its structure and its parents are +unchanged. A changed data file changes nothing here until it is imported again: +the import mints the next version of its source alias, the entries that follow +the alias go stale, and recalc replays their recipes into new entries (at once, +when auto-recalc is on, as it is by default). The old entries keep the rows they +were built from. Importing unchanged bytes is itself a no-op. ## B. Revisiting an entry in the same companion process (RAM-warm) -`GET /api/entry/{hash}` re-reads the small JSON/`expr.py` files (cheap), then -`ensure_session` finds the hash in `_sessions` and **returns the existing -session id without calling `/load_expr`** (`buckaroo_lifecycle.py:535-539`). -No Buckaroo recompute, no stat-cache touch. The SPA reuses the same -websocket. This is the fastest path. +`GET /api/entry/{hash}` re-reads the small JSON/`expr.py` files (cheap). The +session route then runs `ensure_materialized` (a manifest read and a `stat` for a +worthy entry) and POSTs `/load_expr` again with the same derived id. Buckaroo +still holds that session, with the same build dir, and the post carries none of +the config-bearing fields, so it answers from the session it holds. No Buckaroo +recompute, no stat-cache touch. The grid connects to the same session again. +Two exceptions (#202): a promoted diff entry's post carries its +`column_config_overrides`, so Buckaroo re-runs its pipeline for it on every +open, and two opens of one entry at the same moment both post and both run it. ## C. Revisiting after a restart (disk-warm, RAM-cold) -When Buckaroo restarts, the companion detects a new `started` timestamp from -`/health` and clears the in-RAM session map -(`_reset_session_bookkeeping_if_restarted`, `buckaroo_lifecycle.py:302`) — -but it touches nothing on disk. So on the next visit: +When the companion starts (or restarts) the Buckaroo subprocess, it reads the +`started` timestamp from `/health` and, if it is new, clears its in-RAM +bookkeeping of diff sessions (`_reset_session_bookkeeping_if_restarted`). It +touches nothing on disk, and there is no entry-session record to clear. So on +the next visit: -1. `ensure_session` misses the (cleared) session map → POSTs `/load_expr` - again. -2. `ensure_expanded_build` finds the **stable expanded build** already - present with its `.complete` marker → reuses it, no re-expansion. +1. `load_session` POSTs `/load_expr`, and Buckaroo, which no longer holds the + session, creates it. +2. The **stable expanded build** (cheap entry) or **view build** (worthy entry) + is already present with its marker → reused, no re-expansion. 3. Buckaroo is handed the same `cache_storage_path`, finds `.buckaroo_stat_cache/parquet/` populated, and the `ParquetSnapshotCache` **hits** — summary stats are read from disk, not recomputed. (Warm signal: `hits>0, misses=0` in `server.log`; `firstpull.summary_stats secs=Y` with `Y ≪` the cold time.) -The cache survives the restart precisely because (a) nothing in the restart -path deletes it, and (b) the expanded-build path is stable, so the snapshot -key is identical across processes. - -The only thing that *invalidates* the stat cache is a klass-changing event — -a summary-stat / post-processing / display-config change or a project reset, -which routes through `reload_project_sessions` → `_clear_stat_cache` -(`buckaroo_lifecycle.py:378, 431`). A plain restart is not such an event. - -Similarly, the baked **result snapshot** for an expensive entry survives the -restart untouched — it is content-addressed under `compute_cache/` — so +The cache is meant to survive the restart because (a) nothing in the restart +path deletes it, and (b) Buckaroo is handed the same build over the same files, +so it computes the same stat-cache keys. In practice a first load after a +restart has not been seen to hit it (#157). + +The only things that *invalidate* the stat cache are a klass-changing event (a +summary-stat, post-processing or display change), a project reset, a recalc +that moved an alias, and an unfaithful heal. All but the last route through +`reload_project_sessions`, which POSTs `/reload_expr/` for every entry of the +project, one after another (a 404 or 400 means the grid is not open; #201), and +clears the stat cache of each grid it reloaded (`_clear_stat_cache`). A plain +restart is not such an event. + +Similarly, the snapshot of a worthy entry survives the restart untouched — it is +named by the entry's content hash under `compute_cache/` — so `cached_result_expr` returns its `deferred_read_parquet` without recomputing. -It re-checks the snapshot's presence on every call and recomputes only if the -file was evicted (`result_cache.py:451-473`, single-flighted under `_heal_lock`); -a Buckaroo restart never evicts it. +`ensure_materialized` re-creates it only if the user deleted the file; a Buckaroo +restart never does. --- # One-line summary of cache write phases -- **Build time** (`build_and_persist`): the baked result snapshot under - `compute_cache/result_cache/` (expensive entries only) plus any source-read - caches, `schema.json` / `manifest.json`, source digests (cas/salt) and the - `data/.cas/` content clones (cas only), `expr.py`, and the prompt - history. +- **Import time** (`catalog_import_source`, before any recipe reads the data): + the clone under `data/.cas/`, the source entry's snapshot under + `compute_cache/result_cache/`, and its entry directory (`expr.py`, + `xorq_build/`, `schema.json`, `manifest.json` with `provenance`); then the + source alias in `aliases.jsonl`. +- **Build time** (`build_and_persist`): `xorq_build/` and `expr.py`, the snapshot + under `compute_cache/result_cache/` and the expanded build (worthy entries + only), `schema.json` / `manifest.json`, and the prompt history. +- **Naming** (right after the build, in the tool): `aliases.jsonl` via + `set_alias`, and the notebook cell. - **Checkpoint** (when the tool returns, `checkpoint_catalog`): the recipe zip - `entries/.zip`, the `aliases.jsonl` / `entries.jsonl` / - `compute_cache.jsonl` tracked surface, and one git commit + step tag. -- **First view** (`ensure_session` → `/load_expr`): `.buckaroo_stat_cache/`, - the stable expanded build, the Buckaroo session (RAM + `buckaroo_sessions.json`). + `entries/.zip` and `entries.jsonl`, and one git commit + step tag that + also takes in `aliases.jsonl` and the rest of the tracked surface. +- **First view** (`load_session` → `/load_expr`): `.buckaroo_stat_cache/`, and the + stable expanded build (cheap entry, unless a read already wrote it) or view + build (worthy entry). No session record is written anywhere. - **First primary-key scan**: `/primary_key.json`. -Everything keyed on the content hash is immutable and self-heals on deletion; -the only invalidations are Buckaroo restart (RAM session map) and a -klass-changing edit/reset (stat cache). A `reset_to` additionally -garbage-collects orphaned `data/.cas` source clones, which live outside the -catalog git repo (`source_identity.gc_cas`) — an entry's own -`manifest.sources` records every clone its frozen build reads, ancestors' -included, so gc keeps a child's leaves alive even if the parent entry is -evicted. See `caching.md` for the full invalidation table. (The +Everything keyed on the content hash is immutable (apart from the pin an +unfaithful heal adds to a manifest), and a deleted file is made again and +verified the next time something reads it. The only invalidations are a +klass-changing edit, a reset or recalc, or an unfaithful heal (all of them the +stat cache). Only the user deletes a file (the Cache page; a pinned snapshot is +refused), and a failed build leaves the file at its entry's path alone. A +`reset_to` leaves `compute_cache/` alone, and moves the `data/.cas` clones that +no surviving source entry's `manifest.provenance` names into +`/bullpen/cas/` instead of deleting them, so a reset forward brings +them back. A clone therefore stays as long as the source version that imported +it survives. See `caching.md` for the full invalidation table. (The cold-reconstruction staleness hole this section used to reference — #74/#115 — is closed: reads load the frozen build, so a cold read cannot see a post-build -source edit.) +edit of a data file.) diff --git a/docs/installing.md b/docs/installing.md index 95254aa0..f3af4730 100644 --- a/docs/installing.md +++ b/docs/installing.md @@ -58,24 +58,31 @@ A "project" is a named catalog/notebook workspace on disk. uv run tallyman init spike ``` -This creates `~/.tallyman-notebooks/projects/spike/` and a starter fixture. -List or inspect projects later with `uv run tallyman project_list` (via MCP) or -just by looking under the projects root. +This creates `~/.tallyman-notebooks/projects/spike/`, writes a starter fixture +(`data/orders.parquet`; pass `--no-fixture` to skip it) and records the +project's first checkpoint, step 0. There is no CLI command that lists projects: +look under `~/.tallyman-notebooks/projects/`, use the project list in the +browser, or ask Claude to call the `project_list` MCP tool. ## 5. Start the companion app -The companion is a FastAPI app on `http://127.0.0.1:7860`. The MCP server pushes -live updates to it over SSE. +The companion is a FastAPI app on `http://127.0.0.1:7860`. The MCP server tells +it about each change with an HTTP request, and it pushes live updates on to the +browser over SSE (a stream of events the browser keeps open). ```sh uv run tallyman run --project spike ``` Leave this running in its own terminal. Open to watch -the catalog, notebook, lineage, and diffs update as you work. +the catalog, notebook and diffs update as you work. -`tallyman run` also spawns a Buckaroo subprocess on `:8700` for in-table recon. -Disable it with `--no-buckaroo` if you don't need it. +`tallyman run` also spawns a Buckaroo subprocess on `:8700` (or on a random port +if 8700 is taken), which draws each entry's data grid. Its stderr goes to +`~/.tallyman-notebooks/projects/spike/buckaroo.log`, and Buckaroo keeps its own +log in `~/.buckaroo/logs/server.log`. Disable it with `--no-buckaroo` if you +don't need the grids; an entry's page then shows its row count and a note that +Buckaroo is not available. ## 6. Configure and approve the MCP server @@ -90,15 +97,14 @@ server: "args": ["run", "tallyman", "mcp"], "cwd": "", "env": { - "TALLYMAN_PROJECT": "spike", - "TALLYMAN_COMPANION_URL": "http://127.0.0.1:7860" + "TALLYMAN_PROJECT": "spike" } } } } ``` -If you cloned to a different path, update `cwd` to point at your checkout. +The copy in the repo has the owner's path in `cwd`; set it to your checkout. Because `.mcp.json` is a project file, Claude Code will **not** trust it automatically. Launch Claude Code from the repo directory and approve the @@ -128,7 +134,8 @@ it. To remove it later: `claude mcp remove tallyman -s project`. With the companion running (step 5) and the MCP server approved (step 6), the `mcp__tallyman__*` tools are available in chat. Try: -> Use `catalog_load_parquet` to load `orders.parquet`. +> Use `catalog_import_source` to import `orders.parquet` from the project's +> `data/` directory under the alias `orders`. > > Now use `catalog_create` to make a named entry `shoe_sales` that groups orders > by region and totals the price. @@ -140,18 +147,30 @@ of the catalog with a `V_n` chip; previous versions stay in forensic history. ## Sharing a project -A project directory is portable: +A project directory is portable: paths inside each entry's build are stored +relative to the project, so it can be read from any location. ```sh -tar czf my-project.tgz -C ~/.tallyman-notebooks/projects spike +uv run tallyman pack spike -o my-project.tgz # on the other machine: tar xzf my-project.tgz -C ~/projects/ uv run tallyman serve ~/projects/spike ``` +`tallyman pack` tars the whole project directory, including `compute_cache/` +(the result files tallyman can make again), so the archive can be large. + `tallyman serve` runs the companion **read-only** against any project directory -on disk — same catalog and history, no edit affordances (mutation routes return -403). +on disk: same catalog, history and charts. The browser still shows the edit +controls, but the server answers their requests with 403. It starts no Buckaroo +subprocess, so entry grids do not load. + +One known defect affects copies (#209). An archive carries each entry's +`.xorq_build_expanded/` directory, a copy of the build with the original +project's absolute path filled in, and tallyman reuses it without checking the +path. A cheap entry (a filter or selection over one file, with no result file of +its own) then reads from the original location, and fails once that location is +gone. ## Reference @@ -166,21 +185,32 @@ on disk — same catalog and history, no edit affordances (mutation routes retur | Variable | Default | Purpose | |----------|---------|---------| -| `TALLYMAN_PROJECT` | `spike` | Active project name | -| `TALLYMAN_COMPANION_URL` | `http://127.0.0.1:7860` | Where the MCP server pushes updates | | `TALLYMAN_HOME` | `~/.tallyman-notebooks` | Root for all project state | +| `TALLYMAN_PROJECT` | none | Seeds the `active_project` file when that file does not exist yet and the named project does (see below) | +| `TALLYMAN_AUTO_RECALC` | unset | `1`/`0` (or `true`/`false`) overrides the project's auto-recalc switch (on by default) | +| `TALLYMAN_LOG_LEVEL` | `INFO` | Log level of the MCP server | **State on disk** lives under `TALLYMAN_HOME`: ``` ~/.tallyman-notebooks/ ├── active_project # one-line plain text; the active project +├── server.lock # held by the `tallyman run` serving this data dir; names its pid and port └── projects/ └── / # catalog, notebook, build artifacts per project ``` -The active project is resolved from `TALLYMAN_PROJECT` first, then the -`active_project` file. +[architecture.md](architecture.md#on-disk-layout) has the full layout. + +The active project is the one named in the `active_project` file. A command's +`--project` flag (`tallyman run --project spike`) names a project explicitly, and +`tallyman run` writes it to the file. `TALLYMAN_PROJECT` is read only when the +file does not exist yet and the project it names exists, to create the file, so +once the file exists it wins over the variable (#39). (`tallyman serve` is the +exception: it sets the variable to the served directory's name and reads it +while it runs.) The MCP server takes its project from the file at its first tool +call and keeps using it for the rest of the Claude Code session, until +`project_switch` or `project_new` changes it. ## Troubleshooting @@ -191,7 +221,22 @@ The active project is resolved from `TALLYMAN_PROJECT` first, then the - **`uv run tallyman` fails to resolve the command** — make sure you're in the checkout directory (or a subdirectory). `uv run` picks the venv from the nearest `pyproject.toml`. -- **Port 7860 already in use** — another companion is running; stop it, or pass - `--port` to `tallyman run` and update `TALLYMAN_COMPANION_URL` to match. - - +- **"data dir … is in use by another tallyman server"** — one `tallyman run` + serves a data dir (`TALLYMAN_HOME`) at a time, and the error names the one + that holds it (pid, port, start time). Stop it, or run the second tallyman on + its own data dir and port: `TALLYMAN_HOME= uv run tallyman run + --port 7861`. An MCP server started with the same `TALLYMAN_HOME` finds that + companion's port by itself, from the `server.lock` the server writes in the + data dir. With no server on its data dir, the MCP server sends no updates, + its replies carry no entry links, and `project_switch` / `project_new` fail. +- **Port 7860 already in use** — something else is listening there; pass + `--port` to `tallyman run`. +- **An entry's data tab says "Buckaroo not available"** — the companion was + started with `--no-buckaroo`, or Buckaroo failed to start; `tallyman run` + prints why. If the tab shows an error instead, its detail says whether + tallyman could not prepare the entry or Buckaroo could not load it, and + `buckaroo.log` in the project directory has Buckaroo's side. +- **The browser stops responding while Claude builds an entry** — a build holds + the project's write lock, and a page that needs the same lock (to re-create a + deleted result file, or an edit made in the browser) waits for it (#186, + #190). diff --git a/docs/mcp-server.md b/docs/mcp-server.md index 6b171bd5..e2951080 100644 --- a/docs/mcp-server.md +++ b/docs/mcp-server.md @@ -2,10 +2,12 @@ The MCP server (`src/tallyman_mcp/server.py`) is the surface Claude Code drives. It is a [FastMCP](https://github.com/jlowin/fastmcp) server that exposes the -catalog as **30 tools and one prompt** over stdio. Claude Code is the only -caller; a human never invokes these directly. Each tool compiles or mutates the -on-disk catalog through `tallyman_core` / `tallyman_xorq`, commits a git -revision, and best-effort notifies the running companion so the browser updates. +catalog as **31 tools and one prompt** over stdio. Claude Code is the only +caller; a human never invokes these directly. Most tools compile or change the +on-disk catalog through `tallyman_core` / `tallyman_xorq`, commit a git +revision (a checkpoint), and best-effort notify the running companion so the +browser updates. Terms (entry, alias, worthy and cheap entries, snapshot, klass) +follow [architecture.md](architecture.md#terms). This doc lists every tool, its parameters, its return shape, and — the part that matters most when reasoning about a session — its **side effects**: whether it @@ -21,8 +23,7 @@ Claude Code reads `.mcp.json` at the repo root, which registers one server named ```json { "command": "uv", "args": ["run", "tallyman", "mcp"], "cwd": "/Users/paddy/tallyman", - "env": { "TALLYMAN_PROJECT": "spike", - "TALLYMAN_COMPANION_URL": "http://127.0.0.1:7860" } } + "env": { "TALLYMAN_PROJECT": "spike" } } ``` So Claude Code spawns `uv run tallyman mcp`. The `tallyman` entry point is @@ -32,15 +33,21 @@ in `mcp.run()`. With no transport argument FastMCP defaults to **stdio**, so Claude Code talks to the server over the spawned process's stdin/stdout — one MCP process per Claude Code session, owned by Claude Code, not by the companion. -The server reaches the companion through `COMPANION_URL` -(`os.environ.get("TALLYMAN_COMPANION_URL", "http://127.0.0.1:7860")`), firing -best-effort `POST {COMPANION_URL}/internal/notify` calls so the FastAPI companion -refreshes Buckaroo sessions and pushes SSE to open browsers. The companion need -not be up; notifications that fail are logged and dropped. +The server finds the companion per call with `companion_url()` +(`src/tallyman_core/server_lock.py`): the port of the `tallyman run` that holds +this data dir (`TALLYMAN_HOME`), read from its owner record in `server.lock`. +There is no default port: with no server on the data dir there is no companion, +so notifies are skipped, tool replies carry `url: null`, and `project_switch` / +`project_new` return an error. It fires best-effort +`POST /internal/notify` calls so the FastAPI companion refreshes +Buckaroo sessions and pushes SSE to open browsers. Each notify, and each +project switch or creation, names its data dir as `home`, and a companion +serving another data dir answers 409 and changes nothing. Notifications that +fail are logged and dropped. ## Cross-cutting behavior -Four mechanisms apply to (almost) every tool, so they are documented once here +Five mechanisms apply to (almost) every tool, so they are documented once here rather than repeated per tool. **Auto-checkpoint (opt-out).** `mcp.tool` is monkeypatched to @@ -73,8 +80,14 @@ tagging. **Notifications.** `_notify(kind, hash, **extra)` is a 2-second best-effort POST to the companion's `/internal/notify`, which re-publishes it as an SSE event of that `kind`. It never raises. Pass extra data as flat kwargs (`remap=...`), not -`extra={...}`, or it nests and is silently dropped. The "Notifies" line per tool -lists the kinds it emits. +`extra={...}`, or it nests and is silently dropped. The "SSE notify" column +below lists the kinds each tool emits. The SPA listens for `new_entry`, +`build_failed`, `notebook_changed`, `chart_attached`, +`post_processing_changed`, `summary_stat_changed`, `recalc` and +`project_switched`; `alias_changed`, `alias_renamed`, `display_changed` and +`entry_added` reach the browser but trigger no refetch. For +`summary_stat_changed`, `post_processing_changed`, `display_changed` and +`recalc` the companion also reloads the project's Buckaroo sessions. **Sticky active project.** `_resolve_active_project()` returns the in-process `_mcp_active_project` (set by `project_switch` / `project_new`) ahead of the @@ -82,19 +95,20 @@ on-disk `active_project` file, so switching once is sticky for the rest of the session regardless of what another session writes to disk. The first tool call seeds it. `SESSION_ID` (an 8-hex uuid) tags this process in `events.jsonl`. -**Auto-recalc.** `catalog_revise` and `catalog_promote_diff` advance an existing -alias head, which can make followers stale. Both route through +**Auto-recalc.** `catalog_revise`, `catalog_promote_diff` and +`catalog_import_source` (when it mints a new version) advance an existing alias +head, which can make followers stale. All three route through `_auto_recalc_after_head_advance` (when the project's `auto_recalc` switch is on): a checkpoint-free cascade rebuilds the followers in dependency order and leaves them in the working tree, so the head advance and the whole cascade land as one -git revision. A non-empty cascade emits a `recalc` SSE event. +git revision. A cascade whose remap is non-empty emits a `recalc` SSE event. ## Side-effect matrix | Tool | Checkpoints | SSE notify | Auto-recalc | |---|---|---|---| | `catalog_run` | yes | `new_entry`, `build_failed` | no | -| `catalog_load_parquet` | yes | `new_entry`, `build_failed`, `notebook_changed`¹ | no | +| `catalog_import_source` | yes | `new_entry`, `build_failed`, `notebook_changed`¹, `recalc` | yes | | `catalog_create` | yes | `new_entry`, `build_failed`, `notebook_changed` | no | | `catalog_revise` | yes | `new_entry`, `build_failed`, `recalc` | yes | | `catalog_alias` | yes | `alias_changed`, `notebook_changed` | no | @@ -104,6 +118,7 @@ git revision. A non-empty cascade emits a `recalc` SSE event. | `notebook_remove` | yes | `notebook_changed` | no | | `notebook_edit_markdown` | yes | `notebook_changed` | no | | `catalog_chart` | yes | `chart_attached` | no | +| `catalog_chart_errors` | yes ⚠ | — | no | | `catalog_diff` | no | — | no | | `catalog_promote_diff` | self³ | `entry_added`, `recalc`⁴ | yes | | `catalog_scan_staleness` | no | — | no | @@ -124,33 +139,51 @@ git revision. A non-empty cascade emits a `recalc` SSE event. | `project_switch` | no | — ⁵ | no | | `project_new` | no | — ⁵ | no | -¹ `notebook_changed` only when `name` is supplied. ² only when a cell was -actually removed. ³ self-checkpoints exactly one revision; on the opt-out list so -the dispatch boundary does not double-commit. ⁴ only on a committing run that -produced a non-empty remap. ⁵ the *companion* broadcasts `project_switched`; the -MCP tool itself sends no `_notify`. ⚠ see [Known quirks](#known-quirks). +¹ `notebook_changed` only when the import mints v1 of a new alias, which appends +a notebook cell. ² only when a cell was actually removed. ³ self-checkpoints one +revision (for `catalog_recalc`, only a committing run whose remap is non-empty); +on the opt-out list so the dispatch boundary does not double-commit. ⁴ only on a +committing run that produced a non-empty remap. ⁵ the *companion* broadcasts +`project_switched`; the MCP tool itself sends no `_notify`. ⚠ see [Known +quirks](#known-quirks). Any tool whose result carries an `error` key skips its checkpoint and sends no -notify; failures come back inline in the response (most as `{error, error_id}` -for build failures, `{error}` for validation/precondition failures), not as -raised exceptions. +success notify (a failed build in the authoring tools still sends +`build_failed`; `catalog_promote_diff` sends nothing on failure); failures come +back inline in the response (most as `{error, error_id}` for build failures, +`{error}` for validation/precondition failures), not as raised exceptions. --- ## Authoring tools The compile-and-persist family. The `code` argument is a self-contained Python -script that binds a top-level `expr` to a xorq/ibis expression; -`catalog_run`'s docstring is the canonical cookbook for writing it (namespaces, -the datafusion-only backend, data sourcing, and the `xorq.ml` MODELING section). -Data sourcing inside `code` uses four helpers from `tallyman_xorq.io`: -`tracked_expr_from_alias` (a catalog entry, recorded as a lineage parent), -`pinned_expr_from_alias` (a catalog entry by hash or `"name-vN"` version -reference, no following — a bare alias is rejected, #166), -`read_project_file` (a raw parquet under `data/`), and `tallyman_read_csv` (CSV -ingest — injects an `original_row_order` column so the baked snapshot is -byte-stable across builds; use it for all CSV reads instead of -`xo.deferred_read_csv`, #137). +script that binds a top-level `expr` to a xorq/ibis expression; `catalog_run`'s +docstring is the canonical cookbook for writing it (namespaces, the +datafusion-only backend, data sourcing, and the `xorq.ml` MODELING section). +Data sourcing inside `code` uses two helpers from `tallyman_xorq.io`: +`tracked_expr_from_alias` (a catalog entry by alias, recorded as a lineage +parent that the entry follows) and `pinned_expr_from_alias` (one version of an +alias, by `"name-vN"` version reference, no following; a bare alias is rejected, +#166, and so is a bare content hash, ADR-011 D5). A recipe never opens a file: +data files come in through `catalog_import_source` (below) as **source +aliases**, and a recipe reads one like any other alias. `read_project_file`, +`tallyman_read_csv` and `xo.deferred_read_csv` are build errors whose message +names the import to use, and so is `xo.deferred_read_parquet` of any file +outside the project's `compute_cache/`. + +Every entry's result ends in `__row_order`, and pages are ordered by it. A recipe +that only filters, selects or adds columns is *cheap* and must keep the column: +`t.select("region", "price")` fails the build, and the error shows the fix, +`t.select("region", "price", "__row_order")`. A recipe that aggregates, joins, +sorts, uses a window function, union, distinct, unnest, `random()`, `now()` or a +UDF, or reads a second file, is *worthy*: the server writes its result to a +snapshot file when the entry is created (running the query twice) and renumbers +the column to match. Assigning to `__row_order` is an error, an `order_by` that +is followed by more steps is kept if its key columns survive, and joining three +entries in one recipe needs `.drop("__row_order")` on the right-hand inputs +(chains of semi or anti joins are refused too, #199). `catalog_run`'s docstring +has the details. ### `catalog_run(code, prompt="") -> dict` Execute an expression and persist it as an **unnamed (scratch)** entry. Claude's @@ -158,22 +191,44 @@ default authoring tool for a one-off run. - **Params:** `code` (required) — script binding `expr`; `prompt` — optional intent string, recorded on the entry. - **Returns:** `{hash, row_count, execute_seconds, schema, entry_path, url}`, - plus `lint_warnings` when the nondeterminism lint fired. `{error, error_id}` - on a `BuildError`. -- **Writes:** the entry build dir under `catalog/entries//`; a `build_ok` + plus `lint_warnings` when the nondeterminism lint fired, and `reproducible: + false` with `nonreproducible_columns` when a worthy entry's query gave + different results on its two runs at create time. `{error, error_id}` on a + `BuildError`. +- **Writes:** the entry build dir under `catalog/entries//`; for a worthy + entry, its snapshot under `catalog/compute_cache/result_cache/`; a `build_ok` or `build_error` event to `events.jsonl`. - **Promote** a scratch entry to a name afterward with `catalog_alias`. -### `catalog_load_parquet(rel_path, prompt="", name="") -> dict` -Register a parquet under `/data/` as an entry by synthesizing a -`read_project_file(rel_path)` recipe — the no-code "load this file" path. With -`name`, behaves like `catalog_create` (names the entry, appends a notebook cell). -- **Params:** `rel_path` (required) — path under `data/`; `prompt` — optional - intent / cell markdown; `name` — optional alias. -- **Returns:** same as `catalog_run`, plus `alias` and `version` when named. - `{error}` if the alias already exists; `{error, error_id}` on a missing file. -- Unlike `catalog_run`, it does **not** emit a `build_ok` event; success shows - only via the notifies. +### `catalog_import_source(outside_path, alias, pinned_version=None, prompt="", +schema=None, reader_options=None) -> dict` Import a parquet or CSV into the +catalog and point the source alias `alias` at it (ADR-011). The only way a file +enters: the bytes are copied into the project's clone store (`data/.cas/`), one +snapshot of them is written in file order with a `__row_order` column, and a +version of `alias` is minted. Recipes then read +`tracked_expr_from_alias(alias)`; the original path is provenance and is never +read again. +- **Params:** `outside_path` (required) — any path, `data/` is not special; + `alias` (required) — the source alias, which may not name a catalog alias; + `pinned_version` — the version you claim the file is, checked rather than + assumed; `prompt` — optional intent / cell markdown; `schema` and + `reader_options` — CSV column types and `polars.scan_csv` options, recorded on + the entry and never re-derived (ADR-011 D12). +- **Returns:** `hash`, `alias`, `version`, `created`, `row_count`, `schema`, + `path`, `digest`, `url`, plus `recalc` when advancing the alias cascaded. + `{error, error_id}` for a missing path, an alias collision, bytes another + alias already holds (the error names that alias and version), or any row of + the ADR-011 D3 table that is an error. +- Re-running it on **unchanged** bytes is a no-op that returns the current + version, so it is safe in a script. If that version's snapshot is gone, the + no-op heals it from the clone and checks it against the recorded digest; it + never rewrites the version. Re-running it on **changed** bytes mints the next + version and cascades like a revise. +- **One set of bytes is one version under one alias** (ADR-011 D1). To give a + source a second name, `catalog_create` an entry whose recipe is + `tracked_expr_from_alias("")`; it follows the source when it is + re-imported. A CSV read with other reader options is a different entry and + may be imported under its own alias (ADR-011 D12). ### `catalog_create(name, code, prompt="") -> dict` Execute and persist as a **named** entry (alias) and append a notebook cell. @@ -191,7 +246,8 @@ carry chart/display config forward, and cascade-recompute the alias's stale followers. The canonical head-advance path. - **Params:** `name` (required) — existing alias; `code` (required) — a self-contained recipe that **must not reference its own alias by name** - (rejected, #135 — inline the source or pin the previous version by hash); + (rejected, #135 — inline the source, or pin the previous version with + `pinned_expr_from_alias("-v")`; a bare hash is refused); `prompt` — optional. - **Returns:** `catalog_run` shape plus `alias`, `version`; `carried_over` (subset of `chart` / `display_config`) when non-empty; and a `recalc` @@ -212,7 +268,9 @@ cell. Used after a `catalog_run` when a scratch entry earns a permanent name. - **Params:** `hash` (required) — must name an on-disk entry; `name` (required) — must be free. - **Returns:** `{hash, alias, version, url}`. `{error}` if the hash has no entry - or the name is taken. + or the name is taken, and if the entry is a source version: a source entry + belongs to its source alias, and the error names the `catalog_create` over + `tracked_expr_from_alias` that gives the source a second name (ADR-011 D1). ### `catalog_rename(old_name, new_name) -> dict` Rename an alias, preserving its full version history and notebook position. @@ -261,12 +319,25 @@ hash). but-JSON spec only fails in the browser). - **Returns:** `{hash, spec_path}`. `{error}` for an unknown target or invalid JSON. -- **Writes:** `catalog/chart_specs/.vl.json`. +- **Writes:** `catalog/chart_specs/.vl.json`. The browser fetches the + chart's data from `/{project}/api/data/?limit=100000`. + +### `catalog_chart_errors(hash_or_alias) -> list` +Chart render failures the browser reported for an entry, most recent first. The +page posts one to `/{project}/api/chart_error` when vega-embed fails, and it is +stored in `errors.jsonl` with `tool="chart_render"`. Use it after attaching or +editing a chart, once someone has viewed the page. +- **Params:** `hash_or_alias` (required); resolved like `catalog_chart`. +- **Returns:** a list of error records (up to the 500 most recent errors are + searched), empty if none were reported; `[{error}]` for an unknown target. +- It is read-only but not on the opt-out list, so it checkpoints on every call + (see [Known quirks](#known-quirks)). ### `catalog_diff(name, va=-2, vb=-1) -> dict` **Read-only** diff of two versions of an alias (default previous vs latest). -Each side's expression comes from `cached_result_expr`, so it does not depend on -a materialized result existing. +Each side's expression comes from `cached_result_expr`, which makes any missing +file exist first. The diff (`full_diff`) keeps `__row_order` as a data column, so +after an inserted row the stats diff reports it as changed (#200). - **Params:** `name` (required); `va`, `vb` — version indices, 1-based or negative-from-end (`-1` latest, `-2` penultimate). - **Returns:** `{alias, before:{version,hash}, after:{version,hash}, schema, @@ -295,28 +366,39 @@ marimo-exportable. ## Staleness and recalc tools -### `catalog_scan_staleness() -> dict` -**Read-only** scan of every live entry for staleness against its recorded inputs. -Purely diagnostic — never rebuilds, repoints, or checkpoints. Call it first to -see what is stale. +### `catalog_scan_staleness(verify_results=False) -> dict` +**Read-only** scan of every complete entry (current alias heads and superseded +versions alike; only heads can be directly stale) for staleness against its +recorded inputs. Purely diagnostic — never rebuilds, repoints, or checkpoints, +and opens no data file: an entry is stale only when an alias it follows has +moved (ADR-011 D6, one staleness axis). Call it first to see what is stale. +- **Params:** `verify_results` — also check that each snapshot on disk still has + its recorded `result_digest`. It reads the files that exist and writes nothing, + so a snapshot that was deleted is reported and stays deleted. - **Returns:** `{stale, transitively_stale, entries, orphan_stale}`. `stale` is the default root set `catalog_recalc` uses. `orphan_stale` populates only under - auto-recalc mode. + auto-recalc mode. With `verify_results`, also `verify: {results, unfaithful, + absent, errors}`: `results` maps each hash that recorded a digest to `true`, + `false` or `null` (no file to check), `unfaithful` lists the `false` ones, and + `absent` lists the entries whose snapshot file is missing. ### `catalog_recalc(roots=None, dry_run=True) -> dict` Recompute stale entries and their dependents in dependency order. Defaults to a non-committing **dry-run preview**; call again with `dry_run=False` to commit. - **Params:** `roots` — content hashes to recompute with their descendant cone; - `None` defaults to every directly-stale entry. Roots naming no live entry are - silently dropped. `dry_run` — preview when `True` (default). + `None` defaults to every directly-stale entry. Roots that name no complete + entry (no `manifest.json`) are silently dropped; a named root that is a + superseded version is kept and replayed. `dry_run` — preview when `True` + (default). - **Returns:** the `RecalcReport` (`project, roots, cone, entries, dry_run, status, remap, checkpoint_step, error`). Action vocabulary differs by mode (`rebuild`/`cascade`/`unchanged` on dry-run; `rebuilt`/`noop`/`failed`/ `skipped` on a real run). A nothing-stale fast path returns a short `{status:'ok', ..., note:'nothing stale to recompute'}`. - **Self-checkpoints arg-aware:** a dry-run takes no checkpoint; a committing run - takes exactly one for the whole walk, so the recompute is a single revision - `reset-to` can undo atomically. A committing run with a non-empty remap emits + that changed something (a non-empty remap) takes exactly one for the whole + walk, so the recompute is a single revision `reset-to` can undo atomically, + and one that changed nothing takes none. A committing run with a non-empty remap emits `recalc`. Build failures are encoded in `status`/`error`/per-entry `error` (the walk halts, the rebuilt prefix stays committed); a dependency cycle yields `status=='cycle'` rather than raising. @@ -327,9 +409,18 @@ Project-authored per-column statistics. Source runs in a restricted-globals sandbox; errors come back inline rather than later in the Buckaroo log. There is no separate update tool — re-adding the same `name` overwrites. +Buckaroo looks for project stats in `/stats/`, and tallyman sends +`project_root` as `/artifacts/`, while it writes stats to +`/artifacts/catalog/stats/`. So a stat added here is validated, +committed and reloaded, but Buckaroo does not find it and the grid does not show +it (#170). Post-processing functions have the same problem; display klasses, +which live in `artifacts/display/`, do not. + ### `catalog_add_summary_stat(name, source) -> dict` -Validate a `compute(col)` function against a 1-row ibis memtable, write it to -`/stats/.py`, and hot-reload it into open Buckaroo sessions. +Validate a `compute(col)` function against a 3-row, one-column ibis memtable, +write it to `/artifacts/catalog/stats/.py` (tracked in the +catalog repository), and ask the companion to reload the project's open Buckaroo +sessions. - **Params:** `name` (required) — a valid Python identifier; `source` (required) — must define `compute(col)` taking one ibis column and returning an ibis scalar. @@ -357,8 +448,10 @@ styling base classes already in scope; validated in a restricted sandbox before the file lands. ### `catalog_add_display_klass(name, source) -> dict` -Validate and persist a display klass to `display/.py`, then hot-reload it -into open sessions (`display_changed`). The primary use is extending +Validate and persist a display klass to `/artifacts/display/.py`, +then hot-reload it into open sessions (`display_changed`). That directory is +outside the catalog repository, so the checkpoint this tool takes commits +nothing for it, and a reset does not undo it. The primary use is extending `DefaultMainStyling` (`df_display_name='main'`) or `DefaultSummaryStatsStyling` (`'summary'`) with a `pinned_rows` entry so a stat shows as a frozen row. - **Params:** `name` (required) — valid identifier; `source` (required) — must @@ -385,7 +478,8 @@ sandbox as summary stats (third-party imports like numpy/sklearn raise ### `catalog_add_post_processing(name, source) -> dict` Validate a `process(expr)` function and write it to -`/post_processing/.py`. +`/artifacts/catalog/post_processing/.py` (tracked in the catalog +repository). Buckaroo does not find it there yet (#170, above). - **Params:** `name` (required) — valid identifier, becomes the dropdown label; `source` (required) — defines `process(expr)` returning an ibis expression or a pandas DataFrame. @@ -393,8 +487,9 @@ Validate a `process(expr)` function and write it to - **Validator gotcha:** the dry-run table has only columns `{a, b}`, so a `process` that references real column names is rejected here even when `catalog_run_post_processing` previewed it fine — guard with - `if 'col' not in expr.columns: return expr`. The new option appears only on the - next session load (V1 does not hot-swap loaded sessions). + `if 'col' not in expr.columns: return expr`. The tool's docstring says the new + option appears on the next session load; the companion also posts + `/reload_expr` for each of the project's entries, as for any klass change. ### `catalog_remove_post_processing(name) -> dict` Soft-delete to `post_processing/_disabled/`. Still shows in the list with @@ -414,9 +509,9 @@ anything — the iterate-before-commit tool. Note the arg is `code` here, not - **Returns:** `{entry, row_count, columns, preview}` (first 20 rows). `{error}` on any run failure. - Reads the entry's actual result via `cached_result_expr` (a cheap entry - recomputes; an expensive one reads its baked snapshot, self-healing if - evicted), so the preview reflects real data — wider than `add`'s `{a, b}` - dry-run. Persists nothing; on the opt-out list, no notify. + recomputes; a worthy one reads its snapshot, re-creating and verifying it if + it was deleted), so the preview reflects real data — wider than `add`'s + `{a, b}` dry-run. Persists nothing; on the opt-out list, no notify. ## Export and listing tools @@ -442,11 +537,13 @@ it references exact column names rather than guessing. ## Project-lifecycle tools -These require the companion (`tallyman run`) to be active: the **companion** is -the sole writer of the `active_project` file and the genesis baseline, reached -via `_companion_post`. All three are on the opt-out list. On success the -companion broadcasts a `project_switched` SSE event itself (the MCP tool sends no -`_notify`). Each updates the in-process sticky `_mcp_active_project`. +`project_switch` and `project_new` require the companion (`tallyman run`) to be +active: for these tools the **companion** writes the `active_project` file and +the genesis baseline, reached via `_companion_post`, so that it can publish the +SSE event. All three are on the opt-out +list. On success the companion broadcasts a `project_switched` SSE event itself +(the MCP tool sends no `_notify`), and the tool updates the in-process sticky +`_mcp_active_project`. ### `project_list() -> dict` List projects on disk and report the active one. **Safe even when the companion @@ -469,7 +566,7 @@ Create a new project on disk (the companion runs `ensure_project` + genesis baseline) and switch to it. - **Params:** `name` (required) — `^[a-z0-9][a-z0-9_-]{0,31}$`, not reserved, not colliding; `with_fixture` — when `True`, the companion seeds the shoe-orders - demo parquet into `data/`. + demo parquet into `data/`, where `catalog_import_source` can import it. - **Returns:** `{name, active}`. On failure `{error}` (sticky unchanged): same shapes as `project_switch`, with 409 on a name collision. @@ -477,12 +574,13 @@ baseline) and switch to it. ### `ds_modeling_workflow(dataset, target="") -> str` An `@mcp.prompt()` (not a tool) that returns a static guidance string scaffolding -a modeling workflow: load → engineer features → split → fit via `xorq.ml` → +a modeling workflow: import → engineer features → split → fit via `xorq.ml` → score → chart. Pure function — no side effects, no I/O, not routed through the checkpoint or tagging wrappers. -- **Params:** `dataset` (required) — a parquet under `data/`, interpolated into - the text; `target` — column to predict (empty renders as `` for - unsupervised work). +- **Params:** `dataset` (required) — the data file, interpolated into the text; + the workflow's first step imports it with `catalog_import_source`. `target` — + column to predict (empty renders as `` for unsupervised + work). - It encodes the hard constraints the model must follow: fit the model *as a catalog entry* with `xorq.ml` (never in a post-processing sandbox, which blocks sklearn/scipy/numpy); copy exact fit/predict signatures from the MODELING @@ -493,11 +591,18 @@ checkpoint or tagging wrappers. ## Known quirks -- **`catalog_list_display_klasses` checkpoints.** It is the one read-only `*_list*` - tool missing from the `_NO_CHECKPOINT` set (`server.py`), so `_with_checkpoint` - runs `checkpoint_catalog`, which commits with `--allow-empty` and advances the - step tag. Each call therefore appends an **empty** catalog revision — harmless - to correctness but it inflates the git history and the step count. Adding it to - `_NO_CHECKPOINT` (next to `catalog_list_summary_stats` / - `catalog_list_post_processings`) would bring it in line with the other list - tools. +- **`catalog_list_display_klasses` and `catalog_chart_errors` checkpoint.** They + are read-only but missing from the `_NO_CHECKPOINT` set (`server.py`), so + `_with_checkpoint` runs `checkpoint_catalog`, which commits with + `--allow-empty` and advances the step tag. Each call therefore appends an + **empty** catalog revision — harmless to correctness but it inflates the git + history and the step count. Adding them to `_NO_CHECKPOINT` (next to + `catalog_list_summary_stats` / `catalog_list_post_processings`) would bring + them in line with the other read-only tools. +- **Display klasses are outside the catalog repository.** `catalog_add_display_klass` + and `catalog_remove_display_klass` checkpoint, but `artifacts/display/` is not + in the repository, so the revision records nothing about them and a reset does + not undo them. +- **A build holds the project lock for its whole length.** Parallel tool calls + that build queue behind each other, and behind any build or heal the companion + is running (#186). diff --git a/docs/reactive-recalc.md b/docs/reactive-recalc.md index 6d75aecb..14b30dff 100644 --- a/docs/reactive-recalc.md +++ b/docs/reactive-recalc.md @@ -1,27 +1,29 @@ # The reactive system: revise an alias, recompute its dependents -A tallyman catalog is a closed system. Nothing reads a database, and no source -file changes underneath you without your knowing. The thing that moves is *you*: -you revise an alias — a source projection or an aggregation — to a new recipe, -and now everything built on top of the old version is out of date. The reactive -system is what notices that and recomputes the followers, in dependency order, on -demand. +A tallyman catalog is a closed system. Nothing reads a database, and no data +file changes underneath you: a file enters the catalog only when you import it, +and every build reads tallyman's copy. The thing that moves is *you*: you revise +an alias (a named pointer to the latest version of an entry, such as an +aggregation) to a new recipe, or import a new version of a data file under its +source alias, and now everything built on top of the old version is out of +date. The reactive system is what notices +that and recomputes the followers (the entries that read that alias by name), in +dependency order. Terms follow [architecture.md](architecture.md#terms). This doc starts with the two things you actually do — build a small catalog, then revise an alias and recompute its dependents — and then documents the surfaces -(MCP tools and the companion's HTTP API) and the edges (the staleness axes, +(MCP tools and the companion's HTTP API) and the edges (the staleness rule, checkpoints, failure handling) underneath. The short version of how it works today: - Building an alias does not recompute anything downstream; it advances that one - alias. *Revising* an alias, by default, recomputes its stale followers in the - same atomic checkpoint (auto-recalc — see *Trigger model*). The switch is - per-project and env-overridable; turn it off for the explicit path below. + alias. *Revising* an alias, or importing new data under a source alias, by + default recomputes its stale followers in the same atomic checkpoint + (auto-recalc — see *Trigger model*). The switch is per-project and + env-overridable; turn it off for the explicit path below. - A scan tells you which followers are stale. With auto-recalc off, a separate - explicit recalc recomputes them and re-points their aliases — and that explicit - recalc is still the path for source-file drift, which the revise trigger doesn't - cover. + explicit recalc recomputes them and re-points their aliases. - A recalc that rebuilds an entry advances that entry's alias to a **new revision** (the alias history is append-only), and takes **one** catalog checkpoint for the whole walk. @@ -31,29 +33,30 @@ The short version of how it works today: Every catalog entry is content-addressed: its `content_hash` is a function of its recipe and its resolved inputs. An *alias* is a mutable name that points at the latest hash for a concept and keeps the full history of hashes it has pointed at -(`aliases.jsonl`, one line per alias: `{"alias", "latest", "history": [...]}`). +(`aliases.jsonl`, one line per alias: `{"alias", "latest", "history": [...], +"kind"}`). The kind is `source` for an alias whose versions are imported files +and `catalog` for one whose versions are computations. A recipe is a self-contained Python script that binds a top-level `expr`. It -reads a raw file with `read_project_file(...)` and reads another catalog entry with -`tracked_expr_from_alias(...)`. Build a three-node chain — a source projection, an -aggregation over it, and a filter over that: +reads other catalog entries, by alias, with `tracked_expr_from_alias(...)`; it +never opens a file. Data enters through `catalog_import_source`, which makes each +version of a file an entry of its own under a source alias. Build a three-node +chain — an imported file, an aggregation over it, and a filter over that: ```python -from tallyman_mcp.server import catalog_create +from tallyman_mcp.server import catalog_create, catalog_import_source -# Node 1 — a source alias: project some columns out of a raw parquet. -catalog_create("orders", ''' -from tallyman_xorq.io import read_project_file -t = read_project_file("orders.parquet") -expr = t.select("region", "price") -''') -# -> {"hash": , "alias": "orders", "version": 1, ...} +# Node 1 — a source alias: import a parquet file. The import copies the bytes into +# the project and writes them, in file order with a last __row_order column, as +# the snapshot of a new entry. The path is recorded and never read again. +catalog_import_source("/path/to/orders.parquet", "orders") +# -> {"hash": , "alias": "orders", "version": 1, "created": True, ...} # Node 2 — an aggregation alias that FOLLOWS orders by name. catalog_create("by_region", ''' from tallyman_xorq.io import tracked_expr_from_alias t = tracked_expr_from_alias("orders") -expr = t.group_by("region").agg(total=t.price.sum()) +expr = t.group_by("region").aggregate(total=t.price.sum()) ''') # -> {"hash": , "alias": "by_region", "version": 1, ...} @@ -69,18 +72,23 @@ expr = t.filter(t.total > 1000) `catalog_create(name, code)` builds the recipe, persists the entry, and assigns the alias at **version 1**. It errors if the alias already exists (use `catalog_revise` for an existing name). There is no `project` argument on the -tool: it operates on the session's active project (set by `project_switch`); the -optional `project=` kwarg on `read_project_file`/`tracked_expr_from_alias` defaults to that same -active project, so recipes omit it. - -A recipe pulls in data through one of three functions in `tallyman_xorq.io`, and +tool: it operates on the session's active project (set by `project_switch`). The +optional `project=` kwarg on `tracked_expr_from_alias`/`pinned_expr_from_alias` +defaults to the project named in the `active_project` file, which is the +session's project unless another session has switched projects since; then the +two differ and the recipe looks its aliases up in the other project (related to +#39). Recipes omit the kwarg. + +A recipe pulls in data through one of two functions in `tallyman_xorq.io`, and which one you call is what records the dependency edge: | Function | Reads | Records | Goes stale when | |---|---|---|---| -| `read_project_file("orders.parquet")` | a raw file under `/data/` | a **source** leaf (`rel_path → digest`) | the file's bytes change on disk | -| `tracked_expr_from_alias("orders")` | a catalog entry, by **alias** | a **parent** edge, `follow=True` | the alias advances to a new head | -| `pinned_expr_from_alias(...)` | a catalog entry, by alias **or** hash | a **parent** edge, `follow=False` | never (it pinned that exact revision) | +| `tracked_expr_from_alias("orders")` | a catalog entry, by **alias** (a source alias or a catalog alias) | a **parent** edge, `follow=True` | the alias advances to a new head | +| `pinned_expr_from_alias("orders-v1")` | a catalog entry, by version reference | a **parent** edge, `follow=False` | never (it pinned that exact revision) | + +`read_project_file`, `tallyman_read_csv`, and a raw `xo.deferred_read_*` of a +file tallyman did not write are build errors that name the import to use. The follow relationship is the whole game: @@ -89,39 +97,44 @@ The follow relationship is the whole game: follow — resolves to the alias's current head at build time, and records **follow=True**. The child goes stale and recomputes when `orders` advances. This is normal chaining. -- `pinned_expr_from_alias("orders")` or `pinned_expr_from_alias("")` is the - deliberate opt-out. It accepts an alias or a hash, records **follow=False**, and - the child stays on that exact revision: it never goes stale on the alias axis, - and recalc finds it but leaves it alone. -- `read_project_file` is the root of every chain: a raw file with no catalog - identity, the leaf the graph bottoms out in. +- `pinned_expr_from_alias("orders-v1")` is the deliberate opt-out. It accepts + only a version reference (`"-v"`, 1-based into the alias's history). + It refuses a bare alias, which would pin whatever the head happened to be + (#166), and a bare content hash (ADR-011 D5), so every edge an authored recipe + records names an alias (a promoted diff's generated recipe names its two + entries by hash and records no edge). It + records **follow=False**, and the child stays on that exact revision: it never + goes stale, and recalc finds it but leaves it alone. +- Source aliases are the roots of every chain. Each version is a source entry, + an ordinary entry whose snapshot holds an imported file's rows, and the graph + bottoms out in them. So after these three calls: `orders → by_region → top_regions`, each aliased at version 1, each following its parent by name (`tracked_expr_from_alias`). ## Revising an alias and recomputing its dependents -This is the core workflow. You revise an alias — it does not matter whether it is -a source projection (`orders`) or an aggregation (`by_region`) — and then you -recompute whatever followed it. +This is the core workflow. You advance an alias — by importing new data under a +source alias (`orders`) or by revising a catalog alias (`by_region`) — and then +you recompute whatever followed it. The walk-through below is the explicit path, +which runs when auto-recalc is off (`TALLYMAN_AUTO_RECALC=0`, or +`"auto_recalc": false` in the catalog's `config.json`). With auto-recalc on, the +default, the import (or `catalog_revise`) does steps 1 to 3 itself (see *Trigger +model*). ```python -from tallyman_mcp.server import catalog_revise, catalog_scan_staleness, catalog_recalc +from tallyman_mcp.server import catalog_import_source, catalog_scan_staleness, catalog_recalc -# Revise the source alias: keep an extra column this time. -catalog_revise("orders", ''' -from tallyman_xorq.io import read_project_file -t = read_project_file("orders.parquet") -expr = t.select("region", "price", "qty") -''') -# -> {"hash": , "alias": "orders", "version": 2, ...} +# The file has new rows: import it again under the same alias. +catalog_import_source("/path/to/orders.parquet", "orders") +# -> {"hash": , "alias": "orders", "version": 2, "created": True, ...} ``` -`catalog_revise(name, code)` advances `orders` to a new hash and **version 2**. -The previous hash stays in the catalog as a forensic artifact and in -`orders`'s alias history. Nothing downstream has moved: `by_region` and -`top_regions` still point at their version-1 builds, which were computed against -`orders` v1. They are now stale. +The bytes differ from `orders` v1, so the import mints **version 2** of the +source alias under a new hash. (Importing unchanged bytes is a no-op that returns +v1.) The previous version stays in the catalog and in `orders`'s alias history. +Nothing downstream has moved: `by_region` and `top_regions` still point at their +version-1 builds, which were computed against `orders` v1. They are now stale. Step 1 — scan to see what moved: @@ -132,11 +145,11 @@ catalog_scan_staleness() ``` `by_region` followed `orders` by name, and `orders`'s head no longer matches the -hash `by_region` recorded at build, so `by_region` is **directly stale** on the -alias axis. `top_regions` followed `by_region`, whose head has *not* moved yet, so -it is not directly stale — but it is **transitively stale**: a recalc rooted at -the stale set will carry it along. `orders` itself is not stale; it is the fresh -head you just built. +hash `by_region` recorded at build, so `by_region` is **directly stale**. +`top_regions` followed `by_region`, whose head has *not* moved yet, so it is not +directly stale — but it is **transitively stale**: a recalc rooted at the stale +set will carry it along. `orders` is the fresh head you just imported, and it +reads no other entry. Step 2 — preview the recalc (always a dry run first): @@ -163,18 +176,19 @@ The walk replays `by_region` first. Because dependency order re-points `tracked_expr_from_alias("by_region")` reads the advanced parent and rebuilds against it. Both aliases now point at fresh, version-2 hashes; a re-scan is clean. -The same flow drives a revision of an *aggregation* alias. Revise `by_region` and -`top_regions` goes stale and recomputes; `orders` (its parent) is untouched. The -direction is always downstream: revising an alias makes its **followers** stale, -never its parents. +The same flow drives a revision of a catalog alias. Revise `by_region` with +`catalog_revise` and `top_regions` goes stale and recomputes; `orders` (its +parent) is untouched. The direction is always downstream: advancing an alias +makes its **followers** stale, never its parents. A source alias cannot be +revised, since there is no recipe to change: new data is a new import. -This three-step flow (revise, scan, recalc) is the explicit path — the one that -runs when auto-recalc is **off**. By default it is **on**, and `catalog_revise` -folds the recalc into itself: it advances the alias *and* recomputes its stale -followers in the same atomic checkpoint, so the scan-and-recalc dance above -collapses to a single `catalog_revise` call. See *Trigger model* for the switch -and what the auto path reports. The scan stays passive either way; it never -writes. +This three-step flow (advance, scan, recalc) is the explicit path — the one that +runs when auto-recalc is **off**. By default it is **on**, and the import or +`catalog_revise` folds the recalc into itself: it advances the alias *and* +recomputes its stale followers in the same atomic checkpoint, so the +scan-and-recalc dance above collapses to a single call. See *Trigger model* for +the switch and what the auto path reports. The scan stays passive either way: it +rebuilds nothing, opens no data file and changes no catalog state. ## When are new alias revisions created @@ -184,9 +198,12 @@ moves the head (`alias_map[name] = hash`) and appends to the history that history. So a new alias revision is created exactly when an alias head advances to a hash it was not already pointing at: -- **On create** — `catalog_create("orders", ...)` writes version 1. -- **On revise** — `catalog_revise("orders", ...)` appends version 2 for the one - alias you edited. +- **On create** — `catalog_create("by_region", ...)` writes version 1, and so + does the first `catalog_import_source(..., "orders")`. +- **On revise** — `catalog_revise("by_region", ...)` appends version 2 for the + one alias you edited. +- **On import** — `catalog_import_source(..., "orders")` with bytes that differ + from the head appends the next version of the source alias. - **On recalc** — each dependent that **rebuilds to a new hash** has its alias advanced to a new revision. In the example above, the single committed recalc appended `by_region` v2 and `top_regions` v2. This is the "new alias revisions @@ -198,13 +215,14 @@ Three edges make that precise: - A **`noop`** entry — one that replays to the *same* hash because its inputs didn't really move — gets no new revision. Its alias is left untouched (and `set_alias` dedupes a consecutive duplicate hash regardless). -- The revision is appended **per advanced alias head, not per rebuilt entry**. A - rebuilt entry that has no alias on its head — an unnamed scratch entry, or an - intermediate node referenced only by a hash pin — produces a new hash and a - `remap` entry but **zero** alias revisions. A head carrying two aliases advances - both. -- A **hash-pinned** child (`pinned_expr_from_alias("")`) re-resolves to the same - parent and is a `noop`, so it neither rebuilds nor advances its alias. +- The revision is appended **per advanced alias head, not per rebuilt entry**. + Since #154 the cone holds only current alias heads, plus any entry you named + in `roots`. A named root with no alias on it (an unnamed scratch entry, say) + produces a new hash and a `remap` entry but **zero** alias revisions. A head + carrying two aliases advances both. +- A **version-pinned** child (`pinned_expr_from_alias("orders-v1")`) re-resolves + to the same parent and is a `noop`, so it neither rebuilds nor advances its + alias. This per-alias revision is distinct from the catalog-level checkpoint. The alias head advancing is bookkeeping inside `aliases.jsonl`; the checkpoint is the single @@ -222,27 +240,29 @@ each query (cheap for a notebook-sized catalog — one small JSON read per entry ### Where the edges live -Each entry has a `manifest.json` with two fields that carry the graph: +Each entry has a `manifest.json`, and one field in it carries the graph: - **`manifest.parents`** — the resolved cross-entry edges, a list of `{hash, ref, follow}`. `hash` is the parent's build-time content hash, `ref` the - original argument (alias or hash), `follow` its read-intent. Empty for a root (an - entry that reads no other entry). -- **`manifest.sources`** — the raw-file leaves, `{rel_path: digest}`: the project - files the recipe read via `read_project_file`, each with the content digest it - was built against. `None` when the entry was built under `off` identity mode (so - the source axis can't be evaluated); `{}` when the recipe read no raw files. + original argument (an alias or a version reference), `follow` its read-intent. + Empty for a root (an entry that reads no other entry). -`tracked_`/`pinned_expr_from_alias` write `parents`; `read_project_file` writes -`sources`. Together they are the only record of the graph. +`tracked_`/`pinned_expr_from_alias` write it, and it is the only record of the +graph. A source entry's manifest also records `provenance` (the path the file +was imported from, its digest, the reader options and the name it was imported +as), which says where its data came from and is not an edge: the file is never +read again. There used to be a second field, `manifest.sources`, a per-file digest +map that children inherited from their parents; ADR-011 D6 deleted it once every +input became an entry, since the parent edges then record everything. ### Roots, leaves, and direction This doc uses the **data-flow** convention (as in Airflow, dbt, Spark lineage): edges point downstream, the direction data moves. -- A **source / root** has no incoming edges — nothing it depends on: a - `read_project_file` raw input, or an entry with empty `manifest.parents`. +- A **source / root** has no incoming edges — nothing it depends on: a source + entry (a version of an imported file), or any entry with empty + `manifest.parents`. - A **leaf / sink** has no outgoing edges — nothing depends on it: a final aggregation nobody chains off. @@ -250,42 +270,45 @@ A recalc starts at the changed roots and flows down to the leaves. (Build-system tools — Make, Bazel — point the arrows the other way and swap these names, which is the usual source of confusion.) -### Cheap vs expensive parents — and what lands in `manifest.sources` +### Cheap vs worthy parents When a recipe reads a parent with `tracked_expr_from_alias`, what comes back -depends on whether the parent was classified **expensive** or **cheap** at build. +depends on whether the parent was classified **worthy** or **cheap** at build. This is the materialized-vs-not distinction: -- An **expensive** parent (an Aggregate, Join, Sort, or UDF) bakes a result - snapshot when built (materialized). A child reading it gets a *read of that baked - snapshot* — the parent's work is computed once and shared, and the child does not - re-run it. -- A **cheap** parent (row-wise: select, filter, mutate) bakes nothing - (non-materialized). A child reading it gets the parent's *recipe re-run inline* — - pushdown makes that ~free — so the parent's expression, down to its own - `read_project_file` reads, is composed into the child. - -That second case is why a cheap parent's sources show up in the **child's** -`manifest.sources`: the inlined recipe re-runs the parent's `read_project_file` -calls during the child's build, and those reads get collected into the child's own -source set. So editing that raw file makes the child **directly** stale on the -source axis, not merely transitively stale. A child reading an *expensive* parent -reads the snapshot instead, so the parent's raw sources never enter the child's -sources — there the source edit makes the *parent* directly stale, and the child -follows only when the parent's alias advances. +- A **worthy** parent (an aggregate, join, sort, window function, union, distinct, + unnest, a second file, a non-pure operation, or a UDF, and every source entry) + is materialized: its + result was written to a snapshot file when it was built. A child reading it gets + a *bare read of that snapshot* — the parent's work is computed once and shared, + and the child does not re-run it. The snapshot's path contains the parent's + content hash, so the child's own hash changes when the parent it was built on + changes. +- A **cheap** parent (row-preserving over one file: select, filter, mutate) has no + file of its own. A child reading it gets the parent's *frozen graph inlined* — + pushdown makes re-running it ~free — so the parent's expression, down to the + snapshot it reads, is composed into the child. + +Either way the child records one edge, to the parent it named, and nothing about +the parent's own inputs. New data reaches it through the aliases: an import +advances the source alias, the entries that follow it go directly stale, and the +entries built on those are transitively stale. Which work is materialized is otherwise an implementation detail you don't see: `tracked_expr_from_alias` returns an expression either way, and the child never -depends on the parent having a `result.parquet` on disk. +depends on the parent having a `result.parquet` on disk. It does depend on the +parent's snapshot existing when the child is built, so building a child first makes +the parent's files exist (`ensure_materialized`), re-creating a deleted snapshot if +it has to. ### Build-time capture, not live introspection The edges are captured the moment a recipe runs, not by walking the built -expression afterward. While `build_and_persist` imports the recipe, two collectors -are armed, and each loader announces itself as it executes: `read_project_file` -notes the source digest, `tracked_`/`pinned_expr_from_alias` notes the resolved -parent hash. After the import those bags are written into the manifest. The capture -happens once, at build; nothing re-derives it later. +expression afterward. While `build_and_persist` imports the recipe, one +collector is armed (`parent_capture`), and `tracked_`/`pinned_expr_from_alias` +note the resolved parent hash in it as they execute. After the import the +collected edges are written into the manifest. The capture happens once, at +build; nothing re-derives it later. This is a **tallyman** mechanism, not an xorq one. xorq's unit is a single expression and its content hash; it has no concept of a catalog, an alias, an @@ -296,20 +319,25 @@ loaders announced, and writes the manifest beside xorq's artifact. The capture is a side-channel because the parent edge is **not recoverable from xorq's output**. Since the expression-composition change (#73/#74), `tracked_expr_from_alias` composes the parent's *expression* into the child (cheap: -the inlined recipe; expensive: a read of the baked snapshot) rather than reading -the parent's `result.parquet` by a path that named it. The result is one flattened -graph with no node that says "this subtree was entry ``." Before the -change a child read `entries//result.parquet` — a leaf path naming the -parent — and the edge was readable straight out of the expression. After it, the -only place the edge survives is the manifest tallyman wrote. +the parent's frozen graph inlined; worthy: a read of the parent's snapshot) rather +than reading the parent's `result.parquet` by a path that named it. For a cheap +parent the result is one flattened graph with no node that says "this subtree was +entry ``." Before the change a child read +`entries//result.parquet` — a leaf path naming the parent — and the +edge was readable straight out of the expression. After it, the only place the +edge survives for a cheap parent is the manifest tallyman wrote. A worthy parent's +snapshot path, `compute_cache/result_cache/.parquet`, does carry the +parent's hash, but not the alias the recipe named or whether the edge follows or +pins, and those live only in the manifest too. ### Reading it back: one seam All graph reads go through `tallyman_xorq.dependents`, which scans the manifests and assembles the views the reactive system needs: -- `parents_of(hash)` / `sources_of(hash)` — one entry's recorded inputs. -- `build_dag()` — forward edges for every live entry, `{child: [parents]}`. +- `parents_of(hash)` — one entry's recorded parent edges. +- `build_dag()` — forward edges for every complete entry (one with a + `manifest.json`), `{child: [parents]}`. - `dependents_index()` — the reverse index, `{parent: {children}}`: how a recalc goes from a changed entry to everything downstream. - `descendant_cone(roots)` — the roots and all their transitive dependents, @@ -319,65 +347,50 @@ Both `staleness` and `recalc` go through this module rather than touching manife fields directly, so it is the single seam over the recorded graph: swapping the implementation (a persistent index, say) is a change to `dependents` alone. -A consequence worth noting (and a candidate future check): because the parent edge -is gone from xorq's flattened expression, you can't reconcile tallyman's recorded -parents against xorq's graph. What you *can* reconcile is the source leaves — a -child's inlined raw-file reads should equal the union of `sources` over its -transitive cheap parents. That invariant is checkable; the parent edges are only as -good as what the side-channel captured at build. - -## The two staleness axes - -The workflow above drives the **alias axis**. There is a second axis for source -files that change on disk, which a closed notebook catalog mostly doesn't hit, but -it exists and is worth knowing. - -An entry is judged on two independent axes, each tied to a kind of recorded input: - -- **alias** — a `follow=True` parent (recorded when the recipe referenced a parent - by alias name, `tracked_expr_from_alias("orders")`) is stale when that alias now resolves to - a different hash than the one recorded at build. This is what fires when you - revise an upstream alias. A `follow=False` parent (a `pinned_expr_from_alias` - reference, by alias or hash) is never stale on this axis: the recipe asked for - *that* revision and still gets it. -- **source** — a recorded `(rel_path, digest)` is stale when the file on disk no - longer digests to the recorded value. The check forces a faithful re-read so an - in-place content swap can't be masked by a cached digest. In a closed catalog - you don't expect this to fire; it covers the case where a raw parquet under - `data/` is replaced. - -A note on cheap chains: when a child reads a *cheap* (non-materialized) parent, -that parent's recipe is inlined into the child, so the parent's source files land -in the child's own `manifest.sources` (see *Cheap vs expensive parents* above). -Editing such a source makes the child **directly** stale on the source axis, not -merely transitively stale. This is expected. +One consequence: because a cheap parent's edge is gone from xorq's flattened +expression, you can't fully reconcile tallyman's recorded parents against xorq's +graph, and the parent edges are only as good as what the side-channel captured +at build. + +## One staleness axis + +An entry is stale when a `follow=True` parent — recorded when the recipe +referenced a parent by alias name, `tracked_expr_from_alias("orders")` — now +resolves to a different hash than the one recorded at build, and for no other +reason (ADR-011 D6, `plans/ADR-011-sources-are-aliases.md`). This is what fires +when you revise an upstream alias or import new data under a source alias. A +`follow=False` parent (a `pinned_expr_from_alias("orders-v1")` reference) is +never stale: the recipe asked for *that* revision and still gets it. + +A data file that changes outside tallyman is not a reason. Until it is imported +again nothing in the catalog has changed, and the builds read tallyman's copy of +the bytes, never the file. So the scan opens no data file; it reads +`aliases.jsonl` and the manifests. There used to be a second axis, a recorded +source digest no longer matching the file on disk. It could not tell how an +entry had come to depend on a file, so a child pinned to its parent by hash read +as stale forever, and the import model removed it. + +Only an entry that is the current head of an alias can be directly stale (#154). +A superseded version still records inputs that have since moved, but it heads no +alias, so recomputing it would re-point nothing; the scan reports it with +`live: false` and `stale: false`, and keeps its `reasons` for the record. + +A note on chains: an entry records only its own parent edges, so an import makes +the direct followers of the source alias **directly** stale and everything built +on them **transitively** stale, whether the parents in between are cheap or +worthy. `result_digest` is deliberately not a staleness input. An entry that recomputes to the same `content_hash` but a different result is *nondeterministic*, not stale, and recompute can't make it fresh. That is the #83/#121 concern, separate from this system. -### Prerequisite: `cas` source identity (source axis only) - -Source-axis staleness only works under content-addressed source identity, which is -the default. `source_identity.mode()` reads `TALLYMAN_SOURCE_IDENTITY` and falls -back to `"cas"` (`source_identity.py:59`). In `cas` mode, `read_project_file` reads each -source through a copy-on-write clone at `data/.cas/`, so the path -xorq hashes is the content identity: editing a source in place yields a different -digest, the manifest records that digest at build time, and a later scan can tell -the file moved. The clone also means an old entry's recompute still reads the bytes -it was built from after you edit the source. - -Under the legacy `off` mode the manifest records no source digests, so the source -axis cannot be evaluated. A scan reports it as `unknown` rather than silently -fresh. If you have a corpus built under `off`, rebuild it — there is no migration -path and none is wanted (single-user repo). The **alias axis is unaffected** by -this mode; it works regardless. - ## Detecting staleness (the scan) -The scan classifies every live entry and tells you two things per entry: whether it -is **directly stale** (its own recorded inputs moved) and whether it is +The scan classifies every complete entry (every entry directory with a +`manifest.json`, current alias heads and superseded versions alike) and tells +you two things per entry: whether it is **directly stale** (its own recorded +inputs moved, and it is a current alias head) and whether it is **transitively stale** (a clean entry sitting downstream of a directly stale ancestor, so a recalc would carry it along). @@ -399,17 +412,22 @@ Read-only, takes no checkpoint. It returns: "stale": true, "reasons": [{"axis": "alias", "ref": "orders", "was": "", "now": ""}], "unknown_axes": [], - "transitively_stale": false + "transitively_stale": false, + "live": true // the entry is a current alias head }, ... - } + }, + "orphan_stale": [ ... ] // empty unless auto-recalc is on; see Trigger model } ``` -Each `reason` names the axis, the `ref` (alias name or source rel_path), the value -`was` at build time, and the value `now` (or `null` if the input is gone). An axis -that can't be evaluated — alias deleted, source file removed, or the entry was -built under `off` mode — lands in `unknown_axes` and never forces `stale: true`. +Each `reason` names the axis (always `"alias"`; the field is kept so a reason +has the same shape everywhere), the `ref` (the alias name), the parent hash that +`was` recorded at build, and the alias head `now`. A parent alias that no longer +exists can't be evaluated; it lands in `unknown_axes` as `alias:` and never +forces `stale: true`. `catalog_scan_staleness(verify_results=True)` +also checks every snapshot on disk against its recorded `result_digest`; see +[mcp-server.md](mcp-server.md). ### HTTP @@ -420,22 +438,24 @@ GET /{project}/api/staleness Read-only, same data shape plus the project name: ```json -{ "project": "...", "stale": [...], "transitively_stale": [...], "entries": {...} } +{ "project": "...", "stale": [...], "transitively_stale": [...], "entries": {...}, "orphan_stale": [...] } ``` -This is the scan-on-load surface. The SPA is expected to fire it on project load -and after a recalc to drive a per-entry badge (the badge UI itself is not built yet -— see Current limits). +This is meant as the scan-on-load surface for a per-entry staleness badge. The +SPA does not call it yet, and there is no badge (see Current limits). ## Recomputing (the recalc) Recalc takes a set of *roots*, computes their descendant cone (every entry -reachable downstream, topologically ordered parents-before-children), and replays -each recipe in that order through `build_and_persist` — the same primitive -`catalog_revise` uses. A replay reads the *current* alias heads, so once a parent -has been re-pointed to its fresh hash, its dependents replay against the advanced -parent. An entry whose inputs didn't actually move replays to the same -`content_hash` and short-circuits to a no-op. +reachable downstream through recorded parent edges, topologically ordered +parents-before-children), drops from it every entry that is not a current alias +head unless it was named as a root (#154: replaying a superseded version would +only make an entry nothing points at), and replays each recipe in that order +through `build_and_persist` — the same primitive `catalog_revise` uses. A replay +reads the *current* alias heads, so once a parent has been re-pointed to its +fresh hash, its dependents replay against the advanced parent. An entry whose +inputs didn't actually move replays to the same `content_hash` and +short-circuits to a no-op. ### Preview, then commit @@ -447,7 +467,8 @@ Dry-run actions: - `rebuild` — directly stale, its own inputs moved. - `cascade` — clean, but a followed parent inside the cone will advance. -- `unchanged` — pinned to an out-of-cone parent, or a root that isn't stale. +- `unchanged` — not stale itself, and no followed parent inside the cone: a child + that pins a version of its parent, or a root that isn't stale. A real run (`dry_run=False`) replays for real and reports: @@ -502,10 +523,11 @@ Content-Type: application/json Both body fields are optional; `dry_run` defaults to True and `roots` defaults to the stale set, matching the MCP tool. The response is the same report (without the -`note` field on the empty case). A committed run additionally invalidates the -companion's result/compare caches, reloads any live buckaroo sessions, and -publishes a `recalc` SSE event. The route is blocked with a 403 in read-only -(serve) mode. +`note` field on the empty case). A committed run that changed something +additionally clears the companion's result and compare memos and its +`diff_stat_cache/` directory, reloads the project's Buckaroo sessions (one +`/reload_expr` per catalog entry, #201), and publishes a `recalc` SSE event. The +route is blocked with a 403 in read-only (serve) mode. ### One revision, undoable atomically @@ -519,13 +541,14 @@ it was, rolling back the whole cascade at once. ### Failure and cycles -The cascade-failure policy is **stop-and-report**, the plan's default. If a replay -raises, the walk stops at that entry: the already-recomputed prefix stays committed -(so its checkpoint still records), the failing entry is reported with +The cascade-failure policy is **stop-and-report**. If a replay raises, the walk +stops at that entry: the already-recomputed prefix stays committed (so its +checkpoint still records), the failing entry is reported with `action: "failed"` and the error text, and everything after it is `skipped`. The call does not raise — a caller gets a structured `status: "failed"` report, not a -500. The two other policies in the ADR (skip-and-continue, all-or-nothing) are not -implemented. +500. The recalc design also considered skip-and-continue and all-or-nothing; +neither is implemented. The failure, and every skipped entry that was itself +directly stale, is recorded in `errors.jsonl` under its hash. If the dependency graph contains a cycle, recalc detects it while building the cone, before any recipe runs, and returns `status: "cycle"` with the cycle described in @@ -562,21 +585,19 @@ event's `kind`. A client must therefore register a **named** listener, What works today is the backend, end to end, through both the MCP and HTTP surfaces: scan, preview, commit, atomic revision, cross-process invalidation, and -the published SSE event. What is not yet built: - -- **Frontend rendering.** The SPA's `SSEContext` listens for nine named events - (`hello`, `ping`, `new_entry`, `build_failed`, `notebook_changed`, - `chart_attached`, `post_processing_changed`, `summary_stat_changed`, - `project_switched`) but **not** `recalc` — the backend emits it, nothing consumes - it, so open views do not auto-refresh after a recalc yet. Auto-recalc-on-revise - does not change this: it emits the same `recalc` event, which the SPA still - ignores, so a view open on a cascaded dependent needs a manual refresh. Wiring the - listener is the one piece that makes the cascade visible in the viewer. There is - also no staleness badge and no recalc affordance in the UI, and nothing calls - `/api/staleness` or `/api/recalc` from the frontend. For now, drive recalc through - MCP or a direct HTTP call and refresh the view manually. -- **Deletion-side semantics.** Archive / delete-cone behavior and multi-parent - unbind (ADR Q3/Q4) are out of scope for this system as built. +the published SSE event. The SPA listens for `recalc`: each event refetches the +catalog list, and an entry view open in a background tab whose hash was remapped +navigates to the new hash. A focused tab stays put, so that a code edit in +progress is not thrown away; its URL still names the old hash, so a reload shows +the old version too, and the new one is reached from the refreshed catalog list. +What is not yet built: + +- **Staleness and recalc in the UI.** There is no staleness badge and no recalc + button, and nothing in the SPA calls `/api/staleness` or `/api/recalc`. Drive an + explicit recalc through MCP or a direct HTTP call. +- **Deletion-side semantics.** Archive and delete-cone behaviour, and unbinding + one parent of an entry that has several, are out of scope for this system as + built. - **Alternative cascade policies.** Only stop-and-report exists. ## Trigger model @@ -588,12 +609,17 @@ scan-then-recalc path. The scan surfaces (`catalog_scan_staleness` / ### Auto-recalc on revise (default) Revising an alias recomputes its stale followers in the **same** checkpoint as the -head advance. Every surface that re-points an *existing* alias triggers it — -`catalog_revise`, the companion `PUT …/api/code/{alias}`, and `catalog_promote_diff` -when it re-points an existing target. The cascade is scoped to *this* revise's -followers: roots are the entries the advance made directly stale on the alias axis, -and the descendant cone expands from there. Pre-existing staleness from any other -cause is left untouched. +head advance. Every surface that advances an *existing* alias to a new version +triggers it — `catalog_revise`, the companion `PUT …/api/code/{alias}`, a diff +promotion (`catalog_promote_diff` or the companion's promote route) when it +re-points an existing target, and `catalog_import_source` when it mints a new +version of a source alias. `catalog_alias`, which points a name at an entry that +already exists, does not. The cascade is scoped to *this* advance's followers: +roots are the entries it made directly stale, and the descendant cone expands +from there. Pre-existing staleness from any other cause is left untouched. +The companion's two routes build the new entry on its event loop, so the whole +UI stops answering until that build (and any wait for the project lock) is over +(#190); the cascade itself runs on a worker thread. The whole thing is one git revision. The head advance, the rebuilt dependents, and their re-pointed aliases are swept into the surface's single per-operation @@ -604,8 +630,8 @@ the rest are skipped. The revise reply carries a `recalc` sub-report — `catalog_recalc`'s shape (`cone`, `entries`, `remap`, `status`) plus `orphan_stale`. `orphan_stale` lists every entry -that was directly stale but *not* a follower of this revise (a drifted source, a -different alias not yet recalced, or an invariant break): not recomputed, logged at +that was directly stale but *not* a follower of this revise (a different alias +not yet recalced, or an invariant break): not recomputed, logged at WARNING, and classified against the durable error store — "explained by error ``" when a recorded recalc/build failure carries that hash, else "UNEXPLAINED … file a bug." Cascade failures are persisted to `errors.jsonl` (keyed by the @@ -622,5 +648,6 @@ path below. With the switch off, advancing an alias marks its followers stale but recomputes nothing. `catalog_recalc` / `POST …/api/recalc` is then the deliberate action that writes — preview with a dry run, commit with `dry_run=False`. This is also the path -for staleness the auto trigger never sees: a drifted source file, or any alias -left stale because auto-recalc was off when it was revised. +for any alias left stale because auto-recalc was off when it was revised or +imported. A source entry passed as a root fails instead of being skipped +(#238). diff --git a/docs/system-contract.md b/docs/system-contract.md index 67daa228..0a2170d5 100644 --- a/docs/system-contract.md +++ b/docs/system-contract.md @@ -1,9 +1,22 @@ # Tallyman: primitives and the system contract -- **Status:** Normative, implemented (PR #167, 2026-07-31). Describes the - system as built; where code or the descriptive docs disagree with it, the - disagreement is a bug. Design decisions and pre-fix history: - `plans/ADR-006-read-path-loads-builds.md`. +- **Status:** Normative, implemented. Describes the system as built; where + code or the descriptive docs disagree with it, the disagreement is a bug. + The known disagreements, each with its open issue, are listed under + [Known deviations](#known-deviations) at the end. + Design decisions and history: `plans/ADR-006-read-path-loads-builds.md` + (the read path, still in force for what the later ADRs did not change) and + `plans/ADR-007-tallyman-owned-materialization.md`, + `plans/ADR-008-row-order-of-reads.md` and + `plans/ADR-009-digest-stability.md` (tallyman writes its own result files, + every file carries `__row_order`, and the digest is a content digest), + accepted on 2026-09-22 and implemented in #189. + `plans/ADR-010-immutable-store-one-owner.md`, a later proposal to replace + those three, was rejected the same day. + `plans/ADR-011-sources-are-aliases.md` (a raw input is a source alias whose + versions are entries, a file enters only by an explicit import, recipes name + aliases and never bare hashes, and staleness has one axis), accepted the same + day and implemented in #217, #218 and #219. - **Audience:** no prior xorq knowledge assumed. The xorq section below covers exactly as much of xorq as the rest of the doc needs, and no more. @@ -54,11 +67,11 @@ backends shape tallyman's design: The remedy for the identity trap is **`replace_sources`**: given a mapping {old backend object → new backend object}, it rewrites an expression rebinding -every leaf and every cache reference onto the new object. This is the +every leaf onto the new object. This is the sanctioned way to take expressions that arrived on different backend objects and compose them onto one shared backend. -**Rebinding is zero-copy.** A file read or cache node is a path plus a recipe +**Rebinding is zero-copy.** A file read is a path plus a recipe for reading it; swapping which connection executes it moves no data — the new backend reads the same files at execute time. (Tallyman's backends are stateless in-process engines; all data state lives in files.) The one node @@ -69,7 +82,8 @@ it: the build forbids the authoring patterns that create such nodes (in-memory reads are a build error), so every leaf is a file read and the refusal is a loud assertion that the build gate failed, never a live copy path. Beside it, rebinding fails loudly if a build ever spans more than one -distinct backend content profile (ADR D3). +distinct backend content profile (ADR-006 D3, rebind composition onto the +default backend with a one-group guard). ## 3. Builds: freezing an expression to disk @@ -100,45 +114,33 @@ Three properties matter: (build artifacts should be reproducible from what the expression *is*), and it is the single most consequential xorq design choice for tallyman: **if you want content identity, you must put the content in the path.** Tallyman - does (Part 2, CAS). + does (Part 2, "Project, imports and source entries"). -`load_expr(build_dir, cache_dir=...)` is the inverse: it reads the yaml, -mints **fresh backend objects** from the profiles (nothing is memoized — two -loads of the same build yield two distinct backend objects, which is the -identity trap again), and rebases cache storage into `cache_dir` (next -section). The `cache_dir` parameter is the seam that lets the same build run -against any cache directory, including an empty one. +`load_expr(build_dir)` is the inverse: it reads the yaml and mints **fresh +backend objects** from the profiles (nothing is memoized — two loads of the +same build yield two distinct backend objects, which is the identity trap +again). A build that holds no cache node needs nothing else to load, and +tallyman's builds hold none (Part 2, "Materialization"). -## 4. Caching: snapshot keys, and existence as the only check +## 4. xorq's cache nodes, which tallyman does not use Calling `.cache()` on an expression wraps it in a **cache node**. At execution -time xorq computes a **key** from the wrapped subgraph and looks for -`//.parquet`: - -- file exists → read it back; the subgraph does not run; -- file missing → execute the subgraph, write the file (atomically), read it. - -That is the entire mechanism. There is no invalidation step and no validation -of the file's contents: **a cache hit is decided by filename existence -alone**, and the stored bytes are trusted by assumption. Consequently the -cache is only as honest as two things: the key derivation, and whoever writes -the files. - -Tallyman uses xorq's **snapshot** keying everywhere: the key is a token over -the subgraph's structure and its read *paths* — again deliberately blind to -file content and mtime.[^mtime] Under CAS paths (Part 2) this blindness is -harmless: content is in the path, so the key is content-honest. The cache node survives -serialization: a build of a cached expression carries the cache node in its -yaml, so *any* loader of that build reads through the same cache key. What is -**not** serialized is the cache's base directory — every loader must supply -`cache_dir`, which is why the parameter exists. - -One xorq wart tallyman must own: xorq's load-time cache-dir rewrite walks the -graph shallowly and misses cache nodes nested *inside* another cache node's -subgraph. Tallyman's loader therefore does its own deep rewrite (over xorq's -deep-walking primitives) so that every cache node in a loaded build — nested -or not — lands in the project's cache directory, never in the global default -(`~/.cache/xorq`). +time xorq computes a key from the wrapped subgraph and looks for +`//.parquet`: if the file exists it is read back +and the subgraph does not run, and if it is missing the subgraph runs and the +file is written. A hit is decided by filename existence alone, so the cache is +only as honest as the key derivation and whoever writes the files. The node +survives serialization, but its base directory does not: a loader has to supply +one, and xorq's load-time rewrite of it misses nodes nested inside another +cache node's subgraph. + +Tallyman stopped depending on all of this (`plans/ADR-007-tallyman-owned-materialization.md`). +No build holds a cache node, so no loader has to aim one at a directory, the +file names come from tallyman and not from xorq's tokenization of a graph, and +the writer is tallyman's own (Part 2, "Materialization"). A recipe that calls +`.cache()` is a build error, because a node with default storage would write +under `~/.cache/xorq`. What remains of xorq here is the expression, build, load, +hashing and execution layer. ## 5. What the hash sees, and what it cannot @@ -159,66 +161,109 @@ adds the output axis itself (Part 2, `result_digest`). answer to new — built to serve the current file's answer fast, nothing more. Tallyman needs the opposite: entries are history, so identity must hold still while data moves (that is what makes V3-vs-V4 meaningful), and - change is handled explicitly instead — the manifest records what the - world *was* (source digests, parent hashes), the staleness scan detects - by comparing that record against the live world, recalc mints new entries - and advances aliases, and old entries keep serving their old bytes - forever. A key whose job is to move when data moves has nothing to hang - that on. Bolting it under tallyman's caches would also collide with - identity mechanically: the expression hash is hardwired to path-only - normalization (measured in the source-identity ADR), so mtime-keyed - caches would move while entry hashes stood still — fresh bytes under an - old name, the #163 failure shape. CAS serves both needs with one - mechanism: content in the path gives the hash and every cache key an - identity that moves exactly when content moves, and the recorded digests - give staleness something durable to compare. (One stat use survives, as - an accelerator: the source-digest memo skips re-hashing files whose stat - is unchanged; content stays the truth.) + change is handled explicitly instead — new data is imported as a new + version of a source alias, the manifest records which version of each + parent an entry was built on, the staleness scan compares that record with + the alias heads, recalc mints new entries and advances aliases, and old + entries keep serving their old bytes forever. A key whose job is to move + when data moves has nothing to hang that on. Bolting it under tallyman's + caches would also collide with identity mechanically: the expression hash + is hardwired to path-only normalization (measured in the source-identity + ADR), so mtime-keyed caches would move while entry hashes stood still — + fresh bytes under an old name, the #163 failure shape. Content-named files + serve both needs with one mechanism: the hash and every cache key move + exactly when content moves, and an import is the one event that moves + them. --- # Part 2 — The tallyman primitives -## Project, sources, and CAS +## Project, imports and source entries A **project** is a directory (`~/.tallyman-notebooks/projects//`) holding -user data files (`data/`) and the catalog (`artifacts/catalog/`, a git repo). - -A **source** is a user-provided data file. Sources are mutable on disk — the -user can overwrite `trips.parquet` any time — so tallyman never lets an entry -depend on a live source path. At build time each source is: - -1. digested (md5, memoized on stat so unchanged files hash once), -2. cloned copy-on-write to `data/.cas/`, -3. read through the clone. - -This is **CAS** (content-addressed sources), and it is the answer to xorq's -path-only hashing: the path *is* the digest, so every xorq-level key — the -expression hash, every cache key — becomes content-honest for free. Edit a -source and rebuild: the digest changes, the path changes, the hash changes, a -new entry forks. The clone is the entry's immutable input forever; the live -file is merely where the *next* build will look. +the catalog (`artifacts/catalog/`, a git repo) and the clone store +(`data/.cas/`). + +**A recipe names aliases. A build reads only files tallyman owns.** A data file +the user has is mutable and outside tallyman's control, so no build ever reads +one. It enters the catalog by one explicit act, an **import** +(`catalog_import_source`, `source_import.update_and_depend`), which: + +1. digests the file (md5) and clones its bytes, copy-on-write where the + filesystem offers it, to `data/.cas/` (the **clone**), + digesting the clone again and refusing it if it does not match its name; +2. writes one parquet file of its rows, in file order plus a last `__row_order` + column, `0..N-1`, to `compute_cache/result_cache/.parquet`: + pyarrow copies a parquet file, keeping its types, and polars parses a CSV + under the schema and `scan_csv` options the call names (a timestamp column + whose schema type has a zone keeps the wall-clock time of text with no UTC + offset, converts text with one into the zone, and refuses a time the zone + skips or repeats and a column that mixes the two kinds of text); +3. writes an entry, the **source entry**, with a generated recipe, a frozen + build, a schema and a manifest whose `provenance` records the outside path, + the digest, the reader options and the name it was imported as; +4. appends that entry to a **source alias**, as its next version. + +The outside path is provenance from then on and is never read again: editing, +moving or deleting the file changes no build. To bring in new data, import the +file again under the same alias. Different bytes mint the next version, and the +entries that follow the alias go stale exactly as after a revise. The same bytes +are a no-op. + +A source entry's content hash is `md5("source||")`, +truncated to 12 hex characters: the bytes and the reader options, and nothing +else. This matters because xorq hashes a file read by its path alone. Every file +a recipe's expression reads is a snapshot, named by the content hash of the +entry it holds, so every xorq-level key (the expression hash, and the name of +every file tallyman writes) is content-honest, and the chain of names ends at +hashes of bytes. The reader options are fixed at import (a CSV read two ways is +two imports under two aliases), and they must be plain values that the entry can +record: a callable option is refused. + +The import's outcomes form a table (ADR-011 D3). With no `pinned_version`: a new +alias mints v1, bytes that differ from the head mint the next version, and bytes +equal to the head are a no-op. With `pinned_version=N`: the file must be version +N (a no-op) or, as N = head + 1, new bytes; anything else is an error, and +versions cannot be skipped. History is append-only, so bytes equal to a version +older than the head are refused, naming `reset_to` as the way back. And **one set +of bytes, read one way, is one version under one alias**: bytes another alias of +the project already holds are refused, naming that alias; a second name for them +is a catalog entry whose recipe reads `tracked_expr_from_alias("")`. +The rule is per project: two projects importing one file each hold their own +entry, snapshot and clone under the same hash. ## Recipe A **recipe** (`expr.py`) is the LLM-authored Python that defines a computation. It must bind a variable `expr`, built from: -- `read_project_file("trips.parquet")` — read a source (through CAS); -- `tracked_expr_from_alias("trips")` — build on another entry, named by alias, - recording a `follow=True` parent edge: this expression depends on the parent - alias, and when that alias advances, recalc mints a new version of this - expression; -- `pinned_expr_from_alias()` — same, but `follow=False`: - recalc never touches this expression when the parent moves. Pins name an - exact version — a hash, or an explicit `"trips-v3"` version reference; a - bare alias is rejected as ambiguous (it would silently pin whatever the - head happened to be when the recipe was built — #166). - -The defining property of a recipe: **it binds by name.** "trips.parquet" -means whatever bytes sit there right now; "trips" means whatever entry that -alias points at right now. A recipe therefore has a different meaning at -different moments. That is exactly what you want when *authoring* — and +- `tracked_expr_from_alias("trips")` — build on another entry, named by alias + (a source alias or a catalog alias), recording a `follow=True` parent edge: + this expression depends on the parent alias, and when that alias advances, + recalc mints a new version of this expression; +- `pinned_expr_from_alias("trips-v3")` — same, but `follow=False`: recalc never + touches this expression when the parent moves. A pin names an exact version + with an explicit version reference. A bare alias is rejected as ambiguous (it + would silently pin whatever the head happened to be when the recipe was built, + #166), and so is a bare content hash (ADR-011 D5), so every parent edge an + authored recipe records names an alias, and no opaque hash appears in one. An + entry with no alias has to be named before anything can build on it. The one + generated exception is a promoted diff, whose recipe calls + `build_diff_expr(a_hash=..., b_hash=...)` with the two hashes and records no + parent edge (below, "Composition"). + +A recipe never opens a file. `read_project_file`, `tallyman_read_csv`, +`xo.deferred_read_csv`, and `xo.deferred_read_parquet` of a file outside +`compute_cache/` are build errors, each naming the import to use. The one recipe +that calls `read_project_file` is the one the importer generates for a source +entry, where a context variable resolves the call to that entry's own snapshot; +the path it names is provenance. + +The defining property of a recipe: **it binds by name.** "trips" means whatever +entry that alias points at right now, and for a source alias that is whichever +version of the file was imported last. A recipe therefore has a different +meaning at different moments. That is exactly what you want when *authoring* — and exactly what you must never consult again afterward. ## Entry @@ -232,7 +277,7 @@ entries// xorq_build/ # the frozen build (Part 1 §3), paths made portable manifest.json # the closure record (below) schema.json -entries/.zip # git-tracked durable form, written at checkpoint +entries/.zip # git-tracked record, written at checkpoint, never read back ``` An entry carries **two representations of its computation, with different @@ -240,7 +285,7 @@ authority**: | | `expr.py` (recipe) | `xorq_build/` (build) | |---|---|---| -| binds inputs by | name (aliases, live paths) | value (parent graphs inlined, CAS paths) | +| binds inputs by | name (aliases) | value (a worthy parent's snapshot path, a source version's included; a cheap parent's graph inlined) | | meaning over time | drifts as names move | fixed forever | | authoritative for | authoring: revise, display, the *next* build | semantics: every read, forever | @@ -253,30 +298,38 @@ resolving names to current heads is the point. (#163 is what happens when this rule is broken: reads re-ran recipes, so historical entries silently re-bound to today's parents.) +A source entry has both representations too, and the rule holds for it +trivially: its recipe is generated by the import and never re-run to make +anything, and the build the import records reads its own snapshot. + ## Content hash An entry's `content_hash` is xorq's expression hash (Part 1 §3), taken at build time. Two details determine what it covers: -- the expression hashed is the **rewritten** one — cache injection (below) - has already happened — so an entry's caching shape is part of its identity; -- every file read in the graph points at a CAS clone, so each read path - carries a digest of the input bytes; and since parent expressions are - inlined into the graph, the same holds all the way up the lineage. +- the expression hashed is the one after the rewrite (below), which adds the + canonical sort to a worthy entry and adds no cache node to anything; +- every file read in the graph is a worthy parent's snapshot, named by the + parent's content hash, so a child's hash is a function of its parent's. + +A source entry's hash is the exception, and the base of every chain: it is +computed from the imported bytes and the reader options (Part 2, "Project, +imports and source entries") rather than taken from xorq, because its generated +recipe reads the snapshot that the hash names. -The hash therefore names "this computation over these exact input bytes." +So the hash names "this computation over these exact input bytes." That one property makes builds idempotent and history append-only. Rebuild the same computation over unchanged inputs and you land on the existing entry: the build recognizes the hash and stops. Change anything that alters -the computation or its inputs — the recipe's logic, a source's bytes, a -parent's graph — and a new entry forks under a new hash, while every existing -entry keeps its name and its meaning. +the computation or its inputs — the recipe's logic, a parent's identity, and +through a new source version the imported bytes — and a new entry forks under a +new hash, while every existing entry keeps its name and its meaning. Two limits are deliberate. The hash cannot see execution behavior (Part 1 §5): a recipe calling `sample()` or `now()` hashes identically run to run, which is the gap `result_digest` (below) exists to police. And it makes no -attempt at cross-machine portability: absolute path prefixes participate in -the hash. +attempt at cross-machine portability: absolute path prefixes participate in a +computed entry's hash. (A source entry's hash has no path in it.) ## Manifest: the closure record @@ -287,21 +340,43 @@ itself doesn't state, so that no later operation ever needs to resolve a name: |---|---| | `content_hash` | the entry's identity | | `parents` | `[{hash, ref, follow}]` — each alias reference, **resolved to the exact hash it meant at build time** | -| `sources` | `{rel_path: digest}` — each source, pinned to the bytes read | -| `result_digest` | SHA-256 of the baked result snapshot (worthy entries) — the output identity | -| `snapshot_key` | the baked snapshot's cache key, recorded at build (see write path) | -| `cache_worthy`, `cache_worthy_why`, `cache_bytes` | the worthiness verdict and its evidence | +| `provenance` | a source entry only: `{alias, version, path, digest, suffix, reader, imported_at}` — the name it was imported as, the outside path (never read again), the digest of the bytes, and the reader options that let its snapshot be made again from the clone; its presence is what makes an entry a source entry | +| `cache_worthy`, `cache_worthy_why`, `cache_bytes` | whether the entry is materialized, decided once at build, and the evidence | +| `result_digest` | `arrow-sha256:` digest of the snapshot's content (worthy entries) — the output identity | +| `reproducible`, `nonreproducible_columns` | whether two runs at create gave the same digest, and the columns that differed | +| `unfaithful_heal_digest` | the digest the last unfaithful heal wrote; set, it pins the snapshot. The only field written after create | +| `snapshot_format`, `engine_versions` | the format version and the xorq, xorq-datafusion and pyarrow versions at build | | `row_count`, `execute_seconds`, `compile_seconds`, timings | build measurements | -The manifest is written last, atomically: its presence is the "this entry is -complete" sentinel. An entry directory without one is treated as absent. +The manifest is the entry directory's last write, atomic: its presence is the +"this entry is complete" sentinel. After that only an unfaithful heal rewrites +it, to record `unfaithful_heal_digest`, by an atomic replace under the project +lock. The recipe zip, written by the first checkpoint, keeps the manifest as it +was at create. An entry directory without a manifest is treated as absent by the +entry list, the checkpoint, recalc and the build, which builds it again. Every +read refuses it: `result_cache.entry_manifest` raises a `BuildError` naming the +missing `manifest.json` before anything is loaded or written, a child that reads +the entry raises the same error, and no route answers around it. Nothing stands +in for the manifest's `cache_worthy`. ## Alias -An **alias** is a mutable name: `{alias, latest, history}` in a git-tracked -file. `latest` is the head; `history` is every hash it has pointed at (V1…Vn, -oldest first). Revising an alias mints a new entry, advances `latest`, appends -to `history`. Old entries remain, immutable, as the version history. +An **alias** is a mutable name: `{alias, latest, history, kind}` in a +git-tracked file. `latest` is the head; `history` is every hash it has pointed +at (V1…Vn, oldest first). Revising an alias mints a new entry, advances +`latest`, appends to `history`. Old entries remain, immutable, as the version +history. + +An alias has a **kind**. A **catalog alias** names computations and advances by +revise, promote and recalc. A **source alias** names imported data and advances +only by an import: there is no recipe to revise and nothing to promote onto it, +and every surface that offers those refuses a source alias before building +anything. A name is one kind or the other, never both. **An alias's kind +matches its entries' kind**: `set_alias` never points a catalog alias at a +source entry or a source alias at a computed one, whichever route asks. +A source version is named by the alias that holds it now: `provenance` keeps +the name it was imported as, and a rename or an unalias does not rewrite it, so +a message that names a version or advises an import asks the alias store. **Where alias resolution is legal** — names resolve in exactly three situations, all of them *about* choosing or minting, never about serving: @@ -310,50 +385,159 @@ situations, all of them *about* choosing or minting, never about serving: hashes in the new entry's manifest. 2. **Selecting** (UI, diff version arithmetic): resolve "by_hour" or "V-1" to a hash, *then* serve that hash. -3. **Judging** (staleness scan): compare recorded parent hashes and source - digests against current heads and current files — read-only, executing - nothing. +3. **Judging** (staleness scan): compare recorded parent hashes against + current heads — read-only, executing nothing and opening no data file. Once an entry is selected, serving it consults no name again. `follow` is a **recalc** policy (should a parent's advance mint a new version of this entry?), never a read policy: reads are lineage-faithful for followed and pinned parents alike. -## Worthiness: cheap and expensive entries - -At build time, tallyman rewrites the author's expression before freezing it -(**the worthiness rewrite**): - -- every non-parquet file read gets a cache node injected after the parse (so - a CSV is parsed once, shared by every entry reading it); -- if the expression contains an expensive operation — Aggregate, Join, Sort, - Window, or a UDF — the whole expression is wrapped in a top-level cache - node. Such an entry is **worthy**: its result is materialized once at build - (**the baked snapshot**, a parquet under the project's - `compute_cache/result_cache/`) and read back ever after. -- everything else is **cheap**: a projection/filter/rename over columnar - sources re-executes in pushdown time comparable to reading a copy, so no - copy is kept. - -The rewrite happens *before* hashing, so the cache nodes are part of the -entry's identity and travel inside `xorq_build/` — which is why every loader -of the build, in any process, reads through the same snapshot key. +## Worthiness: cheap and worthy entries + +Whether an entry has a file of its own is decided once, when it is built, by +one test on the expression the author wrote (`worthiness.classify_expr`), and +recorded in the manifest as `cache_worthy`. Nothing derives it again. + +An entry is **cheap** only if all of these hold: every relation operation is a +file read, filter, column selection, computed column, rename, cast, column +drop, drop of null rows or fill of nulls; the plan reads exactly one file; and +no value operation multiplies rows (`unnest`), depends on the order rows arrive +in (a window function, which also covers `row_number` and `lag`) or is not pure +(`random()`, `uuid()`, `now()`, `today()`, or any UDF). Anything else is +**worthy**: an aggregate, join, sort, limit, union, distinct, a second file, or +an operation nobody has classified. The list is an allow-list, so an unknown +operation costs a copy and never unstable paging. + +- A **worthy** entry is **materialized**: its result is written once, when the + entry is created, to a **snapshot**, + `compute_cache/result_cache/.parquet`, and every read after that + is a read of that file. +- A **cheap** entry writes nothing. It is a stored plan over files that exist + (a view, in the database sense), which re-runs on every read. Tallyman ran it + in full when the entry was created, so an error in it surfaced there. + +The canonical sort, described next, is added only to a worthy entry. + +## Row order + +Every file tallyman writes ends in an `int64` column named `__row_order` holding +`0..N-1` in the file's physical row order. It is the last column, and it is +visible in every table. Two writers produce it: the import (a source entry's +snapshot, numbered in the imported file's order) and `materialize` (every other +snapshot), and each overwrites an inherited one with positions in its own +file. It is what makes a page of an +entry a function of `(content_hash, sort, offset, limit)`: with no user sort a +page is `ORDER BY __row_order`, and with one the user's keys come first and +`__row_order` is the last key, which breaks every tie. + +- A cheap entry has no file of its own, so it inherits the column and must keep + it. A cheap recipe whose output lacks it fails to build, and the error names + what it reads and shows the corrected select. A worthy entry may drop it, + because the writer numbers its rows. +- A recipe may read the column and copy it under another name + (`__row_order_v1`), and may not assign to it. To change the order of rows, sort + them: the writer numbers the result in that order. +- Every `order_by` in a recipe gets `__row_order`, then the remaining sortable + columns, as its last keys, so the sort is total wherever the recipe put it. A + sort followed only by steps that keep row order (filters, limits, selections, + column drops, drops or fills of nulls) is kept: the top-level sort leads with + its keys, and the build fails, naming the key, if a later step dropped or + changed it or if the sort was by an expression rather than a column. Above an + aggregate, a join or a union the order of rows is gone, and the top-level sort + is the tie-break alone. +- A join of two entries leaves the right side's copy under ibis's collision name, + `__row_order_right`, and the writer drops it. Joining three entries in one + recipe needs `.drop("__row_order")` on the right-hand inputs, and the build + says so. (Today the writer drops any column of that name, including an + author's, #206, and the check also refuses semi and anti join chains, which + cannot collide, #199.) +- A diff carries no row-order column from either side. The compare grid and a + promoted diff drop it; `full_diff`, behind the diff page's summaries and + `catalog_diff`, does not yet (#200). + +## Materialization + +`materialize(project, hash)` is the one routine that writes a computed entry's +snapshot. The build calls it and so does every **heal** of one (the re-creation +of a snapshot that is missing from disk), so result bytes are manufactured in +one place. A source entry's snapshot is not a result: it is written by the +import, and made again from the clone by the same code the import used +(`source_import.rewrite_source_snapshot`), through the same pinned writer. It +runs the entry's frozen build on a **single-partition** connection (so a float +total is merged in one order and is bit-stable on any machine), streams the rows +through a writer with a pinned layout (zstd, row groups of 1,048,576 rows, a +parquet page index, `__row_order` last), writes to a unique temp name and +replaces the final file atomically, all under the project's write lock, and +returns the content digest of the file it wrote, read back. A heal replaces the +file at once. A create leaves the finished file at its temp name +(`materialize(..., publish=False)`), and the build moves it into place +(`publish_snapshot`) after the manifest is written, so a build that fails +removes only its temp file and never the file already at the path. A create runs +the query twice and compares the digests; if they differ the recipe is not +reproducible, the entry still builds, and its file is **pinned**: the Cache +page's delete leaves it alone. A snapshot changes only by an atomic replace of a +complete file. + +**Pins are read from the entry's manifest and, for a source entry, from whether +its clone is on disk** (`pinned_reason`): a snapshot is pinned when the manifest +says `reproducible: false`, when it holds `unfaithful_heal_digest`, or when it +is a source entry whose clone is gone. So a pin moves with its entry through a +reset, and nothing outside the entry, such as the error log, can lift it. A +snapshot whose entry a reset retired is judged by the manifest parked in the +bullpen (the directory a reset moves retired entries into), and a retired source +version's clone counts as present when a reset parked it there too. + +`ensure_materialized(project, hash)` is the one entry point that makes files +exist, and every consumer that composes or executes an entry goes through it +(the canonical read below, chaining, the Buckaroo hand-off): + +1. A worthy entry whose snapshot exists is done, and no build is loaded. +2. Otherwise load the build and collect every file its `Read` nodes point at. +3. Re-create each that is missing. Every one is another entry's snapshot, made + again by recursing on the hash in its file name. +4. If the entry is worthy, materialize it and verify the result against + `result_digest`. A source entry skips steps 2 and 3: its snapshot is written + again from its clone with the reader options in its manifest, and verified + the same way. + +Whether the entry is worthy is read from the manifest, never derived again. +With the manifest missing the read raises; a file at the snapshot path says +nothing about the verdict. + +A file is cache only if this function can re-create it from files that are not +cache. Snapshots satisfy that, a source entry's included, and live under +`compute_cache/`, which anything may delete. Clones are data: a clone is the +only copy tallyman has of bytes it imported, so nothing deletes one (a reset +moves a clone no surviving source entry names into the bullpen). A source entry +whose clone is gone is the one case where a snapshot is the last copy of its +rows, so that snapshot is pinned; if it is deleted anyway, the read fails with +an error naming the missing clone and the import call, reader options included, +that repairs the version. + +Files are deleted only by an explicit user action, and a file is written only +because something is about to read it. The startup warm-up, the verify sweep and +a reset write and delete nothing under `compute_cache/`. ## Result digest -For worthy entries, the build records `result_digest`: the SHA-256 of the -baked snapshot file, taken over a canonically-ordered materialization so an -engine's parallel-scan row reshuffling cannot move it — the worthiness -rewrite sorts by the author's own `order_by` keys, then `original_row_order`, -then the remaining sortable columns (ADR D5). It is -the **output** identity axis, and it has exactly one job: witnessing that a -later rematerialization reproduced the original bytes. It is never a -staleness input (an entry whose recompute differs is *nondeterministic*, not -stale — recomputing cannot make it fresh) and never part of the entry's name. +For worthy entries, the build (for a source entry, the import) records +`result_digest`: `arrow-sha256:`, a +SHA-256 over the snapshot's ordered Arrow data, computed from the file read +back. It is independent of the row-group size, the codec, the writer's version, +whether a text column is `string` or `large_string`, and what a null slot holds; +it depends on every value, on which slots are null, on the order of the rows +(fixed by the canonical sort, or for a source entry by the file's order) and on +the column names and types. It is the +**output** identity axis, and it has exactly one job: witnessing that a later +rematerialization reproduced the original result. It is never a staleness input +(an entry whose recompute differs is *nondeterministic*, not stale — recomputing +cannot make it fresh) and never part of the entry's name. It exists because recipes are LLM-authored: an LLM can write `sample()` or an impure UDF, the hash cannot see it (Part 1 §5), and the digest is the runtime -backstop that catches it. +backstop that catches it. Running the query twice at create moves the catch to +the moment the entry is born. --- @@ -361,95 +545,153 @@ backstop that catches it. ## The write path (build) -`build_and_persist(project, code)`: +`build_and_persist(project, code)` holds the project's write lock for the whole +build (one build at a time per project, so two builds of one entry cannot end +with the failing one deleting the winner's directory). The lock is a file lock, +so it holds between the MCP server and the companion, which both build. One +companion serves a data dir (the directory that holds every project, +`TALLYMAN_HOME`): `tallyman run` holds an exclusive lock on +`/server.lock` while it serves, a second `tallyman run` on the same +data dir is refused, and a client of the data dir reaches only the companion +that lock names. The build: 1. **Import the recipe** — the single moment of name resolution. During the - import, `read_project_file` digests and clones each source (CAS) and - records `{rel_path: digest}`; `tracked_expr_from_alias` resolves each - alias to its current head, records the parent edge, and returns the - parent's expression (loaded from the parent's frozen build — see read - path). -2. **Rewrite** — the worthiness rewrite (cache injection). + import, `tracked_expr_from_alias` resolves each alias to its current head, + records the parent edge, and returns the parent's result (below), and + `pinned_expr_from_alias` does the same for a version reference. A raw file + read, a bare hash and a bare alias raise here. +2. **Check and rewrite** — reject what cannot become a sound entry (an in-memory + read, a `.cache()` call, a raw parquet or CSV read, an assignment to + `__row_order`, a cheap entry that drops it, a join chain over three entries + that all carry it), classify the entry once (cheap or worthy), add the + canonical sort to a worthy entry, and move `__row_order` to the last column + of a cheap one. 3. **Freeze** — `build_expr` serializes the rewritten expression; `content_hash` = the build's name. If an entry with this hash already exists, stop: append the prompt, return the existing entry (idempotency). 4. **Lay down the entry** — copy the build in, make paths portable (`${TALLYMAN_PROJECT_ROOT}` placeholders), write `expr.py`. -5. **Execute once** — load the just-written build against the project's - compute cache and run it. Worthy: the top cache node materializes the - baked snapshot (canonically ordered), and the build records its digest - **and its snapshot key** in the manifest. Cheap: one full streaming pass - (honest evaluation, fails fast, keeps nothing). -6. **Record** — schema, manifest (written last, atomic). +5. **Execute** — a worthy entry is materialized (Part 2, "Materialization"), + which runs its query twice, writes the snapshot and yields its digest, and + the build records the digest, the reproducibility verdict and the schema + read from the written file. The snapshot stays at its temp name. A cheap + entry is streamed once in full and keeps nothing (honest evaluation, fails + fast). +6. **Record** — schema, manifest (atomic, the entry directory's last write), and + then the snapshot, moved into place from its temp name. A build that fails + before that leaves any file already at the snapshot's path as it was. 7. **Checkpoint** — when the MCP tool returns: recipe zip, tracked pointers, one git commit. The build's obligation in one line: **record everything a reader will ever need, because the reader is forbidden from resolving anything.** +## The import path + +`update_and_depend(outside_path, alias, pinned_version=None, schema=None, +**reader_options)`, behind the `catalog_import_source` tool, is the only way a +source alias advances. It fixes the reader from the file's suffix and the +options, digests the file, computes the entry hash, and then, under the project +lock, decides the case (Part 2, "Project, imports and source entries"): + +- **Mint:** clone the bytes and verify the clone, write the snapshot, write the + entry (generated `expr.py`, frozen build, schema, manifest with `provenance`, + written last), and append it to the alias. A failure removes the entry + directory the import created, and nothing else. +- **The version already exists** (a no-op, or a repair): the entry's recipe, + build and manifest are the record of the import that minted it, and none of + them is rewritten. If its snapshot is gone, the clone is restored from the + given file when it is gone too (verified against the digest), and the + snapshot is healed exactly as `ensure_materialized` heals it: from the clone, + verified against `result_digest`. If a reader now parses the bytes + differently, the healed file is served but recorded as an unfaithful heal + (which pins it), and the manifest keeps the digest of the rows the version + was imported with: a repair never re-records a version's rows. A directory a + crash left without a manifest is not an entry, and is written again. +- **Refuse:** a directory, a name that is a catalog alias, a file that is not + parquet or CSV, a CSV option that does not survive JSON, and every error row + of the case table, each with a message saying what to do instead. + +A minted version is a catalog operation like a revise: the tool records the +same events, notifies the companion, runs auto-recalc for the alias's followers +when the project enables it, and lands the import and its cascade as one +checkpoint. + ## The read path One canonical read, used by every consumer: ``` -read(project, content_hash, cache_dir=): - expand the entry's xorq_build (placeholders → real paths, stable per-entry dir) - expr = load_expr(expanded, cache_dir=cache_dir) # + tallyman's deep cache-dir rewrite - return expr +read(project, content_hash): + ensure_materialized(project, content_hash) # every file the plan reads exists + worthy entry: one bare read of its snapshot + cheap entry: load_expr(expanded build), rebound onto the default backend ``` -- **Worthy entry:** the loaded graph carries its cache node, so execution - reads the baked snapshot by key. If the file is missing (evicted), the - cache node re-executes the *frozen* subgraph — whose leaves are CAS clones - and inlined parent graphs, all pinned — writes the snapshot, and the result - is verified against `result_digest` (below). Single-flighted per entry so - concurrent cold readers don't race. -- **Cheap entry:** execution re-runs the frozen graph in pushdown time, - reading CAS clones. Same bytes every time, by construction. -- The `cache_dir` parameter is the **cold seam**: any entry can be - materialized against an empty temp dir, and must produce the same bytes as - the warm read. That property is the standing regression test for every - read-path change. +- **Worthy entry:** a bare read of `compute_cache/result_cache/.parquet`, + served without loading the entry's build when the file exists. If the file is + missing, `ensure_materialized` re-runs the *frozen* build (whose reads are + parents' snapshots, all made to exist first), writes the snapshot, and + verifies it against `result_digest` before it is served. For a source entry + it parses the clone again instead. +- **Cheap entry:** execution re-runs the frozen graph, reading files that exist. + Same rows every time, by construction. +- **The cold state is an empty `compute_cache/`:** any entry can be read after + it is deleted, and must produce the same result as the warm read. That property + is the standing regression test for every read-path change. The known + exceptions are an entry recorded as not reproducible, whose snapshot is pinned + because it cannot be made again faithfully, and the entries built on it + (#185, #208), and a source entry whose clone is gone, whose snapshot is the + last copy of its rows. `cached_result_expr(project, hash)` is the function every in-process consumer -calls, and it is an optimization over the canonical read, nothing more: it -performs the read once per `(project, content_hash)` and keeps the loaded -expression in a bounded in-process LRU, so repeated reads (every pagination -request, both sides of a diff) skip the load. Removing it must change latency -and nothing else. That is what makes its key honest: the cached value is a -pure function of the frozen build the key names. The per-call -snapshot-existence check and the single-flighted heal stay outside the LRU, -because file existence is the one input that remains mutable (eviction). - -**Chaining** (`tracked_expr_from_alias` at build time) uses the same read: the -parent's frozen build is loaded and its graph — including, for a worthy -parent, its cache node — is composed into the child. The child's build -therefore carries its own regeneration knowledge. If the parent's snapshot is -missing when the child executes, the parent's cache node re-runs its frozen -subgraph and rewrites the file, the same as any cache miss; nothing outside -the execution has to arrange for the file to exist.[^preheal] +calls. It is `ensure_materialized` plus the read above, and it memoizes the loaded +plan (and the bare snapshot read, so a snapshot has one table name in the shared +backend) per `(project, content_hash)`. Removing the memo must change latency and +nothing else. The per-call existence check stays outside the memo, because file +existence is the one input that remains mutable. + +**One execution at a time per process.** A process's default backend is one +DataFusion session, which fails with `Already borrowed` when two threads execute +on it at once. So every execution on it (a page, a post-processing run, a +primary-key probe, a diff's summaries) holds `execution.execution_lock`, one +re-entrant lock per process. The read above comes first and the execution after, +because a heal takes the project lock and the order is the project lock first, +then the execution lock: `project_lock` raises in a thread that holds the +execution lock and would take a new file lock. An execution on a connection of +its own (a materialization's stream, a cheap entry's row count at build) needs +no execution lock. + +**Chaining** (`tracked_expr_from_alias` at build time) uses the same read. A +worthy parent, a source entry included, is a bare read of its snapshot, so the +child's build holds the literal path of that file, which contains the parent's +content hash: the child's identity is a function of its parent's. A filter over +an aggregate's snapshot is therefore a cheap entry. A cheap parent's graph is +inlined. The parent's snapshot is made to exist before the child is composed, +since a child cannot be built over a file that is missing. The recipe-reconstruction machinery survives only as a diagnostic. An entry whose build is missing or unloadable is a **hard error** naming the entry and the remedy (rebuild) — there is no automatic recipe fallback, because a warning on a background read is exactly how #163-class behavior stays -invisible (decided in `plans/ADR-006-read-path-loads-builds.md`, D6).[^recon] +invisible (ADR-006 D6, a missing or unloadable build is a hard error, in +`plans/ADR-006-read-path-loads-builds.md`).[^recon] ## Composition: diff and beyond To compose two entries (diff's outer join, a union, any multi-entry expression): -1. read each entry (canonical read above) — each load minted fresh backend - objects; -2. **rebind onto shared backends, one per distinct content profile** (profile - identity = profile minus `idx`), using `replace_sources`. For today's - catalogs every profile is the same embedded engine, so this collapses to - one shared backend; -3. compose. The result is a single-backend expression: it executes - in-process, and `build_expr` serializes it into a normal single-profile - build that Buckaroo's `/load_expr` accepts — the same handoff as an - ordinary entry grid. +1. read each entry (canonical read above) — each read makes its files exist, + and each load minted fresh backend objects; +2. **rebind onto the process's default backend**, using `replace_sources`. + Every profile in a tallyman build is the same embedded engine (profile + identity = profile minus `idx`), and a build that spans more than one + distinct content profile fails loudly instead of being rebound; +3. compose, dropping `__row_order` from both sides first (`full_diff` does not + yet, #200). The result is a single-backend expression: it executes + in-process, and `build_expr` serializes it into a normal single-profile build + that Buckaroo's `/load_expr` accepts. Composition of frozen builds was never the problem; backend object identity was (#75, rediagnosed in #163). The rebind is cheap graph surgery, no data @@ -457,29 +699,89 @@ moves. A **promoted diff** is just an entry whose recipe pins two hashes (`build_diff_expr(a_hash, b_hash)`) — name-free, deterministic, and built -through the ordinary write path. - -## Staleness and recalc (unchanged, stated for completeness) - -**Staleness** is a read-only judgment: an entry is stale on the alias axis -when a `follow=True` parent's recorded hash no longer equals that alias's -head, and on the source axis when a recorded digest no longer matches the -live file's digest. Computing staleness executes nothing and mutates nothing. +through the ordinary write path. It contains a join, so it is worthy and is +materialized like any other. The live diff grid, which is not an entry, still +hands Buckaroo an unmaterialized join: ADR-007 D10, which would have built +every diff as an entry before showing it, was moved out of +`plans/ADR-007-tallyman-owned-materialization.md` to #188. + +## Handing an entry to Buckaroo + +Buckaroo displays: it runs queries only for summary stats, sorting and paging. +Tallyman finishes the entry's computation first (`ensure_materialized`), so no +other process runs an entry's expensive computation, writes result files or +repairs tallyman's cache, and a failure of the computation surfaces in tallyman's +process and never inside a grid query. + +- A **worthy** entry's grid is handed a **view build**: a build whose whole graph + is one bare read of the entry's snapshot, written once to a stable per-entry + directory, so Buckaroo is handed the same build after a restart. +- A **cheap** entry's grid is handed its own expanded build, a stored plan over + files that exist. +- Tallyman keeps no record of Buckaroo's sessions. A **session** (one grid's + state in the Buckaroo process) has the id `entry--`, + posted on every open: Buckaroo skips the work while it holds that session with + the same build directory and the post carries no configuration, and creates + the session again if it dropped it (it does after an hour without a browser). + A klass (a project-authored stat, post-processing or display class) reload + posts `/reload_expr/` for each entry of the project and treats the 404 for + an id Buckaroo does not hold as "not open". +- Every `/load_expr` names `__row_order` as the row-order column, so Buckaroo can + order its pages by it. Buckaroo 0.15.6, the pinned version, ignores the hint; + buckaroo-data/buckaroo#974 is Buckaroo's half of that. +- After an unfaithful heal, the entry's stat cache is wiped and Buckaroo is told + to reload the grid (`force_reload`). + +## Reset + +`reset_to` returns the catalog to an earlier step. It restores every tracked +file with `git reset --hard`, and reconciles the untracked entry directories +through the **bullpen**, the directory a reset moves retired files into so a +reset forward can bring them back. It leaves `compute_cache/` alone: snapshots +are named by content hash, and a file that is missing afterwards is made again +and verified like any other. It moves the clones no surviving source entry names +in its `provenance` into the bullpen and never deletes them, and a reset forward +copies back the clones a restored source entry names. Source aliases rewind with +every other alias, since `aliases.jsonl` is a tracked file. An entry directory +retired when the bullpen already holds one under its name replaces the parked +copy, because the live one agrees with the snapshot on disk; one with no +manifest, left by an interrupted build, is dropped instead. So a reset forward +brings back the manifest that matches the file. The bullpen has one live reader +besides `reset_to`, the Cache page, which reads a retired entry's parked +manifest (and a retired source version's parked clone) to keep its snapshot's +pin. + +## Staleness and recalc + +**Staleness** is a read-only judgment with one axis: an entry is stale when a +`follow=True` parent's recorded hash no longer equals that alias's head, and +for no other reason. A pinned parent never makes its child stale. A data file +that changed outside tallyman is not a reason: until it is imported again +nothing has changed in the catalog, and the import moves a source alias, which +is the one axis. Computing staleness executes nothing, opens no data file and +changes no catalog state. Only an entry that is the current head of an alias is +actionably stale; a superseded version is reported with `live=False` (#154). A +parent alias that no longer exists is reported under `unknown_axes`. **Recalc** is the one sanctioned re-execution of recipes. When an alias head -advances (a revise), the entries that may be affected form its **cone**: -every entry reachable by walking `follow=True` parent edges backwards from -that alias — its followers, their followers, and so on. Pinned -(`follow=False`) edges are not in the cone. It runs automatically after a -revise when the project enables auto-recalc, or on demand. +advances (a revise, or an import that mints a new version of a source alias), +the entries that followed it by name are directly stale; they are the roots, and +the entries that may be affected form their **cone**: the roots and every +current alias head reachable from them through recorded parent edges, followers +of followers and so on. It runs automatically, when the project enables +auto-recalc (the default), after a revise, after an import that mints a version, +and after a promoted diff that re-points an existing alias; otherwise it runs on +demand. Recalc rebuilds the cone in topological order, parents before children. For each member it re-imports the member's *recipe* — the one situation where name resolution is the point, since the goal is a new version against the new heads — builds the result as an ordinary new entry, and advances the member's alias before any of its children replay, so each child chains off -its parent's fresh head. Old entries are untouched; every member gains a -version, none loses one. +its parent's fresh head. A member whose inputs did not move, such as a child +that pins a version of its parent (`follow=False`), replays to the same hash and +is left alone. Old entries are untouched; a member that rebuilds gains a version, +and none loses one. Two disciplines keep it predictable. **Scope:** recalc touches only followers of the alias that moved; pre-existing staleness elsewhere is reported, not @@ -490,50 +792,51 @@ open entries to their new versions. ## Verification -`verify_result_faithful(project, hash)` means exactly: *executing this -entry's frozen build reproduces its recorded `result_digest`.* It runs in -production, not only in tests: - -- on every self-heal (a healed snapshot is checked before it is served); -- on demand, corpus-wide, via `catalog_scan_staleness(verify_results=True)`. - -A failure is surfaced loudly — a durable `unfaithful_heal` record in -`errors.jsonl` (the UI badge and ADR D12's non-evictable marking), a stat -cache wipe, session eviction, an SSE event — never only a log line. Its -attribution has three classes with three different fixes: +`verify_result_faithful(project, hash)` means exactly: *the snapshot on disk +still has the entry's recorded `result_digest`.* Verification runs in production, +not only in tests: + +- on every heal (a snapshot `ensure_materialized` writes is checked before it is + served); +- on demand, corpus-wide, via `catalog_scan_staleness(verify_results=True)`, whose + verify sweep reads and never writes: a snapshot that is missing is reported as + `absent` and checked at the moment it next exists. + +A failure is surfaced loudly, never only as a log line: the pin, the digest the +heal wrote recorded as `unfaithful_heal_digest` in the manifest (the Cache +page's delete leaves the file alone), a durable `unfaithful_heal` record in +`errors.jsonl` (shown in the catalog page's error banner), a stat cache wipe, +and in the companion a forced reload of the entry's Buckaroo grid and an +`unfaithful_heal` SSE event. Today the SPA has no listener for that event, the +forced reload is sent even when no grid is open (#203), and cheap entries that +read the healed snapshot are not flagged (#208). Its attribution has four +classes with four different fixes: | class | detector | meaning | response | |---|---|---|---| -| structural (#88) | recipe re-derives a different hash, sources unchanged | author-time value baked into the graph (`pd.Timestamp.now()`) | lint; rewrite recipe | -| execution (#83) | fixed graph, digest moves across runs | `sample()`, `now()`, impure UDF | lint; pin entry to its snapshot, mark non-evictable | +| engine | the xorq, xorq-datafusion or pyarrow version, or the snapshot format, differs from the one recorded at build | a library upgrade changed the result | rebuild the entry; the recipe is not implicated | +| structural (#88) | recipe re-derives a different hash, inputs unchanged | author-time value baked into the graph (`pd.Timestamp.now()`) | lint; rewrite recipe | +| execution (#83) | fixed graph, digest moves across runs | `sample()`, `now()`, impure UDF | lint; the file is pinned | | lineage (#163) | a read resolved a name post-build | machinery bug | impossible by construction under this contract | -The third row is the point of the whole design: with reads going through the -build, lineage drift has no mechanism left. - -[^preheal]: This replaced a two-layer workaround. Before the #163 fix, - chaining stripped the parent's cache node down to a bare file read of its - snapshot (the #75 single-backend fix), so the child's build pointed at a - parquet without knowing the file was regenerable — evict the snapshot and - the child read nothing. To compensate, the companion called - `cached_result_expr` on an entry just before handing its build to - Buckaroo and discarded the result: the call existed only for its side - effect, forcing any missing ancestor snapshots onto disk in time for - Buckaroo's replay. The stripping and the pre-heal call were both deleted - with the fix. +Creating a materialized entry runs its query twice, so an execution-class entry +is known from birth, and a mismatch at a later heal means something changed +underneath a reproducible entry. The lineage row is the point of the whole +design: with reads going through the build, lineage drift has no mechanism left. [^recon]: The subsystem the fix demoted: `_recipe_expr` re-imports an entry's `expr.py` to recover its expression, patching over the recipe's name-binding with context variables for the duration of the import. - `_RECON_SOURCES` carries the entry's recorded `{path: digest}` map so - that `read_project_file`, when called inside such a re-import, resolves - each source to its frozen `.cas/` clone instead of re-digesting - the live file (#115), and `_resolve_noncyclic_hash` walks a - self-referencing alias back to its previous version to break re-import - cycles (#74). Each hook re-derives a binding the frozen build already - contains, which is why the fix retired their read-time role (and #162's - proposed `_RECON_PARENTS` sibling was never built) rather than adding a - fourth. The diagnostic use stays, and is the only reason the machinery + `_resolve_noncyclic_hash` walks a self-referencing alias back to its + previous version to break re-import cycles (#74). A second hook, + `_RECON_SOURCES`, once resolved each raw file read to its frozen + `.cas/` clone (#115); ADR-011 deleted it with `manifest.sources`, + since a recipe no longer reads files at all, and a source entry's own + generated recipe resolves its one read to its snapshot (`_SOURCE_ENTRY`). + Each hook re-derives a binding the frozen build already contains, which is + why the fix retired their read-time role (and #162's proposed + `_RECON_PARENTS` sibling was never built) rather than adding another. The + diagnostic use stays, and is the only reason the machinery still exists: `recipe_is_structurally_nondeterministic` re-derives the hash from the recipe on purpose, because re-running the recipe is exactly how you detect that a recipe fails to reproduce its own graph. @@ -542,7 +845,7 @@ build, lineage drift has no mechanism left. # Part 4 — The invariants -Everything above compresses to five statements. Any change that breaks one is +Everything above compresses to six statements. Any change that breaks one is wrong even if every test passes. - **I1 — A content hash names a fixed result.** Materializing an entry yields @@ -553,10 +856,13 @@ wrong even if every test passes. - **I2 — Caches affect latency, never results.** Every answer must be byte-identical with all caches empty. Corollaries: cache keys derive only from immutable inputs; cached values are reproducible from the frozen build - alone; every read path has a cold seam (`cache_dir`); a self-heal is - reproduce-and-verify, never manufacture. Result bytes are manufactured - exactly once, at build — a read path that writes bytes the build didn't - define has become a second, unaudited build path. + alone; every read path has a cold seam (an empty `compute_cache/`); a + self-heal is reproduce-and-verify, never manufacture. Result bytes are + manufactured in two places: `materialize`, which the build and every heal of + a computed entry call, and the import's writer (`source_import._write_snapshot`), + which writes a source entry's snapshot at import and at a heal from its clone. + A read path that writes bytes any other way has become a second, unaudited + build path. - **I3 — One read semantics.** Every consumer that materializes an entry reads the frozen build through the one canonical read. The recipe is never re-executed on behalf of an existing entry. @@ -564,20 +870,92 @@ wrong even if every test passes. the manifest; after that the entry is closed. Aliases at read time select *which* entry to serve, never *what* an entry means. (Grep-able form: `get_alias`/`previous_version`/`version_of_hash` appear only in minting, - selecting, and judging code — never in materializing code.) + selecting, and judging code — never in materializing code. The one alias + lookup near materializing code names a source version in the text of an + error or a pin reason, `current_source_version`, and decides no rows.) - **I5 — One question, one path.** Any question answerable two ways (grid vs API bytes, manifest row count vs live count) either shares one canonical path or carries an explicit check tying the two together; disagreement is - surfaced, never averaged over. + surfaced, never averaged over. A worthy entry's grid and its `/api/data` pages + read the same file. +- **I6 — A page is a function of its request.** The same `(content_hash, sort, + offset, limit)` returns the same rows in any process and any cache state, because + every page is ordered by `__row_order`, which has no ties. + +--- + +# Known deviations + +Where the code breaks a rule above today. Each is an open issue; none is a +change of the rule. + +- **Verification reaches one entry.** An unfaithful heal of a worthy parent + changes its cheap children's rows under their hashes without a record + (#208), and the purity of an entry is not passed on to entries built on it, + so a child of a non-reproducible parent is recorded as reproducible (#185). + Both are gaps in I1 and I2. +- **Row order.** The canonical sort's tie-break leaves out nested columns + (#205); the snapshot writer drops any column named `__row_order_right` (#206); + the three-way join check refuses semi and anti joins (#199); `full_diff` keeps + `__row_order` as data (#200); and Buckaroo's grid does not yet order pages by + `__row_order` (buckaroo-data/buckaroo#974), so I6 holds for `/api/data` and + not yet for the grid. +- **Handing an entry to Buckaroo.** Concurrent opens both post `/load_expr`, and + a promoted diff re-runs Buckaroo's statistics on every open (#202); the + forced reload after an unfaithful heal runs under the project lock and opens a + session nobody asked for (#203); a klass reload posts once per entry from the + companion's event loop (#201); Buckaroo is pointed at `artifacts/` and does + not find the project's stats and post-processing functions (#170); the live + diff grid is an unmaterialized join (#188). +- **One writer at a time.** The project lock blocks with no timeout (#186), and + two companion routes build on the event loop and freeze the UI while they wait + (#190). The lock covers builds, materializations, checkpoints and resets + only: alias, notebook, chart, display-config and `config.json` writes replace + their file atomically without it, so an MCP edit and a browser edit of the + same file at the same moment can lose one of the two (#240). +- **Names resolve once, in the right project.** A recipe's alias readers + (`tracked_expr_from_alias`, `pinned_expr_from_alias`) resolve the project from + the `active_project` file, while the MCP tool builds into the session's own + project; after another session switches projects the two differ, and the + recipe looks its aliases up in the other project (#233; related to #39). +- **Portability.** An expanded build does not record the project path it was + filled in with, so a copied project reads the old location (#209). +- **Float totals.** An ungrouped float `SUM` depends on the row-group layout of + the file it reads; the snapshot format version pins that layout, and a change + of it is a corpus rebuild (#187). --- # History This contract began as the proposed design for the #163 fix; the pre-fix -deviations and the decisions that settled the design (D1–D12) are in +deviations and the decisions that settled the design (ADR-006 D1–D12) are in [`plans/ADR-006-read-path-loads-builds.md`](../plans/ADR-006-read-path-loads-builds.md). -The wider audit of the same bug class is +The redesign that replaced xorq's cache nodes with tallyman's own materialization, +added `__row_order` and redefined the digest is in +[`plans/ADR-007-tallyman-owned-materialization.md`](../plans/ADR-007-tallyman-owned-materialization.md), +[`plans/ADR-008-row-order-of-reads.md`](../plans/ADR-008-row-order-of-reads.md) and +[`plans/ADR-009-digest-stability.md`](../plans/ADR-009-digest-stability.md), +accepted on 2026-09-22 and implemented in #189. +[`plans/ADR-010-immutable-store-one-owner.md`](../plans/ADR-010-immutable-store-one-owner.md) +proposed replacing them and was rejected. +[`plans/ADR-011-sources-are-aliases.md`](../plans/ADR-011-sources-are-aliases.md) +made a raw input a source alias whose versions are entries, which removed the +source axis of staleness, the identity modes, `manifest.sources` and the ordered +copy, and with them the defects listed here before it (#191, #197, #198, #207, +#211, and the two staleness defects that had no issue). Four deviations this +section listed were fixed on the same branch: a failed build deleted the +snapshot already at its path (#193, fixed in #222), and a reset could pair a +non-reproducible entry's older manifest with its newer snapshot, a retired +entry's snapshot lost its pin, and dismissing the error banner lifted an +unfaithful heal's pin (#194, #195 and #196, fixed in #223). Three more were +fixed later on the same branch: with its manifest missing, an entry's worthiness +was guessed from whether its snapshot existed, so a worthy entry that had lost +both was read as cheap (#204, fixed in #245, which makes every read refuse such +a directory); two servers on one project went undetected (#183, fixed in #241, +one server per data dir); and concurrent executions on a process's shared +backend could fail with `Already borrowed` (#118, fixed in #242, the execution +lock). The wider audit of the same bug class is [`plans/cache-soundness-audit.md`](../plans/cache-soundness-audit.md) (#168–#172, buckaroo#955–#957) — the contract's rules apply to those axes too. diff --git a/packages/app/package.json b/packages/app/package.json index a4bab08f..91402ac7 100644 --- a/packages/app/package.json +++ b/packages/app/package.json @@ -2,6 +2,7 @@ "name": "@tallyman/app", "version": "0.0.1", "private": true, + "license": "AGPL-3.0-only", "type": "module", "scripts": { "dev": "vite", diff --git a/packages/app/src/api.ts b/packages/app/src/api.ts index 06b640f0..d49a9083 100644 --- a/packages/app/src/api.ts +++ b/packages/app/src/api.ts @@ -145,8 +145,12 @@ export const api = { get(`/${project}/api/result_cache`), deleteResultCache: (project: string, hash: string): Promise<{ ok: boolean; hash: string }> => - fetch(`/${project}/api/result_cache/${hash}`, { method: "DELETE" }).then((r) => { - if (!r.ok) throw new Error(`delete failed: HTTP ${r.status}`); + fetch(`/${project}/api/result_cache/${hash}`, { method: "DELETE" }).then(async (r) => { + if (!r.ok) { + // A pinned snapshot answers 409 with the reason in `detail`; show it instead of a bare status. + const body = await r.json().catch(() => null); + throw new Error(body?.detail ?? `delete failed: HTTP ${r.status}`); + } return r.json() as Promise<{ ok: boolean; hash: string }>; }), }; diff --git a/packages/app/src/pages/CachePage.tsx b/packages/app/src/pages/CachePage.tsx index 73864978..4c8a9efe 100644 --- a/packages/app/src/pages/CachePage.tsx +++ b/packages/app/src/pages/CachePage.tsx @@ -21,13 +21,13 @@ export function CachePage() { const handleDelete = async (hash: string) => { if (!project) return; - if (!confirm(`Delete the cached snapshot for ${hash.slice(0, 12)}?\n\nThe entry, code and alias stay — it re-bakes on next view.`)) return; + if (!confirm(`Delete the snapshot for ${hash.slice(0, 12)}?\n\nThe entry, code and alias stay — the snapshot is made again and verified the next time the entry is opened.`)) return; setDeleting((s) => new Set(s).add(hash)); try { await api.deleteResultCache(project, hash); setEntries((prev) => prev.filter((e) => e.hash !== hash)); - } catch { - alert("delete failed"); + } catch (err) { + alert(err instanceof Error ? err.message : "delete failed"); } finally { setDeleting((s) => { const n = new Set(s); n.delete(hash); return n; }); } @@ -45,7 +45,8 @@ export function CachePage() { {totalFormatted} on disk - Deleting frees the snapshot only — code, build and stat cache stay; the entry re-bakes on next view. + Deleting frees the snapshot only — code, build and stat cache stay; the entry is made again and verified on next view. + A pinned snapshot cannot be made again faithfully, so it is kept. @@ -71,7 +72,11 @@ export function CachePage() { {e.row_count.toLocaleString()} {e.created} - {e.alias ? ( + {e.retired ? ( + + (retired by a reset) + + ) : e.alias ? ( <> {e.alias} {e.version != null && ( @@ -81,14 +86,27 @@ export function CachePage() { current )} + ) : e.orphan ? ( + + (no entry) + ) : ( (scratch) )} + {e.pinned && ( + + pinned + + )} - - {e.hash} - + {e.orphan || e.retired ? ( + {e.hash} + ) : ( + + {e.hash} + + )} {e.prompt ?? —} @@ -96,7 +114,8 @@ export function CachePage() {