diff --git a/.github/workflows/fleet-converge.yml b/.github/workflows/fleet-converge.yml index 08952998fa7..e7321dd8a98 100644 --- a/.github/workflows/fleet-converge.yml +++ b/.github/workflows/fleet-converge.yml @@ -81,7 +81,7 @@ jobs: fetch-depth: 0 ref: ${{ github.event.inputs.expected_revision || github.ref }} - name: Isolate toolchain dirs - run: | + run: |- rm -rf "$RUNNER_TEMP/rustup" "$RUNNER_TEMP/cargo" echo "HOME=$RUNNER_TEMP" >> "$GITHUB_ENV" echo "CARGO_HOME=$RUNNER_TEMP/cargo" >> "$GITHUB_ENV" @@ -104,7 +104,7 @@ jobs: env: HOME: ${{ runner.temp }} - name: Derive native-cache root (rustc-version segment; after setup-rust-toolchain) - run: | + run: |- # 🟡 dissolve-on: ci_native_cache_root_script — orch-emitted foreign-executor prelude step deriving GUNBC_NATIVE_CACHE_ROOT from rustc -V after setup-rust-toolchain; leaf rustc/tr/mkdir/echo strings remain until typed toolchain probe lands on host_effect_apply (#5828 / shell→intent Phase 2) if ! GUNBC_TOOLCHAIN_SEG_RAW=$(rustc -V 2>/dev/null); then echo "::error::rustc -V failed after setup-rust-toolchain: cannot derive native-cache toolchain segment"; exit 1; fi GUNBC_TOOLCHAIN_SEG=$(echo "$GUNBC_TOOLCHAIN_SEG_RAW" | tr ' ' '-') @@ -112,7 +112,7 @@ jobs: mkdir -p "$RUNNER_TOOL_CACHE/gunbc-native/$GUNBC_TOOLCHAIN_SEG" echo "GUNBC_NATIVE_CACHE_ROOT=$RUNNER_TOOL_CACHE/gunbc-native/$GUNBC_TOOLCHAIN_SEG" >> "$GITHUB_ENV" - name: Pin rustup default (isolated RUSTUP_HOME has no default toolchain) - run: | + run: |- # dissolve-on: ci_pin_rustup_default_script -- orch-emitted foreign-executor step selecting a rustup default toolchain inside an isolated RUSTUP_HOME, which starts with none, and resolving the cargo binary that selection implies. The leaf rustup/command/echo strings remain until a typed TOOLCHAIN-SELECTION effect lands on host_effect_apply -- NOT the filesystem-and-environment effect ci_toolchain_home_isolation_script waits on, which is why this is a separate obligation: that effect landing alone would leave this carrier standing rustup default "$(rustup show active-toolchain | awk '{print $1; exit}')" if [ -x "$CARGO_HOME/bin/cargo" ]; then CARGO_BIN="$CARGO_HOME/bin/cargo"; else CARGO_BIN="$(command -v cargo || true)"; fi @@ -121,7 +121,7 @@ jobs: - name: Cache Cargo uses: actions/cache@caa296126883cff596d87d8935842f9db880ef25 with: - path: | + path: |- ${{ runner.temp }}/cargo/registry/index/ ${{ runner.temp }}/cargo/registry/cache/ ${{ runner.temp }}/cargo/git/db/ @@ -131,7 +131,7 @@ jobs: cargo-ci-${{ runner.os }}-${{ runner.arch }}- - name: Build release bins (claim_executor + gunbc) id: release_build - run: | + run: |- # 🟡 dissolve-on: ci_release_build_script — concat-built foreign-executor (GitHub Actions run:) release-build runner (ROOT stamp + verify-artifacts + sccache stats wrapping the orch-emitted EAGAIN retry core); membership of --bin / verify paths is derived from gunbc.ci_release_bins, but the runner transport itself remains hand-shell; DISSOLVES WHEN bash-emit (#5828 / ROADMAP 6-shell-slice0 / shell→intent Phase 2) realizes the release-build runner through orchestration emit or typed host_effect_apply without a medium-as-string concat scaffold ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) set -o pipefail @@ -150,7 +150,7 @@ jobs: timeout-minutes: 45 - name: "Reset-return dispatch admission: refuse an unnamed observer before anything is scheduled" id: reset_observer_dispatch_admission - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/host/host_reset_return_run.dag --function host_reset_return_dispatch_admission_wet env: @@ -206,7 +206,7 @@ jobs: timeout-minutes: 5 - name: "Pair-serving D0 admission (credential-free): the executor is srv1, the consent and the expected revision are named" id: pair_serving_d0_admit - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/spark/pair_serving_d0_door.dag --function pair_serving_d0_admit_executor env: @@ -225,7 +225,7 @@ jobs: create_credentials_file: false timeout-minutes: 5 - name: Materialize fleet key in-run (SM versions/1 pinned -> RUNNER_TEMP 0600 -> ssh-agent -> wipe file) - run: | + run: |- # 🟡 dissolve-on: gunbc_ci_fleet_key_agent_script - orch-emitted foreign-executor (GitHub Actions run:) credential runner: WIF access token by env, one PINNED secret version fetched over curl, a 0600 key file under RUNNER_TEMP with a trap armed BEFORE the credential touches disk, fingerprint verified against the modeled authority, ssh-agent load, file wipe. The pipeline steps are modeled (gunbc_ci_fleet_key_agent_prelude) but the runner transport itself remains hand-shell; DISSOLVES WHEN bash-emit (#5828 / ROADMAP 6-shell-slice0 / shell-to-intent Phase 2) realizes the credential runner through orchestration emit or typed host_effect_apply without a medium-as-string concat scaffold set -euo pipefail umask 077 @@ -255,7 +255,7 @@ jobs: timeout-minutes: 5 - name: Fleet converge plan (membership_reconcile → artifact) id: plan - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/fleet/fleet_converge_plan_cli.dag --function fleet_converge_plan_wet env: @@ -265,7 +265,7 @@ jobs: timeout-minutes: 15 - name: Fleet converge plan — scope:launch-environment (legacy fleet-converge timer retirement) id: launch_environment_plan - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/fleet/fleet_converge_plan_cli.dag --function fleet_converge_launch_environment_plan_wet env: @@ -275,7 +275,7 @@ jobs: timeout-minutes: 5 - name: Fleet allocation-store substrate plan id: allocation_store_plan - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/fleet/fleet_converge_plan_cli.dag --function fleet_converge_allocation_store_plan_wet if: github.event.inputs.mode == 'allocation_store_plan' @@ -303,7 +303,7 @@ jobs: timeout-minutes: 10 - name: Fleet converge apply (CAS + generated apply.sh) id: apply - run: | + run: |- set -euo pipefail ACTUAL="$(cat /tmp/fleet-converge-plan/plan_content.hex)" if [ -z "${EXPECTED_HASH:-}" ] || [ "$EXPECTED_HASH" != "$ACTUAL" ]; then echo "::error::PlanArtifactHashMismatch expected=$EXPECTED_HASH actual=$ACTUAL" >&2; exit 1; fi @@ -328,7 +328,7 @@ jobs: timeout-minutes: 10 - name: Org Actions credential validation + read-only settings diff id: org_actions_observe - run: | + run: |- set -euo pipefail umask 077 KEY_FILE="$RUNNER_TEMP/org-admin-app-key" @@ -379,7 +379,7 @@ jobs: timeout-minutes: 10 - name: "Org runner roster: paginated read of GitHub self-hosted runner registrations" id: org_runner_roster_observe - run: | + run: |- set -euo pipefail umask 077 KEY_FILE="$RUNNER_TEMP/org-admin-app-key" @@ -432,7 +432,7 @@ jobs: timeout-minutes: 10 - name: "GitHub App control plane: registration + webhook config observation" id: app_control_plane_observe - run: | + run: |- set -euo pipefail umask 077 APP_KEY_FILE="$RUNNER_TEMP/app-jwt-key" @@ -482,7 +482,7 @@ jobs: timeout-minutes: 10 - name: "App key version verify: mint an installation token with the exact version" id: app_key_version_verify_mint - run: | + run: |- set -euo pipefail : > "$APP_KEY_VERIFY_RECORD" case "$APP_KEY_VERSION" in ''|0*|*[!0-9]*) echo "AppKeyVersionSelectorRefused: '$APP_KEY_VERSION' is not a Secret Manager version number; the alias is refused rather than resolved. Reading NO key." >&2; exit 1;; esac @@ -528,7 +528,7 @@ jobs: timeout-minutes: 5 - name: "App key version verify: typed verdict and rotation-deadline standing" id: app_key_version_verify_verdict - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/auth/ci_app_key_rotation.dag --function ci_app_key_version_verify_wet cat "$ROOT/target/app-key-version-verify-receipt.txt" @@ -550,7 +550,7 @@ jobs: timeout-minutes: 10 - name: "Mt. Collins fan observation: IPMI SDR fan + temperature time series with power state (reads only)" id: mtcollins1_fan_observe - run: | + run: |- # 🟡 dissolve-on: gunbc_ci_mtcollins1_fan_observe_invoke - orch-emitted foreign-executor (GitHub Actions run:) credential runner: WIF access token by env, ONE pinned BMC secret version fetched over curl through the shared gunbc_ci_mtcollins1_bmc_credential_fetch_steps, a 0600 file under RUNNER_TEMP with a trap armed BEFORE the credential touches disk. The pipeline steps are modeled but each step body is a shell string built by concat, so the transport itself remains hand-shell; DISSOLVES WHEN bash-emit (#5828 / ROADMAP 6-shell-slice0 / shell-to-intent Phase 2) realizes the credential runner through orchestration emit or typed host_effect_apply without a medium-as-string concat scaffold. The SHARED fetch means this row and gunbc_ci_mtcollins1_boot_credential_shell_emit_dissolution_trigger retire together for the fetch half; each keeps its own row because each lane's file names, trap and exports are still its own hand-shell. set -euo pipefail umask 077 @@ -581,7 +581,7 @@ jobs: timeout-minutes: 10 - name: "Fabric writer identity: the login this host presents at the fabric DB's served door (reads only)" id: fabric_writer_identity_observe - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/fabric/fabric_writer_identity_observe.dag --function fabric_writer_identity_observe_wet cat "$ROOT/target/fabric-writer-identity-receipt.txt" @@ -602,7 +602,7 @@ jobs: timeout-minutes: 10 - name: "Pair-serving D0 (Cut D): file the consent, wait for the operator, suspend the group's authority for the keyed successor under the admitted grant" id: pair_serving_d0 - run: | + run: |- # 🟡 dissolve-on: gunbc_ci_pair_serving_d0_invoke - orch-emitted foreign-executor (GitHub Actions run:) credential runner: WIF access token by env, ONE pinned submission-MAC secret version fetched over curl through the shared auth header and SM decode, a 0600 file under RUNNER_TEMP with a trap armed BEFORE the key touches disk. The pipeline steps are modeled (gunbc_ci_pair_serving_d0_prelude) but the runner transport itself remains hand-shell; DISSOLVES WHEN bash-emit (#5828 / ROADMAP 6-shell-slice0 / shell-to-intent Phase 2) realizes the credential runner through orchestration emit or typed host_effect_apply without a medium-as-string concat scaffold -- the same trigger as gunbc_ci_approval_keyring_converge_shell_emit_dissolution_trigger, whose fetch this shares set -euo pipefail umask 077 @@ -623,7 +623,7 @@ jobs: timeout-minutes: 47 - name: "Mt. Collins census image: build the seeded live-server image in /srv/bmc and publish it create-only under its digest name" id: mtcollins1_census_image_publish - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/machine_intake/mtcollins1_census_image.dag --function mtcollins1_census_image_publish_wet cat "$ROOT/target/mtcollins1-census-image.txt" @@ -643,7 +643,7 @@ jobs: timeout-minutes: 10 - name: "R2 mint preflight: bootstrap read + account permission-group listing (reachability, no mutation)" id: r2_mint_preflight - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/cloudflare/r2_permission_group_observe.dag --function observe_account_permission_groups_wet cat "$ROOT/target/r2-mint-preflight-permission-groups.txt" @@ -663,7 +663,7 @@ jobs: timeout-minutes: 10 - name: R2 bucket-admin token mint (AccountTokens.Create + Secret Manager custody) id: r2_bucket_admin_mint - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/cloudflare/r2_token_mint_run.dag --function run_bucket_admin env: @@ -682,7 +682,7 @@ jobs: timeout-minutes: 10 - name: "R2 bucket ensure: entitlement + bucket existence per allocated purpose (create on absence, readback)" id: r2_bucket_ensure - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/cloudflare/r2_bucket_ensure.dag --function ensure cat "$ROOT/target/r2-bucket-ensure-receipt.txt" @@ -702,7 +702,7 @@ jobs: timeout-minutes: 10 - name: R2 object-write token mint (AccountTokens.Create + Secret Manager custody) id: r2_object_write_mint - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/cloudflare/r2_token_mint_run.dag --function run_object_write env: @@ -721,7 +721,7 @@ jobs: timeout-minutes: 10 - name: Materialize approval MAC keys on srv1 (SM pinned -> sha256 prefix -> /etc/gunbc-roadmap 0640) id: approval_keyring_converge - run: | + run: |- # 🟡 dissolve-on: gunbc_ci_approval_keyring_converge_invoke - orch-emitted foreign-executor (GitHub Actions run:) credential runner: WIF access token by env, two PINNED secret versions (capability and submission MAC keys) fetched over curl, 0600 files under RUNNER_TEMP with a trap armed BEFORE the credentials touch disk. The pipeline steps are modeled (gunbc_ci_approval_keyring_converge_prelude) but the runner transport itself remains hand-shell; DISSOLVES WHEN bash-emit (#5828 / ROADMAP 6-shell-slice0 / shell-to-intent Phase 2) realizes the credential runner through orchestration emit or typed host_effect_apply without a medium-as-string concat scaffold set -euo pipefail umask 077 @@ -746,7 +746,7 @@ jobs: timeout-minutes: 5 - name: Mt. Collins unit 1 approval-gated diskless boot (file request, poll, grant, attach, CD handoff) id: mtcollins1_boot - run: | + run: |- # 🟡 dissolve-on: gunbc_ci_mtcollins1_boot_invoke - orch-emitted foreign-executor (GitHub Actions run:) credential runner: WIF access token by env, PINNED BMC and submission-MAC secret versions fetched over curl, 0600 files under RUNNER_TEMP with a trap armed BEFORE the credentials touch disk. SOL hold is gunbc.machine_intake.sol_hold ActivateHeld (its own scaffold). The pipeline steps are modeled (gunbc_ci_mtcollins1_boot_credential_prelude) but the runner transport itself remains hand-shell; DISSOLVES WHEN bash-emit (#5828 / ROADMAP 6-shell-slice0 / shell-to-intent Phase 2) realizes the credential runner through orchestration emit or typed host_effect_apply without a medium-as-string concat scaffold set -euo pipefail umask 077 @@ -788,7 +788,7 @@ jobs: timeout-minutes: 10 - name: "Runner micro-VM guest image: observe base artifacts and their digests" id: guest_image_observe - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/runner/runner_guest_image.dag --function runner_guest_image_observe_wet cat "$ROOT/target/runner-guest-image-standing.txt" @@ -796,7 +796,7 @@ jobs: timeout-minutes: 30 - name: "Runner micro-VM guest image: build the image and digest what was built" id: guest_image_converge - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/runner/runner_guest_image.dag --function runner_guest_image_converge_wet cat "$ROOT/target/runner-guest-image-standing.txt" @@ -814,7 +814,7 @@ jobs: timeout-minutes: 10 - name: "Runner micro-VM host: install the cited Firecracker release and re-observe" id: microvm_host_converge - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/runner/runner_microvm_host_ready.dag --function runner_microvm_host_converge_wet cat "$ROOT/target/runner-microvm-host-standing.txt" @@ -834,7 +834,7 @@ jobs: timeout-minutes: 10 - name: "Runner micro-VM network: read the slot taps, ruleset and forwarding back into a receipt (installs nothing)" id: microvm_network_observe - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/runner/runner_microvm_network_observe.dag --function runner_microvm_network_observe_wet cat "$ROOT/target/microvm-network-observe.txt" @@ -854,7 +854,7 @@ jobs: timeout-minutes: 10 - name: "Runner micro-VM network: stage and install the slot and host network files as the host administrator, then read the ruleset back" id: microvm_network_apply - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/runner/runner_microvm_network_apply.dag --function runner_microvm_network_apply_ci_wet cat "$ROOT/target/microvm-network-apply-receipt.txt" @@ -875,7 +875,7 @@ jobs: timeout-minutes: 10 - name: "Runner micro-VM: boot the guest image and read the serial console" id: microvm_boot_probe - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/runner/runner_microvm_boot_probe.dag --function runner_microvm_boot_probe_wet cat "$ROOT/target/runner-microvm-boot-probe.txt" @@ -895,7 +895,7 @@ jobs: timeout-minutes: 10 - name: Spark managed grant reconcile (materializes its own administrator credential in-run, installs only the grants measured missing at the named target, removes the credential within this same step) id: spark_grants - run: | + run: |- set -euo pipefail umask 077 CRED_FILE="$RUNNER_TEMP/spark-administrator-credential" @@ -924,7 +924,7 @@ jobs: timeout-minutes: 60 - name: Spark managed-access bootstrap (creates gunbc-automation, installs the fleet key, proves key-only auth, lays the narrow grants; one target per run) id: spark_bootstrap - run: | + run: |- set -euo pipefail umask 077 CRED_FILE="$RUNNER_TEMP/spark-administrator-credential" @@ -953,7 +953,7 @@ jobs: timeout-minutes: 60 - name: Spark pair serving apply (promoted fabric groups' vLLM units over the password session, workers before heads) id: spark_serving_apply - run: | + run: |- set -euo pipefail umask 077 CRED_FILE="$RUNNER_TEMP/spark-administrator-credential" @@ -982,7 +982,7 @@ jobs: timeout-minutes: 60 - name: "Spark native serving apply (group B's native four-rank arm as one bounded transaction over the password session: all-host preflight that mutates nothing, the incumbent unit preserved, head before workers, a readback of the complete realization bound to each rank's own incarnation, then commit or roll the whole arm back)" id: spark_native_serving_apply - run: | + run: |- set -euo pipefail umask 077 CRED_FILE="$RUNNER_TEMP/spark-administrator-credential" @@ -1011,7 +1011,7 @@ jobs: timeout-minutes: 60 - name: Spark runtime image probe (pull the pinned image on the selected Spark and ask it, inside its own digest, what it registers and what it admits) id: spark_runtime_image_probe - run: | + run: |- set -euo pipefail umask 077 CRED_FILE="$RUNNER_TEMP/spark-administrator-credential" @@ -1040,7 +1040,7 @@ jobs: timeout-minutes: 60 - name: Spark V4.1 checkpoint materialize (fetch the admitted published files onto the selected Group A Spark, verify each sha256 before publishing, read the storage-backed Engram spans) id: spark_v41_checkpoint_materialize - run: | + run: |- set -euo pipefail umask 077 CRED_FILE="$RUNNER_TEMP/spark-administrator-credential" @@ -1069,7 +1069,7 @@ jobs: timeout-minutes: 60 - name: Spark V4.1 index selection (verify the pinned index on the selected Group A Spark and read its .engram. selection with parse_json_document, in its own process) id: spark_v41_index_selection - run: | + run: |- set -euo pipefail umask 077 CRED_FILE="$RUNNER_TEMP/spark-administrator-credential" @@ -1098,7 +1098,7 @@ jobs: timeout-minutes: 60 - name: Spark V4.1 runtime image build (verify the candidate's kernel wheels, converge the patched source, build the image on the selected Spark, read the produced digest back from inside it and admit it against the candidate) id: spark_v41_runtime_image_build - run: | + run: |- set -euo pipefail umask 077 CRED_FILE="$RUNNER_TEMP/spark-administrator-credential" @@ -1127,7 +1127,7 @@ jobs: timeout-minutes: 180 - name: "Runner host files: observe the teardown drop-in, the needrestart deferral and the loaded teardown (no writes)" id: runner_host_file_observe - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/runner/runner_host_file_converge.dag --function runner_host_file_observe_ci_wet cat "$ROOT/target/runner-host-file-converge-receipt.txt" @@ -1137,7 +1137,7 @@ jobs: timeout-minutes: 15 - name: "Runner host files: write what differs as the administrator, reload if the drop-in changed, read the loaded teardown back" id: runner_host_file_converge - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/runner/runner_host_file_converge.dag --function runner_host_file_converge_ci_wet cat "$ROOT/target/runner-host-file-converge-receipt.txt" @@ -1147,7 +1147,7 @@ jobs: timeout-minutes: 15 - name: "Password-session tools: ensure every tool password_session_host_cli_requirements enrolls is present on this runner, installing only what is measured absent" id: runner_password_session_tool_converge - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/runner/runner_password_session_tool_converge.dag --function runner_password_session_tool_converge_ci_wet cat "$ROOT/target/runner-password-session-tool-converge-receipt.txt" @@ -1167,7 +1167,7 @@ jobs: timeout-minutes: 10 - name: "Host credential custody: deliver the chosen rostered credential from Secret Manager at its row's owner, group and mode, read owner, mode, directory, bytes and staging back" id: host_credential_custody_converge - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/fleet/host_credential_custody_converge.dag --function host_credential_custody_converge_ci_wet cat "$ROOT/target/host-credential-custody-converge-receipt.txt" @@ -1199,7 +1199,7 @@ jobs: timeout-minutes: 10 - name: "Site PXE edge: observe the chainloader digest, dnsmasq, the edge config and unit (no writes)" id: site_pxe_edge_observe - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/fleet/site_pxe_edge_converge.dag --function site_pxe_edge_observe_ci_wet cat "$ROOT/target/site-pxe-edge-converge-receipt.txt" @@ -1209,7 +1209,7 @@ jobs: timeout-minutes: 15 - name: "Site PXE edge: if the gate is open, write what differs, restart the unit, read its active state back" id: site_pxe_edge_converge - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/fleet/site_pxe_edge_converge.dag --function site_pxe_edge_converge_ci_wet cat "$ROOT/target/site-pxe-edge-converge-receipt.txt" @@ -1263,7 +1263,7 @@ jobs: timeout-minutes: 5 - name: "Host reset-return: drive the subject through its controller and measure the return from a peer" id: host_reset_return - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/host/host_reset_return_run.dag --function host_reset_return_wet cat "$ROOT/target/host-reset-return.txt" @@ -1330,7 +1330,7 @@ jobs: timeout-minutes: 10 - name: Deploy dashboard to srv1 (live_deploy_apply_srv1_transaction_wet, receipted) id: dashboard_deploy - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/live_deploy/apply.dag --function live_deploy_apply_srv1_transaction_wet cat "$ROOT/target/live-deploy-srv1-live-receipt/live_deploy_receipt.json" @@ -1382,7 +1382,7 @@ jobs: timeout-minutes: 5 - name: Install the approval broker dark on srv1 (additive unit + slice; does not restart gunbc-roadmap.service) id: approval_broker_dark_install - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/live_deploy/apply.dag --function approval_broker_dark_install_srv1_wet timeout-minutes: 30 @@ -1417,7 +1417,7 @@ jobs: timeout-minutes: 5 - name: Install the microVM slot controller release locus, VMM + jailer and template unit on srv1 (additive; starts no instance) id: microvm_controller_install - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/live_deploy/apply.dag --function microvm_controller_install_srv1_wet timeout-minutes: 30 @@ -1464,7 +1464,7 @@ jobs: create_credentials_file: false timeout-minutes: 5 - name: Materialize fleet key in-run (SM versions/1 pinned -> RUNNER_TEMP 0600 -> ssh-agent -> wipe file) - run: | + run: |- # 🟡 dissolve-on: gunbc_ci_fleet_key_agent_script - orch-emitted foreign-executor (GitHub Actions run:) credential runner: WIF access token by env, one PINNED secret version fetched over curl, a 0600 key file under RUNNER_TEMP with a trap armed BEFORE the credential touches disk, fingerprint verified against the modeled authority, ssh-agent load, file wipe. The pipeline steps are modeled (gunbc_ci_fleet_key_agent_prelude) but the runner transport itself remains hand-shell; DISSOLVES WHEN bash-emit (#5828 / ROADMAP 6-shell-slice0 / shell-to-intent Phase 2) realizes the credential runner through orchestration emit or typed host_effect_apply without a medium-as-string concat scaffold set -euo pipefail umask 077 @@ -1520,7 +1520,7 @@ jobs: timeout-minutes: 10 - name: Roadmap launch deployment receipt (rlm_launch_deployment_receipt_wet) id: rlm_receipt - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/roadmap/roadmap_launch_deployment_cli.dag --function rlm_launch_deployment_receipt_wet cat "$ROOT/target/rlm-launch-deployment-receipt/receipt.json" diff --git a/.github/workflows/fleet-desired.yml b/.github/workflows/fleet-desired.yml index f693eab4d7a..189e8b399fa 100644 --- a/.github/workflows/fleet-desired.yml +++ b/.github/workflows/fleet-desired.yml @@ -19,9 +19,9 @@ jobs: - name: Checkout uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 with: - fetch-depth: 0 + fetch-depth: "0" - name: Isolate toolchain homes - run: | + run: |- # dissolve-on: ci_toolchain_home_isolation_script -- orch-emitted foreign-executor prelude step wiping and setting HOME/CARGO_HOME/RUSTUP_HOME under RUNNER_TEMP so concurrent runner slots stop sharing one toolchain; leaf rm/echo strings remain until a typed per-job filesystem-and-environment effect lands on host_effect_apply (shell-to-intent Phase 2). This obligation covers THIS carrier and ci_isolate_toolchain_script, which share that terminal construction; ci_pin_rustup_default_script carries its own obligation because it does not rm -rf "$RUNNER_TEMP/rustup" "$RUNNER_TEMP/cargo" echo "HOME=$RUNNER_TEMP" >> "$GITHUB_ENV" @@ -34,7 +34,7 @@ jobs: cache: false rustflags: -D warnings - name: Pin rustup default (isolated RUSTUP_HOME has no default toolchain) - run: | + run: |- # dissolve-on: ci_pin_rustup_default_script -- orch-emitted foreign-executor step selecting a rustup default toolchain inside an isolated RUSTUP_HOME, which starts with none, and resolving the cargo binary that selection implies. The leaf rustup/command/echo strings remain until a typed TOOLCHAIN-SELECTION effect lands on host_effect_apply -- NOT the filesystem-and-environment effect ci_toolchain_home_isolation_script waits on, which is why this is a separate obligation: that effect landing alone would leave this carrier standing rustup default "$(rustup show active-toolchain | awk '{print $1; exit}')" if [ -x "$CARGO_HOME/bin/cargo" ]; then CARGO_BIN="$CARGO_HOME/bin/cargo"; else CARGO_BIN="$(command -v cargo || true)"; fi diff --git a/.github/workflows/heal-publish.yml b/.github/workflows/heal-publish.yml index 391756a8d17..b479d2cc9a5 100644 --- a/.github/workflows/heal-publish.yml +++ b/.github/workflows/heal-publish.yml @@ -25,7 +25,7 @@ jobs: steps: - name: Ask the triggering run whether it sealed a candidate id: candidate_probe - run: | + run: |- set -euo pipefail NAMES=$(gh api --paginate "repos/$GITHUB_REPOSITORY/actions/runs/$TRIGGERING_RUN_ID/artifacts" --jq '.artifacts[].name') if printf '%s\n' "$NAMES" | grep -qx "heal-repair-candidate"; then @@ -52,7 +52,7 @@ jobs: github-token: ${{ github.token }} if: steps.candidate_probe.outputs.present == 'true' - name: Isolate toolchain homes - run: | + run: |- # dissolve-on: ci_toolchain_home_isolation_script -- orch-emitted foreign-executor prelude step wiping and setting HOME/CARGO_HOME/RUSTUP_HOME under RUNNER_TEMP so concurrent runner slots stop sharing one toolchain; leaf rm/echo strings remain until a typed per-job filesystem-and-environment effect lands on host_effect_apply (shell-to-intent Phase 2). This obligation covers THIS carrier and ci_isolate_toolchain_script, which share that terminal construction; ci_pin_rustup_default_script carries its own obligation because it does not rm -rf "$RUNNER_TEMP/rustup" "$RUNNER_TEMP/cargo" echo "HOME=$RUNNER_TEMP" >> "$GITHUB_ENV" @@ -65,7 +65,7 @@ jobs: cache: false rustflags: -D warnings - name: Pin rustup default (isolated RUSTUP_HOME has no default toolchain) - run: | + run: |- # dissolve-on: ci_pin_rustup_default_script -- orch-emitted foreign-executor step selecting a rustup default toolchain inside an isolated RUSTUP_HOME, which starts with none, and resolving the cargo binary that selection implies. The leaf rustup/command/echo strings remain until a typed TOOLCHAIN-SELECTION effect lands on host_effect_apply -- NOT the filesystem-and-environment effect ci_toolchain_home_isolation_script waits on, which is why this is a separate obligation: that effect landing alone would leave this carrier standing rustup default "$(rustup show active-toolchain | awk '{print $1; exit}')" if [ -x "$CARGO_HOME/bin/cargo" ]; then CARGO_BIN="$CARGO_HOME/bin/cargo"; else CARGO_BIN="$(command -v cargo || true)"; fi @@ -73,7 +73,7 @@ jobs: echo "CARGO_BIN=$CARGO_BIN" >> "$GITHUB_ENV" - name: Build the publisher this job runs from the default branch id: build_witness_fold - run: | + run: |+ GUNBC_FLOOR_LOG='gunbc-floor-cmd.log' 'set' '+e' 'set' '-o' 'pipefail' @@ -94,8 +94,9 @@ jobs: if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' && '[' '-f' "$GUNBC_FLOOR_LOG" ']' && 'grep' '-q' 'Resource temporarily unavailable' "$GUNBC_FLOOR_LOG"; then GUNBC_FLOOR_CLASS='infra'; GUNBC_FLOOR_SIGNATURE='ResourceTemporarilyUnavailable'; fi if (! '[' '-f' "$GUNBC_FLOOR_RECEIPT" ']') || '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' || ('[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT"))) || ('[' "$GUNBC_FLOOR_CLASS" '=' 'none' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT")) && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=infra' "$GUNBC_FLOOR_RECEIPT"))); then 'printf' 'class=%s\nsignature=%s\nexit=%s\n' "$GUNBC_FLOOR_CLASS" "$GUNBC_FLOOR_SIGNATURE" "$GUNBC_FLOOR_EXIT" > "$GUNBC_FLOOR_RECEIPT"; if '[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']'; then 'echo' '::error title=environment::floor_class='"$GUNBC_FLOOR_CLASS"' signature='"$GUNBC_FLOOR_SIGNATURE"' exit='"$GUNBC_FLOOR_EXIT"'; this is not a verdict about the diff. Attempt receipt: '"$GUNBC_FLOOR_RECEIPT"; fi; if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']'; then 'echo' '::error title=subject::floor_class='"$GUNBC_FLOOR_CLASS"' exit='"$GUNBC_FLOOR_EXIT"'; read the step log for the subject defect'; fi; fi 'exit' "$GUNBC_FLOOR_EXIT" + - name: Admit the candidate against this checkout's policy, with no publication credential - run: | + run: |- set -euo pipefail GUNBC_CG=/sys/fs/cgroup/gunbc-publish echo "+memory" | sudo tee /sys/fs/cgroup/cgroup.subtree_control >/dev/null @@ -121,7 +122,7 @@ jobs: export_environment_variables: false if: hashFiles('heal-candidate/heal-repair-candidate-blob-*') != '' - name: Publish the admitted repair, fast-forward only, and read it back - run: | + run: |- set -euo pipefail GUNBC_CG=/sys/fs/cgroup/gunbc-publish echo "+memory" | sudo tee /sys/fs/cgroup/cgroup.subtree_control >/dev/null diff --git a/.github/workflows/heal.yml b/.github/workflows/heal.yml index 7a0682845c1..a8d4821161e 100644 --- a/.github/workflows/heal.yml +++ b/.github/workflows/heal.yml @@ -27,7 +27,7 @@ concurrency: cancel-in-progress: true env: CARGO_TERM_COLOR: always - MALLOC_ARENA_MAX: 2 + MALLOC_ARENA_MAX: "2" jobs: heal-generated-artifacts: runs-on: [self-hosted, linux, arm64] @@ -43,7 +43,7 @@ jobs: ref: ${{ inputs.head_sha }} persist-credentials: false - name: Isolate toolchain homes - run: | + run: |- # dissolve-on: ci_toolchain_home_isolation_script -- orch-emitted foreign-executor prelude step wiping and setting HOME/CARGO_HOME/RUSTUP_HOME under RUNNER_TEMP so concurrent runner slots stop sharing one toolchain; leaf rm/echo strings remain until a typed per-job filesystem-and-environment effect lands on host_effect_apply (shell-to-intent Phase 2). This obligation covers THIS carrier and ci_isolate_toolchain_script, which share that terminal construction; ci_pin_rustup_default_script carries its own obligation because it does not rm -rf "$RUNNER_TEMP/rustup" "$RUNNER_TEMP/cargo" echo "HOME=$RUNNER_TEMP" >> "$GITHUB_ENV" @@ -56,7 +56,7 @@ jobs: cache: false rustflags: -D warnings - name: Pin rustup default (isolated RUSTUP_HOME has no default toolchain) - run: | + run: |- # dissolve-on: ci_pin_rustup_default_script -- orch-emitted foreign-executor step selecting a rustup default toolchain inside an isolated RUSTUP_HOME, which starts with none, and resolving the cargo binary that selection implies. The leaf rustup/command/echo strings remain until a typed TOOLCHAIN-SELECTION effect lands on host_effect_apply -- NOT the filesystem-and-environment effect ci_toolchain_home_isolation_script waits on, which is why this is a separate obligation: that effect landing alone would leave this carrier standing rustup default "$(rustup show active-toolchain | awk '{print $1; exit}')" if [ -x "$CARGO_HOME/bin/cargo" ]; then CARGO_BIN="$CARGO_HOME/bin/cargo"; else CARGO_BIN="$(command -v cargo || true)"; fi @@ -64,7 +64,7 @@ jobs: echo "CARGO_BIN=$CARGO_BIN" >> "$GITHUB_ENV" - name: Build the regenerator this job runs against its own checkout id: build_witness_fold - run: | + run: |+ GUNBC_FLOOR_LOG='gunbc-floor-cmd.log' 'set' '+e' 'set' '-o' 'pipefail' @@ -85,18 +85,19 @@ jobs: if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' && '[' '-f' "$GUNBC_FLOOR_LOG" ']' && 'grep' '-q' 'Resource temporarily unavailable' "$GUNBC_FLOOR_LOG"; then GUNBC_FLOOR_CLASS='infra'; GUNBC_FLOOR_SIGNATURE='ResourceTemporarilyUnavailable'; fi if (! '[' '-f' "$GUNBC_FLOOR_RECEIPT" ']') || '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' || ('[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT"))) || ('[' "$GUNBC_FLOOR_CLASS" '=' 'none' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT")) && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=infra' "$GUNBC_FLOOR_RECEIPT"))); then 'printf' 'class=%s\nsignature=%s\nexit=%s\n' "$GUNBC_FLOOR_CLASS" "$GUNBC_FLOOR_SIGNATURE" "$GUNBC_FLOOR_EXIT" > "$GUNBC_FLOOR_RECEIPT"; if '[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']'; then 'echo' '::error title=environment::floor_class='"$GUNBC_FLOOR_CLASS"' signature='"$GUNBC_FLOOR_SIGNATURE"' exit='"$GUNBC_FLOOR_EXIT"'; this is not a verdict about the diff. Attempt receipt: '"$GUNBC_FLOOR_RECEIPT"; fi; if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']'; then 'echo' '::error title=subject::floor_class='"$GUNBC_FLOOR_CLASS"' exit='"$GUNBC_FLOOR_EXIT"'; read the step log for the subject defect'; fi; fi 'exit' "$GUNBC_FLOOR_EXIT" + - name: Declare which ledger rows this heal will repair - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/instruments/generated_artifact_gate.dag --function heal_repair_declaration - name: Regenerate every registry-rostered generated artifact (not the stage0 mirrors) - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/instruments/generated_artifact_gate.dag --function main_wet ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/instruments/generated_artifact_gate.dag --function main - name: Produce the sealed repair candidate - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/heal_candidate.dag --function heal_candidate_produce_wet env: diff --git a/.github/workflows/witnesses.yml b/.github/workflows/witnesses.yml index 9d67aa72ad4..c90be8cbd0f 100644 --- a/.github/workflows/witnesses.yml +++ b/.github/workflows/witnesses.yml @@ -32,7 +32,7 @@ jobs: with: persist-credentials: false - name: Isolate toolchain homes - run: | + run: |- # dissolve-on: ci_toolchain_home_isolation_script -- orch-emitted foreign-executor prelude step wiping and setting HOME/CARGO_HOME/RUSTUP_HOME under RUNNER_TEMP so concurrent runner slots stop sharing one toolchain; leaf rm/echo strings remain until a typed per-job filesystem-and-environment effect lands on host_effect_apply (shell-to-intent Phase 2). This obligation covers THIS carrier and ci_isolate_toolchain_script, which share that terminal construction; ci_pin_rustup_default_script carries its own obligation because it does not rm -rf "$RUNNER_TEMP/rustup" "$RUNNER_TEMP/cargo" echo "HOME=$RUNNER_TEMP" >> "$GITHUB_ENV" @@ -45,7 +45,7 @@ jobs: cache: false rustflags: -D warnings - name: Pin rustup default (isolated RUSTUP_HOME has no default toolchain) - run: | + run: |- # dissolve-on: ci_pin_rustup_default_script -- orch-emitted foreign-executor step selecting a rustup default toolchain inside an isolated RUSTUP_HOME, which starts with none, and resolving the cargo binary that selection implies. The leaf rustup/command/echo strings remain until a typed TOOLCHAIN-SELECTION effect lands on host_effect_apply -- NOT the filesystem-and-environment effect ci_toolchain_home_isolation_script waits on, which is why this is a separate obligation: that effect landing alone would leave this carrier standing rustup default "$(rustup show active-toolchain | awk '{print $1; exit}')" if [ -x "$CARGO_HOME/bin/cargo" ]; then CARGO_BIN="$CARGO_HOME/bin/cargo"; else CARGO_BIN="$(command -v cargo || true)"; fi @@ -53,7 +53,7 @@ jobs: echo "CARGO_BIN=$CARGO_BIN" >> "$GITHUB_ENV" - name: Build the compiler id: build_witness_fold - run: | + run: |+ GUNBC_FLOOR_LOG='gunbc-floor-cmd.log' 'set' '+e' 'set' '-o' 'pipefail' @@ -74,6 +74,7 @@ jobs: if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' && '[' '-f' "$GUNBC_FLOOR_LOG" ']' && 'grep' '-q' 'Resource temporarily unavailable' "$GUNBC_FLOOR_LOG"; then GUNBC_FLOOR_CLASS='infra'; GUNBC_FLOOR_SIGNATURE='ResourceTemporarilyUnavailable'; fi if (! '[' '-f' "$GUNBC_FLOOR_RECEIPT" ']') || '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' || ('[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT"))) || ('[' "$GUNBC_FLOOR_CLASS" '=' 'none' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT")) && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=infra' "$GUNBC_FLOOR_RECEIPT"))); then 'printf' 'class=%s\nsignature=%s\nexit=%s\n' "$GUNBC_FLOOR_CLASS" "$GUNBC_FLOOR_SIGNATURE" "$GUNBC_FLOOR_EXIT" > "$GUNBC_FLOOR_RECEIPT"; if '[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']'; then 'echo' '::error title=environment::floor_class='"$GUNBC_FLOOR_CLASS"' signature='"$GUNBC_FLOOR_SIGNATURE"' exit='"$GUNBC_FLOOR_EXIT"'; this is not a verdict about the diff. Attempt receipt: '"$GUNBC_FLOOR_RECEIPT"; fi; if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']'; then 'echo' '::error title=subject::floor_class='"$GUNBC_FLOOR_CLASS"' exit='"$GUNBC_FLOOR_EXIT"'; read the step log for the subject defect'; fi; fi 'exit' "$GUNBC_FLOOR_EXIT" + clippy: runs-on: ubuntu-24.04-arm timeout-minutes: 45 @@ -85,7 +86,7 @@ jobs: with: persist-credentials: false - name: Isolate toolchain homes - run: | + run: |- # dissolve-on: ci_toolchain_home_isolation_script -- orch-emitted foreign-executor prelude step wiping and setting HOME/CARGO_HOME/RUSTUP_HOME under RUNNER_TEMP so concurrent runner slots stop sharing one toolchain; leaf rm/echo strings remain until a typed per-job filesystem-and-environment effect lands on host_effect_apply (shell-to-intent Phase 2). This obligation covers THIS carrier and ci_isolate_toolchain_script, which share that terminal construction; ci_pin_rustup_default_script carries its own obligation because it does not rm -rf "$RUNNER_TEMP/rustup" "$RUNNER_TEMP/cargo" echo "HOME=$RUNNER_TEMP" >> "$GITHUB_ENV" @@ -98,7 +99,7 @@ jobs: cache: false rustflags: -D warnings - name: Pin rustup default (isolated RUSTUP_HOME has no default toolchain) - run: | + run: |- # dissolve-on: ci_pin_rustup_default_script -- orch-emitted foreign-executor step selecting a rustup default toolchain inside an isolated RUSTUP_HOME, which starts with none, and resolving the cargo binary that selection implies. The leaf rustup/command/echo strings remain until a typed TOOLCHAIN-SELECTION effect lands on host_effect_apply -- NOT the filesystem-and-environment effect ci_toolchain_home_isolation_script waits on, which is why this is a separate obligation: that effect landing alone would leave this carrier standing rustup default "$(rustup show active-toolchain | awk '{print $1; exit}')" if [ -x "$CARGO_HOME/bin/cargo" ]; then CARGO_BIN="$CARGO_HOME/bin/cargo"; else CARGO_BIN="$(command -v cargo || true)"; fi @@ -106,7 +107,7 @@ jobs: echo "CARGO_BIN=$CARGO_BIN" >> "$GITHUB_ENV" - name: Lint every target id: build_witness_fold - run: | + run: |+ GUNBC_FLOOR_LOG='gunbc-floor-cmd.log' 'set' '+e' 'set' '-o' 'pipefail' @@ -127,6 +128,7 @@ jobs: if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' && '[' '-f' "$GUNBC_FLOOR_LOG" ']' && 'grep' '-q' 'Resource temporarily unavailable' "$GUNBC_FLOOR_LOG"; then GUNBC_FLOOR_CLASS='infra'; GUNBC_FLOOR_SIGNATURE='ResourceTemporarilyUnavailable'; fi if (! '[' '-f' "$GUNBC_FLOOR_RECEIPT" ']') || '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' || ('[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT"))) || ('[' "$GUNBC_FLOOR_CLASS" '=' 'none' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT")) && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=infra' "$GUNBC_FLOOR_RECEIPT"))); then 'printf' 'class=%s\nsignature=%s\nexit=%s\n' "$GUNBC_FLOOR_CLASS" "$GUNBC_FLOOR_SIGNATURE" "$GUNBC_FLOOR_EXIT" > "$GUNBC_FLOOR_RECEIPT"; if '[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']'; then 'echo' '::error title=environment::floor_class='"$GUNBC_FLOOR_CLASS"' signature='"$GUNBC_FLOOR_SIGNATURE"' exit='"$GUNBC_FLOOR_EXIT"'; this is not a verdict about the diff. Attempt receipt: '"$GUNBC_FLOOR_RECEIPT"; fi; if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']'; then 'echo' '::error title=subject::floor_class='"$GUNBC_FLOOR_CLASS"' exit='"$GUNBC_FLOOR_EXIT"'; read the step log for the subject defect'; fi; fi 'exit' "$GUNBC_FLOOR_EXIT" + emit-build: runs-on: ubuntu-24.04-arm timeout-minutes: 90 @@ -138,7 +140,7 @@ jobs: with: persist-credentials: false - name: Isolate toolchain homes - run: | + run: |- # dissolve-on: ci_toolchain_home_isolation_script -- orch-emitted foreign-executor prelude step wiping and setting HOME/CARGO_HOME/RUSTUP_HOME under RUNNER_TEMP so concurrent runner slots stop sharing one toolchain; leaf rm/echo strings remain until a typed per-job filesystem-and-environment effect lands on host_effect_apply (shell-to-intent Phase 2). This obligation covers THIS carrier and ci_isolate_toolchain_script, which share that terminal construction; ci_pin_rustup_default_script carries its own obligation because it does not rm -rf "$RUNNER_TEMP/rustup" "$RUNNER_TEMP/cargo" echo "HOME=$RUNNER_TEMP" >> "$GITHUB_ENV" @@ -151,7 +153,7 @@ jobs: cache: false rustflags: -D warnings - name: Pin rustup default (isolated RUSTUP_HOME has no default toolchain) - run: | + run: |- # dissolve-on: ci_pin_rustup_default_script -- orch-emitted foreign-executor step selecting a rustup default toolchain inside an isolated RUSTUP_HOME, which starts with none, and resolving the cargo binary that selection implies. The leaf rustup/command/echo strings remain until a typed TOOLCHAIN-SELECTION effect lands on host_effect_apply -- NOT the filesystem-and-environment effect ci_toolchain_home_isolation_script waits on, which is why this is a separate obligation: that effect landing alone would leave this carrier standing rustup default "$(rustup show active-toolchain | awk '{print $1; exit}')" if [ -x "$CARGO_HOME/bin/cargo" ]; then CARGO_BIN="$CARGO_HOME/bin/cargo"; else CARGO_BIN="$(command -v cargo || true)"; fi @@ -159,7 +161,7 @@ jobs: echo "CARGO_BIN=$CARGO_BIN" >> "$GITHUB_ENV" - name: Build the compiler id: build_witness_fold - run: | + run: |+ GUNBC_FLOOR_LOG='gunbc-floor-cmd.log' 'set' '+e' 'set' '-o' 'pipefail' @@ -180,9 +182,10 @@ jobs: if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' && '[' '-f' "$GUNBC_FLOOR_LOG" ']' && 'grep' '-q' 'Resource temporarily unavailable' "$GUNBC_FLOOR_LOG"; then GUNBC_FLOOR_CLASS='infra'; GUNBC_FLOOR_SIGNATURE='ResourceTemporarilyUnavailable'; fi if (! '[' '-f' "$GUNBC_FLOOR_RECEIPT" ']') || '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' || ('[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT"))) || ('[' "$GUNBC_FLOOR_CLASS" '=' 'none' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT")) && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=infra' "$GUNBC_FLOOR_RECEIPT"))); then 'printf' 'class=%s\nsignature=%s\nexit=%s\n' "$GUNBC_FLOOR_CLASS" "$GUNBC_FLOOR_SIGNATURE" "$GUNBC_FLOOR_EXIT" > "$GUNBC_FLOOR_RECEIPT"; if '[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']'; then 'echo' '::error title=environment::floor_class='"$GUNBC_FLOOR_CLASS"' signature='"$GUNBC_FLOOR_SIGNATURE"' exit='"$GUNBC_FLOOR_EXIT"'; this is not a verdict about the diff. Attempt receipt: '"$GUNBC_FLOOR_RECEIPT"; fi; if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']'; then 'echo' '::error title=subject::floor_class='"$GUNBC_FLOOR_CLASS"' exit='"$GUNBC_FLOOR_EXIT"'; read the step log for the subject defect'; fi; fi 'exit' "$GUNBC_FLOOR_EXIT" + - name: emit and build //gunbc/instruments:self-host id: emit_build_self_host - run: | + run: |+ GUNBC_FLOOR_LOG='gunbc-floor-cmd.log' 'set' '+e' 'set' '-o' 'pipefail' @@ -203,9 +206,10 @@ jobs: if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' && '[' '-f' "$GUNBC_FLOOR_LOG" ']' && 'grep' '-q' 'Resource temporarily unavailable' "$GUNBC_FLOOR_LOG"; then GUNBC_FLOOR_CLASS='infra'; GUNBC_FLOOR_SIGNATURE='ResourceTemporarilyUnavailable'; fi if (! '[' '-f' "$GUNBC_FLOOR_RECEIPT" ']') || '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' || ('[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT"))) || ('[' "$GUNBC_FLOOR_CLASS" '=' 'none' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT")) && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=infra' "$GUNBC_FLOOR_RECEIPT"))); then 'printf' 'class=%s\nsignature=%s\nexit=%s\n' "$GUNBC_FLOOR_CLASS" "$GUNBC_FLOOR_SIGNATURE" "$GUNBC_FLOOR_EXIT" > "$GUNBC_FLOOR_RECEIPT"; if '[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']'; then 'echo' '::error title=environment::floor_class='"$GUNBC_FLOOR_CLASS"' signature='"$GUNBC_FLOOR_SIGNATURE"' exit='"$GUNBC_FLOOR_EXIT"'; this is not a verdict about the diff. Attempt receipt: '"$GUNBC_FLOOR_RECEIPT"; fi; if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']'; then 'echo' '::error title=subject::floor_class='"$GUNBC_FLOOR_CLASS"' exit='"$GUNBC_FLOOR_EXIT"'; read the step log for the subject defect'; fi; fi 'exit' "$GUNBC_FLOOR_EXIT" + - name: emit and build //gunbc/instruments:v2-native-cli id: emit_build_v2_native_cli - run: | + run: |+ GUNBC_FLOOR_LOG='gunbc-floor-cmd.log' 'set' '+e' 'set' '-o' 'pipefail' @@ -226,8 +230,9 @@ jobs: if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' && '[' '-f' "$GUNBC_FLOOR_LOG" ']' && 'grep' '-q' 'Resource temporarily unavailable' "$GUNBC_FLOOR_LOG"; then GUNBC_FLOOR_CLASS='infra'; GUNBC_FLOOR_SIGNATURE='ResourceTemporarilyUnavailable'; fi if (! '[' '-f' "$GUNBC_FLOOR_RECEIPT" ']') || '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' || ('[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT"))) || ('[' "$GUNBC_FLOOR_CLASS" '=' 'none' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT")) && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=infra' "$GUNBC_FLOOR_RECEIPT"))); then 'printf' 'class=%s\nsignature=%s\nexit=%s\n' "$GUNBC_FLOOR_CLASS" "$GUNBC_FLOOR_SIGNATURE" "$GUNBC_FLOOR_EXIT" > "$GUNBC_FLOOR_RECEIPT"; if '[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']'; then 'echo' '::error title=environment::floor_class='"$GUNBC_FLOOR_CLASS"' signature='"$GUNBC_FLOOR_SIGNATURE"' exit='"$GUNBC_FLOOR_EXIT"'; this is not a verdict about the diff. Attempt receipt: '"$GUNBC_FLOOR_RECEIPT"; fi; if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']'; then 'echo' '::error title=subject::floor_class='"$GUNBC_FLOOR_CLASS"' exit='"$GUNBC_FLOOR_EXIT"'; read the step log for the subject defect'; fi; fi 'exit' "$GUNBC_FLOOR_EXIT" + - name: Why this lane is red, and what would be a real finding - run: | + run: |- printf '%s\n' 'emit-build is a NON-REQUIRED detector lane. A red here does NOT block your merge:' printf '%s\n' 'the required context is `witnesses`, and this job is not an input to it.' printf '%s\n' '' @@ -263,7 +268,7 @@ jobs: ref: ${{ github.event.pull_request.head.sha }} persist-credentials: false - name: Isolate toolchain homes - run: | + run: |- # dissolve-on: ci_toolchain_home_isolation_script -- orch-emitted foreign-executor prelude step wiping and setting HOME/CARGO_HOME/RUSTUP_HOME under RUNNER_TEMP so concurrent runner slots stop sharing one toolchain; leaf rm/echo strings remain until a typed per-job filesystem-and-environment effect lands on host_effect_apply (shell-to-intent Phase 2). This obligation covers THIS carrier and ci_isolate_toolchain_script, which share that terminal construction; ci_pin_rustup_default_script carries its own obligation because it does not rm -rf "$RUNNER_TEMP/rustup" "$RUNNER_TEMP/cargo" echo "HOME=$RUNNER_TEMP" >> "$GITHUB_ENV" @@ -276,7 +281,7 @@ jobs: cache: false rustflags: -D warnings - name: Pin rustup default (isolated RUSTUP_HOME has no default toolchain) - run: | + run: |- # dissolve-on: ci_pin_rustup_default_script -- orch-emitted foreign-executor step selecting a rustup default toolchain inside an isolated RUSTUP_HOME, which starts with none, and resolving the cargo binary that selection implies. The leaf rustup/command/echo strings remain until a typed TOOLCHAIN-SELECTION effect lands on host_effect_apply -- NOT the filesystem-and-environment effect ci_toolchain_home_isolation_script waits on, which is why this is a separate obligation: that effect landing alone would leave this carrier standing rustup default "$(rustup show active-toolchain | awk '{print $1; exit}')" if [ -x "$CARGO_HOME/bin/cargo" ]; then CARGO_BIN="$CARGO_HOME/bin/cargo"; else CARGO_BIN="$(command -v cargo || true)"; fi @@ -284,7 +289,7 @@ jobs: echo "CARGO_BIN=$CARGO_BIN" >> "$GITHUB_ENV" - name: Build the compiler and the witness executor id: build_witness_fold - run: | + run: |+ GUNBC_FLOOR_LOG='gunbc-floor-cmd.log' 'set' '+e' 'set' '-o' 'pipefail' @@ -305,13 +310,14 @@ jobs: if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' && '[' '-f' "$GUNBC_FLOOR_LOG" ']' && 'grep' '-q' 'Resource temporarily unavailable' "$GUNBC_FLOOR_LOG"; then GUNBC_FLOOR_CLASS='infra'; GUNBC_FLOOR_SIGNATURE='ResourceTemporarilyUnavailable'; fi if (! '[' '-f' "$GUNBC_FLOOR_RECEIPT" ']') || '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' || ('[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT"))) || ('[' "$GUNBC_FLOOR_CLASS" '=' 'none' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT")) && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=infra' "$GUNBC_FLOOR_RECEIPT"))); then 'printf' 'class=%s\nsignature=%s\nexit=%s\n' "$GUNBC_FLOOR_CLASS" "$GUNBC_FLOOR_SIGNATURE" "$GUNBC_FLOOR_EXIT" > "$GUNBC_FLOOR_RECEIPT"; if '[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']'; then 'echo' '::error title=environment::floor_class='"$GUNBC_FLOOR_CLASS"' signature='"$GUNBC_FLOOR_SIGNATURE"' exit='"$GUNBC_FLOOR_EXIT"'; this is not a verdict about the diff. Attempt receipt: '"$GUNBC_FLOOR_RECEIPT"; fi; if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']'; then 'echo' '::error title=subject::floor_class='"$GUNBC_FLOOR_CLASS"' exit='"$GUNBC_FLOOR_EXIT"'; read the step log for the subject defect'; fi; fi 'exit' "$GUNBC_FLOOR_EXIT" + - name: Require isolated toolchain homes run: | if [ -z "${CARGO_HOME:-}" ] || [ -z "${RUSTUP_HOME:-}" ]; then echo "::error::ToolchainHomesNotIsolated CARGO_HOME=${CARGO_HOME:-unset} RUSTUP_HOME=${RUSTUP_HOME:-unset} -- the isolation step did not run, so this job shares a toolchain with every other runner slot on this host and a concurrent install can replace a binary mid-run" >&2; exit 1; fi if: "!cancelled()" - name: Nominal witnesses (one prepared subject, one fold) id: required_ci_measure - run: | + run: |+ GUNBC_FLOOR_LOG='gunbc-floor-cmd.log' 'set' '+e' 'set' '-o' 'pipefail' @@ -335,9 +341,10 @@ jobs: if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' && '[' '-f' "$GUNBC_FLOOR_LOG" ']' && 'grep' '-q' 'Resource temporarily unavailable' "$GUNBC_FLOOR_LOG"; then GUNBC_FLOOR_CLASS='infra'; GUNBC_FLOOR_SIGNATURE='ResourceTemporarilyUnavailable'; fi if (! '[' '-f' "$GUNBC_FLOOR_RECEIPT" ']') || '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' || ('[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT"))) || ('[' "$GUNBC_FLOOR_CLASS" '=' 'none' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT")) && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=infra' "$GUNBC_FLOOR_RECEIPT"))); then 'printf' 'class=%s\nsignature=%s\nexit=%s\n' "$GUNBC_FLOOR_CLASS" "$GUNBC_FLOOR_SIGNATURE" "$GUNBC_FLOOR_EXIT" > "$GUNBC_FLOOR_RECEIPT"; if '[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']'; then 'echo' '::error title=environment::floor_class='"$GUNBC_FLOOR_CLASS"' signature='"$GUNBC_FLOOR_SIGNATURE"' exit='"$GUNBC_FLOOR_EXIT"'; this is not a verdict about the diff. Attempt receipt: '"$GUNBC_FLOOR_RECEIPT"; fi; if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']'; then 'echo' '::error title=subject::floor_class='"$GUNBC_FLOOR_CLASS"' exit='"$GUNBC_FLOOR_EXIT"'; read the step log for the subject defect'; fi; fi 'exit' "$GUNBC_FLOOR_EXIT" + if: "!cancelled() && steps.build_witness_fold.outcome == 'success'" - name: "D0-MEASURE: seal an unreached receipt when the instrument did not return" - run: | + run: |+ GUNBC_FLOOR_LOG='gunbc-floor-cmd.log' 'set' '+e' 'set' '-o' 'pipefail' @@ -359,6 +366,7 @@ jobs: if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' && '[' '-f' "$GUNBC_FLOOR_LOG" ']' && 'grep' '-q' 'Resource temporarily unavailable' "$GUNBC_FLOOR_LOG"; then GUNBC_FLOOR_CLASS='infra'; GUNBC_FLOOR_SIGNATURE='ResourceTemporarilyUnavailable'; fi if (! '[' '-f' "$GUNBC_FLOOR_RECEIPT" ']') || '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' || ('[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT"))) || ('[' "$GUNBC_FLOOR_CLASS" '=' 'none' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT")) && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=infra' "$GUNBC_FLOOR_RECEIPT"))); then 'printf' 'class=%s\nsignature=%s\nexit=%s\n' "$GUNBC_FLOOR_CLASS" "$GUNBC_FLOOR_SIGNATURE" "$GUNBC_FLOOR_EXIT" > "$GUNBC_FLOOR_RECEIPT"; if '[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']'; then 'echo' '::error title=environment::floor_class='"$GUNBC_FLOOR_CLASS"' signature='"$GUNBC_FLOOR_SIGNATURE"' exit='"$GUNBC_FLOOR_EXIT"'; this is not a verdict about the diff. Attempt receipt: '"$GUNBC_FLOOR_RECEIPT"; fi; if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']'; then 'echo' '::error title=subject::floor_class='"$GUNBC_FLOOR_CLASS"' exit='"$GUNBC_FLOOR_EXIT"'; read the step log for the subject defect'; fi; fi 'exit' "$GUNBC_FLOOR_EXIT" + env: REQUIRED_CI_INSTRUMENT_BUILD_OUTCOME: ${{ steps.build_witness_fold.outcome }} if: always() && !cancelled() @@ -371,7 +379,7 @@ jobs: retention-days: 14 if: always() && !cancelled() - name: "D0-ADJUDICATE: consume the published measurement receipt" - run: | + run: |+ GUNBC_FLOOR_LOG='gunbc-floor-cmd.log' 'set' '+e' 'set' '-o' 'pipefail' @@ -394,6 +402,7 @@ jobs: if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' && '[' '-f' "$GUNBC_FLOOR_LOG" ']' && 'grep' '-q' 'Resource temporarily unavailable' "$GUNBC_FLOOR_LOG"; then GUNBC_FLOOR_CLASS='infra'; GUNBC_FLOOR_SIGNATURE='ResourceTemporarilyUnavailable'; fi if (! '[' '-f' "$GUNBC_FLOOR_RECEIPT" ']') || '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']' || ('[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT"))) || ('[' "$GUNBC_FLOOR_CLASS" '=' 'none' ']' && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=structural' "$GUNBC_FLOOR_RECEIPT")) && (! ('[' '-f' "$GUNBC_FLOOR_RECEIPT" ']' && 'grep' '-q' 'class=infra' "$GUNBC_FLOOR_RECEIPT"))); then 'printf' 'class=%s\nsignature=%s\nexit=%s\n' "$GUNBC_FLOOR_CLASS" "$GUNBC_FLOOR_SIGNATURE" "$GUNBC_FLOOR_EXIT" > "$GUNBC_FLOOR_RECEIPT"; if '[' "$GUNBC_FLOOR_CLASS" '=' 'infra' ']'; then 'echo' '::error title=environment::floor_class='"$GUNBC_FLOOR_CLASS"' signature='"$GUNBC_FLOOR_SIGNATURE"' exit='"$GUNBC_FLOOR_EXIT"'; this is not a verdict about the diff. Attempt receipt: '"$GUNBC_FLOOR_RECEIPT"; fi; if '[' "$GUNBC_FLOOR_CLASS" '=' 'structural' ']'; then 'echo' '::error title=subject::floor_class='"$GUNBC_FLOOR_CLASS"' exit='"$GUNBC_FLOOR_EXIT"'; read the step log for the subject defect'; fi; fi 'exit' "$GUNBC_FLOOR_EXIT" + env: REQUIRED_CI_INSTRUMENT_BUILD_OUTCOME: ${{ steps.build_witness_fold.outcome }} if: always() && !cancelled() @@ -405,7 +414,7 @@ jobs: if: "!cancelled() && steps.build_witness_fold.outcome == 'success'" - name: Declare which ledger rows this repair covers id: repair_declaration - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/instruments/generated_artifact_gate.dag --function heal_repair_declaration if: failure() && steps.generated_artifact_gate.outcome == 'failure' && github.actor != 'dependabot[bot]' @@ -417,7 +426,7 @@ jobs: if: failure() && steps.generated_artifact_gate.outcome == 'failure' && github.actor != 'dependabot[bot]' && steps.repair_declaration.outcome == 'success' - name: Produce the sealed repair candidate id: repair_candidate - run: | + run: |- ROOT=$(git rev-parse --show-toplevel 2>/dev/null || pwd) "$ROOT/target/release/gunbc" run --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --entry dag/gunbc/heal_candidate.dag --function heal_candidate_produce_wet env: @@ -436,7 +445,7 @@ jobs: retention-days: 3 if: failure() && steps.generated_artifact_gate.outcome == 'failure' && github.actor != 'dependabot[bot]' && steps.repair_declaration.outcome == 'success' && steps.repair_regen.outcome == 'success' && steps.repair_candidate.outcome == 'success' env: - MALLOC_ARENA_MAX: 2 + MALLOC_ARENA_MAX: "2" GUNBC_EXPECTED_RED_ROSTER_JOIN: expected_red_roster_join.tsv GUNBC_REQUIRED_FLOOR_DISPOSITION: required_floor_disposition.tsv GUNBC_LONG_HOME_STORAGE_AGREEMENT: long_home_storage_agreement.tsv @@ -451,7 +460,7 @@ jobs: contents: read steps: - name: Every required lane must have succeeded - run: | + run: |- echo "required lanes: compiler=$COMPILER clippy=$CLIPPY floor=$FLOOR (same_repo=$SAME_REPO)" if [ "$COMPILER" = failure ] || [ "$CLIPPY" = failure ] || [ "$FLOOR" = failure ]; then echo "::error::a required lane concluded failure (compiler=$COMPILER clippy=$CLIPPY floor=$FLOOR) - open that job's log" >&2; exit 1; fi if [ "$COMPILER" != success ] || [ "$CLIPPY" != success ]; then echo "::error::a required lane produced no conclusion of its own, so this head is unobserved rather than failed (compiler=$COMPILER clippy=$CLIPPY)" >&2; exit 1; fi diff --git a/dag/extdeps/languages/yaml/emit.dag b/dag/extdeps/languages/yaml/emit.dag index 900c44d631e..782fa9bf7c7 100644 --- a/dag/extdeps/languages/yaml/emit.dag +++ b/dag/extdeps/languages/yaml/emit.dag @@ -2,10 +2,16 @@ module extdeps.languages.yaml.emit import extdeps.external_authority { ExternalAuthority } import extdeps.uri { Uri, Https } +import std.algebra { trim } import extdeps.languages.yaml.types { YamlValue, YamlKeyValue, YamlNull, YamlBool, YamlInt, YamlFloat, YamlString, YamlSequence, YamlMapping } +import extdeps.languages.yaml.ingest { + yaml_string_reads_back_plain, yaml_string_reads_back_as_plain_key, yaml_resolve_plain, + yaml_contains_refused_character, yaml_first_refused_character, yaml_refused_character_reason, + yaml_line_containing, yaml_source_lines +} import std.layout { Doc, DocLine, DocConcat, LayoutProtocol, @@ -20,69 +26,129 @@ data extdeps_external_authority_anchor: ExternalAuthority = ExternalAuthority { } } +// THE WRITER FOR THE SUBSET extdeps.languages.yaml.ingest READS. Its contract is the reader's, +// turned around: for every value it emits, ingest_yaml_source(emitted) == IngestedYaml(value), and +// a value it cannot write with that meaning is refused with the path to it, never written as +// something else. Which strings go plain is decided by the reader's own predicates +// (yaml_string_reads_back_plain, yaml_string_reads_back_as_plain_key), so the two cannot disagree +// about what plain text means. + +type YamlEmitResult + = EmittedYaml { text: String } + | YamlEmitRefused { path: String, reason: String } + data yaml_protocol: LayoutProtocol = LayoutProtocol { indent_unit: " " } -fn is_expression(s: String) -> Bool { - starts_with(s: s, prefix: "${{") +// --------------------------------------------------------------------------------------------- +// WHAT CAN BE WRITTEN +// --------------------------------------------------------------------------------------------- + +fn yaml_path_key(path: String, key: String) -> String { + if path == "" { key } else { join([path, ".", key], "") } } -fn is_reserved_yaml_scalar(s: String) -> Bool { - (s == "true") || - (s == "false") || - (s == "null") || - (s == "yes") || - (s == "no") || - (s == "~") +fn yaml_lexeme_reads_back(v: YamlValue) -> Bool { + match v { + YamlInt { lexeme: l } => match yaml_resolve_plain(s: l) { YamlInt { lexeme: _ } => true _ => false } + YamlFloat { lexeme: l } => match yaml_resolve_plain(s: l) { YamlFloat { lexeme: _ } => true _ => false } + _ => true + } } -fn starts_with_yaml_indicator(s: String) -> Bool { - starts_with(s: s, prefix: "- ") || - starts_with(s: s, prefix: "?") || - starts_with(s: s, prefix: ":") || - starts_with(s: s, prefix: ",") || - starts_with(s: s, prefix: "[") || - starts_with(s: s, prefix: "]") || - starts_with(s: s, prefix: "{") || - starts_with(s: s, prefix: "}") || - starts_with(s: s, prefix: "#") || - starts_with(s: s, prefix: "&") || - starts_with(s: s, prefix: "*") || - starts_with(s: s, prefix: "!") || - starts_with(s: s, prefix: "|") || - starts_with(s: s, prefix: ">") || - starts_with(s: s, prefix: "'") || - starts_with(s: s, prefix: "\"") || - starts_with(s: s, prefix: "%") || - starts_with(s: s, prefix: "@") || - starts_with(s: s, prefix: "`") +// WHETHER A MULTI-LINE STRING CAN BE WRITTEN AS A LITERAL BLOCK THE READER TAKES BACK UNCHANGED. +// The literal detects its indentation from its first non-empty line, so that line may not begin +// with a space; the reader refuses a line that ends in whitespace or whose leading whitespace holds +// a tab; and a string of line breaks alone has no content line to carry them. Every other string +// is written double-quoted with its line breaks escaped, which the reader also takes back unchanged. +fn yaml_literal_block_holds(s: String) -> Bool { + let first = fold(split(s: s, delimiter: "\n"), init: "", f: (acc, l) => if acc == "" { l } else { acc }) + if (replace(s, "\n", "") == "") || string_contains(s: s, pattern: "\t") || starts_with(s: first, prefix: " ") { + false + } else { + !string_contains(s: s, pattern: " \n") && !ends_with(s: s, suffix: " ") + } } -fn needs_double_quotes(s: String) -> Bool { - is_reserved_yaml_scalar(s: s) || - starts_with_yaml_indicator(s: s) || - string_contains(s: s, pattern: ": ") || - string_contains(s: s, pattern: " #") +type YamlEmitCheck + = YamlEmitAdmitted + | YamlEmitRefusedAt { path: String, reason: String } + +fn yaml_emit_check(v: YamlValue, path: String) -> YamlEmitCheck { + match v { + YamlInt { lexeme: l } => + if yaml_lexeme_reads_back(v: v) { YamlEmitAdmitted } else { YamlEmitRefusedAt { path: path, reason: join(["the int lexeme `", l, "` does not read back as an int under the core schema"], "") } } + YamlFloat { lexeme: l } => + if yaml_lexeme_reads_back(v: v) { YamlEmitAdmitted } else { YamlEmitRefusedAt { path: path, reason: join(["the float lexeme `", l, "` does not read back as a float under the core schema"], "") } } + YamlSequence { elements: els } => + yaml_emit_check_elements(elements: els, index: 0, path: path) + YamlMapping { entries: ents } => + yaml_emit_check_entries(entries: ents, index: 0, path: path, seen: empty_map()) + _ => YamlEmitAdmitted + } } -fn escape_double_quoted(s: String) -> String { - replace(replace(s, "\\", "\\\\"), "\"", "\\\"") +fn yaml_emit_check_elements(elements: List, index: Int, path: String) -> YamlEmitCheck { + match get(xs: elements, index: index) { + Absent => YamlEmitAdmitted + Present { value: e } => + match yaml_emit_check(v: e, path: join([path, "[", to_string(index), "]"], "")) { + YamlEmitAdmitted => yaml_emit_check_elements(elements: elements, index: index + 1, path: path) + YamlEmitRefusedAt { path: p, reason: r } => YamlEmitRefusedAt { path: p, reason: r } + } + } } +fn yaml_emit_check_entries(entries: List, index: Int, path: String, seen: Map) -> YamlEmitCheck { + match get(xs: entries, index: index) { + Absent => YamlEmitAdmitted + Present { value: e } => + if map_contains_key(seen, e.key) { + YamlEmitRefusedAt { path: yaml_path_key(path: path, key: e.key), reason: "a mapping holds this key twice, and YAML requires mapping keys to be unique" } + } else { + match yaml_emit_check(v: e.value, path: yaml_path_key(path: path, key: e.key)) { + YamlEmitAdmitted => yaml_emit_check_entries(entries: entries, index: index + 1, path: path, seen: map_insert(seen, e.key, true)) + YamlEmitRefusedAt { path: p, reason: r } => YamlEmitRefusedAt { path: p, reason: r } + } + } + } +} + +// --------------------------------------------------------------------------------------------- +// SCALARS +// --------------------------------------------------------------------------------------------- + +fn yaml_double_quoted_text(s: String) -> String { + concat(concat("\"", replace(replace(replace(s, "\\", "\\\\"), "\"", "\\\""), "\n", "\\n")), "\"") +} + +// A single-line string as a value in block context: plain when the reader reads it back plain, +// `''` when empty, double-quoted otherwise. fn emit_inline_scalar_text(s: String) -> String { if s == "" { "''" - } else if is_expression(s: s) { + } else if yaml_string_reads_back_plain(s: s) { s - } else if needs_double_quotes(s: s) { - concat(concat("\"", escape_double_quoted(s: s)), "\"") } else { + yaml_double_quoted_text(s: s) + } +} + +fn emit_key_text(k: String) -> String { + if yaml_string_reads_back_as_plain_key(k: k) { k } else { yaml_double_quoted_text(s: k) } +} + +// In a flow sequence a plain entry may not contain a flow indicator either (section 7.3.3). +fn emit_flow_scalar_text(s: String) -> String { + if (s != "") && yaml_string_reads_back_plain(s: s) && !string_contains(s: s, pattern: ",") && !string_contains(s: s, pattern: "[") && !string_contains(s: s, pattern: "]") && !string_contains(s: s, pattern: "{") && !string_contains(s: s, pattern: "}") { s + } else { + yaml_double_quoted_text(s: s) } } fn emit_scalar(v: YamlValue) -> String { match v { - YamlNull => "" + YamlNull => "null" YamlBool { value: b } => if b { "true" } else { "false" } YamlInt { lexeme: l } => l YamlFloat { lexeme: l } => l @@ -92,61 +158,78 @@ fn emit_scalar(v: YamlValue) -> String { } } +fn emit_flow_entry(v: YamlValue) -> String { + match v { + YamlString { value: s } => emit_flow_scalar_text(s: s) + _ => emit_scalar(v: v) + } +} + fn all_scalars_flowable(elements: List) -> Bool { - fold(elements, init: true, f: (acc, e) => acc && match e { + all(elements, e => match e { YamlString { value: s } => !string_contains(s: s, pattern: "\n") YamlInt { lexeme: _ } => true + YamlFloat { lexeme: _ } => true YamlBool { value: _ } => true + YamlNull => true _ => false }) } fn emit_flow_sequence(elements: List) -> String { - concat( - concat( - "[", - fold(elements, init: "", f: (acc, e) => { - let part = emit_scalar(v: e) - if acc == "" { part } else { concat(concat(acc, ", "), part) } - }) - ), - "]" - ) + join(["[", join(map(elements, e => emit_flow_entry(v: e)), ", "), "]"], "") } +// --------------------------------------------------------------------------------------------- +// BLOCKS +// --------------------------------------------------------------------------------------------- + fn nested_body_doc(body: Doc) -> Doc { doc_nest(levels: 1, body: doc_concat(parts: [DocLine, body])) } +// A LITERAL BLOCK SCALAR CARRIES ITS FINAL LINE BREAKS IN ITS CHOMPING INDICATOR (section 8.1.1.2): +// `|-` for none, `|` for exactly one, `|+` for more. The body is every line of the string before +// the chomped breaks, empty lines included; an empty line renders as a bare newline. fn multiline_block_doc(prefix: String, s: String) -> Doc { - let lines = filter(split(s: s, delimiter: "\n"), l => l != "") - let content = fold(lines, init: [], f: (acc, l) => append(acc, items: [DocLine, doc_text(text: l)])) + let header = if ends_with(s: s, suffix: "\n\n") { "|+" } else if ends_with(s: s, suffix: "\n") { "|" } else { "|-" } + let body = if ends_with(s: s, suffix: "\n") { substring(s: s, start: 0, end: s.length() - 1) } else { s } + let content = fold(split(s: body, delimiter: "\n"), init: [], f: (acc, l) => append(acc, items: [DocLine, doc_text(text: l)])) doc_concat(parts: [ - doc_text(text: concat(prefix, "|")), + doc_text(text: concat(prefix, header)), doc_nest(levels: 1, body: doc_concat(parts: content)), DocLine ]) } +// A string in block context after `prefix`: a literal block when the reader takes it back from +// one, otherwise one line (plain, `''` or double-quoted). +fn string_doc(prefix: String, s: String) -> Doc { + if string_contains(s: s, pattern: "\n") && yaml_literal_block_holds(s: s) { + multiline_block_doc(prefix: prefix, s: s) + } else { + doc_line_of(text: concat(prefix, emit_inline_scalar_text(s: s))) + } +} + fn entry_doc(entry: YamlKeyValue) -> Doc { - let key = entry.key + let key = emit_key_text(k: entry.key) let val = entry.value match val { YamlNull => doc_line_of(text: concat(key, ":")) YamlMapping { entries: nested } => - doc_concat(parts: [doc_text(text: concat(key, ":")), nested_body_doc(body: mapping_body_doc(entries: nested))]) + if (nested |> count) == 0 { + doc_line_of(text: concat(key, ": {}")) + } else { + doc_concat(parts: [doc_text(text: concat(key, ":")), nested_body_doc(body: mapping_body_doc(entries: nested))]) + } YamlSequence { elements: els } => if all_scalars_flowable(elements: els) { doc_line_of(text: concat(concat(key, ": "), emit_flow_sequence(elements: els))) } else { doc_concat(parts: [doc_text(text: concat(key, ":")), nested_body_doc(body: block_sequence_doc(elements: els))]) } - YamlString { value: s } => - if string_contains(s: s, pattern: "\n") { - multiline_block_doc(prefix: concat(key, ": "), s: s) - } else { - doc_line_of(text: concat(concat(key, ": "), emit_inline_scalar_text(s: s))) - } + YamlString { value: s } => string_doc(prefix: concat(key, ": "), s: s) _ => doc_line_of(text: concat(concat(key, ": "), emit_scalar(v: val))) } } @@ -158,18 +241,19 @@ fn mapping_body_doc(entries: List) -> Doc { fn block_sequence_element_doc(e: YamlValue) -> Doc { match e { YamlMapping { entries: ents } => - doc_concat(parts: [doc_text(text: "- "), doc_nest(levels: 1, body: mapping_body_doc(entries: ents))]) - YamlString { value: s } => - if string_contains(s: s, pattern: "\n") { - multiline_block_doc(prefix: "- ", s: s) + if (ents |> count) == 0 { + doc_line_of(text: "- {}") + } else { + doc_concat(parts: [doc_text(text: "- "), doc_nest(levels: 1, body: mapping_body_doc(entries: ents))]) + } + YamlSequence { elements: els } => + if all_scalars_flowable(elements: els) { + doc_line_of(text: concat("- ", emit_flow_sequence(elements: els))) } else { - doc_line_of(text: concat("- ", emit_inline_scalar_text(s: s))) + doc_concat(parts: [doc_text(text: "- "), doc_nest(levels: 1, body: block_sequence_doc(elements: els))]) } - YamlNull => doc_line_of(text: concat("- ", emit_scalar(v: e))) - YamlBool { value: _ } => doc_line_of(text: concat("- ", emit_scalar(v: e))) - YamlInt { lexeme: _ } => doc_line_of(text: concat("- ", emit_scalar(v: e))) - YamlFloat { lexeme: _ } => doc_line_of(text: concat("- ", emit_scalar(v: e))) - YamlSequence { elements: _ } => doc_line_of(text: concat("- ", emit_scalar(v: e))) + YamlString { value: s } => string_doc(prefix: "- ", s: s) + _ => doc_line_of(text: concat("- ", emit_scalar(v: e))) } } @@ -177,28 +261,52 @@ fn block_sequence_doc(elements: List) -> Doc { doc_concat(parts: map(elements, e => block_sequence_element_doc(e: e))) } -fn project_yaml_to_doc(v: YamlValue) -> Doc { +// --------------------------------------------------------------------------------------------- +// DOCUMENT +// --------------------------------------------------------------------------------------------- + +// The reader takes a document whose root is a non-empty block mapping or block sequence, so that +// is what the writer writes. +fn yaml_root_doc_refusal(v: YamlValue) -> String { + match v { + YamlMapping { entries: ents } => if (ents |> count) == 0 { "an empty mapping has no block form, and a document root must be a block collection" } else { "" } + YamlSequence { elements: els } => if (els |> count) == 0 { "an empty sequence has no block form, and a document root must be a block collection" } else { "" } + _ => "a document root must be a block mapping or a block sequence" + } +} + +fn yaml_root_doc(v: YamlValue) -> Doc { match v { YamlMapping { entries: ents } => mapping_body_doc(entries: ents) - YamlSequence { elements: els } => - if all_scalars_flowable(elements: els) { - doc_line_of(text: emit_flow_sequence(elements: els)) - } else { - block_sequence_doc(elements: els) - } - YamlString { value: s } => - if string_contains(s: s, pattern: "\n") { - multiline_block_doc(prefix: "", s: s) - } else { - doc_line_of(text: emit_inline_scalar_text(s: s)) - } - YamlNull => doc_line_of(text: emit_scalar(v: v)) - YamlBool { value: _ } => doc_line_of(text: emit_scalar(v: v)) - YamlInt { lexeme: _ } => doc_line_of(text: emit_scalar(v: v)) - YamlFloat { lexeme: _ } => doc_line_of(text: emit_scalar(v: v)) + YamlSequence { elements: els } => block_sequence_doc(elements: els) + _ => doc_concat(parts: []) + } +} + +fn emit_yaml(v: YamlValue) -> YamlEmitResult { + let root = yaml_root_doc_refusal(v: v) + if root != "" { + YamlEmitRefused { path: "", reason: root } + } else { + match yaml_emit_check(v: v, path: "") { + YamlEmitRefusedAt { path: p, reason: r } => YamlEmitRefused { path: p, reason: r } + YamlEmitAdmitted => yaml_emitted_text(text: render(doc: yaml_root_doc(v: v), proto: yaml_protocol)) + } + } +} + +fn yaml_emitted_text(text: String) -> YamlEmitResult { + if yaml_contains_refused_character(text: text) { + let cp = yaml_first_refused_character(text: text) + YamlEmitRefused { + path: join(["line ", to_string(yaml_line_containing(lines: yaml_source_lines(src: text), pattern: from_code_point(cp: cp), index: 0)), " of the emitted text"], ""), + reason: yaml_refused_character_reason(cp: cp), + } + } else { + EmittedYaml { text: text } } } -fn serialize_yaml(v: YamlValue) -> String { - render(doc: project_yaml_to_doc(v: v), proto: yaml_protocol) +fn yaml_emit_refusal_text(module_path: String, path: String, reason: String) -> String { + join(["REFUSED ", module_path, ": the YAML writer cannot write the value at `", path, "` with its meaning: ", reason], "") } diff --git a/dag/extdeps/languages/yaml/ingest.dag b/dag/extdeps/languages/yaml/ingest.dag index 8a4648a8dbd..06f6f332658 100644 --- a/dag/extdeps/languages/yaml/ingest.dag +++ b/dag/extdeps/languages/yaml/ingest.dag @@ -3,10 +3,10 @@ module extdeps.languages.yaml.ingest import extdeps.external_authority { ExternalAuthority } import extdeps.uri { Uri, Https } import v2.std.optional { Present, Absent } +import std.algebra { trim } import extdeps.languages.yaml.types { YamlValue, YamlKeyValue, - YamlNull, YamlBool, YamlInt, YamlFloat, YamlString, YamlSequence, YamlMapping, - kv + YamlNull, YamlBool, YamlInt, YamlFloat, YamlString, YamlSequence, YamlMapping } data extdeps_external_authority_anchor: ExternalAuthority = ExternalAuthority { @@ -16,61 +16,529 @@ data extdeps_external_authority_anchor: ExternalAuthority = ExternalAuthority { } } +// A BOUNDED SUBSET OF YAML 1.2.2, READ WITH ITS CORRECT MEANING OR REFUSED AT A LINE. Every document +// this module accepts denotes, under the YAML 1.2.2 core schema, exactly the YamlValue it returns; +// every construct outside the subset is refused with the 1-based line it sits on. Nothing outside +// the subset is reinterpreted as a string and nothing is dropped. The subset, construct by construct: +// +// ACCEPTED +// - one document whose root is a block mapping or a block sequence starting at column 0 +// - block mappings with plain keys, or single- or double-quoted keys; plain keys must resolve to a +// string under the core schema and may not start with an indicator or a digit; duplicate keys refuse +// - block sequences, including compact `- key: value` items and compact nested `- - x` items, and a +// sequence under a key at the key's own indentation +// - any indentation that increases, spaces only +// - plain scalars, resolved by the core schema (section 10.3.2): null, bool, int (decimal, 0o, 0x), +// float (decimal, exponent, .inf, .nan), otherwise string +// - single-quoted scalars (`''` is a quote) and double-quoted scalars with every escape section 5.7 +// names, each on one line with nothing after the closing quote +// - flow sequences of scalars on one line, and the empty flow mapping `{}` +// - literal block scalars `|`, `|-`, `|+` with auto-detected indentation +// - blank lines and full-line comments between nodes +// +// REFUSED, each with its own reason +// - folded block scalars (`>`), block scalar indentation indicators, anchors, aliases, tags, +// directives, document markers, explicit `?` keys, flow mappings other than `{}`, nested flow +// collections, multi-line flow collections, multi-line plain or quoted scalars, and a comment +// after a value on the same line +// - tabs in indentation, CR line breaks, a line that ends in whitespace, and every character YAML +// does not allow in a stream or that this reader cannot tell from whitespace +// (yaml_refused_code_points) + type YamlIngestResult = IngestedYaml { value: YamlValue } - | YamlIngestRejected { reason: String } + | YamlIngestRejected { line: Int, reason: String } -type YamlBlockResult - = ParsedYamlBlock { value: YamlValue, next_index: Int } - | YamlBlockRejected { reason: String } +type YamlScalarResult + = ReadYamlScalar { value: YamlValue } + | YamlScalarRefused { reason: String } + +// --------------------------------------------------------------------------------------------- +// CHARACTER REPERTOIRE +// --------------------------------------------------------------------------------------------- + +// Code points refused anywhere in a document. C0 controls other than tab and line feed, DEL, and +// the C1 block are outside YAML's printable set (section 5.1), and CR is a line break this reader +// does not model. The rest -- NEL, no-break space, the Unicode space separators, the line and +// paragraph separators, the byte-order mark and the two noncharacters -- are refused because the +// host's trim treats them as whitespace (or, for the BOM, because it is only legal at the start of +// a stream), so a reader that accepted them could not tell their content from indentation. +data yaml_refused_code_points: List = [ + 0, 1, 2, 3, 4, 5, 6, 7, 8, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, + 28, 29, 30, 31, 127, 128, 129, 130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 141, 142, + 143, 144, 145, 146, 147, 148, 149, 150, 151, 152, 153, 154, 155, 156, 157, 158, 159, 160, 5760, + 8192, 8193, 8194, 8195, 8196, 8197, 8198, 8199, 8200, 8201, 8202, 8232, 8233, 8239, 8287, 12288, + 65279, 65534, 65535 +] + +data yaml_refused_characters: List = map(yaml_refused_code_points, cp => from_code_point(cp: cp)) + +fn yaml_first_refused_character(text: String) -> Int { + fold(yaml_refused_code_points, init: -1, f: (acc, cp) => if (acc == -1) && string_contains(s: text, pattern: from_code_point(cp: cp)) { cp } else { acc }) +} + +fn yaml_contains_refused_character(text: String) -> Bool { + any(yaml_refused_characters, c => string_contains(s: text, pattern: c)) +} + +fn yaml_line_containing(lines: List, pattern: String, index: Int) -> Int { + match get(xs: lines, index: index) { + Absent => 0 + Present { value: raw } => if string_contains(s: raw, pattern: pattern) { index + 1 } else { yaml_line_containing(lines: lines, pattern: pattern, index: index + 1) } + } +} + +fn yaml_refused_character_reason(cp: Int) -> String { + join(["the character U+", yaml_hex_text(n: cp, width: 4), " is outside the supported repertoire (YAML's printable set, less the characters this reader cannot tell from whitespace, and less CR line breaks)"], "") +} + +data yaml_hex_digits: String = "0123456789ABCDEF" + +fn yaml_hex_text(n: Int, width: Int) -> String { + if (n < 16) && (width <= 1) { + substring(s: yaml_hex_digits, start: n, end: n + 1) + } else { + concat(yaml_hex_text(n: n / 16, width: width - 1), substring(s: yaml_hex_digits, start: n % 16, end: (n % 16) + 1)) + } +} + +// --------------------------------------------------------------------------------------------- +// PLAIN SCALARS AND THE CORE SCHEMA +// --------------------------------------------------------------------------------------------- + +// Section 10.3.2: the exact spellings the core schema resolves to null, bool and the special floats. +data yaml_core_schema_words: Map = { + "null": YamlNull, "Null": YamlNull, "NULL": YamlNull, "~": YamlNull, + "true": YamlBool { value: true }, "True": YamlBool { value: true }, "TRUE": YamlBool { value: true }, + "false": YamlBool { value: false }, "False": YamlBool { value: false }, "FALSE": YamlBool { value: false }, + ".inf": YamlFloat { lexeme: ".inf" }, ".Inf": YamlFloat { lexeme: ".Inf" }, ".INF": YamlFloat { lexeme: ".INF" }, + "+.inf": YamlFloat { lexeme: "+.inf" }, "+.Inf": YamlFloat { lexeme: "+.Inf" }, "+.INF": YamlFloat { lexeme: "+.INF" }, + "-.inf": YamlFloat { lexeme: "-.inf" }, "-.Inf": YamlFloat { lexeme: "-.Inf" }, "-.INF": YamlFloat { lexeme: "-.INF" }, + ".nan": YamlFloat { lexeme: ".nan" }, ".NaN": YamlFloat { lexeme: ".NaN" }, ".NAN": YamlFloat { lexeme: ".NAN" } +} -type YamlEntryResult - = ParsedYamlEntry { entry: YamlKeyValue, next_index: Int } - | YamlEntryRejected { reason: String } +data yaml_decimal_digits: String = "0123456789" +data yaml_numeric_starts: String = "0123456789+-." -type YamlSeqItemResult - = ParsedYamlSeqItem { value: YamlValue, next_index: Int } - | YamlSeqItemRejected { reason: String } +fn yaml_run_end(s: String, i: Int, alphabet: String) -> Int { + if i >= s.length() { + i + } else if string_contains(s: alphabet, pattern: char_at(s, i)) { + yaml_run_end(s: s, i: i + 1, alphabet: alphabet) + } else { + i + } +} -type YamlLiteralCollectResult { - text: String - next_index: Int +fn yaml_char_is(s: String, i: Int, alphabet: String) -> Bool { + if i >= s.length() { false } else { string_contains(s: alphabet, pattern: char_at(s, i)) } } -data yaml_digit_chars: String = "0123456789" +fn yaml_sign_width(s: String) -> Int { + if starts_with(s: s, prefix: "-") || starts_with(s: s, prefix: "+") { 1 } else { 0 } +} -// STOPS AT THE FIRST NON-DIGIT, BY BRANCHING. Written as `digit(head) && yaml_all_digits(tail)` it -// recursed to the end of every plain value, because the interpreter evaluates both operands of `&&` -// before dispatching it -- the operand-demand divergence already recorded as -// gunbc.recurring_failure_mode.realization_arms_diverge_on_whether_the_program_refuses. The explicit -// branch is this function's own early exit; it does not resolve that shared defect. -fn yaml_all_digits(s: String) -> Bool { - if s == "" { +fn yaml_is_core_int(s: String) -> Bool { + let sign = yaml_sign_width(s: s) + let decimal_end = yaml_run_end(s: s, i: sign, alphabet: yaml_decimal_digits) + if (decimal_end == s.length()) && (decimal_end > sign) { true - } else if !string_contains(s: yaml_digit_chars, pattern: substring(s: s, start: 0, end: 1)) { + } else if starts_with(s: s, prefix: "0o") && (s.length() > 2) { + yaml_run_end(s: s, i: 2, alphabet: "01234567") == s.length() + } else if starts_with(s: s, prefix: "0x") && (s.length() > 2) { + yaml_run_end(s: s, i: 2, alphabet: "0123456789abcdefABCDEF") == s.length() + } else { + false + } +} + +// [-+]? ( \. [0-9]+ | [0-9]+ ( \. [0-9]* )? ) ( [eE] [-+]? [0-9]+ )? +fn yaml_is_core_float(s: String) -> Bool { + let sign = yaml_sign_width(s: s) + let whole_end = yaml_run_end(s: s, i: sign, alphabet: yaml_decimal_digits) + let has_dot = yaml_char_is(s: s, i: whole_end, alphabet: ".") + let fraction_end = if has_dot { yaml_run_end(s: s, i: whole_end + 1, alphabet: yaml_decimal_digits) } else { whole_end } + let has_mantissa = (whole_end > sign) || (fraction_end > whole_end + 1) + if !has_mantissa { false + } else if yaml_char_is(s: s, i: fraction_end, alphabet: "eE") { + let exponent_digits = if yaml_char_is(s: s, i: fraction_end + 1, alphabet: "+-") { fraction_end + 2 } else { fraction_end + 1 } + let exponent_end = yaml_run_end(s: s, i: exponent_digits, alphabet: yaml_decimal_digits) + (exponent_end > exponent_digits) && (exponent_end == s.length()) + } else { + fraction_end == s.length() + } +} + +// The host's i64 parse accepts exactly [-+]?[0-9]+, the core schema's decimal int, so a value it +// reads is an int in one step; anything it refuses (an overflow, 0o, 0x, a float, a string) is +// decided by the full grammar below. +fn yaml_resolve_numeric(s: String) -> YamlValue { + match parse_int(s: s) { + Present { value: _ } => YamlInt { lexeme: s } + Absent => yaml_resolve_numeric_grammar(s: s) + } +} + +// THE WORD MAP IS STILL THE AUTHORITY FOR `.inf` AND `.nan`: their spellings start with a numeric +// character, so a reader that took the numeric route and never consulted it would read `.NaN` as a +// string -- which is what this reader exists not to do. +fn yaml_resolve_numeric_grammar(s: String) -> YamlValue { + if yaml_is_core_int(s: s) { + YamlInt { lexeme: s } + } else if yaml_is_core_float(s: s) { + YamlFloat { lexeme: s } + } else { + match map_get(yaml_core_schema_words, s) { + Present { value: v } => v + Absent => YamlString { value: s } + } + } +} + +// THE CORE SCHEMA'S TAG RESOLUTION FOR A PLAIN SCALAR (section 10.3.2), for text already known to +// be a well-formed plain scalar. +fn yaml_resolve_plain(s: String) -> YamlValue { + match map_get(yaml_core_schema_words, s) { + Present { value: v } => v + Absent => if string_contains(s: yaml_numeric_starts, pattern: substring(s: s, start: 0, end: 1)) { yaml_resolve_numeric(s: s) } else { YamlString { value: s } } + } +} + +data yaml_inline_comment_reason: String = "inline YAML comments after values are not modeled" + +// Why `s` is not a well-formed single-line plain scalar in block context, or "" when it is. +// `s` is trimmed and non-empty and does not start with an indicator (the caller dispatched those). +// `: ` would make it a mapping and ` #` would start a comment, so neither can be content. +fn yaml_plain_refusal(s: String) -> String { + if !string_contains(s: s, pattern: ":") { + yaml_plain_comment_refusal(s: s) + } else if string_contains(s: s, pattern: ": ") || string_contains(s: s, pattern: ":\t") || ends_with(s: s, suffix: ":") { + "a plain scalar cannot contain `: ` or end with `:` (a nested mapping on one line is not YAML; quote the value)" + } else { + yaml_plain_comment_refusal(s: s) + } +} + +fn yaml_plain_comment_refusal(s: String) -> String { + if !string_contains(s: s, pattern: "#") { + "" + } else if string_contains(s: s, pattern: " #") || string_contains(s: s, pattern: "\t#") { + yaml_inline_comment_reason } else { - yaml_all_digits(s: substring(s: s, start: 1, end: s.length())) + "" } } -fn yaml_looks_like_int(s: String) -> Bool { - if s == "" { +// A plain KEY is further restricted: it may not start with any indicator or a numeric start, and +// it must resolve to a string, because YamlKeyValue.key is a string and the core schema would read +// `true`, `null` or `1` as something else. The emitter quotes every key this refuses. +data yaml_plain_key_refused_starts: String = "-?:,[]{}#&*!|>'\"%@`0123456789+.~ \t" + +fn yaml_plain_key_refusal(k: String) -> String { + if string_contains(s: yaml_plain_key_refused_starts, pattern: substring(s: k, start: 0, end: 1)) { + join(["the plain key `", k, "` is empty or starts with an indicator or a numeric character; quote it"], "") + } else if map_contains_key(yaml_core_schema_words, k) { + join(["the plain key `", k, "` resolves to a non-string under the core schema; quote it"], "") + } else { + yaml_plain_comment_refusal(s: k) + } +} + +// Emit and ingest share one reading of plain text: a string may be written plain exactly when +// reading it plain gives the same string back. +fn yaml_string_reads_back_plain(s: String) -> Bool { + if (s == "") || (trim(s) != s) || string_contains(s: s, pattern: "\n") { + false + } else if string_contains(s: yaml_value_special_starts, pattern: substring(s: s, start: 0, end: 1)) && !yaml_indicator_opens_plain(v: s) { + false + } else if yaml_plain_refusal(s: s) != "" { false - } else if starts_with(s: s, prefix: "-") { - let rest = substring(s: s, start: 1, end: s.length()) - (rest != "") && yaml_all_digits(s: rest) } else { - yaml_all_digits(s: s) + match yaml_resolve_plain(s: s) { + YamlString { value: _ } => true + _ => false + } + } +} + +fn yaml_string_reads_back_as_plain_key(k: String) -> Bool { + (trim(k) == k) && !string_contains(s: k, pattern: "\n") && (yaml_plain_key_refusal(k: k) == "") && !string_contains(s: k, pattern: ": ") && !ends_with(s: k, suffix: ":") +} + +// --------------------------------------------------------------------------------------------- +// QUOTED SCALARS +// --------------------------------------------------------------------------------------------- + +fn yaml_single_quoted(v: String) -> YamlScalarResult { + if (v.length() < 2) || !ends_with(s: v, suffix: "'") { + YamlScalarRefused { reason: "a single-quoted scalar must close on its own line with nothing after the closing quote (multi-line quoted scalars and trailing comments are outside the supported subset)" } + } else { + let inner = substring(s: v, start: 1, end: v.length() - 1) + if string_contains(s: replace(inner, "''", ""), pattern: "'") { + YamlScalarRefused { reason: "a lone `'` ends a single-quoted scalar before the end of the value (write `''` for a quote)" } + } else { + ReadYamlScalar { value: YamlString { value: replace(inner, "''", "'") } } + } + } +} + +// Section 5.7. `\\`, `\x`, `\u` and `\U` are handled by the decoder; every other escape is one +// character. +data yaml_double_quoted_escapes: Map = { + "0": 0, "a": 7, "b": 8, "t": 9, "\t": 9, "n": 10, "v": 11, "f": 12, "r": 13, "e": 27, + " ": 32, "\"": 34, "/": 47, "N": 133, "_": 160, "L": 8232, "P": 8233 +} + +data yaml_hex_values: Map = { + "0": 0, "1": 1, "2": 2, "3": 3, "4": 4, "5": 5, "6": 6, "7": 7, "8": 8, "9": 9, + "a": 10, "b": 11, "c": 12, "d": 13, "e": 14, "f": 15, + "A": 10, "B": 11, "C": 12, "D": 13, "E": 14, "F": 15 +} + +fn yaml_hex_value(s: String, i: Int, acc: Int) -> Int { + if i >= s.length() { + acc + } else { + match map_get(yaml_hex_values, char_at(s, i)) { + Present { value: d } => yaml_hex_value(s: s, i: i + 1, acc: (acc * 16) + d) + Absent => -1 + } } } -// COUNTED IN CHUNKS, NOT ONE RECURSIVE CALL PER SPACE. Recursing per space copied the remaining line -// at every leading space and ran several times per line, so indentation dominated the parse and put -// a census of one hand-authored workflow past a whole claim's budget (gunb-ai/gunbc#11587; the cost -// is re-derived by claim_batch eval_steps over ingest_yaml_source on a .github/workflows file). -// Consuming the largest run of 8, 4, 2 or 1 spaces per call gives the -// identical count (spaces only; a tab still ends the indentation) in a handful of calls. +data yaml_unescaped_quote_reason: String = "an unescaped `\"` ends a double-quoted scalar before the end of the value (multi-line quoted scalars and trailing comments are outside the supported subset)" + +fn yaml_double_quoted(v: String) -> YamlScalarResult { + if (v.length() < 2) || !ends_with(s: v, suffix: "\"") { + YamlScalarRefused { reason: "a double-quoted scalar must close on its own line with nothing after the closing quote (multi-line quoted scalars and trailing comments are outside the supported subset)" } + } else { + let inner = substring(s: v, start: 1, end: v.length() - 1) + if !string_contains(s: inner, pattern: "\\") { + if string_contains(s: inner, pattern: "\"") { + YamlScalarRefused { reason: yaml_unescaped_quote_reason } + } else { + ReadYamlScalar { value: YamlString { value: inner } } + } + } else { + yaml_double_quoted_pieces(pieces: split(s: inner, delimiter: "\\"), index: 0, escape: false, acc: []) + } + } +} + +// THE BODY SPLIT AT EVERY BACKSLASH. Piece 0 is literal. In escape position a piece's first +// character is the escaped one; an EMPTY piece there means the escaped character was the next +// backslash, so the piece after it is literal again. An empty last piece in escape position is a +// backslash with nothing after it: the closing quote was escaped, so the scalar never closed. +fn yaml_double_quoted_pieces(pieces: List, index: Int, escape: Bool, acc: List) -> YamlScalarResult { + match get(xs: pieces, index: index) { + Absent => ReadYamlScalar { value: YamlString { value: join(acc, "") } } + Present { value: p } => + if !escape { + if string_contains(s: p, pattern: "\"") { + YamlScalarRefused { reason: yaml_unescaped_quote_reason } + } else { + yaml_double_quoted_pieces(pieces: pieces, index: index + 1, escape: true, acc: concat(acc, [p])) + } + } else if p == "" { + if index + 1 >= pieces.length() { + YamlScalarRefused { reason: "a double-quoted scalar ends in a lone backslash, which escapes its closing quote" } + } else { + yaml_double_quoted_pieces(pieces: pieces, index: index + 1, escape: false, acc: concat(acc, ["\\"])) + } + } else { + yaml_double_quoted_escape(pieces: pieces, index: index, p: p, acc: acc) + } + } +} + +fn yaml_double_quoted_escape(pieces: List, index: Int, p: String, acc: List) -> YamlScalarResult { + let c = substring(s: p, start: 0, end: 1) + let width = if c == "x" { 2 } else if c == "u" { 4 } else if c == "U" { 8 } else { 0 } + if width == 0 { + match map_get(yaml_double_quoted_escapes, c) { + Absent => YamlScalarRefused { reason: join(["`\\", c, "` is not a YAML escape (section 5.7)"], "") } + Present { value: cp } => yaml_double_quoted_rest(pieces: pieces, index: index, rest: substring(s: p, start: 1, end: p.length()), acc: concat(acc, [from_code_point(cp: cp)])) + } + } else if p.length() < width + 1 { + YamlScalarRefused { reason: join(["`\\", c, "` needs ", to_string(width), " hexadecimal digits"], "") } + } else { + let cp = yaml_hex_value(s: substring(s: p, start: 1, end: width + 1), i: 0, acc: 0) + if cp < 0 { + YamlScalarRefused { reason: join(["`\\", substring(s: p, start: 0, end: width + 1), "` is not a hexadecimal escape"], "") } + } else if ((cp >= 55296) && (cp <= 57343)) || (cp > 1114111) { + YamlScalarRefused { reason: join(["`\\", substring(s: p, start: 0, end: width + 1), "` does not name a Unicode scalar value"], "") } + } else { + yaml_double_quoted_rest(pieces: pieces, index: index, rest: substring(s: p, start: width + 1, end: p.length()), acc: concat(acc, [from_code_point(cp: cp)])) + } + } +} + +fn yaml_double_quoted_rest(pieces: List, index: Int, rest: String, acc: List) -> YamlScalarResult { + if string_contains(s: rest, pattern: "\"") { + YamlScalarRefused { reason: yaml_unescaped_quote_reason } + } else { + yaml_double_quoted_pieces(pieces: pieces, index: index + 1, escape: true, acc: concat(acc, [rest])) + } +} + +// The index just past the quote that closes the quoted scalar opening at 0, or -1 if it does not +// close on this line. Used only where a quoted scalar may be followed by more text: a quoted key. +fn yaml_quoted_end(s: String, i: Int, quote: String) -> Int { + if i >= s.length() { + -1 + } else { + let c = char_at(s, i) + if (quote == "\"") && (c == "\\") { + yaml_quoted_end(s: s, i: i + 2, quote: quote) + } else if c != quote { + yaml_quoted_end(s: s, i: i + 1, quote: quote) + } else if (quote == "'") && yaml_char_is(s: s, i: i + 1, alphabet: "'") { + yaml_quoted_end(s: s, i: i + 2, quote: quote) + } else { + i + 1 + } + } +} + +fn yaml_quoted_scalar(v: String) -> YamlScalarResult { + if starts_with(s: v, prefix: "'") { yaml_single_quoted(v: v) } else { yaml_double_quoted(v: v) } +} + +// --------------------------------------------------------------------------------------------- +// FLOW SEQUENCES +// --------------------------------------------------------------------------------------------- + +// A flow entry is a quoted scalar or a plain scalar that, in flow context, also may not contain a +// flow indicator (section 7.3.3). +fn yaml_flow_entry(e: String) -> YamlScalarResult { + if e == "" { + YamlScalarRefused { reason: "an empty flow sequence entry (including a trailing comma) is outside the supported subset" } + } else if starts_with(s: e, prefix: "\"") || starts_with(s: e, prefix: "'") { + yaml_quoted_scalar(v: e) + } else if string_contains(s: yaml_value_special_starts, pattern: substring(s: e, start: 0, end: 1)) && !yaml_indicator_opens_plain(v: e) { + YamlScalarRefused { reason: join(["the flow entry `", e, "` starts with an indicator; nested flow collections, anchors, aliases and tags are outside the supported subset"], "") } + } else if string_contains(s: e, pattern: "[") || string_contains(s: e, pattern: "]") || string_contains(s: e, pattern: "{") || string_contains(s: e, pattern: "}") { + YamlScalarRefused { reason: join(["the flow entry `", e, "` contains a flow indicator; quote it"], "") } + } else { + let bad = yaml_plain_refusal(s: e) + if bad != "" { YamlScalarRefused { reason: bad } } else { ReadYamlScalar { value: yaml_resolve_plain(s: e) } } + } +} + +fn yaml_flow_entries(entries: List, index: Int, acc: List) -> YamlScalarResult { + match get(xs: entries, index: index) { + Absent => ReadYamlScalar { value: YamlSequence { elements: acc } } + Present { value: e } => + match yaml_flow_entry(e: trim(e)) { + ReadYamlScalar { value: x } => yaml_flow_entries(entries: entries, index: index + 1, acc: concat(acc, [x])) + YamlScalarRefused { reason: r } => YamlScalarRefused { reason: r } + } + } +} + +// Splits at the commas that are outside quotes. Quote state is tracked character by character; a +// backslash inside double quotes skips the character after it. A quoted scalar can only OPEN where +// an entry opens (section 7.3), so `at_entry_start` carries that position: a quote anywhere else is +// ordinary plain content, and entering quote mode there would hide the following commas from the +// splitter and read `[a'b,c]` as one entry instead of two. +fn yaml_flow_split(s: String, i: Int, start: Int, quote: String, at_entry_start: Bool, acc: List) -> List { + if i >= s.length() { + concat(acc, [substring(s: s, start: start, end: s.length())]) + } else { + let c = char_at(s, i) + if quote == "" { + if c == "," { + yaml_flow_split(s: s, i: i + 1, start: i + 1, quote: "", at_entry_start: true, acc: concat(acc, [substring(s: s, start: start, end: i)])) + } else if at_entry_start && ((c == "\"") || (c == "'")) { + yaml_flow_split(s: s, i: i + 1, start: start, quote: c, at_entry_start: false, acc: acc) + } else { + yaml_flow_split(s: s, i: i + 1, start: start, quote: "", at_entry_start: at_entry_start && (c == " "), acc: acc) + } + } else if (quote == "\"") && (c == "\\") { + yaml_flow_split(s: s, i: i + 2, start: start, quote: quote, at_entry_start: false, acc: acc) + } else if c == quote { + yaml_flow_split(s: s, i: i + 1, start: start, quote: "", at_entry_start: false, acc: acc) + } else { + yaml_flow_split(s: s, i: i + 1, start: start, quote: quote, at_entry_start: false, acc: acc) + } + } +} + +fn yaml_flow_sequence(v: String) -> YamlScalarResult { + if !ends_with(s: v, suffix: "]") { + YamlScalarRefused { reason: "a flow sequence must close on its own line with nothing after `]` (multi-line flow collections and trailing comments are outside the supported subset)" } + } else { + let inner = trim(substring(s: v, start: 1, end: v.length() - 1)) + if inner == "" { + ReadYamlScalar { value: YamlSequence { elements: [] } } + } else if string_contains(s: inner, pattern: "\"") || string_contains(s: inner, pattern: "'") { + yaml_flow_entries(entries: yaml_flow_split(s: inner, i: 0, start: 0, quote: "", at_entry_start: true, acc: []), index: 0, acc: []) + } else if yaml_flow_inner_is_plain(inner: inner) { + yaml_flow_plain_sequence(inner: inner) + } else { + yaml_flow_entries(entries: split(s: inner, delimiter: ","), index: 0, acc: []) + } + } +} + +// NO QUOTE, FLOW INDICATOR, `:` OR `#` ANYWHERE INSIDE THE BRACKETS: then no entry can hold a flow +// indicator, a mapping indicator or a comment, and that is decided once for the sequence rather than +// again for every entry. Each entry is then read by its first character alone, and its reading is +// typed: an entry that cannot open a plain scalar here (an empty entry, an anchor, an alias, a tag) +// is a refusal with its reason, never a value standing in for one. +fn yaml_flow_inner_is_plain(inner: String) -> Bool { + if string_contains(s: inner, pattern: "[") { false } + else if string_contains(s: inner, pattern: "]") { false } + else if string_contains(s: inner, pattern: "\{") { false } + else if string_contains(s: inner, pattern: "}") { false } + else if string_contains(s: inner, pattern: ":") { false } + else { !string_contains(s: inner, pattern: "#") } +} + +fn yaml_flow_plain_sequence(inner: String) -> YamlScalarResult { + let reads = map(split(s: inner, delimiter: ","), e => yaml_flow_plain_entry(e: trim(e))) + let elements = flat_map(reads, r => match r { ReadYamlScalar { value: v } => [v] YamlScalarRefused { reason: _ } => [] }) + if (elements |> count) != (reads |> count) { + yaml_first_scalar_refusal(reads: reads, index: 0) + } else { + ReadYamlScalar { value: YamlSequence { elements: elements } } + } +} + +fn yaml_first_scalar_refusal(reads: List, index: Int) -> YamlScalarResult { + match get(xs: reads, index: index) { + Absent => YamlScalarRefused { reason: "a flow sequence counted fewer accepted entries than entries but no entry reading refused; the sequence's reading is inconsistent and is refused rather than passed" } + Present { value: r } => + match r { + YamlScalarRefused { reason: why } => YamlScalarRefused { reason: why } + ReadYamlScalar { value: _ } => yaml_first_scalar_refusal(reads: reads, index: index + 1) + } + } +} + +// One entry of a sequence already known to hold no quote, flow indicator, `:` or `#`: its first +// character decides it. An entry that cannot open a plain scalar here is read by yaml_flow_entry, +// which is the reading that makes that refusal and says why. +fn yaml_flow_plain_entry(e: String) -> YamlScalarResult { + if e == "" { + yaml_flow_entry(e: e) + } else { + match map_get(yaml_value_start_class, substring(s: e, start: 0, end: 1)) { + Absent => ReadYamlScalar { value: YamlString { value: e } } + Present { value: cls } => + if cls != 1 { + ReadYamlScalar { value: yaml_resolve_plain(s: e) } + } else if yaml_indicator_opens_plain(v: e) { + ReadYamlScalar { value: yaml_resolve_plain(s: e) } + } else { + yaml_flow_entry(e: e) + } + } + } +} + +// Spaces only: a tab ends the count. fn yaml_indent_of(line: String) -> Int { if starts_with(s: line, prefix: " ") { 8 + yaml_indent_of(line: substring(s: line, start: 8, end: line.length())) @@ -85,320 +553,870 @@ fn yaml_indent_of(line: String) -> Int { } } -fn yaml_strip_indent(line: String, indent: Int) -> String { - substring(s: line, start: indent, end: line.length()) +// --------------------------------------------------------------------------------------------- +// LITERAL BLOCK SCALARS +// --------------------------------------------------------------------------------------------- + +data yaml_block_scalar_chomping: Map = { "|": "clip", "|-": "strip", "|+": "keep" } + +data yaml_spaces: String = " " +data yaml_newline_spaces: String = concat("\n", yaml_spaces) +data yaml_max_indent: Int = yaml_spaces.length() + +fn yaml_first_nonempty(lines: List, index: Int) -> Int { + match get(xs: lines, index: index) { + Absent => index + Present { value: l } => if l == "" { yaml_first_nonempty(lines: lines, index: index + 1) } else { index } + } } -// ONE LINE, READ ONCE. Every parse function asks the same three questions of a line -- its indent, -// its text after that indent, and whether it is blank or a full-line comment -- and each used to -// re-derive them from the raw string, several times per line and once more at every enclosing block -// level that checks for its own end. That made the parse cost a large constant per line for facts -// that never change after the line is read; the constant is re-derived by claim_batch eval_steps -// over ingest_yaml_source on .github/workflows/fleet-converge.yml. The table computes them once; -// every consumer reads the field. -type YamlLine { - raw: String - indent: Int - text: String - ignorable: Bool +// The first line of a literal's continuation that is neither empty nor indented to the content +// column: the literal's body ends there (section 8.1.2). +fn yaml_literal_end(lines: List, index: Int, prefix: String) -> Int { + match get(xs: lines, index: index) { + Absent => index + Present { value: l } => if (l == "") || starts_with(s: l, prefix: prefix) { yaml_literal_end(lines: lines, index: index + 1, prefix: prefix) } else { index } + } } -fn yaml_line_record(raw: String) -> YamlLine { - let indent = yaml_indent_of(line: raw) - let text = substring(s: raw, start: indent, end: raw.length()) - YamlLine { raw: raw, indent: indent, text: text, ignorable: (text == "") || starts_with(s: text, prefix: "#") } +fn yaml_strip_final_newlines(s: String) -> String { + if ends_with(s: s, suffix: "\n") { yaml_strip_final_newlines(s: substring(s: s, start: 0, end: s.length() - 1)) } else { s } } -fn yaml_lines_from_source(src: String) -> List { - let raw = split(s: src, delimiter: "\n") - let n = raw.length() - let kept = if (n > 0) && (raw.skip(n: n - 1).first() == "") { raw.take(n: n - 1) } else { raw } - map(kept, l => yaml_line_record(raw: l)) +// Chomping (section 8.1.1.2): strip drops every final line break, clip keeps exactly one, keep +// keeps them all. `text` is the body's lines joined, so its final line breaks are its trailing +// empty lines. +fn yaml_chomp(chomping: String, text: String) -> String { + if chomping == "strip" { + yaml_strip_final_newlines(s: text) + } else if chomping == "clip" { + concat(yaml_strip_final_newlines(s: text), "\n") + } else { + concat(text, "\n") + } +} + +// The body with its content indentation removed from every line. Empty lines carry none. +fn yaml_literal_text(body: String, prefix: String) -> String { + let s = replace(concat("\n", body), concat("\n", prefix), "\n") + substring(s: s, start: 1, end: s.length()) +} + +fn yaml_block_scalar(header: String, cont: String, col: Int) -> YamlBlockResult { + match map_get(yaml_block_scalar_chomping, header) { + Absent => YamlBlockRefused { line: 0, reason: join(["the block scalar header `", header, "` is outside the supported subset (only `|`, `|-` and `|+`, with no indentation indicator and no comment)"], "") } + Present { value: chomping } => yaml_literal_body(chomping: chomping, cont: cont, col: col) + } +} + +// `cont` is every line after the header up to the next entry of the enclosing block; its first +// non-empty line fixes the content indentation (section 8.1.1.1), which must exceed `col`. +// The leading whitespace of a line, which the first content line of a literal repeats across a +// document's scripts. +fn yaml_line_lead(line: String) -> Int { + line.length() - trim(line).length() } -fn yaml_has_line_at(lines: List, index: Int) -> Bool { - index < lines.length() +// The number of empty lines in `text`, by one marking rather than a test per line: every line start +// is marked with U+0002 (refused anywhere in an accepted document, so it cannot already be there), +// and an empty line is then exactly a mark followed directly by a line break. +fn yaml_empty_line_count(text: String) -> Int { + (split(s: replace(concat("\n", concat(text, "\n")), "\n", concat("\n", yaml_line_start_mark)), delimiter: concat(yaml_line_start_mark, "\n")) |> count) - 1 } -// Callers guard with yaml_has_line_at; the Absent arm is the out-of-range row those guards exclude, -// shaped so that no predicate below can mistake it for content (indent -1 matches no block). -fn yaml_line_entry(lines: List, index: Int) -> YamlLine { +data yaml_line_start_mark: String = from_code_point(cp: 2) + +fn yaml_literal_body(chomping: String, cont: String, col: Int) -> YamlBlockResult { + let lines = split(s: cont, delimiter: "\n") + let first = join(lines.take(n: 1), "") + let lead = yaml_line_lead(line: first) + if (first == "") || (lead <= col) || (lead > yaml_max_indent) { + yaml_literal_general(chomping: chomping, cont: cont, col: col, lines: lines) + } else { + let prefix = substring(s: yaml_spaces, start: 0, end: lead) + let blanks = if string_contains(s: concat(cont, "\n"), pattern: "\n\n") { yaml_empty_line_count(text: cont) } else { 0 } + if (split(s: concat("\n", cont), delimiter: concat("\n", prefix)) |> count) - 1 + blanks == (lines |> count) { + YamlBlockParsed { value: YamlString { value: yaml_chomp(chomping: chomping, text: yaml_literal_text(body: cont, prefix: prefix)) } } + } else { + yaml_literal_general(chomping: chomping, cont: cont, col: col, lines: lines) + } + } +} + +// Every literal the fast test above does not settle: leading empty lines, a first line whose +// leading whitespace holds a tab, no content, or a body that ends before the chunk does. +fn yaml_literal_general(chomping: String, cont: String, col: Int, lines: List) -> YamlBlockResult { + let first = match get(xs: lines, index: yaml_first_nonempty(lines: lines, index: 0)) { Present { value: l } => l Absent => "" } + let content_indent = yaml_indent_of(line: first) + if (cont == "") || (first == "") || (content_indent <= col) { + if chomping == "keep" { + YamlBlockRefused { line: 0, reason: "a keep-chomped (`|+`) block scalar with no content line is outside the supported subset" } + } else { + yaml_scalar_with_cont(value: YamlString { value: "" }, cont: cont) + } + } else if content_indent > yaml_max_indent { + YamlBlockRefused { line: 1, reason: "block scalar content indented past 64 columns is outside the supported subset" } + } else { + let prefix = substring(s: yaml_spaces, start: 0, end: content_indent) + let end = yaml_literal_end(lines: lines, index: 0, prefix: prefix) + match yaml_scalar_with_cont(value: YamlNull, cont: join(lines.skip(n: end), "\n")) { + YamlBlockRefused { line: l, reason: r } => YamlBlockRefused { line: end + l, reason: r } + YamlBlockParsed { value: _ } => + YamlBlockParsed { value: YamlString { value: yaml_chomp(chomping: chomping, text: yaml_literal_text(body: join(lines.take(n: end), "\n"), prefix: prefix)) } } + } + } +} + +// --------------------------------------------------------------------------------------------- +// BLOCK STRUCTURE +// --------------------------------------------------------------------------------------------- +// +// A BLOCK IS SPLIT INTO ITS ENTRIES WITH ONE SPLIT, NOT READ LINE BY LINE. `text` is a block whose +// first line starts at column `n` (with that indentation already removed) and whose later lines +// keep theirs. Every line break followed by `n` spaces and a non-space is first marked with +// yaml_line_join_mark, a character that cannot occur in the document's source text, so splitting at +// the marks yields one chunk per line at column `n`, with the deeper lines left inside their chunk +// exactly as written. A chunk that is an +// empty line, a full-line comment, or (in a mapping) a sequence item at the mapping's own column is +// joined back to the chunk before it: the first two carry no content there, and the third is the +// value of the key before it. Everything else about a chunk is decided by reading that chunk. +// +// Every result's line is an offset from the first line of the text it was handed; callers add +// their own offset, and ingest_yaml_source reports the 1-based document line. + +type YamlBlockResult + = YamlBlockParsed { value: YamlValue } + | YamlBlockRefused { line: Int, reason: String } + +// ONE ENTRY'S READING: the pair a mapping holds, or the refusal with its line (an offset from the +// entry's first line) and reason. A refusal is an arm of its own and never a value: no key, string or +// sentinel stands for "refused", so no accepted document can be mistaken for a refusing one or the +// reverse. +type YamlEntryRead + = YamlEntryAccepted { entry: YamlKeyValue } + | YamlEntryRefused { line: Int, reason: String } + +// THE MARK IS SOURCE-TEXT PUNCTUATION AND NOTHING ELSE. It is inserted into, and split out of, the +// text of one block, and the document it came from cannot already hold it (ingest_yaml_source refuses +// U+0001 anywhere). It never appears in a value, a key or a refusal: a decoded scalar may hold U+0001 +// (a double-quoted `\x01`), and that is ordinary content. +data yaml_line_join_mark: String = from_code_point(cp: 1) + +fn yaml_is_ignorable_line(l: String) -> Bool { + let t = trim(l) + (t == "") || starts_with(s: t, prefix: "#") +} + +// The offset of the first line of `cont` that carries content, or -1 when none does. +fn yaml_first_content_offset(lines: List, index: Int) -> Int { match get(xs: lines, index: index) { - Present { value: l } => l - Absent => YamlLine { raw: "", indent: -1, text: "", ignorable: false } + Absent => -1 + Present { value: l } => if yaml_is_ignorable_line(l: l) { yaml_first_content_offset(lines: lines, index: index + 1) } else { index } + } +} + +// A scalar's entry is one line: whatever else its chunk holds must be empty lines or comments. +fn yaml_scalar_with_cont(value: YamlValue, cont: String) -> YamlBlockResult { + if cont == "" { + YamlBlockParsed { value: value } + } else { + let at = yaml_first_content_offset(lines: split(s: cont, delimiter: "\n"), index: 0) + if at < 0 { + YamlBlockParsed { value: value } + } else { + YamlBlockRefused { line: at + 1, reason: "unexpected indentation: this line continues a scalar or sits between block levels (multi-line plain scalars and irregular indentation are outside the supported subset)" } + } + } +} + +// A chunk continues the one before it when its first line is empty (a blank line at column 0), +// a comment, or -- in a mapping -- a sequence item at the mapping's own column. +fn yaml_chunk_continues(c: String, seq: Bool) -> Bool { + if starts_with(s: concat(c, "\n"), prefix: "\n") { + true + } else if starts_with(s: c, prefix: "#") { + true + } else if seq { + false + } else { + yaml_is_seq_item(t: yaml_first_line(c: c)) } } -fn yaml_text_at(lines: List, index: Int) -> String { - yaml_line_entry(lines: lines, index: index).text +fn yaml_merge_chunks(chunks: List, sp: String, seq: Bool) -> List { + fold(chunks, init: [], f: (acc, c) => + if ((acc |> count) > 0) && yaml_chunk_continues(c: c, seq: seq) { + concat(acc.take(n: (acc |> count) - 1), [join([join(acc.skip(n: (acc |> count) - 1), ""), "\n", sp, c], "")]) + } else { + concat(acc, [c]) + }) +} + +fn yaml_is_seq_item(t: String) -> Bool { + starts_with(s: concat(t, " "), prefix: "- ") +} + +fn yaml_starts_seq(text: String) -> Bool { + starts_with(s: concat(text, " "), prefix: "- ") || starts_with(s: text, prefix: "-\n") +} + +// Only the root block can hold an empty line at its own column (a deeper block's empty lines stay +// inside the chunk above them), and only a mapping merges a sequence item at its own column. +fn yaml_root_needs_merge(text: String, seq: Bool) -> Bool { + string_contains(s: concat(text, "\n"), pattern: "\n\n") || (!seq && string_contains(s: text, pattern: "\n-")) } -fn yaml_indent_at(lines: List, index: Int) -> Int { - yaml_line_entry(lines: lines, index: index).indent +// THE MARKS OF A BLOCK AT INDENTATION `n` ARE A FUNCTION OF `n` ALONE, so they are built once per +// indentation a document uses rather than once per block. `nl` is a line break followed by `n` +// spaces; `marked` is the same break behind the line-join mark; `deeper`, `comment` and `item` are +// the three things that can follow `nl` and still belong to the chunk above. +type YamlIndentMarks { + nl: String + marked: String + deeper: String + marked_deeper: String + comment: String + item: String } -// OUTSIDE A SCALAR, A BLANK LINE OR A FULL-LINE COMMENT CARRIES NO CONTENT AT ANY INDENTATION -// (YAML 1.2.2 section 6.6, l-comment). Block-literal bodies never pass through here -- they are -// collected by yaml_collect_literal_lines_from, where blank and `#` lines ARE content -- so skipping -// them between structural entries cannot drop scalar text. -fn yaml_line_is_ignorable(lines: List, index: Int) -> Bool { - yaml_has_line_at(lines: lines, index: index) && yaml_line_entry(lines: lines, index: index).ignorable +fn yaml_indent_marks(n: Int) -> YamlIndentMarks { + let nl = substring(s: yaml_newline_spaces, start: 0, end: n + 1) + let marked = concat(yaml_line_join_mark, nl) + YamlIndentMarks { nl: nl, marked: marked, deeper: concat(nl, " "), marked_deeper: concat(marked, " "), comment: concat(nl, "#"), item: concat(nl, "-") } } -fn yaml_skip_ignorable_lines(lines: List, index: Int) -> Int { - if yaml_line_is_ignorable(lines: lines, index: index) { - yaml_skip_ignorable_lines(lines: lines, index: index + 1) +// Deeper lines are joined to the line above by marking every line break at this block's own +// indentation and unmarking the ones a deeper line follows, so the split lands only on this block's +// own lines; with no deeper line the unmarking finds nothing and the split is the plain one. +fn yaml_block(text: String, n: Int, seq: Bool, base: Int) -> YamlBlockResult { + if n > yaml_max_indent { + YamlBlockRefused { line: base, reason: "indentation past 64 columns is outside the supported subset" } } else { - index + let k = yaml_indent_marks(n: n) + if !string_contains(s: text, pattern: k.nl) { + yaml_one_chunk_block(text: text, n: n, seq: seq, base: base) + } else { + let raw = split(s: replace(replace(text, k.nl, k.marked), k.marked_deeper, k.deeper), delimiter: k.marked) + let chunks = if string_contains(s: text, pattern: k.comment) { + yaml_merge_chunks(chunks: raw, sp: substring(s: yaml_spaces, start: 0, end: n), seq: seq) + } else if n == 0 { + if yaml_root_needs_merge(text: text, seq: seq) { yaml_merge_chunks(chunks: raw, sp: "", seq: seq) } else { raw } + } else if seq { + raw + } else if string_contains(s: text, pattern: k.item) { + yaml_merge_chunks(chunks: raw, sp: substring(s: yaml_spaces, start: 0, end: n), seq: seq) + } else { + raw + } + if seq { yaml_seq_block(chunks: chunks, n: n, base: base) } else { yaml_map_block(chunks: chunks, n: n, base: base) } + } } } -// Inline comment separation is not yet modeled. Refuse the recognizable unquoted value form -// instead of silently ingesting its comment as scalar content. This intentionally errs toward -// refusal for richer scalar forms this bounded parser does not model; double-quoted scalars are -// excluded, and block-literal content never passes through this predicate. -fn yaml_has_unmodeled_inline_comment(text: String) -> Bool { - let double_quoted = starts_with(s: text, prefix: "\"") && ends_with(s: text, suffix: "\"") - !double_quoted && ( - string_contains(s: text, pattern: " #") - || string_contains(s: text, pattern: "\t#") - ) +// No line break at this block's own indentation: the whole text is one entry (or one item), so it +// needs neither the marking nor the split. +fn yaml_one_chunk_block(text: String, n: Int, seq: Bool, base: Int) -> YamlBlockResult { + if seq { yaml_seq_block(chunks: [text], n: n, base: base) } else { yaml_single_entry_block(c: text, n: n, base: base) } } -// A VALUE WITH NO BACKSLASH HAS NOTHING TO UNESCAPE, so it is returned whole. The per-character -// recursion below copies the remainder at every character -- quadratic in the value's length -- and -// generated workflows are made of long double-quoted command lines; running it only when an escape -// is present gives the identical result. -fn yaml_unescape_double_quoted(s: String) -> String { - if !string_contains(s: s, pattern: "\\") { - s +fn yaml_single_entry_block(c: String, n: Int, base: Int) -> YamlBlockResult { + match yaml_entry(c: c, n: n) { + YamlEntryAccepted { entry: e } => YamlBlockParsed { value: YamlMapping { entries: [e] } } + YamlEntryRefused { line: l, reason: why } => YamlBlockRefused { line: base + l, reason: why } + } +} + +fn yaml_line_count(c: String) -> Int { + split(s: c, delimiter: "\n") |> count +} + +// A BLOCK'S READINGS ARE TYPED, AND A REFUSAL IS FOUND BY COUNTING THE ACCEPTED ONES. Each chunk's +// reading is either accepted or refused; the accepted pairs are the mapping, and a block with fewer +// pairs than chunks refused somewhere, which the walk below locates in the readings already made. +// +// WHETHER THE KEYS ARE ADMISSIBLE IS A QUESTION ABOUT THE KEY SEQUENCE AND NOTHING ELSE, so it is +// asked of that sequence spelled as one string, and the sequences a document repeats -- every step +// with the same fields, every `with:` of the same action -- are one question. +fn yaml_map_block(chunks: List, n: Int, base: Int) -> YamlBlockResult { + let reads = map(chunks, c => yaml_entry(c: c, n: n)) + let entries = flat_map(reads, r => match r { YamlEntryAccepted { entry: e } => [e] YamlEntryRefused { line: _, reason: _ } => [] }) + if (entries |> count) != (chunks |> count) { + yaml_first_entry_refusal(chunks: chunks, reads: reads, index: 0, line: base) + } else if yaml_key_sequence_is_clean(keys: join(map(entries, e => e.key), yaml_key_separator), count: entries |> count) { + YamlBlockParsed { value: YamlMapping { entries: entries } } } else { - yaml_unescape_double_quoted_chars(s: s) + yaml_map_block_keys(chunks: chunks, entries: entries, base: base) } } -fn yaml_unescape_double_quoted_chars(s: String) -> String { - if s == "" { - "" - } else if starts_with(s: s, prefix: "\\\"") { - concat("\"", yaml_unescape_double_quoted_chars(s: substring(s: s, start: 2, end: s.length()))) - } else if starts_with(s: s, prefix: "\\\\") { - concat("\\", yaml_unescape_double_quoted_chars(s: substring(s: s, start: 2, end: s.length()))) +// U+0003 separates the keys. A decoded key may itself hold it (a double-quoted `\x03`), and the +// spelled sequence then splits into more keys than the block has; that block's keys are then read one +// by one by yaml_map_block_keys rather than trusted to the spelling. +data yaml_key_separator: String = from_code_point(cp: 3) + +// True exactly when the spelled keys split back into `count` keys and no key repeats. +fn yaml_key_sequence_is_clean(keys: String, count: Int) -> Bool { + let ks = split(s: keys, delimiter: yaml_key_separator) + if (ks |> count) != count { + false } else { - concat(substring(s: s, start: 0, end: 1), yaml_unescape_double_quoted_chars(s: substring(s: s, start: 1, end: s.length()))) + (map_keys(fold(ks, init: empty_map(), f: (m, k) => map_insert(m, k, true))) |> count) == count } } -fn yaml_is_flow_sequence_text(text: String) -> Bool { - starts_with(s: text, prefix: "[") && ends_with(s: text, suffix: "]") +fn yaml_map_block_keys(chunks: List, entries: List, base: Int) -> YamlBlockResult { + let keys = fold(entries, init: empty_map(), f: (m, e) => map_insert(m, e.key, true)) + if (map_keys(keys) |> count) != (entries |> count) { + yaml_duplicate_key_refusal(chunks: chunks, entries: entries, index: 0, line: base, seen: empty_map()) + } else { + YamlBlockParsed { value: YamlMapping { entries: entries } } + } } -fn yaml_parse_flow_sequence(text: String) -> YamlValue { - let inner = substring(s: text, start: 1, end: text.length() - 1) - if inner == "" { - YamlSequence { elements: [] } +// THE FIRST REFUSED READING, LOCATED. `line` accumulates the lines of the chunks before it, so the +// refusal's own offset becomes an offset from the block's first line. The readings are the ones the +// block already made -- nothing is read twice -- and a count that promised a refusal the walk does not +// find refuses too, by name, rather than passing the block. +fn yaml_first_entry_refusal(chunks: List, reads: List, index: Int, line: Int) -> YamlBlockResult { + match get(xs: reads, index: index) { + Absent => YamlBlockRefused { line: line, reason: "a mapping counted fewer accepted entries than chunks but no entry reading refused; the block's reading is inconsistent and is refused rather than passed" } + Present { value: r } => + match r { + YamlEntryRefused { line: l, reason: why } => YamlBlockRefused { line: line + l, reason: why } + YamlEntryAccepted { entry: _ } => yaml_first_entry_refusal(chunks: chunks, reads: reads, index: index + 1, line: line + yaml_line_count(c: yaml_chunk_at(chunks: chunks, index: index))) + } + } +} + +fn yaml_chunk_at(chunks: List, index: Int) -> String { + match get(xs: chunks, index: index) { Present { value: c } => c Absent => "" } +} + +fn yaml_duplicate_key_refusal(chunks: List, entries: List, index: Int, line: Int, seen: Map) -> YamlBlockResult { + match get(xs: entries, index: index) { + Absent => YamlBlockRefused { line: line, reason: "a mapping holds a key twice" } + Present { value: e } => + if map_contains_key(seen, e.key) { + YamlBlockRefused { line: line, reason: join(["duplicate mapping key `", e.key, "` (YAML requires the keys of a mapping to be unique)"], "") } + } else { + yaml_duplicate_key_refusal(chunks: chunks, entries: entries, index: index + 1, line: line + yaml_line_count(c: yaml_chunk_at(chunks: chunks, index: index)), seen: map_insert(seen, e.key, true)) + } + } +} + +// A SEQUENCE THE SAME WAY: typed readings, the accepted values are the elements, and a shortfall is +// located in the readings already made. +fn yaml_seq_block(chunks: List, n: Int, base: Int) -> YamlBlockResult { + let reads = map(chunks, c => yaml_item_read(c: c, n: n)) + let elements = flat_map(reads, r => match r { YamlBlockParsed { value: v } => [v] YamlBlockRefused { line: _, reason: _ } => [] }) + if (elements |> count) != (chunks |> count) { + yaml_first_item_refusal(chunks: chunks, reads: reads, index: 0, line: base) } else { - YamlSequence { elements: map(split(s: inner, delimiter: ", "), t => yaml_parse_scalar_text(text: t)) } + YamlBlockParsed { value: YamlSequence { elements: elements } } + } +} + +fn yaml_first_item_refusal(chunks: List, reads: List, index: Int, line: Int) -> YamlBlockResult { + match get(xs: reads, index: index) { + Absent => YamlBlockRefused { line: line, reason: "a sequence counted fewer accepted items than chunks but no item reading refused; the block's reading is inconsistent and is refused rather than passed" } + Present { value: r } => + match r { + YamlBlockRefused { line: l, reason: why } => YamlBlockRefused { line: line + l, reason: why } + YamlBlockParsed { value: _ } => yaml_first_item_refusal(chunks: chunks, reads: reads, index: index + 1, line: line + yaml_line_count(c: yaml_chunk_at(chunks: chunks, index: index))) + } } } -fn yaml_parse_scalar_text(text: String) -> YamlValue { - if text == "''" { - YamlString { value: "" } - } else if starts_with(s: text, prefix: "\"") && ends_with(s: text, suffix: "\"") && (text.length() >= 2) { - YamlString { value: yaml_unescape_double_quoted(s: substring(s: text, start: 1, end: text.length() - 1)) } - } else if text == "true" { - YamlBool { value: true } - } else if text == "false" { - YamlBool { value: false } - } else if yaml_looks_like_int(s: text) { - YamlInt { lexeme: text } +// THE COMMON ITEM, `- key: ...`, A COMPACT MAPPING. Whether an item is one is decided by its first +// line up to the first `: ` -- `- name`, `- uses` -- and by whether a `: ` follows, and nothing +// else, so that is the whole identity of the decision and the heads a sequence repeats are one +// question. Every other item is read by yaml_item, which is the reading that also makes the refusals. +fn yaml_item_read(c: String, n: Int) -> YamlBlockResult { + let f = join(split(s: c, delimiter: "\n").take(n: 1), "") + let head = join(split(s: f, delimiter: ": ").take(n: 1), "") + if yaml_item_head_opens_mapping(head: head, has_sep: head != f) { + yaml_block(text: substring(s: c, start: 2, end: c.length()), n: n + 2, seq: false, base: 0) } else { - YamlString { value: text } + yaml_item(c: c, n: n) } } -// A BLOCK LITERAL IS A SLICE: every line after the indicator that is empty or indented to at least -// the content column. Its end is found with one field test per line, and its text is built once -// from the slice -- not by re-reading each line and copying the growing accumulator on every step. -fn yaml_literal_end(lines: List, index: Int, content_indent: Int) -> Int { - let e = yaml_line_entry(lines: lines, index: index) - if yaml_has_line_at(lines: lines, index: index) && ((e.raw == "") || (e.indent >= content_indent)) { - yaml_literal_end(lines: lines, index: index + 1, content_indent: content_indent) +// `- ` then a character that opens neither an indicator nor whitespace, and either a `: ` later in +// the line or a final `:`: yaml_item would find the same body at column 2 and read it as a mapping. +fn yaml_item_head_opens_mapping(head: String, has_sep: Bool) -> Bool { + if !starts_with(s: head, prefix: "- ") { + false + } else if head.length() < 3 { + false + } else if string_contains(s: yaml_item_body_nonplain_starts, pattern: substring(s: head, start: 2, end: 3)) { + false + } else if has_sep { + true } else { - index + ends_with(s: head, suffix: ":") } } -fn yaml_collect_literal_lines_from(lines: List, index: Int, content_indent: Int) -> YamlLiteralCollectResult { - let end = yaml_literal_end(lines: lines, index: index, content_indent: content_indent) - let body = lines.skip(n: index).take(n: end - index) - YamlLiteralCollectResult { - text: join(map(body, l => if l.raw == "" { "" } else { yaml_strip_indent(line: l.raw, indent: content_indent) }), "\n"), - next_index: end, +data yaml_item_body_nonplain_starts: String = concat(yaml_value_special_starts, " \t") + +// First characters that do not open an ordinary plain scalar in block context: quotes, flow and +// block-scalar indicators, anchors, aliases, tags, reserved indicators, a comment, and `-`, `?`, +// `:`, which open a plain scalar only when a non-space follows (yaml_indicator_opens_plain). +data yaml_value_special_starts: String = "\"'[]{},|>&*!%@`#-?:" + +fn yaml_indicator_opens_plain(v: String) -> Bool { + if (v.length() < 2) || yaml_char_is(s: v, i: 1, alphabet: " \t") { + false + } else { + starts_with(s: v, prefix: "-") || starts_with(s: v, prefix: "?") || starts_with(s: v, prefix: ":") } } -fn yaml_parse_mapping_entry(lines: List, index: Int, indent: Int) -> YamlEntryResult { - let content = yaml_text_at(lines: lines, index: index) - let parts = split(s: content, delimiter: ": ") - if parts.length() > 1 { - match parts.first() { - Absent => YamlEntryRejected { reason: "mapping entry has no key" } - Present { value: key } => - let rest = join(parts.skip(n: 1), ": ") - if yaml_has_unmodeled_inline_comment(text: rest) { - YamlEntryRejected { reason: "inline YAML comments after values are not modeled" } - } else if rest == "|" { - let lit = yaml_collect_literal_lines_from(lines: lines, index: index + 1, content_indent: indent + 2) - ParsedYamlEntry { entry: kv(key: key, value: YamlString { value: lit.text }), next_index: lit.next_index } - } else if yaml_is_flow_sequence_text(text: rest) { - ParsedYamlEntry { entry: kv(key: key, value: yaml_parse_flow_sequence(text: rest)), next_index: index + 1 } - } else { - ParsedYamlEntry { entry: kv(key: key, value: yaml_parse_scalar_text(text: rest)), next_index: index + 1 } +// 1: leaves the one-line fast path (an indicator or whitespace). 2: a plain scalar the core schema +// may resolve to something other than a string. Absent: an ordinary start, always a string. +// 3: a core-schema word can start here (null/true/false and their spellings, and `~`), so the word +// map is consulted. 2: only a number can, so it is not. +data yaml_value_start_class: Map = { + "": 1, " ": 1, "\t": 1, "\"": 1, "'": 1, "[": 1, "]": 1, "{": 1, "}": 1, ",": 1, "|": 1, ">": 1, + "&": 1, "*": 1, "!": 1, "%": 1, "@": 1, "`": 1, "#": 1, "-": 1, "?": 1, ":": 1, + "0": 2, "1": 2, "2": 2, "3": 2, "4": 2, "5": 2, "6": 2, "7": 2, "8": 2, "9": 2, "+": 2, ".": 2, + "n": 3, "N": 3, "t": 3, "T": 3, "f": 3, "F": 3, "~": 3 +} + +// The key half of the one-line fast path, as ONE map lookup instead of a scan of the refused +// starts plus an unconditional core-schema word lookup. Class 1 starts leave the path outright; +// class 3 starts are the only ones a core-schema word can begin with, so only they consult the word +// map; every other start is an ordinary key. A trailing space or tab is still a slow start, because +// `uses : x` names the key `uses ` and only the general path decides what that means. +data yaml_key_start_class: Map = { + "": 1, " ": 1, "\t": 1, "-": 1, "?": 1, ":": 1, ",": 1, "[": 1, "]": 1, "\{": 1, "}": 1, + "#": 1, "&": 1, "*": 1, "!": 1, "|": 1, ">": 1, "'": 1, "\"": 1, "%": 1, "@": 1, "`": 1, + "0": 1, "1": 1, "2": 1, "3": 1, "4": 1, "5": 1, "6": 1, "7": 1, "8": 1, "9": 1, + "+": 1, ".": 1, "~": 1, + "n": 3, "N": 3, "t": 3, "T": 3, "f": 3, "F": 3 +} + +fn yaml_key_start_leaves_fast_path(c0: String, k: String) -> Bool { + match map_get(yaml_key_start_class, c0) { + Absent => false + Present { value: cls } => if cls == 1 { true } else { map_contains_key(yaml_core_schema_words, k) } + } +} + +fn yaml_key_leaves_fast_path(k: String) -> Bool { + if yaml_key_start_leaves_fast_path(c0: substring(s: k, start: 0, end: 1), k: k) { + true + } else { + trim(k) != k + } +} + +// TWO READINGS OF AN ENTRY CHUNK, ONE OF THEM AN ACCELERATOR. yaml_entry_general is THE reading: it +// decides every entry, refusal and line included. yaml_entry is the walk every mapping pays: it reads +// the three shapes a workflow is made of -- `key: plain` on one line, `key:` over a nested block, and +// `key: |` over a literal -- and hands every other chunk to the general reading. It decides no +// refusal of its own: the one-line tests only route, and its nested and literal arms call the same +// readers the general reading would call with the same arguments, so a refusal it returns is the one +// the general reading makes, typed, with its line and reason. +// +// The one-line tests only route. Exactly one `: `, no continuation, no `#` anywhere (so no comment), +// no final `:`, a key that starts ordinarily and is not a core-schema word, and a value that does +// not start with an indicator. +fn yaml_entry(c: String, n: Int) -> YamlEntryRead { + if string_contains(s: c, pattern: "\n") { + yaml_entry_lines(c: c, n: n, f: join(split(s: c, delimiter: "\n").take(n: 1), "")) + } else { + let parts = split(s: c, delimiter: ": ") + if (parts |> count) != 2 { + yaml_entry_general(c: c, n: n) + } else if string_contains(s: c, pattern: "#") { + yaml_entry_general(c: c, n: n) + } else { + let v = join(parts.skip(n: 1), "") + let k = join(parts.take(n: 1), "") + if ends_with(s: v, suffix: ":") || yaml_key_leaves_fast_path(k: k) { + yaml_entry_general(c: c, n: n) + } else { + match map_get(yaml_value_start_class, substring(s: v, start: 0, end: 1)) { + Absent => YamlEntryAccepted { entry: YamlKeyValue { key: k, value: YamlString { value: v } } } + Present { value: cls } => + if cls == 1 { + yaml_entry_indicator_value(c: c, n: n, k: k, v: v) + } else if cls == 2 { + YamlEntryAccepted { entry: YamlKeyValue { key: k, value: yaml_resolve_numeric(s: v) } } + } else { + YamlEntryAccepted { entry: YamlKeyValue { key: k, value: yaml_resolve_plain(s: v) } } + } } - } - } else if ends_with(s: content, suffix: ":") { - let key = substring(s: content, start: 0, end: content.length() - 1) - let body = yaml_skip_ignorable_lines(lines: lines, index: index + 1) - if yaml_has_line_at(lines: lines, index: body) && (yaml_indent_at(lines: lines, index: body) == indent + 2) { - match yaml_parse_block(lines: lines, index: body, indent: indent + 2) { - ParsedYamlBlock { value: nested, next_index: ni } => - ParsedYamlEntry { entry: kv(key: key, value: nested), next_index: ni } - YamlBlockRejected { reason: r } => YamlEntryRejected { reason: r } } + } + } +} + +// A MULTI-LINE ENTRY IS ROUTED BY ITS FIRST LINE, `f`, AND BY NOTHING ELSE: the route reads `f` +// alone, so `f` is the whole identity of that decision, and the first lines that open multi-line +// entries -- `env:`, `with:`, `steps:`, `run: |-` -- repeat across a document. What the rest of the +// chunk holds is read here, once, from the lines after `f`, which are passed down rather than found +// again. +fn yaml_entry_lines(c: String, n: Int, f: String) -> YamlEntryRead { + let route = yaml_first_line_route(f: f) + if route == 1 { + yaml_entry_value(k: substring(s: f, start: 0, end: f.length() - 1), r: yaml_nested(cont: substring(s: c, start: f.length() + 1, end: c.length()), col: n, same_col_seq: true)) + } else if route == 2 { + let k = join(split(s: f, delimiter: ": ").take(n: 1), "") + yaml_entry_value(k: k, r: yaml_block_scalar(header: substring(s: f, start: k.length() + 2, end: f.length()), cont: substring(s: c, start: f.length() + 1, end: c.length()), col: n)) + } else { + yaml_entry_general(c: c, n: n) + } +} + +// A one-line value that opens with an indicator -- a quoted scalar, a flow sequence, `{}` -- under a +// key the tests above already admitted: the general reading would split the line again only to +// reach the same value reader with the same key. +fn yaml_entry_indicator_value(c: String, n: Int, k: String, v: String) -> YamlEntryRead { + if starts_with(s: v, prefix: "|") { + yaml_entry_general(c: c, n: n) + } else { + yaml_entry_value(k: k, r: yaml_value(v: v, cont: "", col: n, same_col_seq: true)) + } +} + +// 1: `key:` with its value on the following lines, where the key is plain, starts ordinarily, is not +// a core-schema word and holds no `#`. 2: `key: |...`, a literal block scalar under a key the plain +// key reading admits and that does not end in whitespace. 0: every other first line, which the +// general reading decides. Nested rather than conjoined because `&&` and `||` evaluate both sides. +// (gunbc.recurring_failure_mode realization_arms_diverge_on_whether_the_program_refuses owns the +// eager-`&&` question itself; this module only declines to depend on short-circuiting.) +fn yaml_first_line_route(f: String) -> Int { + if !ends_with(s: f, suffix: ":") { + yaml_first_line_literal_route(f: f) + } else if string_contains(s: f, pattern: ": ") { + yaml_first_line_literal_route(f: f) + } else if ends_with(s: f, suffix: " :") || ends_with(s: f, suffix: "\t:") { + yaml_first_line_literal_route(f: f) + } else if string_contains(s: "\"' \t", pattern: substring(s: f, start: 0, end: 1)) { + yaml_first_line_literal_route(f: f) + } else if yaml_key_start_leaves_fast_path(c0: substring(s: f, start: 0, end: 1), k: substring(s: f, start: 0, end: f.length() - 1)) { + 0 + } else if string_contains(s: f, pattern: "#") { + 0 + } else { + 1 + } +} + +fn yaml_first_line_literal_route(f: String) -> Int { + let parts = split(s: f, delimiter: ": ") + let k = join(parts.take(n: 1), "") + if (parts |> count) != 2 { + 0 + } else if ends_with(s: k, suffix: " ") || ends_with(s: k, suffix: "\t") { + 0 + } else if !starts_with(s: join(parts.skip(n: 1), ""), prefix: "|") { + 0 + } else if string_contains(s: "\"'", pattern: substring(s: f, start: 0, end: 1)) { + 0 + } else if yaml_plain_key_refusal(k: k) != "" { + 0 + } else { + 2 + } +} + +fn yaml_entry_refused(line: Int, reason: String) -> YamlEntryRead { + YamlEntryRefused { line: line, reason: reason } +} + +// THE FIRST ELEMENT OF A LIST IS READ AS `join(xs.take(n: 1), "")` THROUGHOUT: `first` returns an +// optional, and unwrapping one per line costs more than the idiom, which yields that element or "" +// -- and every reader below either knows its list is a split result (never empty) or reads "" as no +// content and refuses it. + +fn yaml_first_line(c: String) -> String { + join(split(s: c, delimiter: "\n").take(n: 1), "") +} + +fn yaml_rest_lines(c: String, first: String) -> String { + if first == c { "" } else { substring(s: c, start: first.length() + 1, end: c.length()) } +} + +fn yaml_entry_general(c: String, n: Int) -> YamlEntryRead { + let ls = split(s: c, delimiter: "\n") + let f = join(ls.take(n: 1), "") + let cont = if (ls |> count) == 1 { "" } else { join(ls.skip(n: 1), "\n") } + if starts_with(s: f, prefix: "\"") || starts_with(s: f, prefix: "'") { + yaml_quoted_key_entry(f: f, cont: cont, n: n) + } else if starts_with(s: f, prefix: "?") { + yaml_entry_refused(line: 0, reason: "explicit `?` mapping keys are outside the supported subset") + } else if starts_with(s: f, prefix: "\t") { + yaml_entry_refused(line: 0, reason: "indentation contains a tab (YAML indents with spaces only)") + } else { + let parts = split(s: f, delimiter: ": ") + let k = join(parts.take(n: 1), "") + if (parts |> count) > 1 { + yaml_keyed_entry(k: trim(k), v: substring(s: f, start: k.length() + 2, end: f.length()), cont: cont, n: n) + } else if ends_with(s: f, suffix: ":") { + yaml_keyed_entry(k: trim(substring(s: f, start: 0, end: f.length() - 1)), v: "", cont: cont, n: n) } else { - ParsedYamlEntry { entry: kv(key: key, value: YamlNull), next_index: index + 1 } + yaml_entry_refused(line: 0, reason: yaml_not_an_entry_reason(t: f)) } + } +} + +fn yaml_keyed_entry(k: String, v: String, cont: String, n: Int) -> YamlEntryRead { + let bad = yaml_plain_key_refusal(k: k) + if bad != "" { + yaml_entry_refused(line: 0, reason: bad) } else { - YamlEntryRejected { reason: concat("malformed mapping entry: ", content) } + yaml_entry_value(k: k, r: if starts_with(s: v, prefix: "|") { yaml_block_scalar(header: v, cont: cont, col: n) } else { yaml_value(v: v, cont: cont, col: n, same_col_seq: true) }) } } -fn yaml_parse_mapping_entry_from_text(text: String, lines: List, index: Int, indent: Int) -> YamlEntryResult { - let parts = split(s: text, delimiter: ": ") - if parts.length() > 1 { - match parts.first() { - Absent => YamlEntryRejected { reason: "mapping entry has no key" } - Present { value: key } => - let rest = join(parts.skip(n: 1), ": ") - if yaml_has_unmodeled_inline_comment(text: rest) { - YamlEntryRejected { reason: "inline YAML comments after values are not modeled" } - } else if rest == "|" { - let lit = yaml_collect_literal_lines_from(lines: lines, index: index + 1, content_indent: indent + 4) - ParsedYamlEntry { entry: kv(key: key, value: YamlString { value: lit.text }), next_index: lit.next_index } - } else if yaml_is_flow_sequence_text(text: rest) { - ParsedYamlEntry { entry: kv(key: key, value: yaml_parse_flow_sequence(text: rest)), next_index: index + 1 } - } else { - ParsedYamlEntry { entry: kv(key: key, value: yaml_parse_scalar_text(text: rest)), next_index: index + 1 } +fn yaml_entry_value(k: String, r: YamlBlockResult) -> YamlEntryRead { + match r { + YamlBlockParsed { value: x } => YamlEntryAccepted { entry: YamlKeyValue { key: k, value: x } } + YamlBlockRefused { line: l, reason: why } => yaml_entry_refused(line: l, reason: why) + } +} + +fn yaml_not_an_entry_reason(t: String) -> String { + if (t == "---") || starts_with(s: t, prefix: "--- ") || (t == "...") { + "document markers are outside the supported subset (one document, no `---` or `...`)" + } else if starts_with(s: t, prefix: "%") { + "directives are outside the supported subset" + } else { + join(["expected a block mapping entry `key: value`, found `", t, "` (root scalars, multi-line plain scalars and tab-separated values are outside the supported subset)"], "") + } +} + +fn yaml_quoted_key_entry(f: String, cont: String, n: Int) -> YamlEntryRead { + let end = yaml_quoted_end(s: f, i: 1, quote: substring(s: f, start: 0, end: 1)) + let rest = if end < 0 { "" } else { substring(s: f, start: end, end: f.length()) } + if (end < 0) || !starts_with(s: concat(rest, " "), prefix: ": ") { + yaml_entry_refused(line: 0, reason: "a quoted key must close on its line and be followed directly by `:` and a space or the end of the line") + } else { + match yaml_quoted_scalar(v: substring(s: f, start: 0, end: end)) { + YamlScalarRefused { reason: why } => yaml_entry_refused(line: 0, reason: why) + ReadYamlScalar { value: decoded } => + match decoded { + YamlString { value: k } => yaml_entry_value(k: k, r: yaml_value(v: trim(substring(s: rest, start: 1, end: rest.length())), cont: cont, col: n, same_col_seq: true)) + _ => yaml_entry_refused(line: 0, reason: "a quoted key did not decode to a string") } } - } else if ends_with(s: text, suffix: ":") { - let key = substring(s: text, start: 0, end: text.length() - 1) - let body = yaml_skip_ignorable_lines(lines: lines, index: index + 1) - if yaml_has_line_at(lines: lines, index: body) && (yaml_indent_at(lines: lines, index: body) == indent + 4) { - match yaml_parse_block(lines: lines, index: body, indent: indent + 4) { - ParsedYamlBlock { value: nested, next_index: ni } => - ParsedYamlEntry { entry: kv(key: key, value: nested), next_index: ni } - YamlBlockRejected { reason: r } => YamlEntryRejected { reason: r } - } + } +} + +// A value that begins on its entry's (or item's) line, followed by `cont`. `col` is the column a +// nested block or a literal's content must exceed; `same_col_seq` admits a block sequence at `col` +// itself, which YAML allows for a mapping value and not for a sequence item. +fn yaml_value(v: String, cont: String, col: Int, same_col_seq: Bool) -> YamlBlockResult { + if v == "" { + yaml_nested(cont: cont, col: col, same_col_seq: same_col_seq) + } else { + let c0 = substring(s: v, start: 0, end: 1) + if string_contains(s: " \t", pattern: c0) { + yaml_value(v: trim(v), cont: cont, col: col, same_col_seq: same_col_seq) + } else if !string_contains(s: yaml_value_special_starts, pattern: c0) { + yaml_plain_value(v: v, cont: cont) + } else if !string_contains(s: "-?:", pattern: c0) { + yaml_special_value(v: v, c0: c0, cont: cont, col: col) + } else if yaml_indicator_opens_plain(v: v) { + yaml_plain_value(v: v, cont: cont) } else { - ParsedYamlEntry { entry: kv(key: key, value: YamlNull), next_index: index + 1 } + yaml_special_value(v: v, c0: c0, cont: cont, col: col) } + } +} + +fn yaml_plain_value(v: String, cont: String) -> YamlBlockResult { + let bad = yaml_plain_refusal(s: v) + if bad != "" { YamlBlockRefused { line: 0, reason: bad } } else { yaml_scalar_with_cont(value: yaml_resolve_plain(s: v), cont: cont) } +} + +fn yaml_scalar_read(read: YamlScalarResult, cont: String) -> YamlBlockResult { + match read { + ReadYamlScalar { value: x } => yaml_scalar_with_cont(value: x, cont: cont) + YamlScalarRefused { reason: why } => YamlBlockRefused { line: 0, reason: why } + } +} + +fn yaml_special_value(v: String, c0: String, cont: String, col: Int) -> YamlBlockResult { + if (c0 == "\"") || (c0 == "'") { + yaml_scalar_read(read: yaml_quoted_scalar(v: v), cont: cont) + } else if c0 == "[" { + yaml_scalar_read(read: yaml_flow_sequence(v: v), cont: cont) + } else if c0 == "|" { + yaml_block_scalar(header: v, cont: cont, col: col) + } else if v == "{}" { + yaml_scalar_with_cont(value: YamlMapping { entries: [] }, cont: cont) } else { - YamlEntryRejected { reason: concat("malformed sequence-item mapping entry: ", text) } + YamlBlockRefused { line: 0, reason: yaml_indicator_refusal(c: c0) } } } -fn yaml_parse_mapping_entries(lines: List, index: Int, indent: Int, acc: List) -> YamlBlockResult { - let current = yaml_skip_ignorable_lines(lines: lines, index: index) - let ln = yaml_line_entry(lines: lines, index: current) - if !yaml_has_line_at(lines: lines, index: current) { - ParsedYamlBlock { value: YamlMapping { entries: acc }, next_index: current } - } else if ln.indent != indent { - ParsedYamlBlock { value: YamlMapping { entries: acc }, next_index: current } - } else if starts_with(s: ln.text, prefix: "- ") { - ParsedYamlBlock { value: YamlMapping { entries: acc }, next_index: current } +fn yaml_indicator_refusal(c: String) -> String { + if c == "{" { + "flow mappings other than `{}` are outside the supported subset" + } else if c == ">" { + "folded block scalars (`>`) are outside the supported subset" + } else if (c == "&") || (c == "*") { + "anchors and aliases are outside the supported subset" + } else if c == "!" { + "tags are outside the supported subset" + } else if c == "#" { + yaml_inline_comment_reason + } else if (c == "-") || (c == "?") || (c == ":") { + join(["`", c, " ` cannot begin a value on the line of its key or sequence entry (explicit keys and inline block collections are outside the supported subset)"], "") } else { - match yaml_parse_mapping_entry(lines: lines, index: current, indent: indent) { - ParsedYamlEntry { entry: e, next_index: ni } => - yaml_parse_mapping_entries(lines: lines, index: ni, indent: indent, acc: concat(acc, [e])) - YamlEntryRejected { reason: r } => YamlBlockRejected { reason: r } - } + join(["`", c, "` cannot begin a plain scalar (section 7.3.3); quote the value"], "") } } -fn yaml_parse_seq_item(lines: List, index: Int, indent: Int) -> YamlSeqItemResult { - let content = yaml_text_at(lines: lines, index: index) - let item_text = substring(s: content, start: 2, end: content.length()) - if item_text == "|" { - let lit = yaml_collect_literal_lines_from(lines: lines, index: index + 1, content_indent: indent + 2) - ParsedYamlSeqItem { value: YamlString { value: lit.text }, next_index: lit.next_index } - } else if yaml_is_flow_sequence_text(text: item_text) { - ParsedYamlSeqItem { value: yaml_parse_flow_sequence(text: item_text), next_index: index + 1 } - } else if string_contains(s: item_text, pattern: ": ") || ends_with(s: item_text, suffix: ":") { - match yaml_parse_mapping_entry_from_text(text: item_text, lines: lines, index: index, indent: indent) { - ParsedYamlEntry { entry: first_entry, next_index: ni } => - match yaml_parse_mapping_entries(lines: lines, index: ni, indent: indent + 2, acc: [first_entry]) { - ParsedYamlBlock { value: v, next_index: fi } => ParsedYamlSeqItem { value: v, next_index: fi } - YamlBlockRejected { reason: r } => YamlSeqItemRejected { reason: r } - } - YamlEntryRejected { reason: r } => YamlSeqItemRejected { reason: r } - } - } else if yaml_has_unmodeled_inline_comment(text: item_text) { - YamlSeqItemRejected { reason: "inline YAML comments after values are not modeled" } +// A value on the lines after its key: a block deeper than `col`, a sequence at `col` where +// admitted, or -- when the continuation holds only empty lines and comments -- null. +fn yaml_nested(cont: String, col: Int, same_col_seq: Bool) -> YamlBlockResult { + let first = yaml_first_line(c: cont) + let m = yaml_content_indent(line: first) + if m < 0 { + yaml_nested_general(cont: cont, col: col, same_col_seq: same_col_seq) + } else if m > col { + yaml_block(text: substring(s: cont, start: m, end: cont.length()), n: m, seq: yaml_line_opens_seq(line: first), base: 1) + } else if same_col_seq && (m == col) && yaml_line_opens_seq(line: first) { + yaml_block(text: substring(s: cont, start: m, end: cont.length()), n: m, seq: true, base: 1) } else { - ParsedYamlSeqItem { value: yaml_parse_scalar_text(text: item_text), next_index: index + 1 } + YamlBlockRefused { line: 1, reason: yaml_unexpected_indentation_reason } } } -fn yaml_parse_seq_items(lines: List, index: Int, indent: Int, acc: List) -> YamlBlockResult { - let current = yaml_skip_ignorable_lines(lines: lines, index: index) - let ln = yaml_line_entry(lines: lines, index: current) - if !yaml_has_line_at(lines: lines, index: current) { - ParsedYamlBlock { value: YamlSequence { elements: acc }, next_index: current } - } else if ln.indent != indent { - ParsedYamlBlock { value: YamlSequence { elements: acc }, next_index: current } - } else if !starts_with(s: ln.text, prefix: "- ") { - ParsedYamlBlock { value: YamlSequence { elements: acc }, next_index: current } +// A line's indentation when it carries content, or -1 when it is empty or a comment. A function of the +// line alone, because the lines that open nested blocks repeat. +fn yaml_content_indent(line: String) -> Int { + let t = trim(line) + if t == "" { -1 } else if starts_with(s: t, prefix: "#") { -1 } else { line.length() - t.length() } +} + +fn yaml_line_opens_seq(line: String) -> Bool { + starts_with(s: concat(trim(line), " "), prefix: "- ") +} + +data yaml_unexpected_indentation_reason: String = "unexpected indentation: this line continues a scalar or sits between block levels (multi-line plain scalars and irregular indentation are outside the supported subset)" + +// A nested value whose continuation starts with empty lines or comments, or holds nothing. +fn yaml_nested_general(cont: String, col: Int, same_col_seq: Bool) -> YamlBlockResult { + if cont == "" { + YamlBlockParsed { value: YamlNull } } else { - match yaml_parse_seq_item(lines: lines, index: current, indent: indent) { - ParsedYamlSeqItem { value: v, next_index: ni } => - yaml_parse_seq_items(lines: lines, index: ni, indent: indent, acc: concat(acc, [v])) - YamlSeqItemRejected { reason: r } => YamlBlockRejected { reason: r } + let lines = split(s: cont, delimiter: "\n") + let at = yaml_first_content_offset(lines: lines, index: 0) + if at < 0 { + YamlBlockParsed { value: YamlNull } + } else { + let block = if at == 0 { cont } else { join(lines.skip(n: at), "\n") } + let m = yaml_indent_of(line: block) + if starts_with(s: substring(s: block, start: m, end: block.length()), prefix: "\t") { + YamlBlockRefused { line: at + 1, reason: "indentation contains a tab (YAML indents with spaces only)" } + } else if (m > col) || (same_col_seq && (m == col) && yaml_is_seq_item(t: yaml_first_line(c: substring(s: block, start: m, end: block.length())))) { + yaml_block(text: substring(s: block, start: m, end: block.length()), n: m, seq: yaml_starts_seq(text: substring(s: block, start: m, end: block.length())), base: at + 1) + } else { + YamlBlockRefused { line: at + 1, reason: "unexpected indentation: this line continues a scalar or sits between block levels (multi-line plain scalars and irregular indentation are outside the supported subset)" } + } } } } -fn yaml_parse_block(lines: List, index: Int, indent: Int) -> YamlBlockResult { - let current = yaml_skip_ignorable_lines(lines: lines, index: index) - if !yaml_has_line_at(lines: lines, index: current) { - YamlBlockRejected { reason: "expected a YAML block, found end of input" } - } else if yaml_indent_at(lines: lines, index: current) != indent { - YamlBlockRejected { reason: concat("indentation mismatch at line ", to_string(current)) } +fn yaml_item(c: String, n: Int) -> YamlBlockResult { + let f = join(split(s: c, delimiter: "\n").take(n: 1), "") + if f == "-" { + yaml_nested(cont: yaml_rest_lines(c: c, first: f), col: n, same_col_seq: false) + } else if !starts_with(s: f, prefix: "- ") { + YamlBlockRefused { line: 0, reason: "a block sequence's lines at its own indentation must all be `- ` items (a mapping entry cannot continue a sequence)" } } else { - let content = yaml_text_at(lines: lines, index: current) - if starts_with(s: content, prefix: "- ") { - yaml_parse_seq_items(lines: lines, index: current, indent: indent, acc: []) + let body = trim(substring(s: f, start: 2, end: f.length())) + let skip = f.length() - body.length() + if yaml_opens_mapping_entry(t: body) { + yaml_block(text: substring(s: c, start: skip, end: c.length()), n: n + skip, seq: false, base: 0) + } else if yaml_is_seq_item(t: body) { + yaml_block(text: substring(s: c, start: skip, end: c.length()), n: n + skip, seq: true, base: 0) } else { - yaml_parse_mapping_entries(lines: lines, index: current, indent: indent, acc: []) + yaml_value(v: body, cont: yaml_rest_lines(c: c, first: f), col: n, same_col_seq: false) } } } +// Whether a sequence item's text is the first entry of a compact mapping rather than a scalar. +fn yaml_opens_mapping_entry(t: String) -> Bool { + if !string_contains(s: t, pattern: ": ") && !ends_with(s: t, suffix: ":") { + false + } else if !string_contains(s: yaml_value_special_starts, pattern: substring(s: t, start: 0, end: 1)) { + true + } else if starts_with(s: t, prefix: "\"") || starts_with(s: t, prefix: "'") { + yaml_quoted_key_opens_entry(t: t) + } else { + false + } +} + +fn yaml_quoted_key_opens_entry(t: String) -> Bool { + let end = yaml_quoted_end(s: t, i: 1, quote: substring(s: t, start: 0, end: 1)) + (end > 0) && starts_with(s: concat(substring(s: t, start: end, end: t.length()), " "), prefix: ": ") +} + +// --------------------------------------------------------------------------------------------- +// DOCUMENT +// --------------------------------------------------------------------------------------------- + +fn yaml_source_lines(src: String) -> List { + let raw = split(s: src, delimiter: "\n") + let n = raw.length() + if (n > 0) && (join(raw.skip(n: n - 1), "") == "") { raw.take(n: n - 1) } else { raw } +} + +fn yaml_has_leading_tab(src: String) -> Bool { + if string_contains(s: src, pattern: "\t") { yaml_line_with_leading_tab(lines: yaml_source_lines(src: src), index: 0) > 0 } else { false } +} + +fn yaml_line_with_leading_tab(lines: List, index: Int) -> Int { + match get(xs: lines, index: index) { + Absent => 0 + Present { value: l } => + if string_contains(s: substring(s: l, start: 0, end: l.length() - trim(l).length()), pattern: "\t") { index + 1 } else { yaml_line_with_leading_tab(lines: lines, index: index + 1) } + } +} + +fn yaml_has_trailing_whitespace(src: String) -> Bool { + string_contains(s: src, pattern: " \n") || string_contains(s: src, pattern: "\t\n") || ends_with(s: src, suffix: " ") || ends_with(s: src, suffix: "\t") +} + +fn yaml_line_ending_in_whitespace(lines: List, index: Int) -> Int { + match get(xs: lines, index: index) { + Absent => 0 + Present { value: l } => if ends_with(s: l, suffix: " ") || ends_with(s: l, suffix: "\t") { index + 1 } else { yaml_line_ending_in_whitespace(lines: lines, index: index + 1) } + } +} + fn ingest_yaml_source(src: String) -> YamlIngestResult { - let lines = yaml_lines_from_source(src: src) - if lines.length() == 0 { - YamlIngestRejected { reason: "empty document" } - } else { - match yaml_parse_block(lines: lines, index: 0, indent: 0) { - ParsedYamlBlock { value: v, next_index: ni } => - if ni == lines.length() { - IngestedYaml { value: v } - } else { - YamlIngestRejected { reason: concat("trailing content at line ", to_string(ni)) } + if yaml_contains_refused_character(text: src) { + let cp = yaml_first_refused_character(text: src) + YamlIngestRejected { line: yaml_line_containing(lines: yaml_source_lines(src: src), pattern: from_code_point(cp: cp), index: 0), reason: yaml_refused_character_reason(cp: cp) } + } else if yaml_has_leading_tab(src: src) { + YamlIngestRejected { line: yaml_line_with_leading_tab(lines: yaml_source_lines(src: src), index: 0), reason: "a tab in a line's leading whitespace is outside the supported subset (YAML indents with spaces only; a literal line may not begin with a tab either)" } + } else if yaml_has_trailing_whitespace(src: src) { + YamlIngestRejected { line: yaml_line_ending_in_whitespace(lines: yaml_source_lines(src: src), index: 0), reason: "a line ends in whitespace, which is outside the supported subset (it would make an empty line indistinguishable from literal content)" } + } else { + let lines = yaml_source_lines(src: src) + let at = yaml_first_content_offset(lines: lines, index: 0) + if at < 0 { + YamlIngestRejected { line: 0, reason: "the document has no content node" } + } else { + let root = if at == 0 { join(lines, "\n") } else { join(lines.skip(n: at), "\n") } + if starts_with(s: root, prefix: " ") || starts_with(s: root, prefix: "\t") { + YamlIngestRejected { line: at + 1, reason: "the root block must start at column 0" } + } else { + match yaml_block(text: root, n: 0, seq: yaml_starts_seq(text: root), base: 0) { + YamlBlockParsed { value: v } => IngestedYaml { value: v } + YamlBlockRefused { line: l, reason: why } => YamlIngestRejected { line: at + l + 1, reason: why } } - YamlBlockRejected { reason: r } => YamlIngestRejected { reason: r } + } } } } @@ -406,6 +1424,6 @@ fn ingest_yaml_source(src: String) -> YamlIngestResult { fn yaml_source_parses(src: String) -> Bool { match ingest_yaml_source(src: src) { IngestedYaml { value: _ } => true - YamlIngestRejected { reason: _ } => false + YamlIngestRejected { line: _, reason: _ } => false } } diff --git a/dag/extdeps/languages/yaml/types.dag b/dag/extdeps/languages/yaml/types.dag index d174d5a37c9..ba338dbe2c5 100644 --- a/dag/extdeps/languages/yaml/types.dag +++ b/dag/extdeps/languages/yaml/types.dag @@ -2,6 +2,7 @@ module extdeps.languages.yaml.types import extdeps.external_authority { ExternalAuthority } import extdeps.uri { Uri, Https } +import v2.std.optional { Present, Absent } data extdeps_external_authority_anchor: ExternalAuthority = ExternalAuthority { uri: Uri { @@ -96,3 +97,16 @@ fn yaml_sequence(elements: List) -> YamlValue { fn kv(key: String, value: YamlValue) -> YamlKeyValue { YamlKeyValue { key: key, value: value } } + +// THE VALUE AT `key` IN A MAPPING'S ENTRIES. Keys of an ingested or emitted mapping are unique (both +// directions refuse a duplicate), so there is at most one. +type YamlKeyLookup + = YamlKeyPresent { value: YamlValue } + | YamlKeyAbsent + +fn yaml_lookup(entries: List, key: String) -> YamlKeyLookup { + match get(xs: filter(entries, e => e.key == key), index: 0) { + Present { value: e } => YamlKeyPresent { value: e.value } + Absent => YamlKeyAbsent + } +} diff --git a/dag/gunbc/action_use_admission.dag b/dag/gunbc/action_use_admission.dag index 6b1f574ca9b..05f6fd59927 100644 --- a/dag/gunbc/action_use_admission.dag +++ b/dag/gunbc/action_use_admission.dag @@ -2,10 +2,9 @@ module gunbc.action_use_admission import std.types { Bool, Int, List, NonEmptyStr, String, commit_sha_text_holds } import extdeps.external_authority { ExternalAuthority } -import extdeps.languages.yaml.ingest { yaml_indent_of } -import std.dissolution { unbound_dissolution } -import std.roster_frontier { FrontierRow, frontier_row_decl } -import std.decl_ref { decl_ref } +import v2.std.optional { Present, Absent } +import extdeps.languages.yaml.ingest { ingest_yaml_source, IngestedYaml, YamlIngestRejected } +import extdeps.languages.yaml.types { YamlValue, YamlKeyValue, YamlString, YamlSequence, YamlMapping, yaml_lookup, YamlKeyPresent, YamlKeyAbsent } import extdeps.github.actions { Workflow, Job, Step, UsesStep, RunStep } import extdeps.github.action_release { ActionRelease, ActionExecutionMechanism, ActionManifestReading, @@ -45,9 +44,10 @@ import extdeps.github.action.rust_cache { rust_cache_manifest_readings } // Two routes reach the rule, and both must pass it: // - every modeled workflow, before it is emitted (admit_workflow_action_uses, called by each // generator), and -// - every realized workflow file in .github/workflows, whether generated or hand-authored, -// through realized_workflow_action_uses and admit_realized_action_use. The census over the -// live tree lives in test.claim.action_use_admission_witness. +// - every realized workflow file in .github/workflows, whether generated or hand-authored, read +// by the modeled YAML reader (realized_workflow_reading) and admitted per distinct use +// (realized_distinct_uses, admit_realized_action_use). The census over the live tree lives in +// test.claim.action_use_admission_witness, one claim per file. // The second route closes what the first cannot see: a file nobody emitted. // --------------------------------------------------------------------------------------------- @@ -90,8 +90,12 @@ data repository_action_selections: List = [ github_script_action, ] +data repository_action_selection_texts: Map = fold(repository_action_selections, init: empty_map(), f: fn(m, r) { + map_insert(m, action_release_uses_text(action_release: r), true) +}) + fn realized_use_is_a_repository_selection(uses_text: String) -> Bool { - any(repository_action_selections, r => action_release_uses_text(action_release: r) == uses_text) + map_contains_key(repository_action_selection_texts, uses_text) } // WHICH PLATFORM EPOCH THIS REPOSITORY'S RUNNERS ARE IN. Choosing it is a repository fact: every @@ -181,7 +185,26 @@ fn admit_declared_mechanism(readings: List, action_releas // The readings are a parameter so the refusal arms can be driven by a supplied roster; production // always passes action_manifest_readings (admit_action_release). fn admit_action_release_against(readings: List, action_release: ActionRelease) -> ActionUseAdmission { - let found = readings_for_release(readings: readings, action_release: action_release) + admit_action_release_found(readings: readings, action_release: action_release, found: readings_for_release(readings: readings, action_release: action_release)) +} + +// THE REPOSITORY'S READINGS, KEYED BY THE PRODUCER IDENTITY THEY JOIN ON. action_release_uses_text +// spells owner/repo@commit, exactly what action_release_same_producer compares, so a lookup here +// finds what readings_for_release finds over action_manifest_readings, read once per claim instead +// of once per admission. +data action_manifest_readings_by_producer: Map> = fold(action_manifest_readings, init: empty_map(), f: fn(m, r) { + let key = action_release_uses_text(action_release: r.action_release) + map_insert(m, key, concat(match map_get(m, key) { Present { value: rs } => rs Absent => [] }, [r])) +}) + +fn manifest_readings_for_producer(uses_text: String) -> List { + match map_get(action_manifest_readings_by_producer, uses_text) { + Present { value: rs } => rs + Absent => [] + } +} + +fn admit_action_release_found(readings: List, action_release: ActionRelease, found: List) -> ActionUseAdmission { let n = found |> count if n == 0 { RefusedUnobservedDeclaration { action_release: action_release } @@ -195,15 +218,16 @@ fn admit_action_release_against(readings: List, action_re } fn admit_action_release(action_release: ActionRelease) -> ActionUseAdmission { - admit_action_release_against(readings: action_manifest_readings, action_release: action_release) + admit_action_release_found(readings: action_manifest_readings, action_release: action_release, found: manifest_readings_for_producer(uses_text: action_release_uses_text(action_release: action_release))) } // THE ONE ADMISSION BOTH ROUTES USE FOR A WORKFLOW'S OWN `uses:`: executed exactly AND a repository // selection. Nested releases inside a composite are admitted on runtime alone -- the composite's // producer chose them, not this repository. fn admit_selected_action_use(action_release: ActionRelease) -> ActionUseAdmission { - let a = admit_action_release(action_release: action_release) - if action_use_admitted(a: a) && !realized_use_is_a_repository_selection(uses_text: action_release_uses_text(action_release: action_release)) { + let uses_text = action_release_uses_text(action_release: action_release) + let a = admit_action_release_found(readings: action_manifest_readings, action_release: action_release, found: manifest_readings_for_producer(uses_text: uses_text)) + if action_use_admitted(a: a) && !realized_use_is_a_repository_selection(uses_text: uses_text) { RefusedUnselectedRelease { action_release: action_release } } else { a @@ -287,178 +311,166 @@ type RealizedWorkflowReading = RealizedWorkflowUses { workflow_path: String, uses: List } | RealizedWorkflowUnreadable { workflow_path: String, reason: String } -// SCAFFOLD, OPERATOR-APPROVED -- THE MARK LIVES HERE, ON THE CARRIER. realized_workflow_action_uses -// below is a second reader of YAML beside extdeps.languages.yaml.ingest ingest_yaml_source. The -// operator approved landing it (gunb-ai/gunbc#11663, 2026-09-19: "you can fix it in a followup") -// because the census must cover every executed workflow and the modeled reader did not fit one -// claim's eval-step budget on the largest committed workflow. The follow-up cut the modeled -// reader's per-line cost 2-3x at its owning links (a line table read once, a digit test and an -// unescape that were quadratic per value, literals collected as a slice); what remains is the -// recursive parse's per-line constant, re-derived by claim_batch eval_steps over -// ingest_yaml_source on .github/workflows/fleet-converge.yml. -data realized_workflow_projection_frontier_rows: List = [ - frontier_row_decl( - ref: decl_ref(module_path: "gunbc.action_use_admission", decl_name: "realized_workflow_action_uses"), - reason: "a line projection of workflow YAML standing beside the modeled reader, operator-approved on gunb-ai/gunbc#11663 because the modeled reader exceeded the new-witness eval-step budget on the largest committed workflow; it reads a use only as `uses: ` at the head of a line, classifies the site by the value's shape, and refuses every other spelling of a `uses` key", - dissolution: unbound_dissolution(description: "TRIGGER: extdeps.languages.yaml.ingest ingest_yaml_source parses the largest file under .github/workflows within v2.workflow.required_floor's new-witness eval-step budget, with that file's uses admitted in the same claim. SUFFICIENT FOR: the census reading every realized workflow through ingest_yaml_source, one claim per file with a listing-population join, and the deletion of realized_workflow_action_uses, uses_line_reading, uses_value_text and uses_site_of together with this row. NOT satisfied by the modeled reader fitting on the smaller workflows only: the census closes over every executed file or it is not the census"), - ) -] - -// THE REALIZED-FILE PROJECTION DOES STRUCTURAL WORK ONLY ON THE LINES THAT CAN CARRY A USE. The -// census must cover every file GitHub executes, generated ones included, and neither a full YAML -// parse nor a scan that interprets every line's structure fits one claim's eval-step budget -// (claim_batch re-derives the cost on test.claim.action_use_admission_witness). Every line pays one -// cheap test -- whether it opens with a double-quoted key carrying an escape -- because that is the -// one spelling of a `uses` key that need not contain the token `uses`, so it cannot sit behind the -// token filter; every other test runs only on lines that contain `uses`. Admission needs two things per use, and neither needs -// the surrounding structure: -// - the value, from the line's own `uses:` key; and -// - the SITE, from the value's own shape: GitHub resolves a job-level `uses:` only as a reusable -// workflow, which is spelled `//.github/workflows/@` or -// `./.github/workflows/`, and a step-level one never names a workflow file. A job-level -// `uses:` naming a plain action is a workflow GitHub itself rejects, so no substituted runtime -// can run from it. -// Every line that names `uses` is examined structurally, including lines inside script bodies: a script line -// spelled exactly as a `uses:` key is treated as a use and must be admitted, which fails closed. -// Unmodeled spellings refuse rather than pass: a quoted or flow-mapping `uses` key, an anchor or -// alias value, and an empty value. Whole-file facts are read once: exactly one top-level `jobs`, and -// no runner override that makes GitHub execute node20 as declared -// (repository_node_runtime_epoch override_variable). A refusal quotes the offending line. -fn uses_value_text(value: String) -> String { - let bare = if string_contains(s: value, pattern: " #") { fold(split(s: value, delimiter: " #"), init: "", f: fn(acc, part) { if acc == "" { part } else { acc } }) } else { value } - if (bare.length() >= 2) && ((starts_with(s: bare, prefix: "\"") && ends_with(s: bare, suffix: "\"")) || (starts_with(s: bare, prefix: "'") && ends_with(s: bare, suffix: "'"))) { - substring(s: bare, start: 1, end: bare.length() - 1) +// THE REALIZED FILE IS READ BY THE MODELED READER. Every file GitHub executes goes through +// extdeps.languages.yaml.ingest ingest_yaml_source, and its uses are taken from the decoded +// structure, where GitHub takes them: `jobs..uses` is a job-level use, which GitHub resolves only +// as a reusable workflow, and `jobs..steps[].uses` is a step-level one. A `uses` key anywhere +// else is not one GitHub executes, and a script line that spells `uses:` is script content. A +// document the reader refuses is unreadable with the reader's located reason, never read as "no +// uses"; so is a `jobs` that is not a mapping, a job or step that is not a mapping, a `steps` that +// is not a sequence, and a `uses` that is not a string. The runner override that makes GitHub +// execute node20 as declared (repository_node_runtime_epoch override_variable) is refused wherever +// the file spells it, script bodies included, because a script can export it. +fn realized_workflow_reading(path: String, source: String) -> RealizedWorkflowReading { + if string_contains(s: source, pattern: repository_node_runtime_epoch.override_variable as String) { + RealizedWorkflowUnreadable { workflow_path: path, reason: join(["`", repository_node_runtime_epoch.override_variable as String, "` makes the runner execute node20 as declared; a workflow may not set it, in any scope or script"], "") } } else { - bare + match ingest_yaml_source(src: source) { + YamlIngestRejected { line: l, reason: r } => + RealizedWorkflowUnreadable { workflow_path: path, reason: join(["the YAML reader refused line ", to_string(l), ": ", r], "") } + IngestedYaml { value: doc } => + match doc { + YamlMapping { entries: top } => + match yaml_lookup(entries: top, key: "jobs") { + YamlKeyAbsent => RealizedWorkflowUnreadable { workflow_path: path, reason: "the workflow has no top-level `jobs` mapping" } + YamlKeyPresent { value: jobs } => + match jobs { + YamlMapping { entries: job_entries } => realized_jobs_reading(path: path, jobs: job_entries) + _ => RealizedWorkflowUnreadable { workflow_path: path, reason: "the top-level `jobs` is not a mapping" } + } + } + _ => RealizedWorkflowUnreadable { workflow_path: path, reason: "the workflow document is not a mapping" } + } + } } } -fn uses_site_of(value: String) -> ActionUseSite { - if string_contains(s: value, pattern: ".github/workflows/") { JobReusableWorkflowUse } else { StepActionUse } +// ONE READING PER USE SITE, IN DOCUMENT ORDER: a use, or the refusal that site produces. The job-level +// `uses:` comes before the job's steps, and jobs and steps keep their order, so the first refusal in +// the flattened readings is the one a scan that stopped at it would have reported -- and the scan no +// longer carries a stop flag through every step, or copies its accumulated uses at every use. +type RealizedUseRead + = RealizedUseFound { found: RealizedActionUse } + | RealizedUseRefused { reason: String } + +fn realized_jobs_reading(path: String, jobs: List) -> RealizedWorkflowReading { + let reads = flat_map(jobs, job => realized_job_reads(path: path, job_id: job.key, job: job.value)) + match get(xs: flat_map(reads, r => match r { RealizedUseRefused { reason: why } => [why] RealizedUseFound { found: _ } => [] }), index: 0) { + Present { value: why } => RealizedWorkflowUnreadable { workflow_path: path, reason: why } + Absent => RealizedWorkflowUses { workflow_path: path, uses: flat_map(reads, r => match r { RealizedUseFound { found: u } => [u] RealizedUseRefused { reason: _ } => [] }) } + } } -type UsesLineReading - = UsesLineNotAUse - | UsesLineUse { use: RealizedActionUse } - | UsesLineRefused { reason: String } - -fn uses_line_reading(path: String, line: String) -> UsesLineReading { - let text = substring(s: line, start: yaml_indent_of(line: line), end: line.length()) - let body = if starts_with(s: text, prefix: "- ") { substring(s: text, start: 2, end: text.length()) } else { text } - let why = if starts_with(s: body, prefix: "{") { - "a flow-mapping step that names `uses` is not modeled" - } else if starts_with(s: body, prefix: "\"uses\":") || starts_with(s: body, prefix: "'uses':") { - "a quoted `uses` key is not modeled" - } else { "" } - if why != "" { - UsesLineRefused { reason: join([why, " -- at `", line, "`"], "") } - } else if (body == "uses:") || starts_with(s: body, prefix: "uses: ") { - let v = uses_value_text(value: if body == "uses:" { "" } else { substring(s: body, start: 6, end: body.length()) }) - if v == "" { - UsesLineRefused { reason: join(["a `uses:` with no value -- at `", line, "`"], "") } - } else if starts_with(s: v, prefix: "*") || starts_with(s: v, prefix: "&") { - UsesLineRefused { reason: join(["a YAML anchor or alias as a `uses` value is not modeled -- at `", line, "`"], "") } - } else { - UsesLineUse { use: RealizedActionUse { workflow_path: path, job_id: "", site: uses_site_of(value: v), uses_text: v } } - } - } else { - UsesLineNotAUse +fn realized_use_read(path: String, job_id: String, site: ActionUseSite, where_: String, v: YamlValue) -> RealizedUseRead { + match v { + YamlString { value: t } => + if t == "" { + RealizedUseRefused { reason: join(["job ", job_id, ": ", where_, " `uses` is empty"], "") } + } else { + RealizedUseFound { found: RealizedActionUse { workflow_path: path, job_id: job_id, site: site, uses_text: t } } + } + _ => RealizedUseRefused { reason: join(["job ", job_id, ": ", where_, " `uses` is not a string"], "") } } } -// ANY `uses` FOLLOWED BY A COLON IS A KEY SPELLING UNTIL PROVEN OTHERWISE. The projection reads a -// use only in the shapes it models; every other line that spells a `uses` key -- a flow sequence or -// mapping anywhere on the line, a tab separator, a space before the colon -- refuses rather than -// reading as "no use here", which would be the silent narrow DESIGN section 5 forbids. Comment lines -// are the only exemption, and prose that says "uses" with no colon is not a key. -fn uses_line_is_comment(line: String) -> Bool { - starts_with(s: substring(s: line, start: yaml_indent_of(line: line), end: line.length()), prefix: "#") +fn realized_step_reads(path: String, job_id: String, step: YamlValue) -> List { + match step { + YamlMapping { entries: step_fields } => + match yaml_lookup(entries: step_fields, key: "uses") { + YamlKeyAbsent => [] + YamlKeyPresent { value: v } => [realized_use_read(path: path, job_id: job_id, site: StepActionUse, where_: "a step", v: v)] + } + _ => [RealizedUseRefused { reason: join(["job ", job_id, ": a step is not a mapping"], "") }] + } } -// WHITESPACE BETWEEN THE KEY AND ITS COLON IS NOT SIGNIFICANT, so it is removed before the test -// rather than enumerated: `uses:`, `uses :`, `uses :` and `uses\t:` are one spelling here. -fn uses_line_spells_a_uses_key(line: String) -> Bool { - string_contains(s: replace(replace(line, " ", ""), "\t", ""), pattern: "uses:") +fn realized_job_reads(path: String, job_id: String, job: YamlValue) -> List { + match job { + YamlMapping { entries: fields } => { + let named = filter(fields, e => (e.key == "uses") || (e.key == "steps")) + let job_use = match yaml_lookup(entries: named, key: "uses") { + YamlKeyAbsent => [] + YamlKeyPresent { value: v } => [realized_use_read(path: path, job_id: job_id, site: JobReusableWorkflowUse, where_: "the job-level", v: v)] + } + match yaml_lookup(entries: named, key: "steps") { + YamlKeyAbsent => job_use + YamlKeyPresent { value: steps } => + match steps { + YamlSequence { elements: els } => concat(job_use, flat_map(els, st => realized_step_reads(path: path, job_id: job_id, step: st))) + _ => concat(job_use, [RealizedUseRefused { reason: join(["job ", job_id, ": `steps` is not a sequence"], "") }]) + } + } + } + _ => [RealizedUseRefused { reason: join(["job ", job_id, " is not a mapping"], "") }] + } } -// A QUOTED KEY WITH AN ESCAPE CAN SPELL `uses` WITHOUT CONTAINING IT (`"us\u0065s":`), and this -// projection does not decode escapes, so it cannot know what such a key names. It refuses a line on -// whose body OPENS with a double-quoted key whose closing `":` is preceded by a backslash, rather -// than guessing what the key is. Quoted text with no `":` (a script argument), and a value carrying -// escaped quotes, are not keys and pass; see the boundary note on the function below. -fn uses_line_has_escaped_quoted_key(line: String) -> Bool { - if !string_contains(s: line, pattern: "\":") { - false - } else if !string_contains(s: line, pattern: "\\") { - false - } else { - uses_line_quoted_key_has_escape(line: line) - } +// THE CENSUS OVER THE LARGEST WORKFLOW IS JUDGED AGAINST A LINE, NOT A POINT. Its subject, +// .github/workflows/fleet-converge.yml, is a projection of authorities other lanes edit, and it grows +// without anyone touching this module; a flat eval-step budget therefore turns this REQUIRED claim red +// on someone else's change. Since 2026-09-23 the claim is rostered in v2.workflow.required_floor +// corpus_census_roster and judged against corpus_census_eval_step_allowance: a declared base plus a +// declared per-entry rate times the entry count the census's own parse reads from that file. That +// model was admitted by a ruling (stern-carp-604, 2026-09-23, an operator DEFAULT-approval on a +// fifteen-minute timeout, not a deliberated ruling). It is not a raised budget: the rate is the line, +// and a claim whose steps per entry exceed it is red exactly as before, so growth cannot hide a +// cost-shape regression. +// +// WHAT WAS CUT BEFORE THAT MODEL WAS ADMITTED, AND WHY BOTH ARE NEEDED. The work removed is real: the +// reader's accepted-path block walk decides each shape once from the text that decides it, and the +// census traversal reads each use site once with no copied accumulator. Measured that way the claim +// reached a floor over the flat budget, where what remains is per distinct line and per block and +// near its node floor; the allowance exists because the subject outgrows any constant, not because +// the cutting stopped early. +// +// HOW THE NEXT CUT IS CHOSEN, if the rate is ever approached: by measurement, not by reading (DESIGN +// section 6). Add a claim over a document of ONE repeated shape with DISTINCT content per +// repetition -- identical repetitions are memoized by the evaluator and read as nearly free, which +// makes a naive repeated-shape probe say nothing -- read eval_steps from its [witness] line at two +// sizes, and the slope names the path that shape spends in. No standing probe is committed: this is a +// method, not an instrument (review 69135). +// +// A RAISED BUDGET, A SECOND READER BESIDE THE MODELED ONE, AND A DECLARED COST DROP ARE STILL NOT +// REMEDIES: the first moves the line rather than the work -- and the per-entry allowance is not one, +// because its rate is a declared policy constant with its own reason, never a number fitted to this +// claim -- and the other two are the scaffold this change deleted and the debt row DESIGN section +// 4b(3) keeps for capability losses. +// +// The numbers are not transcribed here (DESIGN section 6: name the instrument). +type RealizedCensusCostStanding { + subject: NonEmptyStr + instrument: NonEmptyStr + remedy: NonEmptyStr + refused_remedies: List } -// The branches above are what every line pays -- written as branches, not `&&`, because the -// interpreter evaluates both operands of `&&` (gunbc.recurring_failure_mode -// realization_arms_diverge_on_whether_the_program_refuses), which would run this split on every -// line. Only a line carrying both a `":` and a backslash is split. -// ONLY A KEY AT THE HEAD OF THE LINE, AND THE BOUNDARY IS MEASURED RATHER THAN ASSUMED. A wider -// rule was tried -- refuse wherever a backslash precedes a `":` on the line, so a flow-style -// `[{"us\u0065s": ...}]` would be caught too -- and it refused two committed workflow lines that -// are VALUES, not keys: fleet-converge's `group: "...{\"srv1\": ...}"`, and a witnesses.yml `run:` -// line whose shell script prints literal JSON (`{"version": 1, "receipt": {...}}`). Restricting the -// opening quote to a key position (`{"` / `,"`) does not separate them either: that JSON is -// spelled exactly so. A line projection cannot tell a YAML key from JSON inside a script string -// without the structure it declines to parse, so the refusal stays at the head of the line, and the -// flow-style escaped key is declared UNSEEN on the failure-mode row rather than claimed. -fn uses_line_quoted_key_has_escape(line: String) -> Bool { - let text = substring(s: line, start: yaml_indent_of(line: line), end: line.length()) - let body = if starts_with(s: text, prefix: "- ") { substring(s: text, start: 2, end: text.length()) } else { text } - if !starts_with(s: body, prefix: "\"") { - false - } else { - let parts = split(s: body, delimiter: "\":") - string_contains(s: join(parts.take(n: parts.length() - 1), "\":"), pattern: "\\") - } +data realized_census_cost_standing: RealizedCensusCostStanding = RealizedCensusCostStanding { + subject: "fleet-converge.yml" as NonEmptyStr, + instrument: "claim_batch --entry dag/test/claim/action_use_admission_witness_test.dag --functions census_fleet_converge_uses_are_modeled_and_executed_exactly, against v2.workflow.required_floor corpus_census_eval_step_allowance (the claim is rostered in corpus_census_roster; the floor run prints its derived budget on a [floor-corpus-census] line)" as NonEmptyStr, + remedy: "cut real work, never the population: the per-entry cost of extdeps.languages.yaml.ingest's accepted-path walk, and the census traversal's own cost shape (realized_jobs_reading, realized_distinct_uses), each chosen by measurement rather than by reading -- a claim over a document of one repeated shape with DISTINCT content per repetition (identical repetitions are memoized and read as nearly free), eval_steps from its [witness] line, and the path that shape spends in" as NonEmptyStr, + refused_remedies: [ + "raising the claim's eval-step budget, including its per-entry rate: v2.workflow.required_floor corpus_census_allowance_policy is floor policy with its own reason, admitted by the 2026-09-23 ruling, and is never refitted to this claim" as NonEmptyStr, + "a second, cheaper reader beside extdeps.languages.yaml.ingest" as NonEmptyStr, + "a declared cost drop or debt row for the claim" as NonEmptyStr, + "dropping a workflow file from the census, or splitting one file's parse across claims" as NonEmptyStr, + "respelling a shared predicate a second time to buy margin: the one respelling that happened, std.types commit_sha_text_holds, was admitted on ITS OWN constant-factor cost by an operator ruling and not on this budget, and it is not a precedent this row extends" as NonEmptyStr, + ], } -type UsesScan { +// EACH DISTINCT USE IS ADMITTED ONCE. Admission is a function of a use's site and text alone, and +// a workflow repeats the same few releases across its jobs, so admitting every occurrence asked one +// question many times over (DESIGN section 2: a repeated demand is removed, not cached). The first +// occurrence of each (site, text) stands for the rest, and a refusal names its job. +type RealizedUseSet { + seen: Map uses: List - refusal: String } -// `job_id` is left empty: the projection does not build the job mapping, and a refusal carries the -// offending line's own text as its locus. -fn realized_workflow_action_uses(path: String, source: String) -> RealizedWorkflowReading { - if string_contains(s: source, pattern: repository_node_runtime_epoch.override_variable as String) { - RealizedWorkflowUnreadable { workflow_path: path, reason: join(["`", repository_node_runtime_epoch.override_variable as String, "` makes the runner execute node20 as declared; a workflow may not set it, in any scope or script"], "") } - } else if ((split(s: concat("\n", source), delimiter: "\njobs:") |> count) != 2) { - RealizedWorkflowUnreadable { workflow_path: path, reason: "the workflow does not carry exactly one top-level `jobs` key" } - } else { - let scan = fold(split(s: source, delimiter: "\n"), init: UsesScan { uses: [], refusal: "" }, f: fn(acc, line) { - if acc.refusal != "" { - acc - } else if uses_line_has_escaped_quoted_key(line: line) { - UsesScan { uses: acc.uses, refusal: join(["a quoted key with an escape, which this projection cannot decode -- at `", line, "`"], "") } - } else if !string_contains(s: line, pattern: "uses") { - acc - } else { - match uses_line_reading(path: path, line: line) { - UsesLineNotAUse => - if uses_line_is_comment(line: line) || !uses_line_spells_a_uses_key(line: line) { - acc - } else { - UsesScan { uses: acc.uses, refusal: join(["a `uses` key this projection cannot classify (flow style, a tab, or a key not at the head of the line) -- at `", line, "`"], "") } - } - UsesLineUse { use: u } => UsesScan { uses: concat(acc.uses, [u]), refusal: acc.refusal } - UsesLineRefused { reason: r } => UsesScan { uses: acc.uses, refusal: r } - } - } - }) - if scan.refusal != "" { - RealizedWorkflowUnreadable { workflow_path: path, reason: scan.refusal } - } else { - RealizedWorkflowUses { workflow_path: path, uses: scan.uses } - } - } +fn realized_distinct_uses(uses: List) -> List { + fold(uses, init: RealizedUseSet { seen: empty_map(), uses: [] }, f: fn(acc, u) { + let key = concat(match u.site { StepActionUse => "step " JobReusableWorkflowUse => "job " }, u.uses_text) + if map_contains_key(acc.seen, key) { acc } else { RealizedUseSet { seen: map_insert(acc.seen, key, true), uses: acc.uses |> list_push(u) } } + }).uses } type UsesSpelling { @@ -479,9 +491,7 @@ fn uses_spelling(text: String) -> UsesSpelling { } fn realized_release_candidates(owner: String, repo: String, commit: String) -> List { - fold(action_manifest_readings, init: [], f: fn(acc, r) { - if (r.action_release.owner as String) == owner && (r.action_release.repo as String) == repo && (r.action_release.commit as String) == commit { concat(acc, [r]) } else { acc } - }) + manifest_readings_for_producer(uses_text: join([owner, "/", repo, "@", commit], "")) } // Every spelling GitHub accepts in a step `uses:` is classified, and only one reaches admission: diff --git a/dag/gunbc/assimilate/bmc_token_federation.dag b/dag/gunbc/assimilate/bmc_token_federation.dag index a4fcae1b784..6c94db001cf 100644 --- a/dag/gunbc/assimilate/bmc_token_federation.dag +++ b/dag/gunbc/assimilate/bmc_token_federation.dag @@ -26,7 +26,7 @@ import extdeps.github.actions { } import gunbc.action_use_admission { checkout_action, google_auth_action } import extdeps.languages.yaml.types { YamlKeyValue, kv, yaml_string } -import extdeps.languages.yaml.emit { serialize_yaml } +import extdeps.languages.yaml.emit { emit_yaml, YamlEmitResult } import extdeps.languages.yaml.gha_workflow { project_workflow_to_yaml } import gunbc.runner_spec_from_offer { gunbc_ci_runner_spec } import std.types { NonEmptyStr } @@ -243,6 +243,6 @@ data bmc_token_smoke_workflow: Workflow = { // rename touches BMC production files and their witnesses, and it is separable from the value // consolidation this change lands. -fn emit_bmc_token_smoke_workflow_yaml() -> String { - serialize_yaml(v: project_workflow_to_yaml(workflow: bmc_token_smoke_workflow)) +fn emit_bmc_token_smoke_workflow_yaml() -> YamlEmitResult { + emit_yaml(v: project_workflow_to_yaml(workflow: bmc_token_smoke_workflow)) } diff --git a/dag/gunbc/census_closure_frontier.dag b/dag/gunbc/census_closure_frontier.dag index d96c884c811..828373a14d6 100644 --- a/dag/gunbc/census_closure_frontier.dag +++ b/dag/gunbc/census_closure_frontier.dag @@ -31,7 +31,6 @@ import gunbc.public_workload_census { public_workload_census_declared_runner_frontier_rows, } import gunbc.ci_workflow_source_read { ci_workflow_scan_producer_frontier_rows } -import gunbc.action_use_admission { realized_workflow_projection_frontier_rows } import gunbc.review_sheet_identity { review_sheet_identity_consumer_frontier_rows } import gunbc.review_sheet_converge_cli { review_sheet_whole_estate_consumer_frontier_rows } import gunbc.review_sheet_formula_observation { review_sheet_formula_observation_consumer_frontier_rows } @@ -118,7 +117,6 @@ fn census_closure_frontier_row_groups() -> List> { public_workload_census_run_attempt_consumer_frontier_rows, public_workload_census_declared_runner_frontier_rows, ci_workflow_scan_producer_frontier_rows, - realized_workflow_projection_frontier_rows, attempt_admission_subject_frontier_rows, ipv6_text_codec_staged_frontier_rows, cpu_description_string_leaf_frontier_rows, diff --git a/dag/gunbc/ci/ci_layer_roots.dag b/dag/gunbc/ci/ci_layer_roots.dag index 1ea9c26ff3f..455da3b2905 100644 --- a/dag/gunbc/ci/ci_layer_roots.dag +++ b/dag/gunbc/ci/ci_layer_roots.dag @@ -897,8 +897,8 @@ data witness_exclusion_frontier: List = [ dissolution: excl_install_media_dissolve}, WitnessExclusionRow { pattern: "srv3_seeded_install_media_real_execution_witness_test.dag", - classification: BinWitnessWet, - reason: excl_install_media_reason, + classification: LocalRepoWetLane, + reason: excl_install_media_seeded_local_wet_reason, dissolution: excl_install_media_dissolve}, WitnessExclusionRow { pattern: "http_client_get_real_execution_witness_test.dag", @@ -1457,7 +1457,9 @@ data excl_no_fake_anomalies_dissolve: DissolutionCondition = unbound_dissolution data excl_install_media_local_wet_reason: String = "REAL LOCAL FILE-EFFECT WITNESSES, RECLASSIFIED FROM THE DARK BIN LANE TO THE EXECUTING LOCAL ONE (operator ruling, 2026-09-09). Every fn in this file drives shell.Mktemp.Dir, Filesystem.Write and shell.Remove.RecursiveForce, which dag/extdeps/shell has no mock_response for, so hermetic discovery refuses them by construction. That much was always true and is why the file is excluded.\n\nWHAT WAS WRONG WAS THE LANE. These three fns were enrolled in bin_witness_wet_entries, whose scheduled route has had no executing consumer since 2026-08-15 -- so their execution standing read as covered while nothing ran them. They are not bin-execution witnesses at all: they create a temporary directory, write a file, observe its presence, verify a sha256sum and clean up. No ISO is downloaded, no network is contacted, no remote host is invoked, no Cargo build runs. That is precisely the local-repo wet lane's subject, and that lane executes with candidate-bound terminals joined back at identity grain.\n\nTHE ALTERNATIVE WAS REFUSED AND IS WORTH NAMING. A SourceLocator change touched this file's subject module, which put these identities in the changed set and blocked the floor for want of a terminal verdict. Splitting that change so this file stopped appearing would have removed the DETECTOR while leaving the execution standing false -- gate evasion rather than repair. Reclassification fixes the standing the detector was correctly reporting.\n\nTHE ROUTE-GAP ROWS STAY. Hermetic execution still reaches host operations and is still held; the wet terminal is the independent evidence PAIRED with that route gap, never a reason to delete it." -data install_media_real_execution_wet_note: String = "The four srv3_install_media_fetch_real_execution_witness_test.dag / srv3_seeded_install_media_real_execution_witness_test.dag entries at the tail of bin_witness_wet_entries below share this lane with the bin-execution witnesses (same RunnableDiscoveryBatch Wet mechanism, not a forked one - DESIGN §3) though they are not bin-execution: they drive real shell.Mktemp.Dir/Filesystem.Write/shell.Remove.RecursiveForce effects that dag/extdeps/shell/shell.dag has no mock_response for (only shell.Exec.Run/Check do), so the Hermetic discovery corpus refuses them by construction (typed 'no mock_response for operation Dir' error). Split out of their parent files (shell-to-dag B1, 2026-07-14) so the parent files' pure classification witnesses keep hermetic per-PR discovery undiminished. Dissolve-on: shell.dag's file-effect operations gain mock_response coverage AND these fns' assertions are re-checked for correctness under a fixed mock (some assert behavior-dependent outcomes, e.g. matched == false on a wrong hash, which a naive fixed mock would falsify) - then they re-enroll as ordinary hermetic discovery rows and these rows delete." +data excl_install_media_seeded_local_wet_reason: String = "REAL LOCAL FILE-EFFECT WITNESS, RECLASSIFIED FROM THE DARK BIN LANE TO THE EXECUTING LOCAL ONE on gunb-ai/gunbc#11730, on the same ruling and for the same reason as its sibling (excl_install_media_local_wet_reason, operator ruling 2026-09-09). Its one fn drives shell.Mktemp.Dir, Filesystem.Write/Read, grep and an in-place sed over a grub.cfg fixture in that temporary directory, and shell.Remove.RecursiveForce -- none of which dag/extdeps/shell has a mock_response for, so hermetic discovery refuses it by construction. No ISO is read, no network is contacted, no remote host is invoked, no Cargo build runs: that is the local-repo wet lane's subject. THE TRIGGER WAS THE DETECTOR, AND IT WAS RIGHT: #11730 changed this file's import (the grub cmdline is now read from srv3_seeded_grub_kernel_cmdline), so its identity entered the changed set and the floor refused with changed_witness_planned_without_terminal_verdict (run 35816045125) -- because bin_witness_wet_entries has had no executing consumer since 2026-08-15, the witness had been reading as covered while nothing ran it. Narrowing the change so the file left the changed set would remove the detector and leave that standing false; the witness is instead enrolled in v2.workflow.local_repo_wet_terminal local_repo_wet_schedule. Measured wet over this tree before enrollment: it held (claim_batch --wet)." + +data install_media_real_execution_wet_note: String = "Both srv3 real-execution files (srv3_install_media_fetch_real_execution_witness_test.dag, srv3_seeded_install_media_real_execution_witness_test.dag) have since LEFT this lane for the local-repo wet lane (excl_install_media_local_wet_reason, excl_install_media_seeded_local_wet_reason); what follows is the history of why they were split out. Their entries at the tail of bin_witness_wet_entries shared this lane with the bin-execution witnesses (same RunnableDiscoveryBatch Wet mechanism, not a forked one - DESIGN §3) though they are not bin-execution: they drive real shell.Mktemp.Dir/Filesystem.Write/shell.Remove.RecursiveForce effects that dag/extdeps/shell/shell.dag has no mock_response for (only shell.Exec.Run/Check do), so the Hermetic discovery corpus refuses them by construction (typed 'no mock_response for operation Dir' error). Split out of their parent files (shell-to-dag B1, 2026-07-14) so the parent files' pure classification witnesses keep hermetic per-PR discovery undiminished. Dissolve-on: shell.dag's file-effect operations gain mock_response coverage AND these fns' assertions are re-checked for correctness under a fixed mock (some assert behavior-dependent outcomes, e.g. matched == false on a wrong hash, which a naive fixed mock would falsify) - then they re-enroll as ordinary hermetic discovery rows and these rows delete." data bin_witness_wet_entries: List = [ bin_wet(entry: "dag/test/claim/self_host_artifact_materialization_real_execution_witness_test.dag", f: "a_real_cargo_build_materializes_through_the_real_digest"), @@ -1518,7 +1520,6 @@ data bin_witness_wet_entries: List = [ bin_wet(entry: "dag/test/claim/dag_compile_clean_shard_totality_witness_test.dag", f: "compile_clean_shard_roster_is_well_formed"), bin_wet(entry: "dag/test/claim/dag_compile_clean_shard_totality_witness_test.dag", f: "compile_clean_shard_totality_algebra_holds"), bin_wet(entry: "dag/test/claim/dag_compile_clean_seam_witness_test.dag", f: "cross_shard_seam_algebra_holds"), - bin_wet(entry: "dag/test/claim/srv3/srv3_seeded_install_media_real_execution_witness_test.dag", f: "install_media_remaster_ensure_grub_cmdline_inserts_by_real_execution"), bin_wet(entry: "dag/test/claim/http_client_get_real_execution_witness_test.dag", f: "http_client_get_roundtrip_by_real_execution"), bin_wet(entry: "dag/test/claim/http_client_get_real_execution_witness_test.dag", f: "http_client_get_missing_target_red_control_by_real_execution"), bin_wet(entry: "dag/test/claim/host/host_build_cache_provision_real_execution_witness_test.dag", f: "provision_read_back_failure_not_converged_by_real_execution"), diff --git a/dag/gunbc/fleet/fleet_converge_workflow.dag b/dag/gunbc/fleet/fleet_converge_workflow.dag index 0f30515514c..49128d5b108 100644 --- a/dag/gunbc/fleet/fleet_converge_workflow.dag +++ b/dag/gunbc/fleet/fleet_converge_workflow.dag @@ -162,7 +162,7 @@ import gunbc.host_reset_return { ResetObserverPairing } import gunbc.live_deploy.apply_receipt { rlm_plan_run_id_env_name, rlm_apply_run_id_env_name, rlm_plan_artifact_hash_env_name } import gunbc.roadmap_launch_deployment_cli { rlm_dashboard_run_id_env_name } import gunbc.fleet_converge_plan { fleet_converge_plan_artifact_dir, FleetConvergeScope, FullHost, LaunchEnvironment, FabricAllocationStoreOnly, LiveDeploy } -import extdeps.languages.yaml.emit { serialize_yaml } +import extdeps.languages.yaml.emit { emit_yaml, EmittedYaml, YamlEmitRefused, yaml_emit_refusal_text } import extdeps.languages.json.emit { json_kv, json_object, json_string, serialize_json } import extdeps.languages.yaml.gha_workflow { project_workflow_to_yaml } import std.types { NonEmptyStr, List, String, Int, Bool, list_length } @@ -3229,8 +3229,9 @@ fn fleet_converge_yml_after_reset_assignment() -> FleetConvergeYamlGenerationOut JobTimeoutExceedsCeiling { minutes: _, ceiling: _ } => FleetConvergeYamlGenerationRefused { reason: "fleet-converge job backstop timeout exceeds extdeps.github.actions max_job_timeout_minutes — split the job or reduce a tier's per-step budget, do not raise past the platform ceiling (see gunbc_ci_fleet_job_backstop_timeout_note in gunbc.fleet_workflow_steps)" } JobTimeoutWithinCeiling { minutes: _ } => - FleetConvergeYamlGenerated { - content: serialize_yaml(v: project_workflow_to_yaml(workflow: fleet_converge_workflow)) + match emit_yaml(v: project_workflow_to_yaml(workflow: fleet_converge_workflow)) { + EmittedYaml { text: t } => FleetConvergeYamlGenerated { content: t } + YamlEmitRefused { path: p, reason: r } => FleetConvergeYamlGenerationRefused { reason: yaml_emit_refusal_text(module_path: "gunbc.fleet_converge_workflow", path: p, reason: r) } } } } else { diff --git a/dag/gunbc/fleet/fleet_desired_admission_workflow.dag b/dag/gunbc/fleet/fleet_desired_admission_workflow.dag index fc97a113a59..cff32c4206a 100644 --- a/dag/gunbc/fleet/fleet_desired_admission_workflow.dag +++ b/dag/gunbc/fleet/fleet_desired_admission_workflow.dag @@ -1,7 +1,7 @@ module gunbc.fleet_desired_admission_workflow import extdeps.languages.yaml.types { yaml_string, kv, YamlKeyValue } -import extdeps.languages.yaml.emit { serialize_yaml } +import extdeps.languages.yaml.emit { emit_yaml, EmittedYaml, YamlEmitRefused, yaml_emit_refusal_text } import extdeps.languages.yaml.gha_workflow { project_workflow_to_yaml } import extdeps.github.actions { Workflow, Job, Step, RunStep, UsesStep, @@ -439,9 +439,10 @@ fn expected_fleet_desired_yml() -> FleetDesiredGenerationOutcome { WorkflowActionUsesAdmitted => match admit_workflow_toolchain_homes(workflow: fleet_desired_admission_workflow) { WorkflowToolchainHomesAdmitted => - FleetDesiredGenerated { - content: serialize_yaml(v: project_workflow_to_yaml(workflow: fleet_desired_admission_workflow)) - } + match emit_yaml(v: project_workflow_to_yaml(workflow: fleet_desired_admission_workflow)) { + EmittedYaml { text: t } => FleetDesiredGenerated { content: t } + YamlEmitRefused { path: p, reason: r } => FleetDesiredGenerationRefused { reason: yaml_emit_refusal_text(module_path: "gunbc.fleet_desired_admission_workflow", path: p, reason: r) } + } WorkflowToolchainHomeRefused { job_id: j, refusal: r } => FleetDesiredGenerationRefused { reason: toolchain_home_refusal_reason(job_id: j, refusal: r) } } diff --git a/dag/gunbc/generated_artifact_emit.dag b/dag/gunbc/generated_artifact_emit.dag index 8b83750c179..259a86d9661 100644 --- a/dag/gunbc/generated_artifact_emit.dag +++ b/dag/gunbc/generated_artifact_emit.dag @@ -50,7 +50,7 @@ import gunbc.githooks_pre_push_emit { expected_githooks_pre_push_sh } import gunbc.githooks_pre_commit_emit { expected_githooks_pre_commit_sh } import gunbc.generated_artifact_merge_driver { expected_generated_artifact_merge_driver_sh } import gunbc.ci_spec { generated_artifact_author_regen_command } -import gunbc.os_install_emit { autoinstall_user_data } +import gunbc.os_install_emit { autoinstall_user_data, AutoinstallUserDataRendered, AutoinstallUserDataRefused } import gunbc.runner_host_grants { runner_host_refused_sudoers_content, runner_host_sudoers_content } import gunbc.runner_host_deploy { runner_host_deploy_for, runner_host_spec_for, RunnerHostSpecFound, RunnerHostSpecAbsent, RunnerHostDeployFound, RunnerHostDeployUnmodeled, RunnerHostDeployRefused } @@ -163,7 +163,10 @@ fn artifact_generate_unadorned(a: GeneratedArtifact) -> ArtifactGenerationOutcom Fci1BoundedExecutionContextEmissionRefused { reason } => ArtifactGenerationRefused { reason: reason } } - AutoinstallUserDataArtifact { spec } => artifact_generated(content: autoinstall_user_data(payload: spec.payload)) + AutoinstallUserDataArtifact { spec } => match autoinstall_user_data(payload: spec.payload) { + AutoinstallUserDataRendered { content } => artifact_generated(content: content) + AutoinstallUserDataRefused { reason } => ArtifactGenerationRefused { reason: reason } + } RunnerHostSudoersArtifact { host } => match runner_host_deploy_for(host: host) { RunnerHostDeployFound { deploy } => artifact_generated( diff --git a/dag/gunbc/heal_publisher_workflow.dag b/dag/gunbc/heal_publisher_workflow.dag index 6dae79ce0f7..385c44d5f01 100644 --- a/dag/gunbc/heal_publisher_workflow.dag +++ b/dag/gunbc/heal_publisher_workflow.dag @@ -4,7 +4,7 @@ import std.types { Bool, Int, List, String, NonEmptyStr } import v2.std.optional { Present, Absent } import v2.std.algebra { list_map } import extdeps.languages.yaml.types { yaml_string, yaml_int, yaml_bool, kv, YamlKeyValue, YamlValue } -import extdeps.languages.yaml.emit { serialize_yaml } +import extdeps.languages.yaml.emit { emit_yaml, EmittedYaml, YamlEmitRefused, yaml_emit_refusal_text } import extdeps.languages.yaml.gha_workflow { project_workflow_to_yaml } import extdeps.github.actions { Workflow, Job, Step, RunStep, UsesStep, @@ -451,7 +451,10 @@ fn expected_heal_publisher_workflow_yml() -> HealPublisherWorkflowGenerationOutc } else if !heal_publisher_capability_closure_holds() { HealPublisherWorkflowGenerationRefused { reason: heal_publisher_capability_refusal_reason } } else { - HealPublisherWorkflowGenerated { content: serialize_yaml(v: project_workflow_to_yaml(workflow: heal_publisher_workflow)) } + match emit_yaml(v: project_workflow_to_yaml(workflow: heal_publisher_workflow)) { + EmittedYaml { text: t } => HealPublisherWorkflowGenerated { content: t } + YamlEmitRefused { path: p, reason: r } => HealPublisherWorkflowGenerationRefused { reason: yaml_emit_refusal_text(module_path: "gunbc.heal_publisher_workflow", path: p, reason: r) } + } } } } diff --git a/dag/gunbc/heal_workflow.dag b/dag/gunbc/heal_workflow.dag index 30aa51e7f60..bd24813dffe 100644 --- a/dag/gunbc/heal_workflow.dag +++ b/dag/gunbc/heal_workflow.dag @@ -4,7 +4,7 @@ import std.types { Bool, Int, List, String } import v2.std.optional { Present, Absent } import v2.std.algebra { list_map } import extdeps.languages.yaml.types { yaml_string, yaml_int, yaml_bool, kv, YamlKeyValue } -import extdeps.languages.yaml.emit { serialize_yaml } +import extdeps.languages.yaml.emit { emit_yaml, EmittedYaml, YamlEmitRefused, yaml_emit_refusal_text } import extdeps.languages.yaml.gha_workflow { project_workflow_to_yaml } import extdeps.github.actions { Workflow, Job, Step, RunStep, UsesStep, @@ -432,7 +432,10 @@ fn expected_heal_workflow_yml() -> HealWorkflowGenerationOutcome { } else if !heal_workflow_capability_closure_holds() { HealWorkflowGenerationRefused { reason: heal_workflow_capability_refusal_reason } } else { - HealWorkflowGenerated { content: serialize_yaml(v: project_workflow_to_yaml(workflow: heal_workflow)) } + match emit_yaml(v: project_workflow_to_yaml(workflow: heal_workflow)) { + EmittedYaml { text: t } => HealWorkflowGenerated { content: t } + YamlEmitRefused { path: p, reason: r } => HealWorkflowGenerationRefused { reason: yaml_emit_refusal_text(module_path: "gunbc.heal_workflow", path: p, reason: r) } + } } } } diff --git a/dag/gunbc/machine_intake/mtcollins1_boot_authorization.dag b/dag/gunbc/machine_intake/mtcollins1_boot_authorization.dag index 734f73f7ea3..419f5b83e58 100644 --- a/dag/gunbc/machine_intake/mtcollins1_boot_authorization.dag +++ b/dag/gunbc/machine_intake/mtcollins1_boot_authorization.dag @@ -11,8 +11,8 @@ import std.scoped_authorization { } import gunbc.machine_intake_mtcollins1_boot_artifact { mtcollins1_diskless_image_name, mtcollins1_boot_export_dir } import gunbc.machine_intake_mtcollins1_boot_image_fetch { mtcollins1_boot_image_sha256 } -import gunbc.machine_intake_mtcollins1_census_image { mtcollins1_census_image_input } -import extdeps.provisioning.ubuntu_seeded_install_media { ubuntu_seeded_install_media_built } +import gunbc.machine_intake_mtcollins1_census_image { mtcollins1_census_image_stem, mtcollins1_census_volume_id } +import extdeps.provisioning.ubuntu_seeded_install_media { ubuntu_seeded_install_media_image_name } import gunbc.machine_intake_mtcollins1_actuate { mtcollins1_power_action, mtcollins1_cdrom_selection } import gunbc.machine_intake_mtcollins1_access_observation { mtcollins1_endpoint } import gunbc.fleet_intent_network { operator_host_srv2 } @@ -70,16 +70,18 @@ data mtcollins1_boot_medium: MtCollins1BootMedium = MtCollins1CensusMedium { } // THE CENSUS ARM AUTHORS ONE FACT, THE MEASURED OUTPUT DIGEST from the publish receipt. The image name -// derives from that digest through the same build input that owns the census volume id -// (gunbc.machine_intake_mtcollins1_census_image mtcollins1_census_image_input), so a census arm naming -// a stock installer, or any artifact but the census image, has no spelling: the row cannot disagree -// with the label the verdict binds to (review 5261662425). The published file's name is its digest, -// so the digest is also the sha256 the signed subject and the attach carry. +// derives from that digest through the same stem row the census build input is built from, and the +// label is the same volume-id row (gunbc.machine_intake_mtcollins1_census_image +// mtcollins1_census_image_stem, mtcollins1_census_volume_id), so a census arm naming a stock +// installer, or any artifact but the census image, has no spelling: the row cannot disagree with the +// label the verdict binds to (review 5261662425). Neither fact is seeded content, so neither waits on +// the derivation that renders the seed. The published file's name is its digest, so the digest is +// also the sha256 the signed subject and the attach carry. fn mtcollins1_boot_medium_image_name(medium: MtCollins1BootMedium) -> NonEmptyStr { match medium { MtCollins1StockInstallerMedium => mtcollins1_diskless_image_name MtCollins1CensusMedium { output_digest: d } => - ubuntu_seeded_install_media_built(input: mtcollins1_census_image_input, output_digest: d).image_name + ubuntu_seeded_install_media_image_name(image_stem: mtcollins1_census_image_stem, output_digest: d) } } @@ -93,7 +95,7 @@ fn mtcollins1_boot_medium_image_sha256(medium: MtCollins1BootMedium) -> NonEmpty fn mtcollins1_boot_medium_label(medium: MtCollins1BootMedium) -> NonEmptyStr? { match medium { MtCollins1StockInstallerMedium => none - MtCollins1CensusMedium { output_digest: _ } => Present { value: mtcollins1_census_image_input.volume_id } + MtCollins1CensusMedium { output_digest: _ } => Present { value: mtcollins1_census_volume_id } } } diff --git a/dag/gunbc/machine_intake/mtcollins1_census_image.dag b/dag/gunbc/machine_intake/mtcollins1_census_image.dag index 88a5b578550..e1b6845351a 100644 --- a/dag/gunbc/machine_intake/mtcollins1_census_image.dag +++ b/dag/gunbc/machine_intake/mtcollins1_census_image.dag @@ -14,7 +14,7 @@ import extdeps.provisioning.ubuntu_seeded_install_media { ubuntu_seeded_install_media_grub_kernel_cmdline, } import extdeps.os.ubuntu_autoinstall { UbuntuAutoinstallLiveCommands } -import gunbc.os_install_emit { autoinstall_live_commands_user_data } +import gunbc.os_install_emit { autoinstall_live_commands_user_data, AutoinstallUserData, AutoinstallUserDataRendered, AutoinstallUserDataRefused } import gunbc.seeded_install_media_policy { seeded_install_media_pinned_dates, seeded_install_media_builder_revision } import gunbc.machine_intake_mtcollins1_boot_artifact { mtcollins1_boot_export_dir, mtcollins1_diskless_image_name } import gunbc.machine_intake_boot_image_fetch { boot_image_published_path } @@ -25,6 +25,7 @@ import gunbc.seeded_install_media_publish { SeededImagePublicationRefused, SeededImagePublished, SeededImageResolution, + SeededImageUnobservable, publish_seeded_image, resolve_seeded_image, } @@ -102,16 +103,24 @@ fn mtcollins1_census_program_on_medium() -> NonEmptyStr { // a roster that appends stages after the census ones yields a different image with a different // digest and name, built by the same derivation. Nothing here may run before the census stages or // between them and the envelope's END. -fn mtcollins1_seeded_image_input(stages: List, image_stem: NonEmptyStr, volume_id: NonEmptyStr) -> UbuntuSeededInstallMediaBuildInput { +// THE SEED'S USER-DATA: run the census program before any installer section, leave every section +// to a human. The rendered body is the one build input that is not a constant or a parameter of +// this module, so it is the parameter of the input below and the derivation renders it once. +data mtcollins1_census_live_commands: UbuntuAutoinstallLiveCommands = UbuntuAutoinstallLiveCommands { + autoinstall_version: 1, + early_commands: [concat("sh ", mtcollins1_census_program_on_medium() as String) as NonEmptyStr], +} + +fn mtcollins1_seeded_image_input( + stages: List, + image_stem: NonEmptyStr, + volume_id: NonEmptyStr, + user_data_body: NonEmptyStr, +) -> UbuntuSeededInstallMediaBuildInput { UbuntuSeededInstallMediaBuildInput { stock_artifact: noble_numbat_2404_3_live_server_arm64, nocloud_dir_on_iso: mtcollins1_census_nocloud_dir, - user_data_body: autoinstall_live_commands_user_data( - payload: UbuntuAutoinstallLiveCommands { - autoinstall_version: 1, - early_commands: [concat("sh ", mtcollins1_census_program_on_medium() as String) as NonEmptyStr], - }, - ) as NonEmptyStr, + user_data_body: user_data_body, seed_files: [ NoCloudSeedFile { name: mtcollins1_census_program_file, @@ -130,12 +139,34 @@ fn mtcollins1_seeded_image_input(stages: List, image_stem: Non } } -data mtcollins1_census_image_input: UbuntuSeededInstallMediaBuildInput = mtcollins1_seeded_image_input( +// THE DERIVATION CARRIES THE RENDERER'S REFUSAL, because the census image IS its rendered +// user-data: the body is written into the medium's seed and is in the build key, so a document the +// YAML writer will not write leaves no image to build, publish, or resolve. Publication and +// resolution below carry it through the refusal arms gunbc.seeded_install_media_publish already has. +type Mtcollins1CensusImage + = Mtcollins1CensusImageDerived { input: UbuntuSeededInstallMediaBuildInput } + | Mtcollins1CensusImageRefused { reason: String } + +fn mtcollins1_census_image_derivation(stages: List, image_stem: NonEmptyStr, volume_id: NonEmptyStr) -> Mtcollins1CensusImage { + match autoinstall_live_commands_user_data(payload: mtcollins1_census_live_commands) { + AutoinstallUserDataRefused { reason: r } => Mtcollins1CensusImageRefused { reason: r } + AutoinstallUserDataRendered { content: body } => + Mtcollins1CensusImageDerived { + input: mtcollins1_seeded_image_input(stages: stages, image_stem: image_stem, volume_id: volume_id, user_data_body: body as NonEmptyStr), + } + } +} + +data mtcollins1_census_image: Mtcollins1CensusImage = mtcollins1_census_image_derivation( stages: host_capture_census_stages, image_stem: mtcollins1_census_image_stem, volume_id: mtcollins1_census_volume_id, ) +fn mtcollins1_census_image_refusal_text(reason: String) -> String { + concat("the mtcollins1 census image has no derivation: ", reason) +} + // CONSUMPTION (DESIGN §3c). mtcollins1_census_image_publish is consumed by census stage 0B/0C: // mtcollins1_census_image_publish_wet below, invoked by the fleet-converge mode // mtcollins1_census_image_publish on srv2 (gunbc.fleet_converge_workflow, gunbc.ci_spec), which builds @@ -155,15 +186,25 @@ fn mtcollins1_census_image_stock_path() -> NonEmptyStr { } fn mtcollins1_census_image_publish() -> SeededImagePublication { - publish_seeded_image( - input: mtcollins1_census_image_input, - stock_path: mtcollins1_census_image_stock_path(), - boundary: BmcExportBoundary { export_dir: mtcollins1_boot_export_dir }, - ) + match mtcollins1_census_image { + Mtcollins1CensusImageRefused { reason: r } => + SeededImagePublicationRefused { reason: mtcollins1_census_image_refusal_text(reason: r) } + Mtcollins1CensusImageDerived { input: i } => + publish_seeded_image( + input: i, + stock_path: mtcollins1_census_image_stock_path(), + boundary: BmcExportBoundary { export_dir: mtcollins1_boot_export_dir }, + ) + } } fn mtcollins1_census_image_resolve() -> SeededImageResolution { - resolve_seeded_image(input: mtcollins1_census_image_input, dir: mtcollins1_boot_export_dir) + match mtcollins1_census_image { + Mtcollins1CensusImageRefused { reason: r } => + SeededImageUnobservable { reason: mtcollins1_census_image_refusal_text(reason: r) } + Mtcollins1CensusImageDerived { input: i } => + resolve_seeded_image(input: i, dir: mtcollins1_boot_export_dir) + } } data mtcollins1_census_image_receipt_path: String = "target/mtcollins1-census-image.txt" diff --git a/dag/gunbc/machine_intake/mtcollins1_census_medium_readback.dag b/dag/gunbc/machine_intake/mtcollins1_census_medium_readback.dag index aca3b195008..6bd95763f80 100644 --- a/dag/gunbc/machine_intake/mtcollins1_census_medium_readback.dag +++ b/dag/gunbc/machine_intake/mtcollins1_census_medium_readback.dag @@ -17,8 +17,8 @@ import gunbc.machine_intake_boot_image_fetch { boot_image_published_path } import gunbc.machine_intake_mtcollins1_boot_authorization { MtCollins1BootMedium, MtCollins1CensusMedium, MtCollins1StockInstallerMedium, mtcollins1_boot_medium_image_name, } -import gunbc.machine_intake_mtcollins1_census_image { mtcollins1_census_image_input } -import extdeps.provisioning.ubuntu_seeded_install_media { ubuntu_seeded_install_media_record_path, seeded_install_media_build_key } +import gunbc.machine_intake_mtcollins1_census_image { mtcollins1_census_image, Mtcollins1CensusImage, Mtcollins1CensusImageDerived, Mtcollins1CensusImageRefused } +import extdeps.provisioning.ubuntu_seeded_install_media { UbuntuSeededInstallMediaBuildInput, ubuntu_seeded_install_media_record_path, seeded_install_media_build_key } // THE PUBLICATION RECEIPT IS NOT BOOT-TIME EVIDENCE. The medium row carries the digest the publish // measured; between that run and this boot the file the BMC will mount over NFS could have been @@ -44,17 +44,18 @@ type CensusMediumReadback | CensusMediumReadbackMeasureMalformed { served_path: NonEmptyStr, content: String } | CensusMediumReadbackDiverged { served_path: NonEmptyStr, recorded: NonEmptyStr, measured: NonEmptyStr } | CensusMediumReadbackNotCensus + | CensusMediumReadbackInputUnderived { reason: String } fn mtcollins1_census_medium_served_path(medium: MtCollins1BootMedium) -> NonEmptyStr { boot_image_published_path(export_dir: mtcollins1_boot_export_dir, image_name: mtcollins1_boot_medium_image_name(medium: medium)) } -fn mtcollins1_census_record_path() -> NonEmptyStr { - ubuntu_seeded_install_media_record_path(dir: mtcollins1_boot_export_dir, input: mtcollins1_census_image_input) +fn mtcollins1_census_record_path(input: UbuntuSeededInstallMediaBuildInput) -> NonEmptyStr { + ubuntu_seeded_install_media_record_path(dir: mtcollins1_boot_export_dir, input: input) } -fn mtcollins1_census_build_key() -> NonEmptyStr { - seeded_install_media_build_key(input: mtcollins1_census_image_input) +fn mtcollins1_census_build_key(input: UbuntuSeededInstallMediaBuildInput) -> NonEmptyStr { + seeded_install_media_build_key(input: input) } fn mtcollins1_census_record_argv(record_path: NonEmptyStr) -> List { @@ -101,11 +102,11 @@ type CensusRecordStanding = CensusRecordAdmitted { recorded: NonEmptyStr } | CensusRecordRefused { readback: CensusMediumReadback } -fn mtcollins1_census_record_standing(medium: MtCollins1BootMedium, record: RemoteAnswer) -> CensusRecordStanding { +fn mtcollins1_census_record_standing(medium: MtCollins1BootMedium, input: UbuntuSeededInstallMediaBuildInput, record: RemoteAnswer) -> CensusRecordStanding { match medium { MtCollins1StockInstallerMedium => CensusRecordRefused { readback: CensusMediumReadbackNotCensus } MtCollins1CensusMedium { output_digest: selected } => { - let record_path = mtcollins1_census_record_path() + let record_path = mtcollins1_census_record_path(input: input) if record.exit_code != 0 { CensusRecordRefused { readback: CensusMediumReadbackRecordUnreadable { record_path: record_path, reason: concat("exit=", to_string(record.exit_code), " ", trim(s: record.stderr)) } } } else { @@ -126,6 +127,7 @@ fn mtcollins1_census_record_standing(medium: MtCollins1BootMedium, record: Remot // STAGE TWO: the measurement, joined with an already-admitted record. fn mtcollins1_census_measure_join( medium: MtCollins1BootMedium, + input: UbuntuSeededInstallMediaBuildInput, host: NonEmptyStr, recorded: NonEmptyStr, measure: RemoteAnswer, @@ -140,7 +142,7 @@ fn mtcollins1_census_measure_join( if measured == (recorded as String) { CensusMediumReadbackAgreed { host: host, - build_key: mtcollins1_census_build_key(), + build_key: mtcollins1_census_build_key(input: input), recorded_digest: recorded, image_name: mtcollins1_boot_medium_image_name(medium: medium), served_path: served_path, @@ -153,16 +155,22 @@ fn mtcollins1_census_measure_join( } } -// The two stages composed over supplied answers -- the witness's subject. +// The two stages composed over supplied answers -- the witness's subject. The chain starts from the +// current census build input; a derivation the YAML writer refused has no build key and so no record +// to read, and that is refused before either leg. fn mtcollins1_census_medium_readback_from_answers( medium: MtCollins1BootMedium, host: NonEmptyStr, record: RemoteAnswer, measure: RemoteAnswer, ) -> CensusMediumReadback { - match mtcollins1_census_record_standing(medium: medium, record: record) { - CensusRecordRefused { readback: r } => r - CensusRecordAdmitted { recorded: recorded } => mtcollins1_census_measure_join(medium: medium, host: host, recorded: recorded, measure: measure) + match mtcollins1_census_image { + Mtcollins1CensusImageRefused { reason: r } => CensusMediumReadbackInputUnderived { reason: r } + Mtcollins1CensusImageDerived { input: input } => + match mtcollins1_census_record_standing(medium: medium, input: input, record: record) { + CensusRecordRefused { readback: r } => r + CensusRecordAdmitted { recorded: recorded } => mtcollins1_census_measure_join(medium: medium, input: input, host: host, recorded: recorded, measure: measure) + } } } @@ -184,6 +192,8 @@ fn mtcollins1_census_readback_refusal(readback: CensusMediumReadback) -> String? Present { value: concat("mtcollins1 boot: sha256sum over ", p as String, " answered no sha256 digest (", c, "); refusing") } CensusMediumReadbackDiverged { served_path: p, recorded: r, measured: m } => Present { value: concat("mtcollins1 boot: the served census image ", p as String, " reads ", m as String, " but its derivation record and the medium row say ", r as String, "; refusing to mint a subject over bytes the row does not describe") } + CensusMediumReadbackInputUnderived { reason: why } => + Present { value: concat("mtcollins1 boot: the census build input has no derivation (", why, "), so there is no build key to find a record by; refusing") } } } @@ -230,20 +240,24 @@ fn mtcollins1_census_medium_readback(medium: MtCollins1BootMedium) -> CensusMedi match medium { MtCollins1StockInstallerMedium => CensusMediumReadbackNotCensus MtCollins1CensusMedium { output_digest: _ } => - match prepare_fleet_ssh_agent_context(attempt_raw: mtcollins1_census_readback_attempt_raw) { - FleetSshContextRefused { cause: c } => - CensusMediumReadbackRecordUnreadable { record_path: mtcollins1_census_record_path(), reason: c } - FleetSshContextReady { context: context, receipt: _ } => { - let target = mtcollins1_census_readback_target() - let record = remote_answer_of(outcome: typed_argv_exec_over_fleet_ssh(target: target, context: context, argv: mtcollins1_census_record_argv(record_path: mtcollins1_census_record_path()))) - match mtcollins1_census_record_standing(medium: medium, record: record) { - CensusRecordRefused { readback: r } => r - CensusRecordAdmitted { recorded: recorded } => { - let measure = remote_answer_of(outcome: typed_argv_exec_over_fleet_ssh(target: target, context: context, argv: mtcollins1_census_readback_argv(path: mtcollins1_census_medium_served_path(medium: medium)))) - mtcollins1_census_measure_join(medium: medium, host: operator_host_srv2 as NonEmptyStr, recorded: recorded, measure: measure) + match mtcollins1_census_image { + Mtcollins1CensusImageRefused { reason: r } => CensusMediumReadbackInputUnderived { reason: r } + Mtcollins1CensusImageDerived { input: input } => + match prepare_fleet_ssh_agent_context(attempt_raw: mtcollins1_census_readback_attempt_raw) { + FleetSshContextRefused { cause: c } => + CensusMediumReadbackRecordUnreadable { record_path: mtcollins1_census_record_path(input: input), reason: c } + FleetSshContextReady { context: context, receipt: _ } => { + let target = mtcollins1_census_readback_target() + let record = remote_answer_of(outcome: typed_argv_exec_over_fleet_ssh(target: target, context: context, argv: mtcollins1_census_record_argv(record_path: mtcollins1_census_record_path(input: input)))) + match mtcollins1_census_record_standing(medium: medium, input: input, record: record) { + CensusRecordRefused { readback: r } => r + CensusRecordAdmitted { recorded: recorded } => { + let measure = remote_answer_of(outcome: typed_argv_exec_over_fleet_ssh(target: target, context: context, argv: mtcollins1_census_readback_argv(path: mtcollins1_census_medium_served_path(medium: medium)))) + mtcollins1_census_measure_join(medium: medium, input: input, host: operator_host_srv2 as NonEmptyStr, recorded: recorded, measure: measure) + } + } } } - } } } } diff --git a/dag/gunbc/os_install_emit.dag b/dag/gunbc/os_install_emit.dag index e771f46f8a7..0e5e50f78f9 100644 --- a/dag/gunbc/os_install_emit.dag +++ b/dag/gunbc/os_install_emit.dag @@ -16,7 +16,7 @@ import extdeps.languages.yaml.types { yaml_mapping, yaml_sequence, yaml_string, yaml_int, yaml_bool, kv, empty_yaml_values } -import extdeps.languages.yaml.emit { serialize_yaml } +import extdeps.languages.yaml.emit { emit_yaml, EmittedYaml, YamlEmitRefused, yaml_emit_refusal_text } fn identity_entries(id: AutoinstallIdentity) -> List { [ @@ -66,14 +66,29 @@ fn autoinstall_inner_entries(payload: UbuntuAutoinstallPayload) -> List String { +// THE RENDERED USER-DATA, OR THE WRITER'S REFUSAL. extdeps.languages.yaml.emit emit_yaml refuses a +// value it cannot write with its meaning -- a malformed number lexeme, a key held twice, a character +// outside the repertoire it reads back -- and cloud-init is handed bytes, so that refusal travels to +// whoever wanted the media rather than being rendered into the document it is about. +type AutoinstallUserData + = AutoinstallUserDataRendered { content: String } + | AutoinstallUserDataRefused { reason: String } + +fn autoinstall_user_data(payload: UbuntuAutoinstallPayload) -> AutoinstallUserData { let top = yaml_mapping(entries: [ kv(key: "autoinstall", value: yaml_mapping(entries: autoinstall_inner_entries(payload: payload))), ]) - concat("#cloud-config\n", serialize_yaml(v: top)) + match emit_yaml(v: top) { + EmittedYaml { text: t } => AutoinstallUserDataRendered { content: concat("#cloud-config\n", t) } + YamlEmitRefused { path: p, reason: r } => + AutoinstallUserDataRefused { reason: yaml_emit_refusal_text(module_path: "gunbc.os_install_emit", path: p, reason: r) } + } } -fn autoinstall_live_commands_user_data(payload: UbuntuAutoinstallLiveCommands) -> String { +// The same contract for the live-session document: its keys are literal and distinct and its one +// int is rendered from an Int, so the writer admits it by construction of this function -- but the +// substrate has no way to say so short of the outcome, so the outcome travels. +fn autoinstall_live_commands_user_data(payload: UbuntuAutoinstallLiveCommands) -> AutoinstallUserData { let top = yaml_mapping(entries: [ kv(key: "autoinstall", value: yaml_mapping(entries: [ kv(key: "version", value: yaml_int(n: payload.autoinstall_version)), @@ -81,5 +96,9 @@ fn autoinstall_live_commands_user_data(payload: UbuntuAutoinstallLiveCommands) - kv(key: "early-commands", value: yaml_sequence(elements: map(payload.early_commands, c => yaml_string(s: c)))), ])), ]) - concat("#cloud-config\n", serialize_yaml(v: top)) + match emit_yaml(v: top) { + EmittedYaml { text: t } => AutoinstallUserDataRendered { content: concat("#cloud-config\n", t) } + YamlEmitRefused { path: p, reason: r } => + AutoinstallUserDataRefused { reason: yaml_emit_refusal_text(module_path: "gunbc.os_install_emit", path: p, reason: r) } + } } diff --git a/dag/gunbc/plans/emission_ingestion_inverse.dag b/dag/gunbc/plans/emission_ingestion_inverse.dag index c3c654fecb1..cf37ad97de3 100644 --- a/dag/gunbc/plans/emission_ingestion_inverse.dag +++ b/dag/gunbc/plans/emission_ingestion_inverse.dag @@ -97,18 +97,18 @@ fn emission_ingestion_inverse_body() -> List { p(text: "**Frontier placement** (per [expressibility-frontier](expressibility-frontier.md)): **② now** — flag the two faces over a **shrinking roster** (same shape as §5); **① wall** — once a medium is a `Medium` `Node` emitted via grammar rows, *\"no runner env var accessed\"* is a **model walk** (`workflow.steps.any(s ⇒ s.env.references(RunnerContext.Temp))`) or **unwritable** (the model has no `RunnerEnvContext` variant for those vars), and the grep dissolves; **③** — none, fully decidable. **Dissolve-on:** each medium modeled as a `Node` emitted via grammar rows (GHA-expr first — it is the most anemic) ⇒ its roster entries empty ⇒ the guard flips from ratchet to wall. **First instance** (operator, 2026-06-21): model the GHA env/runner context so `$\{\{ env.RUNNER_TEMP \}\}` is a `Node` (`RunnerContextRef \{ var: RunnerTemp \}`) and rewrite the witness as a model walk — the ① form if the model simply has no such variant."), h3(text: "5.2 Formalizing \"properly modeled\": the round-trip law is the oracle, the partition is the boundary"), p(text: "§5.1 gives the **detector** — it *finds* a leak (a string op, an inline literal) after the fact, so it is still *validation* (§5 of DESIGN: it concedes the bad state is writable). Two further mechanisms turn the rule from a grep-replacement into a constructive discipline: an **oracle** that *proves* a site clean, and a **boundary** that says *where* the cure is even available. With all three, \"properly modeled medium\" becomes a one-line executable bar instead of a judgment call."), - p(text: "**(A) The round-trip law as the single oracle (`ingest ∘ emit = id`).** A medium is *properly modeled* **iff** `ingest ∘ emit = id` holds over its IR, with `DecodeFidelity` marking where that is `Lossless`. This collapses both faces of the §5.1 rule into **one decidable question** — *does a round-trip law hold over the same grammar rows?* A row-driven emitter has an ingest that reads the *same* rows, so the law is **provable by execution**; a hand-rolled emitter (inline literals) has **no shared-row ingest to round-trip against**, and that absence *is* the leak signal. So \"is this a leak?\" stops being the detector's necessarily-heuristic question (\"does it grep target syntax?\") and becomes a structural one (\"is there a round-trip over shared rows?\"). The emit-only `serialize_yaml` (as of 2026-06-22 it renders `project_yaml_to_doc` over the shared `std.layout` `Doc` fold — a forward-only layout projection that bakes `'- '`/`':'`/`'|'` into `Doc` text nodes at projection time, *not* a round-trippable grammar) **fails** this oracle — there is no yaml *ingest* over those same spellings to round-trip against; deep-newt's row-driven `serialize_markdown_source` (#5501) **passes** it the moment a markdown ingest reads the same construct→spelling rows. This is the §4 \"one grammar, both directions\" turned into the *test* — and it is *why* §5.1 pushes every serializer to be row-driven: only a row-driven emitter can satisfy the oracle from day one. **It is therefore the acceptance test for a new medium target** (emission-lane-owned, `quick-seal-137`): a target row-set is *done* iff its round-trip law is green by execution, not iff it merely emits — the §5-of-DESIGN \"green by a real consumer, with a discriminating input\" bar applied to a medium."), + p(text: "**(A) The round-trip law as the single oracle (`ingest ∘ emit = id`).** A medium is *properly modeled* **iff** `ingest ∘ emit = id` holds over its IR, with `DecodeFidelity` marking where that is `Lossless`. This collapses both faces of the §5.1 rule into **one decidable question** — *does a round-trip law hold over the same grammar rows?* A row-driven emitter has an ingest that reads the *same* rows, so the law is **provable by execution**; a hand-rolled emitter (inline literals) has **no shared-row ingest to round-trip against**, and that absence *is* the leak signal. So \"is this a leak?\" stops being the detector's necessarily-heuristic question (\"does it grep target syntax?\") and becomes a structural one (\"is there a round-trip over shared rows?\"). `emit_yaml` **passes** this oracle as of gunb-ai/gunbc#11730, and how it came to is the point: yaml is no longer emit-only. `extdeps.languages.yaml.ingest` reads a declared bounded subset of YAML 1.2.2 and refuses everything outside it at a line, `emit_yaml` decides plain-versus-quoted and its chomping through that reader's own predicates, and `test.claim.yaml_emit_witness` holds `ingest(emit(v)) == v` over decoded values by execution. Note what that does NOT require: the writer is still a hand-authored projection over `std.layout` rather than a row-driven inverse, so the oracle is passed by a shared *reading* of the same subset, not by shared rows -- which is why the law is stated as a round-trip rather than as a test of how the emitter is written. The row-driven inverse remains its dissolution; deep-newt's row-driven `serialize_markdown_source` (#5501) **passes** it the moment a markdown ingest reads the same construct→spelling rows. This is the §4 \"one grammar, both directions\" turned into the *test* — and it is *why* §5.1 pushes every serializer to be row-driven: only a row-driven emitter can satisfy the oracle from day one. **It is therefore the acceptance test for a new medium target** (emission-lane-owned, `quick-seal-137`): a target row-set is *done* iff its round-trip law is green by execution, not iff it merely emits — the §5-of-DESIGN \"green by a real consumer, with a discriminating input\" bar applied to a medium."), p(text: "But *done* here is **medium-complete**, which is **not** the merge bar for an emit PR — collapsing the two would wrongly over-gate. The round-trip law is the medium-**completeness** gate — a *later* step than emit-rows-merge-ready, **never** the emit-PR bar. An emit-only PR with correct, green grammar rows is merge-ready *as the emit step*; the medium is *complete* only once its round-trip law is green (`DecodeFidelity`-bounded). E.g. markdown #5501 and the bash-diagnostic #5505 landed the **emit step** — their round-trip *completeness* is the named **next** step, not a merge precondition."), p(text: "**(B) The per-medium decidability partition (the boundary, not a blanket ban).** The rule is **not** \"ban all string ops over media\" — that is the \"never\" trap (§5 of DESIGN): a ratchet wearing a wall's clothes, and it would wall the realization edge itself (where target syntax is *supposed* to live). Partition each medium-emitting site, per [expressibility-frontier](expressibility-frontier.md):"), ul(items: [ li(text: "**① wall** — the medium has grammar rows + a green round-trip law ⇒ inline syntax is *unwritable* and the check is a model walk (markup today);"), - li(text: "**② lens-residue** — a hand-rolled emitter on the shrinking roster with a named dissolve-on = its grammar rows (`serialize_yaml`, the markdown IR→HTML serializer today — honestly marked SCAFFOLD, not silently forked);"), + li(text: "**② lens-residue** — a hand-rolled emitter on the shrinking roster with a named dissolve-on = its grammar rows (`emit_yaml` -- which has since left this class, see 5.3 -- the markdown IR→HTML serializer today — honestly marked SCAFFOLD, not silently forked);"), li(text: "**③ undecidable / fence** — a medium whose grammar is *not* closed/modeled (an arbitrary embedded DSL with no upstream spec to cite): fence it, mark `DecodeFidelity = lossy`, **never fake-gate**. Honesty replaces the wall exactly here."), ]), p(text: "The formal statement is therefore *per medium*: **wall inline syntax exactly where the medium has a closed grammar and a round-trip; flag where it is hand-rolled-but-closeable; fence where the grammar is genuinely open** — each a decidable test, so the guard never masquerades undecidable residue as a wall (the §5 decidability check)."), p(text: "**Composition.** §5.1's detector *finds* leaks; (A) the round-trip oracle *proves* a site clean (the ① test); (B) the partition decides *which* sites are walled now vs flagged vs fenced. Together they bound the guard against the purity trap (§6 of DESIGN): a medium earns a wall only when it is the cheapest path to a real displaced cost (a grep that goes fail-open on an unanticipated surface form), never for elegance."), h3(text: "5.3 The ② lens-residue slice, generalized — regime-2 emission"), - p(text: "The ② lens-residue emitters above (`serialize_yaml`, plus the `serialize_gitignore` / `serialize_runner_deploy` projections) share a sharper boundary than \"hand-rolled-but-closeable\": they are **emit-only** — we *never ingest* a `.gitignore` or a `ci.yml`-as-config, so the round-trip oracle (A) has nothing to gate against and faithfulness rests on the authored projection (§5 honesty boundary). That is a distinct regime from the grammar-inverse ① wall (markdown/languages), and collapsing the three hand-rolled projection serializers into **one** `render(doc, protocol)` fold over a shared `std.layout` `Doc` IR is the §2 de-boutique-ing of exactly this slice — detailed in [regime-2 shared emission fold](regime2-shared-emission-fold.md). It stays a ② scaffold (a forward-only subset of the ① grammar rows) with dissolution into the v2 `TargetModel` rows; whether `yaml` ever crosses into regime-1 (parseable, round-trip-gated) is the separate decision that doc fences off."), + p(text: "The ② lens-residue emitters `serialize_gitignore` and `serialize_runner_deploy` share a sharper boundary than \"hand-rolled-but-closeable\": they are **emit-only** — we *never ingest* a `.gitignore` or a runner manifest, so the round-trip oracle (A) has nothing to gate against and faithfulness rests on the authored projection (§5 honesty boundary). `serialize_yaml` sat in that class until gunb-ai/gunbc#11730 gave yaml a reader over the same declared subset; the yaml pair is now gated by the round-trip law above, and workflow yaml IS ingested -- gunbc.action_use_admission reads every committed workflow through it. That is a distinct regime from the grammar-inverse ① wall (markdown/languages), and collapsing the three hand-rolled projection serializers into **one** `render(doc, protocol)` fold over a shared `std.layout` `Doc` IR is the §2 de-boutique-ing of exactly this slice — detailed in [regime-2 shared emission fold](regime2-shared-emission-fold.md). It stays a ② scaffold (a forward-only subset of the ① grammar rows) with dissolution into the v2 `TargetModel` rows; whether `yaml` ever crosses into regime-1 (parseable, round-trip-gated) is the separate decision that doc fences off."), h2(text: "6. Independent §3-hygiene cleanup (not a roadmap item)"), p(text: "Found in the same sweep, fixable now with existing authority (dispatched separately): `lit(text: \"dag\")` hardcoded as a policy literal in `compiler_closure_ingest_transport` (×3) + `source_root_ingest_transport` (×1) — should fold `witness_layer_roots` (the shared source-root policy authority) instead of hardcoding the root. A §3 policy-leak (an argv carrying a literal it should receive as a parameter)."), p(text: "Related: [emitter ownership de-fork](emitter-ownership-defork.md) — one authority for clone-vs-move at the emit seam, no silent fallback (lane `node://adhoc-0717d295-672`, PR #6248)."), diff --git a/dag/gunbc/plans/invert_hand_maintained.dag b/dag/gunbc/plans/invert_hand_maintained.dag index 14ea2f7a697..ae48349bf5e 100644 --- a/dag/gunbc/plans/invert_hand_maintained.dag +++ b/dag/gunbc/plans/invert_hand_maintained.dag @@ -21,7 +21,7 @@ fn invert_hand_maintained_body() -> List { ]), p(text: "This is the **construction upgrade of the reachability lens** ([inert-layer-lens](inert-layer-lens.md)): that lens *validates* reachability after the fact (a ② residue); a generated ROADMAP makes unreachability / dangling / stale **unwritable** (a ① wall). Same defect class, moved from lens to wall by emission — the §5 \"construction over validation\" move applied to the doc layer."), h2(text: "3. The pattern — reuse `ci.yml`, don't reinvent"), - p(text: "`ci.yml` is **already** this. `expected_ci_yml()` (`dag/gunbc/ci_yaml_emit.dag`) emits it as `serialize_yaml(project_workflow_to_yaml(ci_workflow))` — a **dag-tree serializer fold** (`extdeps.languages.yaml.emit.serialize_yaml`) on the v1 seed, *not* the v2 `06_translate` emitter; the fold is itself self-marked **SCAFFOLD / §3-fork-of-v2-yaml** with a dissolution trigger to the eventual v2 yaml `TargetModel`. `ci_yaml_parse_witness` (`dag/test/claim/ci_yaml_parse_witness.dag`) fails the build if `ci.yml != expected_ci_yml()`, byte-for-byte. The doc project is the **same three pieces** over the markdown medium:"), + p(text: "`ci.yml` is **already** this. `expected_ci_yml()` (`dag/gunbc/ci_yaml_emit.dag`) emits it as `emit_yaml(project_workflow_to_yaml(ci_workflow))` — a **dag-tree serializer fold** (`extdeps.languages.yaml.emit.emit_yaml`) on the v1 seed, *not* the v2 `06_translate` emitter; the fold is itself self-marked **SCAFFOLD / §3-fork-of-v2-yaml** with a dissolution trigger to the eventual v2 yaml `TargetModel`. `ci_yaml_parse_witness` (`dag/test/claim/ci_yaml_parse_witness.dag`) fails the build if `ci.yml != expected_ci_yml()`, byte-for-byte. The doc project is the **same three pieces** over the markdown medium:"), ol(items: [ li(text: "**authority model** — the work DAG in `.dag` (what `CiFloorSpec` is for ci.yml);"), li(text: "**`serialize_markdown_source(project_roadmap_to_markdown(model))`** — the faithful ci.yml-analogue: a **dag serializer fold on the seed**, *not* the v2 `06_translate` target (forcing markdown onto a v2 `Node` crossing ci.yml itself does **not** do would make the analogue heavier than its own template). Author it **row-driven** — construct→spelling pairs (`Heading`→N×`#`, task→`- [ ] `/`- [x] `, …) live in **rows** the fold looks up, the §6 grammar-inverse, *not* inline string literals — so it dodges the medium-as-Node leak (`emission-ingestion-inverse.md` §5.1) **even as a scaffold**, and the rows relocate unchanged into the eventual v2 `Markdown` `TargetModel`. Mark it **SCAFFOLD** with that dissolution trigger, exactly as `yaml.dag` marks itself;"), @@ -71,7 +71,7 @@ fn invert_hand_maintained_body() -> List { p(text: "**The work-DAG authority model** — what a roadmap line *is* in `.dag` (a work node: title, deps, status-source = its PR/branch, plan-doc carrier). DFS the existing work-item carriers + the dashboard model before minting (§2/§3 — do not re-coin the work graph). **Seed already landed:** `gunbc.roadmap_status.CompletesIff \{ prs \}` (slice-1) is the *first row* of this model — the explicit \"this line completes iff #N merges\" edge. Slice-2 **widens** that same binding from a status surface (`RoadmapStatusEntry`) to the full emitted work node (title/deps/pointer); it must build on `CompletesIff`, not re-coin a parallel completion edge (§3)."), p(text: "**Landed (step 1) — `gunbc.roadmap_model`:** a roadmap line is a `RoadmapNode \{ node, line, carrier \}` that *composes* the existing authorities rather than re-coining them: IDENTITY + DEPENDENCY edges ← `gunbc.process_algebra` (`ProcessNodeId` + the `DecomposeProcessNode \{ parent, children \}` decomposition). `RoadmapEdge` is the complete dependency DAG and `roadmap_parent_ids` preserves every prerequisite for admission. Because Markdown and HTML are trees, `roadmap_primary_parent` chooses one declared dependency for presentation placement only; a multi-parent row also emits `requires all` with the complete set, and edge order never weakens or prioritizes a dependency. STATUS ← the slice-1 `RoadmapStatusEntry` (reused whole — title + box + `CompletionBinding`); TITLE + plan-doc `carrier` are the line's presentation surface (host-fed; the work graph carries neither). The `carrier` projection (`carrier_resolves`/`node_carrier_clean`) is the construction wall for a dangling link (a pointer to a non-existent doc is the emit error). *Open for review:* the work graph's own `ProcessCloseState` (open/closed) vs the PR-grounded `CompletionBinding` are two groundings of \"done\" to reconcile (a node should close iff its completing PR merges) — flagged, not built."), ]), - li(text: "**`serialize_markdown_source`** — the row-driven dag serializer fold on the seed (the ci.yml / `serialize_yaml` pattern, §3), extending `dag/std/markdown.dag`'s `MarkdownDocument` IR with the task-list / table variants the ROADMAP needs. Marked SCAFFOLD; its dissolution trigger is the eventual v2 `Markdown` `TargetModel` in `06_translate` / `extdeps/languages/` (the §6 medium axis, where one grammar reads both directions). `serialize_markdown_source(project_roadmap_to_markdown(model))` produces the ROADMAP bytes."), + li(text: "**`serialize_markdown_source`** — the row-driven dag serializer fold on the seed (the ci.yml / `emit_yaml` pattern, §3), extending `dag/std/markdown.dag`'s `MarkdownDocument` IR with the task-list / table variants the ROADMAP needs. Marked SCAFFOLD; its dissolution trigger is the eventual v2 `Markdown` `TargetModel` in `06_translate` / `extdeps/languages/` (the §6 medium axis, where one grammar reads both directions). `serialize_markdown_source(project_roadmap_to_markdown(model))` produces the ROADMAP bytes."), li(text: "**The drift gate** — `roadmap_gate`: `ROADMAP.md == serialize_markdown_source(…)`, the `ci_yaml_gate` clone."), li(text: "**Status derivation** — wire `[x]` / in-progress from PR-merged / branch-open (closes the stale-checkbox class; this is the host-fed status bridge, see §7)."), li(text: "Then the **doc index** (orphan / dangling become emit errors — the reachability wall by construction), then partial plan-doc frames."), diff --git a/dag/gunbc/plans/regime2_shared_emission_fold.dag b/dag/gunbc/plans/regime2_shared_emission_fold.dag index 8e068924dea..f7933a983e8 100644 --- a/dag/gunbc/plans/regime2_shared_emission_fold.dag +++ b/dag/gunbc/plans/regime2_shared_emission_fold.dag @@ -7,7 +7,7 @@ import gunbc.plans.md_helpers { h2, h3, p, li, ul, ol, cell, row } fn regime2_shared_emission_fold_body() -> List { [ - BlockquoteBlock { blocks: [p(text: "Kills the boutique `*_emit.dag` proliferation by collapsing the **pure-projection** serializers (`serialize_yaml`, `serialize_gitignore`, `serialize_runner_deploy`) into **one** fold of a protocol over a shared emission model. DESIGN refs: §2 (no duplicated layout logic — one concept, every format), §3 (single authority — the layout primitive lives once in `std`), §4 (the destination is emission = ingestion⁻¹; this is the seed-layer, forward-only *subset* of it), §5 (emit-only formats have no round-trip oracle — be honest at the boundary), §6 (DFS the concept DAG before minting; price the work as the displaced boutique-duplication pain, not elegance).")] }, + BlockquoteBlock { blocks: [p(text: "Kills the boutique `*_emit.dag` proliferation by collapsing the **pure-projection** serializers (`emit_yaml`, `serialize_gitignore`, `serialize_runner_deploy`) into **one** fold of a protocol over a shared emission model. DESIGN refs: §2 (no duplicated layout logic — one concept, every format), §3 (single authority — the layout primitive lives once in `std`), §4 (the destination is emission = ingestion⁻¹; this is the seed-layer, forward-only *subset* of it), §5 (emit-only formats have no round-trip oracle — be honest at the boundary), §6 (DFS the concept DAG before minting; price the work as the displaced boutique-duplication pain, not elegance).")] }, h2(text: "1. The two regimes (why this is a clean split, not a fork)"), p(text: "Emission has two regimes, and keeping them separate is correct:"), ul(items: [ @@ -19,7 +19,7 @@ fn regime2_shared_emission_fold_body() -> List { header: row(cells: [cell(text: "emitter"), cell(text: "file"), cell(text: "IR today"), cell(text: "serialize fn")]), alignments: [AlignNone, AlignNone, AlignNone, AlignNone], rows: [ - row(cells: [cell(text: "yaml (ci.yml)"), cell(text: "`dag/extdeps/languages/yaml/types.dag` + `dag/extdeps/languages/yaml/emit.dag` + `dag/gunbc/ci_yaml_emit.dag`"), cell(text: "`YamlValue` (a real structured IR)"), cell(text: "`serialize_yaml` — hand-rolled recursive `match` (block/flow seq, scalar quoting, indent)")]), + row(cells: [cell(text: "yaml (ci.yml)"), cell(text: "`dag/extdeps/languages/yaml/types.dag` + `dag/extdeps/languages/yaml/emit.dag` + `dag/gunbc/ci_yaml_emit.dag`"), cell(text: "`YamlValue` (a real structured IR)"), cell(text: "`emit_yaml` (named `serialize_yaml` when this was written) — hand-rolled recursive `match` (block/flow seq, scalar quoting, indent), now total over a `YamlEmitResult`")]), row(cells: [cell(text: "gitignore"), cell(text: "`dag/gunbc/gitignore_emit.dag`"), cell(text: "**none** — raw `concat`"), cell(text: "`serialize_gitignore` — hand fold over `IgnoreGroup`")]), row(cells: [cell(text: "runner_deploy"), cell(text: "`dag/gunbc/runner/runner_deploy_emit.dag`"), cell(text: "**none** — raw `concat`"), cell(text: "`expected_runner_deploy_manifest` — hand fold over hosts")]), ], @@ -48,7 +48,7 @@ fn regime2_shared_emission_fold_body() -> List { ol(items: [ li(text: "**`std` layout IR + `render` fold** — DFS first; mint the minimal faithful `Doc` + the single render fold; one witness that `render` is format-agnostic."), li(text: "**gitignore + runner_deploy** — the trivial line-oriented formats (no IR today). Project each to `Doc`, delete the hand fold, byte-identical witness."), - li(text: "**yaml (ci.yml)** — the heavy one (nesting, block/flow seq, scalar quoting). `project_yaml_to_doc` maps `YamlValue → Doc`, baking quoting into text nodes; delete `serialize_yaml`'s layout half, byte-identical witness against the committed `ci.yml`."), + li(text: "**yaml (ci.yml)** — the heavy one (nesting, block/flow seq, scalar quoting). `project_yaml_to_doc` maps `YamlValue → Doc`, baking quoting into text nodes; delete the writer's layout half, byte-identical witness against the committed `ci.yml`."), li(text: "**markdown — NOT here.** Markdown is regime-1-or-projection and is bright-stag's lane; this doc does not touch `std.markdown` / `roadmap_emit`. Flag the seam, don't cross it."), ]), h2(text: "6. Open / boundaries"), @@ -65,6 +65,6 @@ data regime2_shared_emission_fold_plan: Plan = Plan { slug: "regime2-shared-emission-fold", title: "Regime-2 emission: one shared fold for the pure-projection formats", body: regime2_shared_emission_fold_body(), - retirement: PlanRetiresWhen { condition: unbound_dissolution(description: "Delete this doc when the three regime-2 pure-projection serializers (serialize_yaml/serialize_gitignore/serialize_runner_deploy) are collapsed into one render fold over a shared std layout Doc IR — byte-identical-witnessed against the committed ci.yml/.gitignore/runner manifest with a format-agnostic-render witness — so the boutique *_emit.dag duplication is gone; the scaffold's own dissolution (the v2 TargetModel grammar-row inverse subsuming the shared fold at self-host) then carries the rest.") + retirement: PlanRetiresWhen { condition: unbound_dissolution(description: "Delete this doc when the three regime-2 pure-projection serializers (emit_yaml/serialize_gitignore/serialize_runner_deploy) are collapsed into one render fold over a shared std layout Doc IR — byte-identical-witnessed against the committed ci.yml/.gitignore/runner manifest with a format-agnostic-render witness — so the boutique *_emit.dag duplication is gone; the scaffold's own dissolution (the v2 TargetModel grammar-row inverse subsuming the shared fold at self-host) then carries the rest.") } } diff --git a/dag/gunbc/recurring_failure_mode/bounded_reader_reinterprets_what_it_does_not_model.dag b/dag/gunbc/recurring_failure_mode/bounded_reader_reinterprets_what_it_does_not_model.dag new file mode 100644 index 00000000000..ac5719f76b9 --- /dev/null +++ b/dag/gunbc/recurring_failure_mode/bounded_reader_reinterprets_what_it_does_not_model.dag @@ -0,0 +1,28 @@ +module gunbc.recurring_failure_mode.bounded_reader_reinterprets_what_it_does_not_model + +import std.types { NonEmptyStr } +import gunbc.recurring_failure_mode { RecurringFailureMode } + +data bounded_reader_reinterprets_what_it_does_not_model: RecurringFailureMode = RecurringFailureMode { + identity: "bounded_reader_reinterprets_what_it_does_not_model" as NonEmptyStr, + + receipts: [ + "**a bounded reader or writer answers outside its subset with a plausible wrong meaning instead of a refusal** (a reader that models part of a format falls through to the most permissive arm -- \"anything else is a string\" -- and a writer drops or reshapes what it cannot write, so a document outside the subset is accepted with a tree it does not denote, and a value is written as text that reads back as something else).", + + "Found by the source audit on gunb-ai/gunbc#11663 (comment 5743133113) against extdeps.languages.yaml: unrecognized plain text became a string with no refusal arm; a flow sequence split on the literal `, ` (so `[a,b]` was one element and `[\"a, b\", c]` split inside quotes); single-quoted scalars kept their quotes; double-quoted escapes were not decoded and `\\q` was not refused; quoted keys kept their quotes; a flow mapping became a string; a `|` literal lost its clipped final line break; `key: # comment` became the string `# comment`; the writer filtered out a literal's empty lines, wrote `|` for text with no final line break, and wrote the string \"123\" unquoted so it read back as an int.", + + "WHY IT SURVIVED: every witness compared serialize(parse(x)) with serialize(expected), and the round trip checked emit(parse(emit(v))) == emit(v). Both compare the WRITER's text, so a loss in the writer masked the same loss in the reader (the blank-lines witness could not fail for that reason), and a type change the writer re-spells identically -- YamlString \"123\" to YamlInt \"123\" -- passed. A witness over a reader must compare decoded structure with an independent expected value.", + + "Repair (this change): extdeps.languages.yaml.ingest is a declared bounded subset of YAML 1.2.2 -- the core schema, both quoted styles with every escape, one-line flow sequences, `|`/`|-`/`|+` literals with detected indentation, block collections at any increasing indentation -- and refuses everything else with the line it sits on; extdeps.languages.yaml.emit decides plain-versus-quoted with the reader's own predicates, picks the chomping indicator from the text, falls back to a double-quoted scalar where no literal can carry the string, and returns YamlEmitRefused with the value's path for what it cannot write. Witnesses compare parse(text) with hand-written values and parse(emit(v)) with v, each construct with a red control (test.claim.yaml_ingest_witness, test.claim.yaml_emit_witness).", + + "Rung: mechanically preventable -- the conformance witnesses expose a regression in either direction, and the refusal arms make an out-of-subset document a located, typed refusal rather than a tree.", + + "Ceiling: structurally guaranteed, once the reader and writer are one grammar read in both directions (DESIGN section 4): a production the grammar does not carry has no arm to fall through to, so reinterpretation is not expressible.", + + "Next-rung trigger: the capability for extdeps.languages.yaml's ingest and emit to be derived from one set of grammar rows in extdeps/languages, sufficient for every construct the subset admits to be read and written by the same production.", + + "REVIEW TELL: a reader function returning the value type with no refusal arm; a writer that filters or rewrites content; a witness that compares re-serialized text where decoded values were available.", + ], + + evidence: [], +} diff --git a/dag/gunbc/recurring_failure_mode/provider_substituted_action_runtime.dag b/dag/gunbc/recurring_failure_mode/provider_substituted_action_runtime.dag index 526f3f360f2..d735b9e0ef1 100644 --- a/dag/gunbc/recurring_failure_mode/provider_substituted_action_runtime.dag +++ b/dag/gunbc/recurring_failure_mode/provider_substituted_action_runtime.dag @@ -15,7 +15,7 @@ data provider_substituted_action_runtime: RecurringFailureMode = RecurringFailur "THE PRODUCER MUST BE READ, NOT THE RELEASE NOTE: actions/download-artifact v6.0.0 is announced as its Node 24 release, and its action.yml at 018cc2cf5baa6db3ef3c5f8a56943fffe632ef53 still declares node20; node24 is declared first at v7.0.0. A model keyed on the marketed major would have admitted a release the runner still substitutes.", - "RUNG: found BELOW the ladder (silent substitution, a warning only). Now MECHANICALLY PREVENTABLE -- the guarantee is CANNOT MERGE -- on two routes over one shared admission (gunbc.action_use_admission admit_selected_action_use: exact execution under repository_node_runtime_epoch AND a repository selection). Route 1: each generator refuses to emit a workflow model with a refused use (admit_workflow_action_uses). Route 2: EVERY file under .github/workflows, generated and hand-authored, is projected to its `uses:` lines and each is joined by commit to a producer manifest reading and admitted (test.claim.action_use_admission_witness every_realized_workflow_action_use_is_modeled_and_executed_exactly, enrolled in v2.workflow.required_floor required_gate_prefixes). A tag ref, an unread commit, an unselected release, a local/docker/subdirectory action, a reusable workflow, a composite with a substituted nested use, and the node20 runner override anywhere in a workflow all refuse. THE PROJECTION'S OWN BOUNDARY: it reads a use only as `uses: ` at the head of a line (optionally after `- `); every other non-comment line that spells `uses` followed by a colon (flow style anywhere, a tab, a space before the colon) refuses as unclassifiable. A double-quoted key at the HEAD of a line carrying an escape refuses, since the projection cannot decode it; the same key inside a FLOW mapping (`[{\"us\\u0065s\": ...}]`) is NOT seen -- a wider rule was measured and refused two committed workflow VALUES (escaped JSON in a group expression and in a shell script), and a line projection cannot separate a YAML key from JSON inside a script string. What it does not see is a `uses` key spelled without that token in a form it does not refuse -- a complex key (`? uses`), or a single-quoted key relying on `''` doubling -- and script text is not exempt: a script line shaped like a `uses:` key is read as a use, and one that mentions `uses:` mid-line refuses -- both fail closed, at the cost of forbidding that spelling in scripts. The projection itself is a second YAML reader beside extdeps.languages.yaml.ingest, operator-approved on gunb-ai/gunbc#11663 and marked on its carrier as gunbc.action_use_admission realized_workflow_projection_frontier_rows, whose trigger is the modeled parser fitting the new-witness budget on the largest workflow. It does NOT reach a workflow executing on the pull request that introduces it: GitHub runs a PR's own workflow before any gate can refuse it, so a PR adding a node20 pin can show the warning once, on that PR, and cannot land.", + "RUNG: found BELOW the ladder (silent substitution, a warning only). Now MECHANICALLY PREVENTABLE -- the guarantee is CANNOT MERGE -- on two routes over one shared admission (gunbc.action_use_admission admit_selected_action_use: exact execution under repository_node_runtime_epoch AND a repository selection). Route 1: each generator refuses to emit a workflow model with a refused use (admit_workflow_action_uses). Route 2: EVERY file under .github/workflows, generated and hand-authored, is read by the modeled YAML reader (extdeps.languages.yaml.ingest ingest_yaml_source, a bounded subset of YAML 1.2.2 that refuses everything outside it at a line) through gunbc.action_use_admission realized_workflow_reading; each `uses` is taken from where GitHub reads it (jobs..uses and jobs..steps[i].uses), joined by commit to a producer manifest reading, and admitted -- one claim per file in test.claim.action_use_admission_witness (census__uses_are_modeled_and_executed_exactly), joined to the directory listing by every_executed_workflow_file_has_a_census_claim_and_no_claim_names_a_missing_file, all enrolled in v2.workflow.required_floor required_gate_prefixes. A tag ref, an unread commit, an unselected release, a local/docker/subdirectory action, a reusable workflow, a composite with a substituted nested use, and the node20 runner override anywhere in a workflow all refuse. THE READER'S BOUNDARY: a document it cannot read with its correct meaning (a flow mapping, an anchor or alias, a comment after a value, a tab in indentation, and the rest its module header lists) is unreadable, never read as having no uses; script text is script content, so a script line spelled like a `uses:` key is not a use, and a quoted or escaped `uses` key is decoded and read. This replaced a line projection of `uses:` lines (operator-approved on gunb-ai/gunbc#11663 as a scaffold standing beside the modeled reader), deleted when the modeled reader fit the new-witness budget on the largest workflow. It does NOT reach a workflow executing on the pull request that introduces it: GitHub runs a PR's own workflow before any gate can refuse it, so a PR adding a node20 pin can show the warning once, on that PR, and cannot land.", "CEILING 4, structurally impossible for modeled workflows: UsesStep.uses constructible only as an admitted release. It is not reached because admission is repository policy and UsesStep is the upstream shape in extdeps.github.actions; putting a gunbc admission type in that field is a layer inversion. The realized-file route stays at rung 2 permanently, because hand-authored YAML is outside the compiler's acceptance path.", diff --git a/dag/gunbc/required_ci_epoch_observation.dag b/dag/gunbc/required_ci_epoch_observation.dag index 1d77a4157b3..137e25a7796 100644 --- a/dag/gunbc/required_ci_epoch_observation.dag +++ b/dag/gunbc/required_ci_epoch_observation.dag @@ -37,7 +37,6 @@ type RequiredCiEpochUnobserved | RequiredCiEpochPathUnreadable { detail: String } | RequiredCiEpochDocumentUnparsable { detail: String } | RequiredCiEpochAbsent - | RequiredCiEpochDuplicated { occurrences: Int } | RequiredCiEpochMalformed { detail: String } type RequiredCiEpochObservation @@ -84,23 +83,18 @@ fn epoch_from_entry(kv: YamlKeyValue) -> RequiredCiEpochObservation { } } +// A SECOND EPOCH MEMBER CANNOT REACH THE MAPPING WALK: the reader refuses a mapping that holds a +// key twice (YAML requires unique keys), so a duplicated epoch is RequiredCiEpochDocumentUnparsable +// carrying the reader's line and reason. fn required_ci_epoch_from_workflow_source(src: String) -> RequiredCiEpochObservation { match ingest_yaml_source(src: src) { - YamlIngestRejected { reason: r } => RequiredCiEpochUnobservable { cause: RequiredCiEpochDocumentUnparsable { detail: r } } + YamlIngestRejected { line: l, reason: r } => RequiredCiEpochUnobservable { cause: RequiredCiEpochDocumentUnparsable { detail: join(["line ", to_string(l), ": ", r], "") } } IngestedYaml { value: doc } => match doc { YamlMapping { entries: es } => { - let hits = epoch_entries_in(entries: yaml_env_entries(entries: es)) - let n = hits.length() - if n == 0 { - RequiredCiEpochUnobservable { cause: RequiredCiEpochAbsent } - } else if n > 1 { - RequiredCiEpochUnobservable { cause: RequiredCiEpochDuplicated { occurrences: n } } - } else { - match first(hits) { - Present { value: kv } => epoch_from_entry(kv: kv) - Absent => RequiredCiEpochUnobservable { cause: RequiredCiEpochAbsent } - } + match first(epoch_entries_in(entries: yaml_env_entries(entries: es))) { + Present { value: kv } => epoch_from_entry(kv: kv) + Absent => RequiredCiEpochUnobservable { cause: RequiredCiEpochAbsent } } } _ => @@ -126,8 +120,6 @@ fn required_ci_epoch_unobserved_message(cause: RequiredCiEpochUnobserved) -> Str concat("the workflow document at the triggering commit does not parse: ", d) RequiredCiEpochAbsent => "the workflow at the triggering commit declares no contract epoch" - RequiredCiEpochDuplicated { occurrences: _ } => - "the workflow at the triggering commit declares the contract epoch more than once" RequiredCiEpochMalformed { detail: d } => concat("the contract epoch at the triggering commit is malformed: ", d) } diff --git a/dag/gunbc/srv3/srv3_os_install_actuate_scope.dag b/dag/gunbc/srv3/srv3_os_install_actuate_scope.dag index ebeb2a1c7d3..72f1caa2001 100644 --- a/dag/gunbc/srv3/srv3_os_install_actuate_scope.dag +++ b/dag/gunbc/srv3/srv3_os_install_actuate_scope.dag @@ -7,7 +7,7 @@ import gunbc.os_install { srv3_seeded_os_install_plan_standing, srv3_os_install_plan_autoinstall, } -import gunbc.srv3_seeded_install_media_artifact { srv3_seeded_install_media_input } +import gunbc.srv3_seeded_install_media_artifact { srv3_seeded_install_media, Srv3SeededInstallMedia, Srv3SeededInstallMediaDerived, Srv3SeededInstallMediaRefused } import gunbc.assimilate.bmc_wif_delegation_chain { bmc_srv3_admin_secret_ref } import gunbc.assimilate.bmc_token_federation { bmc_srv3_secret_name } import gunbc.os_install_mechanism { @@ -111,8 +111,12 @@ fn srv3_os_install_actuate_credential_debt_grounds_bmc_secret_ref() -> Bool { fn srv3_actuator_prep_binds_seeded_iso_path() -> Bool { - string_contains(s: srv3_seeded_install_media_input.image_stem as String, pattern: "srv3-seeded") - && string_contains(s: srv3_seeded_install_media_input.user_data_body as String, pattern: "hostname: srv3") + match srv3_seeded_install_media { + Srv3SeededInstallMediaRefused { reason: _ } => false + Srv3SeededInstallMediaDerived { input: i } => + string_contains(s: i.image_stem as String, pattern: "srv3-seeded") + && string_contains(s: i.user_data_body as String, pattern: "hostname: srv3") + } } // PREP IS NOT INSTALL — still the claim, and now demonstrated more sharply. The diff --git a/dag/gunbc/srv3/srv3_seeded_install_media.dag b/dag/gunbc/srv3/srv3_seeded_install_media.dag index feb5235393c..75fca354885 100644 --- a/dag/gunbc/srv3/srv3_seeded_install_media.dag +++ b/dag/gunbc/srv3/srv3_seeded_install_media.dag @@ -58,7 +58,7 @@ fn srv3_seeded_install_media_actuator_host_is_srv1() -> Bool { } fn srv3_seeded_install_media_install_path_is_versioned() -> Bool { - string_contains(s: srv3_seeded_install_media_input.image_stem as String, pattern: "ubuntu-24.04.3-live-server-srv3-seeded") + string_contains(s: srv3_seeded_install_media_image_stem as String, pattern: "ubuntu-24.04.3-live-server-srv3-seeded") } fn srv3_seeded_autoinstall_delivery_is_on_iso() -> Bool { diff --git a/dag/gunbc/srv3/srv3_seeded_install_media_artifact.dag b/dag/gunbc/srv3/srv3_seeded_install_media_artifact.dag index a45b87a8107..7a632a1aab7 100644 --- a/dag/gunbc/srv3/srv3_seeded_install_media_artifact.dag +++ b/dag/gunbc/srv3/srv3_seeded_install_media_artifact.dag @@ -11,21 +11,19 @@ import extdeps.provisioning.ubuntu_seeded_install_media { ubuntu_seeded_install_media_grub_kernel_cmdline, } import gunbc.os_install { srv3_ubuntu_autoinstall_on_iso } -import gunbc.os_install_emit { autoinstall_user_data } +import gunbc.os_install_emit { autoinstall_user_data, AutoinstallUserData, AutoinstallUserDataRendered, AutoinstallUserDataRefused } import gunbc.seeded_install_media_policy { seeded_install_media_pinned_dates, seeded_install_media_builder_revision } import gunbc.seeded_install_media_publish { HostArtifactsDirectory, SeededImagePublication, + SeededImagePublicationRefused, SeededImageResolution, + SeededImageUnobservable, SeededImageWriteBoundary, publish_seeded_image, resolve_seeded_image, } -import std.types { List, NonEmptyStr } - -data srv3_autoinstall_user_data_body: NonEmptyStr = autoinstall_user_data( - payload: srv3_ubuntu_autoinstall_on_iso, -) as NonEmptyStr +import std.types { List, NonEmptyStr, String } data srv3_seeded_nocloud_dir_on_iso: NonEmptyStr = ubuntu_seeded_install_media_default_nocloud_dir( hostname: srv3_ubuntu_autoinstall_on_iso.identity.hostname, @@ -36,17 +34,50 @@ data srv3_seeded_grub_kernel_cmdline: NonEmptyStr = ubuntu_seeded_install_media_ ) // The install payload is srv3's autoinstall, so the image is an INSTALLER and keeps the label the -// earlier remaster gave it; nothing reads srv3's boot-medium label. -data srv3_seeded_install_media_input: UbuntuSeededInstallMediaBuildInput = UbuntuSeededInstallMediaBuildInput { - stock_artifact: noble_numbat_2404_3_live_server_arm64, - nocloud_dir_on_iso: srv3_seeded_nocloud_dir_on_iso, - user_data_body: srv3_autoinstall_user_data_body, - seed_files: [], - grub_kernel_cmdline: srv3_seeded_grub_kernel_cmdline, - volume_id: "UBUNTU_SEEDED", - pinned_dates: seeded_install_media_pinned_dates, - image_stem: "ubuntu-24.04.3-live-server-srv3-seeded", - builder_revision: seeded_install_media_builder_revision, +// earlier remaster gave it; nothing reads srv3's boot-medium label. The stem is its own row because +// it is made of the stock point release and the host, neither of which is seeded content: a reader +// that wants the name does not need the derivation to have succeeded. +data srv3_seeded_install_media_image_stem: NonEmptyStr = "ubuntu-24.04.3-live-server-srv3-seeded" + +// EVERY BUILD INPUT BUT THE USER-DATA BODY IS A CONSTANT OF THIS MODULE; the body is the one field +// that is rendered, so it is the one parameter. A witness that supplies a body at this boundary gets +// the real input shape without re-running the renderer (DESIGN §3, a witness discriminates at one +// interface); the real route below is the inhabitance claim that the renderer's output reaches it. +fn srv3_seeded_install_media_input_for(user_data_body: NonEmptyStr) -> UbuntuSeededInstallMediaBuildInput { + UbuntuSeededInstallMediaBuildInput { + stock_artifact: noble_numbat_2404_3_live_server_arm64, + nocloud_dir_on_iso: srv3_seeded_nocloud_dir_on_iso, + user_data_body: user_data_body, + seed_files: [], + grub_kernel_cmdline: srv3_seeded_grub_kernel_cmdline, + volume_id: "UBUNTU_SEEDED", + pinned_dates: seeded_install_media_pinned_dates, + image_stem: srv3_seeded_install_media_image_stem, + builder_revision: seeded_install_media_builder_revision, + } +} + +// THE DERIVATION CARRIES THE RENDERER'S REFUSAL, because the seeded image IS its rendered user-data: +// the body is written into the ISO's seed and is in the build key, so a document the YAML writer +// will not write leaves no image to build, publish, or resolve. Every consumer of the row below +// therefore matches this outcome and refuses in its own vocabulary -- publication and resolution +// through the refusal arms gunbc.seeded_install_media_publish already carries. +type Srv3SeededInstallMedia + = Srv3SeededInstallMediaDerived { input: UbuntuSeededInstallMediaBuildInput } + | Srv3SeededInstallMediaRefused { reason: String } + +fn srv3_seeded_install_media_derivation() -> Srv3SeededInstallMedia { + match autoinstall_user_data(payload: srv3_ubuntu_autoinstall_on_iso) { + AutoinstallUserDataRefused { reason: r } => Srv3SeededInstallMediaRefused { reason: r } + AutoinstallUserDataRendered { content: body } => + Srv3SeededInstallMediaDerived { input: srv3_seeded_install_media_input_for(user_data_body: body as NonEmptyStr) } + } +} + +data srv3_seeded_install_media: Srv3SeededInstallMedia = srv3_seeded_install_media_derivation() + +fn srv3_seeded_install_media_refusal_text(reason: String) -> String { + concat("the srv3 seeded image has no derivation: ", reason) } // The actuator host serves the image over its own NBD proxy from its artifacts directory, beside the @@ -56,13 +87,23 @@ data srv3_seeded_install_media_boundary: SeededImageWriteBoundary = HostArtifact } fn srv3_seeded_install_media_publish() -> SeededImagePublication { - publish_seeded_image( - input: srv3_seeded_install_media_input, - stock_path: ubuntu_install_media_install_path(artifact: noble_numbat_2404_3_live_server_arm64), - boundary: srv3_seeded_install_media_boundary, - ) + match srv3_seeded_install_media { + Srv3SeededInstallMediaRefused { reason: r } => + SeededImagePublicationRefused { reason: srv3_seeded_install_media_refusal_text(reason: r) } + Srv3SeededInstallMediaDerived { input: i } => + publish_seeded_image( + input: i, + stock_path: ubuntu_install_media_install_path(artifact: noble_numbat_2404_3_live_server_arm64), + boundary: srv3_seeded_install_media_boundary, + ) + } } fn srv3_seeded_install_media_resolve() -> SeededImageResolution { - resolve_seeded_image(input: srv3_seeded_install_media_input, dir: ubuntu_install_media_artifacts_dir()) + match srv3_seeded_install_media { + Srv3SeededInstallMediaRefused { reason: r } => + SeededImageUnobservable { reason: srv3_seeded_install_media_refusal_text(reason: r) } + Srv3SeededInstallMediaDerived { input: i } => + resolve_seeded_image(input: i, dir: ubuntu_install_media_artifacts_dir()) + } } diff --git a/dag/gunbc/witness/compiler_gate_workflow.dag b/dag/gunbc/witness/compiler_gate_workflow.dag index ef99e28e616..dab06992a80 100644 --- a/dag/gunbc/witness/compiler_gate_workflow.dag +++ b/dag/gunbc/witness/compiler_gate_workflow.dag @@ -29,7 +29,7 @@ import std.types { Bool, Int, List, String } import v2.std.optional { Present, Absent } import v2.std.algebra { list_map, filter, list_append, length } import extdeps.languages.yaml.types { YamlKeyValue, yaml_string, yaml_int, yaml_bool, kv } -import extdeps.languages.yaml.emit { serialize_yaml } +import extdeps.languages.yaml.emit { emit_yaml, EmittedYaml, YamlEmitRefused, yaml_emit_refusal_text } import extdeps.languages.yaml.gha_workflow { project_workflow_to_yaml } import extdeps.github.actions { Workflow, Job, Step, UsesStep, RunStep, @@ -1185,7 +1185,10 @@ fn expected_compiler_gate_yml() -> WitnessFloorGenerationOutcome { WitnessFloorGenerationRefused { reason: toolchain_home_refusal_reason(job_id: j, refusal: r) } WorkflowToolchainHomesAdmitted => if compiler_gate_capability_closure_holds() && compiler_gate_clippy_capability_closure_holds() && compiler_gate_emit_build_capability_closure_holds() && compiler_gate_floor_capability_closure_holds() && compiler_gate_floor_scripts_are_renderable() && compiler_gate_lane_standings_hold() { - WitnessFloorGenerated { content: serialize_yaml(v: project_workflow_to_yaml(workflow: compiler_gate_workflow)) } + match emit_yaml(v: project_workflow_to_yaml(workflow: compiler_gate_workflow)) { + EmittedYaml { text: t } => WitnessFloorGenerated { content: t } + YamlEmitRefused { path: p, reason: r } => WitnessFloorGenerationRefused { reason: yaml_emit_refusal_text(module_path: "gunbc.compiler_gate_workflow", path: p, reason: r) } + } } else { WitnessFloorGenerationRefused { reason: compiler_gate_capability_refusal_reason } } diff --git a/dag/gunbc/witness/witness_floor_workflow.dag b/dag/gunbc/witness/witness_floor_workflow.dag index b4c0459db44..bc877852c9c 100644 --- a/dag/gunbc/witness/witness_floor_workflow.dag +++ b/dag/gunbc/witness/witness_floor_workflow.dag @@ -18,7 +18,7 @@ import gunbc.workflow_capability_closure { capability_closure_is_closed, } import v2.std.algebra { FreeSemigroup, fold_list, list_append, list_map, non_empty_to_list, list_head, skip, HeadFound, HeadAbsent } -import extdeps.languages.yaml.emit { serialize_yaml } +import extdeps.languages.yaml.emit { emit_yaml, EmittedYaml, YamlEmitRefused, yaml_emit_refusal_text } import extdeps.languages.yaml.gha_workflow { project_workflow_to_yaml } import v2.workflow.ci_workflow_run_emit { ci_toolchain_homes_isolated_or_refuse_command } import v2.workflow.ci_heal_revalidation_preflight_emit { ci_heal_revalidation_preflight_script } @@ -2780,9 +2780,10 @@ fn expected_witness_floor_yml() -> WitnessFloorGenerationOutcome { WitnessFloorGenerationRefused { reason: toolchain_home_refusal_reason(job_id: j, refusal: r) } WorkflowToolchainHomesAdmitted => if witness_floor_capability_closure_holds() && build_lane_capability_closure_holds() && required_lanes_gate_is_renderable() && required_ci_measurement_scripts_are_renderable() && floor_attempt_production_wraps_are_renderable() && floor_attempt_receipt_finalize_is_renderable() { - WitnessFloorGenerated { - content: serialize_yaml(v: project_workflow_to_yaml(workflow: witness_floor_workflow)) - } + match emit_yaml(v: project_workflow_to_yaml(workflow: witness_floor_workflow)) { + EmittedYaml { text: t } => WitnessFloorGenerated { content: t } + YamlEmitRefused { path: p, reason: r } => WitnessFloorGenerationRefused { reason: yaml_emit_refusal_text(module_path: "gunbc.witness_floor_workflow", path: p, reason: r) } + } } else { WitnessFloorGenerationRefused { reason: witness_floor_capability_closure_refusal_reason } } diff --git a/dag/std/types.dag b/dag/std/types.dag index 8e5cbd15908..ae2e32c2cf8 100644 --- a/dag/std/types.dag +++ b/dag/std/types.dag @@ -166,10 +166,30 @@ type CommitSha = String // constructor; callers reach the constructor, the invalid state stops being writable, and this // predicate deletes. +// WHAT THIS COMPUTES: forty characters, each a lowercase hex digit. It is spelled as "remove every +// hex digit and ask whether nothing is left" rather than as a test of each character. BOTH SPELLINGS +// ARE LINEAR in the input, so the per-character one was never a cost-shape defect in the sense +// DESIGN section 6 makes standing: what this spelling buys is a CONSTANT FACTOR, because the +// interpreter charges a per-character lambda far more than one whole-string builtin. That is a +// modeling loss (the alphabet is hard-coded and the word `hex` is gone) taken against an evaluator +// cost, i.e. a workaround, and it is named as one here. +// +// DISSOLVE-ON: the per-character evaluation cost is fixed at its source, in the interpreter, so the +// per-character spelling costs about what the whole-string one does; this predicate then returns to +// testing each character against the hex alphabet, and the oracle below becomes its only spelling. +// +// THE RESPELLING IS ADMITTED ON ITS OWN COST AND ON NOTHING DOWNSTREAM OF IT (operator ruling, +// stern-carp-604, 2026-09-20, on gunb-ai/gunbc#11730). A claim budget elsewhere in the tree fitting +// afterwards is a consequence and was never the reason; anyone reading this as buying margin for a +// census should read that ruling, which admits a fix to a shared predicate and refuses only an +// author admitting their own. +// +// THE EQUALITY IS THE CONTRACT, not the shape: test.claim.std_commit_sha_text_witness keeps the +// per-character spelling as an oracle and claims the two answer the same on every input either can +// be given, with a red per refusal cause -- the lengths either side of forty, uppercase, the +// characters adjacent to both ranges, and the empty string. fn commit_sha_text_holds(head: String) -> Bool { - head.length() == 40 && all(chars(s: head), cp => - (cp >= 48 && cp <= 57) || (cp >= 97 && cp <= 102) - ) + (head.length() == 40) && (replace(replace(replace(replace(replace(replace(replace(replace(replace(replace(replace(replace(replace(replace(replace(replace(head, "0", ""), "1", ""), "2", ""), "3", ""), "4", ""), "5", ""), "6", ""), "7", ""), "8", ""), "9", ""), "a", ""), "b", ""), "c", ""), "d", ""), "e", ""), "f", "") == "") } type Sha256 = String diff --git a/dag/test/claim/action_use_admission_witness_test.dag b/dag/test/claim/action_use_admission_witness_test.dag index 515aaeef6e2..14053bc41a9 100644 --- a/dag/test/claim/action_use_admission_witness_test.dag +++ b/dag/test/claim/action_use_admission_witness_test.dag @@ -12,6 +12,7 @@ import extdeps.github.action.download_artifact { download_artifact_v6_0_0, downl import extdeps.github.action.checkout { checkout_v4_4_0 } import extdeps.github.action.setup_rust_toolchain { setup_rust_toolchain_v1_16_0 } import extdeps.github.action.rust_cache { rust_cache_v2_9_1 } +import v2.workflow.floor_naming_hygiene { floor_discovery_scan_test_decl_names } import gunbc.action_use_admission { ActionUseAdmission, AdmittedExact, RefusedForcedSubstitution, RefusedByPlatform, RefusedUnmodeledPlatformDisposition, RefusedUnobservedDeclaration, RefusedConflictingDeclarations, @@ -19,9 +20,10 @@ import gunbc.action_use_admission { ActionUseSite, StepActionUse, JobReusableWorkflowUse, RealizedActionUse, RealizedWorkflowReading, RealizedWorkflowUses, RealizedWorkflowUnreadable, WorkflowActionUsesAdmission, WorkflowActionUsesAdmitted, WorkflowActionUsesRefused, + realized_census_cost_standing, action_use_admitted, action_use_admission_reason, admit_action_release, admit_action_release_against, admit_realized_action_use, - admit_workflow_action_uses, realized_workflow_action_uses, - upload_artifact_action, checkout_action, github_script_action, + admit_workflow_action_uses, realized_workflow_reading, realized_distinct_uses, + upload_artifact_action, checkout_action, github_script_action, setup_rust_action, realized_use_is_a_repository_selection, } import gunbc.fleet_desired_admission_workflow { fleet_desired_admission_workflow } @@ -32,7 +34,8 @@ data live_tree_disposition: LiveTreeDisposition = SubstrateInputsOnly // producer releases: the two refusals that matter most are releases this repository actually ran, // read from their own action.yml, not a fixture spelled to look like one. The census half reads // every workflow file GitHub would execute from this checkout (.github/workflows, which is committed -// input to this build like any .dag source) and admits every `uses:` in it. +// input to this build like any .dag source) through the modeled YAML reader, one claim per file, +// and admits every `uses` at the two places GitHub reads one. fn is_forced_substitution(a: ActionUseAdmission) -> Bool { match a { @@ -179,13 +182,13 @@ test fn green_pinned_selected_release_realized_is_admitted() -> Bool { } // --------------------------------------------------------------------------------------------- -// THE WORKFLOW PARSER finds uses at both places GitHub reads them, and refuses what it cannot read +// THE WORKFLOW READING finds uses at both places GitHub reads them, and refuses what it cannot read // --------------------------------------------------------------------------------------------- data parser_fixture: String = "name: f\non:\n push:\njobs:\n call:\n uses: org/repo/.github/workflows/w.yml@5555555555555555555555555555555555555555\n build:\n runs-on: ubuntu-latest\n steps:\n - name: a\n run: |\n echo not-an-action\n - uses: actions/upload-artifact@v4\n with:\n name: x\n" test fn parser_classifies_job_and_step_uses_and_refuses_both() -> Bool { - match realized_workflow_action_uses(path: "fixture.yml", source: parser_fixture) { + match realized_workflow_reading(path: "fixture.yml", source: parser_fixture) { RealizedWorkflowUnreadable { workflow_path: _, reason: _ } => false RealizedWorkflowUses { workflow_path: _, uses: us } => (us |> count) == 2 @@ -199,7 +202,7 @@ test fn parser_classifies_job_and_step_uses_and_refuses_both() -> Bool { } test fn red_unparseable_workflow_is_unreadable_not_empty() -> Bool { - match realized_workflow_action_uses(path: "fixture.yml", source: "name: f\non:\n push:\n") { + match realized_workflow_reading(path: "fixture.yml", source: "name: f\non:\n push:\n") { RealizedWorkflowUnreadable { workflow_path: _, reason: _ } => true RealizedWorkflowUses { workflow_path: _, uses: _ } => false } @@ -243,62 +246,123 @@ test fn fleet_desired_workflow_action_uses_are_admitted() -> Bool { data realized_workflow_directory: String = ".github/workflows" type RealizedCensus { - files: Int uses: Int refused: List admitted_texts: List } -fn census_one_file(acc: RealizedCensus, path: String) -> RealizedCensus { +// ONE FILE, READ BY THE MODELED READER, EVERY DISTINCT USE ADMITTED. A file the reader refuses is a +// refusal of the census, never an empty file. +fn census_file(name: String) -> RealizedCensus { + let path = join([realized_workflow_directory, "/", name], "") let read = filesystem_read(path: path) - match realized_workflow_action_uses(path: path, source: read.content) { + match realized_workflow_reading(path: path, source: read.content) { RealizedWorkflowUnreadable { workflow_path: p, reason: r } => - RealizedCensus { files: acc.files + 1, uses: acc.uses, refused: concat(acc.refused, [join([p, ": unreadable: ", r], "")]), admitted_texts: acc.admitted_texts } + RealizedCensus { uses: 0, refused: [join([p, ": unreadable: ", r], "")], admitted_texts: [] } RealizedWorkflowUses { workflow_path: p, uses: us } => - fold(us, init: RealizedCensus { files: acc.files + 1, uses: acc.uses, refused: acc.refused, admitted_texts: acc.admitted_texts }, f: fn(c, u) { + fold(realized_distinct_uses(uses: us), init: RealizedCensus { uses: us |> count, refused: [], admitted_texts: [] }, f: fn(c, u) { if action_use_admitted(a: admit_realized_action_use(u: u)) { - RealizedCensus { files: c.files, uses: c.uses + 1, refused: c.refused, admitted_texts: concat(c.admitted_texts, [u.uses_text]) } + RealizedCensus { uses: c.uses, refused: c.refused, admitted_texts: concat(c.admitted_texts, [u.uses_text]) } } else { - RealizedCensus { files: c.files, uses: c.uses + 1, refused: concat(c.refused, [join([p, " job ", u.job_id, ": ", u.uses_text], "")]), admitted_texts: c.admitted_texts } + RealizedCensus { uses: c.uses, refused: concat(c.refused, [join([p, " job ", u.job_id, ": ", u.uses_text], "")]), admitted_texts: c.admitted_texts } } }) } } -// GitHub executes the .yml and .yaml files directly inside .github/workflows; subdirectories and -// other files are not workflows. -fn realized_census() -> RealizedCensus { +fn census_holds(c: RealizedCensus) -> Bool { + ((c.refused |> count) == 0) && (c.uses > 0) +} + +// THE POPULATION: every .yml and .yaml file directly inside .github/workflows is a workflow GitHub +// executes; subdirectories and other files are not. Each has its own claim below, because the +// census must read every executed file and one claim per file keeps each within the new-witness +// eval-step budget. This join is what makes the per-file claims the census: a workflow added without +// a claim, or a claim naming a file that is gone, refuses here, by name. +data realized_workflow_files: List = [ + "fleet-converge.yml", "fleet-desired.yml", "heal-publish.yml", "heal.yml", "mtcollins-canary.yml", "witnesses.yml" +] + +fn listed_workflow_files() -> List { let listed = Filesystem.List(path: realized_workflow_directory, expect: ExpectSuccess) - let empty = RealizedCensus { files: 0, uses: 0, refused: [], admitted_texts: [] } if !listed.success { - RealizedCensus { files: 0, uses: 0, refused: [concat("listing refused: ", listed.error)], admitted_texts: [] } + [concat("listing refused: ", listed.error)] } else { - fold(split(s: listed.entries, delimiter: "\n"), init: empty, f: fn(acc, name) { - let path = join([realized_workflow_directory, "/", name], "") - if !(name.ends_with(suffix: ".yml") || name.ends_with(suffix: ".yaml")) { - acc - } else { - census_one_file(acc: acc, path: path) - } - }) + filter(split(s: listed.entries, delimiter: "\n"), name => name.ends_with(suffix: ".yml") || name.ends_with(suffix: ".yaml")) } } -// EVERY FILE GITHUB EXECUTES FROM THIS CHECKOUT, GENERATED AND HAND-AUTHORED: ZERO forced -// substitutions, ZERO unmodeled uses, ZERO unresolved producers, ZERO unreadable workflows, ZERO -// node20 runner overrides, and every use IS a repository selection. The population is not empty and -// the join demonstrably ran over real lines: the generated floor workflow's upload-artifact step and -// the hand-authored canary's checkout and github-script steps are all among the admitted uses. -test fn every_realized_workflow_action_use_is_modeled_and_executed_exactly() -> Bool { - let c = realized_census() - (c.refused |> count) == 0 - && c.files > 1 - && c.uses > 0 +// THE CLAIM NAME PROMISES A CENSUS CLAIM PER FILE, SO THE CLAIM LOOKS FOR ONE. Joining the roster +// to the directory listing alone would leave a hole with exactly the shape this census exists to +// close (review 69151): a lane adds a workflow, this claim reds, the author adds the NAME to the +// roster, this claim greens -- and that file's `uses:` are admitted by nobody while a claim called +// "has a census claim" reports success. That is a silent widen, so the third join below reads this +// module's own test declarations through the floor's own scanner and requires the per-file claim to +// exist, by the name the file derives. Adding a roster row without its claim now reds here. +fn census_claim_name_for(file: String) -> String { + join(["census_", replace(replace(replace(file, ".yaml", ""), ".yml", ""), "-", "_"), "_uses_are_modeled_and_executed_exactly"], "") +} + +data census_claim_module_path: String = "dag/test/claim/action_use_admission_witness_test.dag" + +test fn every_executed_workflow_file_has_a_census_claim_and_no_claim_names_a_missing_file() -> Bool { + let listed = listed_workflow_files() + let own_source = Filesystem.Read(path: census_claim_module_path, expect: ExpectSuccess) + let declared = floor_discovery_scan_test_decl_names(content: own_source.content) + ((listed |> count) == (realized_workflow_files |> count)) + && all(listed, name => any(realized_workflow_files, f => f == name)) + && all(realized_workflow_files, f => any(listed, name => name == f)) + && all(realized_workflow_files, f => any(declared, n => n == census_claim_name_for(file: f))) +} + +// THE COST STANDING NAMES THE FILE THIS CLAIM READS. gunbc.action_use_admission +// realized_census_cost_standing records that the claim below sits close under the floor's +// new-witness budget, that its subject grows outside this lane, and what the remedy is when it stops +// fitting; this join keeps that row pointing at a file the census actually reads. +test fn the_cost_standing_names_a_file_this_census_reads() -> Bool { + any(realized_workflow_files, f => f == (realized_census_cost_standing.subject as String)) + && ((realized_census_cost_standing.refused_remedies |> count) > 0) +} + +// Every use of the largest generated workflow is read through the modeled reader and admitted, in +// one claim (the trigger the retired line projection waited for). +test fn census_fleet_converge_uses_are_modeled_and_executed_exactly() -> Bool { + census_holds(c: census_file(name: "fleet-converge.yml")) +} + +test fn census_fleet_desired_uses_are_modeled_and_executed_exactly() -> Bool { + census_holds(c: census_file(name: "fleet-desired.yml")) +} + +test fn census_heal_publish_uses_are_modeled_and_executed_exactly() -> Bool { + census_holds(c: census_file(name: "heal-publish.yml")) +} + +// The heal workflow carries the upload-artifact step the floor's receipts are published through. +test fn census_heal_uses_are_modeled_and_executed_exactly() -> Bool { + let c = census_file(name: "heal.yml") + census_holds(c: c) && any(c.admitted_texts, t => t == action_release_uses_text(action_release: upload_artifact_action)) +} + +// The hand-authored canary: its checkout and github-script steps are among the admitted uses, so +// the join ran over real structure, not an empty file. +test fn census_mtcollins_canary_uses_are_modeled_and_executed_exactly() -> Bool { + let c = census_file(name: "mtcollins-canary.yml") + census_holds(c: c) && any(c.admitted_texts, t => t == action_release_uses_text(action_release: checkout_action)) && any(c.admitted_texts, t => t == action_release_uses_text(action_release: github_script_action)) } +// The generated floor workflow: its checkout and setup-rust-toolchain steps are among the admitted +// uses. +test fn census_witnesses_uses_are_modeled_and_executed_exactly() -> Bool { + let c = census_file(name: "witnesses.yml") + census_holds(c: c) + && any(c.admitted_texts, t => t == action_release_uses_text(action_release: checkout_action)) + && any(c.admitted_texts, t => t == action_release_uses_text(action_release: setup_rust_action)) +} + // ADMITTED IS NOT SELECTED. Swatinem/rust-cache v2.9.1 declares node24 and is admitted exactly, but // no selection row chooses it -- only setup-rust-toolchain's composite runs it -- so a workflow step // pinning it directly is refused by the census. @@ -319,44 +383,133 @@ test fn red_modeled_workflow_with_an_unselected_exact_release_is_refused() -> Bo } } -// THE SCAN REFUSES WHAT IT DOES NOT MODEL, AND SKIPS WHAT IS NOT STRUCTURE. +// THE READING REFUSES WHAT THE READER DOES NOT MODEL, AND READS WHAT IT DOES. fn unreadable_scan(source: String) -> Bool { - match realized_workflow_action_uses(path: "fixture.yml", source: source) { + match realized_workflow_reading(path: "fixture.yml", source: source) { RealizedWorkflowUnreadable { workflow_path: _, reason: _ } => true RealizedWorkflowUses { workflow_path: _, uses: _ } => false } } +fn step_uses_read(source: String) -> List { + match realized_workflow_reading(path: "fixture.yml", source: source) { + RealizedWorkflowUnreadable { workflow_path: _, reason: _ } => ["unreadable"] + RealizedWorkflowUses { workflow_path: _, uses: us } => map(us, u => u.uses_text) + } +} + test fn red_node20_runner_override_refuses_in_env_and_in_a_script() -> Bool { unreadable_scan(source: "jobs:\n b:\n env:\n ACTIONS_ALLOW_USE_UNSECURE_NODE_VERSION: true\n steps:\n - run: x\n") && unreadable_scan(source: "jobs:\n b:\n steps:\n - run: |\n echo ACTIONS_ALLOW_USE_UNSECURE_NODE_VERSION=true >> $GITHUB_ENV\n") } -test fn red_unmodeled_uses_spellings_refuse() -> Bool { - unreadable_scan(source: "jobs:\n b:\n steps:\n - { uses: actions/checkout@v4 }\n") - && unreadable_scan(source: "jobs:\n b:\n steps:\n - \"uses\": actions/checkout@v4\n") +// Outside the reader's subset, or not a string where GitHub needs one: unreadable, never "no uses". +test fn red_unmodeled_or_malformed_uses_refuse() -> Bool { + unreadable_scan(source: "jobs:\n b:\n steps:\n - \{ uses: actions/checkout@v4 }\n") && unreadable_scan(source: "jobs:\n b:\n steps:\n - uses: *pinned\n") && unreadable_scan(source: "jobs:\n b:\n steps:\n - uses:\n") - && unreadable_scan(source: "jobs:\n b:\n steps: [{ uses: actions/checkout@v4 }]\n") + && unreadable_scan(source: "jobs:\n b:\n steps: [\{ uses: actions/checkout@v4 }]\n") && unreadable_scan(source: "jobs:\n b:\n steps:\n - uses:\tactions/checkout@v4\n") - && unreadable_scan(source: "jobs:\n b:\n steps:\n - uses : actions/checkout@v4\n") - && unreadable_scan(source: "jobs:\n b:\n steps:\n - uses : actions/checkout@v4\n") - && unreadable_scan(source: "jobs:\n b:\n steps:\n - \"us\\u0065s\": actions/checkout@v4\n") - && unreadable_scan(source: "jobs:\n b:\n steps:\n - name: x\n with: { uses: y }\n") - && unreadable_scan(source: "jobs:\n b:\n steps:\n - run: |\n echo uses: not-a-key\n") -} - -// A quoted value and a trailing comment are read to the ref; the site comes from the value's own -// shape; and a script line spelled exactly as a `uses:` key IS read as a use (fail-closed: it must -// then be admitted). -test fn scan_reads_quoted_and_commented_values_and_classifies_by_shape() -> Bool { - match realized_workflow_action_uses(path: "fixture.yml", source: "jobs:\n a:\n steps:\n - run: |\n echo not-a-key\n uses: script/line@v1\n - uses: 'actions/checkout@v4' # pinned below\n b:\n uses: org/repo/.github/workflows/w.yml@v1\n") { + && unreadable_scan(source: "jobs:\n b:\n steps:\n - name: x\n with: \{ uses: y }\n") + && unreadable_scan(source: "jobs:\n b:\n steps:\n - uses: 'actions/checkout@v4' # pinned\n") + && unreadable_scan(source: "jobs:\n b:\n steps:\n - uses: [a, b]\n") + && unreadable_scan(source: "jobs:\n b: [x]\n") + && unreadable_scan(source: "name: no jobs here\n") +} + +// The spellings a line projection could not decode are read for what they mean: a quoted key, an +// escaped key and a space before the colon all name `uses`; a quoted value is its text; and a +// script line spelled like a `uses:` key is script content, not a use. +test fn quoted_escaped_and_spaced_uses_keys_are_read_and_script_text_is_not() -> Bool { + (step_uses_read(source: "jobs:\n b:\n steps:\n - \"uses\": actions/checkout@v4\n") == ["actions/checkout@v4"]) + && (step_uses_read(source: "jobs:\n b:\n steps:\n - \"us\\u0065s\": actions/checkout@v4\n") == ["actions/checkout@v4"]) + && (step_uses_read(source: "jobs:\n b:\n steps:\n - uses : actions/checkout@v4\n") == ["actions/checkout@v4"]) + && (step_uses_read(source: "jobs:\n b:\n steps:\n - uses: 'actions/checkout@v4'\n") == ["actions/checkout@v4"]) + && (step_uses_read(source: "jobs:\n b:\n steps:\n - run: |\n echo not-a-key\n uses: script/line@v1\n") == []) +} + +// THE ROUTE TEST OVER A REAL WORKFLOW: the first `uses:` line of a committed workflow is rewritten +// into each spelling a line projection would have mis-read, and the census is run over the whole +// rewritten file. Three spellings are outside the reader's subset or not a string where GitHub needs +// one -- a flow sequence, an alias, an empty value -- and each must make the file UNREADABLE: never +// an empty population, and never the unrewritten one. The other two -- a quoted key and a space +// before the colon -- are YAML for the same key, so the correct reading is the ORIGINAL population, +// exactly; refusing them would be the wrong-meaning reader this module replaced. Asserting the +// original population also rules out the rewrite having been missed: every spelling is checked to +// have changed the file. +data route_rewrite_subject: String = "heal-publish.yml" + +fn route_rewrite_source() -> String { + filesystem_read(path: join([realized_workflow_directory, "/", route_rewrite_subject], "")).content +} + +type RouteRewrite { + source: String + seen: Bool +} + +// The first line holding `uses: `, respelled: `before` is everything ahead of `uses: ` on that line +// (its indentation and any `- `), `value` everything after it. +fn with_first_uses_line(source: String, spell: String) -> String { + let rewritten = fold(split(s: source, delimiter: "\n"), init: RouteRewrite { source: "", seen: false }, f: fn(acc, line) { + let next = if acc.seen { + line + } else if string_contains(s: line, pattern: "uses: ") { + let parts = split(s: line, delimiter: "uses: ") + respelled_uses_line(before: join(parts.take(n: 1), ""), value: join(parts.skip(n: 1), "uses: "), spell: spell) + } else { + line + } + RouteRewrite { + source: if acc.source == "" { next } else { join([acc.source, next], "\n") }, + seen: acc.seen || string_contains(s: line, pattern: "uses: "), + } + }) + rewritten.source +} + +fn respelled_uses_line(before: String, value: String, spell: String) -> String { + if spell == "flow" { + join([before, "uses: [", value, ", ", value, "]"], "") + } else if spell == "alias" { + join([before, "uses: *pinned"], "") + } else if spell == "empty" { + join([before, "uses:"], "") + } else if spell == "quoted_key" { + join([before, "\"uses\": ", value], "") + } else { + join([before, "uses : ", value], "") + } +} + +fn route_uses_texts(source: String) -> List { + match realized_workflow_reading(path: "rewritten.yml", source: source) { + RealizedWorkflowUnreadable { workflow_path: _, reason: _ } => ["unreadable"] + RealizedWorkflowUses { workflow_path: _, uses: us } => map(us, u => u.uses_text) + } +} + +test fn a_real_uses_line_respelled_outside_the_subset_refuses_and_respelled_inside_it_reads_the_same() -> Bool { + let original = route_rewrite_source() + let expected = route_uses_texts(source: original) + let refusing = ["flow", "alias", "empty"] + let equivalent = ["quoted_key", "spaced_colon"] + ((expected |> count) > 0) + && !any(expected, t => t == "unreadable") + && all(concat(refusing, equivalent), sp => with_first_uses_line(source: original, spell: sp) != original) + && all(refusing, sp => unreadable_scan(source: with_first_uses_line(source: original, spell: sp))) + && all(equivalent, sp => route_uses_texts(source: with_first_uses_line(source: original, spell: sp)) == expected) +} + +// The site comes from where the key sits, not from the value's shape: a job-level `uses` naming a +// plain action is still job-level (and refused as an unmodeled reusable-workflow call). +test fn the_site_is_where_the_key_sits() -> Bool { + match realized_workflow_reading(path: "fixture.yml", source: "jobs:\n a:\n steps:\n - uses: actions/checkout@v4\n b:\n uses: actions/checkout@v4\n") { RealizedWorkflowUnreadable { workflow_path: _, reason: _ } => false RealizedWorkflowUses { workflow_path: _, uses: us } => - (us |> count) == 3 - && any(us, u => u.uses_text == "script/line@v1" && match u.site { StepActionUse => true JobReusableWorkflowUse => false }) - && any(us, u => u.uses_text == "actions/checkout@v4" && match u.site { StepActionUse => true JobReusableWorkflowUse => false }) - && any(us, u => u.uses_text == "org/repo/.github/workflows/w.yml@v1" && match u.site { JobReusableWorkflowUse => true StepActionUse => false }) + ((us |> count) == 2) + && any(us, u => (u.job_id == "a") && match u.site { StepActionUse => true JobReusableWorkflowUse => false }) + && any(us, u => (u.job_id == "b") && match u.site { JobReusableWorkflowUse => true StepActionUse => false }) } } @@ -369,22 +522,3 @@ test fn forced_substitution_reason_cites_the_producer_manifest_at_the_commit() - pattern: "github.com/actions/upload-artifact/blob/ea165f8d65b6e75b540449e92b4886f43607fa02/action.yml", ) } - -// THE ESCAPED-KEY REFUSAL IS ABOUT KEYS: a quoted script argument carrying a backslash, with no `":` -// key delimiter, is not a key, so the workflow reads normally. -test fn quoted_script_text_with_a_backslash_is_not_an_escaped_key() -> Bool { - quoted_value_with_escaped_quotes_is_not_an_escaped_key() && - match realized_workflow_action_uses(path: "fixture.yml", source: "jobs:\n b:\n steps:\n - run: |\n \"$HOME/bin/tool\" --re '\\d+'\n - uses: actions/checkout@v4\n") { - RealizedWorkflowUnreadable { workflow_path: _, reason: _ } => false - RealizedWorkflowUses { workflow_path: _, uses: us } => (us |> count) == 1 - } -} - -// A VALUE carrying escaped quotes is not a key: the committed fleet-converge workflow has -// `group: "...${{ fromJSON('{\\"srv1\\": ...}') }}"`, whose `{` is followed by an ESCAPED quote. -fn quoted_value_with_escaped_quotes_is_not_an_escaped_key() -> Bool { - match realized_workflow_action_uses(path: "fixture.yml", source: "jobs:\n b:\n group: \"g-${{ fromJSON('{\\\"srv1\\\": \\\"a\\\"}')[x] }}\"\n steps:\n - uses: actions/checkout@v4\n") { - RealizedWorkflowUnreadable { workflow_path: _, reason: _ } => false - RealizedWorkflowUses { workflow_path: _, uses: us } => (us |> count) == 1 - } -} diff --git a/dag/test/claim/bmc/bmc_onboarding_lifecycle_witness_test.dag b/dag/test/claim/bmc/bmc_onboarding_lifecycle_witness_test.dag index a6b87f24e5b..d4a5d9c3c54 100644 --- a/dag/test/claim/bmc/bmc_onboarding_lifecycle_witness_test.dag +++ b/dag/test/claim/bmc/bmc_onboarding_lifecycle_witness_test.dag @@ -61,8 +61,8 @@ test fn bmc_plan_steady_state_is_keyless_wif() -> Bool { let steady = bmc_plan_steady_state_phase(plan: srv3_onboarding_plan) (steady == FabricJoined) && posture_is_keyless(p: bmc_plan_steady_state_posture(plan: srv3_onboarding_plan)) - && !string_contains(s: emit_bmc_token_smoke_workflow_yaml(), pattern: "credentials_json") - && string_contains(s: emit_bmc_token_smoke_workflow_yaml(), pattern: "id-token: write") + && !string_contains(s: bmc_smoke_workflow_text(), pattern: "credentials_json") + && string_contains(s: bmc_smoke_workflow_text(), pattern: "id-token: write") } fn prev_next_is_identity(p: BmcOnboardingPhase) -> Bool { @@ -239,3 +239,10 @@ test fn install_boot_target_refuses_alike_for_both_3_22_00_hosts() -> Bool { InstallBootTargetSelected { target: _ } => false } } + +fn bmc_smoke_workflow_text() -> String { + match emit_bmc_token_smoke_workflow_yaml() { + EmittedYaml { text: t } => t + YamlEmitRefused { path: _, reason: _ } => "" + } +} diff --git a/dag/test/claim/bmc/bmc_token_federation_witness_test.dag b/dag/test/claim/bmc/bmc_token_federation_witness_test.dag index f01ac4ba09f..d12a0216d05 100644 --- a/dag/test/claim/bmc/bmc_token_federation_witness_test.dag +++ b/dag/test/claim/bmc/bmc_token_federation_witness_test.dag @@ -44,7 +44,7 @@ test fn bmc_wif_principal_set_pins_repo() -> Bool { } test fn bmc_smoke_workflow_is_keyless_with_id_token_write() -> Bool { - let y = emit_bmc_token_smoke_workflow_yaml() + let y = bmc_smoke_workflow_text() string_contains(s: y, pattern: "id-token: write") && string_contains(s: y, pattern: action_release_uses_text(action_release: google_auth_action)) && string_contains(s: y, pattern: "workload_identity_provider:") @@ -53,7 +53,15 @@ test fn bmc_smoke_workflow_is_keyless_with_id_token_write() -> Bool { } test fn bmc_smoke_workflow_carries_no_pasted_key() -> Bool { - let y = emit_bmc_token_smoke_workflow_yaml() - !string_contains(s: y, pattern: "credentials_json") + let y = bmc_smoke_workflow_text() + (y != "") + && !string_contains(s: y, pattern: "credentials_json") && !string_contains(s: y, pattern: "service_account_key") } + +fn bmc_smoke_workflow_text() -> String { + match emit_bmc_token_smoke_workflow_yaml() { + EmittedYaml { text: t } => t + YamlEmitRefused { path: _, reason: _ } => "" + } +} diff --git a/dag/test/claim/contract_identity/required_ci_epoch_observation_test.dag b/dag/test/claim/contract_identity/required_ci_epoch_observation_test.dag index 908a027d1b2..26a62e99265 100644 --- a/dag/test/claim/contract_identity/required_ci_epoch_observation_test.dag +++ b/dag/test/claim/contract_identity/required_ci_epoch_observation_test.dag @@ -7,7 +7,6 @@ import gunbc.required_ci_epoch_observation { RequiredCiEpochObserved, RequiredCiEpochUnobservable, RequiredCiEpochAbsent, - RequiredCiEpochDuplicated, RequiredCiEpochMalformed, RequiredCiEpochDocumentUnparsable, required_ci_epoch_from_workflow_source, @@ -49,9 +48,12 @@ test fn witness_epoch_absent_refuses_on_its_own_axis() -> Bool { } } -test fn witness_epoch_duplicated_refuses_and_counts() -> Bool { +// A duplicated epoch is a duplicated mapping key, which the YAML reader refuses at the second +// occurrence: the refusal names that line and the key, not "no epoch". +test fn witness_epoch_duplicated_refuses_at_the_reader_with_its_line() -> Bool { match required_ci_epoch_from_workflow_source(src: src_duplicated) { - RequiredCiEpochUnobservable { cause: RequiredCiEpochDuplicated { occurrences: n } } => n == 2 + RequiredCiEpochUnobservable { cause: RequiredCiEpochDocumentUnparsable { detail: d } } => + starts_with(s: d, prefix: "line 4: ") && string_contains(s: d, pattern: "duplicate mapping key `GUNBC_REQUIRED_CI_CONTRACT_EPOCH`") _ => false } } diff --git a/dag/test/claim/contract_identity/required_ci_epoch_real_execution_witness_test.dag b/dag/test/claim/contract_identity/required_ci_epoch_real_execution_witness_test.dag index 24aaba7a7a0..249e6775464 100644 --- a/dag/test/claim/contract_identity/required_ci_epoch_real_execution_witness_test.dag +++ b/dag/test/claim/contract_identity/required_ci_epoch_real_execution_witness_test.dag @@ -11,7 +11,7 @@ import extdeps.github.push_event { PushRefUpdate } import extdeps.git.object_store { GitObjectId, git_object_id_from_untagged_hex } import std.types { CommitSha } import extdeps.languages.yaml.types { YamlValue, yaml_mapping, yaml_string, kv } -import extdeps.languages.yaml.emit { serialize_yaml } +import extdeps.languages.yaml.emit { emit_yaml, EmittedYaml, YamlEmitRefused } import gunbc.generated_artifact { WitnessFloorYamlArtifact, artifact_path } import gunbc.repo_identity { gunbc_default_branch_name } import gunbc.witness_floor_workflow { witness_floor_workflow_name } @@ -53,7 +53,7 @@ data live_tree_disposition: LiveTreeDisposition = SubstrateInputsOnly // through the production reader, and fed to the production composition. // // THE TWO FIXTURE DOCUMENTS ARE EMITTED, NOT TYPED. Both come from one function differing only in -// its epoch argument, and both are serialized by extdeps.languages.yaml.emit serialize_yaml -- the +// its epoch argument, and both are written by extdeps.languages.yaml.emit emit_yaml -- the // same emitter that writes the committed workflow -- so "the trees differ only in the epoch member" // is a property of the construction rather than of an author's care with two string literals. // @@ -66,7 +66,7 @@ data fixture_mismatching_epoch: String = "1999-01-01.0" data fixture_repo_full_name: String = "gunb-ai/gunbc" fn fixture_workflow_source(epoch: String) -> String { - serialize_yaml(v: yaml_mapping(entries: [ + emitted_yaml_text(v: yaml_mapping(entries: [ kv(key: "name", value: yaml_string(s: witness_floor_workflow_name)), kv(key: "env", value: yaml_mapping(entries: [ kv(key: (required_ci_contract_epoch_env_key as String), value: yaml_string(s: epoch)), @@ -295,3 +295,12 @@ test fn a_readable_commit_without_the_workflow_is_path_absent_not_commit_unreada _ => false } } + +// The emitted text, or "" when the writer refuses: every claim below asserts content the refused +// value does not have, so a refusal cannot pass one. +fn emitted_yaml_text(v: YamlValue) -> String { + match emit_yaml(v: v) { + EmittedYaml { text: t } => t + YamlEmitRefused { path: _, reason: _ } => "" + } +} diff --git a/dag/test/claim/gha_job_projection_witness_test.dag b/dag/test/claim/gha_job_projection_witness_test.dag index 84e38b6b7b2..aa40d48192a 100644 --- a/dag/test/claim/gha_job_projection_witness_test.dag +++ b/dag/test/claim/gha_job_projection_witness_test.dag @@ -1,8 +1,8 @@ module test.claim.gha_job_projection_witness_test import std.types { Bool, Int, String } -import extdeps.languages.yaml.types { yaml_string, kv } -import extdeps.languages.yaml.emit { serialize_yaml } +import extdeps.languages.yaml.types { YamlValue, yaml_string, kv } +import extdeps.languages.yaml.emit { emit_yaml, EmittedYaml, YamlEmitRefused } import extdeps.languages.yaml.gha_workflow { job_yaml } import extdeps.github.actions { Job, Step, RunStep, @@ -95,11 +95,11 @@ fn minimal_job() -> Job { } fn all_fields_yaml() -> String { - serialize_yaml(v: job_yaml(job: all_fields_job())) + emitted_yaml_text(v: job_yaml(job: all_fields_job())) } fn minimal_yaml() -> String { - serialize_yaml(v: job_yaml(job: minimal_job())) + emitted_yaml_text(v: job_yaml(job: minimal_job())) } test fn every_populated_job_field_reaches_the_emitted_yaml() -> Bool { @@ -193,8 +193,8 @@ fn queue_max_only_job() -> Job { } test fn queue_max_and_queue_not_max_no_longer_serialize_identically() -> Bool { - let qmax = serialize_yaml(v: job_yaml(job: queue_max_only_job())) - let qnot = serialize_yaml(v: job_yaml(job: queue_not_max_job())) + let qmax = emitted_yaml_text(v: job_yaml(job: queue_max_only_job())) + let qnot = emitted_yaml_text(v: job_yaml(job: queue_not_max_job())) !(qmax == qnot) && string_contains(s: qmax, pattern: "queue: max") && !string_contains(s: qnot, pattern: "queue: max") @@ -215,7 +215,16 @@ test fn scalar_concurrency_group_still_projects_as_a_bare_string() -> Bool { concurrency: Present { value: ConcurrencyScalar { group: "scalar-group" } }, permissions: none } - let y = serialize_yaml(v: job_yaml(job: j)) + let y = emitted_yaml_text(v: job_yaml(job: j)) string_contains(s: y, pattern: "concurrency: scalar-group") && !string_contains(s: y, pattern: "queue: max") } + +// The emitted text, or "" when the writer refuses: every claim below asserts content the refused +// value does not have, so a refusal cannot pass one. +fn emitted_yaml_text(v: YamlValue) -> String { + match emit_yaml(v: v) { + EmittedYaml { text: t } => t + YamlEmitRefused { path: _, reason: _ } => "" + } +} diff --git a/dag/test/claim/machine_intake/mtcollins1_census_image_local_wet_test.dag b/dag/test/claim/machine_intake/mtcollins1_census_image_local_wet_test.dag index 0ec14e45cc1..6b38392c4d6 100644 --- a/dag/test/claim/machine_intake/mtcollins1_census_image_local_wet_test.dag +++ b/dag/test/claim/machine_intake/mtcollins1_census_image_local_wet_test.dag @@ -29,11 +29,17 @@ import std.measure { byte_size_count } import extdeps.shell import extdeps.tools.sha256sum { Sha256FileDigest, Sha256FileDigestUnavailable, sha256sum_file_digest_via_shell } import extdeps.provisioning.ubuntu_seeded_install_media { + UbuntuSeededInstallMediaBuildInput, ubuntu_seeded_install_media_built, ubuntu_seeded_install_media_record_path, } import gunbc.machine_intake_boot_image_fetch { boot_image_published_path } -import gunbc.machine_intake_mtcollins1_census_image { mtcollins1_census_image_input } +import gunbc.machine_intake_mtcollins1_census_image { + mtcollins1_census_image_stem, + mtcollins1_census_volume_id, + mtcollins1_seeded_image_input, +} +import gunbc.machine_intake_host_capture_envelope { host_capture_census_stages } import gunbc.seeded_install_media_publish { SeededImageResolution, SeededImageResolved, @@ -153,10 +159,24 @@ test fn the_rendered_program_runs_and_its_output_parses_by_real_execution() -> B && string_contains(s: out, pattern: "\nit's quoted\n") } +// A SUPPLIED user-data body: these claims drive the real resolver over a real directory and need +// an input's shape, not the renderer's output; the rendering route is exercised by +// test.claim.mtcollins1_census_image. +data census_user_data_fixture: NonEmptyStr = "#cloud-config\nautoinstall:\n version: 1\n interactive-sections:\n - \"*\"\n early-commands:\n - sh /cdrom/gunbc-census/capture.sh\n" as NonEmptyStr + +fn census_input_supplied() -> UbuntuSeededInstallMediaBuildInput { + mtcollins1_seeded_image_input( + stages: host_capture_census_stages, + image_stem: mtcollins1_census_image_stem, + volume_id: mtcollins1_census_volume_id, + user_data_body: census_user_data_fixture, + ) +} + fn census_image_path_in(dir: NonEmptyStr, digest: NonEmptyStr) -> NonEmptyStr { boot_image_published_path( export_dir: dir, - image_name: ubuntu_seeded_install_media_built(input: mtcollins1_census_image_input, output_digest: digest).image_name, + image_name: ubuntu_seeded_install_media_built(input: census_input_supplied(), output_digest: digest).image_name, ) } @@ -174,20 +194,20 @@ test fn a_removed_image_is_absent_not_diverged_by_real_execution() -> Bool { Sha256FileDigestUnavailable { path: _, reason: _ } => "unreadable" as NonEmptyStr Sha256FileDigest { digest: x } => x.hex } - let record = ubuntu_seeded_install_media_record_path(dir: d, input: mtcollins1_census_image_input) + let record = ubuntu_seeded_install_media_record_path(dir: d, input: census_input_supplied()) let wrote_record = Filesystem.Write(path: record as String, content: concat(digest as String, "\n")) - let absent = match resolve_seeded_image(input: mtcollins1_census_image_input, dir: d) { + let absent = match resolve_seeded_image(input: census_input_supplied(), dir: d) { SeededImageRecordedImageAbsent { path: _, recorded: r } => r == digest _ => false } let image = census_image_path_in(dir: d, digest: digest) let wrote_other = Filesystem.Write(path: image as String, content: "some-other-bytes") - let diverged = match resolve_seeded_image(input: mtcollins1_census_image_input, dir: d) { + let diverged = match resolve_seeded_image(input: census_input_supplied(), dir: d) { SeededImageRecordDiverged { path: _, recorded: _, observed: _ } => true _ => false } let wrote_same = Filesystem.Write(path: image as String, content: "census-image-bytes") - let resolved = match resolve_seeded_image(input: mtcollins1_census_image_input, dir: d) { + let resolved = match resolve_seeded_image(input: census_input_supplied(), dir: d) { SeededImageResolved { built: b, path: p } => b.output_digest == digest && p == image _ => false } diff --git a/dag/test/claim/machine_intake/mtcollins1_census_image_witness_test.dag b/dag/test/claim/machine_intake/mtcollins1_census_image_witness_test.dag index df4a0d3eb34..73c1e917387 100644 --- a/dag/test/claim/machine_intake/mtcollins1_census_image_witness_test.dag +++ b/dag/test/claim/machine_intake/mtcollins1_census_image_witness_test.dag @@ -38,11 +38,16 @@ import gunbc.machine_intake_host_capture_historical_binding { census_stages_with_mutated_command, roster_matches_bindings, } +import gunbc.os_install_emit { autoinstall_live_commands_user_data, AutoinstallUserData, AutoinstallUserDataRendered, AutoinstallUserDataRefused } import gunbc.machine_intake_mtcollins1_census_image { mtcollins1_census_capture_program, mtcollins1_census_console_device, mtcollins1_census_console_tty, - mtcollins1_census_image_input, + mtcollins1_census_image, + Mtcollins1CensusImage, + Mtcollins1CensusImageDerived, + Mtcollins1CensusImageRefused, + mtcollins1_census_live_commands, mtcollins1_census_machine, mtcollins1_census_image_stem, mtcollins1_census_program_file, @@ -289,26 +294,51 @@ test fn the_program_writes_every_section_then_end_then_syncs_then_halts() -> Boo // THE SEED RUNS THE PROGRAM BEFORE ANY INSTALLER SECTION AND LEAVES EVERY SECTION TO A HUMAN, so a // failed power-off parks the unit at the installer rather than letting an unattended install touch // storage. No storage or identity key is present for subiquity to act on. +// This is also THE INHABITANCE CLAIM FOR THE REAL ROUTE: the derived input's body is what the +// renderer emitted for the census live commands, and a refusal from the YAML writer is a red here. test fn the_seed_runs_the_census_early_and_leaves_every_section_interactive() -> Bool { - let ud = mtcollins1_census_image_input.user_data_body as String - string_contains(s: ud, pattern: "#cloud-config") - && string_contains(s: ud, pattern: "early-commands:") - && string_contains(s: ud, pattern: "sh /cdrom/gunbc-census/capture.sh") - && string_contains(s: ud, pattern: "interactive-sections:") - && string_contains(s: ud, pattern: "*") - && string_contains(s: ud, pattern: "storage") == false - && string_contains(s: ud, pattern: "identity") == false + match mtcollins1_census_image { + Mtcollins1CensusImageRefused { reason: _ } => false + Mtcollins1CensusImageDerived { input: i } => { + let ud = i.user_data_body as String + string_contains(s: ud, pattern: "#cloud-config") + && string_contains(s: ud, pattern: "early-commands:") + && string_contains(s: ud, pattern: "sh /cdrom/gunbc-census/capture.sh") + && string_contains(s: ud, pattern: "interactive-sections:") + && string_contains(s: ud, pattern: "*") + && string_contains(s: ud, pattern: "storage") == false + && string_contains(s: ud, pattern: "identity") == false + && match autoinstall_live_commands_user_data(payload: mtcollins1_census_live_commands) { + AutoinstallUserDataRendered { content: c } => ud == c + AutoinstallUserDataRefused { reason: _ } => false + } + } + } } // ONE TTY, BOTH SPELLINGS. The device the program writes to and the console the kernel is told to // use derive from the same row, so they cannot be edited apart. test fn the_boot_configuration_seeds_from_the_medium_on_the_serial_console() -> Bool { - let cmdline = mtcollins1_census_image_input.grub_kernel_cmdline as String + let cmdline = census_input_supplied().grub_kernel_cmdline as String string_contains(s: cmdline, pattern: "autoinstall \"ds=nocloud;s=/cdrom/gunbc-census/\"") && string_contains(s: cmdline, pattern: concat("console=", mtcollins1_census_console_tty as String, ",")) && (mtcollins1_census_console_device as String) == concat("/dev/", mtcollins1_census_console_tty as String) - && mtcollins1_census_image_input.seed_files.length() == 1 - && (mtcollins1_census_image_input.volume_id as String).length() <= 32 + && census_input_supplied().seed_files.length() == 1 + && (census_input_supplied().volume_id as String).length() <= 32 +} + +// A SUPPLIED user-data body: the key and name claims below need the input's shape, not the +// renderer's output (DESIGN §3, a witness discriminates at one interface); the real rendering route +// is the inhabitance claim above. +data census_user_data_fixture: NonEmptyStr = "#cloud-config\nautoinstall:\n version: 1\n interactive-sections:\n - \"*\"\n early-commands:\n - sh /cdrom/gunbc-census/capture.sh\n" as NonEmptyStr + +fn census_input_supplied() -> UbuntuSeededInstallMediaBuildInput { + mtcollins1_seeded_image_input( + stages: host_capture_census_stages, + image_stem: mtcollins1_census_image_stem, + volume_id: mtcollins1_census_volume_id, + user_data_body: census_user_data_fixture, + ) } fn census_input_varying( @@ -318,7 +348,7 @@ fn census_input_varying( pinned_dates: SeededInstallMediaPinnedDates, builder_revision: NonEmptyStr, ) -> UbuntuSeededInstallMediaBuildInput { - let base = mtcollins1_census_image_input + let base = census_input_supplied() UbuntuSeededInstallMediaBuildInput { stock_artifact: base.stock_artifact, nocloud_dir_on_iso: base.nocloud_dir_on_iso, @@ -352,9 +382,10 @@ fn census_key_varying( // (the positive control), and one byte of the capture program, the boot configuration, the label, // either pinned date or the builder revision each move it, as does a roster with a stage appended. // The key names the derivation; the image's identity is its measured bytes (next claim), and whether -// one changed input byte changes THOSE bytes is the wet build's control on srv2. +// one changed input byte changes THOSE bytes is the wet build's control on srv2. The user-data body +// is varied too: it is the one rendered input, and the key must move when the document does. test fn every_build_input_reaches_the_derivation_key() -> Bool { - let base = mtcollins1_census_image_input + let base = census_input_supplied() let program = mtcollins1_census_capture_program(stages: host_capture_census_stages) let cmdline = base.grub_kernel_cmdline let volid = base.volume_id @@ -372,16 +403,23 @@ test fn every_build_input_reaches_the_derivation_key() -> Bool { stages: append(host_capture_census_stages, items: [HostCaptureStage { name: "after-end", command: "true" }]), image_stem: mtcollins1_census_image_stem, volume_id: mtcollins1_census_volume_id, + user_data_body: census_user_data_fixture, + )) + && key != seeded_install_media_build_key(input: mtcollins1_seeded_image_input( + stages: host_capture_census_stages, + image_stem: mtcollins1_census_image_stem, + volume_id: mtcollins1_census_volume_id, + user_data_body: concat(census_user_data_fixture as String, "\n") as NonEmptyStr, )) } test fn the_image_name_is_the_stem_and_the_measured_digest() -> Bool { let a = ubuntu_seeded_install_media_built( - input: mtcollins1_census_image_input, + input: census_input_supplied(), output_digest: "0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef", ) let b = ubuntu_seeded_install_media_built( - input: mtcollins1_census_image_input, + input: census_input_supplied(), output_digest: "fedcba9876543210fedcba9876543210fedcba9876543210fedcba9876543210", ) a.image_name == "gunbc-mtcollins1-census-ubuntu-24.04.3-arm64-0123456789abcdef.iso" diff --git a/dag/test/claim/machine_intake/mtcollins1_census_medium_readback_witness_test.dag b/dag/test/claim/machine_intake/mtcollins1_census_medium_readback_witness_test.dag index c9403524fc8..6f4295c84eb 100644 --- a/dag/test/claim/machine_intake/mtcollins1_census_medium_readback_witness_test.dag +++ b/dag/test/claim/machine_intake/mtcollins1_census_medium_readback_witness_test.dag @@ -3,6 +3,12 @@ module test.claim.machine_intake.mtcollins1_census_medium_readback_witness_test import v2.std.live_tree { LiveTreeDisposition, SubstrateInputsOnly } import std.types { Bool, Int, NonEmptyStr, String } import gunbc.machine_intake_mtcollins1_boot_authorization { MtCollins1CensusMedium, MtCollins1StockInstallerMedium } +import gunbc.machine_intake_mtcollins1_census_image { + mtcollins1_census_image, Mtcollins1CensusImage, Mtcollins1CensusImageDerived, Mtcollins1CensusImageRefused, + mtcollins1_census_image_stem, mtcollins1_census_volume_id, mtcollins1_seeded_image_input, +} +import gunbc.machine_intake_host_capture_envelope { host_capture_census_stages } +import extdeps.provisioning.ubuntu_seeded_install_media { UbuntuSeededInstallMediaBuildInput } import gunbc.machine_intake_mtcollins1_census_medium_readback { CensusMediumReadbackAgreed, CensusMediumReadbackDiverged, CensusMediumReadbackNotCensus, CensusMediumReadbackRecordUnreadable, CensusMediumReadbackRecordMalformed, CensusMediumReadbackRecordDisagrees, @@ -30,15 +36,29 @@ fn ok(stdout: String) -> RemoteAnswer { RemoteAnswer { exit_code: 0, stdout: std fn failed(code: Int, stderr: String) -> RemoteAnswer { RemoteAnswer { exit_code: code, stdout: "", stderr: stderr } } // The refused leg is DERIVED through the producer's adapter, not restated (review 69507). fn refused(reason: String) -> RemoteAnswer { remote_answer_of(outcome: TypedArgvExecRefused { reason: reason }) } +// A SUPPLIED build input for the record-stage fold: the fold is over the record answer, and needs an +// input only to name the record path; the real derivation is read by the chain claim above. +fn census_input() -> UbuntuSeededInstallMediaBuildInput { + mtcollins1_seeded_image_input( + stages: host_capture_census_stages, + image_stem: mtcollins1_census_image_stem, + volume_id: mtcollins1_census_volume_id, + user_data_body: "#cloud-config\nautoinstall:\n version: 1\n" as NonEmptyStr, + ) +} fn record_ok() -> RemoteAnswer { ok(stdout: concat(recorded as String, "\n")) } fn measure_ok() -> RemoteAnswer { ok(stdout: concat(recorded as String, " /srv/bmc/x.iso\n")) } // THE CHAIN DERIVES FROM ONE INPUT: the served path from the digest through the image stem, the // record path from the build key through the same stem, the build key from the census inputs. test fn the_served_path_and_the_record_path_derive_from_the_census_input() -> Bool { - (mtcollins1_census_medium_served_path(medium: census) as String) == "/srv/bmc/gunbc-mtcollins1-census-ubuntu-24.04.3-arm64-6455b3b9ae11214d.iso" - && (mtcollins1_census_record_path() as String) == concat("/srv/bmc/.gunbc-mtcollins1-census-ubuntu-24.04.3-arm64.", mtcollins1_census_build_key() as String, ".built") - && (mtcollins1_census_build_key() as String) == "ad4bf3fd18230604" + match mtcollins1_census_image { + Mtcollins1CensusImageRefused { reason: _ } => false + Mtcollins1CensusImageDerived { input: input } => + (mtcollins1_census_medium_served_path(medium: census) as String) == "/srv/bmc/gunbc-mtcollins1-census-ubuntu-24.04.3-arm64-6455b3b9ae11214d.iso" + && (mtcollins1_census_record_path(input: input) as String) == concat("/srv/bmc/.gunbc-mtcollins1-census-ubuntu-24.04.3-arm64.", mtcollins1_census_build_key(input: input) as String, ".built") + && (mtcollins1_census_build_key(input: input) as String) == "ad4bf3fd18230604" + } } test fn a_record_and_a_file_that_both_repeat_the_selected_digest_agree_and_retain_every_value() -> Bool { @@ -56,11 +76,11 @@ test fn a_record_and_a_file_that_both_repeat_the_selected_digest_agree_and_retai // producer runs sha256sum only inside its admitted arm, so refusing here is refusing before the // multi-gigabyte image is consulted on the executing route, not only in a composed fold. test fn an_unreadable_record_refuses_before_the_file_is_consulted() -> Bool { - (match mtcollins1_census_record_standing(medium: census, record: failed(code: 1, stderr: "cat: No such file or directory")) { + (match mtcollins1_census_record_standing(medium: census, input: census_input(), record: failed(code: 1, stderr: "cat: No such file or directory")) { CensusRecordRefused { readback: CensusMediumReadbackRecordUnreadable { record_path: _, reason: why } } => string_contains(s: why, pattern: "exit=1") _ => false }) - && (match mtcollins1_census_record_standing(medium: census, record: record_ok()) { + && (match mtcollins1_census_record_standing(medium: census, input: census_input(), record: record_ok()) { CensusRecordAdmitted { recorded: r } => (r as String) == (recorded as String) _ => false }) diff --git a/dag/test/claim/nbd_proxy_virtual_media_install_witness_test.dag b/dag/test/claim/nbd_proxy_virtual_media_install_witness_test.dag index af3fb999576..9197567067b 100644 --- a/dag/test/claim/nbd_proxy_virtual_media_install_witness_test.dag +++ b/dag/test/claim/nbd_proxy_virtual_media_install_witness_test.dag @@ -3,12 +3,16 @@ module test.claim.nbd_proxy_virtual_media_install data live_tree_disposition: LiveTreeDisposition = SubstrateInputsOnly +// A SUPPLIED user-data body: the fixture needs the input's shape, not the renderer's output; the +// real rendering route is exercised by test.claim.srv3_seeded_install_media. +data srv3_seeded_user_data_fixture: NonEmptyStr = "#cloud-config\nautoinstall:\n identity:\n hostname: srv3\n" as NonEmptyStr + // A SUPPLIED resolved image (DESIGN §3: a witness discriminates at one interface). What these // claims check does not depend on which bytes the actuator serves, only on the path reaching the // intent; the real resolution route is exercised by test.claim.mtcollins1_census_image. data srv3_resolved_image_fixture: SeededImageResolution = SeededImageResolved { built: ubuntu_seeded_install_media_built( - input: srv3_seeded_install_media_input, + input: srv3_seeded_install_media_input_for(user_data_body: srv3_seeded_user_data_fixture), output_digest: "abababababababababababababababababababababababababababababababab", ), path: "/var/lib/gunbc/artifacts/ubuntu-24.04.3-live-server-srv3-seeded-abababababababab.iso", diff --git a/dag/test/claim/os_install_actuator_selection_witness_test.dag b/dag/test/claim/os_install_actuator_selection_witness_test.dag index 122d93b6e02..bd1bf945578 100644 --- a/dag/test/claim/os_install_actuator_selection_witness_test.dag +++ b/dag/test/claim/os_install_actuator_selection_witness_test.dag @@ -47,10 +47,14 @@ data unsatisfiable_actuator_selection_fixture: OsInstallActuatorSelection = Actu reason: "fixture: no candidate host satisfies the srv3 actuator requirement conjunction", } +// A SUPPLIED user-data body: the fixture needs the input's shape, not the renderer's output; the +// real rendering route is exercised by test.claim.srv3_seeded_install_media. +data srv3_seeded_user_data_fixture: NonEmptyStr = "#cloud-config\nautoinstall:\n identity:\n hostname: srv3\n" as NonEmptyStr + // A SUPPLIED resolved image, so the actuator selection is the only thing here that can refuse. data srv3_resolved_image_fixture: SeededImageResolution = SeededImageResolved { built: ubuntu_seeded_install_media_built( - input: srv3_seeded_install_media_input, + input: srv3_seeded_install_media_input_for(user_data_body: srv3_seeded_user_data_fixture), output_digest: "abababababababababababababababababababababababababababababababab", ), path: "/var/lib/gunbc/artifacts/ubuntu-24.04.3-live-server-srv3-seeded-abababababababab.iso", diff --git a/dag/test/claim/os_install_deduction_witness_test.dag b/dag/test/claim/os_install_deduction_witness_test.dag index 69c9023d444..83d1c1a7c28 100644 --- a/dag/test/claim/os_install_deduction_witness_test.dag +++ b/dag/test/claim/os_install_deduction_witness_test.dag @@ -4,7 +4,7 @@ module test.claim.os_install_deduction import extdeps.uri { Uri, Https } import extdeps.os.ubuntu_autoinstall { NoCloudLocal, StoragePolicyDirectLayout, UbuntuAutoinstallPayload } import gunbc.os_install_deduction { AutoinstallIncompleteNotBootable, ForeignOsPresent, KvmOperatorAttestation, LoginPrompt, RequiresStoragePolicy, SubiquityStorageScreen, UnknownDiskState, fold_nbd_proxy_os_install_preflight_verdict, plan_for_os_install_preflight_verdict, runtime_verdict_from_kvm_attestation } -import gunbc.os_install_emit { autoinstall_user_data } +import gunbc.os_install_emit { autoinstall_user_data, AutoinstallUserDataRendered, AutoinstallUserDataRefused } import std.types { Bool } import v2.std.live_tree { LiveTreeDisposition } data live_tree_disposition: LiveTreeDisposition = SubstrateInputsOnly @@ -145,9 +145,10 @@ test fn incomplete_not_bootable_plans_refuse_until_declared() -> Bool { } test fn direct_layout_emits_storage_yaml() -> Bool { - let ud = autoinstall_user_data(payload: witness_autoinstall_with_storage) - ud.contains("layout:") - && ud.contains("direct") + match autoinstall_user_data(payload: witness_autoinstall_with_storage) { + AutoinstallUserDataRefused { reason: _ } => false + AutoinstallUserDataRendered { content: ud } => ud.contains("layout:") && ud.contains("direct") + } } test fn kvm_storage_screen_with_incomplete_preflight_awaits_storage() -> Bool { diff --git a/dag/test/claim/srv3/srv3_os_install_actuate_witness_test.dag b/dag/test/claim/srv3/srv3_os_install_actuate_witness_test.dag index d327b9b9e41..b1fbf6cc68e 100644 --- a/dag/test/claim/srv3/srv3_os_install_actuate_witness_test.dag +++ b/dag/test/claim/srv3/srv3_os_install_actuate_witness_test.dag @@ -3,12 +3,16 @@ module test.claim.srv3_os_install_actuate data live_tree_disposition: LiveTreeDisposition = SubstrateInputsOnly +// A SUPPLIED user-data body: the fixture needs the input's shape, not the renderer's output; the +// real rendering route is exercised by test.claim.srv3_seeded_install_media. +data srv3_seeded_user_data_fixture: NonEmptyStr = "#cloud-config\nautoinstall:\n identity:\n hostname: srv3\n" as NonEmptyStr + // A SUPPLIED resolved image (DESIGN §3: a witness discriminates at one interface). What these // claims check does not depend on which bytes the actuator serves, only on the path reaching the // intent; the real resolution route is exercised by test.claim.mtcollins1_census_image. data srv3_resolved_image_fixture: SeededImageResolution = SeededImageResolved { built: ubuntu_seeded_install_media_built( - input: srv3_seeded_install_media_input, + input: srv3_seeded_install_media_input_for(user_data_body: srv3_seeded_user_data_fixture), output_digest: "abababababababababababababababababababababababababababababababab", ), path: "/var/lib/gunbc/artifacts/ubuntu-24.04.3-live-server-srv3-seeded-abababababababab.iso", diff --git a/dag/test/claim/srv3/srv3_os_install_diagnostic_witness_test.dag b/dag/test/claim/srv3/srv3_os_install_diagnostic_witness_test.dag index 655b9bacd84..afc67699025 100644 --- a/dag/test/claim/srv3/srv3_os_install_diagnostic_witness_test.dag +++ b/dag/test/claim/srv3/srv3_os_install_diagnostic_witness_test.dag @@ -59,10 +59,14 @@ fn serve_ready_sol_silent_pre_boot_observations() -> OsInstallActuationObservati ) } +// A SUPPLIED user-data body: the fixture needs the input's shape, not the renderer's output; the +// real rendering route is exercised by test.claim.srv3_seeded_install_media. +data srv3_seeded_user_data_fixture: NonEmptyStr = "#cloud-config\nautoinstall:\n identity:\n hostname: srv3\n" as NonEmptyStr + // A SUPPLIED built image: the diagnostic fold decides over the digest it is handed, and which bytes // the actuator actually built is the resolution route's question, not this fold's. data srv3_diagnostic_image_fixture: UbuntuSeededInstallMediaBuilt = ubuntu_seeded_install_media_built( - input: srv3_seeded_install_media_input, + input: srv3_seeded_install_media_input_for(user_data_body: srv3_seeded_user_data_fixture), output_digest: "abababababababababababababababababababababababababababababababab", ) diff --git a/dag/test/claim/srv3/srv3_os_install_emit_test.dag b/dag/test/claim/srv3/srv3_os_install_emit_test.dag index f4a1ecc41f0..4cc80fc96d0 100644 --- a/dag/test/claim/srv3/srv3_os_install_emit_test.dag +++ b/dag/test/claim/srv3/srv3_os_install_emit_test.dag @@ -4,7 +4,7 @@ module test.claim.srv3_os_install_emit data live_tree_disposition: LiveTreeDisposition = SubstrateInputsOnly test fn srv3_user_data_is_cloud_config_autoinstall() -> Bool { - let ud = autoinstall_user_data(payload: srv3_ubuntu_autoinstall) + let ud = srv3_rendered_user_data() string_contains(s: ud, pattern: "#cloud-config") && string_contains(s: ud, pattern: "autoinstall:") && string_contains(s: ud, pattern: " version: 1") @@ -12,14 +12,14 @@ test fn srv3_user_data_is_cloud_config_autoinstall() -> Bool { } test fn srv3_user_data_hostname_projects_fleet_identity() -> Bool { - let ud = autoinstall_user_data(payload: srv3_ubuntu_autoinstall) + let ud = srv3_rendered_user_data() string_contains(s: ud, pattern: concat(" hostname: ", operator_host_srv3)) && !string_contains(s: ud, pattern: concat("hostname: ", operator_host_srv1)) && !string_contains(s: ud, pattern: concat("hostname: ", operator_host_srv2)) } test fn srv3_user_data_enables_remote_access_and_locale() -> Bool { - let ud = autoinstall_user_data(payload: srv3_ubuntu_autoinstall) + let ud = srv3_rendered_user_data() string_contains(s: ud, pattern: " ssh:") && string_contains(s: ud, pattern: " install-server: true") && string_contains(s: ud, pattern: " locale: en_US.UTF-8") @@ -28,9 +28,18 @@ test fn srv3_user_data_enables_remote_access_and_locale() -> Bool { } test fn srv3_user_data_bakes_fleet_keys_and_disables_password_ssh() -> Bool { - let ud = autoinstall_user_data(payload: srv3_ubuntu_autoinstall) + let ud = srv3_rendered_user_data() string_contains(s: ud, pattern: " authorized-keys:") && string_contains(s: ud, pattern: "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIKgumC/OBjsSgM266FLtgKJpF179Aj/MzEZHQH79QRQF macbook-m4") && string_contains(s: ud, pattern: "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIB5ObOQw+v/Rd5xdurhZpwI9awQNqCWNgFWbZCQsXJfN fleet-automation@gunbc") && string_contains(s: ud, pattern: " allow-pw: false") } + +// The rendered user-data, or "" when the writer refuses: every claim above asserts content the +// refused value does not have, so a refusal cannot pass one. +fn srv3_rendered_user_data() -> String { + match autoinstall_user_data(payload: srv3_ubuntu_autoinstall) { + AutoinstallUserDataRendered { content: c } => c + AutoinstallUserDataRefused { reason: _ } => "" + } +} diff --git a/dag/test/claim/srv3/srv3_seeded_install_media_real_execution_witness_test.dag b/dag/test/claim/srv3/srv3_seeded_install_media_real_execution_witness_test.dag index 1bb33e2f77f..f25f1e33437 100644 --- a/dag/test/claim/srv3/srv3_seeded_install_media_real_execution_witness_test.dag +++ b/dag/test/claim/srv3/srv3_seeded_install_media_real_execution_witness_test.dag @@ -2,7 +2,7 @@ module test.claim.srv3_seeded_install_media_real_execution import std.logic { Bool } import std.types { String, NonEmptyStr } -import gunbc.srv3_seeded_install_media_artifact { srv3_seeded_install_media_input } +import gunbc.srv3_seeded_install_media_artifact { srv3_seeded_grub_kernel_cmdline } import extdeps.provisioning.ubuntu_seeded_install_media_remaster { install_media_remaster_ensure_grub_cmdline } import extdeps.shell import extdeps.filesystem.filesystem_io @@ -10,7 +10,7 @@ import v2.std.live_tree { LiveTreeDisposition, SubstrateInputsOnly } data live_tree_disposition: LiveTreeDisposition = SubstrateInputsOnly -data real_execution_home_doc: String = "This witness drives real shell/filesystem effects (shell.Mktemp.Dir, Filesystem.Write/Read, shell.Remove.RecursiveForce have no mock_response - dag/extdeps/shell/shell.dag - so it cannot run in CI's default Hermetic discovery corpus, dag/std/execution_mode.dag). Split out of srv3_seeded_install_media_witness_test.dag so that file's pure witnesses keep hermetic per-PR discovery; excluded from discovery (gunbc.ci_layer_roots witness_exclusion_substrings) and re-enrolled as a declared Wet explicit entry (bin_witness_wet_entries), same lane as the bin-execution witnesses." +data real_execution_home_doc: String = "This witness drives real shell/filesystem effects (shell.Mktemp.Dir, Filesystem.Write/Read, shell.Remove.RecursiveForce have no mock_response - dag/extdeps/shell/shell.dag - so it cannot run in CI's default Hermetic discovery corpus, dag/std/execution_mode.dag). Split out of srv3_seeded_install_media_witness_test.dag so that file's pure witnesses keep hermetic per-PR discovery; excluded from discovery (gunbc.ci_layer_roots witness_exclusion_substrings) and executed per PR by the required floor's local-repo wet lane (v2.workflow.local_repo_wet_terminal local_repo_wet_schedule; gunbc.ci_layer_roots excl_install_media_seeded_local_wet_reason)." data grub_cfg_fixture_without_cmdline: String = "menuentry \"Ubuntu\" \{\n linux /casper/vmlinuz ---\n\}\n" @@ -18,7 +18,7 @@ test fn install_media_remaster_ensure_grub_cmdline_inserts_by_real_execution() - let dir = shell.Mktemp.Dir() let grub_cfg = concat(dir.path as String, "/grub.cfg") as NonEmptyStr let wrote = Filesystem.Write(path: grub_cfg as String, content: grub_cfg_fixture_without_cmdline) - let cmdline = srv3_seeded_install_media_input.grub_kernel_cmdline + let cmdline = srv3_seeded_grub_kernel_cmdline let applied = install_media_remaster_ensure_grub_cmdline(grub_cfg: grub_cfg, cmdline: cmdline) let after = Filesystem.Read(path: grub_cfg as String) let idempotent_reapply = install_media_remaster_ensure_grub_cmdline(grub_cfg: grub_cfg, cmdline: cmdline) diff --git a/dag/test/claim/srv3/srv3_seeded_install_media_witness_test.dag b/dag/test/claim/srv3/srv3_seeded_install_media_witness_test.dag index 96fddee4602..ff242e401e6 100644 --- a/dag/test/claim/srv3/srv3_seeded_install_media_witness_test.dag +++ b/dag/test/claim/srv3/srv3_seeded_install_media_witness_test.dag @@ -5,20 +5,34 @@ data live_tree_disposition: LiveTreeDisposition = SubstrateInputsOnly data real_execution_split_doc: String = "install_media_remaster_ensure_grub_cmdline_inserts_by_real_execution moved to srv3_seeded_install_media_real_execution_witness_test.dag (no mock_response on shell.Mktemp.Dir/Filesystem.Write - dag/extdeps/shell/shell.dag - so it cannot run in CI's default Hermetic discovery corpus and is re-enrolled there as a declared Wet entry)." +// THE INHABITANCE CLAIM FOR THE REAL ROUTE: the derived input's body is what the renderer emitted +// for srv3's autoinstall, and a refusal from the YAML writer is a red here, never an empty document. +// The srv3 image is the same derivation as every other seeded image: its build key reaches the stock +// medium's pinned digest and the rendered autoinstall, and its identity is the measured output. test fn srv3_seeded_autoinstall_user_data_is_cloud_config() -> Bool { - let ud = srv3_autoinstall_user_data_body as String - string_contains(s: ud, pattern: "#cloud-config") - && string_contains(s: ud, pattern: "autoinstall:") - && string_contains(s: ud, pattern: "hostname: srv3") - && ud == autoinstall_user_data(payload: srv3_ubuntu_autoinstall_on_iso) + match srv3_seeded_install_media { + Srv3SeededInstallMediaRefused { reason: _ } => false + Srv3SeededInstallMediaDerived { input: i } => { + let ud = i.user_data_body as String + string_contains(s: ud, pattern: "#cloud-config") + && string_contains(s: ud, pattern: "autoinstall:") + && string_contains(s: ud, pattern: "hostname: srv3") + && match autoinstall_user_data(payload: srv3_ubuntu_autoinstall_on_iso) { + AutoinstallUserDataRendered { content: c } => ud == c + AutoinstallUserDataRefused { reason: _ } => false + } + } + } } -// The srv3 image is the same derivation as every other seeded image: its build key reaches the stock -// medium's pinned digest and the rendered autoinstall, and its identity is the measured output. test fn srv3_seeded_install_media_input_is_the_shared_derivation() -> Bool { - srv3_seeded_install_media_input.stock_artifact.content_sha256 == noble_numbat_2404_3_live_server_arm64.content_sha256 - && srv3_seeded_install_media_input.user_data_body == srv3_autoinstall_user_data_body - && seeded_install_media_build_key(input: srv3_seeded_install_media_input) != "" + match srv3_seeded_install_media { + Srv3SeededInstallMediaRefused { reason: _ } => false + Srv3SeededInstallMediaDerived { input: i } => + i.stock_artifact.content_sha256 == noble_numbat_2404_3_live_server_arm64.content_sha256 + && i.image_stem == srv3_seeded_install_media_image_stem + && seeded_install_media_build_key(input: i) != "" + } } test fn srv3_seeded_install_media_path_and_delivery() -> Bool { diff --git a/dag/test/claim/std_commit_sha_text_witness_test.dag b/dag/test/claim/std_commit_sha_text_witness_test.dag new file mode 100644 index 00000000000..dae607676bb --- /dev/null +++ b/dag/test/claim/std_commit_sha_text_witness_test.dag @@ -0,0 +1,89 @@ +module test.claim.std_commit_sha_text_witness + +import std.logic { Bool } +import v2.std.live_tree { LiveTreeDisposition, SubstrateInputsOnly } +import std.types { String, List, commit_sha_text_holds } + +data live_tree_disposition: LiveTreeDisposition = SubstrateInputsOnly + +// THE SUBJECT IS WHAT `std.types` `commit_sha_text_holds` ANSWERS, and it is homed here rather than +// beside any one caller because the predicate is std's. Its callers -- gunbc.action_use_admission +// admitting an action use, gunbc.ci_workflow_source_read admitting a ref to read at -- each have +// their own subject, and a predicate every one of them trusts owes its own discriminating cases. +// +// THE CLAIM BELOW IS AN EQUALITY, NOT A COST. The declaration was respelled to remove a +// per-character recursion from a hot, widely-shared predicate; what has to hold is that the new +// spelling answers exactly what the per-character one answered, on every input either can be given. +// `reference_per_character_spelling` IS that earlier spelling, kept here as the oracle, so a +// divergence is a red rather than a review question. Nothing about eval steps is asserted anywhere +// in this module: a step count is a property of the evaluator, and the contract is the answer. + +fn reference_per_character_spelling(head: String) -> Bool { + head.length() == 40 && all(chars(s: head), cp => + (cp >= 48 && cp <= 57) || (cp >= 97 && cp <= 102) + ) +} + +data a_real_commit: String = "cb39f95cfdee2fce8e78a73dad8887da4213c646" + +// Every input either spelling can be given, chosen so each refusal has a distinct cause: the two +// lengths either side of forty, a digit outside hex, an uppercase hex letter, the letter just past +// `f`, the characters adjacent to the digit and letter ranges on both sides (`/` 47, `:` 58, +// backtick 96, `g` 103), and the empty string. +data discriminating_inputs: List = [ + a_real_commit, + "0000000000000000000000000000000000000000", + "ffffffffffffffffffffffffffffffffffffffff", + "cb39f95cfdee2fce8e78a73dad8887da4213c64", + "cb39f95cfdee2fce8e78a73dad8887da4213c6466", + "CB39F95CFDEE2FCE8E78A73DAD8887DA4213C646", + "cb39f95cfdee2fce8e78a73dad8887da4213c64G", + "cb39f95cfdee2fce8e78a73dad8887da4213c64g", + "cb39f95cfdee2fce8e78a73dad8887da4213c64/", + "cb39f95cfdee2fce8e78a73dad8887da4213c64:", + "cb39f95cfdee2fce8e78a73dad8887da4213c64`", + "cb39f95cfdee2fce8e78a73dad8887da4213c64 ", + " b39f95cfdee2fce8e78a73dad8887da4213c646", + "", + "v4", + "main" +] + +// THE CLAIM: the respelled predicate is the same function as the one it replaced. +test fn the_respelled_predicate_answers_what_the_per_character_one_answered() -> Bool { + all(discriminating_inputs, h => commit_sha_text_holds(head: h) == reference_per_character_spelling(head: h)) +} + +test fn forty_lowercase_hex_digits_hold() -> Bool { + commit_sha_text_holds(head: a_real_commit) + && commit_sha_text_holds(head: "0000000000000000000000000000000000000000") + && commit_sha_text_holds(head: "ffffffffffffffffffffffffffffffffffffffff") +} + +// The red controls, one per cause. Each is a string the predicate must refuse, and each refuses for +// a different reason, so a spelling that lost one of them fails here rather than in a caller. +test fn red_a_length_either_side_of_forty_refuses() -> Bool { + !commit_sha_text_holds(head: "cb39f95cfdee2fce8e78a73dad8887da4213c64") + && !commit_sha_text_holds(head: "cb39f95cfdee2fce8e78a73dad8887da4213c6466") +} + +test fn red_uppercase_hex_refuses() -> Bool { + !commit_sha_text_holds(head: "CB39F95CFDEE2FCE8E78A73DAD8887DA4213C646") + && !commit_sha_text_holds(head: "cb39f95cfdee2fce8e78a73dad8887da4213c64F") +} + +// `g` is the character immediately after `f`, and `/` `:` backtick sit immediately outside the digit +// and lowercase-hex ranges, so a spelling with an off-by-one boundary is red here. +test fn red_a_character_outside_lowercase_hex_refuses() -> Bool { + !commit_sha_text_holds(head: "cb39f95cfdee2fce8e78a73dad8887da4213c64g") + && !commit_sha_text_holds(head: "cb39f95cfdee2fce8e78a73dad8887da4213c64/") + && !commit_sha_text_holds(head: "cb39f95cfdee2fce8e78a73dad8887da4213c64:") + && !commit_sha_text_holds(head: "cb39f95cfdee2fce8e78a73dad8887da4213c64`") + && !commit_sha_text_holds(head: "cb39f95cfdee2fce8e78a73dad8887da4213c64 ") +} + +test fn red_the_empty_string_and_a_branch_name_refuse() -> Bool { + !commit_sha_text_holds(head: "") + && !commit_sha_text_holds(head: "v4") + && !commit_sha_text_holds(head: "main") +} diff --git a/dag/test/claim/witness_floor_workflow_consolidation_witness_test.dag b/dag/test/claim/witness_floor_workflow_consolidation_witness_test.dag index 511ccd23b81..e40d35b0b08 100644 --- a/dag/test/claim/witness_floor_workflow_consolidation_witness_test.dag +++ b/dag/test/claim/witness_floor_workflow_consolidation_witness_test.dag @@ -10,7 +10,8 @@ import gunbc.fleet_converge_workflow { fleet_converge_workflow } import v2.workflow.bash_command_fold_serialize { bash_fold_serialize_program } import v2.std.diagnostic { Outcome } import extdeps.languages.yaml.gha_workflow { pr_activity_string, workflow_triggers_yaml } -import extdeps.languages.yaml.emit { serialize_yaml } +import extdeps.languages.yaml.types { YamlValue } +import extdeps.languages.yaml.emit { emit_yaml, EmittedYaml, YamlEmitRefused } import extdeps.github.actions { PullRequestActivity, Opened, Synchronize, Reopened, ReadyForReview, Closed, github_event_name_pull_request, github_event_name_push, github_event_name_merge_group, github_event_name_workflow_dispatch } import extdeps.github.actions { RunStep, UsesStep } import extdeps.github.expressions { @@ -477,7 +478,7 @@ test fn the_unrenderable_gate_refusal_serializes_and_stops_the_line() -> Bool { // cover the path end to end. A per-claim `expected_witness_floor_yml()` read is the whole-file // serialize this row must not pay (measured 1823ms against the 500ms line). test fn w_RED_the_pull_request_activity_set_is_the_three_that_carry_a_new_subject() -> Bool { - let yml = serialize_yaml(v: workflow_triggers_yaml(triggers: witness_floor_triggers())) + let yml = emitted_yaml_text(v: workflow_triggers_yaml(triggers: witness_floor_triggers())) string_contains( s: yml, pattern: concat("types: [", concat(pr_activity_string(a: Opened), concat(", ", concat(pr_activity_string(a: Synchronize), concat(", ", concat(pr_activity_string(a: Reopened), "]")))))) @@ -635,3 +636,12 @@ test fn w_RED_unmodeled_format_arity_and_base_ref_refuse() -> Bool { WitnessFloorGroupRefused { cause: _ } => false } } + +// The emitted text, or "" when the writer refuses: every claim below asserts content the refused +// value does not have, so a refusal cannot pass one. +fn emitted_yaml_text(v: YamlValue) -> String { + match emit_yaml(v: v) { + EmittedYaml { text: t } => t + YamlEmitRefused { path: _, reason: _ } => "" + } +} diff --git a/dag/test/claim/yaml_emit_witness_test.dag b/dag/test/claim/yaml_emit_witness_test.dag new file mode 100644 index 00000000000..390cebf22bc --- /dev/null +++ b/dag/test/claim/yaml_emit_witness_test.dag @@ -0,0 +1,128 @@ +module test.claim.yaml_emit_witness + +data live_tree_disposition: LiveTreeDisposition = SubstrateInputsOnly + +// THE WRITER'S CONTRACT IS DECODED STRUCTURE: for every value it writes, the reader returns that +// value -- ingest_yaml_source(emit(v)) == v, compared as values, never as re-serialized text (a +// text comparison is satisfied by YamlString "123" coming back as YamlInt "123"). A value it cannot +// write that way is refused with its path. A few byte-level claims pin the shapes the committed +// workflows depend on. + +fn round_trips(v: YamlValue) -> Bool { + match emit_yaml(v: v) { + EmittedYaml { text: t } => ingest_yaml_source(src: t) == IngestedYaml { value: v } + YamlEmitRefused { path: _, reason: _ } => false + } +} + +fn emitted(v: YamlValue) -> String { + match emit_yaml(v: v) { + EmittedYaml { text: t } => t + YamlEmitRefused { path: _, reason: _ } => "" + } +} + +fn str(s: String) -> YamlValue { + YamlString { value: s } +} + +fn field(k: String, v: YamlValue) -> YamlValue { + YamlMapping { entries: [YamlKeyValue { key: k, value: v }] } +} + +fn refused_at(v: YamlValue, path: String, reason_part: String) -> Bool { + match emit_yaml(v: v) { + EmittedYaml { text: _ } => false + YamlEmitRefused { path: p, reason: r } => (p == path) && string_contains(s: r, pattern: reason_part) + } +} + +// Strings that read as something else when written plain, or that carry an indicator, a comment +// marker, surrounding whitespace, an escape, or a GitHub expression. +data tricky_strings: List = [ + "123", "-3", "1.5", ".inf", "0x1F", "true", "False", "null", "~", "", "yes", "on", + "a: b", "x #c", "#123456", "- item", "? q", ": c", "[a]", "\{b}", "&anchor", "*alias", "!tag", + "|", ">", "'quoted'", "\"quoted\"", "%dir", "@at", "`tick", " lead", "trail ", "a\tb", + "back\\slash", "$\{\{ runner.os }}", "$\{\{ fromJSON('\{\"a\": 1}') }}", "ends:", "a:b", "plain text" +] + +test fn every_tricky_string_round_trips_as_a_value() -> Bool { + all(tricky_strings, s => round_trips(v: field(k: "v", v: str(s: s)))) +} + +test fn every_tricky_string_round_trips_as_a_key() -> Bool { + all(tricky_strings, s => round_trips(v: field(k: s, v: str(s: "x")))) +} + +test fn every_tricky_string_round_trips_in_a_flow_sequence() -> Bool { + round_trips(v: field(k: "v", v: YamlSequence { elements: map(tricky_strings, s => str(s: s)) })) +} + +// The failure the old text round trip could not see: a type change. +test fn red_a_digit_string_is_quoted_so_it_does_not_come_back_an_int() -> Bool { + (emitted(v: field(k: "v", v: str(s: "123"))) == "v: \"123\"\n") + && (emitted(v: field(k: "v", v: YamlInt { lexeme: "123" })) == "v: 123\n") + && round_trips(v: field(k: "v", v: YamlInt { lexeme: "123" })) +} + +data multiline_strings: List = [ + "l1\nl2", "l1\nl2\n", "l1\nl2\n\n", "a\n\nb", "\nlead-empty", "echo a\n# kept\necho b", + "\n", "\n\n", " leading space\nx", "x\n indented\ny", "trailing space \nx", "tab\tinside\nx", "x\n\ty" +] + +test fn every_multiline_string_round_trips_in_a_mapping_and_a_sequence() -> Bool { + all(multiline_strings, s => round_trips(v: field(k: "run", v: str(s: s)))) + && all(multiline_strings, s => round_trips(v: field(k: "items", v: YamlSequence { elements: [str(s: s), str(s: "after")] }))) +} + +// KEYS CARRY THE SAME CONTRACT AS VALUES. A key holding a line break cannot be written plain: the +// predicate used to admit it, and the writer then reported success on a document the reader either +// refuses or reads as different structure. +test fn every_multiline_string_round_trips_as_a_key() -> Bool { + all(multiline_strings, s => round_trips(v: field(k: s, v: str(s: "x")))) +} + +test fn red_a_key_with_a_line_break_is_double_quoted() -> Bool { + emitted(v: field(k: "a\nb", v: str(s: "x"))) == "\"a\\nb\": x\n" +} + +// The literal carries its final line breaks in its chomping indicator, and keeps empty lines. +test fn literal_blocks_pick_the_chomping_indicator_and_keep_empty_lines() -> Bool { + (emitted(v: field(k: "run", v: str(s: "a\n\nb"))) == "run: |-\n a\n\n b\n") + && (emitted(v: field(k: "run", v: str(s: "a\n"))) == "run: |\n a\n") + && (emitted(v: field(k: "run", v: str(s: "a\n\n"))) == "run: |+\n a\n\n") +} + +// A string a literal cannot carry is written double-quoted with escaped line breaks. +test fn red_a_string_no_literal_can_carry_is_double_quoted() -> Bool { + emitted(v: field(k: "run", v: str(s: " lead\nx"))) == "run: \" lead\\nx\"\n" +} + +test fn nested_collections_round_trip() -> Bool { + round_trips(v: YamlMapping { entries: [ + YamlKeyValue { key: "jobs", value: field(k: "build", v: field(k: "steps", v: YamlSequence { elements: [ + YamlMapping { entries: [ + YamlKeyValue { key: "uses", value: str(s: "actions/checkout@v4") }, + YamlKeyValue { key: "with", value: field(k: "fetch-depth", v: YamlInt { lexeme: "0" }) } + ] }, + field(k: "run", v: str(s: "make\nmake test")), + YamlSequence { elements: [str(s: "nested"), YamlSequence { elements: [str(s: "deeper")] }] }, + YamlSequence { elements: [YamlMapping { entries: [YamlKeyValue { key: "m", value: YamlNull }] }] }, + YamlMapping { entries: [] }, + YamlNull, + YamlBool { value: false }, + YamlFloat { lexeme: "2.5" } + ] })) }, + YamlKeyValue { key: "empty", value: YamlSequence { elements: [] } }, + YamlKeyValue { key: "none", value: YamlMapping { entries: [] } } + ] }) +} + +test fn red_values_the_writer_cannot_write_refuse_with_their_path() -> Bool { + refused_at(v: field(k: "jobs", v: field(k: "timeout", v: YamlInt { lexeme: "ten" })), path: "jobs.timeout", reason_part: "int lexeme `ten`") + && refused_at(v: field(k: "s", v: YamlSequence { elements: [YamlFloat { lexeme: "1.2.3" }] }), path: "s[0]", reason_part: "float lexeme") + && refused_at(v: YamlMapping { entries: [YamlKeyValue { key: "a", value: YamlNull }, YamlKeyValue { key: "a", value: YamlNull }] }, path: "a", reason_part: "twice") + && refused_at(v: str(s: "top"), path: "", reason_part: "document root") + && refused_at(v: YamlMapping { entries: [] }, path: "", reason_part: "empty mapping") + && refused_at(v: field(k: "a", v: str(s: concat("x", from_code_point(cp: 160)))), path: "line 1 of the emitted text", reason_part: "U+00A0") +} diff --git a/dag/test/claim/yaml_ingest_witness_test.dag b/dag/test/claim/yaml_ingest_witness_test.dag index ae9e7fe15c4..beeee9521dd 100644 --- a/dag/test/claim/yaml_ingest_witness_test.dag +++ b/dag/test/claim/yaml_ingest_witness_test.dag @@ -2,132 +2,304 @@ module test.claim.yaml_ingest_witness data live_tree_disposition: LiveTreeDisposition = SubstrateInputsOnly -fn round_trips(v: YamlValue) -> Bool { - let original = serialize_yaml(v: v) - match ingest_yaml_source(src: original) { - IngestedYaml { value: reparsed } => serialize_yaml(v: reparsed) == original - YamlIngestRejected { reason: _ } => false +// EVERY CLAIM HERE COMPARES DECODED STRUCTURE WITH A HAND-WRITTEN VALUE. Nothing goes through the +// emitter, so a loss in the writer cannot hide a loss in the reader. Each construct pairs its +// accepted reading with a red control: a refusal that must name its line and its reason, or a +// neighbouring reading the construct must not collapse into. + +fn reads(src: String, v: YamlValue) -> Bool { + ingest_yaml_source(src: src) == IngestedYaml { value: v } +} + +fn one(key: String, v: YamlValue) -> YamlValue { + YamlMapping { entries: [YamlKeyValue { key: key, value: v }] } +} + +fn str(s: String) -> YamlValue { + YamlString { value: s } +} + +fn refuses_at(src: String, line: Int, reason_part: String) -> Bool { + match ingest_yaml_source(src: src) { + IngestedYaml { value: _ } => false + YamlIngestRejected { line: l, reason: r } => (l == line) && string_contains(s: r, pattern: reason_part) } } -test fn witness_scalar_mapping_round_trips() -> Bool { - round_trips(v: yaml_mapping(entries: [ - kv(key: "name", value: yaml_string(s: "ci")), - kv(key: "fetch-depth", value: yaml_int(n: 0)), - kv(key: "quoted", value: yaml_bool(b: true)) - ])) +// --------------------------------------------------------------------------------------------- +// PLAIN SCALARS: THE CORE SCHEMA, NOT "UNRECOGNIZED TEXT IS A STRING" +// --------------------------------------------------------------------------------------------- + +test fn plain_text_is_a_string() -> Bool { + reads(src: "a: ubuntu-latest\n", v: one(key: "a", v: str(s: "ubuntu-latest"))) + && reads(src: "a: -D warnings\n", v: one(key: "a", v: str(s: "-D warnings"))) + && reads(src: "a: 2026-09-01.1\n", v: one(key: "a", v: str(s: "2026-09-01.1"))) } -test fn witness_nested_sequence_of_mappings_round_trips() -> Bool { - round_trips(v: yaml_mapping(entries: [ - kv(key: "jobs", value: yaml_sequence(elements: [ - yaml_mapping(entries: [ - kv(key: "uses", value: yaml_string(s: "actions/checkout@v4")), - kv(key: "with", value: yaml_mapping(entries: [ - kv(key: "fetch-depth", value: yaml_int(n: 1)) - ])) - ]) - ])) - ])) +test fn core_schema_resolves_null_bool_int_float() -> Bool { + reads(src: "a: null\nb: ~\nc:\nd: true\ne: FALSE\nf: 010\ng: -3\nh: 0x1F\ni: 0o17\nj: 1.5\nk: .5e3\nl: -.inf\nm: .NaN\n", v: YamlMapping { entries: [ + YamlKeyValue { key: "a", value: YamlNull }, + YamlKeyValue { key: "b", value: YamlNull }, + YamlKeyValue { key: "c", value: YamlNull }, + YamlKeyValue { key: "d", value: YamlBool { value: true } }, + YamlKeyValue { key: "e", value: YamlBool { value: false } }, + YamlKeyValue { key: "f", value: YamlInt { lexeme: "010" } }, + YamlKeyValue { key: "g", value: YamlInt { lexeme: "-3" } }, + YamlKeyValue { key: "h", value: YamlInt { lexeme: "0x1F" } }, + YamlKeyValue { key: "i", value: YamlInt { lexeme: "0o17" } }, + YamlKeyValue { key: "j", value: YamlFloat { lexeme: "1.5" } }, + YamlKeyValue { key: "k", value: YamlFloat { lexeme: ".5e3" } }, + YamlKeyValue { key: "l", value: YamlFloat { lexeme: "-.inf" } }, + YamlKeyValue { key: "m", value: YamlFloat { lexeme: ".NaN" } } + ] }) } -test fn witness_block_literal_round_trips() -> Bool { - round_trips(v: yaml_mapping(entries: [ - kv(key: "run", value: yaml_string(s: "line one\nline two")) - ])) +// Red control for the resolution: the digits are an int, never the string "123", and YAML 1.1's +// words are strings under the 1.2 core schema. +test fn red_digits_are_not_a_string_and_yes_is_not_a_bool() -> Bool { + !reads(src: "a: 123\n", v: one(key: "a", v: str(s: "123"))) + && reads(src: "a: 123\n", v: one(key: "a", v: YamlInt { lexeme: "123" })) + && reads(src: "a: yes\nb: on\n", v: YamlMapping { entries: [YamlKeyValue { key: "a", value: str(s: "yes") }, YamlKeyValue { key: "b", value: str(s: "on") }] }) } -test fn witness_empty_string_scalar_round_trips() -> Bool { - round_trips(v: yaml_mapping(entries: [kv(key: "rustflags", value: yaml_string(s: ""))])) +test fn red_a_plain_value_holding_a_mapping_indicator_refuses() -> Bool { + refuses_at(src: "x: 1\na: b: c\n", line: 2, reason_part: "cannot contain `: `") } -test fn witness_empty_document_rejected() -> Bool { - yaml_source_parses(src: "") == false +// --------------------------------------------------------------------------------------------- +// QUOTED SCALARS +// --------------------------------------------------------------------------------------------- + +test fn single_quoted_scalar_drops_its_quotes_and_decodes_doubled_quotes() -> Bool { + reads(src: "a: 'it''s'\nb: '123'\nc: ''\n", v: YamlMapping { entries: [ + YamlKeyValue { key: "a", value: str(s: "it's") }, + YamlKeyValue { key: "b", value: str(s: "123") }, + YamlKeyValue { key: "c", value: str(s: "") } + ] }) } -test fn witness_column_zero_full_line_comment_is_lexically_ignored() -> Bool { - match ingest_yaml_source(src: "# generated provenance\nname: witnesses\n") { - IngestedYaml { value } => serialize_yaml(v: value) == "name: witnesses\n" - YamlIngestRejected { reason: _ } => false - } +test fn red_a_single_quoted_scalar_with_trailing_text_refuses() -> Bool { + refuses_at(src: "a: 'x' y\n", line: 1, reason_part: "single-quoted") } -test fn witness_indented_full_line_comment_is_lexically_ignored() -> Bool { - match ingest_yaml_source(src: "root:\n # comment between block tokens\n child: x\n") { - IngestedYaml { value } => serialize_yaml(v: value) == "root:\n child: x\n" - YamlIngestRejected { reason: _ } => false - } +test fn double_quoted_escapes_decode() -> Bool { + reads(src: "a: \"x\\ny\\t\\\"q\\\" \\\\ \\u00e9 \\x41\"\n", v: one(key: "a", v: str(s: join(["x\ny\t\"q\" \\ ", from_code_point(cp: 233), " A"], "")))) + && reads(src: "a: \"#123456\"\n", v: one(key: "a", v: str(s: "#123456"))) } -test fn witness_inline_trailing_comment_refuses_instead_of_becoming_scalar_content() -> Bool { - match ingest_yaml_source(src: "name: x # trailing comment\n") { - IngestedYaml { value: _ } => false - YamlIngestRejected { reason } => reason == "inline YAML comments after values are not modeled" - } +test fn red_an_invalid_escape_refuses_and_names_itself() -> Bool { + refuses_at(src: "a: 1\nb: \"bad \\q\"\n", line: 2, reason_part: "`\\q` is not a YAML escape") + && refuses_at(src: "a: \"x\\\"\n", line: 1, reason_part: "lone backslash") + && refuses_at(src: "a: \"a\" \"b\"\n", line: 1, reason_part: "unescaped") } -test fn witness_tab_separated_inline_comment_refuses_instead_of_becoming_scalar_content() -> Bool { - match ingest_yaml_source(src: "name: x\t# trailing comment\n") { - IngestedYaml { value: _ } => false - YamlIngestRejected { reason } => reason == "inline YAML comments after values are not modeled" - } +test fn quoted_keys_decode() -> Bool { + reads(src: "\"uses\": x\n'on': y\n", v: YamlMapping { entries: [ + YamlKeyValue { key: "uses", value: str(s: "x") }, + YamlKeyValue { key: "on", value: str(s: "y") } + ] }) } -test fn witness_hash_inside_double_quoted_scalar_is_content() -> Bool { - round_trips(v: yaml_mapping(entries: [kv(key: "color", value: yaml_string(s: "#123456"))])) +test fn red_a_plain_key_the_core_schema_reads_as_non_string_refuses() -> Bool { + refuses_at(src: "true: x\n", line: 1, reason_part: "non-string") + && reads(src: "\"true\": x\n", v: one(key: "true", v: str(s: "x"))) } -test fn witness_shell_issue_reference_inside_block_literal_is_content() -> Bool { - let source = "run: |\n echo begin\n # dissolve-on: typed shell realization (#5828)\n" - match ingest_yaml_source(src: source) { - IngestedYaml { value } => serialize_yaml(v: value) == source - YamlIngestRejected { reason: _ } => false - } +// --------------------------------------------------------------------------------------------- +// FLOW COLLECTIONS +// --------------------------------------------------------------------------------------------- + +test fn flow_sequences_split_on_commas_outside_quotes() -> Bool { + reads(src: "a: [x,y]\nb: [\"x, y\", z]\nc: []\nd: [1, true]\n", v: YamlMapping { entries: [ + YamlKeyValue { key: "a", value: YamlSequence { elements: [str(s: "x"), str(s: "y")] } }, + YamlKeyValue { key: "b", value: YamlSequence { elements: [str(s: "x, y"), str(s: "z")] } }, + YamlKeyValue { key: "c", value: YamlSequence { elements: [] } }, + YamlKeyValue { key: "d", value: YamlSequence { elements: [YamlInt { lexeme: "1" }, YamlBool { value: true }] } } + ] }) } -test fn witness_comment_only_document_is_rejected() -> Bool { - !yaml_source_parses(src: "# no document node follows\n") +// A quote can only OPEN a scalar where an entry opens. Inside a plain scalar it is ordinary +// content, so the commas after it are still separators. +test fn a_quote_inside_a_plain_flow_entry_is_content_not_an_opener() -> Bool { + reads(src: "a: [a'b,c]\nb: [x\"y, z]\nc: [ \"p, q\" , r]\n", v: YamlMapping { entries: [ + YamlKeyValue { key: "a", value: YamlSequence { elements: [str(s: "a'b"), str(s: "c")] } }, + YamlKeyValue { key: "b", value: YamlSequence { elements: [str(s: "x\"y"), str(s: "z")] } }, + YamlKeyValue { key: "c", value: YamlSequence { elements: [str(s: "p, q"), str(s: "r")] } } + ] }) } -test fn witness_malformed_scalar_line_rejected() -> Bool { - yaml_source_parses(src: "just a bare word\n") == false +// Red control: the entry may not absorb the separator that follows the embedded quote. +test fn red_an_embedded_quote_does_not_hide_the_separator() -> Bool { + !reads(src: "a: [a'b,c]\n", v: one(key: "a", v: YamlSequence { elements: [str(s: "a'b,c")] })) } -test fn witness_bad_nested_indent_rejected() -> Bool { - yaml_source_parses(src: "a:\n b: 1\n") == false +test fn red_nested_and_unclosed_flow_sequences_refuse() -> Bool { + refuses_at(src: "a: [x, [y]]\n", line: 1, reason_part: "indicator") + && refuses_at(src: "a: [x,\n y]\n", line: 1, reason_part: "close on its own line") } -test fn witness_rejection_discriminates_from_valid() -> Bool { - yaml_source_parses(src: "a: 1\n") && - !yaml_source_parses(src: "just a bare word\n") +test fn red_a_flow_mapping_refuses_rather_than_becoming_a_string() -> Bool { + refuses_at(src: "steps:\n - \{uses: x}\n", line: 2, reason_part: "flow mappings") + && reads(src: "a: \{}\n", v: one(key: "a", v: YamlMapping { entries: [] })) } -// Blank lines and full-line comments between entries -- including a comment indented unlike its -// siblings, and one directly after a `key:` that opens a nested block -- carry no content. -test fn witness_blank_and_comment_lines_between_entries_are_ignored() -> Bool { - match ingest_yaml_source(src: "# header\nname: ci\n\njobs:\n # about the job\n build:\n\n # odd indent\n steps:\n - run: a\n\n # between steps\n - run: b\n") { - IngestedYaml { value: v } => serialize_yaml(v: v) == serialize_yaml(v: yaml_mapping(entries: [ - kv(key: "name", value: yaml_string(s: "ci")), - kv(key: "jobs", value: yaml_mapping(entries: [ - kv(key: "build", value: yaml_mapping(entries: [ - kv(key: "steps", value: yaml_sequence(elements: [ - yaml_mapping(entries: [kv(key: "run", value: yaml_string(s: "a"))]), - yaml_mapping(entries: [kv(key: "run", value: yaml_string(s: "b"))]) - ])) - ])) - ])) - ])) - YamlIngestRejected { reason: _ } => false - } +// --------------------------------------------------------------------------------------------- +// LITERAL BLOCK SCALARS +// --------------------------------------------------------------------------------------------- + +test fn literal_block_scalars_chomp_as_their_indicator_says() -> Bool { + reads(src: "a: |\n echo x\nb: |-\n echo y\nc: |+\n echo z\n\nd: 1\n", v: YamlMapping { entries: [ + YamlKeyValue { key: "a", value: str(s: "echo x\n") }, + YamlKeyValue { key: "b", value: str(s: "echo y") }, + YamlKeyValue { key: "c", value: str(s: "echo z\n\n") }, + YamlKeyValue { key: "d", value: YamlInt { lexeme: "1" } } + ] }) } -// ...and inside a block literal they ARE content: skipping them there would silently rewrite scripts. -test fn witness_blank_and_hash_lines_inside_a_block_literal_are_kept() -> Bool { - match ingest_yaml_source(src: "run: |\n echo a\n\n # kept\n echo b\nnext: x\n") { - IngestedYaml { value: v } => serialize_yaml(v: v) == serialize_yaml(v: yaml_mapping(entries: [ - kv(key: "run", value: yaml_string(s: "echo a\n\n# kept\necho b")), - kv(key: "next", value: yaml_string(s: "x")) - ])) - YamlIngestRejected { reason: _ } => false +// Blank lines and `#` lines inside a literal are content; the indentation is detected from the +// first content line, not assumed to be two spaces. +test fn literal_content_keeps_blank_and_hash_lines_and_detects_indentation() -> Bool { + reads(src: "run: |\n echo a\n\n # kept\n deeper\nnext: x\n", v: YamlMapping { entries: [ + YamlKeyValue { key: "run", value: str(s: "echo a\n\n# kept\n deeper\n") }, + YamlKeyValue { key: "next", value: str(s: "x") } + ] }) +} + +// EACH CHOMPING INDICATOR WITH ZERO, ONE AND THREE TRAILING EMPTY LINES, AT THE END OF THE DOCUMENT +// AND BEFORE A FOLLOWING KEY, OVER A BODY WITH A MORE-INDENTED MIDDLE LINE. The expected values are +// not this reader's opinion: each was produced by an independent YAML 1.2 implementation (the npm +// package `yaml`, v2, parsing the same bytes with `version: "1.2"`) and is written here as a literal, +// so a disagreement between the two readers is a red here and not a regenerated expectation. +fn literal_reads(source: String, expected: String) -> Bool { + if string_contains(s: source, pattern: "\nnext: 1\n") { + reads(src: source, v: YamlMapping { entries: [YamlKeyValue { key: "k", value: str(s: expected) }, YamlKeyValue { key: "next", value: YamlInt { lexeme: "1" } }] }) + } else { + reads(src: source, v: YamlMapping { entries: [YamlKeyValue { key: "k", value: str(s: expected) }] }) } } + +test fn literal_clip_chomping_matches_the_independent_reader_for_every_trailing_blank_count() -> Bool { + (literal_reads(source: "k: |\n a\n b\n c\n", expected: "a\n b\nc\n")) + && (literal_reads(source: "k: |\n a\n b\n c\nnext: 1\n", expected: "a\n b\nc\n")) + && (literal_reads(source: "k: |\n a\n b\n c\n\n", expected: "a\n b\nc\n")) + && (literal_reads(source: "k: |\n a\n b\n c\n\nnext: 1\n", expected: "a\n b\nc\n")) + && (literal_reads(source: "k: |\n a\n b\n c\n\n\n\n", expected: "a\n b\nc\n")) + && (literal_reads(source: "k: |\n a\n b\n c\n\n\n\nnext: 1\n", expected: "a\n b\nc\n")) +} + +test fn literal_strip_chomping_matches_the_independent_reader_for_every_trailing_blank_count() -> Bool { + (literal_reads(source: "k: |-\n a\n b\n c\n", expected: "a\n b\nc")) + && (literal_reads(source: "k: |-\n a\n b\n c\nnext: 1\n", expected: "a\n b\nc")) + && (literal_reads(source: "k: |-\n a\n b\n c\n\n", expected: "a\n b\nc")) + && (literal_reads(source: "k: |-\n a\n b\n c\n\nnext: 1\n", expected: "a\n b\nc")) + && (literal_reads(source: "k: |-\n a\n b\n c\n\n\n\n", expected: "a\n b\nc")) + && (literal_reads(source: "k: |-\n a\n b\n c\n\n\n\nnext: 1\n", expected: "a\n b\nc")) +} + +test fn literal_keep_chomping_matches_the_independent_reader_for_every_trailing_blank_count() -> Bool { + (literal_reads(source: "k: |+\n a\n b\n c\n", expected: "a\n b\nc\n")) + && (literal_reads(source: "k: |+\n a\n b\n c\nnext: 1\n", expected: "a\n b\nc\n")) + && (literal_reads(source: "k: |+\n a\n b\n c\n\n", expected: "a\n b\nc\n\n")) + && (literal_reads(source: "k: |+\n a\n b\n c\n\nnext: 1\n", expected: "a\n b\nc\n\n")) + && (literal_reads(source: "k: |+\n a\n b\n c\n\n\n\n", expected: "a\n b\nc\n\n\n\n")) + && (literal_reads(source: "k: |+\n a\n b\n c\n\n\n\nnext: 1\n", expected: "a\n b\nc\n\n\n\n")) +} + +// A REFUSAL IS AN ARM, NEVER A VALUE. The reader's block walk once carried a refused sequence item, +// flow entry or mapping entry as a value holding U+0001 and found refusals by scanning for it; a +// document whose decoded content legitimately held U+0001 (a double-quoted `\x01`) was then refused as +// if one of its items had been. Decoded U+0001 is ordinary content, as an item, a value and a key. +test fn a_decoded_u0001_is_content_as_an_item_a_value_and_a_key() -> Bool { + reads(src: "a:\n - \"\\x01\"\n - b\n", v: one(key: "a", v: YamlSequence { elements: [str(s: from_code_point(cp: 1)), str(s: "b")] })) + && reads(src: "a: \"\\x01\"\nb: c\n", v: YamlMapping { entries: [YamlKeyValue { key: "a", value: str(s: from_code_point(cp: 1)) }, YamlKeyValue { key: "b", value: str(s: "c") }] }) + && reads(src: "\"\\x01\": a\nb: c\n", v: YamlMapping { entries: [YamlKeyValue { key: from_code_point(cp: 1), value: str(s: "a") }, YamlKeyValue { key: "b", value: str(s: "c") }] }) +} + +// THE FLOW FAST PATH'S REFUSALS ARE TYPED ON THEIR OWN, WITH THE DOCUMENT PRE-CHECK BYPASSED: these +// call the entry and sequence readers directly, so ingest_yaml_source's refusal of U+0001 in the +// source never runs. An alias, a tag and an empty entry come back as refusals naming why, and an +// entry that IS U+0001 comes back as that string -- accepted, not mistaken for a refusal. +test fn red_a_flow_fast_path_refusal_is_typed_without_the_document_precheck() -> Bool { + (match yaml_flow_plain_entry(e: "*a") { YamlScalarRefused { reason: r } => string_contains(s: r, pattern: "starts with an indicator") ReadYamlScalar { value: _ } => false }) + && (match yaml_flow_plain_entry(e: "!t") { YamlScalarRefused { reason: r } => string_contains(s: r, pattern: "starts with an indicator") ReadYamlScalar { value: _ } => false }) + && (match yaml_flow_plain_entry(e: "") { YamlScalarRefused { reason: r } => string_contains(s: r, pattern: "empty flow sequence entry") ReadYamlScalar { value: _ } => false }) + && (match yaml_flow_plain_entry(e: from_code_point(cp: 1)) { ReadYamlScalar { value: v } => v == str(s: from_code_point(cp: 1)) YamlScalarRefused { reason: _ } => false }) + && (match yaml_flow_plain_sequence(inner: "a, *b, c") { YamlScalarRefused { reason: r } => string_contains(s: r, pattern: "`*b`") ReadYamlScalar { value: _ } => false }) + && (match yaml_flow_plain_sequence(inner: concat("a, ", from_code_point(cp: 1))) { ReadYamlScalar { value: v } => v == YamlSequence { elements: [str(s: "a"), str(s: from_code_point(cp: 1))] } YamlScalarRefused { reason: _ } => false }) +} + +test fn red_folded_scalars_and_indentation_indicators_refuse() -> Bool { + refuses_at(src: "a: >\n x\n", line: 1, reason_part: "folded") + && refuses_at(src: "a: |2\n x\n", line: 1, reason_part: "block scalar header") +} + +// --------------------------------------------------------------------------------------------- +// BLOCK STRUCTURE +// --------------------------------------------------------------------------------------------- + +test fn nested_mappings_and_sequences_decode() -> Bool { + reads(src: "jobs:\n build:\n steps:\n - uses: actions/checkout@v4\n with:\n fetch-depth: 1\n - run: make\n", v: one(key: "jobs", v: one(key: "build", v: one(key: "steps", v: YamlSequence { elements: [ + YamlMapping { entries: [ + YamlKeyValue { key: "uses", value: str(s: "actions/checkout@v4") }, + YamlKeyValue { key: "with", value: one(key: "fetch-depth", v: YamlInt { lexeme: "1" }) } + ] }, + one(key: "run", v: str(s: "make")) + ] })))) +} + +test fn a_sequence_at_its_keys_column_and_a_compact_nested_sequence_decode() -> Bool { + reads(src: "a:\n- x\n- - y\n - z\nb: 1\n", v: YamlMapping { entries: [ + YamlKeyValue { key: "a", value: YamlSequence { elements: [str(s: "x"), YamlSequence { elements: [str(s: "y"), str(s: "z")] }] } }, + YamlKeyValue { key: "b", value: YamlInt { lexeme: "1" } } + ] }) +} + +test fn wider_indentation_decodes() -> Bool { + reads(src: "a:\n b: 1\n c:\n - x\n", v: one(key: "a", v: YamlMapping { entries: [ + YamlKeyValue { key: "b", value: YamlInt { lexeme: "1" } }, + YamlKeyValue { key: "c", value: YamlSequence { elements: [str(s: "x")] } } + ] })) +} + +test fn red_irregular_indentation_and_multi_line_plain_scalars_refuse() -> Bool { + refuses_at(src: "a:\n b: 1\n c: 2\n", line: 3, reason_part: "unexpected indentation") + && refuses_at(src: "a: one\n two\n", line: 2, reason_part: "unexpected indentation") +} + +test fn red_a_duplicate_key_refuses_at_its_second_occurrence() -> Bool { + refuses_at(src: "a: 1\nb: 2\na: 3\n", line: 3, reason_part: "duplicate mapping key `a`") +} + +// --------------------------------------------------------------------------------------------- +// COMMENTS AND BLANK LINES +// --------------------------------------------------------------------------------------------- + +test fn comments_and_blank_lines_between_nodes_carry_no_content() -> Bool { + reads(src: "# header\nname: ci\n\njobs:\n # about the job\n build:\n\n # odd indent\n steps:\n - run: a\n\n # between steps\n - run: b\n", v: YamlMapping { entries: [ + YamlKeyValue { key: "name", value: str(s: "ci") }, + YamlKeyValue { key: "jobs", value: one(key: "build", v: one(key: "steps", v: YamlSequence { elements: [one(key: "run", v: str(s: "a")), one(key: "run", v: str(s: "b"))] })) } + ] }) +} + +test fn red_a_comment_after_a_value_refuses_rather_than_becoming_content() -> Bool { + refuses_at(src: "name: x # trailing comment\n", line: 1, reason_part: "inline YAML comments") + && refuses_at(src: "name: # comment\n", line: 1, reason_part: "inline YAML comments") + && refuses_at(src: "a: 'x' # comment\n", line: 1, reason_part: "single-quoted") +} + +// --------------------------------------------------------------------------------------------- +// DOCUMENT AND CHARACTER REPERTOIRE +// --------------------------------------------------------------------------------------------- + +test fn red_document_level_refusals_are_located() -> Bool { + refuses_at(src: "", line: 0, reason_part: "no content node") + && refuses_at(src: "# only a comment\n", line: 0, reason_part: "no content node") + && refuses_at(src: "just a bare word\n", line: 1, reason_part: "expected a block mapping entry") + && refuses_at(src: "---\na: 1\n", line: 1, reason_part: "document markers") + && refuses_at(src: "a: &x 1\n", line: 1, reason_part: "anchors") + && refuses_at(src: "a: !!str 1\n", line: 1, reason_part: "tags") + && refuses_at(src: "a:\n\tb: 1\n", line: 2, reason_part: "tab") + && refuses_at(src: "a: 1\nb: 2 \n", line: 2, reason_part: "ends in whitespace") + && refuses_at(src: concat("a: 1\nb: x", concat(from_code_point(cp: 160), "y\n")), line: 2, reason_part: "U+00A0") + && refuses_at(src: "a: 1\r\nb: 2\r\n", line: 1, reason_part: "U+000D") +} diff --git a/dag/test/claim/yaml_multiline_block_witness_test.dag b/dag/test/claim/yaml_multiline_block_witness_test.dag deleted file mode 100644 index 4fb1f8f8913..00000000000 --- a/dag/test/claim/yaml_multiline_block_witness_test.dag +++ /dev/null @@ -1,32 +0,0 @@ -module test.claim.yaml_multiline_block_witness - -data live_tree_disposition: LiveTreeDisposition = SubstrateInputsOnly - -test fn witness_top_level_multiline_is_block_literal() -> Bool { - let out = serialize_yaml(v: yaml_string(s: "l1\nl2")) - (out == "|\n l1\n l2\n") && - (out != "l1\nl2\n") -} - -test fn witness_multiline_sequence_element_is_block_literal() -> Bool { - let out = serialize_yaml(v: yaml_mapping(entries: [ - kv(key: "items", value: yaml_sequence(elements: [yaml_string(s: "l1\nl2")])) - ])) - (out == "items:\n - |\n l1\n l2\n") && - (out != "items: [l1\nl2]\n") -} - -test fn witness_multiline_mapping_value_still_block_literal() -> Bool { - let out = serialize_yaml(v: yaml_mapping(entries: [ - kv(key: "run", value: yaml_string(s: "l1\nl2")) - ])) - out == "run: |\n l1\n l2\n" -} - -test fn witness_no_embedded_newline_in_any_position() -> Bool { - let top = serialize_yaml(v: yaml_string(s: "a\nb")) - let seq = serialize_yaml(v: yaml_sequence(elements: [yaml_string(s: "a\nb")])) - (top == "|\n a\n b\n") && - (seq == "- |\n a\n b\n") -} - diff --git a/dag/test/claim/yaml_quoting_witness_test.dag b/dag/test/claim/yaml_quoting_witness_test.dag deleted file mode 100644 index e67f51dc019..00000000000 --- a/dag/test/claim/yaml_quoting_witness_test.dag +++ /dev/null @@ -1,46 +0,0 @@ -module test.claim.yaml_quoting_witness - -data live_tree_disposition: LiveTreeDisposition = SubstrateInputsOnly - -fn ser_kv(k: String, s: String) -> String { - serialize_yaml(v: yaml_mapping(entries: [kv(key: k, value: yaml_string(s: s))])) -} - -test fn witness_reserved_word_is_quoted() -> Bool { - (ser_kv(k: "v", s: "true") == "v: \"true\"\n") && - (ser_kv(k: "v", s: "yes") == "v: \"yes\"\n") && - (ser_kv(k: "v", s: "null") == "v: \"null\"\n") -} - -test fn witness_plain_word_not_quoted() -> Bool { - ser_kv(k: "v", s: "tru") == "v: tru\n" -} - -test fn witness_reserved_quoting_discriminates() -> Bool { - ser_kv(k: "v", s: "true") != ser_kv(k: "v", s: "tru") -} - -test fn witness_colon_space_forces_quotes() -> Bool { - ser_kv(k: "v", s: "a: b") == "v: \"a: b\"\n" -} - -test fn witness_hash_after_space_forces_quotes() -> Bool { - ser_kv(k: "v", s: "x #c") == "v: \"x #c\"\n" -} - -test fn witness_empty_string_is_single_quotes() -> Bool { - ser_kv(k: "v", s: "") == "v: ''\n" -} - -test fn witness_leading_indicator_quoted_and_escaped() -> Bool { - ser_kv(k: "v", s: "\"hi\"") == "v: \"\\\"hi\\\"\"\n" -} - -test fn witness_backslash_escaped_when_quoted() -> Bool { - ser_kv(k: "v", s: "\": a\\b") == "v: \"\\\": a\\\\b\"\n" -} - -test fn witness_expression_passthrough_not_quoted() -> Bool { - ser_kv(k: "v", s: "${{ runner.os }}") == "v: ${{ runner.os }}\n" -} - diff --git a/src/v1/stage0/src/std_types.rs b/src/v1/stage0/src/std_types.rs index 40a1612118a..992a486035b 100644 --- a/src/v1/stage0/src/std_types.rs +++ b/src/v1/stage0/src/std_types.rs @@ -202,21 +202,72 @@ pub fn list_length(items: Rc>) -> i64 { pub type CommitSha = String; pub fn commit_sha_text_holds(head: String) -> bool { - ((v1_rt::string_length(&head) == 40) && { - let mut __all = true; - for cp in Rc::new(head.clone().chars().map(|c| c as i64).collect::>()) - .iter() - .cloned() - { - if !(((cp.clone() >= 48) && (cp.clone() <= 57)) - || ((cp.clone() >= 97) && (cp.clone() <= 102))) - { - __all = false; - break; - } - } - __all - }) + ((v1_rt::string_length(&head) == 40) + && (v1_rt::replace( + v1_rt::replace( + v1_rt::replace( + v1_rt::replace( + v1_rt::replace( + v1_rt::replace( + v1_rt::replace( + v1_rt::replace( + v1_rt::replace( + v1_rt::replace( + v1_rt::replace( + v1_rt::replace( + v1_rt::replace( + v1_rt::replace( + v1_rt::replace( + v1_rt::replace( + head.clone(), + "0".to_string(), + "".to_string(), + ), + "1".to_string(), + "".to_string(), + ), + "2".to_string(), + "".to_string(), + ), + "3".to_string(), + "".to_string(), + ), + "4".to_string(), + "".to_string(), + ), + "5".to_string(), + "".to_string(), + ), + "6".to_string(), + "".to_string(), + ), + "7".to_string(), + "".to_string(), + ), + "8".to_string(), + "".to_string(), + ), + "9".to_string(), + "".to_string(), + ), + "a".to_string(), + "".to_string(), + ), + "b".to_string(), + "".to_string(), + ), + "c".to_string(), + "".to_string(), + ), + "d".to_string(), + "".to_string(), + ), + "e".to_string(), + "".to_string(), + ), + "f".to_string(), + "".to_string(), + ) == "".to_string())) } pub type Sha256 = String; diff --git a/src/v2/test/claim/gha_workflow_yaml_fold_serialize_test.dag b/src/v2/test/claim/gha_workflow_yaml_fold_serialize_test.dag index 63da7c6ffc5..4b05834ab1e 100644 --- a/src/v2/test/claim/gha_workflow_yaml_fold_serialize_test.dag +++ b/src/v2/test/claim/gha_workflow_yaml_fold_serialize_test.dag @@ -1,6 +1,7 @@ module v2.test.claim.gha_workflow_yaml_fold_serialize -import extdeps.languages.yaml.emit { serialize_yaml } +import extdeps.languages.yaml.types { YamlValue } +import extdeps.languages.yaml.emit { emit_yaml, EmittedYaml, YamlEmitRefused } import extdeps.github.actions { Job } import gunbc.gha_yaml_fold_pilot { gha_fold_pilot_job } import extdeps.languages.yaml.gha_workflow { job_yaml } @@ -25,7 +26,7 @@ import v2.std.live_tree { LiveTreeDisposition, SubstrateInputsOnly } data live_tree_disposition: LiveTreeDisposition = SubstrateInputsOnly fn gha_fold_oracle_job(job: Job) -> String { - serialize_yaml(v: job_yaml(job: job)) + emitted_yaml_text(v: job_yaml(job: job)) } fn gha_fold_serialize_emitted(target: TargetModel, emitted: Node) -> Outcome { @@ -91,3 +92,12 @@ test fn gha_fold_flat_steps_perturbation_rejects_oracle_holds() -> Bool { Violates { diagnostic: _ } => true } } + +// The emitted text, or "" when the writer refuses: every claim below asserts content the refused +// value does not have, so a refusal cannot pass one. +fn emitted_yaml_text(v: YamlValue) -> String { + match emit_yaml(v: v) { + EmittedYaml { text: t } => t + YamlEmitRefused { path: _, reason: _ } => "" + } +} diff --git a/src/v2/workflow/local_repo_wet_terminal.dag b/src/v2/workflow/local_repo_wet_terminal.dag index faaa63ce9dc..4190316be69 100644 --- a/src/v2/workflow/local_repo_wet_terminal.dag +++ b/src/v2/workflow/local_repo_wet_terminal.dag @@ -707,6 +707,12 @@ fn local_repo_wet_schedule() -> List { function: "observe_install_media_fetch_covers_absent_verified_and_mismatch_by_real_execution", expectation: ExpectedToHold {} }, + WetScheduledClaim { + identity: WitnessIdentity { module_path: "test.claim.srv3_seeded_install_media_real_execution", function: "install_media_remaster_ensure_grub_cmdline_inserts_by_real_execution" }, + entry: "dag/test/claim/srv3/srv3_seeded_install_media_real_execution_witness_test.dag", + function: "install_media_remaster_ensure_grub_cmdline_inserts_by_real_execution", + expectation: ExpectedToHold {} + }, WetScheduledClaim { identity: WitnessIdentity { module_path: "test.claim.mtcollins1_census_image_local_wet", function: "the_attempt_three_capture_binds_and_reads_by_real_execution" }, entry: "dag/test/claim/machine_intake/mtcollins1_census_image_local_wet_test.dag", diff --git a/src/v2/workflow/required_floor.dag b/src/v2/workflow/required_floor.dag index 8f70ae904e5..caa8314e713 100644 --- a/src/v2/workflow/required_floor.dag +++ b/src/v2/workflow/required_floor.dag @@ -627,9 +627,10 @@ fn fixture_home_prefixes() -> List { // IT DECLARES ReadsLiveTree TRUTHFULLY -- it reads the four fixture files from the checkout -- and // that is no longer a decline: `DeclinedLiveTree` is deleted at the root above. // -// THE REALIZED-WORKFLOW ACTION-USE CENSUS IS ADMITTED AT EXACT MODULE GRAIN. Its -// every_realized_workflow_action_use_is_modeled_and_executed_exactly row is the only route that -// admits hand-authored .github/workflows files, which no generator sees, and +// THE REALIZED-WORKFLOW ACTION-USE CENSUS IS ADMITTED AT EXACT MODULE GRAIN. Its per-file census +// rows (census__uses_are_modeled_and_executed_exactly, joined to the directory listing by +// every_executed_workflow_file_has_a_census_claim_and_no_claim_names_a_missing_file) are the only +// route that admits hand-authored .github/workflows files, which no generator sees, and // gunbc.recurring_failure_mode.provider_substituted_action_runtime cites it as the executing wall. // Outside the gate it would disposition DeclinedOutsideRequiredGate, and changed-witness selection // fires only on edits to the witness's own test fns -- so a tag or node20 pin reintroduced into a @@ -1450,12 +1451,17 @@ type CorpusCensusMember { subject_path: String } -// THE ROSTER. #11730 adds the realized-workflow census row (fleet-converge.yml); until then it holds -// no production row, and the arm is exercised by the witnesses in -// test.claim.corpus_census_allowance_witness over synthetic subjects. An empty roster is a legitimate -// state -- no whole-file census needs a line -- unlike the grandfathered roster, whose emptiness would -// reclassify the corpus. -data corpus_census_roster: List = [] +// THE ROSTER. Its one production row is the realized-workflow census over the largest committed +// workflow (gunbc.action_use_admission, gunb-ai/gunbc#11730), whose subject is a generated file other +// lanes grow; the arm is also exercised by the witnesses in test.claim.corpus_census_allowance_witness +// over synthetic subjects. An empty roster is a legitimate state -- no whole-file census needs a line +// -- unlike the grandfathered roster, whose emptiness would reclassify the corpus. +data corpus_census_roster: List = [ + CorpusCensusMember { + identity: "test.claim.action_use_admission_witness.census_fleet_converge_uses_are_modeled_and_executed_exactly", + subject_path: ".github/workflows/fleet-converge.yml", + }, +] // THE POLICY, WITH ITS REASON IN THE CARRIER RATHER THAN BESIDE IT. type CorpusCensusAllowancePolicy {