diff --git a/.github/workflows/ci.yaml b/.github/workflows/ci.yaml index caf4e2d3cb00..21916baf2282 100644 --- a/.github/workflows/ci.yaml +++ b/.github/workflows/ci.yaml @@ -75,16 +75,6 @@ jobs: if: needs.detect.outputs.python == 'true' uses: ./.github/workflows/tests.yml - # macOS + Windows lanes. The main `tests` lane above is Linux-only, and - # the OS-marked tests it collects are skipped there by design (see the - # `_OS_MARKS` comment in tests/conftest.py) — this is where they run. - # Same `python` lane gate: if no Python changed, neither runs. - tests-os: - name: OS-specific tests - needs: detect - if: needs.detect.outputs.python == 'true' - uses: ./.github/workflows/tests-os.yml - lint: name: Python lints needs: detect @@ -218,7 +208,6 @@ jobs: needs: - detect - tests - - tests-os - lint - js-tests - installer-tests diff --git a/.github/workflows/docker.yml b/.github/workflows/docker.yml index f245708486fb..284b9103a22c 100644 --- a/.github/workflows/docker.yml +++ b/.github/workflows/docker.yml @@ -172,10 +172,9 @@ jobs: NOUS_API_KEY: "" run: | # Each of these tests drives a container, so the docker daemon sets - # the limit and not the processor. This caps the workers. The - # default from run_tests.sh is cpu_count*2, which starts 64 - # containers together on the 32-core amd64 runner. - HERMES_TEST_WORKERS=$(nproc) scripts/run_tests.sh tests/docker/ --file-timeout 600 + # the limit and not the processor. This pins the xdist worker count + # to the core count. + HERMES_TEST_WORKERS=$(nproc) scripts/run_tests.sh tests/docker/ # --------------------------------------------------------------------------- # Rebuild and push each architecture only after the unprivileged build/test diff --git a/.github/workflows/tests-os.yml b/.github/workflows/tests-os.yml deleted file mode 100644 index 12719a853c54..000000000000 --- a/.github/workflows/tests-os.yml +++ /dev/null @@ -1,154 +0,0 @@ -name: OS-specific tests - -# Runs the tests that can only be trusted on their own host OS. -# -# The main Python suite (.github/workflows/tests.yml) runs on -# ubuntu-latest and covers everything that is either platform-agnostic or -# genuinely Linux-specific. Tests whose subject is macOS- or -# Windows-specific behaviour carry a marker (see the ``_OS_MARKS`` block -# comment in tests/conftest.py) and are SKIPPED on Linux, because faking -# ``sys.platform`` on a Linux runner selects the branch under test without -# reproducing any of the OS behaviour that branch exists for. This workflow -# is where those markers actually execute: -# -# macos → ``-m macos_only`` on macos-latest -# windows → ``-m windows_only`` on windows-latest -# -# Deliberately NOT sliced. The marked set is small (tens of tests, not -# thousands), so one plain ``pytest`` process per OS is both faster and far -# less machinery than the per-file parallel runner the Linux lane uses. -# If either lane grows past its timeout, that is the signal to reach for -# scripts/run_tests.sh here too. -# -# Each lane FAILS when it selects zero tests (pytest exit code 5). Without -# that guard, a renamed marker or a bad selector would report a green job -# that ran nothing — the exact silent-coverage-loss failure this workflow -# exists to prevent. - -on: - workflow_call: - -permissions: - contents: read - -concurrency: - group: tests-os-${{ github.ref }} - cancel-in-progress: true - -jobs: - os-tests: - name: ${{ matrix.name }} - runs-on: ${{ matrix.runner }} - timeout-minutes: 30 - strategy: - fail-fast: false - matrix: - include: - - name: macOS-only tests - runner: macos-latest - marker: macos_only - - name: Windows-only tests - runner: windows-latest-32-core - marker: windows_only - steps: - - name: Checkout code - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 - - - name: Install uv - uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0 - with: - # Pinned for the same reason as the Linux lane: unpinned, setup-uv - # resolves "latest" by fetching a manifest on every job and a - # transient fetch failure fails the whole job. - version: "0.9.28" - enable-cache: true - cache-dependency-glob: | - pyproject.toml - uv.lock - - - name: Set up Python 3.11 - uses: ./.github/actions/retry - with: - command: uv python install 3.11 - - - name: Install dependencies - # Same extras as the Linux test lane so an OS-marked test can import - # anything its Linux siblings can. ``[all]`` is deliberately - # Windows/macOS-installable (see the policy comment on the extra in - # pyproject.toml — matrix/python-olm was removed from it precisely - # because it could not build here). - uses: ./.github/actions/retry - with: - command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra hindsight --extra parallel-web - - - name: Minimize uv cache - run: uv cache prune --ci - - - name: Run ${{ matrix.marker }} tests - # Two-step selection: - # - # 1. scripts/ci/list_os_marked_tests.py narrows WHICH FILES are - # imported. ``-m`` filters after collection, and collection - # imports every module under tests/ — on this host that would - # drag ~900 unrelated test modules through import, where a - # single unrelated ImportError would fail a job whose own - # subject is fine. The helper exits non-zero if the marker - # matches no file at all. - # 2. ``-m`` decides WHICH TESTS run, and stays authoritative. - # Passing it on the command line REPLACES pyproject's - # ``-m 'not integration'`` addopts (same option, last wins) — - # hence repeating ``not integration``, or the integration - # suite would return through the side door. - # - # ``--timeout-method`` needs no override: tests/conftest.py's - # pytest_configure already downgrades the signal-based timer on - # Windows, which has no SIGALRM. - shell: bash - run: | - set -uo pipefail - - LIST="${RUNNER_TEMP:-.}/selected-tests.txt" - - # Process substitution would hide the helper's exit status, so write - # to a file and check it explicitly. - if ! uv run --no-sync python scripts/ci/list_os_marked_tests.py \ - "${{ matrix.marker }}" > "$LIST"; then - echo "::error::could not enumerate ${{ matrix.marker }} test files" - exit 1 - fi - if [ ! -s "$LIST" ]; then - echo "::error::empty ${{ matrix.marker }} file list" - exit 1 - fi - - # Deliberately NOT `mapfile`: that is a bash 4 builtin and the macOS - # runner's /bin/bash is 3.2. Word-splitting is safe here because the - # helper emits repo-relative test paths, which contain no spaces. - # shellcheck disable=SC2046 - set -- $(cat "$LIST") - echo "selected $# file(s) for ${{ matrix.marker }}:" - cat "$LIST" - - # ``shell: bash`` runs this script with ``-e`` injected, which - # ``set -uo pipefail`` above does not clear. A bare pytest call - # would therefore abort the script on any non-zero exit and the - # exit-5 branch below would be unreachable dead code — the job - # would still fail red, but the diagnostic would never print. - status=0 - uv run --no-sync python -m pytest \ - "$@" \ - -m "${{ matrix.marker }} and not integration" \ - -v --tb=short || status=$? - if [ "$status" -eq 5 ]; then - echo "::error::No tests matched -m ${{ matrix.marker }}. Either the" \ - "marker was renamed/dropped or selection is broken — this job" \ - "must never pass without running its OS's tests." - exit 1 - fi - exit "$status" - env: - # Belt-and-suspenders with tests/conftest.py's env blanking: no - # test may reach a real provider API. - OPENROUTER_API_KEY: "" - OPENAI_API_KEY: "" - NOUS_API_KEY: "" diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index a0dca54ad003..5838f7beb500 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -13,21 +13,80 @@ concurrency: jobs: test: - name: Run tests - # One 96-core runner for the whole suite. There is no slicing. Slicing - # existed to spread the suite over 4-core runners. It cost a matrix job, a - # duration cache, a per-slice artifact and a merge job to do it. - # - # 96 cores clear the floor that the slowest single test file sets (about - # 82s). A second slice divides work that is already at that floor, and - # adds a second setup. - runs-on: ubuntu-latest-96-core - timeout-minutes: 30 + # One matrix covers three host OSes. Linux and Windows run the full suite. + # macOS runs only its platform-gated macOS tests for now (the suite is not yet + # macOS-clean). tests/conftest.py's ``pytest_collection_modifyitems`` hook + # skips foreign-OS markers on each host, so an un-gated full run selects + # "generic + this host's own marker" — no ``-m`` filter is needed for the + # full-suite lanes. Nothing slices by platform now: every lane runs the + # whole discoverable suite and lets the markers do the gating. + name: Python tests (${{ matrix.os }}) + runs-on: ${{ matrix.runner }} + # The windows lane needs a bigger budget than the others: the full suite + # there has a long tail of files that spawn servers/sockets/subprocesses + # and run minutes-per-file on the CI runner (they pass in seconds on an + # idle dev box) — measured 88.6% completion at 60 minutes on run + # 33338310764. + timeout-minutes: ${{ matrix.timeout_min }} + strategy: + fail-fast: false + matrix: + include: + - os: linux + runner: ubuntu-latest-96-core + marker: "" + workers: "96" + timeout_min: 30 + - os: windows + runner: windows-latest-32-core + marker: "" + workers: "32" + timeout_min: 150 + - os: macos + runner: macos-latest + marker: macos + workers: "" + timeout_min: 30 steps: + - name: Disable Windows Defender real-time scanning + # Process-spawn-heavy test files pay Defender's real-time scan on + # every python.exe spawn / temp write, which inflated the lane's tail + # to minutes-per-file (seconds on an idle dev box). The runner is + # ephemeral and single-purpose; scanning it protects nothing. Best + # effort — some runner images may refuse, and the lane still passes, + # just slower. + if: matrix.os == 'windows' + shell: pwsh + run: | + try { + Set-MpPreference -DisableRealtimeMonitoring $true + Write-Host "Defender real-time monitoring disabled." + } catch { + Write-Host "Could not disable windows defender" + } + + - name: Put Git bash ahead of the WSL stub on PATH + # windows-2025 ships C:\Windows\System32\bash.exe — the WSL launcher + # stub with no distro installed. PATH puts System32 before Git, so + # every test that resolves "bash" spawns the stub and gets its UTF-16 + # "Windows Subsystem for Linux" banner with exit 1. Prepend Git's bin + # so "bash" resolves to the real MSYS bash the harnesses expect. + if: matrix.os == 'windows' + shell: pwsh + run: | + $gitBin = "C:\Program Files\Git\bin" + if (Test-Path "$gitBin\bash.exe") { + Add-Content $env:GITHUB_PATH $gitBin + Write-Host "Prepended $gitBin to PATH" + } else { + Write-Host "::warning::Git bash not found at $gitBin" + } + - name: Checkout code uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 - name: Install ripgrep (prebuilt binary) + if: matrix.os == 'linux' run: | set -euo pipefail RG_VERSION=15.1.0 @@ -88,38 +147,79 @@ jobs: run: uv cache prune --ci - name: Run tests - # Per-file isolation via scripts/run_tests.sh: each test file runs - # in its own freshly-spawned `python -m pytest ` subprocess - # with bounded parallelism. No xdist, no shared workers, no - # module-level state leakage between files. + # Two shapes: # - # No --files: the runner discovers the suite itself. The discovered - # set is identical to the list the removed matrix job used to pass in. + # * Full-suite lanes (Linux + Windows): scripts/run_tests.sh, + # which dispatches on host because the cost profiles are + # opposites. Linux: per-file subprocess isolation (spawn is + # ~15ms there, each file collected exactly once, zero + # cross-file state pollution). Windows: pytest-xdist + # --dist loadfile (per-file spawn is 0.5-1.5s there, a ~6-min + # floor the old model paid; persistent workers amortize the + # interpreter+import wall, and loadfile pins a file's tests to + # ONE worker so remaining hazards are stateful-test bugs to + # fix, not runner bugs). + # * macOS (marker set): plain pytest over the files that carry the + # marker. list_os_marked_tests.py narrows WHICH FILES are + # imported (collection otherwise drags ~900 unrelated modules + # through import on the macOS host), and `-m` stays the + # authoritative selector. Passing `-m` REPLACES pyproject's + # ``-m 'not integration'`` addopts, so ``not integration`` is + # repeated or the integration suite returns through the side + # door. The lane FAILS on pytest exit 5 (zero tests selected) so + # a renamed/broken marker can never report green while running + # nothing. + shell: bash run: | - source .venv/bin/activate + set -uo pipefail + + + if [ -n "${{ matrix.marker }}" ]; then + LIST="${RUNNER_TEMP:-.}/selected-tests.txt" + + if ! uv run --no-sync python scripts/ci/list_os_marked_tests.py \ + "${{ matrix.marker }}" > "$LIST"; then + echo "::error::could not enumerate ${{ matrix.marker }} test files" + exit 1 + fi + if [ ! -s "$LIST" ]; then + echo "::error::empty ${{ matrix.marker }} file list" + exit 1 + fi + + # Deliberately NOT `mapfile`: that is a bash 4 builtin and the + # macOS runner's /bin/bash is 3.2. Word-splitting is safe here + # because the helper emits repo-relative paths with no spaces. + # shellcheck disable=SC2046 + set -- $(cat "$LIST") + echo "selected $# file(s) for ${{ matrix.marker }}:" + cat "$LIST" + + status=0 + uv run --no-sync python -m pytest \ + "$@" \ + -m "platforms and not integration" \ + -v --tb=short || status=$? + if [ "$status" -eq 5 ]; then + echo "::error::No tests matched -m ${{ matrix.marker }}. Either the" \ + "marker was renamed/dropped or selection is broken — this job" \ + "must never pass without running its OS's tests." + exit 1 + fi + exit "$status" + fi + + # Linux full suite. run_tests.sh locates the venv itself (both the + # POSIX bin/ and the Windows Scripts/ layout), so no per-OS + # activation. scripts/run_tests.sh env: - # This is the maximum number of test FILES that run together. - # run_tests_parallel.py starts one pytest subprocess for each file - # from a single ThreadPoolExecutor, so this value IS the limit. The - # default is cpu_count*2, which is 192 here. - # - # Measured on this runner (96-core EPYC 7763, 377GB). Whole suite, - # two repetitions for each value. See run 32549672063: - # - # workers x cores mean - # 48 0.5x 138s - # 96 1.0x 126s <- fastest - # 144 1.5x 132s - # 192 2.0x 132s - # 240 2.5x 140s - # 288 3.0x 142s - # - # One worker for each core wins. The curve is shallow: 126s to 142s - # across a 6x range. The suite has sufficient concurrency at this - # size. The remaining time is the slowest files plus the setup. - # Workers above the core count only add contention. - HERMES_TEST_WORKERS: 96 + # Parallelism. Linux pins 96 (the measured whole-suite sweep — + # one worker per core is fastest on the 96-core runner, and the + # curve is shallow; the per-file runner reads this as its + # subprocess cap). Windows pins 32 (xdist -n). macOS leaves this + # unset (it runs the marked-file lane, not the full suite). + HERMES_TEST_WORKERS: ${{ matrix.workers }} # Ensure tests don't accidentally call real APIs OPENROUTER_API_KEY: "" OPENAI_API_KEY: "" @@ -149,16 +249,7 @@ jobs: - name: Install uv uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0 with: - # Pin the uv version: unpinned, setup-uv resolves "latest" by - # fetching a manifest from raw.githubusercontent.com on EVERY job — - # a transient fetch failure fails the whole job (2026-07-28 slice-5 - # incident). Pinned, the binary downloads directly; no manifest hop. version: "0.9.28" - # Persist uv's download/wheel cache (~/.cache/uv) across runs. - # Keyed on the dependency manifests, so the cache is reused until - # pyproject.toml or uv.lock changes. `uv sync` still runs every - # time, but resolves from the warm cache instead of re-downloading - # and re-building wheels. enable-cache: true cache-dependency-glob: | pyproject.toml @@ -168,22 +259,11 @@ jobs: run: uv python install 3.11 - name: Install dependencies - # `uv sync --locked` installs the exact pinned set from uv.lock (and - # fails if the lock is out of sync with pyproject.toml), giving a - # reproducible env. It also creates .venv itself, so no separate - # `uv venv` step is needed. - # - # Same extras as the test job's sync above: the hermetic test env - # forbids mid-run pip installs (HERMES_DISABLE_LAZY_INSTALLS=1 in - # tests/conftest.py), so lazy-install SDKs exercised by tests must be - # in the venv up front. uses: ./.github/actions/retry with: command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra hindsight --extra parallel-web - name: Minimize uv cache - # Optimized for CI: prunes pre-built wheels that are cheap to - # re-download, keeping the persisted cache small and fast to restore. run: uv cache prune --ci - name: Run e2e tests diff --git a/.gitignore b/.gitignore index f1d51492aa7d..383effcc7dc7 100644 --- a/.gitignore +++ b/.gitignore @@ -1,198 +1,191 @@ -.DS_Store -/venv/ -/venv.old/ -/venv.stale.runtime-*/ -/bin/ -/.hermes-runtime/ -/_pycache/ + +!apps/desktop/src/global.d.ts +!apps/desktop/src/plugins/*/plugin.js +!apps/desktop/src/vite-env.d.ts +!hermes_cli/data/ +!hermes_cli/data/plugin_index.json +# — ignore so `git status` stays clean and update's autostash skips them. +# (launch-time stale-bytecode sweep). Runtime state, never a code change. +# `data/` pattern above would otherwise swallow it. +# `npm run sync-assets` (see web/package.json). +# also created in-repo when an agent operates in this checkout). Plans, audit +# and `hermes update`'s untracked autostash does not treat it as a local edit (#66189 / #54855). +# automation-blueprints-index.json is a build artifact emitted by +# bootstrap installer. It is Hermes-managed runtime state, never a code change — +# Bundled community plugin index seed (shipped as package data) — the bare +# by `hermes update` / launch-time self-heal. Runtime state, never a code change +# by accident via 3a69e34702, removed in the #72002 salvage). +# Checkout fingerprint the __pycache__ tree was last validated against +# CLI config (may contain sensitive SSH paths) +# committed to the repo root. See the hermes-release skill. +# Cross-process web UI build lock (flock target, always empty) +# cut (passed to `gh release create --notes-file`); the GitHub Release itself +# Desktop demo-run scratch output (hermes writes demo/*.txt during recorded +# Desktop/bootstrap install marker written into the managed checkout root by the +# e.g. hermes-bots): plugin.js IS the source, not tsc output. No .tsx +# every build). +# EXCEPT bundled plain-ESM plugin entries (adopted SDK-consumer plugins, +# git for the same reason as skills-index.json (large, generated, change +# ignore it so `hermes update`'s `git stash push --include-untracked` does not +# image-provider (fal.media) URL — they are NEVER committed to the repo. The +# infographic-check CI job is what actually enforces this. +# Installer-written method stamp in the managed checkout root (scripts/install.sh). +# interrupted; consumed by launch-time recovery. Never commit it (was tracked +# Interrupted-update breadcrumb + recovery lock written next to the shared venv +# Local editor / agent tooling (machine-specific; keep in global config, not the repo) +# logs, and per-session caches are never artifacts of the codebase. +# Nix +# No trailing slash: also matches node_modules SYMLINKS (worktrees often +# Per-release changelog drafts. These exist only transiently during a release +# Persistent dev sandbox dir (scripts/dev-sandbox.sh --persistent) +# Playwright visual regression baselines — cached from main in CI, not committed +# PR body is the archive. See the hermes-agent-dev skill's +# PR infographics are rendered locally and embedded in PR descriptions via the +# pr-infographic-workflow reference (storage rule + lapse #8 / #COMMIT-1). +# Private keys +# regenerated by scripts/tool_search_livetest.py. Never an artifact of the repo. +# Release script temp files +# Repo-root build/debug artifacts that must never be committed +# resolves the stale .js OVER the .tsx — never track these) +# Runtime marker written by hermes update when a lazy dependency refresh is +# Runtime metadata only — never a code change. Ignore so `git status` stays clean +# Sandbox dirs used by the install/update E2E (tests/install/). The suffix is +# sibling exists, so the stale-shadow hazard above cannot apply. +# sidestepped by an `infograficos/` directory (#70552). .gitignore is only +# Skills Hub state (lives in ~/.hermes/skills/.hub/ at runtime, but just in case) +# skills.json + skills-meta.json are build artifacts emitted by +# slip into a commit and break `npm ci` on CI with ENOTDIR). +# Spelling variants are listed because a single `infographic/` pattern was +# stores the published notes. They are not a build artifact and must never be +# symlink node_modules to the main checkout; the dir-only pattern let one +# the first line of defence and cannot stop `git add -f` at all — the +# the route name, so each route gets its own tree and two can run at once. +# Tool Search live-test harness output — non-deterministic model transcripts, +# treat it as a local edit and autostash it on every run (#38529). +# tsc-emitted artifacts (a stray `tsc -b` compiles into src/, and vite then +# walkthroughs). Throwaway artifacts, never part of the app. +# Web UI assets — synced from @nous-research/ui at build time via +# Web UI build output +# website/scripts/extract-automation-blueprints.py during prebuild. +# website/scripts/extract-skills.py during prebuild — keep them out of +# Working directory for the Hermes Agent's session state (~/.hermes/ at runtime; +# +%SystemDrive%/ +*.pem +*.ppk *.pyc* -__pycache__/ -act/ +*.tsbuildinfo +*-snapshots/ .act-sandbox-agent.* -.venv/ -.venv -.vscode/ +.bytecode-fingerprint +.bytecode-fingerprint.tmp +.codex/ +.cursor/ +.direnv/ +.DS_Store .env -.op.env -.env.local +.env.development .env.development.local -.env.test.local +.env.local .env.production.local -.env.development .env.test +.env.test.local +.gemini/ +.hermes/ +.hermes-bootstrap-complete .hermes-docker/ -.notebooklm-home/ +.hermes-sandbox/ +.hermes-sandbox-e2e*/ +.lazy-refresh-incomplete +.mcp.json +.nix-stamps/ .notebooklm-cli-venv/ +.notebooklm-home/ .notebooklm-playwright/ +.op.env .pip-cache/ +.pytest_cache/ +.pytest-cache/ +.release_notes.md +.skills_prompt_snapshot.json +.update-incomplete +.update-incomplete.lock .uv-cache/ -compose.hermes.local.yml -export* +.venv +.venv/ +.vscode/ +.web_ui_build.lock +.worktrees/ +.zed/ +/*.png.bak +/.hermes-runtime/ +/.install_method +/_pycache/ +/bin/ +/default.tar.gz +/log.txt +/sqlite_leak_fix.png +/venv.old/ +/venv.stale.runtime-*/ +/venv/ +__pycache__/ __pycache__/model_tools.cpython-310.pyc __pycache__/web_tools.cpython-310.pyc -logs/ -data/ -# Bundled community plugin index seed (shipped as package data) — the bare -# `data/` pattern above would otherwise swallow it. -!hermes_cli/data/ -!hermes_cli/data/plugin_index.json -.pytest_cache/ -test_durations.json -.pytest-cache/ -tmp/ -temp_vision_images/ -hermes-*/* -examples/ -tests/quick_test_dataset.jsonl -tests/sample_dataset.jsonl -run_datagen_kimik2-thinking.sh -run_datagen_megascience_glm4-6.sh -run_datagen_sonnet.sh -source-data/* -run_datagen_megascience_glm4-6.sh -data/* -# No trailing slash: also matches node_modules SYMLINKS (worktrees often -# symlink node_modules to the main checkout; the dir-only pattern let one -# slip into a commit and break `npm ci` on CI with ENOTDIR). -node_modules -browser-use/ +act/ agent-browser/ -# Private keys -*.ppk -*.pem -privvy* -images/ -__pycache__/ -hermes_agent.egg-info/ -wandb/ -testlogs -playwright-report/ -test-results/ -# Playwright visual regression baselines — cached from main in CI, not committed -*-snapshots/ - -# CLI config (may contain sensitive SSH paths) -cli-config.yaml - -# Skills Hub state (lives in ~/.hermes/skills/.hub/ at runtime, but just in case) -skills/.hub/ -ignored/ -.worktrees/ -environments/benchmarks/evals/ - -# Web UI build output -hermes_cli/web_dist/ -# Cross-process web UI build lock (flock target, always empty) -.web_ui_build.lock apps/desktop/build/ +apps/desktop/demo/ apps/desktop/dist/ - -# tsc-emitted artifacts (a stray `tsc -b` compiles into src/, and vite then -# resolves the stale .js OVER the .tsx — never track these) +apps/desktop/release/ +apps/desktop/src/**/*.d.ts apps/desktop/src/**/*.js apps/desktop/src/**/*.js.map -apps/desktop/src/**/*.d.ts -# EXCEPT bundled plain-ESM plugin entries (adopted SDK-consumer plugins, -# e.g. hermes-bots): plugin.js IS the source, not tsc output. No .tsx -# sibling exists, so the stale-shadow hazard above cannot apply. -!apps/desktop/src/plugins/*/plugin.js -!apps/desktop/src/global.d.ts -!apps/desktop/src/vite-env.d.ts - -# Repo-root build/debug artifacts that must never be committed -/log.txt -/sqlite_leak_fix.png -/*.png.bak -/default.tar.gz +apps/shared/src/**/*.d.ts apps/shared/src/**/*.js apps/shared/src/**/*.js.map -apps/shared/src/**/*.d.ts -apps/desktop/release/ -*.tsbuildinfo - -# Web UI assets — synced from @nous-research/ui at build time via -# `npm run sync-assets` (see web/package.json). -web/public/fonts/ -web/public/ds-assets/ - -# Release script temp files -.release_notes.md +browser-use/ +cli-config.yaml +compose.hermes.local.yml +config/mcporter.json +data/ +data/* +docs/superpowers/* +environments/benchmarks/evals/ +examples/ +export* +hermes-*/* +hermes_agent.egg-info/ +hermes_cli/scripts/ +hermes_cli/tui_dist/* +hermes_cli/web_dist/ +ignored/ +images/ +infografico/ +infograficos/ +infographic/ +infographics/ +logs/ mini-swe-agent/ - -# Nix -.direnv/ -.nix-stamps/ -result -website/static/api/skills-index.json -# skills.json + skills-meta.json are build artifacts emitted by -# website/scripts/extract-skills.py during prebuild — keep them out of -# git for the same reason as skills-index.json (large, generated, change -# every build). -website/static/api/skills.json -website/static/api/skills-meta.json -# automation-blueprints-index.json is a build artifact emitted by -# website/scripts/extract-automation-blueprints.py during prebuild. -website/static/api/automation-blueprints-index.json models-dev-upstream/ - -# Local editor / agent tooling (machine-specific; keep in global config, not the repo) -.codex/ -.cursor/ -.gemini/ -.zed/ -.mcp.json +native/fts5_cjk/*.so +node_modules opencode.json -config/mcporter.json - -hermes_cli/tui_dist/* -hermes_cli/scripts/ -docs/superpowers/* -# Working directory for the Hermes Agent's session state (~/.hermes/ at runtime; -# also created in-repo when an agent operates in this checkout). Plans, audit -# logs, and per-session caches are never artifacts of the codebase. -.hermes/ - -# Desktop/bootstrap install marker written into the managed checkout root by the -# bootstrap installer. It is Hermes-managed runtime state, never a code change — -# ignore it so `hermes update`'s `git stash push --include-untracked` does not -# treat it as a local edit and autostash it on every run (#38529). -.hermes-bootstrap-complete - -# Persistent dev sandbox dir (scripts/dev-sandbox.sh --persistent) -.hermes-sandbox/ -# Sandbox dirs used by the install/update E2E (tests/install/). The suffix is -# the route name, so each route gets its own tree and two can run at once. -.hermes-sandbox-e2e*/ - -# Interrupted-update breadcrumb + recovery lock written next to the shared venv -# by `hermes update` / launch-time self-heal. Runtime state, never a code change -# — ignore so `git status` stays clean and update's autostash skips them. -.update-incomplete -.update-incomplete.lock - -# Checkout fingerprint the __pycache__ tree was last validated against -# (launch-time stale-bytecode sweep). Runtime state, never a code change. -.bytecode-fingerprint -.bytecode-fingerprint.tmp - -# Installer-written method stamp in the managed checkout root (scripts/install.sh). -# Runtime metadata only — never a code change. Ignore so `git status` stays clean -# and `hermes update`'s untracked autostash does not treat it as a local edit (#66189 / #54855). -/.install_method - -# Tool Search live-test harness output — non-deterministic model transcripts, -# regenerated by scripts/tool_search_livetest.py. Never an artifact of the repo. +playwright-report/ +privvy* +RELEASE_v*.md +result +run_datagen_kimik2-thinking.sh +run_datagen_megascience_glm4-6.sh +run_datagen_sonnet.sh scripts/out/ # Per-release changelog drafts. These exist only transiently during a release # cut (passed to `gh release create --notes-file`); the GitHub Release itself # stores the published notes. They are not a build artifact and must never be # committed to the repo root. See the hermes-release skill. -RELEASE_v*.md # Desktop demo-run scratch output (hermes writes demo/*.txt during recorded # walkthroughs). Throwaway artifacts, never part of the app. -apps/desktop/demo/ # PR infographics are rendered locally and embedded in PR descriptions via the # image-provider (fal.media) URL — they are NEVER committed to the repo. The @@ -203,16 +196,25 @@ apps/desktop/demo/ # sidestepped by an `infograficos/` directory (#70552). .gitignore is only # the first line of defence and cannot stop `git add -f` at all — the # infographic-check CI job is what actually enforces this. -infographic/ -infographics/ -infograficos/ -infografico/ -native/fts5_cjk/*.so # Runtime marker written by hermes update when a lazy dependency refresh is # interrupted; consumed by launch-time recovery. Never commit it (was tracked # by accident via 3a69e34702, removed in the #72002 salvage). -.lazy-refresh-incomplete -.skills_prompt_snapshot.json # Disposable profile created by scripts/probe_active_session_exclusivity.py .probe-home/ +skills/.hub/ +source-data/* +temp_vision_images/ +test_durations.json +testlogs +test-results/ +tests/quick_test_dataset.jsonl +tests/sample_dataset.jsonl +tmp/ +wandb/ +web/public/ds-assets/ +web/public/fonts/ +website/static/api/automation-blueprints-index.json +website/static/api/skills.json +website/static/api/skills-index.json +website/static/api/skills-meta.json \ No newline at end of file diff --git a/AGENTS.md b/AGENTS.md index e140522e964a..05237d043426 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1579,30 +1579,34 @@ def profile_env(tmp_path, monkeypatch): ### Python **ALWAYS use `scripts/run_tests.sh`** — do not call `pytest` directly. The script enforces hermetic environment parity with CI (unset credential vars, TZ=UTC, LANG=C.UTF-8, -per-file subprocess isolation via `scripts/run_tests_parallel.py` — no xdist, -worker count auto-scaled from CPU count). Direct `pytest` +and a host-dispatched runner: per-file subprocess isolation on POSIX — spawn is +~15ms there, so each file gets a clean interpreter and is collected exactly +once — and pytest-xdist `--dist loadfile` on Windows, where per-file spawn +costs 0.5-1.5s and persistent workers amortize it). Direct `pytest` on a 16+ core developer machine with API keys set diverges from CI in ways that have caused multiple "works locally, fails in CI" incidents (and the reverse). ```bash scripts/run_tests.sh # full suite, CI-parity scripts/run_tests.sh tests/gateway/ # one directory -scripts/run_tests.sh tests/agent/test_foo.py -k test_x # one test (file + -k; the runner is file-granular) +scripts/run_tests.sh tests/agent/test_foo.py -k test_x # one test (file + -k) scripts/run_tests.sh -v --tb=long # pass-through pytest flags ``` -**Flake policy:** the runner auto-retries a failing test FILE once in a fresh -subprocess (`--file-retries`, default 1; `HERMES_TEST_FILE_RETRIES=0` to -disable). Pass-on-retry counts as green but is printed in a `⚠ FLAKY` summary -section with both attempts' output. A FLAKY report is a bug to fix, not noise -to ignore — timing-sensitive tests must not assume a quiet runner (loose -wall-clock bounds ≥ 2s, event-based sync, no `assert not _wait_until(...)` -negative-timing races). +**Flake policy:** on POSIX every file runs in its own subprocess (a failure +is the file's own bug). On Windows, xdist `loadfile` co-schedules files on +workers — an order-dependent failure there is a stateful-test bug to fix, not +noise; fix it at the test, don't reach for runner changes. Timing-sensitive +tests must not assume a quiet runner (loose wall-clock bounds ≥ 2s, +event-based sync, no `assert not _wait_until(...)` negative-timing races). -#### Subprocess-per-test-file isolation +#### Host-dispatched isolation -Every test file runs in a freshly-spawned Python subprocess via `run_tests_parallel.py`. This means module-level dicts/sets and -ContextVars from one test file cannot leak into the next. +On POSIX, each test file runs in a freshly-spawned `python -m pytest ` +subprocess (via `scripts/run_tests_parallel.py`), so module-level state cannot +leak between files at all. On Windows, `--dist loadfile` pins each file to ONE +xdist worker, bounding pollution to co-scheduled files — state shared across +files is a bug to fix at the tests. #### Why the wrapper @@ -1632,11 +1636,24 @@ genuinely differs per host. Those differences are tested by running on the host, not by patching `sys.platform`. ```python -@pytest.mark.linux_only -@pytest.mark.macos_only -@pytest.mark.windows_only +@pytest.mark.platforms("linux") +@pytest.mark.platforms("macos") +@pytest.mark.platforms("windows") ``` +The `platforms` marker takes any number of spec strings (any-of semantics) +plus optional arch filters, so it can express more than the legacy trio: + +```python +@pytest.mark.platforms("not macos") # anywhere except macOS +@pytest.mark.platforms("windows", arch="arm64") # native Windows on arm64 +@pytest.mark.platforms("posix") # linux or macOS +``` + +Specs: `linux`, `macos`, `windows`, `posix`, `any`, and `not `. +The historic `linux_only` / `macos_only` / `windows_only` markers have been +fully replaced — `platforms` is the only host-gating marker in the tree. + Things that are host-independent can stay unmarked: - **Pure functions that take a platform as data** — @@ -1668,13 +1685,15 @@ argv, not the direct parent (the venv shim makes every spawn a launcher/worker chain). **Use the marker, never a bare `skipif`.** `scripts/ci/list_os_marked_tests.py` -decides which files the macOS/Windows lanes import by grepping for the marker -*name*, and the lane then filters with `-m `. A test gated with -`@pytest.mark.skipif(sys.platform != "win32")` therefore skips on Linux AND is -never imported on the Windows lane — it runs on no host at all, silently. The -same trap catches a file-local alias (`windows_only = pytest.mark.skipif(...)`): -the grep matches the name, so the file *is* listed, but `-m windows_only` -deselects every test in it and the lane reports green over zero coverage. +decides which files the macOS lane imports by grepping for the quoted spec +inside `platforms(...)`, and the lane then selects with `-m platforms` while +the conftest's per-test host skips do the actual gating. A test gated with +`@pytest.mark.skipif(sys.platform != "win32")` therefore runs on no host at +all, silently — it is never imported by the lane that would run it, and the +full-suite lanes skip it. Don't stack a module-level `pytestmark = +platforms(...)` on a file whose tests carry their own host marker — the +conftest hard-rejects tests carrying two `platforms()` markers (a test +skipped on every host, reported green everywhere). Equally, don't `pytest.skip()` the non-host rows of a `@parametrize` over platforms — split it into one marked test per OS, or only the host's row ever executes. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 61bacafbd888..03277e041321 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -201,8 +201,8 @@ ln -sf "$(pwd)/venv/bin/hermes" ~/.local/bin/hermes ### Run tests ```bash -# Preferred — matches CI (hermetic `env -i`, per-file subprocess isolation -# via run_tests_parallel.py, worker count auto-scaled); see AGENTS.md +# Preferred — matches CI (hermetic `env -i`; per-file subprocess +# isolation on POSIX, pytest-xdist --dist loadfile on Windows); see AGENTS.md scripts/run_tests.sh # Alternative (activate the venv first). The wrapper is still recommended @@ -836,9 +836,9 @@ that touches the OS, assume *any* platform can hit your code path. Tests that excercise behavior on specific platforms must run on their target platforms. ```python -@pytest.mark.linux_only -@pytest.mark.macos_only -@pytest.mark.windows_only +@pytest.mark.platforms("linux") +@pytest.mark.platforms("macos") +@pytest.mark.platforms("windows") ``` Avoid monkeypatching `sys.platform` unless absolutely needed, but if you do, also patch `platform.system()` / `platform.release()` / `platform.mac_ver()`. Symlinks, 0o600 permissions, SIGALRM, os.setsid/fork are all unix-only. diff --git a/acp_adapter/server.py b/acp_adapter/server.py index 792e44b70d4b..408355a118b8 100644 --- a/acp_adapter/server.py +++ b/acp_adapter/server.py @@ -325,8 +325,15 @@ def _path_from_file_uri(uri: str) -> Path | None: if len(path_text) >= 3 and path_text[0] == "/" and path_text[2] == ":" and path_text[1].isalpha(): drive = path_text[1].lower() rest = path_text[3:].lstrip("/\\").replace("\\", "/") + if os.name == "nt": + # Native Windows: /C:/Users/... is already an absolute path once the + # leading slash is dropped. The /mnt/ form only makes sense + # for Hermes running in WSL. + return Path(path_text[1:]) return Path("/mnt") / drive / rest if len(path_text) >= 2 and path_text[1] == ":" and path_text[0].isalpha(): + if os.name == "nt": + return Path(path_text) drive = path_text[0].lower() rest = path_text[2:].lstrip("/\\").replace("\\", "/") return Path("/mnt") / drive / rest diff --git a/agent/browser_registry.py b/agent/browser_registry.py index 4348237af1df..405e8c762fa2 100644 --- a/agent/browser_registry.py +++ b/agent/browser_registry.py @@ -41,7 +41,7 @@ from typing import Dict, List, Optional from agent.browser_provider import BrowserProvider -from hermes_constants import hermes_home_key +from hermes_constants import hermes_home_key, normalize_scope logger = logging.getLogger(__name__) @@ -69,6 +69,7 @@ def register_provider(provider: BrowserProvider, *, scope: Optional[str] = None) if not isinstance(raw_name, str) or not raw_name.strip(): raise ValueError("Browser provider .name must be a non-empty string") name = raw_name.strip() + scope = normalize_scope(scope) global _generation with _lock: target = _providers if scope is None else _scoped_providers.setdefault(scope, {}) @@ -94,7 +95,7 @@ def list_providers(*, scope: Optional[str] = None) -> List[BrowserProvider]: """Return all registered providers, sorted by name.""" with _lock: merged = dict(_providers) - merged.update(_scoped_providers.get(scope or hermes_home_key(), {})) + merged.update(_scoped_providers.get(hermes_home_key(scope), {})) items = list(merged.values()) return sorted(items, key=lambda p: p.name) @@ -105,12 +106,13 @@ def get_provider(name: str, *, scope: Optional[str] = None) -> Optional[BrowserP return None with _lock: key = name.strip() - return _scoped_providers.get(scope or hermes_home_key(), {}).get(key) or _providers.get(key) + return _scoped_providers.get(hermes_home_key(scope), {}).get(key) or _providers.get(key) def snapshot_registration( name: str, *, scope: Optional[str] = None ) -> Optional[BrowserProvider]: + scope = normalize_scope(scope) with _lock: target = _providers if scope is None else _scoped_providers.get(scope, {}) return target.get(name.strip()) @@ -118,7 +120,7 @@ def snapshot_registration( def registry_generation(*, scope: Optional[str] = None) -> tuple[int, int]: """Return a cache fingerprint for the global base and one profile.""" - active_scope = scope or hermes_home_key() + active_scope = hermes_home_key(scope) with _lock: return _generation, _scoped_generations.get(active_scope, 0) @@ -132,6 +134,7 @@ def restore_registration( ) -> bool: """Restore a plugin registration only when *current* is still installed.""" key = name.strip() + scope = normalize_scope(scope) global _generation with _lock: target = _providers if scope is None else _scoped_providers.setdefault(scope, {}) diff --git a/agent/deadline.py b/agent/deadline.py index e3357f8a240b..6b4a753614c7 100644 --- a/agent/deadline.py +++ b/agent/deadline.py @@ -90,12 +90,22 @@ # Upper bound for any timeout handed to platform wait primitives. # # CPython converts ``threading.Lock.acquire(timeout=...)`` / -# ``Thread.join(timeout=...)`` deadlines to an absolute timestamp; very large -# relative timeouts overflow ``time_t`` on macOS and raise -# ``OverflowError: timestamp out of range for platform time_t`` (#83220). -# One year is semantically "unbounded" for every wait in this codebase while -# staying far below any platform conversion limit. -MAX_SAFE_TIMEOUT_S = 31_536_000.0 # 365 days +# ``Thread.join(timeout=...)`` deadlines through the platform wait primitive; +# a value too large for that primitive's conversion limit raises +# ``OverflowError``. The ceilings differ per host: +# +# * POSIX — ``time_t`` on macOS (32-bit in #83220) overflows on absurd +# relative timeouts. One year is semantically "unbounded" for every wait +# in this codebase while staying far below ``time_t``. +# * Windows — ``WaitForSingleObject`` takes a DWORD of milliseconds +# (2**32 - 1 ms ≈ 49.7 days), so 365 days would overflow it. +if sys.platform == "win32": + # Leave headroom for consumers that add a small margin on top (e.g. the + # 60s human-wait margin): MAX + margin must stay under the DWORD-millisecond + # ceiling (4294967.295s), so 4_294_907 + 60 rounds to exactly 4_294_967s. + MAX_SAFE_TIMEOUT_S = 4_294_907.0 +else: + MAX_SAFE_TIMEOUT_S = 31_536_000.0 # 365 days # Grace period after a deadline fires before concluding the event loop thread # is blocked in a synchronous call and dumping stacks (family A diagnostics). diff --git a/agent/image_gen_registry.py b/agent/image_gen_registry.py index 6239bbe891f9..8ea14d355675 100644 --- a/agent/image_gen_registry.py +++ b/agent/image_gen_registry.py @@ -25,7 +25,7 @@ from typing import Dict, List, Optional from agent.image_gen_provider import ImageGenProvider -from hermes_constants import hermes_home_key +from hermes_constants import hermes_home_key, normalize_scope logger = logging.getLogger(__name__) @@ -52,6 +52,7 @@ def register_provider(provider: ImageGenProvider, *, scope: Optional[str] = None raise ValueError("Image gen provider .name must be a non-empty string") name = raw_name.strip() with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.setdefault(scope, {}) existing = target.get(name) target[name] = provider @@ -65,7 +66,7 @@ def list_providers(*, scope: Optional[str] = None) -> List[ImageGenProvider]: """Return all registered providers, sorted by name.""" with _lock: merged = dict(_providers) - merged.update(_scoped_providers.get(scope or hermes_home_key(), {})) + merged.update(_scoped_providers.get(hermes_home_key(scope), {})) items = list(merged.values()) return sorted(items, key=lambda p: p.name) @@ -76,13 +77,14 @@ def get_provider(name: str, *, scope: Optional[str] = None) -> Optional[ImageGen return None with _lock: key = name.strip() - return _scoped_providers.get(scope or hermes_home_key(), {}).get(key) or _providers.get(key) + return _scoped_providers.get(hermes_home_key(scope), {}).get(key) or _providers.get(key) def snapshot_registration( name: str, *, scope: Optional[str] = None ) -> Optional[ImageGenProvider]: with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.get(scope, {}) return target.get(name.strip()) @@ -97,6 +99,7 @@ def restore_registration( """Restore a plugin registration only when *current* is still installed.""" key = name.strip() with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.setdefault(scope, {}) if target.get(key) is not current: return False diff --git a/agent/image_routing.py b/agent/image_routing.py index 3412efe585de..ec01b433c1bc 100644 --- a/agent/image_routing.py +++ b/agent/image_routing.py @@ -64,10 +64,13 @@ _IMAGE_EXT_PATTERN = "|".join(e.lstrip(".") for e in _IMAGE_EXTS) # Absolute / home-relative local image path. Matches the same shape gateway's -# extract_local_files() uses: anchors to ``~/`` or ``/``, ignores matches inside -# URLs (the ``(? bool: if _in_code(match.start()): continue raw = match.group(0) - expanded = os.path.expanduser(raw) + expanded = os.path.normpath(os.path.expanduser(raw)) try: if not os.path.isfile(expanded): continue diff --git a/agent/secret_sources/registry.py b/agent/secret_sources/registry.py index bbc52089e843..ac6809379f15 100644 --- a/agent/secret_sources/registry.py +++ b/agent/secret_sources/registry.py @@ -44,7 +44,7 @@ reset_source_environment, set_source_environment, ) -from hermes_constants import hermes_home_key +from hermes_constants import hermes_home_key, normalize_scope logger = logging.getLogger(__name__) @@ -138,6 +138,7 @@ def register_source( name, getattr(source, "shape", None), ) return False + scope = normalize_scope(scope) with _REGISTRY_LOCK: effective = dict(_SOURCES) if scope is not None: @@ -169,7 +170,7 @@ def register_source( def get_source(name: str, *, scope: Optional[str] = None) -> Optional[SecretSource]: _ensure_builtin_sources() with _REGISTRY_LOCK: - return _SCOPED_SOURCES.get(scope or hermes_home_key(), {}).get( + return _SCOPED_SOURCES.get(hermes_home_key(scope), {}).get( name ) or _SOURCES.get(name) @@ -179,6 +180,7 @@ def snapshot_registration( ) -> Optional[SecretSource]: """Return the registration owned by exactly one registry layer.""" _ensure_builtin_sources() + scope = normalize_scope(scope) with _REGISTRY_LOCK: target = _SOURCES if scope is None else _SCOPED_SOURCES.get(scope, {}) return target.get(name) @@ -193,6 +195,7 @@ def restore_registration( ) -> bool: """Restore a host-owned source registration if it is still current.""" _ensure_builtin_sources() + scope = normalize_scope(scope) with _REGISTRY_LOCK: target = _SOURCES if scope is None else _SCOPED_SOURCES.setdefault(scope, {}) if target.get(name) is not current: @@ -210,7 +213,7 @@ def list_sources(*, scope: Optional[str] = None) -> List[SecretSource]: _ensure_builtin_sources() with _REGISTRY_LOCK: merged = dict(_SOURCES) - merged.update(_SCOPED_SOURCES.get(scope or hermes_home_key(), {})) + merged.update(_SCOPED_SOURCES.get(hermes_home_key(scope), {})) return list(merged.values()) diff --git a/agent/skill_commands.py b/agent/skill_commands.py index 6b776e1f68fb..64a1fed723b0 100644 --- a/agent/skill_commands.py +++ b/agent/skill_commands.py @@ -381,8 +381,10 @@ def _build_skill_message( if subdir_path.exists(): for f in sorted(subdir_path.rglob("*")): if f.is_file() and not f.is_symlink(): - rel = str(f.relative_to(skill_dir)) - supporting.append(rel) + # as_posix so the listed path matches the footer's + # examples (scripts/foo.js) on every OS — + # str(relative_to) emits backslashes on Windows. + supporting.append(f.relative_to(skill_dir).as_posix()) if supporting and skill_dir: try: diff --git a/agent/terminal_env_registry.py b/agent/terminal_env_registry.py index 3c65cd2242b8..4478bdfc6fd9 100644 --- a/agent/terminal_env_registry.py +++ b/agent/terminal_env_registry.py @@ -30,7 +30,7 @@ from typing import Dict, List, Optional from agent.terminal_env_provider import TerminalEnvironmentProvider -from hermes_constants import hermes_home_key +from hermes_constants import hermes_home_key, normalize_scope logger = logging.getLogger(__name__) @@ -78,6 +78,7 @@ def register_provider( ) global _generation with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.setdefault(scope, {}) existing = target.get(name) target[name] = provider @@ -101,7 +102,7 @@ def list_providers(*, scope: Optional[str] = None) -> List[TerminalEnvironmentPr """Return all registered providers, sorted by name.""" with _lock: merged = dict(_providers) - merged.update(_scoped_providers.get(scope or hermes_home_key(), {})) + merged.update(_scoped_providers.get(hermes_home_key(scope), {})) items = list(merged.values()) return sorted(items, key=lambda p: p.name) @@ -115,7 +116,7 @@ def get_provider( key = name.strip().lower() with _lock: return ( - _scoped_providers.get(scope or hermes_home_key(), {}).get(key) + _scoped_providers.get(hermes_home_key(scope), {}).get(key) or _providers.get(key) ) @@ -173,13 +174,14 @@ def snapshot_registration( name: str, *, scope: Optional[str] = None ) -> Optional[TerminalEnvironmentProvider]: with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.get(scope, {}) return target.get(name.strip().lower()) def registry_generation(*, scope: Optional[str] = None) -> tuple: """Return a cache fingerprint for the global base and one profile.""" - active_scope = scope or hermes_home_key() + active_scope = hermes_home_key(scope) with _lock: return _generation, _scoped_generations.get(active_scope, 0) @@ -195,6 +197,7 @@ def restore_registration( key = name.strip().lower() global _generation with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.setdefault(scope, {}) if target.get(key) is not current: return False diff --git a/agent/transcription_registry.py b/agent/transcription_registry.py index a167e710b55a..feb52f293321 100644 --- a/agent/transcription_registry.py +++ b/agent/transcription_registry.py @@ -24,7 +24,7 @@ from typing import Dict, List, Optional from agent.transcription_provider import TranscriptionProvider -from hermes_constants import hermes_home_key +from hermes_constants import hermes_home_key, normalize_scope logger = logging.getLogger(__name__) @@ -87,6 +87,7 @@ def register_provider(provider: TranscriptionProvider, *, scope: Optional[str] = ) return with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.setdefault(scope, {}) existing = target.get(key) target[key] = provider @@ -106,7 +107,7 @@ def list_providers(*, scope: Optional[str] = None) -> List[TranscriptionProvider """Return all registered providers, sorted by name.""" with _lock: merged = dict(_providers) - merged.update(_scoped_providers.get(scope or hermes_home_key(), {})) + merged.update(_scoped_providers.get(hermes_home_key(scope), {})) items = list(merged.values()) return sorted(items, key=lambda p: p.name) @@ -122,7 +123,7 @@ def get_provider(name: str, *, scope: Optional[str] = None) -> Optional[Transcri return None key = name.strip().lower() with _lock: - return _scoped_providers.get(scope or hermes_home_key(), {}).get(key) or _providers.get(key) + return _scoped_providers.get(hermes_home_key(scope), {}).get(key) or _providers.get(key) def snapshot_registration( @@ -130,6 +131,7 @@ def snapshot_registration( ) -> Optional[TranscriptionProvider]: key = name.strip().lower() with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.get(scope, {}) return target.get(key) @@ -144,6 +146,7 @@ def restore_registration( """Restore a plugin registration only when *current* is still installed.""" key = name.strip().lower() with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.setdefault(scope, {}) if target.get(key) is not current: return False diff --git a/agent/tts_registry.py b/agent/tts_registry.py index 5e8f94b4598d..5c4c45d8c0d9 100644 --- a/agent/tts_registry.py +++ b/agent/tts_registry.py @@ -33,7 +33,7 @@ from typing import Dict, List, Optional from agent.tts_provider import TTSProvider -from hermes_constants import hermes_home_key +from hermes_constants import hermes_home_key, normalize_scope logger = logging.getLogger(__name__) @@ -97,6 +97,7 @@ def register_provider(provider: TTSProvider, *, scope: Optional[str] = None) -> ) return with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.setdefault(scope, {}) existing = target.get(key) target[key] = provider @@ -116,7 +117,7 @@ def list_providers(*, scope: Optional[str] = None) -> List[TTSProvider]: """Return all registered providers, sorted by name.""" with _lock: merged = dict(_providers) - merged.update(_scoped_providers.get(scope or hermes_home_key(), {})) + merged.update(_scoped_providers.get(hermes_home_key(scope), {})) items = list(merged.values()) return sorted(items, key=lambda p: p.name) @@ -132,7 +133,7 @@ def get_provider(name: str, *, scope: Optional[str] = None) -> Optional[TTSProvi return None key = name.strip().lower() with _lock: - return _scoped_providers.get(scope or hermes_home_key(), {}).get(key) or _providers.get(key) + return _scoped_providers.get(hermes_home_key(scope), {}).get(key) or _providers.get(key) def snapshot_registration( @@ -140,6 +141,7 @@ def snapshot_registration( ) -> Optional[TTSProvider]: key = name.strip().lower() with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.get(scope, {}) return target.get(key) @@ -154,6 +156,7 @@ def restore_registration( """Restore a plugin registration only when *current* is still installed.""" key = name.strip().lower() with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.setdefault(scope, {}) if target.get(key) is not current: return False diff --git a/agent/video_gen_registry.py b/agent/video_gen_registry.py index 5b4c9cea5494..975fcd03e649 100644 --- a/agent/video_gen_registry.py +++ b/agent/video_gen_registry.py @@ -29,7 +29,7 @@ from typing import Dict, List, Optional from agent.video_gen_provider import VideoGenProvider -from hermes_constants import hermes_home_key +from hermes_constants import hermes_home_key, normalize_scope logger = logging.getLogger(__name__) @@ -56,6 +56,7 @@ def register_provider(provider: VideoGenProvider, *, scope: Optional[str] = None raise ValueError("Video gen provider .name must be a non-empty string") name = raw_name.strip() with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.setdefault(scope, {}) existing = target.get(name) target[name] = provider @@ -69,7 +70,7 @@ def list_providers(*, scope: Optional[str] = None) -> List[VideoGenProvider]: """Return all registered providers, sorted by name.""" with _lock: merged = dict(_providers) - merged.update(_scoped_providers.get(scope or hermes_home_key(), {})) + merged.update(_scoped_providers.get(hermes_home_key(scope), {})) items = list(merged.values()) return sorted(items, key=lambda p: p.name) @@ -80,13 +81,14 @@ def get_provider(name: str, *, scope: Optional[str] = None) -> Optional[VideoGen return None with _lock: key = name.strip() - return _scoped_providers.get(scope or hermes_home_key(), {}).get(key) or _providers.get(key) + return _scoped_providers.get(hermes_home_key(scope), {}).get(key) or _providers.get(key) def snapshot_registration( name: str, *, scope: Optional[str] = None ) -> Optional[VideoGenProvider]: with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.get(scope, {}) return target.get(name.strip()) @@ -101,6 +103,7 @@ def restore_registration( """Restore a plugin registration only when *current* is still installed.""" key = name.strip() with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.setdefault(scope, {}) if target.get(key) is not current: return False diff --git a/agent/web_search_registry.py b/agent/web_search_registry.py index 5260272133fb..c2b59d1a8e16 100644 --- a/agent/web_search_registry.py +++ b/agent/web_search_registry.py @@ -37,7 +37,7 @@ from typing import Dict, List, Optional from agent.web_search_provider import WebSearchProvider -from hermes_constants import hermes_home_key +from hermes_constants import hermes_home_key, normalize_scope logger = logging.getLogger(__name__) @@ -64,6 +64,7 @@ def register_provider(provider: WebSearchProvider, *, scope: Optional[str] = Non raise ValueError("Web provider .name must be a non-empty string") name = raw_name.strip() with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.setdefault(scope, {}) existing = target.get(name) target[name] = provider @@ -83,7 +84,7 @@ def list_providers(*, scope: Optional[str] = None) -> List[WebSearchProvider]: """Return all registered providers, sorted by name.""" with _lock: merged = dict(_providers) - merged.update(_scoped_providers.get(scope or hermes_home_key(), {})) + merged.update(_scoped_providers.get(hermes_home_key(scope), {})) items = list(merged.values()) return sorted(items, key=lambda p: p.name) @@ -94,13 +95,14 @@ def get_provider(name: str, *, scope: Optional[str] = None) -> Optional[WebSearc return None with _lock: key = name.strip() - return _scoped_providers.get(scope or hermes_home_key(), {}).get(key) or _providers.get(key) + return _scoped_providers.get(hermes_home_key(scope), {}).get(key) or _providers.get(key) def snapshot_registration( name: str, *, scope: Optional[str] = None ) -> Optional[WebSearchProvider]: with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.get(scope, {}) return target.get(name.strip()) @@ -115,6 +117,7 @@ def restore_registration( """Restore a plugin registration only when *current* is still installed.""" key = name.strip() with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.setdefault(scope, {}) if target.get(key) is not current: return False diff --git a/hermes_cli/dashboard_auth/registry.py b/hermes_cli/dashboard_auth/registry.py index 33b58305179d..867e043d0991 100644 --- a/hermes_cli/dashboard_auth/registry.py +++ b/hermes_cli/dashboard_auth/registry.py @@ -10,7 +10,7 @@ import threading from typing import List, Optional -from hermes_constants import hermes_home_key +from hermes_constants import hermes_home_key, normalize_scope from hermes_cli.dashboard_auth.base import ( DashboardAuthProvider, assert_protocol_compliance, @@ -24,7 +24,7 @@ def _merged(scope: Optional[str] = None) -> dict[str, DashboardAuthProvider]: providers = dict(_providers) - providers.update(_scoped_providers.get(scope or hermes_home_key(), {})) + providers.update(_scoped_providers.get(hermes_home_key(scope), {})) return providers @@ -41,6 +41,7 @@ def register_provider( """ assert_protocol_compliance(type(provider)) with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.setdefault(scope, {}) effective = target if scope is None else _merged(scope) if provider.name in effective: @@ -70,6 +71,7 @@ def snapshot_registration( scope: Optional[str] = None, ) -> Optional[DashboardAuthProvider]: with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.get(scope, {}) return target.get(name) @@ -83,6 +85,7 @@ def restore_registration( ) -> bool: """Restore a host-owned provider registration if it is still current.""" with _lock: + scope = normalize_scope(scope) target = _providers if scope is None else _scoped_providers.setdefault(scope, {}) if target.get(name) is not current: return False diff --git a/hermes_cli/doctor.py b/hermes_cli/doctor.py index 1af14d388202..b929be4d1034 100644 --- a/hermes_cli/doctor.py +++ b/hermes_cli/doctor.py @@ -3231,7 +3231,16 @@ def _gh_authenticated() -> bool: capture_output=True, timeout=10, ) return result.returncode == 0 - except (FileNotFoundError, subprocess.TimeoutExpired): + except ( + FileNotFoundError, + subprocess.TimeoutExpired, + PermissionError, + OSError, + ): + # Unspawnable gh (Windows Store/MSIX reparse-point shims raise + # WinError 5; PATH lookups can also land on broken wrappers) is + # the same outcome for this check as no gh at all — a diagnostic + # command must never crash the whole doctor run. return False github_token = get_env_value("GITHUB_TOKEN") or get_env_value("GH_TOKEN") diff --git a/hermes_cli/fs_utils.py b/hermes_cli/fs_utils.py new file mode 100644 index 000000000000..3fc791a2223c --- /dev/null +++ b/hermes_cli/fs_utils.py @@ -0,0 +1,23 @@ +"""Filesystem helpers shared across Hermes CLI subsystems.""" + +import os +import shutil +import stat +from pathlib import Path + + +def rmtree_force(path: Path) -> None: + """``shutil.rmtree`` that clears read-only bits before deleting. + + Windows refuses to delete a tree containing read-only files — git clones + store objects read-only — raising ``PermissionError`` where POSIX unlinks + them fine. The ``onerror`` hook chmods the file writable and retries. + """ + def _onerror(func, p, _exc_info): + try: + os.chmod(p, stat.S_IWRITE) + except OSError: + pass + func(p) + + shutil.rmtree(path, onerror=_onerror) diff --git a/hermes_cli/kanban.py b/hermes_cli/kanban.py index e23eedc7fa98..ffbcfad8caed 100644 --- a/hermes_cli/kanban.py +++ b/hermes_cli/kanban.py @@ -3474,7 +3474,15 @@ def run_slash(rest: str) -> str: import io import contextlib - tokens = shlex.split(rest) if rest and rest.strip() else [] + # Non-posix split (Windows) keeps backslashes as path separators but + # leaves quote characters in the tokens — strip a fully wrapping pair + # so `"my task"` reaches argparse as `my task`, not `"my task"`. + tokens = [] + if rest and rest.strip(): + for tok in shlex.split(rest, posix=os.name == "posix"): + if len(tok) >= 2 and tok[0] == tok[-1] and tok[0] in ("'", '"'): + tok = tok[1:-1] + tokens.append(tok) # Bare ``/kanban`` or ``/kanban help`` / ``--help`` / ``-h`` / ``?``: # show the curated short-help block instead of dumping argparse's full diff --git a/hermes_cli/plugins.py b/hermes_cli/plugins.py index 9028e285307e..1ec926e0b8d6 100644 --- a/hermes_cli/plugins.py +++ b/hermes_cli/plugins.py @@ -3741,8 +3741,9 @@ class PluginManager: def __init__(self, scope_key: Optional[str] = None) -> None: # Capture the home immutably. Unload can run from a different ambient # profile context, but every inverse must target the registration's - # original scope. - self.scope_key = scope_key or hermes_home_key() + # original scope. Normalize through hermes_home_key so the scope + # matches the key the registries store under (normcase on Windows). + self.scope_key = hermes_home_key(scope_key) self.home_path = Path(self.scope_key) self._discovery_lock = threading.RLock() self._plugins: Dict[str, LoadedPlugin] = {} diff --git a/hermes_cli/plugins_cmd.py b/hermes_cli/plugins_cmd.py index 912d93208d21..de9f5b4536c4 100644 --- a/hermes_cli/plugins_cmd.py +++ b/hermes_cli/plugins_cmd.py @@ -28,6 +28,7 @@ from hermes_cli._subprocess_compat import noninteractive_git_env from hermes_cli.config import cfg_get from hermes_cli.secret_prompt import masked_secret_prompt +from hermes_cli.fs_utils import rmtree_force as _rmtree_force from utils import atomic_write_text logger = logging.getLogger(__name__) @@ -867,7 +868,7 @@ def _install_plugin_core( _write_install_metadata(new_metadata) except Exception: if target.exists(): - shutil.rmtree(target) + _rmtree_force(target) if replaced_existing and backup.exists(): os.replace(backup, target) if old_metadata: @@ -1208,7 +1209,7 @@ def _remove_plugin_core(target: Path) -> None: """Remove one plugin and its metadata without splitting their state.""" metadata = _read_install_metadata() if target.name not in metadata: - shutil.rmtree(target) + _rmtree_force(target) return updated = dict(metadata) @@ -1230,7 +1231,7 @@ def _remove_plugin_core(target: Path) -> None: ) from restore_exc shutil.rmtree(staging, ignore_errors=True) raise - shutil.rmtree(staging) + _rmtree_force(staging) def cmd_remove(name: str) -> None: diff --git a/hermes_constants.py b/hermes_constants.py index 4bd9d583b27f..d5c7cb1255a9 100644 --- a/hermes_constants.py +++ b/hermes_constants.py @@ -142,15 +142,36 @@ def get_hermes_home() -> Path: def hermes_home_key(path: str | Path | None = None) -> str: """Return a stable key for a Hermes home/profile directory. + A falsy path (``None`` or empty) resolves to the active default home. Runtime registries use this key to isolate plugin-owned entries while keeping built-in registrations process-global. ``strict=False`` preserves useful behavior for profiles whose directories have not been created yet. """ - candidate = Path(path) if path is not None else get_hermes_home() + candidate = Path(path) if path else get_hermes_home() resolved = candidate.expanduser().resolve(strict=False) return os.path.normcase(str(resolved)) +def normalize_scope(scope: str | Path | None) -> str | None: + """Normalize a WRITE-side registry scope key, preserving ``None``. + + Two different contracts live on the same registries — do not unify them: + + * **Write / slot paths** (``register_*``, ``snapshot_registration``, + ``restore_registration``, tool-registry slot lookup): ``None`` means + the process-global layer and must stay ``None``. Use this function. + * **Read paths** (``list_providers``, ``get_provider``): ``None`` means + "the active home's scope" and must go through :func:`hermes_home_key` + (falsy input resolves to the active default home). Using this + function there hides every scoped registration — the exact bug + fixed after e66a627aa5. + + Both normalize non-None values identically (resolved absolute path, + normcase on Windows) so writes and reads agree on the key. + """ + return hermes_home_key(scope) if scope is not None else None + + def get_process_hermes_home() -> Path: """Return the Hermes home for the running process, ignoring task overrides. diff --git a/hermes_state.py b/hermes_state.py index 831fcbc78dd4..2ed2733bc635 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -15091,10 +15091,12 @@ def retag_kanban_worker_sessions(self, workspaces_root: str) -> int: return 0 def _do(conn): + esc = _escape_like(prefix) cursor = conn.execute( "UPDATE sessions SET source = 'kanban' " - "WHERE source = 'cli' AND (cwd = ? OR cwd LIKE ? ESCAPE '\\')", - (prefix, _escape_like(prefix) + "/%"), + "WHERE source = 'cli' AND (cwd = ? OR cwd LIKE ? ESCAPE '\\' " + "OR cwd LIKE ? ESCAPE '\\')", + (prefix, f"{esc}/%", f"{esc}\\\\%"), ) # Read rowcount before set_meta reuses this cursor for its INSERT, # which would otherwise overwrite it with the meta write's count. diff --git a/optional-skills/creative/comfyui/tests/README.md b/optional-skills/creative/comfyui/tests/README.md index d27fa97e32aa..783735ac4c73 100644 --- a/optional-skills/creative/comfyui/tests/README.md +++ b/optional-skills/creative/comfyui/tests/README.md @@ -45,7 +45,7 @@ When you change a script: The parent hermes-agent repo used to enable `pytest-xdist` by default (`-n auto`); the canonical runner has since moved to per-file subprocess -isolation via `scripts/run_tests_parallel.py` and no longer uses xdist. +pytest-xdist with `--dist loadfile` via `scripts/run_tests.sh`. This suite is small enough that parallelism isn't worth the complexity, and pytest-xdist isn't always installed in the user's environment. The `-c tests/pytest.ini -o addopts="-p no:xdist"` flags make the suite run diff --git a/plugins/disk-cleanup/__init__.py b/plugins/disk-cleanup/__init__.py index 71d44b1c8916..21db4e622526 100644 --- a/plugins/disk-cleanup/__init__.py +++ b/plugins/disk-cleanup/__init__.py @@ -21,6 +21,7 @@ from __future__ import annotations import logging +import os import re import shlex import threading @@ -42,7 +43,9 @@ # Tool-call result shapes we can parse _WRITE_FILE_PATH_KEY = "path" -_TERMINAL_PATH_REGEX = re.compile(r"(?:^|\s)(/[^\s'\"`]+|\~/[^\s'\"`]+)") +_TERMINAL_PATH_REGEX = re.compile( + r"(?:^|\s)(/[^\s'\"`]+|~/[^\s'\"`]+|[A-Za-z]:[\\/][^\s'\"`]+)" +) # --------------------------------------------------------------------------- @@ -107,10 +110,16 @@ def _extract_paths_from_terminal(args: Dict[str, Any], result: str) -> Set[str]: paths: Set[str] = set() cmd = args.get("command") or "" if isinstance(cmd, str) and cmd: - # Tokenise the command — catches `touch /tmp/hermes-x/test_foo.py` + # Tokenise the command — catches `touch /tmp/hermes-x/test_foo.py`. + # ``posix`` follows the host so Windows backslash paths survive + # (``shlex.split(posix=True)`` would eat them as escapes). + # Non-posix mode keeps quote characters in tokens, so strip a fully + # wrapping pair (`"C:\file"` → `C:\file`) before matching. try: - for tok in shlex.split(cmd, posix=True): - if tok.startswith(("/", "~")): + for tok in shlex.split(cmd, posix=os.name == "posix"): + if len(tok) >= 2 and tok[0] == tok[-1] and tok[0] in ("'", '"'): + tok = tok[1:-1] + if tok.startswith(("/", "~")) or re.match(r"^[A-Za-z]:[\\/]", tok): paths.add(tok) except ValueError: pass diff --git a/pyproject.toml b/pyproject.toml index f455c40ed59b..8c21ee1ef2a2 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -196,7 +196,7 @@ modal = ["modal==1.3.4"] daytona = ["daytona==0.155.0"] vercel = ["vercel==0.7.2"] hindsight = ["hindsight-client==0.6.1"] -dev = ["debugpy==1.8.20", "pytest==9.1.1", "pytest-asyncio==1.3.0", "mcp==2.0.0", "httpx2==2.7.0", "starlette==1.3.1", "ty==0.0.21", "ruff==0.15.10", "setuptools==83.0.0"] # starlette: CVE-2026-48710; setuptools: 83 (torch >=2.13 requires setuptools 83) +dev = ["debugpy==1.8.20", "pytest==9.1.1", "pytest-asyncio==1.3.0", "pytest-xdist==3.8.0", "mcp==2.0.0", "httpx2==2.7.0", "starlette==1.3.1", "ty==0.0.21", "ruff==0.15.10", "setuptools==83.0.0"] # starlette: CVE-2026-48710; setuptools: 83 (torch >=2.13 requires setuptools 83) messaging = ["python-telegram-bot[webhooks]==22.8", "discord.py[voice]==2.7.1", "aiohttp==3.14.3", "brotlicffi==1.2.0.1", "slack-bolt==1.30.0", "slack-sdk==3.43.0", "qrcode==7.4.2"] # aiohttp 3.14.3: prior CVEs + GHSA-cq5v-8q36-5273/GHSA-mfx4-hv73-q22v/GHSA-mq44-7p77-q5h7 cron = [] # croniter is now a core dependency; this extra kept for back-compat slack = ["slack-bolt==1.30.0", "slack-sdk==3.43.0", "aiohttp==3.14.3"] @@ -526,6 +526,7 @@ pydantic = false pyjwt = false pytest = false pytest-asyncio = false +pytest-xdist = false python-dotenv = false python-multipart = false python-olm = false @@ -608,9 +609,7 @@ markers = [ "requires_wal: needs the runtime to actually enable SQLite WAL mode (skipped where Hermes falls back to journal_mode=DELETE)", "no_isolate: opt out of per-file subprocess isolation (tests share mutable module-level state)", "ssh: marks tests requiring a reachable SSH server (skipped in normal CI)", - "linux_only: exercises Linux-specific behaviour; skipped on other hosts", - "macos_only: exercises macOS-specific behaviour; skipped on other hosts", - "windows_only: exercises native-Windows behaviour; skipped on other hosts", + "platforms(*specs, arch=None, arch_negate=False): run only on hosts matching at least one spec — linux/macos/windows/posix/any, 'not X' negation, optional arch filter", ] # integration tests take way too long to run in the normal CI environments addopts = "-m 'not integration'" diff --git a/scripts/check_subprocess_stdin.py b/scripts/check_subprocess_stdin.py index 884923e2e04c..8502e13f1821 100644 --- a/scripts/check_subprocess_stdin.py +++ b/scripts/check_subprocess_stdin.py @@ -172,8 +172,9 @@ def main() -> int: for py_file in dirpath.rglob("*.py"): rel = str(py_file.relative_to(repo_root)) - # Skip known-safe files. - if rel in KNOWN_SAFE: + # Skip known-safe files. ``relative_to`` returns a host-separated + # path (backslashes on Windows) while KNOWN_SAFE is forward-slash. + if py_file.relative_to(repo_root).as_posix() in KNOWN_SAFE: continue # Skip test files inside tools/ etc. diff --git a/scripts/ci/classify_changes.py b/scripts/ci/classify_changes.py index 935703c870ed..454db7531720 100644 --- a/scripts/ci/classify_changes.py +++ b/scripts/ci/classify_changes.py @@ -144,7 +144,7 @@ def _py_test_only(p: str) -> bool: Product jobs (Desktop E2E's ``hermes serve`` backend, the Docker image) run installed code — nothing under ``tests/`` is packaged or importable - there. scripts/run_tests.sh and run_tests_parallel.py are deliberately + there. scripts/run_tests.sh and scripts/run_tests_parallel.py are deliberately NOT test-only: they are runner infrastructure, and a bad edit there can mask real failures, so they stay conservative (python_prod=true). """ diff --git a/scripts/ci/list_os_marked_tests.py b/scripts/ci/list_os_marked_tests.py index bfbc0a60c1d6..1e65c4d4e3dd 100644 --- a/scripts/ci/list_os_marked_tests.py +++ b/scripts/ci/list_os_marked_tests.py @@ -1,27 +1,34 @@ #!/usr/bin/env python3 -"""List the test files that carry a given OS marker. +"""List the test files that carry a platforms() gate for a given platform. -Used by ``.github/workflows/tests-os.yml`` to scope what the macOS and -Windows lanes import. +Used by the marked-OS lane of ``.github/workflows/tests.yml`` to scope what +the macOS lane imports. -Why scope at all, when ``pytest -m macos_only`` already selects correctly? +Why scope at all, when ``pytest -m platforms`` already selects correctly? Because ``-m`` filters AFTER collection, and collection IMPORTS every test module under ``tests/``. On the Linux lane that is fine (it runs them all -anyway), but on the macOS/Windows lanes it would drag ~900 unrelated modules -through import on a host they were never expected to import on — one -unrelated ImportError would fail a job whose actual subject passed. Narrowing -the paths keeps each lane's failure signal about its own tests. - -``-m`` is still passed by the workflow and remains the authoritative -selector: this script only decides which files get imported, never which -tests run. Over-selecting here is harmless (``-m`` drops the extras); the -failure mode to care about is UNDER-selecting, which is why the workflow -fails the job when zero tests end up selected. +anyway), but on the macOS lane it would drag ~900 unrelated modules through +import on a host they were never expected to import on — one unrelated +ImportError would fail a job whose actual subject passed. Narrowing the +paths keeps each lane's failure signal about its own tests. + +``-m platforms`` (plus the conftest's per-test host skips) remains the +authoritative selector: this script only decides which files get imported, +never which tests run. Over-selecting here is harmless (the skips drop the +extras); the failure mode to care about is UNDER-selecting, which is why +the workflow fails the job when zero tests end up selected. + +A file matches when a quoted ``platforms("...")`` spec names the platform — +including negated (``"not macos"``) and any-of lists. The match is anchored +inside the string literal so bare identifiers (a variable named ``windows``) +don't produce false positives. Usage: - python scripts/ci/list_os_marked_tests.py macos_only [tests_root] + python scripts/ci/list_os_marked_tests.py macos [tests_root] -Prints one path per line (POSIX separators, repo-relative), sorted. +Prints one path per line (POSIX separators, repo-relative), sorted. Exits +non-zero when no file matches (a renamed spec or broken selection would +otherwise report a green lane that ran nothing). """ from __future__ import annotations @@ -30,18 +37,14 @@ import sys from pathlib import Path -_VALID_MARKERS = ("linux_only", "macos_only", "windows_only") - +_VALID_PLATFORMS = ("linux", "macos", "windows") -def find_marked_files(marker: str, root: Path) -> list[Path]: - """Return every ``test_*.py`` under *root* that references *marker*. - Matches the marker as a whole word so ``macos_only`` doesn't pick up a - hypothetical ``macos_only_extra``. Catches both the decorator form - (``@pytest.mark.macos_only``, on a function or a class) and the - module-level ``pytestmark`` form. - """ - pattern = re.compile(rf"\b{re.escape(marker)}\b") +def find_marked_files(platform: str, root: Path) -> list[Path]: + """Return every ``test_*.py`` under *root* gating on *platform*.""" + pattern = re.compile( + rf'platforms\(\s*[^)]*?"[^")]*\b{re.escape(platform)}\b[^")]*"' + ) hits: list[Path] = [] for path in sorted(root.rglob("test_*.py")): try: @@ -57,59 +60,29 @@ def main(argv: list[str]) -> int: if len(argv) < 2: print(__doc__, file=sys.stderr) return 2 - marker = argv[1] - if marker not in _VALID_MARKERS: + platform = argv[1] + if platform not in _VALID_PLATFORMS: print( - f"error: unknown marker {marker!r} (expected one of " - f"{', '.join(_VALID_MARKERS)})", + f"unknown platform {platform!r}; valid: {', '.join(_VALID_PLATFORMS)}", file=sys.stderr, ) return 2 - - repo_root = Path(__file__).resolve().parents[2] - root = Path(argv[2]) if len(argv) > 2 else repo_root / "tests" - if not root.exists(): - print(f"error: no such directory: {root}", file=sys.stderr) + tests_root = Path(argv[2]) if len(argv) > 2 else Path("tests") + if not tests_root.is_dir(): + print(f"no such directory: {tests_root}", file=sys.stderr) return 2 - - files = find_marked_files(marker, root) - if not files: + hits = find_marked_files(platform, tests_root) + for path in hits: + print(path.as_posix()) + if not hits: print( - f"error: no test file references @pytest.mark.{marker} — the marker " - "was probably renamed or dropped. Refusing to emit an empty list, " - "which would let the OS lane pass without running anything.", + f"no test files gate on {platform!r} under {tests_root} — " + "either the spec vocabulary changed or selection is broken", file=sys.stderr, ) return 1 - - lines: list[str] = [] - for path in files: - # POSIX separators so the output is safe to paste into a bash - # command line on the Windows runner (Git Bash accepts them). - # - # Relative to the repo root when the path is inside it (the CI case — - # pytest is invoked from the repo root). A root outside the repo is a - # test/manual invocation; emit it as-is rather than raising, since - # ``relative_to`` refuses non-descendant paths. - try: - rel = path.resolve().relative_to(repo_root) - except ValueError: - lines.append(path.as_posix()) - else: - lines.append(rel.as_posix()) - - # Write bytes with explicit LF rather than print(), which on Windows - # translates "\n" to "\r\n" in text mode. The consumer reads this list with - # ``$(cat ...)`` in bash, and word splitting uses IFS (space/tab/newline) — - # a CR is NOT a separator, so it stays glued to each path and pytest then - # fails with "file or directory not found: tests/...py" for a path that - # looks correct in the log because the CR is invisible. Emitting bytes makes - # the output identical on every host instead of depending on the platform's - # newline translation. - sys.stdout.buffer.write(b"".join(line.encode("utf-8") + b"\n" for line in lines)) - sys.stdout.buffer.flush() return 0 if __name__ == "__main__": - sys.exit(main(sys.argv)) + raise SystemExit(main(sys.argv)) diff --git a/scripts/run_tests.sh b/scripts/run_tests.sh index 4445d6d5428b..3a529390b7ef 100755 --- a/scripts/run_tests.sh +++ b/scripts/run_tests.sh @@ -2,34 +2,35 @@ # Canonical test runner for hermes-agent. Run this instead of calling # `pytest` directly to guarantee your local run matches CI behavior. # -# What this script enforces: -# * Per-file isolation via scripts/run_tests_parallel.py — each test -# file runs in its own freshly-spawned `python -m pytest ` -# subprocess. No xdist, no shared workers, no module-level leakage -# between files. -# * TZ=UTC, LANG=C.UTF-8, PYTHONHASHSEED=0 (deterministic) -# * Env vars blanked (conftest.py also does this, but this -# is belt-and-suspenders for anyone running pytest outside our -# conftest path — e.g. on a single file) -# * Proper venv activation (probes .venv, venv, then ~/.hermes/...) +# The runner dispatches on host, because the two cost profiles are opposites: +# +# * POSIX — per-file subprocess isolation (scripts/run_tests_parallel.py): +# each test FILE runs in its own freshly-spawned `python -m pytest ` +# process. The spawn floor is ~15ms there, so process isolation is nearly +# free; in exchange there is no cross-file state pollution and each file +# is collected exactly once (pytest's per-item fixture-closure machinery — +# tens of millions of dict walks over ~42k items against the conftest's +# autouse fixtures — is paid once, not once per xdist worker; measured +# 37-65s of pure collection that a persistent-worker model multiplies by +# the worker count). +# * Windows — pytest-xdist with --dist loadfile. The per-file model pays a +# 0.5-1.5s spawn+import wall per file (~3400 files ≈ a 6-minute floor that +# dominated the lane); persistent workers pay the interpreter+import wall +# once per worker. loadfile pins each file's tests to ONE worker, so the +# remaining hazard is state shared by files co-scheduled on a worker — +# which is a stateful-test bug to fix, not a runner bug. +# +# Both paths enforce the same hermetic environment: TZ=UTC, LANG=C.UTF-8, +# PYTHONHASHSEED=0, `env -i` scrubbing (credential vars can't leak), and +# proper venv activation (probes .venv, venv, then ~/.hermes/...). # # Usage: # scripts/run_tests.sh # full suite -# scripts/run_tests.sh -j 4 # cap parallelism +# scripts/run_tests.sh -j 4 # cap workers/parallelism # scripts/run_tests.sh tests/agent/ # discover only here -# scripts/run_tests.sh tests/agent/ tests/acp/ # multiple roots # scripts/run_tests.sh tests/foo.py # single file # scripts/run_tests.sh tests/foo.py -q # path + bare pytest flag -# scripts/run_tests.sh tests/foo.py -v --tb=long # bare flags "just work" # scripts/run_tests.sh -k 'pattern' # value flags pass through too -# scripts/run_tests.sh tests/foo.py -- --tb=long # explicit '--' still works -# -# Bare pytest flags (anything starting with '-' that isn't one of this -# runner's own options: -j/--jobs, --paths, --slice, --file-timeout, etc.) -# are forwarded to each per-file pytest invocation automatically — no '--' -# separator required. The explicit '--' form still works and stacks with -# bare flags. Positional path arguments override the default discovery -# root (tests/). set -euo pipefail @@ -37,6 +38,14 @@ set -euo pipefail SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)" +# ── Host model ─────────────────────────────────────────────────────────────── +# Git-bash / MSYS on Windows reports uname -s like MINGW64_NT-10.0-... or +# MSYS_NT-...; POSIX hosts report Linux / Darwin. +case "$(uname -s)" in + Linux|Darwin) IS_WINDOWS=0 ;; + *) IS_WINDOWS=1 ;; +esac + # ── Locate python ─────────────────────────────────────────────────────────── # Probe local venvs first; fall back to the Nix devShell's editable venv # (HERMES_PYTHON is exported by the devShell hook and ships [dev] extras: @@ -59,12 +68,7 @@ for candidate in "$REPO_ROOT/.venv" "$REPO_ROOT/venv" "$HOME/.hermes/hermes-agen break fi SKIPPED_VENVS="$SKIPPED_VENVS $candidate" - fi - # Native Windows venv layout: python.exe and activate live under - # Scripts/, and there is no bin/. Anyone running this script from - # Git Bash / MSYS with a `python -m venv`- or uv-created venv hits - # this branch — without it the canonical runner refuses to start. - if [ -f "$candidate/Scripts/activate" ]; then + elif [ -f "$candidate/Scripts/activate" ]; then if "$candidate/Scripts/python.exe" -c 'import pytest' 2>/dev/null; then VENV="$candidate" VENV_PYTHON="$candidate/Scripts/python.exe" @@ -73,40 +77,19 @@ for candidate in "$REPO_ROOT/.venv" "$REPO_ROOT/venv" "$HOME/.hermes/hermes-agen SKIPPED_VENVS="$SKIPPED_VENVS $candidate" fi done - -if [ -n "$SKIPPED_VENVS" ]; then - for skipped in $SKIPPED_VENVS; do - echo "▶ skipping venv without pytest: $skipped" >&2 - done -fi - -if [ -n "$VENV" ]; then - PYTHON="$VENV_PYTHON" -elif [ -n "${HERMES_PYTHON:-}" ] && [ -x "$HERMES_PYTHON" ] \ - && "$HERMES_PYTHON" -c 'import pytest' 2>/dev/null; then - # Guard with an import check: HERMES_PYTHON may point at the RELEASE - # venv (no pytest) when inherited from a wrapped `hermes` binary rather - # than the devShell hook. - PYTHON="$HERMES_PYTHON" - echo "▶ no local venv — using Nix dev venv via HERMES_PYTHON: $PYTHON" -else - echo "error: no virtualenv with pytest found in $REPO_ROOT/.venv or $REPO_ROOT/venv," >&2 - echo " and HERMES_PYTHON is not a python with pytest (enter the Nix devShell or create a venv)" >&2 - if [ -n "$SKIPPED_VENVS" ]; then - echo " (skipped for missing pytest:$SKIPPED_VENVS — install dev extras there, or create $REPO_ROOT/.venv)" >&2 +if [ -z "$VENV_PYTHON" ]; then + if [ -n "${HERMES_PYTHON:-}" ] && "${HERMES_PYTHON}" -c 'import pytest' 2>/dev/null; then + VENV_PYTHON="$HERMES_PYTHON" + else + echo "✗ No venv with pytest found. Install dev extras:" >&2 + echo " uv sync --extra dev" >&2 + if [ -n "$SKIPPED_VENVS" ]; then + echo " (skipped for missing pytest:$SKIPPED_VENVS — install dev extras there, or create $REPO_ROOT/.venv)" >&2 + fi + exit 1 fi - exit 1 fi - - -# ── Live-gateway plugin (computed before we drop env) ─────────────────────── -EXTRA_PYTHONPATH="" -EXTRA_PYTEST_PLUGINS="" -if [ -f "$HOME/.hermes/pytest_live_guard.py" ]; then - EXTRA_PYTHONPATH="$HOME/.hermes" - EXTRA_PYTEST_PLUGINS="pytest_live_guard" -fi - +PYTHON="$VENV_PYTHON" # ── Windows location variables (computed before we drop env) ─────────────── # `env -i` forwards HOME, which is enough on POSIX. Native Windows CPython @@ -123,19 +106,43 @@ for _win_var in USERPROFILE HOMEDRIVE HOMEPATH LOCALAPPDATA APPDATA SYSTEMROOT T fi done -# ── Test-runner knobs (computed before we drop env) ──────────────────────── -# The runner's own documented environment knobs must survive the hermetic -# `env -i` below, or they are silent no-ops for anyone invoking this script: -# -# * HERMES_TEST_WORKERS / PATHS / FILE_TIMEOUT / FILE_RETRIES / SLICE are -# read by run_tests_parallel.py at argparse-default time — inside the -# stripped environment. +# ── Live-gateway plugin (computed before we drop env) ─────────────────────── +EXTRA_PYTHONPATH="" +EXTRA_PYTEST_PLUGINS="" +if [ -f "$HOME/.hermes/pytest_live_guard.py" ]; then + EXTRA_PYTHONPATH="$HOME/.hermes" + EXTRA_PYTEST_PLUGINS="pytest_live_guard" +fi + +# ── Our -j/--jobs flag: consumed here, forwarded via HERMES_TEST_WORKERS ──── +# (both backends read that env knob: run_tests_parallel.py as its worker cap, +# the xdist path as -n). +JOBS="${HERMES_TEST_WORKERS:-}" +PASS_THROUGH=() +while [ $# -gt 0 ]; do + case "$1" in + -j|--jobs) + JOBS="$2"; shift 2 ;; + -j*) + JOBS="${1#-j}"; shift ;; + --jobs=*) + JOBS="${1#--jobs=}"; shift ;; + *) + PASS_THROUGH+=("$1"); shift ;; + esac +done +set -- ${PASS_THROUGH[@]+"${PASS_THROUGH[@]}"} +if [ -n "$JOBS" ]; then + export HERMES_TEST_WORKERS="$JOBS" + TEST_ENV_KNOB="HERMES_TEST_WORKERS" +fi + +# ── Test-runner knobs (computed before we drop env) ────────────────────────── # * HERMES_TEST_IMAGE is read by tests/docker/conftest.py to skip its -# session-scoped `docker build`. CI's docker.yml sets it to the image -# the build step just loaded; stripping it made every per-file pytest -# subprocess rebuild the 5GB image from a cold builder cache instead -# (~4 min per worker per run, and the rebuilt image lacked the -# HERMES_GIT_SHA build-arg the workflow bakes in). +# session-scoped `docker build`. +# * POSIX per-file path: HERMES_TEST_WORKERS / PATHS / FILE_TIMEOUT / +# FILE_RETRIES / SLICE are read by run_tests_parallel.py at argparse- +# default time — inside the stripped environment. # # These are test-infrastructure knobs, not credentials — same class as the # HERMES_RUN_SLOW_PET_TESTS / HERMES_E2E_BROWSER opt-ins already forwarded. @@ -152,32 +159,35 @@ done # ── Run in hermetic env ────────────────────────────────────────────────────── # env -i: start with empty environment, opt-in only what we need. # No credential var can leak — you'd have to explicitly add it here. -echo "▶ running per-file parallel test suite via run_tests_parallel.py" -echo " (TZ=UTC LANG=C.UTF-8 PYTHONHASHSEED=0; clean env)" - cd "$REPO_ROOT" -# ── Pre-compile .pyc bytecode cache ───────────────────────────────────────── -# Each test file runs in its own subprocess via run_tests_parallel.py. -# Pre-building the bytecode cache once here (instead of each subprocess -# compiling on first import) avoids redundant work across ~2000 processes. -# Uses git to list tracked .py files (skips venv, node_modules, etc). echo "▶ pre-compiling bytecode cache" "$PYTHON" -m compileall -q -j 0 -- $(git ls-files '*.py') >/dev/null 2>&1 || true -echo "▶ launching test runner" -exec env -i \ - PATH="$PATH" \ - HOME="$HOME" \ - ${WIN_ENV[@]+"${WIN_ENV[@]}"} \ - ${TEST_ENV[@]+"${TEST_ENV[@]}"} \ - TZ=UTC \ - LANG=C.UTF-8 \ - LC_ALL=C.UTF-8 \ - PYTHONHASHSEED=0 \ - PYTHONUTF8=1 \ - ${HERMES_RUN_SLOW_PET_TESTS:+HERMES_RUN_SLOW_PET_TESTS="$HERMES_RUN_SLOW_PET_TESTS"} \ - ${HERMES_E2E_BROWSER:+HERMES_E2E_BROWSER="$HERMES_E2E_BROWSER"} \ - ${EXTRA_PYTHONPATH:+PYTHONPATH="$EXTRA_PYTHONPATH"} \ - ${EXTRA_PYTEST_PLUGINS:+PYTEST_PLUGINS="$EXTRA_PYTEST_PLUGINS"} \ +HERMETIC_ENV=( + PATH="$PATH" + HOME="$HOME" + ${WIN_ENV[@]+"${WIN_ENV[@]}"} + ${TEST_ENV[@]+"${TEST_ENV[@]}"} + TZ=UTC + LANG=C.UTF-8 + LC_ALL=C.UTF-8 + PYTHONHASHSEED=0 + PYTHONUTF8=1 + ${HERMES_RUN_SLOW_PET_TESTS:+HERMES_RUN_SLOW_PET_TESTS="$HERMES_RUN_SLOW_PET_TESTS"} + ${HERMES_E2E_BROWSER:+HERMES_E2E_BROWSER="$HERMES_E2E_BROWSER"} + ${EXTRA_PYTHONPATH:+PYTHONPATH="$EXTRA_PYTHONPATH"} + ${EXTRA_PYTEST_PLUGINS:+PYTEST_PLUGINS="$EXTRA_PYTEST_PLUGINS"} +) + +if [ "$IS_WINDOWS" -eq 1 ]; then + echo "▶ windows: pytest-xdist (-n ${HERMES_TEST_WORKERS:-auto} --dist loadfile)" + exec env -i "${HERMETIC_ENV[@]}" \ + "$PYTHON" -m pytest -n "${HERMES_TEST_WORKERS:-auto}" --dist loadfile \ + -p no:cacheprovider -m "not integration" -q --tb=line "$@" +fi + +echo "▶ posix: per-file parallel suite via run_tests_parallel.py" +echo " (TZ=UTC LANG=C.UTF-8 PYTHONHASHSEED=0; clean env)" +exec env -i "${HERMETIC_ENV[@]}" \ "$PYTHON" "$SCRIPT_DIR/run_tests_parallel.py" "$@" diff --git a/scripts/run_tests_parallel.py b/scripts/run_tests_parallel.py old mode 100755 new mode 100644 index eb923cc9b43b..ea989635d91c --- a/scripts/run_tests_parallel.py +++ b/scripts/run_tests_parallel.py @@ -146,8 +146,8 @@ def _split_pathspec(value: str) -> List[str]: # behaviour, and names the CI lane where those tests actually execute. _OS_MARKERS = { "linux_only": ("linux", "the main Linux CI lane"), - "macos_only": ("darwin", "the tests-os CI lane (macos-latest)"), - "windows_only": ("win32", "the tests-os CI lane (windows-latest)"), + "macos_only": ("darwin", "the macOS Python-tests lane"), + "windows_only": ("win32", "the Windows Python-tests lane"), } @@ -781,8 +781,8 @@ def main() -> int: "-j", "--jobs", type=int, - default=int(os.environ.get("HERMES_TEST_WORKERS") or (os.cpu_count() or 4) * 2), - help="Parallel worker count (default: $HERMES_TEST_WORKERS or cpu_count*2)", + default=int(os.environ.get("HERMES_TEST_WORKERS") or (os.cpu_count() or 4)), + help="Parallel worker count (default: $HERMES_TEST_WORKERS or cpu_count)", ) parser.add_argument( "--paths", diff --git a/skills/autonomous-ai-agents/hermes-agent/references/contributor-guide.md b/skills/autonomous-ai-agents/hermes-agent/references/contributor-guide.md index a578f06d25c1..3db9ed055a79 100644 --- a/skills/autonomous-ai-agents/hermes-agent/references/contributor-guide.md +++ b/skills/autonomous-ai-agents/hermes-agent/references/contributor-guide.md @@ -86,7 +86,7 @@ run_conversation(): Use the canonical runner — it enforces CI-parity (hermetic `env -i`, unset credentials, TZ=UTC, per-file subprocess isolation via -`scripts/run_tests_parallel.py` — no xdist, worker count auto-scaled): +`scripts/run_tests.sh` — pytest-xdist, `--dist loadfile`, worker count = CPU count): ```bash scripts/run_tests.sh # full suite diff --git a/skills/software-development/python-debugpy/SKILL.md b/skills/software-development/python-debugpy/SKILL.md index c94dfdf52699..b5279161fbd8 100644 --- a/skills/software-development/python-debugpy/SKILL.md +++ b/skills/software-development/python-debugpy/SKILL.md @@ -107,7 +107,7 @@ scripts/run_tests.sh tests/path/to/test_file.py::test_name --trace scripts/run_tests.sh tests/path/to/test_file.py --showlocals --tb=long ``` -Note: `scripts/run_tests.sh` runs each test file in a captured subprocess via `run_tests_parallel.py` (no xdist), so interactive pdb does NOT work under the wrapper. Run pytest directly for `--pdb`: +Note: `scripts/run_tests.sh` runs pytest under xdist workers, so interactive pdb does NOT work under the wrapper. Run pytest directly for `--pdb`: ```bash source .venv/bin/activate diff --git a/tests/acp/test_ping_suppression.py b/tests/acp/test_ping_suppression.py index 1682c6ad5446..38a67aae7900 100644 --- a/tests/acp/test_ping_suppression.py +++ b/tests/acp/test_ping_suppression.py @@ -99,6 +99,7 @@ def on_connect(self, conn): # noqa: ANN001 pass +@pytest.mark.platforms("linux") @pytest.mark.asyncio async def test_bare_ping_request_produces_proper_response_and_no_stderr_noise( caplog: pytest.LogCaptureFixture, diff --git a/tests/acp/test_session.py b/tests/acp/test_session.py index 7922846725d6..4d3a1a202124 100644 --- a/tests/acp/test_session.py +++ b/tests/acp/test_session.py @@ -164,6 +164,7 @@ class TestSymlinkAliasNormalization: ``/private/tmp``) must compare equal, or ACP history filters silently drop a workspace's own sessions.""" + @pytest.mark.require_symlinks def test_symlink_alias_compares_equal(self, tmp_path): real = tmp_path / "real" real.mkdir() @@ -188,8 +189,9 @@ def test_missing_path_keeps_lexical_normalization(self): # exactly as the old normpath comparison did. assert acp_session._normalize_cwd_for_compare( "/nonexistent-hermes-test/x/../y" - ) == "/nonexistent-hermes-test/y" + ) == acp_session._normalize_cwd_for_compare("/nonexistent-hermes-test/y") + @pytest.mark.require_symlinks def test_list_sessions_matches_symlink_alias_cwd(self, manager, tmp_path): real = tmp_path / "proj" real.mkdir() diff --git a/tests/acp_adapter/test_acp_images.py b/tests/acp_adapter/test_acp_images.py index 3ef1f48fe2e5..296701c2a040 100644 --- a/tests/acp_adapter/test_acp_images.py +++ b/tests/acp_adapter/test_acp_images.py @@ -38,7 +38,7 @@ def test_text_only_acp_blocks_stay_string_for_legacy_prompt_path(): def test_acp_resource_link_file_is_inlined_as_text(tmp_path): attached = tmp_path / "notes.md" - attached.write_text("# Notes\n\nAttached file body", encoding="utf-8") + attached.write_text("# Notes\n\nAttached file body", encoding="utf-8", newline="\n") content = _content_blocks_to_openai_user_content([ TextContentBlock(type="text", text="Please read this file"), diff --git a/tests/agent/lsp/test_install_and_lint_fixes.py b/tests/agent/lsp/test_install_and_lint_fixes.py index b614cee50f3e..314e96750453 100644 --- a/tests/agent/lsp/test_install_and_lint_fixes.py +++ b/tests/agent/lsp/test_install_and_lint_fixes.py @@ -45,7 +45,7 @@ def fake_run(cmd, **kwargs): from agent.lsp import install as install_mod monkeypatch.setattr(install_mod.subprocess, "run", fake_run) - monkeypatch.setattr(install_mod.shutil, "which", lambda c: "/usr/bin/npm" if c == "npm" else None) + monkeypatch.setattr(install_mod, "find_node_executable", lambda name: "/usr/bin/npm") install_mod._install_npm("pyright", "pyright-langserver") @@ -61,11 +61,11 @@ def fake_run(cmd, **kwargs): -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_install_pip_finds_windows_scripts_launcher(tmp_path, monkeypatch): """pip console scripts can land in Scripts/ on native Windows. - ``windows_only``: the ``Scripts/`` layout and the ``.exe`` launcher are + ``platforms("windows")``: the ``Scripts/`` layout and the ``.exe`` launcher are what pip actually produces on Windows. Faking ``_is_windows()`` on Linux made the test assert against a directory tree the test itself created, on a host where pip would never lay it out that way. diff --git a/tests/agent/lsp/test_workspace.py b/tests/agent/lsp/test_workspace.py index 8d96f902d682..9e075b9d0063 100644 --- a/tests/agent/lsp/test_workspace.py +++ b/tests/agent/lsp/test_workspace.py @@ -69,6 +69,7 @@ def test_resolve_workspace_for_file_uses_cwd_first(tmp_path: Path, monkeypatch): +@pytest.mark.platforms("linux") def test_normalize_path_expands_tilde(monkeypatch): monkeypatch.setenv("HOME", "/home/user") p = normalize_path("~/x.py") diff --git a/tests/agent/test_anthropic_keychain.py b/tests/agent/test_anthropic_keychain.py index 7f1174967c27..7673be47ffd5 100644 --- a/tests/agent/test_anthropic_keychain.py +++ b/tests/agent/test_anthropic_keychain.py @@ -19,11 +19,11 @@ pytestmark = pytest.mark.allow_macos_keychain -@pytest.mark.macos_only +@pytest.mark.platforms("macos") class TestReadClaudeCodeCredentialsFromKeychain: """Bug 4: macOS Keychain support for Claude Code >=2.1.114. - ``macos_only``: the reader is gated on ``platform.system() == "Darwin"`` + ``platforms("macos")``: the reader is gated on ``platform.system() == "Darwin"`` and shells out to the ``security`` CLI. Faking Darwin on Linux selected the branch but proved nothing about the host it exists for; on the real macOS runner only ``subprocess.run`` is mocked (via the @@ -51,7 +51,7 @@ def test_returns_none_on_nonzero_exit_code(self): -@pytest.mark.macos_only +@pytest.mark.platforms("macos") class TestReadClaudeCodeCredentialsPriority: """Bug 4: Keychain must be checked before the JSON file.""" @@ -122,7 +122,7 @@ def test_returns_none_when_neither_keychain_nor_json_has_creds(self, tmp_path, m assert creds is None -@pytest.mark.macos_only +@pytest.mark.platforms("macos") class TestReadClaudeCodeCredentialsDesync: """Reconciliation when Keychain and JSON file disagree. diff --git a/tests/agent/test_anthropic_oauth_stress.py b/tests/agent/test_anthropic_oauth_stress.py index b558eb4a6aaf..fc6b00d2eae5 100644 --- a/tests/agent/test_anthropic_oauth_stress.py +++ b/tests/agent/test_anthropic_oauth_stress.py @@ -261,7 +261,7 @@ def _run(idx: int) -> None: @pytest.mark.live_system_guard_bypass -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_distinct_profiles_share_one_claude_refresh_without_duplicate_post( hermes_home, ): diff --git a/tests/agent/test_command_token_source.py b/tests/agent/test_command_token_source.py index e7210480ef4c..803bece06c8a 100644 --- a/tests/agent/test_command_token_source.py +++ b/tests/agent/test_command_token_source.py @@ -14,6 +14,8 @@ from __future__ import annotations +import base64 +import sys import time from types import SimpleNamespace @@ -27,21 +29,38 @@ ) +def _pycmd(code: str) -> str: + """A key_cmd that works under every host shell. + + ``_mint`` runs the command through ``shell=True``: POSIX sh on Linux/macOS, + ``cmd.exe`` on Windows — where ``printf``/``date`` don't exist and + single-quote wrapping means something else. Base64-encoding the payload + keeps the command pure printable ASCII with no quote characters for either + shell to mangle, and ``exec`` runs it unchanged. + """ + encoded = base64.b64encode(code.encode("utf-8")).decode("ascii") + return ( + f'"{sys.executable}" -c ' + f'"import base64; exec(base64.b64decode(\'{encoded}\').decode())"' + ) + + class TestMinting: def test_bare_token_stdout(self): - source = CommandTokenSource("printf 'tok-abc'", "dbx") + source = CommandTokenSource(_pycmd("print('tok-abc')"), "dbx") assert source() == "tok-abc" def test_json_access_token(self): """The OAuth 2.0 token-endpoint response shape.""" source = CommandTokenSource( - """printf '{"access_token":"tok-json","expires_in":3600}'""", "dbx" + _pycmd("""print('{"access_token":"tok-json","expires_in":3600}')"""), + "dbx", ) assert source() == "tok-json" def test_trailing_newline_is_stripped(self): """A raw newline in the credential would corrupt the auth header.""" - assert CommandTokenSource("echo tok-nl", "dbx")() == "tok-nl" + assert CommandTokenSource(_pycmd("print('tok-nl')"), "dbx")() == "tok-nl" def test_multiline_output_is_rejected_not_guessed(self): """Only the token may land on stdout. @@ -50,26 +69,30 @@ def test_multiline_output_is_rejected_not_guessed(self): warning, two tokens) into a corrupt-credential 401 that is much harder to diagnose than an explicit refusal. """ - source = CommandTokenSource("printf 'banner\\ntok-real'", "dbx") + source = CommandTokenSource( + _pycmd("print('banner'); print('tok-real')"), "dbx" + ) with pytest.raises(CommandTokenError, match="multiple lines"): source() def test_json_without_access_token_is_an_error(self): - source = CommandTokenSource("""printf '{"nope":1}'""", "dbx") + source = CommandTokenSource(_pycmd("""print('{"nope":1}')"""), "dbx") with pytest.raises(CommandTokenError, match="access_token"): source() def test_empty_output_is_an_error(self): with pytest.raises(CommandTokenError, match="no output"): - CommandTokenSource("true", "dbx")() + CommandTokenSource(_pycmd("pass"), "dbx")() def test_nonzero_exit_is_an_error(self): with pytest.raises(CommandTokenError, match="exited 3"): - CommandTokenSource("exit 3", "dbx")() + CommandTokenSource(_pycmd("import sys; sys.exit(3)"), "dbx")() def test_failure_message_is_actionable_without_echoing_the_command(self): """Actionable, but never echoes the command (it may embed a secret).""" - secret_cmd = "print-token --client-secret=SENTINEL-SECRET; exit 1" + secret_cmd = _pycmd( + "import sys; sys.stderr.write('SENTINEL-SECRET\\n'); sys.exit(1)" + ) with pytest.raises(CommandTokenError) as excinfo: CommandTokenSource(secret_cmd, "dbx")() message = str(excinfo.value) @@ -82,7 +105,10 @@ class TestNoCredentialLeak: def test_failure_message_excludes_command_output(self): """A failing auth helper may print a token — it must not be surfaced.""" source = CommandTokenSource( - "printf 'SENTINEL-SECRET'; printf 'stderr-SENTINEL' >&2; exit 1", + _pycmd( + "import sys; sys.stdout.write('SENTINEL-SECRET\\n');" + " sys.stderr.write('stderr-SENTINEL\\n'); sys.exit(1)" + ), "dbx", ) with pytest.raises(CommandTokenError) as excinfo: @@ -91,12 +117,14 @@ def test_failure_message_excludes_command_output(self): class TestCaching: + @pytest.mark.platforms("linux") def test_token_is_cached_between_calls(self): """Without caching the command would run on every request.""" # A command whose output changes each run: equal results prove caching. source = CommandTokenSource("date +%s%N", "dbx") assert source() == source() + @pytest.mark.platforms("linux") def test_expired_token_is_reminted(self): # date +%s%N changes every run; $RANDOM would be bash-only (empty # under dash, which is what /bin/sh is on Debian-family CI). @@ -109,6 +137,7 @@ def test_expired_token_is_reminted(self): source._expires_at = 0.0 assert source() != first + @pytest.mark.platforms("linux") def test_no_advertised_ttl_caches_on_a_bounded_window(self): """No TTL means a bounded cache, not a process-lifetime one. @@ -128,7 +157,7 @@ def test_no_advertised_ttl_caches_on_a_bounded_window(self): def test_advertised_ttl_sets_an_expiry(self): source = CommandTokenSource( - """printf '{"access_token":"tok","expires_in":3600}'""", "dbx" + _pycmd("""print('{"access_token":"tok","expires_in":3600}')"""), ) source() assert source._expires_at is not None @@ -136,7 +165,7 @@ def test_advertised_ttl_sets_an_expiry(self): def test_ttl_shorter_than_the_leeway_still_caches_briefly(self): """A leeway larger than the TTL must not disable caching entirely.""" source = CommandTokenSource( - """printf '{"access_token":"tok","expires_in":1}'""", "dbx" + _pycmd("""print('{"access_token":"tok","expires_in":1}')"""), ) source() assert source._expires_at is not None @@ -149,7 +178,7 @@ def test_returns_none_when_unset(self): assert build_command_token_provider(" ") is None def test_returns_callable_when_set(self): - provider = build_command_token_provider("printf tok", "dbx") + provider = build_command_token_provider(_pycmd("print('tok')"), "dbx") assert callable(provider) assert provider() == "tok" @@ -166,7 +195,7 @@ def test_key_cmd_entry_resolves_to_a_callable(self, monkeypatch): "base_url": "https://example.invalid/v1", "api_mode": "chat_completions", "model": "m1", - "key_cmd": "printf minted-token", + "key_cmd": _pycmd("print('minted-token')"), } } } @@ -188,7 +217,7 @@ def test_explicit_api_key_still_wins(self, monkeypatch): "base_url": "https://example.invalid/v1", "api_mode": "chat_completions", "model": "m1", - "key_cmd": "printf minted-token", + "key_cmd": _pycmd("print('minted-token')"), } } } @@ -249,36 +278,38 @@ def _iso(seconds_from_now: float) -> str: def test_iso_expiry_yields_a_ttl(self): deadline = self._iso(3600) - _, ttl = _mint(f"printf '%s' '{{\"access_token\":\"t\",\"expiry\":\"{deadline}\"}}'", "p") + _, ttl = _mint(_pycmd(f"""print('{{"access_token":"t","expiry":"{deadline}"}}')"""), "p") assert ttl is not None, "an advertised deadline must produce a TTL" assert 3500 < ttl <= 3600 def test_azure_expires_on_spelling(self): deadline = self._iso(1800) - _, ttl = _mint(f"printf '%s' '{{\"access_token\":\"t\",\"expiresOn\":\"{deadline}\"}}'", "p") + _, ttl = _mint(_pycmd(f"""print('{{"access_token":"t","expiresOn":"{deadline}"}}')"""), "p") assert ttl is not None and 1700 < ttl <= 1800 def test_expires_in_still_wins_when_both_present(self): """The RFC 6749 field is authoritative where a helper sends both.""" deadline = self._iso(3600) _, ttl = _mint( - f"printf '%s' '{{\"access_token\":\"t\",\"expires_in\":120,\"expiry\":\"{deadline}\"}}'", + _pycmd(f"""print('{{"access_token":"t","expires_in":120,"expiry":"{deadline}"}}')"""), "p", ) assert ttl == 120.0 def test_unparseable_expiry_is_not_a_ttl(self): """Junk must fall back to refresh-on-401, never to a guessed deadline.""" - _, ttl = _mint('printf \'%s\' \'{"access_token":"t","expiry":"whenever"}\'', "p") + _, ttl = _mint(_pycmd("""print('{"access_token":"t","expiry":"whenever"}')"""), "p") assert ttl is None def test_already_past_expiry_is_not_a_ttl(self): """A stale deadline must not become a negative or zero TTL.""" _, ttl = _mint( - f"printf '%s' '{{\"access_token\":\"t\",\"expiry\":\"{self._iso(-60)}\"}}'", "p" + _pycmd(f"""print('{{"access_token":"t","expiry":"{self._iso(-60)}"}}')"""), + "p", ) assert ttl is None + @pytest.mark.platforms("linux") def test_the_token_actually_gets_re_minted(self, tmp_path): """The regression that mattered: a deadline must expire the cache.""" counter = tmp_path / "calls" @@ -327,7 +358,7 @@ def _spy(*, api_key, base_url, **kw): BASE = {"base_url": "https://example.invalid/v1", "model": "m1"} def test_key_cmd_resolves_to_a_callable(self, monkeypatch): - api_key = self._resolve(monkeypatch, {**self.BASE, "key_cmd": "printf minted-token"}) + api_key = self._resolve(monkeypatch, {**self.BASE, "key_cmd": _pycmd("print('minted-token')")}) assert callable(api_key), "auxiliary tasks must mint per request too" assert api_key() == "minted-token" @@ -335,7 +366,7 @@ def test_key_cmd_beats_static_credentials(self, monkeypatch): """Precedence matches the runtime resolver, so both agree on one entry.""" api_key = self._resolve( monkeypatch, - {**self.BASE, "api_key": "stale-static", "key_cmd": "printf minted-token"}, + {**self.BASE, "api_key": "stale-static", "key_cmd": _pycmd("print('minted-token')")}, ) assert callable(api_key) and api_key() == "minted-token" diff --git a/tests/agent/test_file_safety_sandbox_mirror.py b/tests/agent/test_file_safety_sandbox_mirror.py index bb59c1ecfb27..b974f5e8affc 100644 --- a/tests/agent/test_file_safety_sandbox_mirror.py +++ b/tests/agent/test_file_safety_sandbox_mirror.py @@ -14,6 +14,7 @@ """ from __future__ import annotations +import os from pathlib import Path import pytest @@ -42,9 +43,9 @@ def test_docker_mirror_soul_md_classified(self, tmp_path): assert result is not None assert result["target_path"] == str(target.resolve()) assert result["mirror_root"].endswith( - "sandboxes/docker/default/home/.hermes" + os.path.join("sandboxes", "docker", "default", "home", ".hermes") ) - assert result["inner_path"] == "profiles/group1/SOUL.md" + assert result["inner_path"] == os.path.join("profiles", "group1", "SOUL.md") @pytest.mark.parametrize( "backend,inner", @@ -68,7 +69,7 @@ def test_other_backends_and_inner_files_match(self, tmp_path, backend, inner): result = classify_sandbox_mirror_target(str(target)) assert result is not None - assert result["inner_path"] == inner + assert result["inner_path"] == str(Path(inner)) assert backend in result["mirror_root"] @@ -107,9 +108,9 @@ def test_mirror_warning_names_mirror_root_and_inner_path(self, tmp_path): warn = get_sandbox_mirror_warning(str(target)) assert warn is not None # Must name the mirror root so the user can locate the sandbox. - assert "sandboxes/docker/default/home/.hermes" in warn + assert os.path.join("sandboxes", "docker", "default", "home", ".hermes") in warn # Must hint at what the agent likely meant. - assert "profiles/group1/SOUL.md" in warn + assert os.path.join("profiles", "group1", "SOUL.md") in warn # Must name the bypass kwarg shared with the cross-profile guard. assert "cross_profile=True" in warn diff --git a/tests/agent/test_image_routing.py b/tests/agent/test_image_routing.py index f545909e1fa5..b26dae5dfe07 100644 --- a/tests/agent/test_image_routing.py +++ b/tests/agent/test_image_routing.py @@ -378,8 +378,13 @@ def test_finds_absolute_path(self, tmp_path: Path): assert urls == [] def test_finds_home_relative_path(self, tmp_path: Path, monkeypatch): - # Simulate ~/foo.png by pointing HOME at tmp_path and creating the file + # Simulate ~/foo.png by pointing the home env vars at tmp_path and + # creating the file. Windows resolves ~ via USERPROFILE (before HOME), + # so set both and clear the HOMEDRIVE/HOMEPATH fallback. monkeypatch.setenv("HOME", str(tmp_path)) + monkeypatch.setenv("USERPROFILE", str(tmp_path)) + monkeypatch.delenv("HOMEDRIVE", raising=False) + monkeypatch.delenv("HOMEPATH", raising=False) img = tmp_path / "foo.png" img.write_bytes(_png_bytes()) paths, urls = extract_image_refs("see ~/foo.png please") diff --git a/tests/agent/test_proxy_and_url_validation.py b/tests/agent/test_proxy_and_url_validation.py index 766c361630a8..c230bf68b27b 100644 --- a/tests/agent/test_proxy_and_url_validation.py +++ b/tests/agent/test_proxy_and_url_validation.py @@ -28,7 +28,7 @@ ]) def test_proxy_env_rejects_malformed_port(monkeypatch, key): monkeypatch.setenv(key, "http://127.0.0.1:6153export") - with pytest.raises(RuntimeError, match=rf"Malformed proxy environment variable {key}=.*6153export"): + with pytest.raises(RuntimeError, match=rf"(?i)Malformed proxy environment variable {key}=.*6153export"): _validate_proxy_env_urls() diff --git a/tests/agent/test_relay_runtime_plugins.py b/tests/agent/test_relay_runtime_plugins.py index a906f94f29bd..4cd1c0372110 100644 --- a/tests/agent/test_relay_runtime_plugins.py +++ b/tests/agent/test_relay_runtime_plugins.py @@ -1025,7 +1025,7 @@ def test_real_binding_ignores_project_config_without_explicit_opt_in( [[components.config.atof.sinks]] type = "file" -output_directory = "{atof_dir}" +output_directory = "{atof_dir.as_posix()}" filename = "events.jsonl" mode = "overwrite" """.strip(), @@ -1081,7 +1081,7 @@ def test_real_binding_layers_project_config_after_explicit_opt_in( [[components.config.atof.sinks]] type = "file" -output_directory = "{atof_dir}" +output_directory = "{atof_dir.as_posix()}" filename = "events.jsonl" mode = "overwrite" """.strip(), @@ -1142,13 +1142,13 @@ def test_real_binding_keeps_two_profile_trajectories_separate_in_shared_exporter [[components.config.atof.sinks]] type = "file" -output_directory = "{atof_dir}" +output_directory = "{atof_dir.as_posix()}" filename = "events.jsonl" mode = "overwrite" [components.config.atif] enabled = true -output_directory = "{atif_dir}" +output_directory = "{atif_dir.as_posix()}" filename_template = "trajectory-{{session_id}}.json" agent_name = "Hermes Native Test" agent_version = "test" diff --git a/tests/agent/test_save_url_image.py b/tests/agent/test_save_url_image.py index 3737871f6756..3cf698c26cd0 100644 --- a/tests/agent/test_save_url_image.py +++ b/tests/agent/test_save_url_image.py @@ -13,6 +13,7 @@ from __future__ import annotations import http.server +import os import socketserver import threading @@ -104,7 +105,7 @@ def test_writes_real_bytes_to_hermes_home_cache(self, http_server): assert path.read_bytes() == PNG_1PX # The cache directory must be under HERMES_HOME — gateway cleanup # relies on this being the canonical location. - assert "cache/images" in str(path) + assert os.path.join("cache", "images") in str(path) assert path.suffix == ".png" diff --git a/tests/agent/test_shell_hooks.py b/tests/agent/test_shell_hooks.py index ef7c45ebf706..e29b24233886 100644 --- a/tests/agent/test_shell_hooks.py +++ b/tests/agent/test_shell_hooks.py @@ -142,6 +142,7 @@ def test_whitespace_only_matcher_becomes_none(self): # ── End-to-end subprocess behaviour ─────────────────────────────────────── +@pytest.mark.platforms("linux") class TestCallbackSubprocess: @@ -678,6 +679,7 @@ def test_fail_closed_on_non_blocking_event_still_fails_open(self): class TestFailSemanticsEndToEnd: + @pytest.mark.platforms("linux") def test_exit_2_script_blocks(self, tmp_path): script = _write_script( tmp_path, "exit2.sh", @@ -705,6 +707,7 @@ def test_fail_closed_missing_command_blocks(self, tmp_path): assert result is not None and result["action"] == "block" assert "failed closed" in result["message"] + @pytest.mark.platforms("linux") def test_run_once_reflects_exit_2_block(self, tmp_path): """hermes hooks test must mirror production semantics.""" script = _write_script( @@ -722,6 +725,7 @@ def test_run_once_reflects_exit_2_block(self, tmp_path): assert result["returncode"] == 2 assert result["parsed"] == {"action": "block", "message": "denied"} + @pytest.mark.platforms("linux") def test_run_once_reflects_fail_closed_timeout(self, tmp_path): script = _write_script( tmp_path, "sleepy.sh", diff --git a/tests/agent/test_shell_hooks_consent.py b/tests/agent/test_shell_hooks_consent.py index 6835e06c3d39..0ac00d14f2e4 100644 --- a/tests/agent/test_shell_hooks_consent.py +++ b/tests/agent/test_shell_hooks_consent.py @@ -144,6 +144,7 @@ class TestAllowlistOps: + @pytest.mark.platforms("linux") def test_tilde_path_approval_records_resolvable_mtime(self, tmp_path, monkeypatch): """If the command uses ~ the approval must still find the file.""" monkeypatch.setenv("HOME", str(tmp_path)) diff --git a/tests/agent/test_skill_commands.py b/tests/agent/test_skill_commands.py index 623f5a5c05ca..691e4c18532c 100644 --- a/tests/agent/test_skill_commands.py +++ b/tests/agent/test_skill_commands.py @@ -762,6 +762,7 @@ class TestInlineShellExpansion: + @pytest.mark.platforms("linux") def test_inline_shell_runs_in_skill_directory(self, tmp_path): """Inline snippets get the skill dir as CWD so relative paths work.""" with ( diff --git a/tests/agent/test_subdirectory_hints.py b/tests/agent/test_subdirectory_hints.py index 85b89f647ef5..e310aed62e0e 100644 --- a/tests/agent/test_subdirectory_hints.py +++ b/tests/agent/test_subdirectory_hints.py @@ -182,6 +182,7 @@ class TestContentDeduplication: """The same context content must never be injected twice (ref: symlinked shared workspaces, hardlinks, and copied backups all alias one file).""" + @pytest.mark.require_symlinks def test_symlinked_duplicate_not_reinjected(self, tmp_path): """Two directories whose AGENTS.md is the same file yield one injection.""" real = tmp_path / "real" diff --git a/tests/agent/test_treekill_consolidation.py b/tests/agent/test_treekill_consolidation.py index 902629986c26..f48242a994e9 100644 --- a/tests/agent/test_treekill_consolidation.py +++ b/tests/agent/test_treekill_consolidation.py @@ -135,6 +135,7 @@ def test_delegates_sigterm_tree_first(self, monkeypatch): code_execution_tool._kill_process_group(proc) assert calls == [(5555, _signal.SIGTERM)] + @pytest.mark.platforms("linux") def test_escalate_waits_then_sigkills_tree(self, monkeypatch): import signal as _signal diff --git a/tests/ci/test_classify_changes.py b/tests/ci/test_classify_changes.py index 81e93d680944..dc39f0332591 100644 --- a/tests/ci/test_classify_changes.py +++ b/tests/ci/test_classify_changes.py @@ -176,7 +176,9 @@ def _lanes(python=False, frontend=False, site=False, scan=False, deps=False, uv_ _lanes(python=True, scan=True), ), # Runner infrastructure is NOT tests-only — a bad runner edit can mask - # real failures, so it keeps the conservative full lane set. + # real failures, so it keeps the conservative full lane set. The .py + # runner additionally trips the supply-chain scan lane (executable + # .py/.pth payloads are what it scans for). "test runner script → python_prod stays on": ( ["scripts/run_tests_parallel.py"], _lanes(python=True, scan=True), diff --git a/tests/ci/test_list_os_marked_tests.py b/tests/ci/test_list_os_marked_tests.py index 9b44a7596c98..724fbb2ea369 100644 --- a/tests/ci/test_list_os_marked_tests.py +++ b/tests/ci/test_list_os_marked_tests.py @@ -1,11 +1,9 @@ -"""Tests for ``scripts/ci/list_os_marked_tests.py``. - -The helper decides which files the macOS / Windows CI lanes import. Its -failure modes matter more than its happy path: if it silently returned an -empty list, the OS lane would run zero tests and still report green — the -exact silent-coverage-loss the lanes exist to prevent. So the contracts under -test are "finds real markers", "refuses to emit nothing", and "rejects an -unknown marker". +"""Tests for scripts/ci/list_os_marked_tests.py. + +The properties this tool must keep are "finds real gates", "refuses to emit +nothing", and "rejects unknown platforms" — the macOS lane imports exactly +what it emits, so under-selection silently drops coverage while +over-selection is corrected by the per-test host skips. """ from __future__ import annotations @@ -37,67 +35,91 @@ def _write(root: Path, relpath: str, body: str) -> Path: return path -@pytest.mark.parametrize("marker", ["linux_only", "macos_only", "windows_only"]) -def test_finds_decorator_and_pytestmark_forms(tmp_path, marker): +@pytest.mark.parametrize("platform", ["linux", "macos", "windows"]) +def test_finds_decorator_and_pytestmark_forms(tmp_path, platform): """Both the decorator form and module-level ``pytestmark`` are detected.""" _write( tmp_path, "test_decorated.py", - f"import pytest\n\n\n@pytest.mark.{marker}\ndef test_x():\n pass\n", + f'import pytest\n\n\n@pytest.mark.platforms("{platform}")\ndef test_x():\n pass\n', ) _write( tmp_path, "nested/test_module_level.py", - f"import pytest\n\npytestmark = pytest.mark.{marker}\n\n\ndef test_y():\n pass\n", + f'import pytest\n\npytestmark = pytest.mark.platforms("{platform}")\n\n\ndef test_y():\n pass\n', ) - # A file with no marker at all must not be selected. + # A file with no gate at all must not be selected. _write(tmp_path, "test_plain.py", "def test_z():\n pass\n") + # A platforms() file gating a DIFFERENT platform must not be selected. + other = "windows" if platform != "windows" else "linux" + _write( + tmp_path, + "test_other_platform.py", + f'import pytest\n\n\n@pytest.mark.platforms("{other}")\ndef test_w():\n pass\n', + ) + # A bare identifier that merely LOOKS like the platform name must not + # match — the spec has to appear inside the quoted string literal. + _write( + tmp_path, + "test_bare_identifier.py", + f"import pytest\n\n\n@pytest.mark.parametrize(\"kind\", [\"{platform}\"])\ndef test_v():\n assert kind\n", + ) - result = _run(marker, str(tmp_path)) + result = _run(platform, str(tmp_path)) assert result.returncode == 0, result.stderr listed = result.stdout.split() assert any(p.endswith("test_decorated.py") for p in listed) assert any(p.endswith("test_module_level.py") for p in listed) assert not any(p.endswith("test_plain.py") for p in listed) + assert not any(p.endswith("test_other_platform.py") for p in listed) + assert not any(p.endswith("test_bare_identifier.py") for p in listed) + +def test_negated_spec_lists_for_that_lane(tmp_path): + """``platforms("not macos")`` lists the file for the macOS lane import. -def test_marker_matched_as_whole_word(tmp_path): - """``macos_only`` must not match a longer identifier that contains it.""" + The script only decides which files are IMPORTED; the per-test host + skips remain authoritative. A negated-spec file may still contain other + tests, so it must be imported on the macOS lane — under-selecting is + the failure mode this tool exists to prevent. + """ _write( tmp_path, - "test_lookalike.py", - "import pytest\n\n\n@pytest.mark.macos_only_extra\ndef test_x():\n pass\n", + "test_negated.py", + 'import pytest\n\n\n@pytest.mark.platforms("not macos")\ndef test_x():\n pass\n', ) - result = _run("macos_only", str(tmp_path)) - - # No genuine match: the helper must fail rather than emit nothing. - assert result.returncode == 1 - assert "macos_only" in result.stderr + result = _run("macos", str(tmp_path)) + assert result.returncode == 0, result.stderr + listed = result.stdout.split() + assert any(p.endswith("test_negated.py") for p in listed) -def test_exits_nonzero_when_no_file_carries_the_marker(tmp_path): - """The load-bearing guard: an empty result is an error, never a silent pass.""" - _write(tmp_path, "test_plain.py", "def test_z():\n pass\n") - result = _run("windows_only", str(tmp_path)) +def test_unknown_platform_is_rejected(tmp_path): + result = _run("amiga", str(tmp_path)) + assert result.returncode == 2 + assert "unknown platform" in result.stderr - assert result.returncode == 1 - assert result.stdout.strip() == "" - assert "renamed or dropped" in result.stderr +def test_exits_nonzero_when_no_file_gates_on_the_platform(tmp_path): + """A platform with zero gated files must fail, not emit nothing.""" + _write( + tmp_path, + "test_unrelated.py", + 'import pytest\n\n\n@pytest.mark.platforms("linux")\ndef test_x():\n pass\n', + ) -def test_rejects_unknown_marker(tmp_path): - result = _run("bsd_only", str(tmp_path)) + result = _run("macos", str(tmp_path)) - assert result.returncode == 2 - assert "unknown marker" in result.stderr + # No genuine match: the helper must fail rather than emit nothing. + assert result.returncode != 0 + assert "no test files" in result.stderr def test_rejects_missing_root(): - result = _run("macos_only", "/nonexistent/path/for/this/test") - + result = _run("macos", "/nonexistent/path/for/this/test") assert result.returncode == 2 assert "no such directory" in result.stderr @@ -110,7 +132,7 @@ def test_emits_repo_relative_posix_paths(): exercises — an out-of-repo root can't be made repo-relative and is emitted absolute instead. """ - result = _run("windows_only") + result = _run("windows") assert result.returncode == 0, result.stderr listed = result.stdout.split() @@ -120,17 +142,9 @@ def test_emits_repo_relative_posix_paths(): assert not Path(line).is_absolute() -def test_real_tree_selects_files_for_every_marker(): - """Against the actual ``tests/`` tree each marker resolves to real files. - - This is the invariant the CI lanes depend on — not a snapshot of which - files those are, only that each marker is in use and every listed path - exists. - """ - for marker in ("linux_only", "macos_only", "windows_only"): - result = _run(marker) - assert result.returncode == 0, f"{marker}: {result.stderr}" - listed = result.stdout.split() - assert listed, f"{marker} selected no files" - for rel in listed: - assert (REPO_ROOT / rel).is_file(), f"{marker} listed missing {rel}" +def test_real_tree_selects_files_for_every_platform(): + """Against the actual ``tests/`` tree each platform resolves to real files.""" + for platform in ("linux", "macos", "windows"): + result = _run(platform) + assert result.returncode == 0, (platform, result.stderr) + assert result.stdout.split(), f"{platform} selected nothing in the real tree" diff --git a/tests/cli/test_bang_shell_mode.py b/tests/cli/test_bang_shell_mode.py index 63fedde30d33..d9a197c32ac5 100644 --- a/tests/cli/test_bang_shell_mode.py +++ b/tests/cli/test_bang_shell_mode.py @@ -89,6 +89,7 @@ def test_disabled_in_non_cli_contexts(self, monkeypatch, var, value): # ── execution ────────────────────────────────────────────────────────────── class TestBangExecution: + @pytest.mark.platforms("linux") def test_output_is_streamed_to_writer(self): lines = [] code = run_bang_command("echo bang-one; echo bang-two", writer=lines.append) @@ -96,6 +97,7 @@ def test_output_is_streamed_to_writer(self): assert "bang-one" in lines assert "bang-two" in lines + @pytest.mark.platforms("linux") def test_stderr_is_merged_into_output(self): lines = [] run_bang_command("echo to-stderr >&2", writer=lines.append) @@ -106,6 +108,7 @@ def test_nonzero_exit_code_is_returned(self): code = run_bang_command("exit 42", writer=lines.append) assert code == 42 + @pytest.mark.platforms("linux") def test_runs_in_requested_cwd(self, tmp_path): lines = [] code = run_bang_command("pwd", cwd=str(tmp_path), writer=lines.append) diff --git a/tests/cli/test_cli_file_drop.py b/tests/cli/test_cli_file_drop.py index e08e37f95f64..cf0477fe5c2f 100644 --- a/tests/cli/test_cli_file_drop.py +++ b/tests/cli/test_cli_file_drop.py @@ -168,10 +168,10 @@ def test_tilde_prefixed_path(self, tmp_path, monkeypatch): assert result["remainder"] == "what is this?" - # ``windows_only`` rather than ``skipif(os.name != "nt")``: the Windows CI - # job selects ``-m windows_only``, so a bare skipif would leave this + # ``platforms("windows")`` rather than ``skipif(os.name != "nt")``: the Windows CI + # job selects ``-m platforms("windows")``, so a bare skipif would leave this # skipped on Linux AND unselected there — dead on every host. - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_drive_letter_file_uri_drops_url_leading_slash(self, tmp_path): image = tmp_path / "drive-uri.png" image.write_bytes(b"\x89PNG\r\n\x1a\n") diff --git a/tests/cli/test_cli_init.py b/tests/cli/test_cli_init.py index cca40f831e76..8234a58f4271 100644 --- a/tests/cli/test_cli_init.py +++ b/tests/cli/test_cli_init.py @@ -154,6 +154,7 @@ def test_interrupt_mode_routes_busy_enter_to_interrupt(self): class TestPromptToolkitTerminalCompatibility: + @pytest.mark.platforms("linux") def test_lf_enter_binding_respects_multiline_shortcuts(self): """Ctrl+J is reserved by default, with legacy LF-submit available as an opt-out. @@ -222,7 +223,7 @@ def submit_handler(event): assert bindings[("c-m",)] is submit_handler assert ("c-j",) not in bindings - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_leaves_ctrl_j_unbound(self): """On native Windows only enter submits; c-j is free for the newline binding added separately in the prompt setup.""" @@ -251,12 +252,13 @@ def test_cpr_warning_callback_is_disabled(self): + @pytest.mark.platforms("linux") def test_cpr_gating_posix_suppresses_without_ssh(self, monkeypatch): """POSIX suppresses CPR without SSH. The native-Windows arm (``_terminal_may_leak_cpr() is False``, plus the ``PROMPT_TOOLKIT_NO_CPR`` override that outranks it) lives in - ``tests/cli/test_cpr_local_leak.py`` under ``windows_only``, where it + ``tests/cli/test_cpr_local_leak.py`` under ``platforms("windows")``, where it runs against a real Windows console. """ from cli import _terminal_may_leak_cpr diff --git a/tests/cli/test_cli_light_mode.py b/tests/cli/test_cli_light_mode.py index 0763cc4ff67c..a8b28e13c02a 100644 --- a/tests/cli/test_cli_light_mode.py +++ b/tests/cli/test_cli_light_mode.py @@ -176,6 +176,7 @@ def test_skin_color_passthrough_in_dark_mode(self, cli_mod, monkeypatch): assert skin.get_color("banner_text") == "#FFF8DC" +@pytest.mark.platforms("linux") class TestOsc11DrainGuard: """Regression: a late-arriving OSC 11 reply must not leak into prompt_toolkit's input buffer (#40250). diff --git a/tests/cli/test_cpr_local_leak.py b/tests/cli/test_cpr_local_leak.py index 3d636e00b58f..37340e8d295a 100644 --- a/tests/cli/test_cpr_local_leak.py +++ b/tests/cli/test_cpr_local_leak.py @@ -33,12 +33,12 @@ def _clear_cpr_env(monkeypatch): class TestClassicCliOutputSelection: - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_preserves_default_output_selection(self): assert _terminal_may_leak_cpr() is False assert _select_classic_cli_pt_output(sys.stdout) is None - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_honors_explicit_no_cpr(self, monkeypatch): monkeypatch.setenv("PROMPT_TOOLKIT_NO_CPR", "1") assert _terminal_may_leak_cpr() is True diff --git a/tests/cli/test_ctrl_enter_newline.py b/tests/cli/test_ctrl_enter_newline.py index a7eaef772611..bb88795f1eeb 100644 --- a/tests/cli/test_ctrl_enter_newline.py +++ b/tests/cli/test_ctrl_enter_newline.py @@ -11,7 +11,7 @@ ``_preserve_ctrl_enter_newline()`` short-circuits to True on native Windows before it ever looks at the environment, so the env-driven cases below are POSIX assertions and run on the Linux job. The native-Windows short-circuit -is marked ``windows_only`` and asserted on the real host — patching +is marked ``platforms("windows")`` and asserted on the real host — patching ``sys.platform`` to ``"win32"`` here would only re-assert the literal in the ``if``, on an interpreter where none of the Windows terminal behaviour it exists for is present. @@ -52,7 +52,7 @@ def _bind_submit_keys_for_local_linux(cli_mod, *, multiline_shortcuts_enabled): return kb -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_native_windows_preserves_newline(): import cli as cli_mod @@ -211,11 +211,11 @@ def test_non_ghostty_terminals_still_push_kitty_protocol(): assert b"\x1b[>4;2m" in out.written -@pytest.mark.linux_only +@pytest.mark.platforms("linux") def test_proc_version_microsoft_marker_preserves_newline(): """WSL detection via /proc when env vars are scrubbed (sudo etc.). - ``linux_only``: the fallback reads ``/proc/version`` — a Linux-only + ``platforms("linux")``: the fallback reads ``/proc/version`` — a Linux-only interface, and the WSL kernels it sniffs for are Linux kernels. """ import cli as cli_mod diff --git a/tests/cli/test_slash_confirm_windows.py b/tests/cli/test_slash_confirm_windows.py index 6563162870f2..962d1950d735 100644 --- a/tests/cli/test_slash_confirm_windows.py +++ b/tests/cli/test_slash_confirm_windows.py @@ -16,7 +16,7 @@ (on win32, off-thread) a scheduling failure degrades to a clean cancel. 4. Empty choices returns None. -**Why the Windows cases are ``windows_only`` rather than ``sys.platform`` +**Why the Windows cases are ``platforms("windows")`` rather than ``sys.platform`` patches.** The deadlock #33961 fixed is a real property of the Windows console: a raw ``input()`` off the main thread blocks forever against prompt_toolkit's stdin ownership. On Linux the same call returns immediately (or EOFs), so a test @@ -151,7 +151,7 @@ def test_main_thread_with_app_uses_modal(self): mock_stdin.assert_not_called() assert result == "once" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_scheduling_failure_clean_cancels(self): """win32 off the main thread: if marshaling onto the app loop fails, cancel cleanly (None) rather than fall to raw input() (which deadlocks on native @@ -183,7 +183,7 @@ class TestConfirmDestructiveSlash: This is the flow bug #33961 froze on native Windows. The fix made it platform-agnostic (modal via the app loop), so the assertion holds on whichever host runs it. The class carries no OS marker, so the - ``windows_only`` lane deselects it — the deadlock tests above are the + ``platforms("windows")`` lane deselects it — the deadlock tests above are the Windows-side regression guard. """ @@ -222,7 +222,7 @@ def test_confirm_destructive_slash_uses_modal(self, response, expected): assert outcome["result"] == expected -@pytest.mark.windows_only +@pytest.mark.platforms("windows") class TestNativeWindowsNoRawInputDeadlock: """Anti-regression guard exercising the REAL ``_prompt_text_input``. @@ -238,7 +238,7 @@ class TestNativeWindowsNoRawInputDeadlock: pre-#33961 code (win32 → ``_prompt_text_input`` → off-main ``input()``) and pass once the modal path / clean-cancel fallback is in place. - ``windows_only``: the deadlock is a property of the Windows console's stdin + ``platforms("windows")``: the deadlock is a property of the Windows console's stdin ownership. Running this with a patched ``sys.platform`` on Linux exercises a blocking ``input()`` that does not actually deadlock there, so it could never have caught the regression it is named for. diff --git a/tests/cli/test_worktree.py b/tests/cli/test_worktree.py index c2fe4119c06f..4f7702d87f5d 100644 --- a/tests/cli/test_worktree.py +++ b/tests/cli/test_worktree.py @@ -1342,6 +1342,7 @@ def test_real_unpushed_work_survives_deepening(self, tmp_path): ) +@pytest.mark.platforms("linux") class TestPrMergedEscapeHatch: """Rebase-merged PRs whose diff changed during salvage defeat ``git cherry`` (patch-id mismatch), so the pruner asks GitHub whether the diff --git a/tests/computer_use/test_cua_no_overlay.py b/tests/computer_use/test_cua_no_overlay.py index 12582a34a2ea..f5f15ca06961 100644 --- a/tests/computer_use/test_cua_no_overlay.py +++ b/tests/computer_use/test_cua_no_overlay.py @@ -30,7 +30,7 @@ def test_explicit_true_overrides(self): assert cua_backend._cua_no_overlay() is True - @pytest.mark.macos_only + @pytest.mark.platforms("macos") def test_config_load_failure_falls_through_to_auto_detect_macos(self): """Unreadable config => auto-detect (macOS defaults to overlay off). @@ -41,7 +41,7 @@ def test_config_load_failure_falls_through_to_auto_detect_macos(self): side_effect=RuntimeError("boom")): assert cua_backend._cua_no_overlay() is True - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_config_load_failure_falls_through_to_auto_detect_linux(self, monkeypatch): """Unreadable config must not raise; headless Linux auto-detects off. @@ -53,7 +53,7 @@ def test_config_load_failure_falls_through_to_auto_detect_linux(self, monkeypatc side_effect=RuntimeError("boom")): assert cua_backend._cua_no_overlay() is True - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_linux_x11_auto_detects_off(self, monkeypatch): """X11 desktop (DISPLAY set, no Wayland) defaults the overlay off. @@ -68,7 +68,7 @@ def test_linux_x11_auto_detects_off(self, monkeypatch): with patch("hermes_cli.config.load_config", return_value={}): assert cua_backend._cua_no_overlay() is True - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_linux_x11_explicit_session_type_also_off(self, monkeypatch): """XDG_SESSION_TYPE=x11 without Wayland env is still X11.""" monkeypatch.setenv("DISPLAY", ":0") @@ -77,7 +77,7 @@ def test_linux_x11_explicit_session_type_also_off(self, monkeypatch): with patch("hermes_cli.config.load_config", return_value={}): assert cua_backend._cua_no_overlay() is True - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_linux_wayland_keeps_overlay(self, monkeypatch): """Wayland desktop keeps the overlay: the compositor owns the overlay surface lifecycle, so it cannot get stuck above every @@ -88,7 +88,7 @@ def test_linux_wayland_keeps_overlay(self, monkeypatch): with patch("hermes_cli.config.load_config", return_value={}): assert cua_backend._cua_no_overlay() is False - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_linux_x11_explicit_false_overrides_auto_detect(self, monkeypatch): """An explicit ``no_overlay: false`` must restore the cursor even on X11 — auto-detection is the default, never a hard lock.""" diff --git a/tests/conformance/persistence/test_cell1_prefix_durability.py b/tests/conformance/persistence/test_cell1_prefix_durability.py index 8427c6b06544..5f47b5fe5826 100644 --- a/tests/conformance/persistence/test_cell1_prefix_durability.py +++ b/tests/conformance/persistence/test_cell1_prefix_durability.py @@ -75,6 +75,7 @@ def _recover(db_path: Path) -> list[tuple]: @pytest.mark.parametrize("requested_mode", [None, "DELETE", "WAL"]) +@pytest.mark.platforms("linux") def test_acknowledged_appends_survive_sigkill(tmp_path, requested_mode): db_path = tmp_path / "state.db" journal = tmp_path / "acked.jsonl" diff --git a/tests/conformance/persistence/test_cell3_rotation_atomicity.py b/tests/conformance/persistence/test_cell3_rotation_atomicity.py index a8a700ddeb52..9a8294041386 100644 --- a/tests/conformance/persistence/test_cell3_rotation_atomicity.py +++ b/tests/conformance/persistence/test_cell3_rotation_atomicity.py @@ -78,6 +78,7 @@ def _completed_rotations(journal: Path) -> int: @pytest.mark.parametrize("requested_mode", [None, "DELETE", "WAL"]) +@pytest.mark.platforms("linux") def test_rotation_is_atomic_under_sigkill(tmp_path, requested_mode): db_path = tmp_path / "state.db" journal = tmp_path / "rotations.jsonl" diff --git a/tests/conftest.py b/tests/conftest.py index ffa5e7b9282c..58485c96d5c7 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -115,20 +115,19 @@ def _hermes_home_points_at_production(value: str) -> bool: HERMES_HOME_AT_CONFTEST_IMPORT = os.environ.get("HERMES_HOME", "") -# ── Per-file process isolation ────────────────────────────────────────────── -# Tests run via ``scripts/run_tests_parallel.py``, which spawns a fresh -# ``python -m pytest `` subprocess per test file. Cross-file state -# leakage (module-level dicts, ContextVars, caches) is impossible: each -# file gets a clean Python interpreter. Intra-file ordering is the test -# author's responsibility — if test A in foo.py mutates state that test B -# in foo.py reads, that's a real bug to fix in the file (it would also -# bite anyone running ``pytest tests/foo.py`` directly). +# ── File-level scheduling isolation ────────────────────────────────────────── +# Tests run via ``scripts/run_tests.sh``, which dispatches on host. POSIX: +# every file in its own freshly-spawned ``python -m pytest `` subprocess +# (``scripts/run_tests_parallel.py``) — cross-file state leakage is +# impossible. Windows: pytest-xdist with ``--dist loadfile``, which pins every +# test of a FILE to ONE worker — leakage is bounded to co-scheduled files, and +# any such leak is a stateful-test bug to fix at the tests. Intra-file +# ordering is the test author's responsibility on every host — if test A in +# foo.py mutates state that test B in foo.py reads, that's a real bug to fix +# in the file (it would also bite anyone running ``pytest tests/foo.py`` +# directly). # -# This replaces the historic _reset_module_state autouse fixture (manual -# state clearing) and the brief experiment with subprocess-per-test -# isolation (too slow at ~17k tests). -# -# See ``scripts/run_tests_parallel.py`` for the runner. +# See ``scripts/run_tests.sh`` for the runner. # ── Credential env-var filter ────────────────────────────────────────────── @@ -783,16 +782,14 @@ def _state_db_write_guard(request, monkeypatch): yield -# ── Module-level state reset — replaced by per-file process isolation ────── -# -# Each test FILE runs in a freshly-spawned ``python -m pytest `` -# subprocess via ``scripts/run_tests_parallel.py``, so module-level dicts / -# sets / ContextVars from tests in one file cannot leak into tests in -# another file. No manual per-module clearing needed. +# ── Module-level state reset — replaced by file-pinned xdist workers ──────── # -# Within a single file, ordering is the author's responsibility. If your -# tests in the same file share mutable state, either reset it explicitly -# in a fixture or split them across files. +# ``--dist loadfile`` pins each test FILE to ONE worker, so heavy +# co-scheduling pollution (module-level dicts / sets / ContextVars shared +# by many files) is bounded to files that land on the same worker. Within +# a single file, ordering is the author's responsibility. If your tests +# in the same file share mutable state, either reset it explicitly in a +# fixture or split them across files. # # The skill ``test-suite-cascade-diagnosis`` documents the cascade patterns # this replaces; the running example was ``test_command_guards`` failing @@ -1107,12 +1104,12 @@ def _wal_is_usable() -> bool: # So: a test whose subject is genuinely OS-specific declares the OS it # belongs to and runs there for real — # -# @pytest.mark.windows_only → only on native Windows (``sys.platform == "win32"``) -# @pytest.mark.macos_only → only on macOS (``sys.platform == "darwin"``) -# @pytest.mark.linux_only → only on Linux (``sys.platform.startswith("linux")``) +# @pytest.mark.platforms("windows") → only on native Windows (``sys.platform == "win32"``) +# @pytest.mark.platforms("macos") → only on macOS (``sys.platform == "darwin"``) +# @pytest.mark.platforms("linux") → only on Linux (``sys.platform.startswith("linux")``) # # Elsewhere the test is skipped, not faked. CI runs a dedicated macOS job -# (``-m macos_only``) and a dedicated Windows job (``-m windows_only``) so +# (``-m platforms("macos")``) and a dedicated Windows job (``-m platforms("windows")``) so # those markers are actually exercised on their own host rather than # quietly skipped everywhere. # @@ -1132,22 +1129,80 @@ def _wal_is_usable() -> bool: # another OS in order to pass, it belongs on that OS. # --------------------------------------------------------------------------- -_OS_MARKS = { - "linux_only": ( - lambda: sys.platform.startswith("linux"), - "Linux", - ), - "macos_only": ( - lambda: sys.platform == "darwin", - "macOS", - ), - "windows_only": ( - lambda: sys.platform == "win32", - "native Windows", - ), +_PLATFORM_ALIASES = { + "linux": ("linux",), + "macos": ("darwin", "macos"), + "windows": ("win32", "windows"), + "posix": ("linux", "darwin"), + "any": (), } +def _platform_machine() -> str: + import platform as _platform + + machine = (_platform.machine() or "").lower() + return {"amd64": "x86_64", "x86": "x86_64", "aarch64": "arm64"}.get(machine, machine) + + +def _host_matches_platforms(conditions, arch=None, arch_negate=False): + """Evaluate a platforms() marker payload against the running host. + + Returns ``(ok, skip_reason)``. + """ + host = sys.platform.lower() + machine = _platform_machine() + specs = [str(c).strip().lower() for c in conditions if str(c).strip()] + if not specs: + return True, "platforms() with no specs matches every host" + for spec in specs: + negate = spec.startswith("not ") + leaf = spec[4:].strip() if negate else spec + if leaf not in _PLATFORM_ALIASES: + return False, f"platforms(): unknown spec {spec!r}" + wanted = _PLATFORM_ALIASES[leaf] + matched = (not wanted) or host in wanted + if negate: + matched = not matched + if matched: + break + else: + return False, f"platforms({', '.join(specs)}); host is {sys.platform}" + if arch is not None: + arch_l = str(arch).lower() + arch_hit = machine == arch_l or ( + arch_l in {"arm64", "aarch64"} and machine == "arm64" + ) + if arch_negate: + arch_hit = not arch_hit + if not arch_hit: + return False, ( + f"platforms(arch={'not ' if arch_negate else ''}{arch}); " + f"host machine is {machine or 'unknown'}" + ) + return True, "" + + +def _platforms_gate_reason(item): + """Skip reason when the item's platforms() gating excludes this host.""" + for mark in item.iter_markers("platforms"): + kwargs = dict(mark.kwargs) + conds = list(mark.args) + ok, reason = _host_matches_platforms( + conds, + arch=kwargs.pop("arch", None), + arch_negate=kwargs.pop("arch_negate", False), + ) + if kwargs: + raise pytest.UsageError( + f"{item.nodeid}: platforms() got unexpected keyword(s) " + f"{sorted(kwargs)} — valid: arch, arch_negate" + ) + if not ok: + return reason + return None + + def pytest_configure(config): # noqa: D401 — pytest hook """Register markers used by hermetic conftest.""" config.addinivalue_line( @@ -1168,6 +1223,18 @@ def pytest_configure(config): # noqa: D401 — pytest hook "for tests that genuinely need real TTS synthesis and speaker " "playback — there are none in the default suite).", ) + config.addinivalue_line( + "markers", + "platforms(*specs, arch=None, arch_negate=False): run only on hosts " + "matching at least one spec — linux/macos/windows/posix/any, " + "'not X' negation, optional arch filter (e.g. arch='arm64')", + ) + config.addinivalue_line( + "markers", + "platforms(*specs, arch=None, arch_negate=False): run only on hosts " + "matching at least one spec — linux/macos/windows/posix/any, " + "'not X' negation, optional arch filter (e.g. arch='arm64')", + ) config.addinivalue_line( "markers", f"{_ALLOW_MACOS_KEYCHAIN_MARK}: allow a test to exercise the macOS " @@ -1185,7 +1252,7 @@ def pytest_configure(config): # noqa: D401 — pytest hook "dispatcher's memory guard to 'no data' — only for tests that " "exercise the guard itself with their own patched samples.", ) - # NOTE: linux_only / macos_only / windows_only are declared in + # NOTE: platforms("linux") / platforms("macos") / platforms("windows") are declared in # pyproject.toml's ``markers`` list, not here — they are part of the # project's public marker vocabulary (``pytest --markers``, and the CI # lanes select on them), whereas the marks above are conftest-internal @@ -1231,52 +1298,51 @@ def pytest_runtest_setup(item): ) -def _reject_multiple_os_marks(items): - """Fail collection when one test carries two host-OS markers. +def _reject_contradictory_platform_marks(items): + """Fail collection when one test carries two platforms() markers. - Every marker in ``_OS_MARKS`` skips on all but one host, so two of them - on the same item means it is skipped on *every* host — a test that never - runs anywhere, reported as green by both the Linux suite and the - tests-os lanes. That is the exact silent-coverage-loss the markers were - introduced to remove, so it is a hard collection error rather than a - warning nobody reads. + Two markers are ANDed by the gate, so a stacked pair is not always wrong + in principle — but the historic failure this guard exists for (a + module-level gate stacking with a per-test gate so the test is skipped + on every host while both lanes report green) is only diagnosable at + collection time. A test that needs a compound condition writes ONE + marker: platforms("linux", arch="arm64"). """ offenders = [] for item in items: - marks = sorted({m.name for m in item.iter_markers() if m.name in _OS_MARKS}) + marks = list(item.iter_markers("platforms")) if len(marks) > 1: - offenders.append(f" {item.nodeid}: {', '.join(marks)}") + offenders.append(f" {item.nodeid}: {len(marks)} platforms() marks") if offenders: raise pytest.UsageError( - "a test may carry at most one host-OS marker " - f"({', '.join(_OS_MARKS)}); these carry several and would be " - "skipped on every host:\n" + "\n".join(offenders) + "a test may carry at most one platforms() marker — combine the " + 'specs into one call (platforms("linux", arch="arm64") instead ' + "of stacking two markers); these carry several:\n" + + "\n".join(offenders) ) def pytest_collection_modifyitems(config, items): # noqa: D401 — pytest hook """Apply host-OS gating, then skip ``requires_wal`` where WAL is unusable. - OS gating: a test marked ``linux_only`` / ``macos_only`` / - ``windows_only`` runs only on that host. See the ``_OS_MARKS`` block - comment above for why these tests are skipped rather than run against a - patched ``sys.platform``. + OS gating: a test marked ``platforms(...)`` runs only on hosts its + specs match. See the platform-gating block comment above for why these + tests are skipped rather than run against a patched ``sys.platform``. WAL gating is cheaper and more honest than each test hand-rolling a version check: the reason string names the actual linked version so the skip is diagnosable rather than mysterious. """ - _reject_multiple_os_marks(items) + _reject_contradictory_platform_marks(items) - for mark_name, (is_host, label) in _OS_MARKS.items(): - if is_host(): - continue - skip_os = pytest.mark.skip( - reason=f"{label}-only test (marked {mark_name}); host is {sys.platform}" - ) - for item in items: - if item.get_closest_marker(mark_name) is not None: - item.add_marker(skip_os) + # platforms() gating: skip items whose specs exclude this host. The skip + # markers (not -m expressions) are the authoritative host filter on + # every lane, so a lane selects with plain ``-m platforms`` and lets the + # specs decide per-test. + for item in items: + reason = _platforms_gate_reason(item) + if reason is not None: + item.add_marker(pytest.mark.skip(reason=reason)) if _wal_is_usable(): return diff --git a/tests/cron/test_cron_script.py b/tests/cron/test_cron_script.py index 5c51134d25bb..c5ff375ce35a 100644 --- a/tests/cron/test_cron_script.py +++ b/tests/cron/test_cron_script.py @@ -133,7 +133,7 @@ def test_script_subprocess_env_sanitized(self, cron_env, monkeypatch): assert success is True assert output == "ABSENT" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_uv_venv_python_script_bypasses_launcher(self, cron_env, tmp_path, monkeypatch): # Windows-only: the fake ``sys.platform`` could not reproduce the # ``Scripts/python.exe`` launcher layout or the CREATE_NO_WINDOW diff --git a/tests/cron/test_cron_workdir.py b/tests/cron/test_cron_workdir.py index 013de5dca5ad..5eb4c0a8140d 100644 --- a/tests/cron/test_cron_workdir.py +++ b/tests/cron/test_cron_workdir.py @@ -45,6 +45,7 @@ def test_absolute_existing_dir_returns_resolved_str(self, tmp_path): result = _normalize_workdir(str(tmp_path)) assert result == str(tmp_path.resolve()) + @pytest.mark.platforms("linux") def test_tilde_expands(self, tmp_path, monkeypatch): from cron.jobs import _normalize_workdir monkeypatch.setenv("HOME", str(tmp_path)) diff --git a/tests/cron/test_file_permissions.py b/tests/cron/test_file_permissions.py index 20f57a82b74a..969e3d2fb1af 100644 --- a/tests/cron/test_file_permissions.py +++ b/tests/cron/test_file_permissions.py @@ -7,6 +7,10 @@ from pathlib import Path from unittest.mock import patch +import pytest + +pytestmark = pytest.mark.platforms("linux") + class TestCronFilePermissions(unittest.TestCase): """Verify cron files get secure permissions.""" diff --git a/tests/cron/test_media_delivery_parity.py b/tests/cron/test_media_delivery_parity.py index 8e6a6dd70d8a..b71344419644 100644 --- a/tests/cron/test_media_delivery_parity.py +++ b/tests/cron/test_media_delivery_parity.py @@ -25,6 +25,7 @@ MEDIA paths under different policy than the gateway's scheduled tick. """ +import json import os from pathlib import Path @@ -231,7 +232,7 @@ def test_bridge_helper_exists_and_applies_config(self, monkeypatch, tmp_path): (home / "config.yaml").write_text( "gateway:\n" " strict: true\n" - f" media_delivery_allow_dirs: [{str(allow_dir)!r}]\n" + f" media_delivery_allow_dirs: [{json.dumps(str(allow_dir))}]\n" " trust_recent_files: false\n" ) monkeypatch.setenv("HERMES_HOME", str(home)) diff --git a/tests/cron/test_script_claim_heartbeat.py b/tests/cron/test_script_claim_heartbeat.py index 8d81470576c7..b9fb0ca07288 100644 --- a/tests/cron/test_script_claim_heartbeat.py +++ b/tests/cron/test_script_claim_heartbeat.py @@ -10,6 +10,7 @@ import pytest +@pytest.mark.platforms("linux") def test_cancel_event_terminates_script_process_tree(tmp_path, monkeypatch): """Losing a fire claim must stop both the script and its descendants.""" import cron.scheduler as scheduler @@ -540,8 +541,10 @@ def run_body(_job, **kwargs): } monkeypatch.setattr(scheduler, "heartbeat_fire_claim", heartbeat) monkeypatch.setattr(scheduler, "_run_one_job_body", run_body) - monkeypatch.setattr(scheduler, "_RUN_CLAIM_HEARTBEAT_SECONDS", 0.01) - monkeypatch.setattr(scheduler, "_FIRE_CLAIM_HEARTBEAT_GRACE_SECONDS", 0.03) + # Intervals well above Windows' ~15ms timer resolution so the grace window + # reliably spans 3+ heartbeat ticks (10ms/30ms left only ~2 on win32). + monkeypatch.setattr(scheduler, "_RUN_CLAIM_HEARTBEAT_SECONDS", 0.05) + monkeypatch.setattr(scheduler, "_FIRE_CLAIM_HEARTBEAT_GRACE_SECONDS", 0.3) assert scheduler.run_one_job(job) is True assert calls >= 3 diff --git a/tests/gateway/test_73771_media_resend_dedup.py b/tests/gateway/test_73771_media_resend_dedup.py index d0490e8056d0..3be77f0eca6d 100644 --- a/tests/gateway/test_73771_media_resend_dedup.py +++ b/tests/gateway/test_73771_media_resend_dedup.py @@ -27,6 +27,7 @@ import time from types import SimpleNamespace from unittest.mock import AsyncMock +from urllib.parse import unquote import pytest @@ -495,7 +496,7 @@ async def test_streamed_explicit_media_resend_is_delivered(tmp_path, monkeypatch adapter.send_multiple_images.assert_awaited_once() sent_paths = [p for p, _cap in adapter.send_multiple_images.await_args.kwargs["images"]] - assert str(img) in sent_paths[0] + assert str(img) in unquote(sent_paths[0]) def test_stream_rescan_accepts_no_history_dedup_input(): diff --git a/tests/gateway/test_complete_path_at_filter.py b/tests/gateway/test_complete_path_at_filter.py index 8add45006bb8..5e764ca9f42c 100644 --- a/tests/gateway/test_complete_path_at_filter.py +++ b/tests/gateway/test_complete_path_at_filter.py @@ -223,6 +223,7 @@ def test_leading_slash_matches_the_bare_form(tmp_path, monkeypatch): assert slashed == bare +@pytest.mark.platforms("linux") def test_leading_slash_prefers_a_real_absolute_path(tmp_path, monkeypatch): """When the absolute reading resolves, it wins — no silent rewrite. diff --git a/tests/gateway/test_feishu.py b/tests/gateway/test_feishu.py index a923ea5f5d46..a834a45e5809 100644 --- a/tests/gateway/test_feishu.py +++ b/tests/gateway/test_feishu.py @@ -15,6 +15,9 @@ from gateway.platforms.base import ProcessingOutcome +import os as _os +_SYS_ENV = {k: _os.environ[k] for k in ("SYSTEMROOT", "USERPROFILE", "HOMEDRIVE", "HOMEPATH", "HOME") if k in _os.environ} + try: import lark_oapi _HAS_LARK_OAPI = True @@ -182,7 +185,7 @@ async def _fake_disconnect() -> None: self.assertIsNone(adapter._ws_client) - @patch.dict(os.environ, { + @patch.dict(os.environ, {**_SYS_ENV, "FEISHU_APP_ID": "cli_app", "FEISHU_APP_SECRET": "secret_app", }, clear=True) @@ -242,7 +245,7 @@ def is_closed(self): "extra_ua_tags must be ['channel'] to enable group event routing") - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_edit_message_falls_back_to_text_when_post_update_is_rejected(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -368,7 +371,7 @@ def _admits_group(adapter, message, sender_id, chat_id=""): class TestAdapterBehavior(unittest.TestCase): - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_build_event_handler_registers_reaction_and_card_processors(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -450,7 +453,7 @@ def builder(_encrypt_key, _verification_token): ], ) - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_bot_origin_reactions_are_dropped_to_avoid_feedback_loops(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -471,7 +474,7 @@ def test_bot_origin_reactions_are_dropped_to_avoid_feedback_loops(self): adapter._on_reaction_event("im.message.reaction.created_v1", data) run_threadsafe.assert_not_called() - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_user_reaction_with_managed_emoji_is_still_routed(self): # Operator-origin filter is enough to prevent feedback loops; we must # not additionally swallow user-origin reactions just because their @@ -529,7 +532,7 @@ def _build_reaction_adapter(self, *, msg_sender_id: str): adapter.get_chat_info = AsyncMock(return_value={"name": "Test Chat"}) return adapter - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_reaction_on_peer_bot_message_is_not_routed(self): # GET im/v1/messages sender for bot messages carries id=app_id; a peer # bot's message has a different app_id than ours, so it must be dropped. @@ -639,7 +642,7 @@ def test_default_group_policy_fallback_for_chats_without_explicit_rule(self): ) - @patch.dict(os.environ, {"FEISHU_GROUP_POLICY": "open"}, clear=True) + @patch.dict(os.environ, {**_SYS_ENV, "FEISHU_GROUP_POLICY": "open"}, clear=True) def test_group_message_matches_bot_name_when_only_name_available(self): """Name fallback engages when either side lacks an open_id. When BOTH the mention and the bot carry open_ids, IDs are authoritative — a @@ -697,7 +700,7 @@ def test_group_message_matches_bot_name_when_only_name_available(self): ) - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_extract_post_message_downloads_embedded_resources(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -734,7 +737,7 @@ def test_extract_post_message_downloads_embedded_resources(self): ) - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_extract_audio_message_downloads_and_caches(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -761,7 +764,7 @@ def test_extract_audio_message_downloads_and_caches(self): self.assertEqual(media_types, ["audio/ogg"]) - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_extract_text_message_starting_with_slash_becomes_command(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -799,7 +802,7 @@ def test_extract_text_message_starting_with_slash_becomes_command(self): self.assertEqual(event.message_type.value, "command") self.assertEqual(event.text, "/help test") - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_extract_text_file_injects_content(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -817,7 +820,7 @@ def test_extract_text_file_injects_content(self): self.assertIn("hello from feishu", text) self.assertIn("[Content of", text) - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_message_event_submits_to_adapter_loop(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -852,7 +855,7 @@ def _submit(coro, _loop): self.assertTrue(submit.called) - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_webhook_request_uses_same_message_dispatch_path(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -876,7 +879,7 @@ def test_webhook_request_uses_same_message_dispatch_path(self): self.assertEqual(response.status, 200) adapter._on_message_event.assert_called_once() - @patch.dict(os.environ, {"FEISHU_VERIFICATION_TOKEN": "expected-token"}, clear=True) + @patch.dict(os.environ, {**_SYS_ENV, "FEISHU_VERIFICATION_TOKEN": "expected-token"}, clear=True) def test_url_verification_requires_configured_verification_token(self): """url_verification must be rejected when token is set but mismatched. @@ -904,7 +907,7 @@ def test_url_verification_requires_configured_verification_token(self): self.assertEqual(response.status, 401) - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_process_inbound_message_uses_event_sender_identity_only(self): from gateway.config import PlatformConfig from gateway.platforms.base import MessageType @@ -954,6 +957,7 @@ def test_process_inbound_message_uses_event_sender_identity_only(self): @patch.dict( os.environ, { + **_SYS_ENV, "HERMES_FEISHU_TEXT_BATCH_MAX_MESSAGES": "2", }, clear=True, @@ -1001,7 +1005,7 @@ async def _run() -> None: self.assertEqual(first.text, "A\nB") self.assertEqual(second.text, "C") - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_media_batch_merges_rapid_photo_messages(self): from gateway.config import PlatformConfig from gateway.platforms.base import MessageEvent, MessageType @@ -1191,7 +1195,7 @@ def test_dedup_state_persists_across_adapter_restart(self): self.assertTrue(second._is_duplicate("om_same")) - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_send_document_reply_uses_thread_flag(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -1247,7 +1251,7 @@ async def _direct(func, *args, **kwargs): self.assertTrue(captured["request"].request_body.reply_in_thread) - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_send_uses_post_for_every_chunk_of_multi_chunk_markdown(self): """Regression for #26841: when a long Markdown message is split across multiple chunks, every chunk must go out as @@ -1307,7 +1311,7 @@ async def _direct(func, *args, **kwargs): self.assertEqual(msg_types, ["post", "post"]) - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_send_splits_fenced_code_blocks_into_separate_post_rows(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -1387,7 +1391,7 @@ def _make_adapter(self): return FeishuAdapter(PlatformConfig()) - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_hydration_populates_open_id_from_bot_info(self): adapter = self._make_adapter() adapter._client = Mock() @@ -1411,6 +1415,7 @@ def test_hydration_populates_open_id_from_bot_info(self): @patch.dict( os.environ, { + **_SYS_ENV, "FEISHU_BOT_OPEN_ID": "ou_env", "FEISHU_BOT_NAME": "Env Hermes", }, @@ -1446,7 +1451,7 @@ class TestPendingInboundQueue(unittest.TestCase): before or during adapter loop transitions must be queued for replay rather than silently dropped.""" - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_event_queued_when_loop_not_ready(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -1466,7 +1471,7 @@ def test_event_queued_when_loop_not_ready(self): # Drain scheduled flag set. self.assertTrue(adapter._pending_drain_scheduled) - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_drainer_replays_queued_events_when_loop_becomes_ready(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -1521,7 +1526,7 @@ def _make_adapter(self, encrypt_key: str = "") -> "FeishuAdapter": from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter - with patch.dict(os.environ, {"FEISHU_APP_ID": "cli", "FEISHU_APP_SECRET": "sec", "FEISHU_ENCRYPT_KEY": encrypt_key}, clear=True): + with patch.dict(os.environ, {**_SYS_ENV, "FEISHU_APP_ID": "cli", "FEISHU_APP_SECRET": "sec", "FEISHU_ENCRYPT_KEY": encrypt_key}, clear=True): return FeishuAdapter(PlatformConfig()) def test_signature_valid_passes(self): @@ -1576,7 +1581,7 @@ def test_webhook_request_rejects_oversized_chunked_body_while_reading(self): self.assertEqual(content.read_sizes, [_FEISHU_WEBHOOK_MAX_BODY_BYTES + 1]) - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_webhook_connect_requires_inbound_auth_secret(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -1589,7 +1594,7 @@ def test_webhook_connect_requires_inbound_auth_secret(self): ) self.assertFalse(asyncio.run(adapter.connect())) - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_webhook_loads_auth_secrets_from_platform_extra(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -1613,7 +1618,7 @@ def test_webhook_loads_auth_secrets_from_platform_extra(self): class TestDedupTTL(unittest.TestCase): """Tests for TTL-aware deduplication.""" - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_duplicate_within_ttl_is_rejected(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -1625,7 +1630,7 @@ def test_duplicate_within_ttl_is_rejected(self): self.assertTrue(adapter._is_duplicate("om_dup")) - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_load_tolerates_malformed_timestamp_values(self): """Regression #13632 — a non-numeric timestamp in the persisted dedup state must not crash adapter startup. The bad key is @@ -1636,7 +1641,7 @@ def test_load_tolerates_malformed_timestamp_values(self): from plugins.platforms.feishu.adapter import FeishuAdapter with tempfile.TemporaryDirectory() as temp_home: - with patch.dict(os.environ, {"HERMES_HOME": temp_home}, clear=True): + with patch.dict(os.environ, {**_SYS_ENV, "HERMES_HOME": temp_home}, clear=True): adapter = FeishuAdapter(PlatformConfig()) adapter._dedup_state_path.parent.mkdir(parents=True, exist_ok=True) adapter._dedup_state_path.write_text( @@ -1661,7 +1666,7 @@ class TestGroupMentionAtAll(unittest.TestCase): """Tests for @_all (Feishu @everyone) group mention routing.""" - @patch.dict(os.environ, {"FEISHU_GROUP_POLICY": "allowlist", "FEISHU_ALLOWED_USERS": "ou_allowed"}, clear=True) + @patch.dict(os.environ, {**_SYS_ENV, "FEISHU_GROUP_POLICY": "allowlist", "FEISHU_ALLOWED_USERS": "ou_allowed"}, clear=True) def test_at_all_still_requires_policy_gate(self): """@_all bypasses mention gating but NOT the allowlist policy.""" from gateway.config import PlatformConfig @@ -1682,7 +1687,7 @@ class TestSenderNameResolution(unittest.TestCase): """Tests for _resolve_sender_name_from_api (contact API + cache).""" - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_returns_cached_name_within_ttl(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -1694,7 +1699,7 @@ def test_returns_cached_name_within_ttl(self): result = asyncio.run(adapter._resolve_sender_name_from_api("ou_cached")) self.assertEqual(result, "Alice") - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_fetches_and_caches_name_from_api(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -1751,7 +1756,7 @@ def _fake_request(request): adapter._client = SimpleNamespace(request=_fake_request) return adapter, calls - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_returns_cached_bot_name_without_api_call(self): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter @@ -1764,7 +1769,7 @@ def test_returns_cached_bot_name_without_api_call(self): result = asyncio.run(adapter._resolve_sender_name_from_api("ou_peer", is_bot=True)) self.assertEqual(result, "Peer Bot") - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_fetches_and_caches_bot_name(self): adapter, calls = self._build_adapter_with_bots({"ou_peer": "Peer Bot"}) @@ -1850,7 +1855,7 @@ async def _direct(func, *args, **kwargs): return patch("plugins.platforms.feishu.adapter.asyncio.to_thread", side_effect=_direct) # ------------------------------------------------------------------ start - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_start_adds_typing_and_caches_reaction_id(self): adapter, tracker = self._build_adapter(next_reaction_id="r_typing") with self._patch_to_thread(): @@ -1860,7 +1865,7 @@ def test_start_adds_typing_and_caches_reaction_id(self): # --------------------------------------------------------------- complete - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_success_removes_typing_and_adds_nothing(self): adapter, tracker = self._build_adapter(next_reaction_id="r_typing") with self._patch_to_thread(): @@ -1872,7 +1877,7 @@ def test_success_removes_typing_and_adds_nothing(self): self.assertEqual(tracker.delete_calls, ["r_typing"]) self.assertNotIn("om_msg", adapter._pending_processing_reactions) - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_failure_removes_typing_then_adds_cross_mark(self): adapter, tracker = self._build_adapter(next_reaction_id="r_typing") with self._patch_to_thread(): @@ -1885,7 +1890,7 @@ def test_failure_removes_typing_then_adds_cross_mark(self): # ------------------------- delete failure: don't stack badges ----------- - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_delete_failure_on_failure_outcome_skips_cross_mark(self): # Removing Typing is best-effort — but if it fails, we must NOT # additionally add CrossMark, or the UI would show two contradictory diff --git a/tests/gateway/test_media_spaced_paths_and_history_dedupe.py b/tests/gateway/test_media_spaced_paths_and_history_dedupe.py index bc214dcdd896..a7bee1abd6de 100644 --- a/tests/gateway/test_media_spaced_paths_and_history_dedupe.py +++ b/tests/gateway/test_media_spaced_paths_and_history_dedupe.py @@ -87,6 +87,8 @@ def test_quoted_spaced_home_path_is_collected_in_delivery_form( monkeypatch, ): monkeypatch.setenv("HOME", str(tmp_path)) + # On Windows os.path.expanduser("~") reads USERPROFILE, not HOME. + monkeypatch.setenv("USERPROFILE", str(tmp_path)) history = [ { "role": "assistant", @@ -96,7 +98,7 @@ def test_quoted_spaced_home_path_is_collected_in_delivery_form( paths = _collect_history_media_paths(history) - assert str(tmp_path / "audio cache" / "old.ogg") in paths + assert os.path.expanduser("~/audio cache/old.ogg") in paths def test_empty_history_empty_set(self): assert _collect_history_media_paths([]) == set() diff --git a/tests/gateway/test_platform_base.py b/tests/gateway/test_platform_base.py index 4d12ec90dac2..41a34cf23754 100644 --- a/tests/gateway/test_platform_base.py +++ b/tests/gateway/test_platform_base.py @@ -561,6 +561,8 @@ def test_recency_trust_denies_system_paths_even_when_fresh(self, tmp_path, monke secret = ssh_dir / "id_rsa.txt" secret.write_bytes(b"-----BEGIN ...") # mtime = now monkeypatch.setenv("HOME", str(fake_home)) + # On Windows os.path.expanduser("~") reads USERPROFILE, not HOME. + monkeypatch.setenv("USERPROFILE", str(fake_home)) assert BasePlatformAdapter.validate_media_delivery_path(str(secret)) is None @@ -709,6 +711,8 @@ def test_root_home_deliverable_is_accepted(self, tmp_path, monkeypatch): doc = workdir / "proposal.docx" doc.write_bytes(b"PK\x03\x04") monkeypatch.setenv("HOME", str(fake_home)) + # On Windows os.path.expanduser("~") reads USERPROFILE, not HOME. + monkeypatch.setenv("USERPROFILE", str(fake_home)) # $HOME is itself on the denied-prefix list, mirroring /root. monkeypatch.setattr( "gateway.platforms.base._MEDIA_DELIVERY_DENIED_PREFIXES", @@ -758,6 +762,7 @@ def test_profile_scoped_cache_delivers_under_symlinked_root(self, tmp_path, monk ) + @pytest.mark.require_symlinks def test_root_home_workdir_symlink_to_credential_blocked(self, tmp_path, monkeypatch): """A symlink in the workdir pointing at a credential is rejected on its resolved target, even under the $HOME exception. @@ -782,6 +787,7 @@ def test_root_home_workdir_symlink_to_credential_blocked(self, tmp_path, monkeyp assert BasePlatformAdapter.validate_media_delivery_path(str(link)) is None +@pytest.mark.platforms("linux") class TestDockerContainerMediaPathTranslation: """MEDIA:/workspace (and configured mounts) must resolve to host paths.""" @@ -1321,6 +1327,7 @@ async def test_caption_is_preserved_in_fallback(self): assert self.SENSITIVE_PATH not in sent_text +@pytest.mark.platforms("linux") class TestDockerProfileSandboxMediaTranslation: """MEDIA from persistent Docker sandboxes must resolve to the host directory the profile's container actually bind-mounts (#93950). diff --git a/tests/gateway/test_post_stream_media_delivery.py b/tests/gateway/test_post_stream_media_delivery.py index bbb1f7c6fe78..964715fca7ae 100644 --- a/tests/gateway/test_post_stream_media_delivery.py +++ b/tests/gateway/test_post_stream_media_delivery.py @@ -14,6 +14,7 @@ from types import SimpleNamespace from unittest.mock import AsyncMock +from urllib.parse import unquote import pytest @@ -108,6 +109,6 @@ async def test_explicit_media_tag_still_delivers_post_stream(tmp_path, monkeypat adapter.send_multiple_images.assert_awaited_once() images_kwargs = adapter.send_multiple_images.await_args.kwargs assert images_kwargs["chat_id"] == "C123CHAN" - assert str(media_file) in images_kwargs["images"][0][0] + assert str(media_file) in unquote(images_kwargs["images"][0][0]) diff --git a/tests/gateway/test_replace_child_reap.py b/tests/gateway/test_replace_child_reap.py index 6409ea2d22d5..2700e289d3cc 100644 --- a/tests/gateway/test_replace_child_reap.py +++ b/tests/gateway/test_replace_child_reap.py @@ -64,6 +64,7 @@ def _fake_psutil(monkeypatch, *, wait_gone=None, wait_alive=None): return fake +@pytest.mark.platforms("linux") class TestReapGatewayChildren: def test_reaps_orphaned_children_sigterm_then_wait(self, monkeypatch): fake = _fake_psutil(monkeypatch) @@ -87,6 +88,7 @@ def test_survivors_of_sigterm_get_sigkill(self, monkeypatch): assert reaped == 1 +@pytest.mark.platforms("linux") class TestSnapshotGatewayChildren: def test_snapshot_walks_descendants_recursively(self, monkeypatch): fake = _fake_psutil(monkeypatch) diff --git a/tests/gateway/test_restart_drain.py b/tests/gateway/test_restart_drain.py index 0d643e657196..d5d6b855864f 100644 --- a/tests/gateway/test_restart_drain.py +++ b/tests/gateway/test_restart_drain.py @@ -265,7 +265,7 @@ async def _decoy(): ) -@pytest.mark.windows_only +@pytest.mark.platforms("windows") @pytest.mark.asyncio async def test_windows_detached_restart_scrubs_gateway_marker(monkeypatch, tmp_path): """Faking sys.platform="win32" on Linux could not reach the real Windows @@ -308,7 +308,7 @@ def fake_popen(cmd, **kwargs): assert kwargs["stderr"] is subprocess.DEVNULL -@pytest.mark.windows_only +@pytest.mark.platforms("windows") @pytest.mark.asyncio async def test_windows_detached_restart_watcher_keeps_console_python(monkeypatch, tmp_path): """The restart watcher must run sys.executable (console python) under the diff --git a/tests/gateway/test_run_progress_topics.py b/tests/gateway/test_run_progress_topics.py index 583c48d52dbe..14cbd66c140d 100644 --- a/tests/gateway/test_run_progress_topics.py +++ b/tests/gateway/test_run_progress_topics.py @@ -6,6 +6,7 @@ import time import types from types import SimpleNamespace +from urllib.parse import quote import pytest @@ -1288,7 +1289,7 @@ async def test_run_agent_queued_message_delivers_first_response_media(monkeypatc "image_batches": [ { "chat_id": "discord-thread", - "images": [(media_path.as_uri(), "")], + "images": [(f"file://{quote(str(media_path))}", "")], "metadata": {"thread_id": "discord-thread"}, } ], @@ -1329,7 +1330,7 @@ async def test_run_agent_queued_message_delivers_streamed_first_response_media( assert adapter.image_batches == [ { "chat_id": "discord-thread", - "images": [(media_path.as_uri(), "")], + "images": [(f"file://{quote(str(media_path))}", "")], "metadata": {"thread_id": "discord-thread"}, } ] diff --git a/tests/gateway/test_runtime_footer.py b/tests/gateway/test_runtime_footer.py index 1ca63a90c618..5f382964466f 100644 --- a/tests/gateway/test_runtime_footer.py +++ b/tests/gateway/test_runtime_footer.py @@ -36,10 +36,12 @@ def test_model_short_drops_vendor_prefix(model, expected): def test_home_relative_cwd_collapses_home(tmp_path, monkeypatch): monkeypatch.setenv("HOME", str(tmp_path)) + # On Windows os.path.expanduser("~") reads USERPROFILE, not HOME. + monkeypatch.setenv("USERPROFILE", str(tmp_path)) sub = tmp_path / "projects" / "hermes" sub.mkdir(parents=True) result = _home_relative_cwd(str(sub)) - assert result == "~/projects/hermes" + assert result == os.path.join("~", "projects", "hermes") # --------------------------------------------------------------------------- @@ -48,6 +50,8 @@ def test_home_relative_cwd_collapses_home(tmp_path, monkeypatch): def test_format_footer_all_fields(monkeypatch, tmp_path): monkeypatch.setenv("HOME", str(tmp_path)) + # On Windows os.path.expanduser("~") reads USERPROFILE, not HOME. + monkeypatch.setenv("USERPROFILE", str(tmp_path)) monkeypatch.setenv("TERMINAL_CWD", str(tmp_path / "projects" / "hermes")) (tmp_path / "projects" / "hermes").mkdir(parents=True) out = format_runtime_footer( @@ -57,7 +61,7 @@ def test_format_footer_all_fields(monkeypatch, tmp_path): cwd=None, # falls back to TERMINAL_CWD env var fields=("model", "context_pct", "cwd"), ) - assert out == "gpt-5.4 · 68% · ~/projects/hermes" + assert out == "gpt-5.4 · 68% · " + os.path.join("~", "projects", "hermes") def test_format_footer_skips_missing_context_length(): @@ -71,7 +75,7 @@ def test_format_footer_skips_missing_context_length(): # context_pct dropped silently; no "?%" artifact assert "%" not in out assert "gpt-5.4" in out - assert "/tmp/wd" in out + assert os.path.abspath("/tmp/wd") in out # --------------------------------------------------------------------------- @@ -214,6 +218,8 @@ def test_format_footer_latency_zero_renders_sub_second(): def test_format_footer_latency_in_field_order(monkeypatch, tmp_path): monkeypatch.setenv("HOME", str(tmp_path)) + # On Windows os.path.expanduser("~") reads USERPROFILE, not HOME. + monkeypatch.setenv("USERPROFILE", str(tmp_path)) out = format_runtime_footer( model="openai/gpt-5.4", context_tokens=68_000, @@ -256,6 +262,9 @@ def test_build_footer_line_threads_turn_seconds(monkeypatch): # --------------------------------------------------------------------------- _LEGACY_DEFAULT_FIELDS = ["model", "context_pct", "cwd"] +# An absolute path that survives _home_relative_cwd unchanged on every host +# (os.path.abspath("/var/data") -> "C:\\var\\data" on Windows, "/var/data" on POSIX). +_VAR_DATA = os.path.abspath("/var/data") def test_latency_not_in_default_fields(): @@ -275,10 +284,10 @@ def test_resolve_footer_config_default_fields_exclude_latency(): @pytest.mark.parametrize( "model,tokens,window,cwd,expected", [ - ("openai/gpt-5.4", 50_247, 1_000_000, "/var/data", "gpt-5.4 · 5% · /var/data"), - ("claude-opus-4-8", 68_000, 100_000, "/var/data", "claude-opus-4-8 · 68% · /var/data"), - ("m", 0, None, "/var/data", "m · /var/data"), - ("", 10, 100, "/var/data", "10% · /var/data"), + ("openai/gpt-5.4", 50_247, 1_000_000, _VAR_DATA, f"gpt-5.4 · 5% · {_VAR_DATA}"), + ("claude-opus-4-8", 68_000, 100_000, _VAR_DATA, f"claude-opus-4-8 · 68% · {_VAR_DATA}"), + ("m", 0, None, _VAR_DATA, f"m · {_VAR_DATA}"), + ("", 10, 100, _VAR_DATA, f"10% · {_VAR_DATA}"), ("m", 10, 100, "", "m · 10%"), ], ) @@ -311,9 +320,9 @@ def test_default_build_footer_line_ignores_turn_seconds(monkeypatch): model="openai/gpt-5.4", context_tokens=50_247, context_length=1_000_000, - cwd="/var/data", + cwd=_VAR_DATA, ) baseline = build_footer_line(**common) with_timing = build_footer_line(**common, turn_seconds=125.0) - assert baseline == "gpt-5.4 · 5% · /var/data" + assert baseline == f"gpt-5.4 · 5% · {_VAR_DATA}" assert with_timing == baseline diff --git a/tests/gateway/test_scale_to_zero.py b/tests/gateway/test_scale_to_zero.py index 0f604850c47b..622081151cc4 100644 --- a/tests/gateway/test_scale_to_zero.py +++ b/tests/gateway/test_scale_to_zero.py @@ -150,6 +150,7 @@ def serve(): return sock_path, t +@pytest.mark.platforms("linux") def test_suspend_self_posts_suspend_for_this_machine(tmp_path): captured: list[bytes] = [] sock_path, t = _fake_flaps(tmp_path, "200 OK", captured) @@ -164,6 +165,7 @@ def test_suspend_self_posts_suspend_for_this_machine(tmp_path): assert "Host: flaps\r\n" in request +@pytest.mark.platforms("linux") def test_suspend_self_non_2xx_is_false_not_raise(tmp_path): captured: list[bytes] = [] sock_path, t = _fake_flaps(tmp_path, "412 Precondition Failed", captured) @@ -171,6 +173,7 @@ def test_suspend_self_non_2xx_is_false_not_raise(tmp_path): t.join(timeout=5) +@pytest.mark.platforms("linux") def test_suspend_self_missing_socket_is_false_not_raise(tmp_path): # Fail-awake: a dead/absent flaps socket must never raise out of the watcher. assert suspend_self(_FLY_ENV, socket_path=str(tmp_path / "nope.sock")) is False diff --git a/tests/gateway/test_setup_feishu.py b/tests/gateway/test_setup_feishu.py index 6ae9fe228e91..70ad2216df1b 100644 --- a/tests/gateway/test_setup_feishu.py +++ b/tests/gateway/test_setup_feishu.py @@ -7,6 +7,9 @@ import os from unittest.mock import patch +import os as _os +_SYS_ENV = {k: _os.environ[k] for k in ("SYSTEMROOT", "USERPROFILE", "HOMEDRIVE", "HOMEPATH", "HOME") if k in _os.environ} + # --------------------------------------------------------------------------- # Helpers @@ -211,12 +214,12 @@ def _make_env_from_setup(self, dm_idx=0, group_idx=0): ) return env - @patch.dict(os.environ, {}, clear=True) + @patch.dict(os.environ, _SYS_ENV, clear=True) def test_qr_env_produces_valid_adapter_settings(self): """QR setup → adapter initializes with websocket mode.""" env = self._make_env_from_setup() - with patch.dict(os.environ, env, clear=True): + with patch.dict(os.environ, {**_SYS_ENV, **env}, clear=True): from gateway.config import PlatformConfig from plugins.platforms.feishu.adapter import FeishuAdapter adapter = FeishuAdapter(PlatformConfig()) diff --git a/tests/gateway/test_status.py b/tests/gateway/test_status.py index d17e47fa86b0..c6a9ab0bf32e 100644 --- a/tests/gateway/test_status.py +++ b/tests/gateway/test_status.py @@ -390,7 +390,7 @@ def test_live_process_is_stable_int(self): class TestTerminatePid: - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_force_uses_taskkill_on_windows(self, monkeypatch): # Faking _IS_WINDOWS on POSIX could not reproduce the real # CREATE_NO_WINDOW creationflags value that windows_hide_flags() @@ -438,7 +438,7 @@ def test_windows_force_refuses_reused_pid(self, monkeypatch): class TestScopedLocks: - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_file_lock_uses_high_offset(self, tmp_path, monkeypatch): # Faking _IS_WINDOWS on POSIX could not reproduce the msvcrt # byte-range locking path at all: msvcrt does not exist off Windows, @@ -1059,6 +1059,7 @@ def test_consume_still_rejects_start_time_mismatch_when_both_known( class TestReadProcessCmdlinePsFallback: """Tests for _read_process_cmdline falling back to ps on non-Linux.""" + @pytest.mark.platforms("linux") def test_ps_fallback_when_proc_unavailable(self, monkeypatch): monkeypatch.setattr(status.Path, "read_bytes", lambda self: (_ for _ in ()).throw(FileNotFoundError)) monkeypatch.setattr( diff --git a/tests/gateway/test_status_command.py b/tests/gateway/test_status_command.py index ec179816c767..5f2016e88925 100644 --- a/tests/gateway/test_status_command.py +++ b/tests/gateway/test_status_command.py @@ -3,6 +3,7 @@ from datetime import datetime import time +from pathlib import Path from types import SimpleNamespace from unittest.mock import AsyncMock, MagicMock, patch @@ -456,7 +457,13 @@ async def test_profile_command_reports_source_stamped_profile(monkeypatch, tmp_p result = await runner._handle_profile_command(event) assert "**Profile:** `milo`" in result - assert f"**Home:** `{profile_home}`" in result + # /profile reports the home via display_hermes_home(), which collapses a + # home-relative path to "~/…" (Windows tmp paths live under USERPROFILE). + try: + expected_home = "~/" + str(profile_home.relative_to(Path.home())) + except ValueError: + expected_home = str(profile_home) + assert f"**Home:** `{expected_home}`" in result # ── /context command tests ──────────────────────────────────────────────── diff --git a/tests/gateway/test_systemd_notify.py b/tests/gateway/test_systemd_notify.py index b0dea324cb32..a8dcd2f36064 100644 --- a/tests/gateway/test_systemd_notify.py +++ b/tests/gateway/test_systemd_notify.py @@ -7,6 +7,8 @@ import pytest +pytestmark = pytest.mark.platforms("linux") + @pytest.mark.skipif( not hasattr(socket, "AF_UNIX"), reason="Unix datagram sockets are unavailable" diff --git a/tests/gateway/test_tts_media_routing.py b/tests/gateway/test_tts_media_routing.py index fd3ed9ee443b..a25def6a1f63 100644 --- a/tests/gateway/test_tts_media_routing.py +++ b/tests/gateway/test_tts_media_routing.py @@ -12,6 +12,7 @@ import types from types import SimpleNamespace from unittest.mock import AsyncMock +from urllib.parse import quote, unquote import pytest @@ -240,7 +241,7 @@ async def test_queued_followup_delivery_strips_media_tag_from_text_and_sends_ima ) adapter.send_multiple_images.assert_awaited_once_with( chat_id="chat-1", - images=[(f"file://{media_file.as_posix()}", "")], + images=[(f"file://{quote(str(media_file))}", "")], metadata={"thread_id": "topic-1"}, ) @@ -290,7 +291,7 @@ async def test_queued_followup_delivery_reuses_routing_metadata_for_media( ) adapter.send_multiple_images.assert_awaited_once_with( chat_id="chat-1", - images=[(f"file://{media_file.as_posix()}", "")], + images=[(f"file://{quote(str(media_file))}", "")], metadata=routing_metadata, ) @@ -578,6 +579,6 @@ async def test_queued_resend_branch_delivers_media_and_preserves_protected_examp assert first_texts, f"expected queued resend of first response, got: {adapter.sent!r}" assert f"MEDIA:{media_file}" not in first_texts[0] assert "`MEDIA:/tmp/example.png`" in first_texts[0] - assert any(str(media_file) in img["image_path"] for img in adapter.images), ( + assert any(str(media_file) in unquote(img["image_path"]) for img in adapter.images), ( f"expected native image delivery via queued resend, got: {adapter.images!r}" ) diff --git a/tests/gateway/test_update_command.py b/tests/gateway/test_update_command.py index 69916e922960..72796cb96084 100644 --- a/tests/gateway/test_update_command.py +++ b/tests/gateway/test_update_command.py @@ -138,6 +138,7 @@ async def test_writes_pending_marker(self, tmp_path): @pytest.mark.asyncio + @pytest.mark.platforms("linux") async def test_fallback_when_no_setsid(self, tmp_path): """Falls back to start_new_session=True when setsid is not available.""" runner = _make_runner() diff --git a/tests/gateway/test_update_streaming.py b/tests/gateway/test_update_streaming.py index 1e1134b7ad05..940ab9646bd1 100644 --- a/tests/gateway/test_update_streaming.py +++ b/tests/gateway/test_update_streaming.py @@ -130,6 +130,7 @@ class TestUpdateCommandGatewayFlag: """Verify the gateway spawns hermes update --gateway.""" @pytest.mark.asyncio + @pytest.mark.platforms("linux") async def test_spawns_with_gateway_flag(self, tmp_path): """The spawned update command includes --gateway and PYTHONUNBUFFERED.""" runner = _make_runner() diff --git a/tests/gateway/test_whatsapp_connect.py b/tests/gateway/test_whatsapp_connect.py index 756282ba2949..32a22e947898 100644 --- a/tests/gateway/test_whatsapp_connect.py +++ b/tests/gateway/test_whatsapp_connect.py @@ -315,9 +315,9 @@ def poll_side_effect(): class TestKillPortProcess: """Verify _kill_port_process uses platform-appropriate commands.""" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_uses_netstat_and_taskkill_on_windows(self): - """``windows_only``: netstat/taskkill are Windows binaries. The old + """``platforms("windows")``: netstat/taskkill are Windows binaries. The old ``_IS_WINDOWS`` patch selected this branch on Linux, where neither exists, so the mocked argv was the only thing under test.""" from plugins.platforms.whatsapp.adapter import _kill_port_process @@ -352,7 +352,7 @@ def run_side_effect(cmd, **kwargs): for call in mock_run.call_args_list ) - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_refuses_taskkill_on_non_bridge_pid(self): """#89614 class: the netstat-scanned PID is a bare number — if the live process is not a node bridge, taskkill must never fire.""" @@ -378,7 +378,7 @@ def run_side_effect(cmd, **kwargs): ) - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_kills_only_listeners_on_linux(self): """POSIX path SIGTERMs only LISTENer PIDs (never clients) — the #43846 fix. @@ -387,7 +387,7 @@ def test_kills_only_listeners_on_linux(self): processes (a browser tab on the same port). The implementation now resolves listeners via ``_listener_pids_on_port`` and signals only those. - ``linux_only``: asserts the POSIX ``os.kill``/SIGTERM path, which is + ``platforms("linux")``: asserts the POSIX ``os.kill``/SIGTERM path, which is genuinely selected here without patching ``_IS_WINDOWS``. """ from plugins.platforms.whatsapp import adapter as wa @@ -404,7 +404,7 @@ def test_kills_only_listeners_on_linux(self): mock_listeners.assert_called_once_with(3000) assert kills == [(55555, signal.SIGTERM)] - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_non_bridge_listener_is_never_killed(self): """#89614 class: a listener that is not a node bridge is refused.""" from plugins.platforms.whatsapp import adapter as wa @@ -429,11 +429,11 @@ class TestHttpSessionLifecycle: """Verify persistent aiohttp.ClientSession is created and cleaned up.""" @pytest.mark.asyncio - @pytest.mark.windows_only + @pytest.mark.platforms("windows") async def test_disconnect_uses_taskkill_tree_on_windows(self): """Windows disconnect should target the bridge process tree, not just the parent PID. - ``windows_only``: ``taskkill /T`` is the Windows tree-kill primitive; + ``platforms("windows")``: ``taskkill /T`` is the Windows tree-kill primitive; on Linux the branch was reachable only by faking ``_IS_WINDOWS``. """ adapter = _make_adapter() diff --git a/tests/hermes_cli/test_agent_import.py b/tests/hermes_cli/test_agent_import.py index d00ae59f3ad3..1f910d37ddaf 100644 --- a/tests/hermes_cli/test_agent_import.py +++ b/tests/hermes_cli/test_agent_import.py @@ -643,6 +643,7 @@ def test_config_write_is_atomic_and_leaves_no_temp_files( leftovers = [p.name for p in hermes_home.glob(".tmp*")] assert leftovers == [] + @pytest.mark.require_symlinks def test_symlinked_config_stays_a_symlink( self, claude_tree, hermes_home, tmp_path, config_path): """A non-atomic ``write_text`` would follow it; ``os.replace`` would diff --git a/tests/hermes_cli/test_agent_plugins.py b/tests/hermes_cli/test_agent_plugins.py index 2c070faa5fad..bed441915eb3 100644 --- a/tests/hermes_cli/test_agent_plugins.py +++ b/tests/hermes_cli/test_agent_plugins.py @@ -67,11 +67,11 @@ def test_loads_manifest_skill_and_stdio_server(tmp_path: Path) -> None: assert package.skills[0].root == skill_dir.resolve() server = package.mcp_servers["worker"] assert server["command"] == "python" - assert server["args"] == [str(root.resolve() / "server.py"), "${UNKNOWN}"] + assert server["args"] == [str(root.resolve()) + "/server.py", "${UNKNOWN}"] assert server["cwd"] == str(root.resolve()) assert server["env"]["PLUGIN_ROOT"] == str(root.resolve()) assert server["env"]["PLUGIN_DATA"] == str((tmp_path / "data").resolve()) - assert server["env"]["CACHE"] == str((tmp_path / "data").resolve() / "cache") + assert server["env"]["CACHE"] == str((tmp_path / "data").resolve()) + "/cache" assert (tmp_path / "data").is_dir() @@ -133,6 +133,7 @@ def test_rejects_invalid_optional_skill_fields( assert package.skills == () +@pytest.mark.require_symlinks def test_symlink_escape_is_isolated_to_component(tmp_path: Path) -> None: root = tmp_path / "plugin" root.mkdir() diff --git a/tests/hermes_cli/test_auth_nous_provider.py b/tests/hermes_cli/test_auth_nous_provider.py index a1772fd32671..657f3e72aa71 100644 --- a/tests/hermes_cli/test_auth_nous_provider.py +++ b/tests/hermes_cli/test_auth_nous_provider.py @@ -889,6 +889,7 @@ def test_shared_store_seat_belt_refuses_real_home_under_pytest(monkeypatch): _nous_shared_store_path() +@pytest.mark.platforms("linux") def test_shared_store_write_and_read_roundtrip(shared_store_env): """Write → read must preserve refresh_token + OAuth URLs.""" from hermes_cli.auth import ( diff --git a/tests/hermes_cli/test_auth_ssl_macos.py b/tests/hermes_cli/test_auth_ssl_macos.py index 9d9abc1dbfeb..2320af0addc3 100644 --- a/tests/hermes_cli/test_auth_ssl_macos.py +++ b/tests/hermes_cli/test_auth_ssl_macos.py @@ -48,13 +48,13 @@ def real_bundle_file(tmp_path: Path) -> str: class TestDefaultVerify: - @pytest.mark.macos_only + @pytest.mark.platforms("macos") def test_returns_ssl_context_on_darwin(self): result = _default_verify() assert isinstance(result, ssl.SSLContext) - @pytest.mark.macos_only + @pytest.mark.platforms("macos") def test_darwin_falls_back_to_true_when_certifi_missing(self, monkeypatch): real_import = __import__ @@ -71,7 +71,7 @@ class TestResolveVerifyIntegration: """_resolve_verify should defer to _default_verify in the no-CA path.""" - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_no_ca_uses_default_verify_on_linux(self, monkeypatch): for var in ("HERMES_CA_BUNDLE", "SSL_CERT_FILE", "REQUESTS_CA_BUNDLE"): monkeypatch.delenv(var, raising=False) diff --git a/tests/hermes_cli/test_auth_store_lock_concurrent.py b/tests/hermes_cli/test_auth_store_lock_concurrent.py index 30788abe1704..584c637dd20a 100644 --- a/tests/hermes_cli/test_auth_store_lock_concurrent.py +++ b/tests/hermes_cli/test_auth_store_lock_concurrent.py @@ -39,7 +39,7 @@ def hermes_home(tmp_path, monkeypatch): return tmp_path -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_many_concurrent_lock_acquisitions_do_not_raise_permission_error(hermes_home): """CONCURRENCY threads race to acquire/release the same auth-store lock. diff --git a/tests/hermes_cli/test_backup.py b/tests/hermes_cli/test_backup.py index e78cfb57f42e..6f5424c8d7d0 100644 --- a/tests/hermes_cli/test_backup.py +++ b/tests/hermes_cli/test_backup.py @@ -1056,9 +1056,10 @@ def test_import_skips_profile_dirs_without_config(self, tmp_path, monkeypatch): from hermes_cli.backup import run_import run_import(args) - # Only valid profile should get a wrapper - assert (wrapper_dir / "valid").exists() - assert not (wrapper_dir / "empty").exists() + # Only valid profile should get a wrapper (Windows uses a .bat shim). + suffix = ".bat" if os.name == "nt" else "" + assert (wrapper_dir / f"valid{suffix}").exists() + assert not (wrapper_dir / f"empty{suffix}").exists() # --------------------------------------------------------------------------- @@ -1492,7 +1493,7 @@ def _spy(src, dst): monkeypatch.setattr(bk, "_safe_copy_db", _spy) snap_id = create_quick_snapshot(hermes_home=hermes_home) # The board db was copied via _safe_copy_db (not raw copy). - assert any(s.endswith("boards/work/kanban.db") for s in called["db"]), called["db"] + assert any(s.endswith(os.path.join("boards", "work", "kanban.db")) for s in called["db"]), called["db"] copy = hermes_home / "state-snapshots" / snap_id / "kanban" / "boards" / "work" / "kanban.db" rows = sqlite3.connect(str(copy)).execute("SELECT * FROM tasks").fetchall() assert rows == [("w1", "ship")] @@ -1822,6 +1823,7 @@ def test_backup_skips_external_paths_outside_home(self, tmp_path, monkeypatch): (outside / "leak.json").unlink() outside.rmdir() + @pytest.mark.platforms("linux") def test_import_restores_external_to_home_relative_location(self, tmp_path, monkeypatch): """_external/ members restore to ~/, not under HERMES_HOME, and credential-shaped files get 0600.""" diff --git a/tests/hermes_cli/test_backup_stability.py b/tests/hermes_cli/test_backup_stability.py index d461d507220b..bb88e430ee25 100644 --- a/tests/hermes_cli/test_backup_stability.py +++ b/tests/hermes_cli/test_backup_stability.py @@ -53,7 +53,7 @@ def test_atomic_output_keeps_previous_file_after_failure(tmp_path) -> None: def test_quick_snapshot_is_published_with_manifest(tmp_path, monkeypatch) -> None: home = tmp_path / ".hermes" home.mkdir() - (home / "config.yaml").write_text("model: {}\n", encoding="utf-8") + (home / "config.yaml").write_bytes(b"model: {}\n") published: list[tuple[Path, Path]] = [] from hermes_cli import backup diff --git a/tests/hermes_cli/test_browser_connect_default_chromium.py b/tests/hermes_cli/test_browser_connect_default_chromium.py index cdc29ad4c5ce..5dcdf90ff34a 100644 --- a/tests/hermes_cli/test_browser_connect_default_chromium.py +++ b/tests/hermes_cli/test_browser_connect_default_chromium.py @@ -8,6 +8,8 @@ import pytest +import posixpath + import hermes_cli.browser_connect as bc @@ -146,31 +148,37 @@ def test_missing_xdg_settings_fails_closed(self): class TestLinuxProfileDir: def _env(self, monkeypatch, home): - monkeypatch.setenv("HOME", str(home)) + # The code under test resolves the user's home via + # os.path.expanduser("~"), which reads HOME on POSIX but USERPROFILE + # on Windows — patch expanduser to a POSIX-FORM home so the + # Linux-target path resolution is exercised identically on every + # host (posixpath.join only inserts '/' between components; a + # backslash-drive home would leak host separators into the result). + monkeypatch.setattr(bc.os.path, "expanduser", lambda _p: home.as_posix()) monkeypatch.delenv("XDG_CONFIG_HOME", raising=False) def test_native_path_when_nothing_exists(self, tmp_path, monkeypatch): self._env(monkeypatch, tmp_path) - assert bc.real_profile_data_dir("chromium", "Linux") == str(tmp_path / ".config" / "chromium") + assert bc.real_profile_data_dir("chromium", "Linux") == posixpath.join(tmp_path.as_posix(), ".config", "chromium") def test_snap_chromium_profile_is_found(self, tmp_path, monkeypatch): self._env(monkeypatch, tmp_path) snap = tmp_path / "snap" / "chromium" / "common" / "chromium" snap.mkdir(parents=True) - assert bc.real_profile_data_dir("chromium", "Linux") == str(snap) + assert bc.real_profile_data_dir("chromium", "Linux") == snap.as_posix() def test_flatpak_chrome_profile_is_found(self, tmp_path, monkeypatch): self._env(monkeypatch, tmp_path) flatpak = tmp_path / ".var" / "app" / "com.google.Chrome" / "config" / "google-chrome" flatpak.mkdir(parents=True) - assert bc.real_profile_data_dir("chrome", "Linux") == str(flatpak) + assert bc.real_profile_data_dir("chrome", "Linux") == flatpak.as_posix() def test_native_profile_wins_when_present(self, tmp_path, monkeypatch): self._env(monkeypatch, tmp_path) native = tmp_path / ".config" / "BraveSoftware" / "Brave-Browser" native.mkdir(parents=True) (tmp_path / ".var" / "app" / "com.brave.Browser" / "config" / "BraveSoftware" / "Brave-Browser").mkdir(parents=True) - assert bc.real_profile_data_dir("brave", "Linux") == str(native) + assert bc.real_profile_data_dir("brave", "Linux") == native.as_posix() def test_xdg_config_home_is_honoured(self, tmp_path, monkeypatch): monkeypatch.setenv("HOME", str(tmp_path)) diff --git a/tests/hermes_cli/test_claw.py b/tests/hermes_cli/test_claw.py index d9472cd8c858..79891f3b2dd0 100644 --- a/tests/hermes_cli/test_claw.py +++ b/tests/hermes_cli/test_claw.py @@ -350,6 +350,7 @@ def test_empty_report(self, capsys): class TestDetectOpenclawProcesses: + @pytest.mark.platforms("linux") def test_returns_match_when_pgrep_finds_openclaw(self): with patch.object(claw_mod, "subprocess") as mock_subprocess: # systemd check misses, pgrep finds openclaw @@ -363,7 +364,7 @@ def test_returns_match_when_pgrep_finds_openclaw(self): assert "1234" in result[0] - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_returns_empty_on_windows_when_nothing_found(self): """Faking win32 picked the tasklist/powershell branch on a host that has neither; only a real Windows host resolves those executables. diff --git a/tests/hermes_cli/test_clipboard_text_write.py b/tests/hermes_cli/test_clipboard_text_write.py index 379ab5b798ee..5ef8ceae894e 100644 --- a/tests/hermes_cli/test_clipboard_text_write.py +++ b/tests/hermes_cli/test_clipboard_text_write.py @@ -17,7 +17,7 @@ def _completed(returncode=0): return subprocess.CompletedProcess(args=[], returncode=returncode) -@pytest.mark.macos_only +@pytest.mark.platforms("macos") def test_darwin_uses_pbcopy(): with patch.object(clip.subprocess, "run", return_value=_completed()) as run: assert clip.write_clipboard_text("hello") is True @@ -26,7 +26,7 @@ def test_darwin_uses_pbcopy(): assert run.call_args[1]["input"] == b"hello" -@pytest.mark.linux_only +@pytest.mark.platforms("linux") def test_linux_falls_through_backends_until_success(): calls = [] diff --git a/tests/hermes_cli/test_cmd_update.py b/tests/hermes_cli/test_cmd_update.py index 13a2beae7ad8..6ede05145a5e 100644 --- a/tests/hermes_cli/test_cmd_update.py +++ b/tests/hermes_cli/test_cmd_update.py @@ -10,6 +10,16 @@ from hermes_cli.main import cmd_update, PROJECT_ROOT +@pytest.fixture(autouse=True) +def _isolate_venv_holders(monkeypatch): + """The update flow's venv-holder guard sees the live gateway processes on + a dev machine and aborts with SystemExit 2 before reaching the branch + logic under test. Isolate it so the test exercises the intended path.""" + import hermes_cli.main as cli_main + + monkeypatch.setattr(cli_main, "_detect_venv_python_processes", lambda: []) + + def _make_run_side_effect(branch="main", verify_ok=True, commit_count="0"): """Build a side_effect function for subprocess.run that simulates git commands.""" diff --git a/tests/hermes_cli/test_config.py b/tests/hermes_cli/test_config.py index 968649e58a1d..d3756050fa05 100644 --- a/tests/hermes_cli/test_config.py +++ b/tests/hermes_cli/test_config.py @@ -35,21 +35,11 @@ class TestGetHermesHome: def test_default_path(self): + from hermes_constants import _get_platform_default_hermes_home with patch.dict(os.environ, {}, clear=False): os.environ.pop("HERMES_HOME", None) home = get_hermes_home() - if sys.platform == "win32": - # Windows default is %LOCALAPPDATA%\hermes — see - # hermes_constants._get_platform_default_hermes_home. - local_appdata = os.environ.get("LOCALAPPDATA", "").strip() - base = ( - Path(local_appdata) - if local_appdata - else Path.home() / "AppData" / "Local" - ) - assert home == base / "hermes" - else: - assert home == Path.home() / ".hermes" + assert home == _get_platform_default_hermes_home() class TestEnsureHermesHome: @@ -1315,7 +1305,7 @@ def test_windows_env_assignment_matching_is_case_insensitive(self, prefix): assert _env_line_defines_key(line, "PATH", is_windows=True) assert not _env_line_defines_key(line, "PATH", is_windows=False) - @pytest.mark.windows_only + @pytest.mark.platforms("windows") @pytest.mark.parametrize( "protected_key", [ diff --git a/tests/hermes_cli/test_container_boot.py b/tests/hermes_cli/test_container_boot.py index 5838ffdb4519..8134af5a9cf6 100644 --- a/tests/hermes_cli/test_container_boot.py +++ b/tests/hermes_cli/test_container_boot.py @@ -19,6 +19,8 @@ reconcile_profile_gateways, ) +pytestmark = pytest.mark.platforms("linux") + # --------------------------------------------------------------------------- # Fixtures + helpers diff --git a/tests/hermes_cli/test_dashboard_auth_gate.py b/tests/hermes_cli/test_dashboard_auth_gate.py index 40f1c000897d..de3240657675 100644 --- a/tests/hermes_cli/test_dashboard_auth_gate.py +++ b/tests/hermes_cli/test_dashboard_auth_gate.py @@ -161,7 +161,7 @@ def test_start_server_loopback_sets_auth_required_false(monkeypatch): # Force a fresh state to detect that start_server actually set it. web_server.app.state.auth_required = None web_server.start_server( - host="127.0.0.1", port=9119, + host="127.0.0.1", port=0, open_browser=False, allow_public=False, ) assert web_server.app.state.auth_required is False @@ -179,7 +179,7 @@ def test_start_server_insecure_public_no_longer_bypasses_gate(monkeypatch): web_server.app.state.auth_required = None with pytest.raises(SystemExit): web_server.start_server( - host="0.0.0.0", port=9119, + host="0.0.0.0", port=0, open_browser=False, allow_public=True, ) assert web_server.app.state.auth_required is True @@ -198,7 +198,7 @@ def test_start_server_public_without_insecure_records_auth_required(monkeypatch) web_server.app.state.auth_required = None with pytest.raises(SystemExit): web_server.start_server( - host="0.0.0.0", port=9119, + host="0.0.0.0", port=0, open_browser=False, allow_public=False, ) assert web_server.app.state.auth_required is True @@ -226,7 +226,7 @@ def test_start_server_gate_with_provider_proceeds_and_sets_proxy_headers(monkeyp try: web_server.app.state.auth_required = None web_server.start_server( - host="0.0.0.0", port=9119, + host="0.0.0.0", port=0, open_browser=False, allow_public=False, ) assert web_server.app.state.auth_required is True @@ -371,7 +371,7 @@ def test_start_server_loopback_public_url_enables_gate(monkeypatch): ) try: web_server.start_server( - host="127.0.0.1", port=9119, + host="127.0.0.1", port=0, open_browser=False, allow_public=False, ) assert web_server.app.state.auth_required is True @@ -404,7 +404,7 @@ def test_start_server_loopback_public_url_without_provider_fails_closed(monkeypa with pytest.raises(SystemExit, match=r"no auth providers"): web_server.start_server( - host="127.0.0.1", port=9119, + host="127.0.0.1", port=0, open_browser=False, allow_public=False, ) assert web_server.app.state.auth_required is True @@ -435,7 +435,7 @@ def test_loopback_public_url_fail_closed_message_is_actionable(monkeypatch): with pytest.raises(SystemExit) as exc: web_server.start_server( - host="127.0.0.1", port=9119, + host="127.0.0.1", port=0, open_browser=False, allow_public=False, ) msg = str(exc.value) diff --git a/tests/hermes_cli/test_dashboard_spawn_executable.py b/tests/hermes_cli/test_dashboard_spawn_executable.py index de1bae8485f0..d6921e9b19c9 100644 --- a/tests/hermes_cli/test_dashboard_spawn_executable.py +++ b/tests/hermes_cli/test_dashboard_spawn_executable.py @@ -14,6 +14,8 @@ from pathlib import Path from unittest.mock import patch +import pytest + import hermes_cli.web_server as web_server @@ -68,6 +70,7 @@ def test_no_venv_falls_back_to_sys_executable(self, tmp_path): ): assert web_server._dashboard_spawn_executable() == str(base_interp) + @pytest.mark.require_symlinks def test_venv_symlink_to_base_is_still_preferred_unresolved(self, tmp_path): """The Linux-standard layout: venv/bin/python is a SYMLINK to the base interpreter. The chooser must return the UNRESOLVED venv path — diff --git a/tests/hermes_cli/test_dashboard_unified_launch.py b/tests/hermes_cli/test_dashboard_unified_launch.py index 3bb122136ff8..faeee95da4eb 100644 --- a/tests/hermes_cli/test_dashboard_unified_launch.py +++ b/tests/hermes_cli/test_dashboard_unified_launch.py @@ -35,20 +35,42 @@ def test_profile_launch_reexecs_machine_dashboard(self, main_mod, monkeypatch): "hermes_cli.profiles.get_active_profile_name", lambda: "worker_x" ) monkeypatch.setattr(main_mod, "_dashboard_listening", lambda host, port: False) - execs = [] - - def fake_exec(exe, argv, env): - execs.append((exe, argv, env)) - raise SystemExit(0) # execvpe never returns - monkeypatch.setattr(main_mod.os, "execvpe", fake_exec) - - with pytest.raises(SystemExit): - main_mod.cmd_dashboard(_args()) + if sys.platform == "win32": + # Windows cannot truly replace the process, so cmd_dashboard + # re-execs via subprocess.Popen + sys.exit(code) instead of + # os.execvpe (which doesn't exist on Windows). + spawns = [] + + class _Done: + def wait(self): + return 0 + + monkeypatch.setattr( + main_mod.subprocess, + "Popen", + lambda argv, env=None, **kw: spawns.append((argv, env)) or _Done(), + ) + with pytest.raises(SystemExit): + main_mod.cmd_dashboard(_args()) + assert len(spawns) == 1 + argv, env = spawns[0] + else: + execs = [] + + def fake_exec(exe, argv, env): + execs.append((exe, argv, env)) + raise SystemExit(0) # execvpe never returns + + monkeypatch.setattr(main_mod.os, "execvpe", fake_exec) + + with pytest.raises(SystemExit): + main_mod.cmd_dashboard(_args()) + + assert len(execs) == 1 + exe, argv, env = execs[0] + assert exe == sys.executable - assert len(execs) == 1 - exe, argv, env = execs[0] - assert exe == sys.executable # Pinned to the default profile + launching profile preselected. assert "-p" in argv and argv[argv.index("-p") + 1] == "default" assert "--open-profile" in argv diff --git a/tests/hermes_cli/test_debug.py b/tests/hermes_cli/test_debug.py index 2a3d0206b142..220156246ff4 100644 --- a/tests/hermes_cli/test_debug.py +++ b/tests/hermes_cli/test_debug.py @@ -134,7 +134,7 @@ def test_keeps_first_line_when_truncation_on_boundary(self, hermes_home): # backward-reading loop so the truncation path actually fires. line = "A" * 99 + "\n" # 100 bytes per line num_lines = 200 # 20000 bytes - (hermes_home / "logs" / "agent.log").write_text(line * num_lines) + (hermes_home / "logs" / "agent.log").write_bytes((line * num_lines).encode("utf-8")) # max_bytes = 1000 = 100 * 10 → cut at byte 20000 - 1000 = 19000, # and byte 19000 - 1 is '\n'. Boundary hit → keep all 10 lines. diff --git a/tests/hermes_cli/test_dep_ensure.py b/tests/hermes_cli/test_dep_ensure.py index 17b37d1f5d63..21afa8f1c79d 100644 --- a/tests/hermes_cli/test_dep_ensure.py +++ b/tests/hermes_cli/test_dep_ensure.py @@ -3,11 +3,11 @@ import pytest -@pytest.mark.linux_only +@pytest.mark.platforms("linux") def test_find_install_script_from_checkout(tmp_path): """_find_install_script finds scripts/install.sh in a git checkout. - ``linux_only``: the POSIX arm picks ``install.sh`` + ``bash``, which is + ``platforms("linux")``: the POSIX arm picks ``install.sh`` + ``bash``, which is already what ``_IS_WINDOWS`` reports here — nothing needs faking. """ from hermes_cli.dep_ensure import _find_install_script @@ -98,9 +98,9 @@ def counting_find_agent_browser(*, validate=True): assert validate_calls == [True, False] -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_ensure_dependency_uses_powershell_on_windows(tmp_path): - """``windows_only``: the assertion is that we shell out to a real + """``platforms("windows")``: the assertion is that we shell out to a real PowerShell. Faking ``_IS_WINDOWS`` on Linux also required faking ``shutil.which`` into inventing a powershell.exe that isn't there.""" from hermes_cli.dep_ensure import ensure_dependency diff --git a/tests/hermes_cli/test_desktop_exe_integrity.py b/tests/hermes_cli/test_desktop_exe_integrity.py index 6e9d3dd06e97..3644e49a6811 100644 --- a/tests/hermes_cli/test_desktop_exe_integrity.py +++ b/tests/hermes_cli/test_desktop_exe_integrity.py @@ -138,13 +138,13 @@ def _windll(name, *args, **kwargs): return _windll -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_native_machine_reports_os_arch_not_process_arch(): """The #69179 WoA regression: x64 Python under ARM64 emulation must report ARM64 (the OS), not AMD64 (the process) — otherwise the integrity gate rejects the correct ARM64 rebuild. - ``windows_only``: the probe under test is a ``ctypes.WinDLL("kernel32")`` + ``platforms("windows")``: the probe under test is a ``ctypes.WinDLL("kernel32")`` call to ``IsWow64Process2``. A patched ``sys.platform`` only got the branch entered — there is no kernel32 to bind on Linux, so nothing below the branch (the HANDLE typing that #71218 was about) was ever executed. @@ -157,13 +157,13 @@ def test_native_machine_reports_os_arch_not_process_arch(): assert cli_main._windows_native_machine() == "ARM64" -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_expected_machines_prefers_user_runnable_api_over_arch_name(monkeypatch): """GetMachineTypeAttributes answers "can this host load PE machine X?" directly, so a WoA host that reports AMD64 everywhere else still accepts an ARM64 exe. - ``windows_only``: ``GetMachineTypeAttributes`` is a real kernel32 export + ``platforms("windows")``: ``GetMachineTypeAttributes`` is a real kernel32 export the fake host could not provide. """ import ctypes @@ -235,9 +235,9 @@ def test_rollback_restores_backup_and_keeps_corrupt_copy(tmp_path): -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_gate_fails_clearly_without_backup(tmp_path, capsys): - """``windows_only``: ``_ensure_desktop_exe_launchable`` is a documented + """``platforms("windows")``: ``_ensure_desktop_exe_launchable`` is a documented no-op off Windows, so the fake was the only reason the gate ran at all. """ desktop_dir, exe = _win_tree(tmp_path) @@ -274,13 +274,13 @@ def _ns(**kw): return argparse.Namespace(**defaults) -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_build_only_fails_when_pack_produces_corrupt_exe(tmp_path, monkeypatch, capsys): """The updater chain's contract: a rebuild whose Hermes.exe cannot launch must exit nonzero (so hermes-setup's retry-once kicks in) and must restore the previous working build instead of leaving the corrupt one. - ``windows_only``: the whole chain is Windows-gated — ``win-unpacked`` + ``platforms("windows")``: the whole chain is Windows-gated — ``win-unpacked`` candidate discovery in ``_desktop_packaged_executable`` and the integrity gate itself both short-circuit off Windows. """ diff --git a/tests/hermes_cli/test_doctor.py b/tests/hermes_cli/test_doctor.py index 9e37dc2cc2f9..4fd472b02f7d 100644 --- a/tests/hermes_cli/test_doctor.py +++ b/tests/hermes_cli/test_doctor.py @@ -318,6 +318,20 @@ def _run_doctor_and_capture(self, monkeypatch, tmp_path, provider=""): except Exception: pass + # Keep doctor from probing the AMBIENT gh CLI. A PATH lookup that + # resolves (any dev box has gh) makes doctor shell out to + # `gh auth status`, which both leaks the runner's real auth state + # into the assertion surface and -- on Windows -- dies with WinError 5 + # when gh resolves to a Store/MSIX reparse-point shim. The gh- + # specific doctor behaviors have their own dedicated tests below, + # which mock gh explicitly. + real_which = doctor_mod.shutil.which + monkeypatch.setattr( + doctor_mod.shutil, + "which", + lambda cmd: None if cmd == "gh" else real_which(cmd), + ) + import io, contextlib buf = io.StringIO() with contextlib.redirect_stdout(buf): diff --git a/tests/hermes_cli/test_doctor_journal_modes.py b/tests/hermes_cli/test_doctor_journal_modes.py index 55828de63134..e8f7773201df 100644 --- a/tests/hermes_cli/test_doctor_journal_modes.py +++ b/tests/hermes_cli/test_doctor_journal_modes.py @@ -142,7 +142,7 @@ def test_locked_database_is_still_readable(self, tmp_path): holder.close() @pytest.mark.skipif(os.name == "nt", reason="chmod is a no-op on Windows") - @pytest.mark.skipif(os.geteuid() == 0, reason="root ignores file permissions") + @pytest.mark.skipif(hasattr(os, "geteuid") and os.geteuid() == 0, reason="root ignores file permissions") def test_read_only_directory_is_still_readable(self, tmp_path): db = tmp_path / "state.db" _make_db(db, journal_mode="WAL") @@ -277,6 +277,7 @@ def test_an_untracked_lock_holder_does_not_block_the_probe(self, tmp_path): class TestUnreadableReason: + @pytest.mark.platforms("linux") def test_missing_file_keeps_the_os_error_text(self, tmp_path): reason = doctor._unreadable_reason(tmp_path / "gone.db") @@ -363,7 +364,8 @@ def test_lists_every_managed_database(self, tmp_path, capsys): assert "state.db is in WAL mode" in out assert "projects.db: rollback journal mode" in out assert "kanban.db: rollback journal mode" in out - assert "kanban/boards/myboard/kanban.db is in WAL mode" in out + board_rel = os.path.join("kanban", "boards", "myboard", "kanban.db") + assert f"{board_rel} is in WAL mode" in out def test_missing_databases_are_skipped(self, tmp_path, capsys): doctor._report_database_journal_modes(tmp_path, VULNERABLE) @@ -387,7 +389,7 @@ def test_locked_database_does_not_crash_or_block(self, tmp_path, capsys): assert "state.db: rollback journal mode" in out @pytest.mark.skipif(os.name == "nt", reason="chmod is a no-op on Windows") - @pytest.mark.skipif(os.geteuid() == 0, reason="root ignores file permissions") + @pytest.mark.skipif(hasattr(os, "geteuid") and os.geteuid() == 0, reason="root ignores file permissions") def test_unreadable_database_does_not_crash(self, tmp_path, capsys): db = tmp_path / "state.db" _make_db(db) diff --git a/tests/hermes_cli/test_ensure_acp_launcher.py b/tests/hermes_cli/test_ensure_acp_launcher.py index fac777858d4a..7e61965871ad 100644 --- a/tests/hermes_cli/test_ensure_acp_launcher.py +++ b/tests/hermes_cli/test_ensure_acp_launcher.py @@ -31,6 +31,7 @@ def fake_home(tmp_path, monkeypatch): +@pytest.mark.require_symlinks def test_does_not_follow_symlink_into_venv(fake_home, tmp_path): """#21454 failure mode: never write through a symlinked hermes-acp.""" (fake_home / "hermes").write_text("#!/bin/sh\n", encoding="utf-8") diff --git a/tests/hermes_cli/test_ensure_hermes_home_uid_34107.py b/tests/hermes_cli/test_ensure_hermes_home_uid_34107.py index ceb9bfb382a8..0937768ee1c1 100644 --- a/tests/hermes_cli/test_ensure_hermes_home_uid_34107.py +++ b/tests/hermes_cli/test_ensure_hermes_home_uid_34107.py @@ -27,6 +27,7 @@ class TestResolveHermesUidGid: + @pytest.mark.platforms("linux") def test_returns_parsed_values_when_both_set(self, monkeypatch): monkeypatch.setenv("HERMES_UID", "1000") monkeypatch.setenv("HERMES_GID", "911") @@ -36,11 +37,11 @@ def test_returns_parsed_values_when_both_set(self, monkeypatch): assert gid == 911 - # ``windows_only`` rather than ``skipif(sys.platform != "win32")``: the - # Windows CI job selects ``-m windows_only``, so a bare skipif would leave + # ``platforms("windows")`` rather than ``skipif(sys.platform != "win32")``: the + # Windows CI job selects ``-m platforms("windows")``, so a bare skipif would leave # this test skipped on Linux AND unselected on the Windows lane — dead on # every host. - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_returns_none_none(self, monkeypatch): monkeypatch.setenv("HERMES_UID", "1000") monkeypatch.setenv("HERMES_GID", "911") @@ -55,6 +56,7 @@ def test_windows_returns_none_none(self, monkeypatch): # --------------------------------------------------------------------------- +@pytest.mark.platforms("linux") class TestChownToHermesUid: def test_calls_os_chown_when_both_set(self, tmp_path, monkeypatch): monkeypatch.setenv("HERMES_UID", "1000") diff --git a/tests/hermes_cli/test_gateway.py b/tests/hermes_cli/test_gateway.py index 5defbb01cd4e..0425b2ba4d96 100644 --- a/tests/hermes_cli/test_gateway.py +++ b/tests/hermes_cli/test_gateway.py @@ -130,7 +130,7 @@ async def start_gateway(*, replace, verbosity): return json.loads(line.removeprefix("DIAG_JSON=")) -@pytest.mark.windows_only +@pytest.mark.platforms("windows") @pytest.mark.parametrize( ("marker", "expected_breakaway"), [("1", True), ("0", False), (None, None)], diff --git a/tests/hermes_cli/test_gateway_foreign_xdg_runtime.py b/tests/hermes_cli/test_gateway_foreign_xdg_runtime.py index 7aaf83e3b488..d5356e2ea9f7 100644 --- a/tests/hermes_cli/test_gateway_foreign_xdg_runtime.py +++ b/tests/hermes_cli/test_gateway_foreign_xdg_runtime.py @@ -16,6 +16,8 @@ import hermes_cli.gateway as gateway_cli +pytestmark = pytest.mark.platforms("linux") + def _eacces(self): raise PermissionError(13, "Permission denied", str(self)) diff --git a/tests/hermes_cli/test_gateway_linger.py b/tests/hermes_cli/test_gateway_linger.py index ac0efcfcd8f1..0a27048b0586 100644 --- a/tests/hermes_cli/test_gateway_linger.py +++ b/tests/hermes_cli/test_gateway_linger.py @@ -2,8 +2,12 @@ from types import SimpleNamespace +import pytest + import hermes_cli.gateway as gateway +pytestmark = pytest.mark.platforms("linux") + class TestEnsureLingerEnabled: def test_linger_already_enabled_via_file(self, monkeypatch, capsys): diff --git a/tests/hermes_cli/test_gateway_platform_gating.py b/tests/hermes_cli/test_gateway_platform_gating.py index e1c4dabe256c..e0ff6aa9bf89 100644 --- a/tests/hermes_cli/test_gateway_platform_gating.py +++ b/tests/hermes_cli/test_gateway_platform_gating.py @@ -16,7 +16,7 @@ class TestMatrixHiddenOnWindows: - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_matrix_present_on_linux(self): """Sanity: matrix is still in the picker on Linux. @@ -29,7 +29,7 @@ def test_matrix_present_on_linux(self): keys = {p["key"] for p in platforms} assert "matrix" in keys, "matrix must be available on Linux" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_matrix_absent_on_windows(self): """The gate itself: matrix must be dropped on a real Windows host. @@ -43,7 +43,7 @@ def test_matrix_absent_on_windows(self): keys = {p["key"] for p in platforms} assert "matrix" not in keys, "matrix must be hidden on Windows" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_other_platforms_unaffected_on_windows(self): """Gating must only drop matrix, not collateral damage.""" import hermes_cli.gateway as gateway_mod diff --git a/tests/hermes_cli/test_gateway_proc_fallback.py b/tests/hermes_cli/test_gateway_proc_fallback.py index b11fecad1f34..f9c2cf99829c 100644 --- a/tests/hermes_cli/test_gateway_proc_fallback.py +++ b/tests/hermes_cli/test_gateway_proc_fallback.py @@ -13,6 +13,8 @@ import hermes_cli.gateway as gateway_mod +pytestmark = pytest.mark.platforms("linux") + # --------------------------------------------------------------------------- # Helpers @@ -53,7 +55,7 @@ def _open(path, mode="r", **kwargs): # --------------------------------------------------------------------------- -@pytest.mark.linux_only +@pytest.mark.platforms("linux") class TestProcFallback: """_scan_gateway_pids reads /proc when available, skips ps. diff --git a/tests/hermes_cli/test_gateway_restart_loop.py b/tests/hermes_cli/test_gateway_restart_loop.py index a005b9c2e8cc..65fc7994b85e 100644 --- a/tests/hermes_cli/test_gateway_restart_loop.py +++ b/tests/hermes_cli/test_gateway_restart_loop.py @@ -556,6 +556,7 @@ def test_force_true_cannot_bypass_block(self, monkeypatch): assert result["exit_code"] == 1 assert "Blocked" in result["error"] + @pytest.mark.platforms("linux") def test_blocks_lifecycle_command_hidden_in_referenced_script( self, monkeypatch, tmp_path ): @@ -665,6 +666,7 @@ def execute(self, cmd, **kwargs): assert result["exit_code"] == 0 assert calls == ["hermes gateway restart"] + @pytest.mark.platforms("linux") def test_blocks_launchctl_submit_hidden_in_referenced_script( self, monkeypatch, tmp_path ): @@ -700,6 +702,7 @@ def execute(self, command, **kwargs): # pragma: no cover assert result["exit_code"] == 1 assert "referenced script" in result["error"] + @pytest.mark.platforms("linux") def test_blocks_executable_shebang_script(self, monkeypatch, tmp_path): import tools.terminal_tool as tt @@ -723,6 +726,7 @@ def test_launchctl_submit_parser_handles_shell_quoting(self, monkeypatch): assert result["exit_code"] == 1 assert "KeepAlive" in result["error"] + @pytest.mark.platforms("linux") def test_shell_option_with_value_still_scans_script(self, monkeypatch, tmp_path): import tools.terminal_tool as tt @@ -756,6 +760,7 @@ def execute(self, command, **kwargs): # pragma: no cover assert result["exit_code"] == 1 + @pytest.mark.platforms("linux") def test_nested_wrapper_script_is_scanned(self, monkeypatch, tmp_path): import tools.terminal_tool as tt @@ -776,6 +781,7 @@ def execute(self, command, **kwargs): # pragma: no cover assert result["exit_code"] == 1 + @pytest.mark.platforms("linux") def test_non_regular_referenced_script_fails_closed(self, monkeypatch, tmp_path): import tools.terminal_tool as tt @@ -809,6 +815,7 @@ def execute(self, command, **kwargs): assert result["exit_code"] == 0 assert calls == [command] + @pytest.mark.platforms("linux") def test_safe_referenced_script_passes_through(self, monkeypatch, tmp_path): import tools.terminal_tool as tt @@ -861,6 +868,7 @@ def execute(self, command, **kwargs): class TestLifecycleGuardModule: """Direct tests for cron.lifecycle_guard.check_gateway_lifecycle.""" + @pytest.mark.platforms("linux") def test_dot_operator_sourced_script_is_scanned(self, tmp_path): """`. ./script.sh` must reach the referenced-script scan. @@ -879,6 +887,7 @@ def test_dot_operator_sourced_script_is_scanned(self, tmp_path): is True ) + @pytest.mark.platforms("linux") def test_nul_padded_script_is_still_scanned(self, tmp_path): """A NUL byte in a *text* script must not disable the scan. @@ -897,6 +906,7 @@ def test_nul_padded_script_is_still_scanned(self, tmp_path): is True ) + @pytest.mark.platforms("linux") def test_source_builtin_sourced_script_is_scanned(self, tmp_path): """The `source` spelling must stay blocked (it already was).""" from cron.lifecycle_guard import ( @@ -922,6 +932,7 @@ def test_dot_operator_clean_script_not_blocked(self, tmp_path): is False ) + @pytest.mark.platforms("linux") def test_nul_padded_script_without_shebang_is_scanned(self, tmp_path): """Same bypass without a shebang — bash still runs it, so still scan. @@ -972,6 +983,7 @@ def test_macho_binary_is_not_scanned_as_script(self, tmp_path): is False ) + @pytest.mark.platforms("linux") def test_oversized_nul_bearing_text_still_fails_closed(self, tmp_path): """An oversized *text* script must keep failing closed. @@ -1193,6 +1205,7 @@ def test_shell_script_reference_walk_still_works(self, tmp_path): with pytest.raises(GatewayLifecycleBlocked): check_gateway_lifecycle("daily ops", str(script)) + @pytest.mark.require_symlinks def test_cloud_backed_symlink_fails_closed_without_opening_target( self, tmp_path, monkeypatch ): @@ -1235,6 +1248,7 @@ def reject_cloud_open(path, flags, *args, **kwargs): str(launcher) ) is True + @pytest.mark.platforms("linux") def test_third_party_cloudstorage_path_fails_closed_without_opening( self, tmp_path, monkeypatch ): @@ -1464,6 +1478,7 @@ def test_cron_guard_total_when_home_unresolvable(self, monkeypatch): # Defense 2 (chokepoint): cron.jobs.create_job blocks the AGENT model-tool path # --------------------------------------------------------------------------- +@pytest.mark.platforms("linux") class TestDotSourceIsScannedLikeSource: """`.` and `source` are the same POSIX builtin and must scan alike. @@ -1522,6 +1537,7 @@ def test_sourcing_a_clean_script_is_allowed(self, tmp_path): assert self._scan(f". {clean}", cwd=str(tmp_path)) is False +@pytest.mark.platforms("linux") class TestTransparentWrapperPrefixes: """`sudo`/`env`/`nohup`/... exec their argument tail, so the command that actually runs sits further right. Reading only the first token made the @@ -1841,6 +1857,7 @@ def _patch_env(self, monkeypatch, fake_env, *, inside_gateway: bool): lambda: inside_gateway, ) + @pytest.mark.platforms("linux") def test_remote_backend_script_read_uses_env_execute(self, monkeypatch, tmp_path): import tools.terminal_tool as tt diff --git a/tests/hermes_cli/test_gateway_service.py b/tests/hermes_cli/test_gateway_service.py index b3464258479e..a2572f62f3ad 100644 --- a/tests/hermes_cli/test_gateway_service.py +++ b/tests/hermes_cli/test_gateway_service.py @@ -346,7 +346,7 @@ def test_launchd_plist_omits_nofile_block_when_disabled(self, monkeypatch): class TestGatewayStopCleanup: - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_stop_only_kills_current_profile_by_default(self, tmp_path, monkeypatch): """Without --all, stop uses systemd (if available) and does NOT call the global kill_gateway_processes(). @@ -1120,7 +1120,7 @@ def test_launchd_restart_forces_kickstart_when_no_replacement_appears( - @pytest.mark.macos_only + @pytest.mark.platforms("macos") def test_gateway_restart_does_not_fallback_to_foreground_when_launchd_restart_fails(self, tmp_path, monkeypatch): """macOS-gated: the branch under test is ``elif is_macos() and get_launchd_plist_path().exists()``. Faking the platform flags on Linux diff --git a/tests/hermes_cli/test_gateway_windows.py b/tests/hermes_cli/test_gateway_windows.py index 9580964a93f0..c0c79a5622c8 100644 --- a/tests/hermes_cli/test_gateway_windows.py +++ b/tests/hermes_cli/test_gateway_windows.py @@ -32,7 +32,7 @@ def _boom(*args, **kwargs): -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_build_gateway_argv_keeps_venv_console_python_for_uv_venv(monkeypatch, tmp_path): """No pythonw / base-interpreter detour: the venv console python.exe is launched hidden (CREATE_NO_WINDOW) so descendants inherit its hidden @@ -79,7 +79,7 @@ def test_build_gateway_argv_keeps_venv_console_python_for_uv_venv(monkeypatch, t assert str(project) in env_overlay["PYTHONPATH"].split(gateway_windows.os.pathsep) -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_spawn_detached_marks_primary_breakaway_success(monkeypatch, tmp_path, caplog): """A successful breakaway spawn reports true without a warning.""" argv = ["python.exe", "-m", "hermes_cli.main", "gateway", "run"] @@ -111,7 +111,7 @@ def fake_popen(call_argv, **kwargs): assert not caplog.records -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_spawn_detached_warns_and_marks_no_breakaway_fallback( monkeypatch, tmp_path, caplog ): @@ -230,7 +230,7 @@ def fake_install_startup_entry(path: Path) -> Path: -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_elevated_gateway_command_uses_hidden_console_python(monkeypatch): """UAC handoff launches console python with SW_HIDE — a single hidden console, not console-less pythonw (#54220/#56747), and no visible diff --git a/tests/hermes_cli/test_gateway_wsl.py b/tests/hermes_cli/test_gateway_wsl.py index 6e7ff3793275..5f677d196b0d 100644 --- a/tests/hermes_cli/test_gateway_wsl.py +++ b/tests/hermes_cli/test_gateway_wsl.py @@ -59,7 +59,7 @@ def test_running(self, monkeypatch): class TestSupportsSystemdServicesWSL: """Test that supports_systemd_services() handles WSL correctly.""" - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_wsl_with_systemd(self, monkeypatch): """WSL + working systemd → True. @@ -74,7 +74,7 @@ def test_wsl_with_systemd(self, monkeypatch): monkeypatch.setattr(gateway, "_wsl_systemd_operational", lambda: True) assert gateway.supports_systemd_services() is True - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_termux_still_excluded(self, monkeypatch): """Termux → False regardless of WSL status. @@ -92,7 +92,7 @@ def test_termux_still_excluded(self, monkeypatch): class TestGatewayCommandWSLMessages: """Test that WSL users see appropriate guidance.""" - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_install_wsl_no_systemd(self, monkeypatch, capsys): """hermes gateway install on WSL without systemd shows guidance. @@ -122,7 +122,7 @@ def test_install_wsl_no_systemd(self, monkeypatch, capsys): assert "tmux" in out - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_status_wsl_running_manual(self, monkeypatch, capsys): """hermes gateway status on WSL with manual process shows WSL note. diff --git a/tests/hermes_cli/test_goal_gates.py b/tests/hermes_cli/test_goal_gates.py index 4dfc79478784..7be845ab5469 100644 --- a/tests/hermes_cli/test_goal_gates.py +++ b/tests/hermes_cli/test_goal_gates.py @@ -64,7 +64,10 @@ def test_run_gate_pass(): assert "hello" in out +@pytest.mark.platforms("linux") def test_run_gate_fail_captures_output(): + # POSIX shell syntax (`>&2`, `;`, `exit`) — cmd.exe (shell=True on Windows) + # doesn't parse it, so the gate "passes" instead of failing. passed, code, out = run_gate(GoalGate(command="echo broken >&2; exit 3")) assert passed is False assert code == 3 diff --git a/tests/hermes_cli/test_graphical_browser_detection.py b/tests/hermes_cli/test_graphical_browser_detection.py index a7c461a88ac8..ca250b5f2704 100644 --- a/tests/hermes_cli/test_graphical_browser_detection.py +++ b/tests/hermes_cli/test_graphical_browser_detection.py @@ -40,7 +40,7 @@ def _force_resolved_browser(monkeypatch, name: str): monkeypatch.setattr(webbrowser, "get", lambda *_a, **_kw: _FakeController(name)) -@pytest.mark.linux_only +@pytest.mark.platforms("linux") def test_headless_linux_no_display_refuses(monkeypatch): """The reported bug: headless Linux, no display server → don't auto-open. diff --git a/tests/hermes_cli/test_gui_command.py b/tests/hermes_cli/test_gui_command.py index a3480f96d10c..0649f36d2c75 100644 --- a/tests/hermes_cli/test_gui_command.py +++ b/tests/hermes_cli/test_gui_command.py @@ -99,7 +99,10 @@ def _make_packaged_executable(root: Path, monkeypatch) -> Path: return exe +@pytest.mark.platforms("linux") def test_gui_installs_packages_and_launches_desktop_app(tmp_path, monkeypatch): + # Exercises the npm-pack → packaged-exe launch path; Windows desktop is + # MSIX-only (Electron autoUpdater) and takes a different launch route. root = _make_desktop_tree(tmp_path) desktop_dir = root / "apps" / "desktop" monkeypatch.setattr(cli_main, "PROJECT_ROOT", root) @@ -308,7 +311,7 @@ def test_electron_dist_ok_on_this_host(): assert cli_main._electron_dist_ok(root) is True -@pytest.mark.linux_only +@pytest.mark.platforms("linux") def test_electron_dist_binary_basename_linux(): """``dist/electron`` on Linux — asserted against the live function. @@ -324,7 +327,7 @@ def test_electron_dist_binary_basename_linux(): ) -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_electron_dist_binary_basename_windows(): """``dist/electron.exe`` on Windows — the ``.exe`` suffix is the whole point.""" root = Path("C:/does-not-need-to-exist") @@ -333,7 +336,7 @@ def test_electron_dist_binary_basename_windows(): ) -@pytest.mark.macos_only +@pytest.mark.platforms("macos") def test_electron_dist_binary_basename_macos(): """``dist/Electron.app/Contents/MacOS/Electron`` on macOS. @@ -454,7 +457,7 @@ def test_desktop_macos_local_codesign_signs_native_binaries(tmp_path, monkeypatc -@pytest.mark.macos_only +@pytest.mark.platforms("macos") def test_relaunchable_fixup_falls_back_to_legacy_adhoc_on_failure(tmp_path, monkeypatch, capsys): """A failing stable sign must still leave a launchable (deep ad-hoc) bundle. @@ -462,7 +465,7 @@ def test_relaunchable_fixup_falls_back_to_legacy_adhoc_on_failure(tmp_path, monk with the fallback sign and strict verification succeeding, the fixup reports ``True`` per its documented contract. - ``macos_only``: the subject is ``codesign`` against a real ``.app`` bundle + ``platforms("macos")``: the subject is ``codesign`` against a real ``.app`` bundle layout (``exe.parents[2]``), which only the macOS packaged tree produces. """ root = _make_desktop_tree(tmp_path) @@ -747,7 +750,7 @@ def test_cmd_gui_setup_tcc_identity_exits_before_build(tmp_path, monkeypatch): -@pytest.mark.macos_only +@pytest.mark.platforms("macos") def test_relaunchable_fixup_stable_identity_never_touches_keychain(tmp_path, monkeypatch): """A successful stable-identity re-sign must NOT delete the safeStorage item. @@ -758,7 +761,7 @@ def test_relaunchable_fixup_stable_identity_never_touches_keychain(tmp_path, mon so after the first launch the keychain ACL already matches and deleting the item would destroy working credentials on every update. - ``macos_only``: the fixup no-ops on non-macOS (sys.platform guard), and + ``platforms("macos")``: the fixup no-ops on non-macOS (sys.platform guard), and the subject is codesign against a real ``.app`` bundle layout. """ root = _make_desktop_tree(tmp_path) @@ -784,7 +787,7 @@ def test_relaunchable_fixup_stable_identity_never_touches_keychain(tmp_path, mon assert not any("delete-generic-password" in c for c in calls) -@pytest.mark.macos_only +@pytest.mark.platforms("macos") def test_relaunchable_fixup_default_noconfig_success_never_touches_keychain(tmp_path, monkeypatch): """Default no-config path (identity == '-') must not delete the keychain item. @@ -792,7 +795,7 @@ def test_relaunchable_fixup_default_noconfig_success_never_touches_keychain(tmp_ ``desktop.macos_signing_identity`` configured, the fixup signs ad-hoc with identifier-pinned requirements and must leave the safeStorage item alone. - ``macos_only``: the fixup no-ops on non-macOS (sys.platform guard), and + ``platforms("macos")``: the fixup no-ops on non-macOS (sys.platform guard), and the subject is codesign against a real ``.app`` bundle layout. """ root = _make_desktop_tree(tmp_path) @@ -816,7 +819,7 @@ def test_relaunchable_fixup_default_noconfig_success_never_touches_keychain(tmp_ assert not any("delete-generic-password" in c for c in calls) -@pytest.mark.macos_only +@pytest.mark.platforms("macos") def test_relaunchable_fixup_legacy_adhoc_failure_never_touches_keychain(tmp_path, monkeypatch): """A failed fallback re-sign must preserve the keychain item (no deletion). @@ -827,7 +830,7 @@ def test_relaunchable_fixup_legacy_adhoc_failure_never_touches_keychain(tmp_path successor app/key identity. The fixup must check the codesign result, run strict verification, and leave the keychain untouched on failure. - ``macos_only``: the fixup no-ops on non-macOS (sys.platform guard), and + ``platforms("macos")``: the fixup no-ops on non-macOS (sys.platform guard), and the subject is codesign against a real ``.app`` bundle layout. """ root = _make_desktop_tree(tmp_path) @@ -866,7 +869,7 @@ def boom(*a, **kw): assert not any("delete-generic-password" in c for c in calls) -@pytest.mark.macos_only +@pytest.mark.platforms("macos") def test_relaunchable_fixup_legacy_adhoc_success_still_verifies_and_never_deletes(tmp_path, monkeypatch): """A successful fallback re-sign runs strict verification, no deletion. @@ -876,7 +879,7 @@ def test_relaunchable_fixup_legacy_adhoc_success_still_verifies_and_never_delete ("Always Allow" updates the ACL partition list and preserves the key); deletion is not. - ``macos_only``: the fixup no-ops on non-macOS (sys.platform guard), and + ``platforms("macos")``: the fixup no-ops on non-macOS (sys.platform guard), and the subject is codesign against a real ``.app`` bundle layout. """ root = _make_desktop_tree(tmp_path) @@ -919,7 +922,7 @@ def boom(*a, **kw): # --- Linux launcher entry registration ------------------------------------ -@pytest.mark.linux_only +@pytest.mark.platforms("linux") def test_gui_registers_linux_desktop_entry_before_launch(tmp_path, monkeypatch): """`hermes desktop` gives the app a launcher presence on Linux.""" root = _make_desktop_tree(tmp_path) @@ -945,7 +948,7 @@ def test_gui_registers_linux_desktop_entry_before_launch(tmp_path, monkeypatch): assert registered == [root] -@pytest.mark.linux_only +@pytest.mark.platforms("linux") def test_gui_launches_even_when_desktop_entry_install_fails(tmp_path, monkeypatch): """Launcher plumbing is a convenience — it must never block the app.""" root = _make_desktop_tree(tmp_path) @@ -971,7 +974,7 @@ def boom(_project_root): assert mock_run.call_args.args[0] == [str(packaged_exe)] -@pytest.mark.macos_only +@pytest.mark.platforms("macos") def test_gui_skips_desktop_entry_off_linux(tmp_path, monkeypatch): root = _make_desktop_tree(tmp_path) monkeypatch.setattr(cli_main, "PROJECT_ROOT", root) @@ -1036,6 +1039,7 @@ def test_desktop_launch_options_ozone_hint_defaults_auto(): assert cli_main._desktop_launch_options()[3] == "auto" +@pytest.mark.platforms("linux") def test_gui_bridges_ozone_hint_to_launch_env(tmp_path, monkeypatch): """COSMIC HUD: ``desktop.ozone_platform_hint: x11`` sets ``ELECTRON_OZONE_PLATFORM_HINT`` on the launched Electron process.""" @@ -1134,7 +1138,7 @@ def test_detect_linux_password_store_none_when_no_keychain(monkeypatch): assert cli_main._detect_linux_password_store() is None -@pytest.mark.linux_only +@pytest.mark.platforms("linux") def test_gui_linux_packaged_launch_bridges_detected_password_store(tmp_path, monkeypatch): _clear_keychain_env(monkeypatch) root = _make_desktop_tree(tmp_path) @@ -1160,7 +1164,7 @@ def test_gui_linux_packaged_launch_bridges_detected_password_store(tmp_path, mon assert launch_env["HERMES_DESKTOP_PASSWORD_STORE"] == "gnome-libsecret" -@pytest.mark.linux_only +@pytest.mark.platforms("linux") def test_gui_linux_source_launch_bridges_detected_password_store(tmp_path, monkeypatch): _clear_keychain_env(monkeypatch) root = _make_desktop_tree(tmp_path) @@ -1184,7 +1188,7 @@ def test_gui_linux_source_launch_bridges_detected_password_store(tmp_path, monke assert launch_env["HERMES_DESKTOP_PASSWORD_STORE"] == "kwallet6" -@pytest.mark.linux_only +@pytest.mark.platforms("linux") def test_gui_config_password_store_skips_detection(tmp_path, monkeypatch): _clear_keychain_env(monkeypatch) root = _make_desktop_tree(tmp_path) @@ -1212,7 +1216,7 @@ def test_gui_config_password_store_skips_detection(tmp_path, monkeypatch): assert launch_env["HERMES_DESKTOP_PASSWORD_STORE"] == "kwallet6" -@pytest.mark.linux_only +@pytest.mark.platforms("linux") def test_gui_explicit_password_store_env_wins_over_config_and_detection(tmp_path, monkeypatch): _clear_keychain_env(monkeypatch) monkeypatch.setenv("HERMES_DESKTOP_PASSWORD_STORE", "basic") @@ -1241,7 +1245,7 @@ def test_gui_explicit_password_store_env_wins_over_config_and_detection(tmp_path assert launch_env["HERMES_DESKTOP_PASSWORD_STORE"] == "basic" -@pytest.mark.macos_only +@pytest.mark.platforms("macos") def test_gui_password_store_bridge_is_linux_only(tmp_path, monkeypatch): _clear_keychain_env(monkeypatch) root = _make_desktop_tree(tmp_path) diff --git a/tests/hermes_cli/test_hooks_cli.py b/tests/hermes_cli/test_hooks_cli.py index 0ffb6ad04789..f6f1340255bc 100644 --- a/tests/hermes_cli/test_hooks_cli.py +++ b/tests/hermes_cli/test_hooks_cli.py @@ -79,6 +79,7 @@ def test_shows_configured_and_consent_status(self, tmp_path): # ── test ────────────────────────────────────────────────────────────────── +@pytest.mark.platforms("linux") class TestHooksTest: def test_synthetic_payload_matches_production_shape(self, tmp_path): """`hermes hooks test` must feed the script stdin in the same diff --git a/tests/hermes_cli/test_install_cua_driver.py b/tests/hermes_cli/test_install_cua_driver.py index 9740fc5c9fe2..0b6eaf73297a 100644 --- a/tests/hermes_cli/test_install_cua_driver.py +++ b/tests/hermes_cli/test_install_cua_driver.py @@ -144,6 +144,7 @@ def test_non_upgrade_on_unsupported_platform_warns(self): assert tools_config.install_cua_driver(upgrade=False) is False warn.assert_called() + @pytest.mark.platforms("linux") def test_upgrade_with_binary_present_runs_installer(self): from hermes_cli import tools_config @@ -163,6 +164,7 @@ def test_upgrade_with_binary_present_runs_installer(self): kwargs = runner.call_args.kwargs assert kwargs.get("verbose") is False + @pytest.mark.platforms("linux") def test_upgrade_without_binary_runs_installer(self): from hermes_cli import tools_config @@ -173,9 +175,9 @@ def test_upgrade_without_binary_runs_installer(self): assert tools_config.install_cua_driver(upgrade=True) is True runner.assert_called_once() - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_quiet_refresh_prints_single_contextual_progress_line(self): - """``linux_only``: reaches Popen through the POSIX download-then-exec + """``platforms("linux")``: reaches Popen through the POSIX download-then-exec branch, which this lane takes for real.""" from unittest.mock import MagicMock @@ -207,9 +209,9 @@ def test_quiet_refresh_prints_single_contextual_progress_line(self): "→ Refreshing cua-driver (Computer Use)..." ) - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_quiet_refresh_can_suppress_progress_line(self): - """``linux_only``: same POSIX Popen path as the test above.""" + """``platforms("linux")``: same POSIX Popen path as the test above.""" from unittest.mock import MagicMock from hermes_cli import tools_config @@ -279,6 +281,7 @@ def test_quiet_refresh_closes_stdin_and_honors_custom_timeout(self): assert popen.call_args.kwargs["stdin"] is subprocess.DEVNULL fake_proc.communicate.assert_called_once_with(timeout=120) + @pytest.mark.platforms("linux") def test_upgrade_can_suppress_installer_progress(self): from hermes_cli import tools_config @@ -342,11 +345,11 @@ def test_fresh_install_non_writable_install_target_skips_install(self): for call in info.call_args_list ) - @pytest.mark.macos_only + @pytest.mark.platforms("macos") def test_install_target_writability_is_probed_for_real_on_macos(self): """The ``_cua_install_target_writable`` seam the two tests above patch. - ``macos_only``: ``/Applications`` is the only install target Hermes + ``platforms("macos")``: ``/Applications`` is the only install target Hermes checks, and the probe short-circuits to True on every other platform — so this is the one host where the real filesystem answer means anything. @@ -456,6 +459,7 @@ def test_missing_explicit_override_does_not_install_standard_driver( runner.assert_not_called() + @pytest.mark.platforms("linux") def test_non_upgrade_without_binary_runs_installer(self): from hermes_cli import tools_config @@ -559,7 +563,7 @@ def test_confirmed_update_runs_installer_bounded(self): runner.assert_called_once() assert runner.call_args.kwargs["installer_timeout"] == 120 - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_incompatible_driver_defers_interactive_repair(self): incompatible = { "ready": False, @@ -579,7 +583,7 @@ def test_windows_incompatible_driver_defers_interactive_repair(self): for call in info.call_args_list ) - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_missing_binary_defers_interactive_install(self): """Driver enabled but never installed (or wiped by a failed install): the automatic update must not launch install.ps1 either — this path @@ -706,14 +710,15 @@ def fake_run(cmd, **kw): cua_backend.cua_driver_update_check() return captured.get("timeout") - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_default_is_generous(self): - """``windows_only``: the 25s default exists because a real Windows + """``platforms("windows")``: the 25s default exists because a real Windows first-spawn is delayed by Defender/SmartScreen scanning — a faked platform asserted the constant, never the host it is chosen for. """ assert self._captured_timeout() == 25.0 + @pytest.mark.platforms("linux") def test_posix_default_unchanged(self): # Unmarked: the POSIX default is what this (Linux) host already picks, # so no platform faking is involved. @@ -880,9 +885,9 @@ def test_no_lock_is_noop(self, tmp_path): class TestWindowsStaleInstallLockClearDispatch: - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_branch_uses_file_lock_probe(self): - """``windows_only``: which lock protocol applies IS the host fact under + """``platforms("windows")``: which lock protocol applies IS the host fact under test — on Linux the faked platform asserted the dispatch and skipped the ``.install.lock.d`` directory that really exists here. """ @@ -896,10 +901,10 @@ def test_windows_branch_uses_file_lock_probe(self): clear_windows.assert_called_once_with() -# ``windows_only`` rather than ``skipif(sys.platform != "win32")``: the -# dedicated Windows CI job selects ``-m windows_only``, so a bare skipif left +# ``platforms("windows")`` rather than ``skipif(sys.platform != "win32")``: the +# dedicated Windows CI job selects ``-m platforms("windows")``, so a bare skipif left # these real-CreateFileW tests running on no host at all. -@pytest.mark.windows_only +@pytest.mark.platforms("windows") class TestWindowsStaleInstallLockClear: def _make_lock(self, tmp_path): import os @@ -970,11 +975,11 @@ class TestInstallerTimeoutKillsProcessGroup: The POSIX cases drop the old ``platform.system`` → "Linux" fake: this lane IS Linux, so the branch is selected for real. The Windows cases are - ``windows_only`` — the psutil tree-kill only runs when ``is_windows``, and + ``platforms("windows")`` — the psutil tree-kill only runs when ``is_windows``, and on Linux the fake picked that branch on a host with no such process model. """ - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_timeout_kills_process_group_and_returns_false(self): import signal import subprocess @@ -1021,7 +1026,7 @@ def test_timeout_ceiling_exceeds_upstream_lock_window(self): # lock; our ceiling must give that window room to complete. assert tools_config._CUA_INSTALLER_TIMEOUT > tools_config._CUA_LOCK_STALE_AFTER - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_installer_runs_in_new_session_on_posix(self): from unittest.mock import MagicMock from hermes_cli import tools_config @@ -1045,7 +1050,7 @@ def fake_popen(*args, **kwargs): assert captured.get("start_new_session") is True - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_timeout_kills_descendants_and_parent(self): import subprocess from unittest.mock import MagicMock @@ -1082,7 +1087,7 @@ def test_windows_timeout_kills_descendants_and_parent(self): == tools_config._CUA_INSTALLER_DRAIN_GRACE ) - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_tree_enumeration_failure_falls_back_to_direct_kill(self): import psutil import subprocess @@ -1141,9 +1146,9 @@ def test_drain_grace_is_short_relative_to_the_run_ceiling(self): < tools_config._CUA_INSTALLER_TIMEOUT / 10 ) - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_post_kill_drain_passes_a_deadline(self): - """``linux_only``: reaches the timeout handler through the real POSIX + """``platforms("linux")``: reaches the timeout handler through the real POSIX ``killpg`` branch, so the drain under test is the one this lane runs. """ import subprocess @@ -1175,7 +1180,7 @@ def test_post_kill_drain_passes_a_deadline(self): ) assert drain_timeout == tools_config._CUA_INSTALLER_DRAIN_GRACE - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_verbose_path_drains_under_the_same_deadline(self): """The streaming install has the same handler and had the same hole. @@ -1212,7 +1217,7 @@ def test_verbose_path_drains_under_the_same_deadline(self): ) assert drain_timeout == tools_config._CUA_INSTALLER_DRAIN_GRACE - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_unkillable_elevated_descendant_does_not_stall_the_drain(self): """The reported scenario, with the kill refused exactly where it is. @@ -1258,7 +1263,7 @@ def test_unkillable_elevated_descendant_does_not_stall_the_drain(self): ) assert drain_timeout == tools_config._CUA_INSTALLER_DRAIN_GRACE - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_drain_that_times_out_still_surfaces_the_run_timeout(self): """A guardrail, not a regression test — it passes without the fix too. @@ -1296,14 +1301,14 @@ def test_drain_that_times_out_still_surfaces_the_run_timeout(self): assert "timed out after" in warned -@pytest.mark.linux_only +@pytest.mark.platforms("linux") class TestInstallerNoShell: """The POSIX installer path must not use shell=True or command substitution: the script is downloaded to a mkstemp file and exec'd as a plain argv list (salvage of #34974's intent, without the fixed /tmp path TOCTOU that PR introduced). - ``linux_only``: the download-then-exec argv IS the POSIX branch, and this + ``platforms("linux")``: the download-then-exec argv IS the POSIX branch, and this lane already takes it — the old ``platform.system`` → "Linux" fake was asserting a branch the host had already selected. """ @@ -1463,12 +1468,12 @@ def test_missing_latest_version_falls_back_unpinned(self): assert runner.call_args.kwargs.get("pin_version") is None -@pytest.mark.linux_only +@pytest.mark.platforms("linux") class TestRunInstallerPinEnv: """_run_cua_driver_installer(pin_version=...) exports CUA_DRIVER_RS_VERSION into the installer child env; unpinned runs leave it untouched. - ``linux_only``: the helper reaches Popen through the POSIX + ``platforms("linux")``: the helper reaches Popen through the POSIX download-then-exec branch, which this lane takes for real — no ``platform.system`` fake needed. The pin itself is host-agnostic (``TestConfirmedVersionPinning`` covers the caller side unmarked). @@ -1515,9 +1520,9 @@ def test_no_pin_leaves_env_untouched(self): class TestWindowsAutostartRepair: - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_existing_task_skips_elevated_powershell_repair(self): - """``windows_only``: ``_repair_cua_driver_autostart_windows`` returns + """``platforms("windows")``: ``_repair_cua_driver_autostart_windows`` returns True unconditionally off Windows, so only the fake made the schtasks probe run at all. """ @@ -1541,9 +1546,9 @@ def fake_run(cmd, **kwargs): ] which.assert_not_called() - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_installer_runs_autostart_repair_after_success(self): - """``windows_only``: the PowerShell install argv and the autostart + """``platforms("windows")``: the PowerShell install argv and the autostart repair hook are both inside the ``is_windows`` branch, so on Linux the fake selected a branch whose `powershell` doesn't exist on PATH.""" from unittest.mock import MagicMock @@ -1585,9 +1590,9 @@ def fake_which(name: str): verbose=False, ) - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_autostart_repair_quotes_username_space_path_via_file_path(self): - """``windows_only``: same early return off Windows — the elevated + """``platforms("windows")``: same early return off Windows — the elevated PowerShell command string is only built on a real Windows host. """ from hermes_cli import tools_config diff --git a/tests/hermes_cli/test_kanban_boards.py b/tests/hermes_cli/test_kanban_boards.py index 6fa004879fe3..d3465ae396cb 100644 --- a/tests/hermes_cli/test_kanban_boards.py +++ b/tests/hermes_cli/test_kanban_boards.py @@ -152,8 +152,11 @@ def test_remove_clears_init_cache_for_recreated_db(self, fresh_home, archive): # contains the resolved path, the CREATE TABLE pass is skipped and # downstream readers hit `no such table: task_events`. kb.create_board("recycle") - # First connect populates _INITIALIZED_PATHS for this DB. - with kb.connect(board="recycle") as conn: + # First connect populates _INITIALIZED_PATHS for this DB. Use + # connect_closing: `with connect() as conn` does NOT close the fd, and + # on Windows an open connection locks kanban.db so remove_board's + # rename below fails with WinError 5/32. + with kb.connect_closing(board="recycle") as conn: kb.create_task(conn, title="t1", assignee="dev") db_path = kb.board_dir("recycle") / "kanban.db" assert str(db_path.resolve()) in kb._INITIALIZED_PATHS @@ -165,7 +168,7 @@ def test_remove_clears_init_cache_for_recreated_db(self, fresh_home, archive): # Simulate the event-stream poll: re-open the same slug. connect() # recreates the directory + empty .db; the schema must be re-applied. - with kb.connect(board="recycle") as conn: + with kb.connect_closing(board="recycle") as conn: tables = { row[0] for row in conn.execute( diff --git a/tests/hermes_cli/test_kanban_core_functionality.py b/tests/hermes_cli/test_kanban_core_functionality.py index 0a47445ec110..7e219ee68c05 100644 --- a/tests/hermes_cli/test_kanban_core_functionality.py +++ b/tests/hermes_cli/test_kanban_core_functionality.py @@ -1320,6 +1320,7 @@ def _drive_nonzero_crash(conn, tid, fake_pid): return _drive_worker_exit(conn, tid, fake_pid, 256) +@pytest.mark.platforms("linux") def test_protocol_violation_budget_not_consumed_by_other_failures(kanban_home): """Mixed failure kinds must not consume the violation retry budget. diff --git a/tests/hermes_cli/test_kanban_db.py b/tests/hermes_cli/test_kanban_db.py index e4dce15bd86a..0b47a915281f 100644 --- a/tests/hermes_cli/test_kanban_db.py +++ b/tests/hermes_cli/test_kanban_db.py @@ -49,7 +49,7 @@ def _init_git_repo(repo: Path) -> None: -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_cross_process_init_lock_uses_windows_byte_range_lock(tmp_path, monkeypatch): """Windows must use a real (non-blocking) process lock, not a no-op open. @@ -57,7 +57,7 @@ def test_cross_process_init_lock_uses_windows_byte_range_lock(tmp_path, monkeypa wedged holder can never block connect() forever; a clean acquire takes the lock once and releases it once. - ``windows_only``: ``msvcrt`` does not exist off Windows, so faking + ``platforms("windows")``: ``msvcrt`` does not exist off Windows, so faking ``_IS_WINDOWS`` on Linux meant injecting a fake ``msvcrt`` module too — the test then asserted against its own stub rather than the byte-range locking API. Here the platform is real; only ``msvcrt.locking`` is @@ -264,6 +264,7 @@ def _exited_status(code: int) -> int: +@pytest.mark.platforms("linux") def test_rate_limit_exit_requeues_without_counting_failure( kanban_home, monkeypatch, ): @@ -566,7 +567,7 @@ def test_worktree_workspace_explicit_target_materializes_linked_worktree(kanban_ capture_output=True, text=True, ).stdout - assert f"worktree {target}" in listed + assert f"worktree {target.as_posix()}" in listed assert f"branch refs/heads/{branch}" in listed @@ -1189,6 +1190,9 @@ def test_resolve_hermes_argv_falls_back_to_module_form_when_no_path_shim(monkeyp monkeypatch.delenv("HERMES_BIN", raising=False) monkeypatch.setattr(shutil, "which", lambda name: None) + # On Windows _resolve_hermes_argv() uses _safe_which_no_cwd() instead of + # shutil.which — mock that too so the "no shim" path is the one tested. + monkeypatch.setattr(kb, "_safe_which_no_cwd", lambda name: None) argv = kb._resolve_hermes_argv() assert argv == [sys.executable, "-m", "hermes_cli.main"] diff --git a/tests/hermes_cli/test_linux_desktop_entry.py b/tests/hermes_cli/test_linux_desktop_entry.py index 8d8c246af52a..bbdb384965fd 100644 --- a/tests/hermes_cli/test_linux_desktop_entry.py +++ b/tests/hermes_cli/test_linux_desktop_entry.py @@ -12,6 +12,7 @@ from hermes_cli import linux_desktop_entry as lde + @pytest.fixture def xdg_home(tmp_path, monkeypatch) -> Path: data_home = tmp_path / "xdg-data" @@ -40,6 +41,7 @@ def _parse(entry_text: str) -> dict: return values +@pytest.mark.platforms("linux") def test_install_writes_entry_with_absolute_exec_and_icon( tmp_path, xdg_home, monkeypatch ): @@ -69,6 +71,7 @@ def test_install_writes_entry_with_absolute_exec_and_icon( icon_path = Path(values["Icon"]) assert icon_path.is_absolute() assert icon_path == lde.icon_path(root) +@pytest.mark.platforms("linux") def test_install_prefers_themed_icon_from_hicolor(tmp_path, xdg_home, monkeypatch): @@ -98,6 +101,7 @@ def test_install_prefers_themed_icon_from_hicolor(tmp_path, xdg_home, monkeypatc dest = xdg_home / "icons" / "hicolor" / "scalable" / "apps" / "hermes.png" assert dest.is_file() assert dest.read_bytes() == lde.icon_path(root).read_bytes() +@pytest.mark.platforms("linux") def test_install_icon_copy_failure_falls_back_to_absolute( @@ -129,6 +133,7 @@ def _boom(src, dst): assert values["Terminal"] == "false" +@pytest.mark.platforms("linux") def test_installed_entry_is_executable(tmp_path, xdg_home, monkeypatch): root = _make_project(tmp_path) monkeypatch.setattr( @@ -141,6 +146,7 @@ def test_installed_entry_is_executable(tmp_path, xdg_home, monkeypatch): assert entry.stat().st_mode & stat.S_IXUSR +@pytest.mark.platforms("linux") def test_exec_falls_back_to_interpreter_module(tmp_path, xdg_home, monkeypatch): root = _make_project(tmp_path) monkeypatch.setattr("hermes_cli.relaunch.resolve_hermes_bin", lambda: None) @@ -158,6 +164,7 @@ def test_exec_falls_back_to_interpreter_module(tmp_path, xdg_home, monkeypatch): # interpreter when the DE spawns the .desktop entry → ModuleNotFoundError, # silent (Terminal=false). The Exec line must prefix sys.executable for any # resolved bin that is a python script escaping the running venv. +@pytest.mark.platforms("linux") def test_exec_prefixes_interpreter_for_env_shebang_python_script( tmp_path, xdg_home, monkeypatch ): @@ -184,6 +191,7 @@ def test_exec_prefixes_interpreter_for_env_shebang_python_script( assert exec_line.endswith("desktop") +@pytest.mark.platforms("linux") def test_exec_leaves_shell_wrapper_launchers_alone(tmp_path, xdg_home, monkeypatch): root = _make_project(tmp_path) hermes_bin = tmp_path / "bin" / "hermes" @@ -204,6 +212,7 @@ def test_exec_leaves_shell_wrapper_launchers_alone(tmp_path, xdg_home, monkeypat assert exec_line == f"{hermes_bin} desktop" +@pytest.mark.platforms("linux") def test_exec_leaves_venv_shebang_scripts_alone(tmp_path, xdg_home, monkeypatch): import sys @@ -235,6 +244,7 @@ def _argv0_context(monkeypatch, argv0: str) -> None: import sys monkeypatch.setattr(sys, "argv", [argv0, "desktop"]) +@pytest.mark.platforms("linux") def test_exec_converges_from_repo_script_argv0_to_installed_wrapper( @@ -272,6 +282,7 @@ def test_exec_converges_from_repo_script_argv0_to_installed_wrapper( # Converged on the durable wrapper — NOT the repo script, and NOT an # interpreter-prefixed form pinning sys.executable. assert exec_line == f"{wrapper} desktop" +@pytest.mark.platforms("linux") def test_exec_never_persists_a_bare_interpreter_command( @@ -309,6 +320,7 @@ def test_exec_never_persists_a_bare_interpreter_command( and "desktop" in exec_line.split(" ", 1)[1] ), f"persisted an unrunnable bare-interpreter Exec: {exec_line}" assert exec_line == f"{wrapper} desktop" +@pytest.mark.platforms("linux") def test_exec_keeps_resolver_fallback_when_no_wrapper_on_path( @@ -349,6 +361,7 @@ def fake_resolve(): assert exec_line.endswith("-m hermes_cli.main desktop") assert Path(exec_line.split(" ")[0].strip('"')).is_absolute() assert str(repo_script) not in exec_line +@pytest.mark.platforms("linux") def test_exec_uses_known_wrapper_when_path_lookup_misses( @@ -397,6 +410,7 @@ def fake_resolve(): # The probe found the wrapper despite the PATH miss. assert exec_line == f"{known_wrapper} desktop" +@pytest.mark.platforms("linux") def test_exec_rejects_known_wrapper_from_another_checkout( @@ -478,6 +492,7 @@ def fake_resolve(): ), ], ) +@pytest.mark.platforms("linux") def test_known_wrapper_candidates_cover_installer_layouts( layout, env_overrides, expected, monkeypatch ): @@ -513,6 +528,7 @@ def test_known_wrapper_candidates_cover_installer_layouts( if layout == "non-root-no-fhs": # Non-root euid: /usr/local/bin must be excluded outright. assert "/usr/local/bin/hermes" not in candidates +@pytest.mark.platforms("linux") def test_install_is_idempotent_and_skips_cache_refresh(tmp_path, xdg_home, monkeypatch): @@ -531,6 +547,7 @@ def test_install_is_idempotent_and_skips_cache_refresh(tmp_path, xdg_home, monke # Unchanged content → no rewrite, no menu-cache churn on every launch. lde.install_desktop_entry(root) assert len(calls) == 1 +@pytest.mark.platforms("linux") def test_install_without_source_icon_uses_themed_name(tmp_path, xdg_home, monkeypatch): @@ -548,14 +565,14 @@ def test_install_without_source_icon_uses_themed_name(tmp_path, xdg_home, monkey assert _parse(entry.read_text(encoding="utf-8"))["Icon"] == "hermes" -@pytest.mark.macos_only +@pytest.mark.platforms("macos") def test_install_is_a_noop_on_macos(tmp_path): """Faking darwin only renamed the host — the real macOS runner is the only place the `sys.platform` guard is exercised against a real host.""" assert lde.install_desktop_entry(_make_project(tmp_path)) is None -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_install_is_a_noop_on_windows(tmp_path): """As above for Windows: a fake left POSIX paths and a POSIX XDG layout in place, so the no-op was never proven against a real one.""" @@ -576,6 +593,7 @@ def _stub_tools(monkeypatch, available: "set[str]") -> "list[list[str]]": ) monkeypatch.setattr(lde, "_run_quiet", lambda cmd: ran.append(cmd) or True) return ran +@pytest.mark.platforms("linux") def test_refresh_runs_kbuildsycoca6_when_present(monkeypatch, tmp_path): @@ -588,6 +606,7 @@ def test_refresh_runs_kbuildsycoca6_when_present(monkeypatch, tmp_path): ["/usr/bin/update-desktop-database", str(tmp_path)], ["/usr/bin/kbuildsycoca6", "--noincremental"], ] +@pytest.mark.platforms("linux") def test_refresh_falls_back_to_kbuildsycoca5(monkeypatch, tmp_path): @@ -597,6 +616,7 @@ def test_refresh_falls_back_to_kbuildsycoca5(monkeypatch, tmp_path): assert tools == ["kbuildsycoca5"] assert ran == [["/usr/bin/kbuildsycoca5", "--noincremental"]] +@pytest.mark.platforms("linux") def test_refresh_prefers_kbuildsycoca6_over_5(monkeypatch, tmp_path): @@ -605,6 +625,7 @@ def test_refresh_prefers_kbuildsycoca6_over_5(monkeypatch, tmp_path): lde.refresh_desktop_databases(tmp_path) assert [cmd[0] for cmd in ran] == ["/usr/bin/kbuildsycoca6"] +@pytest.mark.platforms("linux") def test_refresh_skips_missing_tools(monkeypatch, tmp_path): @@ -612,6 +633,7 @@ def test_refresh_skips_missing_tools(monkeypatch, tmp_path): assert lde.refresh_desktop_databases(tmp_path) == [] assert ran == [] +@pytest.mark.platforms("linux") def test_refresh_reports_only_tools_that_succeeded(monkeypatch, tmp_path): @@ -620,12 +642,14 @@ def test_refresh_reports_only_tools_that_succeeded(monkeypatch, tmp_path): monkeypatch.setattr(lde, "_run_quiet", lambda cmd: "kbuildsycoca" in cmd[0]) assert lde.refresh_desktop_databases(tmp_path) == ["kbuildsycoca6"] +@pytest.mark.platforms("linux") def test_run_quiet_swallows_missing_binary(tmp_path): assert lde._run_quiet([str(tmp_path / "definitely-not-a-binary")]) is False +@pytest.mark.platforms("linux") def test_exec_arg_quoting_handles_spaces(tmp_path, xdg_home, monkeypatch): root = _make_project(tmp_path) spaced = tmp_path / "my apps" / "hermes" @@ -643,6 +667,7 @@ def test_exec_arg_quoting_handles_spaces(tmp_path, xdg_home, monkeypatch): @pytest.mark.skipif( sys.platform == "win32", reason="Symlinks require elevated privileges on Windows" ) +@pytest.mark.platforms("linux") def test_running_interpreter_keeps_venv_semantic_path(tmp_path, monkeypatch): """Lexical preserved only when pyvenv.cfg marks the path as a venv.""" # venv layout: bin/python symlink -> base, pyvenv.cfg at venv root @@ -666,6 +691,7 @@ def test_running_interpreter_keeps_venv_semantic_path(tmp_path, monkeypatch): plain_link.symlink_to(base) monkeypatch.setattr(lde.sys, "executable", str(plain_link)) assert lde._running_interpreter() == str(base) +@pytest.mark.platforms("linux") def test_running_interpreter_resolves_plain_interpreter(monkeypatch): @@ -673,6 +699,7 @@ def test_running_interpreter_resolves_plain_interpreter(monkeypatch): monkeypatch.setattr(lde.sys, "executable", "/usr/bin/python3") out = lde._running_interpreter() assert Path(out).is_absolute() +@pytest.mark.platforms("linux") def test_can_import_probe_runs_and_caches(tmp_path): @@ -701,6 +728,7 @@ def test_can_import_probe_runs_and_caches(tmp_path): assert time.monotonic() - t0 < 0.05 # cache hit: no subprocess finally: lde._probe_cache.pop(str(real), None) +@pytest.mark.platforms("linux") def test_exec_falls_back_to_running_interpreter_when_probe_fails( @@ -747,6 +775,7 @@ def fake_resolve(): "suffix", ["-old", ".bak", "-copy"], ) +@pytest.mark.platforms("linux") def test_wrapper_ownership_rejects_sibling_extensions(suffix, tmp_path): """A shim execing `/...` must NOT pass ownership. @@ -769,6 +798,7 @@ def test_wrapper_ownership_rejects_sibling_extensions(suffix, tmp_path): @pytest.mark.skipif( sys.platform == "win32", reason="Symlinks require elevated privileges on Windows" ) +@pytest.mark.platforms("linux") def test_wrapper_ownership_accepts_shim_via_symlinked_home(tmp_path, monkeypatch): """Installer writes $INSTALL_DIR lexically; the root stays lexical too. @@ -806,6 +836,7 @@ def test_wrapper_ownership_accepts_shim_via_symlinked_home(tmp_path, monkeypatch resolve_fn=lambda: sys.argv[0] if sys.argv[0] else None, checkout_root=lexical_checkout, ) == str(shim) +@pytest.mark.platforms("linux") def test_needs_interpreter_case_insensitive_match(tmp_path, monkeypatch): @@ -826,6 +857,7 @@ def test_needs_interpreter_case_insensitive_match(tmp_path, monkeypatch): monkeypatch.setattr(lde.sys, "executable", str(interpreter)) assert lde._needs_interpreter(console_script) is False +@pytest.mark.platforms("linux") def test_needs_interpreter_rejects_sibling_directory(tmp_path, monkeypatch): @@ -848,6 +880,7 @@ def test_needs_interpreter_rejects_sibling_directory(tmp_path, monkeypatch): encoding="utf-8", ) assert lde._needs_interpreter(sibling_script) is True +@pytest.mark.platforms("linux") def test_needs_interpreter_strips_flags_before_comparing(tmp_path, monkeypatch): @@ -861,6 +894,7 @@ def test_needs_interpreter_strips_flags_before_comparing(tmp_path, monkeypatch): flagged = tmp_path / "flagged" flagged.write_text(f"#!{interp} -S\nimport hermes_cli\n", encoding="utf-8") assert lde._needs_interpreter(flagged) is False +@pytest.mark.platforms("linux") def test_needs_interpreter_env_shebang_always_escapes(tmp_path, monkeypatch): @@ -885,6 +919,7 @@ def test_needs_interpreter_env_shebang_always_escapes(tmp_path, monkeypatch): f"#!/usr/bin/env -S {interp}\nimport hermes_cli\n", encoding="utf-8" ) assert lde._needs_interpreter(env_abs) is False +@pytest.mark.platforms("linux") def test_probe_skips_wrapper_with_escaping_python_shebang( @@ -929,6 +964,7 @@ def fake_resolve(): assert str(broken_wrapper) not in exec_line assert exec_line.endswith("-m hermes_cli.main desktop") +@pytest.mark.platforms("linux") def test_probe_accepts_shell_launcher_wrapper(tmp_path, xdg_home, monkeypatch): @@ -961,6 +997,7 @@ def fake_resolve(): entry = lde.install_desktop_entry(root) exec_line = _parse(entry.read_text(encoding="utf-8"))["Exec"] assert exec_line == f"{good_wrapper} desktop" +@pytest.mark.platforms("linux") def test_install_icon_handles_truncated_png_header(tmp_path, xdg_home, monkeypatch): diff --git a/tests/hermes_cli/test_macos_tcc_anchor.py b/tests/hermes_cli/test_macos_tcc_anchor.py index f07eff9f4caf..3212f9ee14fa 100644 --- a/tests/hermes_cli/test_macos_tcc_anchor.py +++ b/tests/hermes_cli/test_macos_tcc_anchor.py @@ -6,7 +6,7 @@ and alias symlinks to the copied interpreter lost the venv prefix (#95541). Linux tests use fake checkout/uv-store layouts with ``platform.system`` -monkeypatched. The real-interpreter E2E is ``macos_only`` so it runs on +monkeypatched. The real-interpreter E2E is ``platforms("macos")`` so it runs on the existing macOS CI job, not against one-byte fixtures. """ @@ -127,6 +127,7 @@ def test_rejects_uv_store_on_linux(self): assert not tcc._is_uv_macos_store(path) +@pytest.mark.platforms("linux") class TestEnsureTccAnchor: def test_noop_on_non_macos(self, tmp_path, monkeypatch): _linux(monkeypatch) @@ -367,6 +368,7 @@ def test_store_root_marker_tracks_managed_uv_constant(self): assert f"/{_RUNTIME_DIR_NAME}/python/" in tcc._STORE_ROOT_MARKERS +@pytest.mark.platforms("linux") class TestBootGate: """Direct branch coverage for _passes_boot_gate. @@ -453,6 +455,7 @@ def test_accepts_matching_prefix(self, tmp_path, monkeypatch): assert tcc._passes_boot_gate(tmp_path / "staged", venv) +@pytest.mark.platforms("linux") class TestTccAnchorState: def test_state_active_through_unpatched_home_symlink(self, tmp_path, monkeypatch): # The managed-runtime layout symlinks cpython-3.11-macos-* → @@ -523,6 +526,7 @@ def test_state_stale_after_patch_bump(self, tmp_path, monkeypatch): assert status == "active" +@pytest.mark.platforms("linux") class TestDoctorCheck: def test_missing_warns_without_fix(self, monkeypatch, capsys): monkeypatch.setattr( @@ -568,13 +572,13 @@ def boom(*a, **k): assert "macOS TCC anchor check failed" in out -@pytest.mark.macos_only +@pytest.mark.platforms("macos") class TestAnchoredAliasesBootE2E: """Real-interpreter proof that the re-land stays bootable (#95596). Copies the running interpreter's real base binary into a fake uv-store layout (stdlib via a ``lib`` symlink) and actually executes every - entry point after anchoring. ``macos_only`` so Linux CI cannot + entry point after anchoring. ``platforms("macos")`` so Linux CI cannot greenwash this with a fixture. """ diff --git a/tests/hermes_cli/test_managed_uv.py b/tests/hermes_cli/test_managed_uv.py index a81fd52501d2..ccbf3a3596bf 100644 --- a/tests/hermes_cli/test_managed_uv.py +++ b/tests/hermes_cli/test_managed_uv.py @@ -157,6 +157,7 @@ def test_skips_non_macos(self, tmp_path, monkeypatch): class TestResolveUv: + @pytest.mark.platforms("linux") def test_existing_executable(self, tmp_path): _make_executable(tmp_path / "bin" / "uv") with patch("hermes_cli.managed_uv.get_hermes_home", return_value=tmp_path): @@ -181,6 +182,7 @@ def test_non_executable_file_returns_none(self, tmp_path): class TestEnsureUv: + @pytest.mark.platforms("linux") def test_installs_if_missing(self, tmp_path): with patch("hermes_cli.managed_uv.get_hermes_home", return_value=tmp_path), \ patch("hermes_cli.managed_uv.repair_vulnerable_runtime", return_value=_RRR("not-applicable")), \ @@ -195,6 +197,7 @@ def fake_install(target): assert path == str(tmp_path / "bin" / "uv") mock_install.assert_called_once() + @pytest.mark.platforms("linux") def test_install_reports_runtime_repair_to_observer(self, tmp_path): from hermes_cli.managed_uv import ( RuntimeRepairResult, @@ -302,9 +305,9 @@ def test_uvresult_would_break_windows_list2cmdline(self): with pytest.raises(TypeError): subprocess.list2cmdline([_UvResult("C:\\hermes\\uv.exe"), "pip"]) - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_returns_plain_str_safe_for_subprocess(self, tmp_path): - """``windows_only``: the subject is the real Windows opt-out branch and + """``platforms("windows")``: the subject is the real Windows opt-out branch and ``subprocess.list2cmdline`` — the faked ``platform.system`` only ever proved the branch existed, not that the field crash was fixed on the host that reported it.""" @@ -329,6 +332,7 @@ class TestUpdateManagedUv: + @pytest.mark.platforms("linux") def test_fresh_stamp_skips_network_self_update_but_not_repair(self, tmp_path, monkeypatch): """A recent success stamp must skip `uv self update` entirely while the vulnerable-runtime repair probe still runs (CVE repair is never gated).""" @@ -355,6 +359,7 @@ def test_fresh_stamp_skips_network_self_update_but_not_repair(self, tmp_path, mo mock_repair.assert_called_once_with(str(uv)) + @pytest.mark.platforms("linux") def test_stale_stamp_runs_self_update_and_refreshes_stamp(self, tmp_path): import os as _os import time as _time @@ -674,6 +679,7 @@ def test_post_swap_smoke_failure_rolls_back_live_venv(self, tmp_path): # --------------------------------------------------------------------------- class TestInstallUvInternals: + @pytest.mark.platforms("linux") def test_posix_sets_uv_unmanaged_install(self, tmp_path): target = tmp_path / "bin" / "uv" with patch("hermes_cli.managed_uv._install_uv_posix") as mock_posix: @@ -1258,6 +1264,7 @@ def _checkout(self, tmp_path, *dirs): (bin_dir / "python").write_text("py", encoding="utf-8") return root + @pytest.mark.platforms("linux") def test_dot_venv_only_is_targeted(self, tmp_path): from hermes_cli.managed_uv import _default_live_venv diff --git a/tests/hermes_cli/test_migrate_xai.py b/tests/hermes_cli/test_migrate_xai.py index 74099b2969fd..420210f6bdea 100644 --- a/tests/hermes_cli/test_migrate_xai.py +++ b/tests/hermes_cli/test_migrate_xai.py @@ -248,6 +248,7 @@ def boom(fd): # ...and the aborted write must not leave a temp file behind. assert list(trap_config.parent.glob("*.tmp")) == [] + @pytest.mark.require_symlinks def test_symlinked_config_is_replaced_in_place(self, tmp_path: Path): """A config.yaml symlinked into a dotfiles repo must stay a symlink.""" real_dir = tmp_path / "dotfiles" diff --git a/tests/hermes_cli/test_nous_subscription.py b/tests/hermes_cli/test_nous_subscription.py index d9f71a9af160..533b3ada35a0 100644 --- a/tests/hermes_cli/test_nous_subscription.py +++ b/tests/hermes_cli/test_nous_subscription.py @@ -3,6 +3,8 @@ import shutil import sys +import pytest + from hermes_cli.nous_account import NousPortalAccountInfo, NousToolAccessInfo from hermes_cli import nous_subscription as ns @@ -477,6 +479,7 @@ def test_has_agent_browser_import_failure_falls_back_to_path_check(monkeypatch): assert ns._has_agent_browser() is True +@pytest.mark.platforms("linux") def test_has_agent_browser_import_failure_falls_back_to_hermes_managed_node_path( monkeypatch, tmp_path ): diff --git a/tests/hermes_cli/test_npm_engine.py b/tests/hermes_cli/test_npm_engine.py index 56e6e7e42647..1464b7436a5b 100644 --- a/tests/hermes_cli/test_npm_engine.py +++ b/tests/hermes_cli/test_npm_engine.py @@ -85,6 +85,7 @@ def test_malformed_required_block_is_ignored(self): assert required_npm_range(broken) is None +@pytest.mark.require_symlinks class TestManagedDetection: """The upgrade must fire for every spelling of the managed npm, and for no other npm — this is the boundary between "Hermes fixes it" and "the diff --git a/tests/hermes_cli/test_oneshot_surrogate.py b/tests/hermes_cli/test_oneshot_surrogate.py index 9039690a8406..1d5eb02c43f6 100644 --- a/tests/hermes_cli/test_oneshot_surrogate.py +++ b/tests/hermes_cli/test_oneshot_surrogate.py @@ -2,6 +2,7 @@ from __future__ import annotations +import os import subprocess import sys import textwrap @@ -35,5 +36,5 @@ def test_oneshot_replaces_lone_surrogate_and_exits_zero(): # U+FFFD as UTF-8; no raw surrogate bytes assert "\ufffd".encode("utf-8") in result.stdout assert b"answer " in result.stdout - assert b" here\n" in result.stdout + assert (" here" + os.linesep).encode("ascii") in result.stdout assert b"Traceback" not in result.stderr diff --git a/tests/hermes_cli/test_plugin_install_ref.py b/tests/hermes_cli/test_plugin_install_ref.py index f7f1dbec1e39..0fcfbfb73e48 100644 --- a/tests/hermes_cli/test_plugin_install_ref.py +++ b/tests/hermes_cli/test_plugin_install_ref.py @@ -350,7 +350,7 @@ def test_metadata_write_failure_rolls_back_removal(monkeypatch, tmp_path): def test_reinstall_after_manual_directory_removal_retains_pin(monkeypatch, tmp_path): - from hermes_cli.plugins_cmd import _install_plugin_core + from hermes_cli.plugins_cmd import _install_plugin_core, _rmtree_force repo, old_sha, _new_sha = _plugin_repo(tmp_path) home = tmp_path / "home" @@ -358,7 +358,7 @@ def test_reinstall_after_manual_directory_removal_retains_pin(monkeypatch, tmp_p target, _manifest, _name = _install_plugin_core( repo.as_uri(), force=False, ref=old_sha ) - shutil.rmtree(target) + _rmtree_force(target) target, _manifest, _name = _install_plugin_core(repo.as_uri(), force=False) diff --git a/tests/hermes_cli/test_plugins_cmd.py b/tests/hermes_cli/test_plugins_cmd.py index 4002698b0e78..283e410531c3 100644 --- a/tests/hermes_cli/test_plugins_cmd.py +++ b/tests/hermes_cli/test_plugins_cmd.py @@ -103,6 +103,7 @@ def test_valid_nested_subdir(self, tmp_path): + @pytest.mark.require_symlinks def test_rejects_symlink_escape(self, tmp_path): clone = tmp_path / "clone" clone.mkdir() @@ -447,7 +448,7 @@ class TestCmdRemove: @patch("hermes_cli.plugins_cmd._sanitize_plugin_name") @patch("hermes_cli.plugins_cmd._plugins_dir") - @patch("hermes_cli.plugins_cmd.shutil.rmtree") + @patch("hermes_cli.fs_utils.shutil.rmtree") def test_remove_deletes_plugin(self, mock_rmtree, mock_plugins_dir, mock_sanitize): from hermes_cli.plugins_cmd import cmd_remove @@ -458,7 +459,12 @@ def test_remove_deletes_plugin(self, mock_rmtree, mock_plugins_dir, mock_sanitiz cmd_remove("test-plugin") - mock_rmtree.assert_called_once_with(mock_target) + # Deletion goes through the read-only-clearing rmtree wrapper: the + # target is deleted with an onerror hook attached (Windows read-only + # git objects), never a bare rmtree. + assert mock_rmtree.call_count == 1 + assert mock_rmtree.call_args.args == (mock_target,) + assert callable(mock_rmtree.call_args.kwargs.get("onerror")) @patch("hermes_cli.plugins_cmd._sanitize_plugin_name") @patch("hermes_cli.plugins_cmd._plugins_dir") diff --git a/tests/hermes_cli/test_profile_export_credentials.py b/tests/hermes_cli/test_profile_export_credentials.py index 0ab6a4e23be9..3b6524cac750 100644 --- a/tests/hermes_cli/test_profile_export_credentials.py +++ b/tests/hermes_cli/test_profile_export_credentials.py @@ -11,6 +11,8 @@ import tarfile +import pytest + from hermes_cli.profiles import export_profile, _DEFAULT_EXPORT_EXCLUDE_ROOT # Long enough to match agent.redact prefix patterns (sk- + 10+ chars). @@ -107,6 +109,7 @@ def test_named_profile_export_redacts_secrets_in_text(self, tmp_path, monkeypatc assert _LEAKED_KEY in skill.read_text() assert _LEAKED_KEY in memory.read_text() + @pytest.mark.require_symlinks def test_export_redacts_through_symlink_without_touching_source( self, tmp_path, monkeypatch ): diff --git a/tests/hermes_cli/test_profiles.py b/tests/hermes_cli/test_profiles.py index ac5e1e27948f..79b464bf1f6e 100644 --- a/tests/hermes_cli/test_profiles.py +++ b/tests/hermes_cli/test_profiles.py @@ -115,6 +115,7 @@ class TestCreateProfile: """Tests for create_profile().""" + @pytest.mark.platforms("linux") def test_seeds_placeholder_env_file(self, profile_env): """Fresh profiles get their own .env (owner-only) so channel/env writes are profile-scoped from day one instead of falling through @@ -269,6 +270,7 @@ class TestBackfillProfileEnvs: gives pre-#44792 profiles (created before .env seeding) their own .env, copied from the default install so credentials don't break.""" + @pytest.mark.platforms("linux") def test_copies_default_env_into_envless_profiles(self, profile_env): import stat tmp_path = profile_env @@ -593,7 +595,7 @@ class TestAliasCollision: - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_checks_bat_extension(self, profile_env): wrapper_dir = profile_env / ".local" / "bin" wrapper_dir.mkdir(parents=True, exist_ok=True) @@ -622,6 +624,7 @@ def test_traversal_alias_rejected_before_path_lookup(self, profile_env): class TestWrapperScript: """Tests for create_wrapper_script() and remove_wrapper_script().""" + @pytest.mark.platforms("linux") def test_creates_sh_on_posix(self, profile_env, monkeypatch): monkeypatch.setattr("hermes_cli.profiles.shutil.which", lambda name: "/opt/hermes/bin/hermes") from hermes_cli.profiles import create_wrapper_script @@ -633,7 +636,7 @@ def test_creates_sh_on_posix(self, profile_env, monkeypatch): assert "exec /opt/hermes/bin/hermes -p mybot" in content - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_remove_finds_bat_on_windows(self, profile_env): from hermes_cli.profiles import create_wrapper_script, remove_wrapper_script wrapper = create_wrapper_script("mybot") @@ -696,6 +699,7 @@ def test_ignores_unrelated_files(self, profile_env): assert find_alias_for_profile("steve") is None + @pytest.mark.platforms("linux") def test_list_profiles_surfaces_custom_alias(self, profile_env): from hermes_cli.profiles import ( create_profile, @@ -800,6 +804,7 @@ def test_export_default_includes_profile_data(self, profile_env, tmp_path): assert "default/memories/MEMORY.md" in names + @pytest.mark.require_symlinks def test_export_default_handles_broken_symlinks(self, profile_env, tmp_path): """Broken symlinks inside allowed artifacts are preserved, not crashed (#58394). @@ -977,6 +982,7 @@ def test_emoji_description_is_written_as_real_utf8(self, tmp_path): assert "🧙" in raw assert profiles.read_profile_meta(profile_dir)["description"] == "Code wizard 🧙 ✨" + @pytest.mark.require_symlinks def test_symlinked_profile_yaml_survives_the_write(self, tmp_path): """Guard on the conversion, not a behavior change. diff --git a/tests/hermes_cli/test_projects_db.py b/tests/hermes_cli/test_projects_db.py index 81a8bb1efdaa..c33ef261a172 100644 --- a/tests/hermes_cli/test_projects_db.py +++ b/tests/hermes_cli/test_projects_db.py @@ -46,8 +46,8 @@ def test_create_get_list(conn): assert proj.slug == "hermes-agent" assert proj.name == "Hermes Agent" # First folder becomes primary. - assert proj.primary_path == "/tmp/hermes" - assert [f.path for f in proj.folders] == ["/tmp/hermes"] + assert proj.primary_path == os.path.abspath("/tmp/hermes") + assert [f.path for f in proj.folders] == [os.path.abspath("/tmp/hermes")] assert proj.folders[0].is_primary is True # Lookup by slug too. @@ -136,7 +136,7 @@ def test_per_profile_isolation(tmp_path): assert [p.slug for p in pdb.list_projects(a)] == ["only-in-a"] assert pdb.list_projects(b) == [] assert [row["root"] for row in pdb.list_discovered_repos(a)] == [ - "/a/scanned" + os.path.abspath("/a/scanned") ] assert pdb.list_discovered_repos(b) == [] finally: diff --git a/tests/hermes_cli/test_prompt_compose_command.py b/tests/hermes_cli/test_prompt_compose_command.py index 69461279e60d..cd34337b1d35 100644 --- a/tests/hermes_cli/test_prompt_compose_command.py +++ b/tests/hermes_cli/test_prompt_compose_command.py @@ -46,6 +46,7 @@ def test_command_registered(): assert resolve_command("compose").name == "prompt" +@pytest.mark.platforms("linux") def test_compose_reads_and_strips_header(monkeypatch): monkeypatch.setenv("EDITOR", _fake_editor("Refactor the auth module.\nUse pytest.")) out = _Stub()._compose_in_editor("") @@ -54,6 +55,7 @@ def test_compose_reads_and_strips_header(monkeypatch): assert "#!" not in out # the instructional header is stripped +@pytest.mark.platforms("linux") def test_empty_buffer_does_not_seed(monkeypatch): monkeypatch.setenv("EDITOR", _fake_editor("", mode="clear")) s = _Stub() diff --git a/tests/hermes_cli/test_relaunch.py b/tests/hermes_cli/test_relaunch.py index ffb611950e45..fbbc1a073a64 100644 --- a/tests/hermes_cli/test_relaunch.py +++ b/tests/hermes_cli/test_relaunch.py @@ -99,6 +99,7 @@ def test_can_disable_preserve(self, monkeypatch): class TestRelaunch: + @pytest.mark.platforms("linux") def test_calls_execvp(self, monkeypatch): calls = [] @@ -114,14 +115,14 @@ def fake_execvp(path, argv): assert calls == [("/usr/bin/hermes", ["/usr/bin/hermes", "--resume", "abc"])] - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_uses_subprocess_not_execvp(self, monkeypatch): """On Windows, os.execvp raises OSError "Exec format error" when the target is a .cmd shim or console-script wrapper (both common for hermes). relaunch() must detect win32 and use subprocess.run + sys.exit instead. - ``windows_only``: the bug is that ``os.execvp`` cannot exec a Windows + ``platforms("windows")``: the bug is that ``os.execvp`` cannot exec a Windows console-script shim. On Linux ``execvp`` works fine, so a patched platform only re-asserted the branch we wrote, never the constraint that motivated it. @@ -129,7 +130,7 @@ def test_windows_uses_subprocess_not_execvp(self, monkeypatch): monkeypatch.setattr(relaunch_mod, "resolve_hermes_bin", lambda: r"C:\Users\test\hermes.exe") # Pin sys.argv: relaunch() preserves inherited flags from the LIVE # argv, so under pytest it happily inherited the runner's own - # "-m 'windows_only and not integration'" and the assertion below saw + # "-m 'platforms("windows") and not integration'" and the assertion below saw # them in the child argv. Nothing to do with Windows — it only showed # up here because this is the first lane that actually executes the # test, and -m is how that lane selects it. @@ -163,7 +164,7 @@ def fake_execvp(*args, **kwargs): assert execvp_calls == [] assert captured_argv == [[r"C:\Users\test\hermes.exe", "chat"]] - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_propagates_child_exit_code(self, monkeypatch): """A non-zero exit from the child should flow through to sys.exit.""" monkeypatch.setattr(relaunch_mod, "resolve_hermes_bin", lambda: r"C:\hermes.exe") @@ -190,12 +191,12 @@ class TestResolveHermesBinWindowsPyGuard: subprocess.run can't actually exec a .py directly, so the relaunch would fail with the cryptic "%1 is not a valid Win32 application" error. - The Windows cases are ``windows_only``: the PATHEXT-driven ``os.access`` + The Windows cases are ``platforms("windows")``: the PATHEXT-driven ``os.access`` result the guard defends against simply does not occur on POSIX, so a faked ``sys.platform`` could never reproduce the hazard. """ - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_rejects_py_argv0_falls_through_to_path(self, monkeypatch, tmp_path): """On Windows, if sys.argv[0] is a .py file, we must skip the argv[0] fast-path and fall through to PATH / python -m.""" @@ -215,7 +216,7 @@ def test_windows_rejects_py_argv0_falls_through_to_path(self, monkeypatch, tmp_p # Must NOT be the .py — must be the hermes.exe PATH entry. assert bin_path == r"C:\venv\Scripts\hermes.exe" - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_posix_still_accepts_py_argv0(self, monkeypatch, tmp_path): """POSIX behaviour unchanged: argv[0] pointing at an executable script (including .py with a shebang + chmod +x) is fine to return @@ -226,7 +227,7 @@ def test_posix_still_accepts_py_argv0(self, monkeypatch, tmp_path): monkeypatch.setattr(relaunch_mod.sys, "argv", [str(script), "chat"]) assert relaunch_mod.resolve_hermes_bin() == str(script) - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_py_argv0_with_no_hermes_on_path_returns_none(self, monkeypatch, tmp_path): """Bulletproof fallback: if argv0 is .py on Windows AND hermes.exe isn't on PATH, return None so the caller falls back to diff --git a/tests/hermes_cli/test_resume_latest_and_in_dir.py b/tests/hermes_cli/test_resume_latest_and_in_dir.py index 5aa8348884e1..d537ad63eb14 100644 --- a/tests/hermes_cli/test_resume_latest_and_in_dir.py +++ b/tests/hermes_cli/test_resume_latest_and_in_dir.py @@ -215,6 +215,7 @@ def test_in_dir_missing_directory_exits(main_mod, monkeypatch, tmp_path, capsys) assert "--in directory not found" in capsys.readouterr().out +@pytest.mark.platforms("linux") def test_in_dir_expands_user_home(main_mod, launched, monkeypatch, tmp_path): import os diff --git a/tests/hermes_cli/test_serve_runtime_inventory.py b/tests/hermes_cli/test_serve_runtime_inventory.py index ce3ad99cc337..793582436fc6 100644 --- a/tests/hermes_cli/test_serve_runtime_inventory.py +++ b/tests/hermes_cli/test_serve_runtime_inventory.py @@ -197,11 +197,14 @@ def test_scan_dashboard_processes_includes_ledger_only_serves(monkeypatch): fake_pi = SimpleNamespace(ledger_entries=lambda **k: [profiled]) monkeypatch.setitem(sys.modules, "hermes_cli.process_identity", fake_pi) - # Force the ps/wmic scan itself to find nothing. + # Force the ps/wmic scan itself to find nothing. The Windows scan runs + # through ``bounded_probe_run`` (a subprocess.run wrapper), not + # ``subprocess.run`` directly, so patch both. fake_run = SimpleNamespace(returncode=0, stdout="") - monkeypatch.setattr( - dp.subprocess, "run", lambda *a, **k: fake_run - ) + monkeypatch.setattr(dp.subprocess, "run", lambda *a, **k: fake_run) + import hermes_cli._subprocess_compat as _sc + + monkeypatch.setattr(_sc, "bounded_probe_run", lambda *a, **k: fake_run) result = dp._scan_dashboard_processes() assert (8123, profiled["argv"]) in result @@ -214,5 +217,8 @@ def test_scan_dashboard_processes_ledger_respects_exclusions(monkeypatch): monkeypatch.setitem(sys.modules, "hermes_cli.process_identity", fake_pi) fake_run = SimpleNamespace(returncode=0, stdout="") monkeypatch.setattr(dp.subprocess, "run", lambda *a, **k: fake_run) + import hermes_cli._subprocess_compat as _sc + + monkeypatch.setattr(_sc, "bounded_probe_run", lambda *a, **k: fake_run) assert dp._scan_dashboard_processes(exclude_pids={8124}) == [] diff --git a/tests/hermes_cli/test_service_manager.py b/tests/hermes_cli/test_service_manager.py index 706611af90ce..22f99ffb6151 100644 --- a/tests/hermes_cli/test_service_manager.py +++ b/tests/hermes_cli/test_service_manager.py @@ -192,6 +192,7 @@ def _fake(cmd, **kw): # tests/docker/test_s6_profile_gateway_integration.py. +@pytest.mark.platforms("linux") def test_seed_supervise_skeleton_creates_expected_layout(tmp_path) -> None: """Verifies the dirs + FIFO the helper lays down.""" import stat @@ -225,7 +226,7 @@ def test_seed_supervise_skeleton_creates_expected_layout(tmp_path) -> None: assert stat.S_IMODE(control.stat().st_mode) == 0o660 -@pytest.mark.linux_only +@pytest.mark.platforms("linux") def test_seed_supervise_skeleton_sets_setgid_on_event_dirs(tmp_path) -> None: """The event dirs carry setgid so s6-supervise's EEXIST path leaves them alone. @@ -371,6 +372,7 @@ def _log_run_setup_fragment(rendered: str) -> str: return "#!/bin/sh\n" + "".join(keep) +@pytest.mark.platforms("linux") def test_s6_log_run_creates_leaf_as_hermes_without_chown( s6_scandir, fake_subprocess_run, ) -> None: diff --git a/tests/hermes_cli/test_session_browse.py b/tests/hermes_cli/test_session_browse.py index 16ccf384b008..dd2cc6f733c1 100644 --- a/tests/hermes_cli/test_session_browse.py +++ b/tests/hermes_cli/test_session_browse.py @@ -9,6 +9,8 @@ import time from unittest.mock import MagicMock, patch +import pytest + from hermes_cli.main import _session_browse_picker @@ -95,6 +97,7 @@ def mock_import(name, *args, **kwargs): # ─── Curses-based picker (mocked curses) ──────────────────────────────────── +@pytest.mark.platforms("linux") class TestCursesBrowse: """Tests for the curses-based interactive picker via simulated key sequences.""" diff --git a/tests/hermes_cli/test_skin_cmd.py b/tests/hermes_cli/test_skin_cmd.py index 09cf29422ae6..e4750651750b 100644 --- a/tests/hermes_cli/test_skin_cmd.py +++ b/tests/hermes_cli/test_skin_cmd.py @@ -99,6 +99,7 @@ def _tracking_fsync(fd): assert [p.name for p in _skins().iterdir() if p.name.endswith(".tmp")] == [] +@pytest.mark.require_symlinks def test_set_preserves_a_symlinked_skin_file(): """Guard on the conversion, not a behavior change. diff --git a/tests/hermes_cli/test_ssh_ownership_endpoint.py b/tests/hermes_cli/test_ssh_ownership_endpoint.py index ee6154ba80eb..a56d7f1937fa 100644 --- a/tests/hermes_cli/test_ssh_ownership_endpoint.py +++ b/tests/hermes_cli/test_ssh_ownership_endpoint.py @@ -112,6 +112,7 @@ def test_ssh_runtime_marker_survives_in_place_installs(tmp_path, monkeypatch): web_server._apply_ssh_owner_nonce(None) +@pytest.mark.platforms("linux") def test_ssh_runtime_readonly_purelib_falls_back_to_stat(tmp_path, monkeypatch): """When site-packages is unwritable the marker can't be placed; the stat-snapshot fallback still arms (weaker, never a false stale).""" diff --git a/tests/hermes_cli/test_ssh_session_token_parser.py b/tests/hermes_cli/test_ssh_session_token_parser.py index fd50b7a37328..fcc6c529ea04 100644 --- a/tests/hermes_cli/test_ssh_session_token_parser.py +++ b/tests/hermes_cli/test_ssh_session_token_parser.py @@ -123,6 +123,7 @@ def test_token_file_rejects_symlink(tmp_path, monkeypatch): reset_hermes_home_override(override) +@pytest.mark.platforms("linux") def test_token_file_rejects_parent_escape(tmp_path, monkeypatch): home = tmp_path / "home" monkeypatch.setenv("HOME", str(home)) diff --git a/tests/hermes_cli/test_terminal_breadcrumbs.py b/tests/hermes_cli/test_terminal_breadcrumbs.py index caeeb4cd738b..da2b641a4491 100644 --- a/tests/hermes_cli/test_terminal_breadcrumbs.py +++ b/tests/hermes_cli/test_terminal_breadcrumbs.py @@ -16,6 +16,9 @@ from hermes_cli import terminal_breadcrumbs as tb +pytestmark = pytest.mark.platforms("linux") # os.ttyname is POSIX-only + + TERMINAL_ENV_VARS = ( "ZELLIJ_PANE_ID", "TMUX_PANE", diff --git a/tests/hermes_cli/test_tui_npm_install.py b/tests/hermes_cli/test_tui_npm_install.py index 49a3346722b8..7495daeaa722 100644 --- a/tests/hermes_cli/test_tui_npm_install.py +++ b/tests/hermes_cli/test_tui_npm_install.py @@ -332,7 +332,7 @@ def fail_run(*_args, **_kwargs): argv, cwd = main_mod._make_tui_argv(tmp_path, tui_dev=False) - assert argv == ["/bin/node", "--expose-gc", str(tmp_path / "dist" / "entry.js")] + assert argv[1:] == ["--expose-gc", str(tmp_path / "dist" / "entry.js")] assert cwd == tmp_path @@ -352,7 +352,7 @@ def fail_run(*_args, **_kwargs): argv, cwd = main_mod._make_tui_argv(tmp_path, tui_dev=False) - assert argv == ["/bin/node", "--expose-gc", str(tmp_path / "dist" / "entry.js")] + assert argv[1:] == ["--expose-gc", str(tmp_path / "dist" / "entry.js")] assert cwd == tmp_path @@ -382,8 +382,7 @@ def fake_run(*args, **kwargs): main_mod._make_tui_argv(tui_dir, tui_dev=False) install_cmd = calls[0][0][0] - assert install_cmd[:7] == [ - "/bin/npm", + assert install_cmd[1:7] == [ "install", "--workspace", "ui-tui", @@ -418,8 +417,7 @@ def fake_run(*args, **kwargs): main_mod._make_tui_argv(tui_dir, tui_dev=False) - assert calls[0][0][0] == [ - "/bin/npm", + assert calls[0][0][0][1:] == [ "install", "--workspace", "ui-tui", @@ -463,7 +461,7 @@ def fake_run(*args, **kwargs): main_mod._make_tui_argv(tui_dir, tui_dev=False) install_cmd = calls[0][0][0] - assert install_cmd[:2] == ["/bin/npm", "install"] + assert install_cmd[1] == "install" assert "--include=dev" in install_cmd @@ -487,7 +485,7 @@ def fake_run(*args, **kwargs): main_mod._make_tui_argv(tmp_path, tui_dev=False) assert calls - assert calls[0][0][0] == ["/bin/npm", "run", "build"] + assert calls[0][0][0][-2:] == ["run", "build"] _assert_utf8_replace_capture(calls[0][1]) @@ -514,7 +512,7 @@ def fake_run(*args, **kwargs): assert argv == [str(tsx), "src/entry.tsx"] assert cwd == tmp_path - assert calls[0][0][0] == ["/bin/npm", "run", "build"] + assert calls[0][0][0][-2:] == ["run", "build"] assert calls[0][1]["cwd"] == str(ink_dir) _assert_utf8_replace_capture(calls[0][1]) @@ -550,7 +548,7 @@ def fail_run(*_args, **_kwargs): argv, cwd = main_mod._make_tui_argv(tui_dir, tui_dev=False) - assert argv == ["/usr/bin/node", "--expose-gc", str(bundled_entry)] + assert argv[1:] == ["--expose-gc", str(bundled_entry)] assert cwd == bundled_entry.parent diff --git a/tests/hermes_cli/test_tui_resume_flow.py b/tests/hermes_cli/test_tui_resume_flow.py index 33ff8f8700f6..b3d36ccaa16c 100644 --- a/tests/hermes_cli/test_tui_resume_flow.py +++ b/tests/hermes_cli/test_tui_resume_flow.py @@ -126,7 +126,7 @@ def test_oneshot_subprocess_exits_without_teardown_abort(): ) assert result.returncode == 0 - assert result.stdout == b"ok\n" + assert result.stdout in (b"ok\n", b"ok\r\n") # Don't demand byte-empty stderr — an import-time warning from the heavy # CLI import chain shouldn't fail this. What matters is no crash traceback. assert b"Traceback" not in result.stderr @@ -281,7 +281,9 @@ def fake_run(cmd, cwd=None, **_kwargs): assert argv == [str(tsx), "src/entry.tsx"] assert cwd == tui_dir - assert calls == [(["/usr/bin/npm", "run", "build"], str(ink_dir))] + assert len(calls) == 1 + assert calls[0][0][-2:] == ["run", "build"] + assert calls[0][1] == str(ink_dir) diff --git a/tests/hermes_cli/test_uninstall_node_symlinks.py b/tests/hermes_cli/test_uninstall_node_symlinks.py index a32b062cb798..4c01ffed29d5 100644 --- a/tests/hermes_cli/test_uninstall_node_symlinks.py +++ b/tests/hermes_cli/test_uninstall_node_symlinks.py @@ -37,6 +37,7 @@ def _make_hermes_node(hermes_home: Path) -> Path: +@pytest.mark.require_symlinks def test_leaves_unrelated_symlinks_untouched(fake_home): """A node symlink the user repointed at nvm must survive uninstall.""" hermes_home = fake_home / ".hermes" @@ -64,6 +65,7 @@ def test_leaves_unrelated_symlinks_untouched(fake_home): +@pytest.mark.require_symlinks def test_removes_fhs_symlinks_in_usr_local_bin(fake_home, tmp_path, monkeypatch): """Root FHS installs place node symlinks in /usr/local/bin. diff --git a/tests/hermes_cli/test_uninstall_shell_configs.py b/tests/hermes_cli/test_uninstall_shell_configs.py index 4cc9bc1b9741..1ba87c9551f0 100644 --- a/tests/hermes_cli/test_uninstall_shell_configs.py +++ b/tests/hermes_cli/test_uninstall_shell_configs.py @@ -88,6 +88,7 @@ def boom(fd): # The aborted write must not leave a temp file behind in $HOME. assert list(fake_home.glob("*.tmp")) == [] + @pytest.mark.require_symlinks def test_symlinked_shell_config_stays_a_symlink(self, fake_home: Path): """A dotfiles-repo ``~/.zshrc`` is a symlink; replacing it with a regular file silently detaches the user's dotfiles.""" diff --git a/tests/hermes_cli/test_update_gateway_launcher_refresh.py b/tests/hermes_cli/test_update_gateway_launcher_refresh.py index be92a79a3d36..2257838a658c 100644 --- a/tests/hermes_cli/test_update_gateway_launcher_refresh.py +++ b/tests/hermes_cli/test_update_gateway_launcher_refresh.py @@ -14,7 +14,7 @@ ``_resolve_detached_python`` is a pure path helper and runs on any host. ``windowless_gateway_restart_spec`` returns its argv unchanged off Windows, -so the test that exercises the rewrite is ``windows_only`` rather than run +so the test that exercises the rewrite is ``platforms("windows")`` rather than run against a faked ``sys.platform``. """ @@ -57,13 +57,13 @@ def test_resolve_detached_python_swaps_legacy_pythonw_for_console_sibling(tmp_pa -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_restart_spec_normalizes_legacy_pythonw_argv(tmp_path): """A pre-rework Scheduled Task argv snapshot (leading pythonw.exe) must be respawned through the console python + hidden-console launch, with every argument after the interpreter preserved verbatim. - ``windows_only``: ``windowless_gateway_restart_spec`` returns the argv + ``platforms("windows")``: ``windowless_gateway_restart_spec`` returns the argv untouched off Windows, so the fake was the only thing making the rewrite (and its ``Scripts/``-layout venv derivation) run at all. """ diff --git a/tests/hermes_cli/test_update_head_moved_gate.py b/tests/hermes_cli/test_update_head_moved_gate.py index f363dac41171..5a576997bc2a 100644 --- a/tests/hermes_cli/test_update_head_moved_gate.py +++ b/tests/hermes_cli/test_update_head_moved_gate.py @@ -11,11 +11,21 @@ from types import SimpleNamespace +import os + import pytest from hermes_cli import main as hermes_main +@pytest.fixture(autouse=True) +def _isolate_venv_holders(monkeypatch): + """The update flow's venv-holder guard sees the live gateway processes on + a dev machine and aborts with SystemExit 2 before reaching the HEAD-move + gate under test. Isolate it so the test exercises the intended path.""" + monkeypatch.setattr(hermes_main, "_detect_venv_python_processes", lambda: []) + + def _make_head_moved_side_effect(pre_sha="abc123", post_sha="def456"): """Simulate git commands where HEAD advances from pre_sha to post_sha.""" calls = {"n": 0} @@ -119,6 +129,14 @@ def _patch_update_deps(monkeypatch, tmp_path, run_side_effect): monkeypatch.setattr( hermes_gateway, "find_profile_gateway_processes", lambda *a, **k: [] ) + # _restart_gateways_after_update purges and re-imports hermes_cli.gateway, + # which undoes the discovery mocks above (the fresh import finds the live + # gateways on this dev machine). Neutralize the kill itself so the phase + # is a no-op regardless of what the re-imported discovery returns. + monkeypatch.setattr(os, "kill", lambda *a, **k: None) + # ...and stop the purge from evicting the mocked module in the first place, + # so find_gateway_pids stays [] and the fleet-restart phase is a clean no-op. + monkeypatch.setattr(hermes_main, "_purge_stale_hermes_modules", lambda: None) def test_update_success_when_head_moves(monkeypatch, tmp_path, capsys): diff --git a/tests/hermes_cli/test_update_launchd_unloaded_gateway.py b/tests/hermes_cli/test_update_launchd_unloaded_gateway.py index 315fa724422a..938e679f349b 100644 --- a/tests/hermes_cli/test_update_launchd_unloaded_gateway.py +++ b/tests/hermes_cli/test_update_launchd_unloaded_gateway.py @@ -133,6 +133,7 @@ def test_no_plist_is_not_a_launchd_install(self, launchd, capsys): """ +@pytest.mark.platforms("macos") class TestServicePidSweepExclusion: """Regression for the PR #75021 review: `_get_service_pids()` must not rely on `launchctl list` alone. diff --git a/tests/hermes_cli/test_update_stale_dashboard.py b/tests/hermes_cli/test_update_stale_dashboard.py index cf3a199dc4ba..44341edd6d38 100644 --- a/tests/hermes_cli/test_update_stale_dashboard.py +++ b/tests/hermes_cli/test_update_stale_dashboard.py @@ -175,11 +175,11 @@ def _assert_ps_timeout_returns_empty(self): with patch("subprocess.run", side_effect=sp.TimeoutExpired("ps", 10)): assert _find_stale_dashboard_pids() == [] - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_ps_timeout_returns_empty_linux(self): self._assert_ps_timeout_returns_empty() - @pytest.mark.macos_only + @pytest.mark.platforms("macos") def test_ps_timeout_returns_empty_macos(self): self._assert_ps_timeout_returns_empty() @@ -268,9 +268,9 @@ def fake_run(args, *a, **kw): class TestKillStaleDashboardWindows: """Kill path on Windows: taskkill /F.""" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_taskkill_invoked_for_each_pid(self, capsys): - """``windows_only``: ``taskkill.exe`` only exists on Windows, and the + """``platforms("windows")``: ``taskkill.exe`` only exists on Windows, and the faked platform also silently skipped the POSIX-only cgroup/argv snapshot the real Windows path must not take. """ @@ -838,9 +838,9 @@ def fake_run(args, *a, **kw): assert argv == ["hermes", "serve", "--port", "8300"] - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_returns_none_on_windows(self): - """``windows_only``: the contract is "no graceful-argv capture on a + """``platforms("windows")``: the contract is "no graceful-argv capture on a real Windows host" — asserting it against a faked platform only restated the branch condition. """ diff --git a/tests/hermes_cli/test_update_wedged_gateway.py b/tests/hermes_cli/test_update_wedged_gateway.py index 71349e3e7a13..a2e1bf77c218 100644 --- a/tests/hermes_cli/test_update_wedged_gateway.py +++ b/tests/hermes_cli/test_update_wedged_gateway.py @@ -473,9 +473,13 @@ def test_unknown_liveness_keeps_full_drain_budget(self, monkeypatch): assert ("drain", 4242, 195.0) in events +@pytest.mark.platforms("linux") class TestLoopTickWitness: """Two-witness liveness (#90502 review). + Windows CPython builds don't expose AF_UNIX, so this POSIX-socket + transport class gates to Linux. + The heartbeat write moved off-loop, so a stale file no longer proves a wedged loop and a fresh file no longer proves an alive one. The loop answers a UNIX socket instead; the probe only escalates when BOTH diff --git a/tests/hermes_cli/test_update_yes_flag.py b/tests/hermes_cli/test_update_yes_flag.py index ef5a03185952..3603caf5e921 100644 --- a/tests/hermes_cli/test_update_yes_flag.py +++ b/tests/hermes_cli/test_update_yes_flag.py @@ -12,9 +12,21 @@ from types import SimpleNamespace from unittest.mock import patch +import pytest + from hermes_cli.main import cmd_update +@pytest.fixture(autouse=True) +def _isolate_venv_holders(monkeypatch): + """The update flow's venv-holder guard sees the live gateway processes on + a dev machine and aborts with SystemExit 2 before reaching the branch + logic under test. Isolate it so the test exercises the intended path.""" + import hermes_cli.main as cli_main + + monkeypatch.setattr(cli_main, "_detect_venv_python_processes", lambda: []) + + def _make_run_side_effect( branch="main", verify_ok=True, commit_count="1", dirty=False ): diff --git a/tests/hermes_cli/test_verify_core_dependencies.py b/tests/hermes_cli/test_verify_core_dependencies.py index 0404d2bb0e6a..1c8362584e4f 100644 --- a/tests/hermes_cli/test_verify_core_dependencies.py +++ b/tests/hermes_cli/test_verify_core_dependencies.py @@ -70,7 +70,7 @@ def test_skips_deps_excluded_by_environment_markers(self, temp_pyproject, fake_v verification step would false-positive on every cross-platform exclusion and chase its tail installing something inapplicable here. - Deliberately host-invariant rather than ``windows_only``: the subject + Deliberately host-invariant rather than ``platforms("windows")``: the subject is ``packaging``'s marker *evaluation*, not any OS facility. The pyproject fixture declares one dep gated to non-Windows and one gated to Windows, so exactly one of the pair is filtered on any host — the diff --git a/tests/hermes_cli/test_web_ui_build.py b/tests/hermes_cli/test_web_ui_build.py index e1b05d183bca..073e6b2d8d6c 100644 --- a/tests/hermes_cli/test_web_ui_build.py +++ b/tests/hermes_cli/test_web_ui_build.py @@ -145,7 +145,7 @@ def test_web_install_omits_workspace_and_scrubs_esbuild_override( install_cp = __import__("subprocess").CompletedProcess([], 0, stdout="", stderr="") build_cp = __import__("subprocess").CompletedProcess([], 0, stdout="", stderr="") - with patch("hermes_cli.main.shutil.which", return_value="/usr/bin/npm"), \ + with patch("hermes_cli.main._resolve_node_runtime_npm", return_value="/usr/bin/npm"), \ patch("hermes_cli.main.subprocess.run", return_value=install_cp) as mock_run, \ patch("hermes_cli.main._run_with_idle_timeout", return_value=build_cp) as mock_build: result = _build_web_ui(web_dir) @@ -178,7 +178,7 @@ def test_workspace_root_install_names_update_closure(self, tmp_path, monkeypatch install_cp = __import__("subprocess").CompletedProcess([], 0, stdout="", stderr="") build_cp = __import__("subprocess").CompletedProcess([], 0, stdout="", stderr="") - with patch("hermes_cli.main.shutil.which", return_value="/usr/bin/npm"), \ + with patch("hermes_cli.main._resolve_node_runtime_npm", return_value="/usr/bin/npm"), \ patch("hermes_cli.main.subprocess.run", return_value=install_cp) as mock_run, \ patch("hermes_cli.main._run_with_idle_timeout", return_value=build_cp): result = _build_web_ui(web_dir) @@ -201,7 +201,7 @@ def test_workspace_root_install_skips_missing_ui_tui(self, tmp_path, monkeypatch install_cp = __import__("subprocess").CompletedProcess([], 0, stdout="", stderr="") build_cp = __import__("subprocess").CompletedProcess([], 0, stdout="", stderr="") - with patch("hermes_cli.main.shutil.which", return_value="/usr/bin/npm"), \ + with patch("hermes_cli.main._resolve_node_runtime_npm", return_value="/usr/bin/npm"), \ patch("hermes_cli.main.subprocess.run", return_value=install_cp) as mock_run, \ patch("hermes_cli.main._run_with_idle_timeout", return_value=build_cp): result = _build_web_ui(web_dir) @@ -223,7 +223,7 @@ def test_web_build_uses_idle_timeout_helper(self, tmp_path): install_cp = __import__("subprocess").CompletedProcess([], 0, stdout="", stderr="") build_cp = __import__("subprocess").CompletedProcess([], 0, stdout="", stderr="") - with patch("hermes_cli.main.shutil.which", return_value="/usr/bin/npm"), \ + with patch("hermes_cli.main._resolve_node_runtime_npm", return_value="/usr/bin/npm"), \ patch("hermes_cli.main.subprocess.run", return_value=install_cp), \ patch("hermes_cli.main._run_with_idle_timeout", return_value=build_cp) as mock_idle: result = _build_web_ui(web_dir) @@ -233,7 +233,7 @@ def test_web_build_uses_idle_timeout_helper(self, tmp_path): mock_idle.assert_called_once() args, kwargs = mock_idle.call_args # Positional: [npm, "run", "build"]; cwd passed as kwarg. - assert args[0] == ["/usr/bin/npm", "run", "build"] + assert args[0][-2:] == ["run", "build"] assert kwargs["cwd"] == web_dir @@ -247,7 +247,7 @@ def test_retries_build_once_on_failure(self, tmp_path): # build attempt 1: fail; build attempt 2: success. build_fail = Subprocess.CompletedProcess([], 1, stdout="EPERM", stderr="") build_ok = Subprocess.CompletedProcess([], 0, stdout="", stderr="") - with patch("hermes_cli.main.shutil.which", return_value="/usr/bin/npm"), \ + with patch("hermes_cli.main._resolve_node_runtime_npm", return_value="/usr/bin/npm"), \ patch("hermes_cli.main._time.sleep") as mock_sleep, \ patch("hermes_cli.main.subprocess.run", return_value=install_ok), \ patch("hermes_cli.main._run_with_idle_timeout", @@ -267,7 +267,7 @@ def test_falls_back_to_stale_dist_when_retry_also_fails(self, tmp_path, capsys): Subprocess = __import__("subprocess") install_ok = Subprocess.CompletedProcess([], 0, stdout="", stderr="") build_fail = Subprocess.CompletedProcess([], 1, stdout="vite ENOMEM", stderr="") - with patch("hermes_cli.main.shutil.which", return_value="/usr/bin/npm"), \ + with patch("hermes_cli.main._resolve_node_runtime_npm", return_value="/usr/bin/npm"), \ patch("hermes_cli.main._time.sleep"), \ patch("hermes_cli.main.subprocess.run", return_value=install_ok), \ patch("hermes_cli.main._run_with_idle_timeout", @@ -282,6 +282,7 @@ def test_falls_back_to_stale_dist_when_retry_also_fails(self, tmp_path, capsys): assert "vite ENOMEM" in out # combined output surfaced to user +@pytest.mark.platforms("linux") class TestBuildWebUIFlock: """Cross-process build serialization (salvaged from PR #63455). @@ -317,7 +318,7 @@ def release_after_building(): t = threading.Timer(0.2, release_after_building) t.start() try: - with patch("hermes_cli.main.shutil.which", return_value="/usr/bin/npm"), \ + with patch("hermes_cli.main._resolve_node_runtime_npm", return_value="/usr/bin/npm"), \ patch("hermes_cli.main.subprocess.run") as mock_run: result = build(web_dir) finally: diff --git a/tests/hermes_cli/test_win_pty_bridge.py b/tests/hermes_cli/test_win_pty_bridge.py index fff0da000d47..87e6a222ab28 100644 --- a/tests/hermes_cli/test_win_pty_bridge.py +++ b/tests/hermes_cli/test_win_pty_bridge.py @@ -25,9 +25,9 @@ # must never raise, otherwise the web_server import branch becomes a trap. from hermes_cli.win_pty_bridge import PtyUnavailableError, WinPtyBridge -# ``pytest.mark.windows_only`` rather than a local ``skipif`` alias: the +# ``pytest.mark.platforms("windows")`` rather than a local ``skipif`` alias: the # dedicated Windows CI job selects its files by grepping for the marker name -# and then filters with ``-m windows_only``. A file-local skipif alias matched +# and then filters with ``-m platforms("windows")``. A file-local skipif alias matched # the grep (so the file was listed) but carried no marker, so every test below # was deselected — the lane looked like it covered ConPTY and ran none of it. @@ -81,7 +81,7 @@ def test_spawn_raises_unavailable_off_windows(self): # --------------------------------------------------------------------------- -@pytest.mark.windows_only +@pytest.mark.platforms("windows") class TestWinPtyBridgeSpawn: def test_spawn_returns_bridge_with_pid(self): @@ -98,7 +98,7 @@ def test_spawn_raises_on_missing_argv0(self, tmp_path): WinPtyBridge.spawn([bogus]) -@pytest.mark.windows_only +@pytest.mark.platforms("windows") class TestWinPtyBridgeIO: def test_write_sends_to_child_stdin(self): @@ -137,7 +137,7 @@ def test_read_returns_none_after_child_exits(self): bridge.close() -@pytest.mark.windows_only +@pytest.mark.platforms("windows") class TestWinPtyBridgeResize: def test_resize_does_not_raise_on_live_child(self): # ConPTY exposes no ioctl-equivalent for reading the child's current @@ -165,7 +165,7 @@ def test_resize_after_close_is_silent(self): bridge.resize(cols=100, rows=40) -@pytest.mark.windows_only +@pytest.mark.platforms("windows") class TestClampDimension: """The clamp helper is the load-bearing piece — the dashboard sends untrusted winsize values straight from xterm.js, and pywinpty's @@ -187,7 +187,7 @@ def test_non_numeric_falls_back_to_min(self): assert _clamp(float("inf"), _MAX_COLS) == 1 # type: ignore[arg-type] -@pytest.mark.windows_only +@pytest.mark.platforms("windows") class TestWinPtyBridgeClose: def test_close_terminates_long_running_child(self): @@ -209,7 +209,7 @@ def test_close_terminates_long_running_child(self): ) -@pytest.mark.windows_only +@pytest.mark.platforms("windows") class TestWinPtyBridgeEnv: def test_cwd_is_respected(self, tmp_path): bridge = WinPtyBridge.spawn( diff --git a/tests/hermes_cli/test_worktree_command.py b/tests/hermes_cli/test_worktree_command.py index bf9443e77182..163943e401a8 100644 --- a/tests/hermes_cli/test_worktree_command.py +++ b/tests/hermes_cli/test_worktree_command.py @@ -98,7 +98,7 @@ def test_new_outside_repo(tmp_path, monkeypatch): @requires_git def test_list_shows_worktrees(repo): out = _run(_Stub(), "/worktree list") - assert str(repo) in out + assert repo.as_posix() in out @requires_git diff --git a/tests/honcho_plugin/test_client.py b/tests/honcho_plugin/test_client.py index 630a5bf5cd13..743e19ce3e33 100644 --- a/tests/honcho_plugin/test_client.py +++ b/tests/honcho_plugin/test_client.py @@ -22,6 +22,9 @@ resolve_global_config_path, ) +import os as _os +_SYS_ENV = {k: _os.environ[k] for k in ("SYSTEMROOT", "USERPROFILE", "HOMEDRIVE", "HOMEPATH", "HOME") if k in _os.environ} + class TestHonchoClientConfigDefaults: def test_default_values(self): @@ -48,7 +51,7 @@ def test_reads_api_key_from_env(self): def test_defaults_without_env(self): - with patch.dict(os.environ, {}, clear=True): + with patch.dict(os.environ, _SYS_ENV, clear=True): # Remove HONCHO_API_KEY if it exists os.environ.pop("HONCHO_API_KEY", None) os.environ.pop("HONCHO_ENVIRONMENT", None) @@ -92,7 +95,7 @@ def test_honcho_base_url_wins_over_honcho_url(self): class TestFromGlobalConfig: def test_missing_config_falls_back_to_env(self, tmp_path): - with patch.dict(os.environ, {}, clear=True): + with patch.dict(os.environ, _SYS_ENV, clear=True): config = HonchoClientConfig.from_global_config( config_path=tmp_path / "nonexistent.json" ) @@ -108,7 +111,7 @@ def test_missing_config_still_reads_honcho_url(self, tmp_path): absent, so a fallback that only from_global_config() understood would silently do nothing for users with no ~/.honcho/config.json. """ - with patch.dict(os.environ, {"HONCHO_URL": "http://localhost:8000"}, clear=True): + with patch.dict(os.environ, {**_SYS_ENV, "HONCHO_URL": "http://localhost:8000"}, clear=True): config = HonchoClientConfig.from_global_config( config_path=tmp_path / "nonexistent.json" ) @@ -124,7 +127,7 @@ def test_base_url_from_sdk_native_endpoint_block(self, tmp_path): "endpoint": {"baseUrl": "http://localhost:8000"}, })) - with patch.dict(os.environ, {}, clear=True): + with patch.dict(os.environ, _SYS_ENV, clear=True): config = HonchoClientConfig.from_global_config(config_path=config_file) assert config.base_url == "http://localhost:8000" @@ -137,7 +140,7 @@ def test_endpoint_base_url_wins_over_top_level_and_env(self, tmp_path): "base_url": "http://localhost:9002", })) - with patch.dict(os.environ, {"HONCHO_BASE_URL": "http://localhost:9003"}, clear=True): + with patch.dict(os.environ, {**_SYS_ENV, "HONCHO_BASE_URL": "http://localhost:9003"}, clear=True): config = HonchoClientConfig.from_global_config(config_path=config_file) assert config.base_url == "http://localhost:8000" @@ -150,7 +153,7 @@ def test_endpoint_block_non_dict_is_ignored(self, tmp_path): "baseUrl": "http://localhost:9001", })) - with patch.dict(os.environ, {}, clear=True): + with patch.dict(os.environ, _SYS_ENV, clear=True): config = HonchoClientConfig.from_global_config(config_path=config_file) assert config.base_url == "http://localhost:9001" @@ -245,7 +248,7 @@ def test_base_url_full_precedence_chain(self, tmp_path): if i >= 4: env_dict.pop("HONCHO_BASE_URL") config_file.write_text(json.dumps(cfg_dict)) - with patch.dict(os.environ, env_dict, clear=True): + with patch.dict(os.environ, {**_SYS_ENV, **env_dict}, clear=True): config = HonchoClientConfig.from_global_config(config_path=config_file) assert config.base_url == want, f"layer {i}: got {config.base_url!r}, want {want!r}" @@ -711,7 +714,7 @@ def test_lan_default_host_empty_key_uses_local_placeholder(self, tmp_path): }, })) - with patch.dict(os.environ, {}, clear=True), \ + with patch.dict(os.environ, _SYS_ENV, clear=True), \ patch("hermes_cli.profiles.get_active_profile_name", return_value="default"), \ patch("plugins.memory.honcho.client.resolve_config_path", return_value=config_file): cfg = HonchoClientConfig.from_global_config(config_path=config_file) diff --git a/tests/honcho_plugin/test_oauth_flow.py b/tests/honcho_plugin/test_oauth_flow.py index f590932e0222..07317b79b1f5 100644 --- a/tests/honcho_plugin/test_oauth_flow.py +++ b/tests/honcho_plugin/test_oauth_flow.py @@ -271,7 +271,7 @@ def test_display_config_path_never_leaks_absolute_path(): # Under home → collapsed to ~/…; outside home → bare filename only. under_home = Path.home() / ".hermes" / "profiles" / "work" / "honcho.json" - assert oauth_flow._display_config_path(under_home) == "~/.hermes/profiles/work/honcho.json" + assert oauth_flow._display_config_path(under_home) == "~/" + str(Path(".hermes") / "profiles" / "work" / "honcho.json") assert oauth_flow._display_config_path("/var/folders/tmp/honcho.json") == "honcho.json" diff --git a/tests/plugins/image_gen/test_krea_provider.py b/tests/plugins/image_gen/test_krea_provider.py index 32c7ae45750f..f67e7c9b791f 100644 --- a/tests/plugins/image_gen/test_krea_provider.py +++ b/tests/plugins/image_gen/test_krea_provider.py @@ -166,7 +166,7 @@ def test_successful_generation(self): result = KreaImageGenProvider().generate(prompt="A cinematic lamp", upscale=False) assert result["success"] is True - assert result["image"] == "/tmp/krea_krea-2-medium_test.png" + assert result["image"] == str(Path("/tmp/krea_krea-2-medium_test.png")) assert result["provider"] == "krea" assert result["model"] == "krea-2-medium" assert result["aspect_ratio"] == "landscape" diff --git a/tests/plugins/image_gen/test_openai_provider.py b/tests/plugins/image_gen/test_openai_provider.py index f662b13f8dd9..8c8abd98c37b 100644 --- a/tests/plugins/image_gen/test_openai_provider.py +++ b/tests/plugins/image_gen/test_openai_provider.py @@ -236,7 +236,7 @@ def test_url_response_is_cached_locally(self, provider): result = provider.generate("a cat") assert result["success"] is True - assert result["image"].startswith("/") + assert result["image"].startswith(str(Path("/"))) assert "example.com" not in result["image"] mock_save_url.assert_called_once() diff --git a/tests/plugins/image_gen/test_openrouter_compat_provider.py b/tests/plugins/image_gen/test_openrouter_compat_provider.py index b3647376b9e4..e5c51af09ebe 100644 --- a/tests/plugins/image_gen/test_openrouter_compat_provider.py +++ b/tests/plugins/image_gen/test_openrouter_compat_provider.py @@ -325,7 +325,7 @@ def test_success_data_uri(self): result = _openrouter().generate(prompt="a pet") assert result["success"] is True - assert result["image"] == "/tmp/openrouter_gen.png" + assert result["image"] == str(Path("/tmp/openrouter_gen.png")) assert result["provider"] == "openrouter" mock_save.assert_called_once() @@ -701,7 +701,7 @@ def test_cost_and_extras_are_surfaced(self): assert result["cost_usd"] == 0.0336 assert result["total_tokens"] == 1128 assert result["exact_aspect_ratio"] == "9:16" - assert result["image"] == "/tmp/i.png" + assert result["image"] == str(Path("/tmp/i.png")) def test_multiple_images_land_in_additional_images(self): entries = [ @@ -714,8 +714,8 @@ def test_multiple_images_land_in_additional_images(self): side_effect=[Path("/tmp/a.png"), Path("/tmp/b.png")]): result = _openrouter_image_api().generate(prompt="p", model="openai/gpt-image-2") - assert result["image"] == "/tmp/a.png" - assert result["additional_images"] == ["/tmp/b.png"] + assert result["image"] == str(Path("/tmp/a.png")) + assert result["additional_images"] == [str(Path("/tmp/b.png"))] def test_empty_data_is_typed(self): with patch(_RUNTIME, return_value=_runtime_ok()), \ diff --git a/tests/plugins/memory/test_holographic_store.py b/tests/plugins/memory/test_holographic_store.py index 9fa17728410e..78c7d1485fb4 100644 --- a/tests/plugins/memory/test_holographic_store.py +++ b/tests/plugins/memory/test_holographic_store.py @@ -68,6 +68,7 @@ def test_different_paths_get_distinct_connections(self, tmp_path): a.close() b.close() + @pytest.mark.require_symlinks def test_symlinked_path_shares_connection(self, tmp_path): """A symlink to the same DB file must hit the same registry entry — otherwise two connections to one file silently reintroduce the diff --git a/tests/plugins/memory/test_openviking_provider.py b/tests/plugins/memory/test_openviking_provider.py index 315511d15657..bfff736ac7e1 100644 --- a/tests/plugins/memory/test_openviking_provider.py +++ b/tests/plugins/memory/test_openviking_provider.py @@ -1912,7 +1912,7 @@ def test_non_utf8_env_preserves_unrelated_bytes(self, tmp_path): _write_env_vars(env, {"OPENAI_API_KEY": "new"}) - assert env.read_bytes() == b"NAME=caf\xe9\nOPENAI_API_KEY=new\n" + assert env.read_bytes() == f"NAME=caf\xe9{os.linesep}OPENAI_API_KEY=new{os.linesep}".encode("latin-1") def test_plain_env_is_unchanged_apart_from_the_write(self, tmp_path): from plugins.memory.openviking import _write_env_vars diff --git a/tests/plugins/platforms/photon/test_sidecar_paths.py b/tests/plugins/platforms/photon/test_sidecar_paths.py index 66ee07feb6d6..b1f1dab7fc96 100644 --- a/tests/plugins/platforms/photon/test_sidecar_paths.py +++ b/tests/plugins/platforms/photon/test_sidecar_paths.py @@ -88,6 +88,7 @@ def test_mirror_refresh_updates_changed_files_and_keeps_node_modules( assert (mirror / "node_modules" / "installed.txt").exists() +@pytest.mark.platforms("linux") def test_dir_writable_probe(tmp_path) -> None: assert sidecar_paths.dir_writable(tmp_path) is True ro = tmp_path / "ro" diff --git a/tests/plugins/test_a2a_plugin.py b/tests/plugins/test_a2a_plugin.py index 346284b27796..678d558e7e45 100644 --- a/tests/plugins/test_a2a_plugin.py +++ b/tests/plugins/test_a2a_plugin.py @@ -1570,6 +1570,7 @@ def test_reserved_paths_and_duplicate_tenants_are_ignored(self): assert "one" in adapter._agents assert "two" not in adapter._agents + @pytest.mark.platforms("linux") def test_forward_to_profile_first_contact_creates_then_resumes_fake_hermes(self, monkeypatch, tmp_path): from plugins.platforms.a2a.adapter import A2AAdapter from gateway.config import PlatformConfig diff --git a/tests/plugins/video_gen/test_deepinfra_provider.py b/tests/plugins/video_gen/test_deepinfra_provider.py index 6902625e460c..d33fc92465a5 100644 --- a/tests/plugins/video_gen/test_deepinfra_provider.py +++ b/tests/plugins/video_gen/test_deepinfra_provider.py @@ -10,6 +10,7 @@ from __future__ import annotations from contextlib import contextmanager +from pathlib import Path from types import SimpleNamespace from unittest.mock import MagicMock, patch @@ -118,7 +119,7 @@ def test_generate_text_to_video_downloads_url_and_saves_locally(): ) assert result["success"] is True assert result["modality"] == "text" - assert result["video"].endswith(".mp4") and "cache/videos" in result["video"] + assert result["video"].endswith(".mp4") and str(Path("cache") / "videos") in result["video"] assert captured["url"] == "https://cdn.example/out.mp4" assert "deepinfra" in captured["base_url"] assert captured["api_key"] == "test-key" diff --git a/tests/run_agent/test_compression_boundary_hook.py b/tests/run_agent/test_compression_boundary_hook.py index 2595c0ee646d..88c0828d5d35 100644 --- a/tests/run_agent/test_compression_boundary_hook.py +++ b/tests/run_agent/test_compression_boundary_hook.py @@ -97,6 +97,7 @@ def test_on_session_start_called_with_compression_boundary(self): assert call.kwargs.get("old_session_id") == original_sid, \ f"Expected old_session_id={original_sid!r}, got {call.kwargs!r}" assert len(comp_calls) == 1 + db.close() def test_automatic_notification_follows_core_persistence(self): from hermes_state import SessionDB @@ -137,6 +138,7 @@ def _record_publish(*args, **kwargs): ) assert events == ["persist", "compression"] + db.close() def test_failure_before_persistence_does_not_notify(self): from hermes_state import SessionDB @@ -156,6 +158,7 @@ def test_failure_before_persistence_does_not_notify(self): ) compressor.on_session_start.assert_not_called() + db.close() def test_no_progress_does_not_notify(self): @@ -178,6 +181,7 @@ def test_no_progress_does_not_notify(self): assert returned is messages compressor.on_session_start.assert_not_called() + db.close() def test_no_hook_when_no_session_db(self): @@ -251,6 +255,7 @@ def _raise_on_compression(*args, **kwargs): ) assert compressed assert agent.session_id != original_sid + db.close() class TestSessionCompressEvent: @@ -311,6 +316,7 @@ def test_event_emitted_on_compression(self): assert ctx["session_id"] == agent.session_id assert ctx["old_session_id"] == original_sid assert ctx["compression_count"] == 1 + db.close() def test_no_callback_is_safe(self): """Compression must work when no event_callback is wired.""" @@ -324,4 +330,5 @@ def test_no_callback_is_safe(self): [{"role": "user", "content": "m"}], "sys", approx_tokens=100 ) assert compressed + db.close() diff --git a/tests/run_agent/test_compression_persistence.py b/tests/run_agent/test_compression_persistence.py index a99448d1cbb2..340f4c610800 100644 --- a/tests/run_agent/test_compression_persistence.py +++ b/tests/run_agent/test_compression_persistence.py @@ -99,6 +99,7 @@ def test_flush_after_compression_with_long_history(self): f"Expected 5 compressed messages in new session, got {len(new_rows)}. " f"Compression persistence bug: messages not written to SQLite." ) + db.close() def test_flush_with_stale_history_loses_messages(self): """Stale conversation_history no longer causes data loss.""" @@ -128,6 +129,7 @@ def test_flush_with_stale_history_loses_messages(self): rows = db.get_messages("new-session") assert len(rows) == 2 assert [row["content"] for row in rows] == ["summary", "continuing..."] + db.close() def test_in_place_compression_rebaseline_prevents_duplicate_compacted_rows(self): """In-place compaction already persisted the compacted transcript. @@ -190,6 +192,7 @@ def test_in_place_compression_rebaseline_prevents_duplicate_compacted_rows(self) "tool result", "final answer", ] + db.close() def test_abort_after_in_place_compaction_preserves_flush_baseline(self): """An aborted retry must survive flush, restart, and resume.""" @@ -428,7 +431,7 @@ def test_stored_prompt_fresh_when_cwd_matches(self): from agent.conversation_loop import _stored_prompt_matches_runtime agent = self._make_agent() - current_cwd = "/project/current" + current_cwd = str(Path("/project/current")) stored_prompt = ( self._host_block(current_cwd) + "Model: test/model\n" @@ -456,7 +459,7 @@ def test_project_context_cannot_force_a_rebuild(self): from agent.conversation_loop import _stored_prompt_matches_runtime agent = self._make_agent() - current_cwd = "/project/current" + current_cwd = str(Path("/project/current")) stored_prompt = ( self._host_block(current_cwd) + "\n# AGENTS.md\n\n" @@ -531,3 +534,4 @@ def test_built_prompt_contains_platform_line(self): assert "Platform: cli" in parts["volatile"], ( "Built prompt missing 'Platform: cli' — drift detection cannot read it" ) + db.close() diff --git a/tests/run_agent/test_in_place_compaction.py b/tests/run_agent/test_in_place_compaction.py index 93d8236151c2..22289607811d 100644 --- a/tests/run_agent/test_in_place_compaction.py +++ b/tests/run_agent/test_in_place_compaction.py @@ -123,6 +123,7 @@ def test_in_place_keeps_same_session_id(self): assert agent._last_compaction_in_place is True # Live transcript actually shrank. assert len(compressed) == 2 + db.close() def test_in_place_alternation_preserved(self): """The compacted list must not introduce consecutive same-role messages.""" @@ -140,6 +141,7 @@ def test_in_place_alternation_preserved(self): ) roles = [m["role"] for m in compressed if m.get("role") != "system"] assert all(roles[i] != roles[i + 1] for i in range(len(roles) - 1)) + db.close() def test_rotation_still_preflushes(self): @@ -161,6 +163,7 @@ def test_rotation_still_preflushes(self): approx_tokens=100_000, system_message="sys", ) assert calls["n"] == 1 + db.close() class TestRotationFallbackWhenFlagOff: @@ -205,6 +208,7 @@ def test_rotation_when_flag_off(self): ] # Rotation mode does NOT set the in-place signal. assert getattr(agent, "_last_compaction_in_place", False) is False + db.close() class TestInPlaceSignalForGateway: @@ -234,6 +238,7 @@ def test_signal_set_on_in_place_unset_on_rotation(self): approx_tokens=100_000, system_message="sys", ) assert a_rot._last_compaction_in_place is False + db.close() class TestInPlaceConfigDefault: @@ -292,6 +297,7 @@ def _growing_compress(messages, current_tokens=None, focus_topic=None, force=Fal # Session identity untouched. assert agent.session_id == sid assert db.get_session(sid)["end_reason"] is None + db.close() def test_in_place_salvages_near_break_even_growth(self): """Fat retained tool output + todo state should be salvaged and committed.""" @@ -348,6 +354,7 @@ def test_in_place_salvages_near_break_even_growth(self): # Tool stubbing alone got under budget, so the todo snapshot (the # only in-transcript todo re-injection) survives the salvage. assert any(m.get("_todo_snapshot_synthetic") for m in compressed) + db.close() def test_in_place_still_commits_shrinking_compression(self): """The guard must not block legitimate compressions — a result SMALLER @@ -374,6 +381,7 @@ def test_in_place_still_commits_shrinking_compression(self): "[CONTEXT COMPACTION] summary of prior turns", "recent reply", ] + db.close() class TestCompactedTurnsStaySearchable: @@ -417,6 +425,7 @@ def test_compacted_turns_found_by_default_search(self): assert {m["id"] for m in after} == {1, 4} # Live context still excludes them. assert len(db.get_messages_as_conversation(sid)) == 2 + db.close() def test_rewound_turns_stay_hidden(self): """Rewind/undo (active=0, compacted=0) must NOT leak into default @@ -436,3 +445,4 @@ def test_rewound_turns_stay_hidden(self): "ZEBRAWORD", role_filter=["user", "assistant"], include_inactive=True ) assert len(recovered) == 1 + db.close() diff --git a/tests/run_agent/test_tool_batch_segmentation.py b/tests/run_agent/test_tool_batch_segmentation.py index 6f367878b915..da0d858c0683 100644 --- a/tests/run_agent/test_tool_batch_segmentation.py +++ b/tests/run_agent/test_tool_batch_segmentation.py @@ -667,6 +667,7 @@ def test_relative_and_absolute_same_target_use_separate_segments(self, tmp_path) "Absolute and relative paths pointing to the same file must overlap" ) + @pytest.mark.require_symlinks def test_symlink_aliases_are_not_parallelized(self, tmp_path): """A symlink alias and the real path must be detected as overlapping so they are never placed in the same parallel segment.""" @@ -724,10 +725,10 @@ def test_execution_cwd_used_over_process_cwd(self, tmp_path, monkeypatch): ) - # ``windows_only`` rather than ``skipif(sys.platform != "win32")``: the + # ``platforms("windows")`` rather than ``skipif(sys.platform != "win32")``: the # Windows CI job greps for the marker to decide which files to import, so # a bare skipif leaves this running on no host at all. - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_case_insensitive_paths_overlap_windows(self, tmp_path): """On Windows, FILE.txt and file.txt are the same file — they must be detected as overlapping after normcase() canonicalisation.""" diff --git a/tests/run_agent/test_tool_call_guardrail_runtime.py b/tests/run_agent/test_tool_call_guardrail_runtime.py index ca6e80aac4c4..83f327e2cb22 100644 --- a/tests/run_agent/test_tool_call_guardrail_runtime.py +++ b/tests/run_agent/test_tool_call_guardrail_runtime.py @@ -2,6 +2,7 @@ import json import uuid +from pathlib import Path from types import SimpleNamespace from unittest.mock import MagicMock, patch @@ -293,7 +294,7 @@ def dispatch(name, args, task_id, **kwargs): assert observed["start"] == expected assert observed["dispatch"] == expected assert observed["checkpoint"] == [ - ("/approved/path", "before write_file") + (str(Path("/approved/path")), "before write_file") ] diff --git a/tests/skills/test_openclaw_migration.py b/tests/skills/test_openclaw_migration.py index bfd69cc2d64e..c456844fc09f 100644 --- a/tests/skills/test_openclaw_migration.py +++ b/tests/skills/test_openclaw_migration.py @@ -5,6 +5,8 @@ import sys from pathlib import Path +import pytest + SCRIPT_PATH = ( Path(__file__).resolve().parents[2] @@ -247,6 +249,7 @@ def test_absent_config_is_still_created(tmp_path: Path): assert "anthropic/claude-sonnet-4" in config_path.read_text(encoding="utf-8") +@pytest.mark.require_symlinks def test_symlinked_config_stays_a_symlink(tmp_path: Path): """Managed deployments symlink ~/.hermes/config.yaml into a dotfiles repo. diff --git a/tests/state/test_fts_runtime_rebuild.py b/tests/state/test_fts_runtime_rebuild.py index 8779186e72c2..f8dec25d343f 100644 --- a/tests/state/test_fts_runtime_rebuild.py +++ b/tests/state/test_fts_runtime_rebuild.py @@ -183,6 +183,7 @@ def process_iter(_attrs): (222, f"{db_path}-wal (deleted)") ] + @pytest.mark.platforms("linux") def test_foreign_holder_detection_proc_readlink_deleted_wal( self, db, tmp_path, monkeypatch ): @@ -227,6 +228,7 @@ def _readlink(path): holders = db._foreign_state_db_holders() assert holders == [(222, db_path_wal + " (deleted)")] + @pytest.mark.platforms("linux") def test_foreign_holder_uninspectable_process_cmdline_fallback( self, db, tmp_path, monkeypatch ): diff --git a/tests/test_atomic_replace_symlinks.py b/tests/test_atomic_replace_symlinks.py index 42c85568892d..d3fd4b575fbb 100644 --- a/tests/test_atomic_replace_symlinks.py +++ b/tests/test_atomic_replace_symlinks.py @@ -224,6 +224,7 @@ def test_atomic_replace_broken_symlink_creates_target(tmp_path: Path) -> None: +@pytest.mark.require_symlinks def test_atomic_replace_copy_fallback_preserves_symlink( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: @@ -289,7 +290,7 @@ def test_atomic_replace_real_cross_device(tmp_path: Path) -> None: # *source* reports. The cross-platform tests below therefore simulate # winerror 5, matching what production actually raises. # -# The real-handle tests use @pytest.mark.windows_only rather than a bare +# The real-handle tests use @pytest.mark.platforms("windows") rather than a bare # `skip(os.name != "nt")`: scripts/ci/list_os_marked_tests.py greps for the # MARKER NAME to decide which files the Windows lane imports, so a plain # skipif would leave them running on no host at all. @@ -506,6 +507,7 @@ def test_in_place_rewrite_never_exposes_a_truncated_file( assert observed == [5000] +@pytest.mark.require_symlinks def test_symlinked_target_survives_a_contended_rename( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, fast_replace_retries: None ) -> None: @@ -531,7 +533,7 @@ def test_symlinked_target_survives_a_contended_rename( # ── native Windows: real contended handles ──────────────────────────────── -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_windows_real_held_read_handle_lands_the_write(tmp_path: Path) -> None: """The reported bug, end to end against a real held handle.""" target = tmp_path / "gateway_state.json" @@ -545,7 +547,7 @@ def test_windows_real_held_read_handle_lands_the_write(tmp_path: Path) -> None: assert not tmp.exists() -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_windows_real_held_handle_reports_access_denied(tmp_path: Path) -> None: """Pin the premise this fix is built on: a held *target* handle raises winerror 5, not 32. If CPython ever changes that, the classification in @@ -565,7 +567,7 @@ def test_windows_real_held_handle_reports_access_denied(tmp_path: Path) -> None: assert utils_mod._is_contended_windows_replace_error(caught.value) -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_windows_atomic_json_write_with_concurrent_reader( tmp_path: Path, ) -> None: @@ -582,7 +584,7 @@ def test_windows_atomic_json_write_with_concurrent_reader( assert leftovers == [], f"orphaned temp files: {leftovers}" -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_windows_readonly_target_still_raises(tmp_path: Path) -> None: """A genuinely unwritable target must not be rescued by the fallback.""" import subprocess diff --git a/tests/test_bitwarden_secrets.py b/tests/test_bitwarden_secrets.py index 420605416c9f..ef197e79d400 100644 --- a/tests/test_bitwarden_secrets.py +++ b/tests/test_bitwarden_secrets.py @@ -135,6 +135,7 @@ def test_safe_extract_member_rejects_traversal(tmp_path, evil_name): +@pytest.mark.platforms("linux") def test_install_bws_happy_path(hermes_home, monkeypatch): fake_binary = b"#!/bin/sh\necho 'bws fake 2.0.0'\n" zip_bytes = _make_fake_zip(fake_binary) @@ -334,6 +335,7 @@ def fake_run(*a, **kw): +@pytest.mark.platforms("linux") def test_encrypted_cache_writes_without_plaintext(monkeypatch, tmp_path): """Encrypted cache stores last-good secrets without raw values on disk.""" home = tmp_path / ".hermes" diff --git a/tests/test_desktop_update_windows_pipe_drain.py b/tests/test_desktop_update_windows_pipe_drain.py index d3cd1e8ee809..61beb2274d03 100644 --- a/tests/test_desktop_update_windows_pipe_drain.py +++ b/tests/test_desktop_update_windows_pipe_drain.py @@ -39,7 +39,7 @@ So the contract is: bounded when a descendant holds the pipe open, never slower than the step can write, and bounded when the step itself remains alive without observable progress. All arms live in the script's own -``-SelfTestPipeDrain`` fixture, which is ``windows_only`` because Linux CI +``-SelfTestPipeDrain`` fixture, which is ``platforms("windows")`` because Linux CI cannot execute the PowerShell hand-off. """ @@ -68,7 +68,7 @@ class TestIdleWatchdogCountsUpdateLogGrowth: exit 124. These are source-contract assertions (the executable proof is the - ``logstall`` arm of ``-SelfTestPipeDrain``, ``windows_only`` below): + ``logstall`` arm of ``-SelfTestPipeDrain``, ``platforms("windows")`` below): Linux CI cannot run the PowerShell hand-off, but it CAN pin that the drain loop consults update-log growth before terminating the tree. Sabotage-proof: removing the ``Get-StepProgressLogStamp`` consult from @@ -124,7 +124,7 @@ def test_self_test_has_silent_but_logging_arm(self): assert "silent but logging" in src -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_update_step_survives_pipe_leak_flood_and_live_child_stall( tmp_path: Path, ) -> None: diff --git a/tests/test_desktop_update_windows_progress.py b/tests/test_desktop_update_windows_progress.py index 031971570ec5..38e1a7fa57a2 100644 --- a/tests/test_desktop_update_windows_progress.py +++ b/tests/test_desktop_update_windows_progress.py @@ -20,7 +20,7 @@ import pytest -pytestmark = pytest.mark.windows_only +pytestmark = pytest.mark.platforms("windows") REPO_ROOT = Path(__file__).resolve().parent.parent WINDOWS_UPDATE_PS1 = REPO_ROOT / "scripts" / "desktop-update" / "windows.ps1" diff --git a/tests/test_desktop_update_windows_retry_policy.py b/tests/test_desktop_update_windows_retry_policy.py index 8fafbca24c53..68e86c6ff5a7 100644 --- a/tests/test_desktop_update_windows_retry_policy.py +++ b/tests/test_desktop_update_windows_retry_policy.py @@ -13,7 +13,7 @@ RETRY_POLICY = REPO_ROOT / "scripts" / "desktop-update" / "retry-policy.ps1" -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_retry_policy_distinguishes_self_lock_deferral(tmp_path: Path) -> None: install_root = tmp_path / "hermes-agent" install_root.mkdir() diff --git a/tests/test_hermes_bootstrap.py b/tests/test_hermes_bootstrap.py index b95b50d9ad0e..d9f2035c7f79 100644 --- a/tests/test_hermes_bootstrap.py +++ b/tests/test_hermes_bootstrap.py @@ -46,7 +46,7 @@ def _fresh_import(): class TestWindowsBehavior: """Windows: the bootstrap does its job.""" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_env_vars_set_on_windows(self, monkeypatch): # Clear any pre-existing values and re-run bootstrap. monkeypatch.delenv("PYTHONUTF8", raising=False) @@ -57,7 +57,7 @@ def test_env_vars_set_on_windows(self, monkeypatch): assert os.environ.get("PYTHONIOENCODING") == "utf-8" assert hb._bootstrap_applied is True - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_stdout_reconfigured_to_utf8_on_windows(self): # The live process's stdout should now be UTF-8 (the Hermes CLI # runs on Windows with a pytest console that's cp1252 by default). @@ -77,7 +77,7 @@ def test_stdout_reconfigured_to_utf8_on_windows(self): "reconfigured it to UTF-8" ) - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_child_process_inherits_utf8_mode(self): """A subprocess spawned from this process should inherit PYTHONUTF8=1 and be able to print non-ASCII to stdout.""" @@ -110,7 +110,7 @@ class TestUserOptOut: """If the user has explicitly set PYTHONUTF8 / PYTHONIOENCODING in their environment, we respect that (setdefault, not overwrite).""" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_user_pythonutf8_zero_preserved(self, monkeypatch): monkeypatch.setenv("PYTHONUTF8", "0") _fresh_import() @@ -125,6 +125,7 @@ class TestPosixNoOp: stdio. The goal is that Linux/macOS behave identically before and after this module is imported.""" + @pytest.mark.platforms("linux") def test_noop_on_posix_host(self, monkeypatch): """Even when imported, the bootstrap function must return False and leave env untouched on a POSIX host (``_IS_WINDOWS`` is @@ -163,12 +164,12 @@ class TestStdioReconfigureErrorHandling: don't support reconfigure (e.g. by a test harness), the bootstrap must degrade gracefully rather than crash.""" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_non_reconfigurable_stream_does_not_crash(self, monkeypatch): """Replace sys.stdout with a BytesIO (no reconfigure method), then run the bootstrap and make sure it doesn't raise. - ``windows_only``: forcing ``_IS_WINDOWS = True`` on Linux was the only + ``platforms("windows")``: forcing ``_IS_WINDOWS = True`` on Linux was the only thing that made the reconfigure block reachable — off Windows the bootstrap returns before touching stdio, so the test proved nothing about the guard it names. @@ -336,6 +337,7 @@ def test_env_var_used_when_no_arg(self): class TestSuppressPlatformVerConsole: """suppress_platform_ver_console: stub applied on Windows, no-op on POSIX.""" + @pytest.mark.platforms("linux") def test_noop_on_posix(self): import platform hb = _fresh_import() @@ -343,7 +345,7 @@ def test_noop_on_posix(self): hb.suppress_platform_ver_console() assert getattr(platform, "_syscmd_ver", None) is original - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_stub_applied_when_windows(self): # Faking _IS_WINDOWS on Linux asserted only that the stub was # installed; the reason it exists — ``platform.win32_ver()`` shelling diff --git a/tests/test_hermes_constants.py b/tests/test_hermes_constants.py index e5bb62947b1c..56985cd778f6 100644 --- a/tests/test_hermes_constants.py +++ b/tests/test_hermes_constants.py @@ -33,6 +33,7 @@ class TestGetDefaultHermesRoot: """Tests for get_default_hermes_root() — Docker/custom deployment awareness.""" + @pytest.mark.platforms("linux") def test_no_hermes_home_returns_native(self, tmp_path, monkeypatch): """When HERMES_HOME is not set, returns ~/.hermes.""" monkeypatch.delenv("HERMES_HOME", raising=False) @@ -54,7 +55,7 @@ def test_docker_profile_active(self, tmp_path, monkeypatch): monkeypatch.setenv("HERMES_HOME", str(profile)) assert get_default_hermes_root() == docker_root - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_no_hermes_home_returns_localappdata_root_on_windows(self, tmp_path, monkeypatch): """Native Windows falls back to %LOCALAPPDATA%\\hermes, not ~/.hermes.""" local_appdata = tmp_path / "LocalAppData" @@ -127,7 +128,7 @@ def counting_resolve(self, *a, **k): class TestGetHermesHome: """Tests for get_hermes_home() platform-aware fallback.""" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_fallback_uses_localappdata(self, tmp_path, monkeypatch): """When HERMES_HOME is unset on Windows, use %LOCALAPPDATA%\\hermes.""" local_appdata = tmp_path / "LocalAppData" @@ -156,7 +157,7 @@ def test_env_set_returns_that_path(self, tmp_path, monkeypatch): class TestHermesManagedNode: - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_node_dir_prefers_portable_root(self, tmp_path, monkeypatch): home = tmp_path / "hermes" node_dir = home / "node" @@ -167,7 +168,7 @@ def test_windows_node_dir_prefers_portable_root(self, tmp_path, monkeypatch): assert iter_hermes_node_dirs() == [node_dir, bin_dir] - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_finds_npm_cmd_before_path(self, tmp_path, monkeypatch): home = tmp_path / "hermes" node_dir = home / "node" @@ -181,7 +182,7 @@ def test_windows_finds_npm_cmd_before_path(self, tmp_path, monkeypatch): - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_skips_broken_managed_npm_without_path_fallback(self, tmp_path, monkeypatch): home = tmp_path / "hermes" managed_npm = home / "node" / "npm.cmd" @@ -725,6 +726,7 @@ def test_legacy_is_file_treated_as_content(self, tmp_path, monkeypatch): + @pytest.mark.require_symlinks def test_dangling_legacy_symlink_returns_new(self, tmp_path, monkeypatch): """A dangling legacy symlink must NOT shadow populated new-layout data. @@ -743,6 +745,7 @@ def test_dangling_legacy_symlink_returns_new(self, tmp_path, monkeypatch): result = get_hermes_dir("platforms/pairing", "pairing") assert result == new + @pytest.mark.require_symlinks def test_symlink_to_populated_dir_returns_legacy(self, tmp_path, monkeypatch): """A legacy symlink pointing at a populated directory is honoured.""" self._set_home(tmp_path, monkeypatch) diff --git a/tests/test_hermes_home_profile_warning.py b/tests/test_hermes_home_profile_warning.py index 5ece51bc5c3b..c369d75a1341 100644 --- a/tests/test_hermes_home_profile_warning.py +++ b/tests/test_hermes_home_profile_warning.py @@ -30,6 +30,7 @@ def fresh_constants(monkeypatch, tmp_path): class TestGetHermesHomeProfileWarning: + @pytest.mark.platforms("linux") def test_classic_mode_no_active_profile_no_warning( self, fresh_constants, tmp_path, capsys ): @@ -39,6 +40,7 @@ def test_classic_mode_no_active_profile_no_warning( assert "HERMES_HOME fallback" not in capsys.readouterr().err + @pytest.mark.platforms("linux") def test_named_profile_unset_home_warns_once( self, fresh_constants, tmp_path, capsys ): @@ -77,6 +79,7 @@ def test_hermes_home_set_suppresses_warning( assert result == profile_dir assert "HERMES_HOME fallback" not in capsys.readouterr().err + @pytest.mark.platforms("linux") def test_unreadable_active_profile_no_crash( self, fresh_constants, tmp_path, capsys ): diff --git a/tests/test_hermes_logging.py b/tests/test_hermes_logging.py index bd87d9334031..0eff810c0a4e 100644 --- a/tests/test_hermes_logging.py +++ b/tests/test_hermes_logging.py @@ -382,6 +382,7 @@ def test_no_session_filter_on_handler(self, tmp_path): logger.removeHandler(h) h.close() + @pytest.mark.platforms("linux") def test_managed_mode_initial_open_sets_group_writable(self, tmp_path): log_path = tmp_path / "managed-open.log" logger = logging.getLogger("_test_rotating_managed_open") @@ -424,7 +425,7 @@ def _make_logger_and_handler(self, log_path: Path): logger.addHandler(handler) return logger, handler - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_helper_only_matches_windows_concurrent_lock_timeout(self): # Windows-only: concurrent-log-handler (and therefore its cross-process # lock timeout) is only installed on Windows — faking sys.platform @@ -436,7 +437,7 @@ def test_helper_only_matches_windows_concurrent_lock_timeout(self): RuntimeError("some other logging failure") ) - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_helper_never_matches_off_windows(self): # On POSIX the suppression must stay inert: stdlib RotatingFileHandler # is in use, so this RuntimeError text is never a CLH lock timeout. @@ -444,7 +445,7 @@ def test_helper_never_matches_off_windows(self): RuntimeError("Cannot acquire lock after 20 attempts") ) - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_lock_timeout_routed_to_handle_error_is_suppressed(self, tmp_path, capsys): """Mirror CLH's real control flow. diff --git a/tests/test_hermes_state.py b/tests/test_hermes_state.py index 28480c1d1548..bfc0c7762996 100644 --- a/tests/test_hermes_state.py +++ b/tests/test_hermes_state.py @@ -813,17 +813,20 @@ def test_search_fields_project_results_without_changing_default(self, db): ] assert all("context" in row and row["context"] for row in default) - def test_search_projection_skips_context_enrichment_queries(self, db): + def test_search_projection_skips_context_enrichment_queries(self, db, monkeypatch): + # Force the single-connection (non-WAL) read path so the trace callback + # on db._conn observes every statement. Under WAL the read pool hands a + # *different* pooled connection to each _read_ctx() checkout, so the + # enrichment query would run on a connection this test never traces — + # the assertion only holds when reads fall back to the writer conn. + monkeypatch.setattr(db, "_wal_active", False) db.create_session(session_id="s1", source="cli") db.append_message("s1", role="user", content="before") db.append_message("s1", role="assistant", content="projectionneedle") db.append_message("s1", role="user", content="after") statements = [] - read_conn = db._get_read_conn() or db._conn traced_connections = [db._conn] - if read_conn is not db._conn: - traced_connections.append(read_conn) for conn in traced_connections: conn.set_trace_callback(statements.append) diff --git a/tests/test_install_autostash_conflict_recovery.py b/tests/test_install_autostash_conflict_recovery.py index 94e19aeec598..bd2f0561ded1 100644 --- a/tests/test_install_autostash_conflict_recovery.py +++ b/tests/test_install_autostash_conflict_recovery.py @@ -79,6 +79,7 @@ def _assert_conflict_was_recovered(repo: Path, output: str) -> None: @pytest.mark.live_system_guard_bypass +@pytest.mark.platforms("linux") @pytest.mark.skipif( shutil.which("git") is None or shutil.which("bash") is None, reason="needs git and bash", @@ -137,6 +138,7 @@ def test_install_ps1_repository_stage_recovers_from_autostash_conflict( @pytest.mark.live_system_guard_bypass +@pytest.mark.platforms("linux") @pytest.mark.skipif( shutil.which("git") is None or shutil.which("bash") is None, reason="needs git and bash", diff --git a/tests/test_install_macos_launcher.py b/tests/test_install_macos_launcher.py index fd46c347a7a3..200c16a85ff5 100644 --- a/tests/test_install_macos_launcher.py +++ b/tests/test_install_macos_launcher.py @@ -9,6 +9,10 @@ import subprocess from pathlib import Path +import pytest + +pytestmark = pytest.mark.platforms("macos") + REPO_ROOT = Path(__file__).resolve().parent.parent INSTALL_SH = REPO_ROOT / "scripts" / "install.sh" diff --git a/tests/test_install_sh_acp_launcher.py b/tests/test_install_sh_acp_launcher.py index 27cb4c8f1aea..ec394bbfd6f9 100644 --- a/tests/test_install_sh_acp_launcher.py +++ b/tests/test_install_sh_acp_launcher.py @@ -16,6 +16,8 @@ from pathlib import Path from unittest.mock import patch +import pytest + INSTALL_SH = Path(__file__).resolve().parent.parent / "scripts" / "install.sh" ACP_BLOCK = re.compile( @@ -65,6 +67,7 @@ def _run_block(tmp_path: Path, use_venv: str) -> Path: return command_link_dir / "hermes-acp" +@pytest.mark.platforms("linux") def test_venv_install_writes_executable_acp_launcher(tmp_path): shim = _run_block(tmp_path, "true") assert shim.is_file() @@ -78,12 +81,14 @@ def test_venv_install_writes_executable_acp_launcher(tmp_path): assert re.search(r'exec .*\bacp\b', text), text +@pytest.mark.platforms("linux") def test_non_venv_install_writes_acp_launcher(tmp_path): shim = _run_block(tmp_path, "false") text = shim.read_text(encoding="utf-8") assert re.search(r'exec .*\bacp\b', text), text +@pytest.mark.platforms("linux") def test_acp_launcher_does_not_follow_a_symlink_into_the_venv(tmp_path): """Guards the #21454 failure mode for the new launcher. @@ -185,6 +190,7 @@ def _run_hermes_agent_block(tmp_path: Path, use_venv: str) -> Path | None: return command_link_dir / "hermes-agent" +@pytest.mark.platforms("linux") def test_venv_install_writes_executable_hermes_agent_launcher(tmp_path): """venv install must write a user-executable hermes-agent launcher.""" shim = _run_hermes_agent_block(tmp_path, "true") diff --git a/tests/test_install_sh_bootstrap_marker.py b/tests/test_install_sh_bootstrap_marker.py index 6eab5d6599de..7ef97f8d3764 100644 --- a/tests/test_install_sh_bootstrap_marker.py +++ b/tests/test_install_sh_bootstrap_marker.py @@ -16,6 +16,8 @@ import pytest +pytestmark = pytest.mark.platforms("linux") + REPO_ROOT = Path(__file__).resolve().parent.parent INSTALL_SH = REPO_ROOT / "scripts" / "install.sh" diff --git a/tests/test_install_sh_node_deps_failure.py b/tests/test_install_sh_node_deps_failure.py index c42e0cb32ca4..d009e5f8ee42 100644 --- a/tests/test_install_sh_node_deps_failure.py +++ b/tests/test_install_sh_node_deps_failure.py @@ -7,6 +7,10 @@ import subprocess from pathlib import Path +import pytest + +pytestmark = pytest.mark.platforms("linux") + REPO_ROOT = Path(__file__).resolve().parent.parent INSTALL_SH = REPO_ROOT / "scripts" / "install.sh" diff --git a/tests/test_install_sh_symlink_stomp.py b/tests/test_install_sh_symlink_stomp.py index 0fbe508509ff..7a35522e0e41 100644 --- a/tests/test_install_sh_symlink_stomp.py +++ b/tests/test_install_sh_symlink_stomp.py @@ -20,6 +20,8 @@ import subprocess from pathlib import Path +import pytest + REPO_ROOT = Path(__file__).resolve().parent.parent @@ -57,6 +59,7 @@ def test_setup_path_shim_block_removes_old_link_before_writing() -> None: ) +@pytest.mark.platforms("linux") def test_re_running_setup_path_block_preserves_pip_entry_point(tmp_path: Path) -> None: """Behavioral repro: simulate prior-install symlink + new-install heredoc. diff --git a/tests/test_iron_proxy.py b/tests/test_iron_proxy.py index cdd563041b7b..33e00abef4a0 100644 --- a/tests/test_iron_proxy.py +++ b/tests/test_iron_proxy.py @@ -341,6 +341,7 @@ def test_subprocess_env_strips_unrelated_secrets(hermes_home, monkeypatch): # --------------------------------------------------------------------------- +@pytest.mark.platforms("linux") def test_ca_key_created_with_0o600(hermes_home, monkeypatch): """The CA private key must NEVER exist on disk with default umask permissions, even transiently. Fix: open with explicit mode=0o600 @@ -376,6 +377,7 @@ def fake_run(args, **kwargs): # --------------------------------------------------------------------------- +@pytest.mark.platforms("linux") def test_ensure_audit_log_creates_with_0o600(hermes_home, tmp_path): audit = tmp_path / "audit.log" ip.ensure_audit_log(audit) @@ -384,6 +386,7 @@ def test_ensure_audit_log_creates_with_0o600(hermes_home, tmp_path): assert mode == 0o600 +@pytest.mark.platforms("linux") def test_ensure_audit_log_tightens_existing_perms(hermes_home, tmp_path): audit = tmp_path / "audit.log" audit.write_text("preexisting content\n") @@ -398,6 +401,7 @@ def test_ensure_audit_log_tightens_existing_perms(hermes_home, tmp_path): # --------------------------------------------------------------------------- +@pytest.mark.platforms("linux") def test_proxy_state_dir_is_0o700(hermes_home): state = ip._proxy_state_dir() mode = state.stat().st_mode & 0o777 @@ -485,6 +489,7 @@ def test_mappings_roundtrip_preserves_headers_and_aliases(hermes_home): +@pytest.mark.platforms("linux") def test_ensure_management_token_persists_and_is_stable(hermes_home): t1 = ip.ensure_management_token() t2 = ip.ensure_management_token() @@ -539,6 +544,7 @@ def fake_urlopen(req, timeout=None): assert captured["auth"] == f"Bearer {token}" +@pytest.mark.platforms("linux") def test_start_proxy_injects_management_key_env(hermes_home, monkeypatch): """When the generated config has a management listener, start_proxy must inject the bearer key env var — v0.39 refuses to start when diff --git a/tests/test_live_system_guard_self_test.py b/tests/test_live_system_guard_self_test.py index 40e9e6b5e1a4..aa44d7205982 100644 --- a/tests/test_live_system_guard_self_test.py +++ b/tests/test_live_system_guard_self_test.py @@ -105,6 +105,7 @@ def test_fail_closed_probe_classifies_raw_builtin_as_unguarded(): # ──────────────────── kill primitives ───────────────────────── +@pytest.mark.platforms("linux") def test_os_kill_blocks_foreign_pid(): with pytest.raises(RuntimeError, match="live-system guard"): os.kill(FOREIGN_PID, signal.SIGTERM) @@ -218,6 +219,7 @@ def test_os_popen_systemctl_blocked(): # ──────────────────── pty.spawn ──────────────────────────────── +@pytest.mark.platforms("linux") def test_pty_spawn_systemctl_blocked(): import pty with pytest.raises(RuntimeError, match="live-system guard"): diff --git a/tests/test_os_marker_gating.py b/tests/test_os_marker_gating.py index 5d348fd73e98..8e4dad72affb 100644 --- a/tests/test_os_marker_gating.py +++ b/tests/test_os_marker_gating.py @@ -1,8 +1,8 @@ -"""The collection guard against a test carrying two host-OS markers. +"""The collection guard against a test carrying two platforms() markers. -Every marker in ``_OS_MARKS`` skips on all but one host, so two of them on one -item means it runs on no host at all while both the Linux suite and the -tests-os lanes report green. tests/conftest.py fails collection instead; this +A module-level gate stacked on a per-test gate ran on no host at all while +both the full-suite and marked lanes reported green — the silent coverage +loss the guard exists for. tests/conftest.py fails collection instead; this pins that behaviour so the guard can't be dropped silently. """ @@ -10,51 +10,54 @@ import pytest -from tests.conftest import _OS_MARKS, _reject_multiple_os_marks +from tests.conftest import _reject_contradictory_platform_marks class _FakeItem: """Stands in for a collected item: the guard reads only these two.""" - def __init__(self, nodeid: str, *marks: str) -> None: + def __init__(self, nodeid: str, *marks) -> None: self.nodeid = nodeid - self._marks = [getattr(pytest.mark, name).mark for name in marks] + self._marks = list(marks) - def iter_markers(self): - return iter(self._marks) + def iter_markers(self, name=None): + if name is None: + return iter(self._marks) + return iter(m for m in self._marks if m.name == name) -def test_single_os_marker_is_accepted(): - items = [_FakeItem(f"t.py::test_{name}", name) for name in _OS_MARKS] - _reject_multiple_os_marks(items) # must not raise +def test_single_platforms_marker_is_accepted(): + items = [ + _FakeItem("t.py::test_linux", pytest.mark.platforms("linux")), + _FakeItem("t.py::test_win", pytest.mark.platforms("windows", arch="arm64")), + _FakeItem("t.py::test_not", pytest.mark.platforms("not macos")), + ] + _reject_contradictory_platform_marks(items) # must not raise -def test_unmarked_and_non_os_markers_are_accepted(): - _reject_multiple_os_marks([ - _FakeItem("t.py::test_plain"), - _FakeItem("t.py::test_other", "slow", "integration"), - ]) +def test_unmarked_and_non_platform_markers_are_accepted(): + _reject_contradictory_platform_marks( + [ + _FakeItem("t.py::test_plain"), + _FakeItem("t.py::test_slow", pytest.mark.slow), + ] + ) -def test_two_os_markers_fail_collection(): +def test_two_platforms_markers_fail_collection(): items = [ - _FakeItem("t.py::test_ok", "linux_only"), - _FakeItem("t.py::test_bad", "linux_only", "windows_only"), + _FakeItem("t.py::test_ok", pytest.mark.platforms("linux")), + _FakeItem( + "t.py::test_bad", + pytest.mark.platforms("linux"), + pytest.mark.platforms("windows"), + ), ] with pytest.raises(pytest.UsageError) as excinfo: - _reject_multiple_os_marks(items) + _reject_contradictory_platform_marks(items) message = str(excinfo.value) assert "t.py::test_bad" in message - assert "linux_only, windows_only" in message + assert "at most one platforms()" in message # The passing item must not be named — the error is a list of offenders. assert "t.py::test_ok" not in message - - -def test_all_three_markers_are_reported_together(): - item = _FakeItem("t.py::test_worst", *_OS_MARKS) - with pytest.raises(pytest.UsageError) as excinfo: - _reject_multiple_os_marks([item]) - - for name in _OS_MARKS: - assert name in str(excinfo.value) diff --git a/tests/test_platforms_marker.py b/tests/test_platforms_marker.py new file mode 100644 index 000000000000..6d7af9e40808 --- /dev/null +++ b/tests/test_platforms_marker.py @@ -0,0 +1,163 @@ +"""The composable ``platforms`` marker: spec evaluation + collection gating. + +Behavior-contract tests for the gate introduced alongside the fixed +platforms("linux")/platforms("macos")/platforms("windows") trio: any-of semantics, negation, +POSIX grouping, arch filters, and the hard errors on unknown specs and +stray keyword arguments. +""" + +from __future__ import annotations + +import pytest + +from tests.conftest import _host_matches_platforms + + +class TestSpecEvaluation: + """Pure evaluation: takes the host as data, no host faking.""" + + # (specs, host_platform, expected_ok) + CASES = [ + (("linux",), "linux", True), + (("linux",), "win32", False), + (("macos",), "darwin", True), + (("windows",), "win32", True), + (("windows",), "linux", False), + (("posix",), "linux", True), + (("posix",), "darwin", True), + (("posix",), "win32", False), + (("not macos",), "linux", True), + (("not macos",), "darwin", False), + (("not windows",), "win32", False), + (("not windows",), "linux", True), + (("linux", "win32host-mismatch"), "win32", False), # unknown spec never matches + (("any",), "linux", True), + ((), "linux", True), # no specs = documentation form, matches all + ] + + @pytest.mark.parametrize(("specs", "host", "expected"), CASES) + def test_spec_matrix(self, specs, host, expected, monkeypatch): + monkeypatch.setattr("tests.conftest.sys.platform", host) + ok, _reason = _host_matches_platforms(specs) + assert ok is expected + + def test_unknown_spec_is_reported_not_matched(self, monkeypatch): + monkeypatch.setattr("tests.conftest.sys.platform", "linux") + ok, reason = _host_matches_platforms(("amiga",)) + assert ok is False + assert "unknown spec" in reason + + def test_negation_of_unknown_spec_is_rejected(self, monkeypatch): + monkeypatch.setattr("tests.conftest.sys.platform", "linux") + ok, reason = _host_matches_platforms(("not amiga",)) + assert ok is False + assert "unknown spec" in reason + + def test_case_insensitive_specs(self, monkeypatch): + monkeypatch.setattr("tests.conftest.sys.platform", "win32") + ok, _ = _host_matches_platforms(("WINDOWS",)) + assert ok is True + + +class TestArchFilter: + @pytest.mark.parametrize( + ("arch", "machine", "negate", "expected"), + [ + ("arm64", "arm64", False, True), + ("arm64", "x86_64", False, False), + ("aarch64", "arm64", False, True), # alias + ("arm64", "arm64", True, False), + ("arm64", "x86_64", True, True), + ], + ) + def test_arch_matrix(self, arch, machine, negate, expected, monkeypatch): + monkeypatch.setattr("tests.conftest.sys.platform", "win32") + monkeypatch.setattr("tests.conftest._platform_machine", lambda: machine) + ok, reason = _host_matches_platforms(("windows",), arch=arch, arch_negate=negate) + assert ok is expected, reason + + def test_arch_reason_names_the_machine(self, monkeypatch): + monkeypatch.setattr("tests.conftest.sys.platform", "win32") + monkeypatch.setattr("tests.conftest._platform_machine", lambda: "x86_64") + ok, reason = _host_matches_platforms(("windows",), arch="arm64") + assert ok is False + assert "x86_64" in reason + + +class TestAnyOfSemantics: + def test_multiple_specs_are_any_of(self, monkeypatch): + monkeypatch.setattr("tests.conftest.sys.platform", "darwin") + ok, _ = _host_matches_platforms(("linux", "macos")) + assert ok is True + + def test_first_matching_spec_wins_over_later_unknown(self, monkeypatch): + # any-of: a matching spec satisfies the gate even if a later spec + # is garbage — unknown specs only matter when nothing matched. + monkeypatch.setattr("tests.conftest.sys.platform", "linux") + ok, _ = _host_matches_platforms(("linux", "amiga")) + assert ok is True + + +class TestCollectionGating: + """The marker must actually skip/gate collected items on this host.""" + + @pytest.mark.platforms("not " + __import__("sys").platform.split("_")[0]) + def test_never_runs_on_this_host_shape(self): + # The spec is built to exclude whatever this host is (linux → "not + # linux", win32 → "not windows"); if it RUNS the gate is broken. + raise AssertionError("platforms() gate failed to skip this host") + + @pytest.mark.platforms("any") + def test_any_spec_runs_everywhere(self): + assert True + + @pytest.mark.skipif( + __import__("sys").platform == "win32", + reason="linux-host assertion; inverted on the linux lane below", + ) + @pytest.mark.platforms("linux") + def test_runs_on_linux(self): + assert True + + +class TestHardErrors: + def test_stray_kwarg_is_a_usage_error(self): + # The gate raises UsageError (surfaced by pytest as a collection + # error) for keyword arguments it does not understand — evaluated + # directly because the raise happens inside the project conftest's + # collection hook. + import pytest as _pytest + + from tests.conftest import _platforms_gate_reason + + class _Item: + nodeid = "tests/x.py::test_x" + + @staticmethod + def iter_markers(name): + yield _pytest.mark.platforms("linux", bogus=True).mark + + with _pytest.raises(_pytest.UsageError, match="unexpected keyword"): + _platforms_gate_reason(_Item) + + +class TestMachineAliases: + """_platform_machine normalizes the raw platform.machine() spellings.""" + + @pytest.mark.parametrize( + ("raw", "normalized"), + [ + ("AMD64", "x86_64"), + ("x86", "x86_64"), + ("aarch64", "arm64"), + ("arm64", "arm64"), + ("x86_64", "x86_64"), + ], + ) + def test_alias_matrix(self, raw, normalized, monkeypatch): + import platform as _platform + + monkeypatch.setattr(_platform, "machine", lambda: raw) + from tests.conftest import _platform_machine + + assert _platform_machine() == normalized diff --git a/tests/test_plugin_skills.py b/tests/test_plugin_skills.py index b2f49061fa3d..b1747eadcb27 100644 --- a/tests/test_plugin_skills.py +++ b/tests/test_plugin_skills.py @@ -8,6 +8,8 @@ import json import logging +import os +import sys import pytest @@ -212,7 +214,7 @@ def test_reads_supporting_file_with_containment(self, tmp_path): reference.write_text("API details.") main = json.loads(skill_view("superpowers:writing-plans")) - assert main["linked_files"] == {"references": ["references/api.md"]} + assert main["linked_files"] == {"references": [os.path.join("references", "api.md")]} result = json.loads( skill_view("superpowers:writing-plans", file_path="references/api.md") ) @@ -222,11 +224,12 @@ def test_reads_supporting_file_with_containment(self, tmp_path): def test_platform_gate_applies_before_supporting_file(self, tmp_path): from tools.skills_tool import skill_view + other_platform = "linux" if sys.platform.startswith("win") else "windows" md = self._register_skill( tmp_path, content=( "---\nname: writing-plans\ndescription: desc\n" - "platforms: [windows]\n---\nBody.\n" + f"platforms: [{other_platform}]\n---\nBody.\n" ), ) reference = md.parent / "references" / "guide.md" diff --git a/tests/test_profile_isolation_runtime.py b/tests/test_profile_isolation_runtime.py index ffa75d943f2f..1e0cfd73e8c5 100644 --- a/tests/test_profile_isolation_runtime.py +++ b/tests/test_profile_isolation_runtime.py @@ -14,6 +14,7 @@ probes used to confirm the bug class. """ +import os import threading from pathlib import Path @@ -108,7 +109,7 @@ def test_store_path_follows_override(self, two_profiles, monkeypatch): b_seen = _under_override(prof_b, lambda: rss._store_path()) assert b_seen.startswith(str(prof_b)) - assert b_seen.endswith("state/rich_sent_index.json") + assert b_seen.endswith(os.path.join("state", "rich_sent_index.json")) # --------------------------------------------------------------------------- diff --git a/tests/test_resource_limits.py b/tests/test_resource_limits.py index 6f24ce0e512d..0dcef090b8c2 100644 --- a/tests/test_resource_limits.py +++ b/tests/test_resource_limits.py @@ -249,6 +249,7 @@ def test_serve_startup_applies_limit_before_web_server(monkeypatch): assert calls == ["limit", "server"] +@pytest.mark.platforms("linux") def test_named_profile_reroute_defers_limit_to_final_process(monkeypatch, tmp_path): """The launcher profile must not leak its limit across machine re-exec.""" from hermes_cli import main as cli_main diff --git a/tests/test_subprocess_home_isolation.py b/tests/test_subprocess_home_isolation.py index 48bf1ed65266..2c016b864ef6 100644 --- a/tests/test_subprocess_home_isolation.py +++ b/tests/test_subprocess_home_isolation.py @@ -107,8 +107,8 @@ def test_two_profiles_get_different_homes(self, tmp_path, monkeypatch): assert home_a is not None assert home_b is not None assert home_a != home_b - assert home_a.endswith("alpha/home") - assert home_b.endswith("beta/home") + assert home_a.endswith(os.path.join("alpha", "home")) + assert home_b.endswith(os.path.join("beta", "home")) diff --git a/tests/test_web_server.py b/tests/test_web_server.py index ed81bce22aca..f848fd56b2fd 100644 --- a/tests/test_web_server.py +++ b/tests/test_web_server.py @@ -148,7 +148,7 @@ def test_start_server_enables_ws_ping_for_half_open_detection(monkeypatch): assert captured["ws_ping_timeout"] >= captured["ws_ping_interval"] -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_start_server_runs_on_uvicorns_loop_factory(monkeypatch): """The dashboard/desktop backend must serve uvicorn on the loop *uvicorn* selects, not the interpreter default. @@ -207,6 +207,7 @@ def _guard_asyncio_run(coro): ) +@pytest.mark.platforms("linux") def test_start_server_keeps_bare_asyncio_run_on_posix(monkeypatch): """POSIX continues to serve via the plain ``asyncio.run(_serve())`` path, never the Windows loop-factory branch. @@ -273,7 +274,7 @@ def _raise_keyboard_interrupt(coro): ) -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_start_server_treats_windows_keyboardinterrupt_as_clean_shutdown(monkeypatch): """Console Ctrl+C on the Windows loop-factory branch is a clean exit too. @@ -306,7 +307,7 @@ def _raise_keyboard_interrupt(coro, *, loop_factory=None): ) -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_start_server_treats_windows_fallback_keyboardinterrupt_as_clean_shutdown( monkeypatch, ): diff --git a/tests/test_windows_subprocess_no_window_flags.py b/tests/test_windows_subprocess_no_window_flags.py index cc127452844c..706bfc06023b 100644 --- a/tests/test_windows_subprocess_no_window_flags.py +++ b/tests/test_windows_subprocess_no_window_flags.py @@ -68,12 +68,12 @@ def kill(self): # pragma: no cover - never reached on the fast path return _FakePopen -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_bounded_git_probe_fast_path_spawn_contract_windows(monkeypatch): """The normal-path spawn contract survives the run()->Popen rewrite: PIPE/PIPE/DEVNULL, text + utf-8/replace, hidden-window flags on Windows. - ``windows_only``: the ``creationflags`` assertion is the point, and + ``platforms("windows")``: the ``creationflags`` assertion is the point, and ``bounded_git_probe`` only sets that key when ``IS_WINDOWS`` — which the helper caches from the real platform at import. ``windows_hide_flags`` is still stubbed so the expected value is a fixed constant rather than @@ -156,9 +156,9 @@ def boom(cmd, **kwargs): -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_shell_hooks_hide_hook_command_windows(monkeypatch): - """``windows_only``: ``shell_hooks._spawn`` only adds ``creationflags`` + """``platforms("windows")``: ``shell_hooks._spawn`` only adds ``creationflags`` under its module-level ``IS_WINDOWS``, so on Linux the flag patch was what created the thing being asserted.""" from agent import shell_hooks @@ -402,12 +402,12 @@ def fake_run(cmd, **kwargs): -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_suppress_platform_ver_console_stubs_syscmd_ver(monkeypatch): """``_syscmd_ver`` is replaced by an in-process echo stub so win32_ver() takes its ValueError fallback instead of shelling out to `cmd /c ver`. - ``windows_only``: ``suppress_platform_ver_console()`` is a no-op unless + ``platforms("windows")``: ``suppress_platform_ver_console()`` is a no-op unless ``IS_WINDOWS``, and the console flash it prevents (``cmd /c ver``) only exists on Windows — the old flag patch installed the stub on a host where ``win32_ver`` is never consulted at all. diff --git a/tests/tools/test_approval.py b/tests/tools/test_approval.py index 0d0a666bcbb0..262a6cdb232a 100644 --- a/tests/tools/test_approval.py +++ b/tests/tools/test_approval.py @@ -99,6 +99,7 @@ def test_rm_flags_after_operands_detected(self): assert "delete" in desc.lower() + @pytest.mark.platforms("linux") def test_nonrecursive_verification_artifact_cleanup_is_not_dangerous(self): with mock_patch("tempfile.gettempdir", return_value="/tmp"): for prefix in ("hermes-verify-", "hermes-ad-hoc-"): @@ -108,6 +109,7 @@ def test_nonrecursive_verification_artifact_cleanup_is_not_dangerous(self): None, ) + @pytest.mark.require_symlinks def test_symlinked_temp_dir_only_exempts_canonical_target(self, tmp_path): real_temp = tmp_path / "real-temp" real_temp.mkdir() diff --git a/tests/tools/test_approval_timeout_overflow.py b/tests/tools/test_approval_timeout_overflow.py index 18d7105880a9..57649b4574b6 100644 --- a/tests/tools/test_approval_timeout_overflow.py +++ b/tests/tools/test_approval_timeout_overflow.py @@ -77,7 +77,10 @@ def _blocked(name, *args, **kwargs): monkeypatch.setattr(builtins, "__import__", _blocked) with _with_configured_timeout(10**18): value = _get_approval_timeout() - assert value == 365 * 24 * 3600 + # The fallback mirrors the per-host MAX_SAFE_TIMEOUT_S ceiling + # (imported BEFORE the block — the check itself must not depend on + # the patched-away import). + assert value == int(MAX_SAFE_TIMEOUT_S) # Still platform-safe for the crashing primitive. lock = threading.Lock() assert lock.acquire(timeout=value) diff --git a/tests/tools/test_approved_command_clean_slate.py b/tests/tools/test_approved_command_clean_slate.py index 04a924fe57dd..817a99ab485e 100644 --- a/tests/tools/test_approved_command_clean_slate.py +++ b/tests/tools/test_approved_command_clean_slate.py @@ -86,6 +86,7 @@ def test_non_approved_command_still_interrupts_on_stale_bit(monkeypatch): assert "[Command interrupted]" in result["output"] +@pytest.mark.platforms("linux") def test_approved_command_genuine_interrupt_after_start_still_kills(tmp_path): """The clean-slate clear must NOT make approved commands un-interruptible: an interrupt that arrives after execution starts still SIGINTs (130).""" @@ -112,6 +113,7 @@ def worker(): set_interrupt(False, thread_id=t.ident) +@pytest.mark.platforms("linux") def test_approved_note_enriched_not_misleading_on_interrupt(monkeypatch, tmp_path): """On a genuine post-start interrupt of an approved command, the note must read '...approved by the user, then interrupted.' — the bare diff --git a/tests/tools/test_async_delegation.py b/tests/tools/test_async_delegation.py index 5d0a01c22a63..a5852d45ded2 100644 --- a/tests/tools/test_async_delegation.py +++ b/tests/tools/test_async_delegation.py @@ -100,7 +100,7 @@ def test_schema_init_preserves_shared_state_db_wal_mode(tmp_path): conn.close() -@pytest.mark.macos_only +@pytest.mark.platforms("macos") def test_connect_preserves_wal_and_applies_macos_durability_barriers( tmp_path, monkeypatch ): diff --git a/tests/tools/test_base_environment.py b/tests/tools/test_base_environment.py index cb883ea5c483..8529597eef06 100644 --- a/tests/tools/test_base_environment.py +++ b/tests/tools/test_base_environment.py @@ -3,6 +3,7 @@ Tests _wrap_command(), _extract_cwd_from_output(), _embed_stdin_heredoc(), init_session() failure handling, and the CWD marker contract. """ +import pytest from unittest.mock import MagicMock @@ -190,6 +191,7 @@ def _run(self, script): import subprocess return subprocess.run(["/bin/bash", "-c", script], capture_output=True, text=True) + @pytest.mark.platforms("linux") def test_concurrent_writes_never_tear_the_snapshot(self, tmp_path): import shutil if not shutil.which("bash"): @@ -227,6 +229,7 @@ def test_concurrent_writes_never_tear_the_snapshot(self, tmp_path): final = self._run(f"source {_q(snap)} >/dev/null 2>&1 && echo OK || echo BROKEN") assert "OK" in final.stdout, f"final snapshot not sourceable: {final.stdout} {final.stderr}" + @pytest.mark.platforms("linux") def test_failed_export_does_not_destroy_good_snapshot(self, tmp_path): """If ``export -p`` fails, the ``&&``-chained mv must NOT clobber the existing good snapshot.""" @@ -254,6 +257,7 @@ def test_failed_export_does_not_destroy_good_snapshot(self, tmp_path): class TestSnapshotFileModes: """Snapshot metadata files are private without changing user command umask.""" + @pytest.mark.platforms("linux") def test_snapshot_and_cwd_files_are_0600(self, tmp_path): import os from pathlib import Path diff --git a/tests/tools/test_bot_mode_dm.py b/tests/tools/test_bot_mode_dm.py index e83ec30ae070..670dc2062cff 100644 --- a/tests/tools/test_bot_mode_dm.py +++ b/tests/tools/test_bot_mode_dm.py @@ -533,7 +533,7 @@ def test_real_delivery_command_round_trip(tmp_path, stdin_file): assert not dm_file.exists() -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_delivery_command_round_trip_through_windows_local_shell(tmp_path): """Native runner paths must survive the Git Bash process boundary.""" from tools.environments.local import _find_shell @@ -666,6 +666,7 @@ def test_sweeper_removes_only_stale_dm_files(tmp_path, monkeypatch): assert unrelated.exists() +@pytest.mark.platforms("linux") def test_dm_dir_is_private_and_uid_scoped_on_posix(tmp_path, monkeypatch): monkeypatch.setattr(bot_mode_dm.tempfile, "gettempdir", lambda: str(tmp_path)) @@ -678,6 +679,7 @@ def test_dm_dir_is_private_and_uid_scoped_on_posix(tmp_path, monkeypatch): assert dm_dir.stat().st_mode & 0o777 == 0o700 +@pytest.mark.platforms("linux") def test_dm_dir_repairs_restrictive_owner_mode(tmp_path, monkeypatch): monkeypatch.setattr(bot_mode_dm.tempfile, "gettempdir", lambda: str(tmp_path)) uid = os.getuid() if hasattr(os, "getuid") else None diff --git a/tests/tools/test_bot_relay_windows_paths.py b/tests/tools/test_bot_relay_windows_paths.py index e6123a7371f3..a60461f4c030 100644 --- a/tests/tools/test_bot_relay_windows_paths.py +++ b/tests/tools/test_bot_relay_windows_paths.py @@ -28,6 +28,7 @@ import tools.bot_mode_dm as bot_mode_dm import tools.bot_relay as bot_relay +import pytest ENV = {"id": "d" * 32, "target_handle": "researcher", "target_connection": "ssh-vps"} @@ -50,6 +51,7 @@ def test_waiter_windows_path_compiles_after_backslash_folding(): compile(folded, "", "exec") +@pytest.mark.platforms("linux") def test_waiter_posix_path_and_label_values_roundtrip(): """On POSIX (backslash-free paths) the raw prefix changes nothing.""" root = Path("/tmp/hermes-home") @@ -88,6 +90,7 @@ def test_waiter_raw_prefix_keeps_injection_defense(): assert "x'); __import__('sys').exit(2); print('x" in code +@pytest.mark.platforms("linux") def test_local_delivery_resolves_sibling_hermes(tmp_path, monkeypatch): bin_dir = tmp_path / "venv" / "bin" bin_dir.mkdir(parents=True) diff --git a/tests/tools/test_bot_turn_lock.py b/tests/tools/test_bot_turn_lock.py index 1b02ae6febf3..3d9c6d7d59dc 100644 --- a/tests/tools/test_bot_turn_lock.py +++ b/tests/tools/test_bot_turn_lock.py @@ -9,7 +9,10 @@ from __future__ import annotations -import fcntl +try: + import fcntl # POSIX-only; on Windows the module is skipped wholesale +except ImportError: # pragma: no cover - Windows + fcntl = None import json import os import re @@ -18,6 +21,8 @@ import pytest +pytestmark = pytest.mark.platforms("linux") + from tools import bot_mode_dm, bot_relay from tools.bot_relay import TurnBusyError, acquire_turn_lock, turn_lock_path diff --git a/tests/tools/test_browser_hardening.py b/tests/tools/test_browser_hardening.py index 6ad1e9f5daf8..c67d68abc78d 100644 --- a/tests/tools/test_browser_hardening.py +++ b/tests/tools/test_browser_hardening.py @@ -229,6 +229,7 @@ def test_stored_snapshot_is_secret_redacted(self): content = Path(stored).read_text(encoding="utf-8") assert "STOREDSNAPSHOTSECRET" not in content + @pytest.mark.require_symlinks def test_stored_snapshot_refuses_planted_symlink(self, tmp_path, monkeypatch): """A pre-planted symlink at the content-hash path must not be followed to its target — only the link itself may be replaced. diff --git a/tests/tools/test_browser_homebrew_paths.py b/tests/tools/test_browser_homebrew_paths.py index 13669f1fe97c..62237a512bb3 100644 --- a/tests/tools/test_browser_homebrew_paths.py +++ b/tests/tools/test_browser_homebrew_paths.py @@ -406,6 +406,7 @@ def capture_popen(cmd, **kwargs): ] assert captured_cmd[5:9] == ["--session", "test-session", "--json", "navigate"] + @pytest.mark.platforms("linux") def test_subprocess_path_includes_termux_fallback_dirs(self, tmp_path): """Termux fallback dirs should survive browser PATH rebuilding.""" captured_env = {} diff --git a/tests/tools/test_browser_npx_warmup.py b/tests/tools/test_browser_npx_warmup.py index b866c43073fa..a61101caaeb2 100644 --- a/tests/tools/test_browser_npx_warmup.py +++ b/tests/tools/test_browser_npx_warmup.py @@ -18,6 +18,8 @@ import subprocess from unittest.mock import MagicMock, patch +import pytest + from tools.browser_tool import ( AGENT_BROWSER_NPX_SPEC, _legacy_kill_process_tree, @@ -123,6 +125,7 @@ def test_merges_extended_path_so_managed_only_npx_can_find_sibling_node(): assert kwargs["env"]["PATH"] == "/opt/hermes/node/bin:/usr/bin" +@pytest.mark.platforms("linux") def test_runs_in_its_own_process_group_on_posix(monkeypatch): monkeypatch.setattr("os.name", "posix") with patch("tools.browser_tool._resolve_npx_bin", return_value="/usr/bin/npx"), \ @@ -216,6 +219,7 @@ def test_returns_false_instead_of_raising_on_unexpected_communicate_exception(): mock_kill.assert_called_once_with(proc) +@pytest.mark.platforms("linux") class TestLegacyKillProcessTree: """Contract of the pre-#85125 local fallback (used when agent.deadline delegation fails); the delegating wrapper is covered in diff --git a/tests/tools/test_browser_real_profile.py b/tests/tools/test_browser_real_profile.py index a18146919719..4009b76ef7cc 100644 --- a/tests/tools/test_browser_real_profile.py +++ b/tests/tools/test_browser_real_profile.py @@ -89,8 +89,14 @@ def _make_profile(self, root): (root / "Default" / "Cache" / "Cache_Data" / "big").write_text("x" * 1000) (root / "Code Cache" / "js" / "blob").write_text("y" * 1000) (root / "Crashpad" / "dump").write_text("z") - # Live-instance leftovers that must never reach the copy - os.symlink("dead-target-1", root / "SingletonLock") + # Live-instance leftovers that must never reach the copy. On Windows + # without the SeCreateSymbolicLink privilege os.symlink raises + # WinError 1314 — a regular dangling-ish placeholder exercises the + # same exclusion contract there. + try: + os.symlink("dead-target-1", root / "SingletonLock") + except OSError: + (root / "SingletonLock").write_text("dead-target-1") return root def test_fresh_snapshot_copies_auth_and_skips_caches(self, tmp_path, monkeypatch): @@ -139,6 +145,7 @@ def test_missing_source_fails_closed(self, tmp_path, monkeypatch): assert dst is None assert err and "was not found" in err + @pytest.mark.platforms("linux") def test_snapshot_files_are_owner_only(self, tmp_path, monkeypatch): """Every copied file must be 0600 and every dir 0700 (#96729). @@ -170,6 +177,7 @@ def test_snapshot_files_are_owner_only(self, tmp_path, monkeypatch): offenders.append((os.path.join(root, f), oct(mode))) assert not offenders, f"group/world-accessible snapshot entries: {offenders}" + @pytest.mark.platforms("linux") def test_existing_lax_snapshot_heals_on_refresh(self, tmp_path, monkeypatch): """A snapshot left 0644 by an older build tightens on the next pass.""" import stat @@ -964,16 +972,27 @@ def test_relaunch_path_does_snapshot(self, tmp_path): import tools.browser_tool as bt bt._real_profile_cdp_cache.clear() proc = Mock(returncode=0, stdout="", stderr="") + + class FakeChrome: + def poll(self): + return None + + def fake_popen(argv, **kw): + (tmp_path / "DevToolsActivePort").write_text("9251\n/devtools/browser/x\n") + return FakeChrome() + with patch.object(bt, "_use_real_profile", return_value=True), \ patch.object(bt, "_using_lightpanda_engine", return_value=False), \ patch("hermes_cli.browser_connect.detect_default_chromium", return_value="chrome"), \ patch("hermes_cli.browser_connect.real_profile_copy_dir", return_value=str(tmp_path)), \ + patch("hermes_cli.browser_connect.chromium_executable", return_value="/usr/bin/chrome"), \ patch("hermes_cli.browser_connect.snapshot_real_profile", return_value=(str(tmp_path), None)) as snap, \ patch.object(bt, "_agent_browser_get_cdp", side_effect=[None, "http://127.0.0.1:9251"]), \ patch.object(bt, "_find_agent_browser", return_value="/usr/bin/agent-browser"), \ patch.object(bt.subprocess, "run", return_value=proc), \ + patch.object(bt.subprocess, "Popen", side_effect=fake_popen), \ patch.object(bt, "_is_headed_mode", return_value=False): cdp, err = bt._real_profile_cdp() assert err is None diff --git a/tests/tools/test_browser_use_cli.py b/tests/tools/test_browser_use_cli.py index efcbfe5ebad3..8b84e07839e8 100644 --- a/tests/tools/test_browser_use_cli.py +++ b/tests/tools/test_browser_use_cli.py @@ -348,6 +348,7 @@ def test_auto_detect_without_key_does_not_migrate(self, monkeypatch): monkeypatch.setattr(bu_cli, "_find_cli", lambda: None) assert bu_cli.is_browser_use_cli_mode() is False + @pytest.mark.platforms("linux") def test_migrated_config_gets_bu_autospawn(self, tmp_path, monkeypatch): monkeypatch.setattr("hermes_cli.config.read_raw_config", lambda: self._LEGACY) monkeypatch.setenv("BROWSER_USE_API_KEY", "bu-key") @@ -356,6 +357,7 @@ def test_migrated_config_gets_bu_autospawn(self, tmp_path, monkeypatch): result = json.loads(bu_cli.browser_exec("print(1)")) assert "autospawn:1" in result["output"] + @pytest.mark.platforms("linux") def test_explicit_backend_does_not_set_bu_autospawn(self, tmp_path, monkeypatch): monkeypatch.setattr( "hermes_cli.config.read_raw_config", @@ -449,6 +451,7 @@ def test_provider_without_cdp_returns_error(self, monkeypatch): err = bu_cli._resolve_backend_cdp(self._env(), "t1") assert err and "no" in err.lower() and "CDP" in err + @pytest.mark.platforms("linux") def test_named_session_composes_with_provider_backend(self, tmp_path, monkeypatch): """session= composes with a configured provider backend: the name keys its OWN provider browser (bu-named-), so concurrent @@ -534,6 +537,7 @@ def _run(self, tmp_path, monkeypatch, *, session="", private=False, provider=Fal monkeypatch.setattr(bu_cli, "_find_cli", lambda: [cli]) return json.loads(bu_cli.browser_exec("print('payload')", session=session)) + @pytest.mark.platforms("linux") def test_named_shared_browser_gets_preamble(self, tmp_path, monkeypatch): result = self._run(tmp_path, monkeypatch, session="r7k2") assert result["success"] is True @@ -541,17 +545,20 @@ def test_named_shared_browser_gets_preamble(self, tmp_path, monkeypatch): # model code still present, after the preamble assert result["output"].index("_hermes_ensure_own_tab") < result["output"].index("print('payload')") + @pytest.mark.platforms("linux") def test_unnamed_session_gets_no_preamble(self, tmp_path, monkeypatch): result = self._run(tmp_path, monkeypatch, session="") assert result["success"] is True assert "_hermes_ensure_own_tab" not in result["output"] + @pytest.mark.platforms("linux") def test_named_provider_browser_skips_preamble(self, tmp_path, monkeypatch): """Per-name provider browsers are private — preamble would leak a tab.""" result = self._run(tmp_path, monkeypatch, session="r7k2", provider=True) assert result["success"] is True assert "_hermes_ensure_own_tab" not in result["output"] + @pytest.mark.platforms("linux") def test_sentinel_never_reaches_subprocess_env(self, tmp_path, monkeypatch): import tools.browser_tool as bt @@ -713,6 +720,7 @@ def test_find_screenshot_rejects_stale_and_missing(self, tmp_path): out = f"{stale}\n/nonexistent/dir/x.png\n" assert bu_cli._find_screenshot(out, since=time.time()) is None + @pytest.mark.platforms("linux") def test_vision_model_gets_multimodal_envelope(self, tmp_path, monkeypatch): shot = self._shot(tmp_path) cli = _fake_cli(tmp_path, f'cat > /dev/null\necho "{shot}"\n') @@ -731,6 +739,7 @@ def test_vision_model_gets_multimodal_envelope(self, tmp_path, monkeypatch): assert result["meta"]["screenshot_path"] == shot assert shot in result["text_summary"] + @pytest.mark.platforms("linux") def test_text_only_model_gets_plain_result_with_path(self, tmp_path, monkeypatch): shot = self._shot(tmp_path) cli = _fake_cli(tmp_path, f'cat > /dev/null\necho "{shot}"\n') @@ -851,6 +860,7 @@ def test_empty_code_rejected(self): result = json.loads(bu_cli.browser_exec(" ")) assert "error" in result + @pytest.mark.platforms("linux") def test_code_piped_on_stdin(self, tmp_path, monkeypatch): cli = _fake_cli(tmp_path, 'code=$(cat)\necho "got:$code"\n') monkeypatch.setattr(bu_cli, "_find_cli", lambda: [cli]) @@ -860,6 +870,7 @@ def test_code_piped_on_stdin(self, tmp_path, monkeypatch): assert 'got:print("hi")' in result["output"] assert "session" not in result + @pytest.mark.platforms("linux") def test_session_sets_bu_name(self, tmp_path, monkeypatch): cli = _fake_cli(tmp_path, 'cat > /dev/null\necho "bu:$BU_NAME"\n') monkeypatch.setattr(bu_cli, "_find_cli", lambda: [cli]) @@ -874,6 +885,7 @@ def test_invalid_session_name_rejected(self, monkeypatch, tmp_path): assert "error" in result assert "session" in result["error"].lower() + @pytest.mark.platforms("linux") def test_nonzero_exit_reports_failure_and_stderr(self, tmp_path, monkeypatch): cli = _fake_cli(tmp_path, 'cat > /dev/null\necho "boom" >&2\nexit 3\n') monkeypatch.setattr(bu_cli, "_find_cli", lambda: [cli]) @@ -882,6 +894,7 @@ def test_nonzero_exit_reports_failure_and_stderr(self, tmp_path, monkeypatch): assert result["exit_code"] == 3 assert "boom" in result["stderr"] + @pytest.mark.platforms("linux") def test_timeout_returns_actionable_error(self, tmp_path, monkeypatch): cli = _fake_cli(tmp_path, "cat > /dev/null\nsleep 30\n") monkeypatch.setattr(bu_cli, "_find_cli", lambda: [cli]) @@ -902,6 +915,7 @@ def _hermetic_home(self, tmp_path, monkeypatch): monkeypatch.setenv("HERMES_HOME", str(tmp_path / "home")) monkeypatch.setenv("PATH", str(tmp_path / "empty")) + @pytest.mark.platforms("linux") def test_managed_bin_browser_use_found(self, tmp_path, monkeypatch): bin_dir = tmp_path / "home" / "bin" bin_dir.mkdir(parents=True) @@ -910,6 +924,7 @@ def test_managed_bin_browser_use_found(self, tmp_path, monkeypatch): bu.chmod(bu.stat().st_mode | stat.S_IXUSR) assert bu_cli._find_cli_unpatched() == [str(bu)] + @pytest.mark.platforms("linux") def test_managed_bin_uvx_fallback(self, tmp_path, monkeypatch): bin_dir = tmp_path / "home" / "bin" bin_dir.mkdir(parents=True) @@ -921,6 +936,7 @@ def test_managed_bin_uvx_fallback(self, tmp_path, monkeypatch): def test_nothing_found(self, tmp_path, monkeypatch): assert bu_cli._find_cli_unpatched() is None + @pytest.mark.platforms("linux") def test_user_local_bin_browser_use_found(self, tmp_path, monkeypatch): """#83788: Desktop/TUI workers spawn with a minimal PATH that omits ~/.local/bin, where `uv tool install browser-use` links the binary @@ -932,6 +948,7 @@ def test_user_local_bin_browser_use_found(self, tmp_path, monkeypatch): cli.chmod(cli.stat().st_mode | stat.S_IXUSR) assert bu_cli._find_cli_unpatched() == [str(cli)] + @pytest.mark.platforms("linux") def test_managed_bin_precedes_user_local_bin(self, tmp_path, monkeypatch): """MANAGED-FIRST: Hermes' managed copy wins over a user-level side install — every backend selection provisions/updates the managed @@ -949,6 +966,7 @@ def test_managed_bin_precedes_user_local_bin(self, tmp_path, monkeypatch): managed_cli.chmod(managed_cli.stat().st_mode | stat.S_IXUSR) assert bu_cli._find_cli_unpatched() == [str(managed_cli)] + @pytest.mark.platforms("linux") def test_managed_bin_precedes_path(self, tmp_path, monkeypatch): """MANAGED-FIRST: the managed copy also wins over one on PATH.""" path_dir = tmp_path / "onpath" @@ -964,6 +982,7 @@ def test_managed_bin_precedes_path(self, tmp_path, monkeypatch): managed_cli.chmod(managed_cli.stat().st_mode | stat.S_IXUSR) assert bu_cli._find_cli_unpatched() == [str(managed_cli)] + @pytest.mark.platforms("linux") def test_user_local_bin_uvx_fallback(self, tmp_path, monkeypatch): cli_dir = tmp_path / "userhome" / ".local" / "bin" cli_dir.mkdir(parents=True) @@ -992,6 +1011,7 @@ def test_path_install_does_not_short_circuit(self, tmp_path, monkeypatch): assert ok is False assert "already installed" not in msg + @pytest.mark.platforms("linux") def test_already_installed_in_managed_bin(self, tmp_path, monkeypatch): bin_dir = tmp_path / "home" / "bin" bin_dir.mkdir(parents=True) @@ -1016,6 +1036,7 @@ def test_no_uv_anywhere_fails_with_guidance(self, tmp_path, monkeypatch): assert ok is False assert "uv" in msg + @pytest.mark.platforms("linux") def test_successful_install_via_fake_uv(self, tmp_path, monkeypatch): home = tmp_path / "home" bin_dir = home / "bin" @@ -1044,6 +1065,7 @@ def test_successful_install_via_fake_uv(self, tmp_path, monkeypatch): assert ok is True, msg assert (bin_dir / "browser-use").exists() + @pytest.mark.platforms("linux") def test_failed_install_surfaces_stderr_tail(self, tmp_path, monkeypatch): home = tmp_path / "home" monkeypatch.setenv("HERMES_HOME", str(home)) diff --git a/tests/tools/test_checkpoint_manager.py b/tests/tools/test_checkpoint_manager.py index 67cdaac9f640..4eaa0992ccbd 100644 --- a/tests/tools/test_checkpoint_manager.py +++ b/tests/tools/test_checkpoint_manager.py @@ -650,7 +650,7 @@ def test_env_pins_store_worktree_and_ignores_ambient_git_state( env = _git_env( store, str(work), index_file=store / "indexes" / "abc", ) - assert env["GIT_INDEX_FILE"].endswith("indexes/abc") + assert env["GIT_INDEX_FILE"].endswith(os.path.join("indexes", "abc")) # ~ in the work tree is expanded. tilde_work = fake_home / "work" diff --git a/tests/tools/test_clipboard.py b/tests/tools/test_clipboard.py index 6e08d7f9bb7e..eca303661cf0 100644 --- a/tests/tools/test_clipboard.py +++ b/tests/tools/test_clipboard.py @@ -402,7 +402,7 @@ def setup_method(self): import hermes_cli.clipboard as cb cb._wsl_detected = None - @pytest.mark.macos_only + @pytest.mark.platforms("macos") def test_macos_dispatch(self): """Faking darwin selected the branch but left `_macos_has_image`'s real facility (osascript) absent — only a real macOS host has it.""" @@ -410,7 +410,7 @@ def test_macos_dispatch(self): assert has_clipboard_image() is True m.assert_called_once() - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_wsl_falls_through_to_wayland_when_windows_path_empty(self): """WSLg often bridges images to wl-paste even when powershell.exe check fails. diff --git a/tests/tools/test_code_execution.py b/tests/tools/test_code_execution.py index 66ed46a68663..1a82344bf7ec 100644 --- a/tests/tools/test_code_execution.py +++ b/tests/tools/test_code_execution.py @@ -905,6 +905,7 @@ def _run(): t.join(timeout=5) return responses + @pytest.mark.platforms("linux") def test_missing_token_rejected(self): """A request with no token is rejected as Unauthorized.""" resp = self._drive_server( diff --git a/tests/tools/test_code_execution_windows_env.py b/tests/tools/test_code_execution_windows_env.py index 45d0058f4bdc..d5c57642f471 100644 --- a/tests/tools/test_code_execution_windows_env.py +++ b/tests/tools/test_code_execution_windows_env.py @@ -191,11 +191,11 @@ def test_passthrough_still_works_on_windows(self): assert "OPENAI_API_KEY" not in scrubbed -# ``windows_only`` rather than ``skipif(sys.platform != "win32")``: the +# ``platforms("windows")`` rather than ``skipif(sys.platform != "win32")``: the # dedicated Windows CI job selects its files by grepping for the marker, so a # bare skipif is invisible to it — the file is never imported there and these # tests run on no host at all. -@pytest.mark.windows_only +@pytest.mark.platforms("windows") class TestWindowsSocketSmokeTest: """Integration-ish smoke test: spawn a child Python with a scrubbed env and confirm it can create an AF_INET socket. This is the @@ -480,7 +480,7 @@ def test_stub_source_roundtrips_through_utf8(self): finally: os.unlink(tmp_path) - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_default_encoding_would_have_failed(self): """Negative control: prove that on Windows, writing the stub *without* ``encoding="utf-8"`` would corrupt the file. If this @@ -614,7 +614,7 @@ def test_live_child_can_print_non_ascii(self): assert "\u2192" in decoded, f"arrow missing from output: {decoded!r}" assert "\U0001f680" in decoded, f"emoji missing from output: {decoded!r}" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_child_without_utf8_env_would_fail(self): """Negative control: spawn a Python child *without* our env overrides and prove that on Windows, printing non-ASCII fails. diff --git a/tests/tools/test_computer_use.py b/tests/tools/test_computer_use.py index 0ac3fed17e77..44e9b970c7b5 100644 --- a/tests/tools/test_computer_use.py +++ b/tests/tools/test_computer_use.py @@ -1409,8 +1409,11 @@ class FakeProc: returncode = 0 stderr = "" # Daemon returns a path, not inline base64. - stdout = ('{"element_count": 7, "tree_markdown": "- [0] AXButton",' - ' "screenshot_file_path": "%s"}' % str(shot)) + stdout = json.dumps({ + "element_count": 7, + "tree_markdown": "- [0] AXButton", + "screenshot_file_path": str(shot), + }) import subprocess as _sp orig_run = _sp.run diff --git a/tests/tools/test_computer_use_cua_backend_linux.py b/tests/tools/test_computer_use_cua_backend_linux.py index f106ea14f081..43ad30050f19 100644 --- a/tests/tools/test_computer_use_cua_backend_linux.py +++ b/tests/tools/test_computer_use_cua_backend_linux.py @@ -85,7 +85,7 @@ def test_parse_xprop_net_active_window_standard_output(): assert _parse_xprop_net_active_window(raw) == 0x503000b -@pytest.mark.linux_only +@pytest.mark.platforms("linux") def test_default_capture_prefers_x11_active_window_when_z_index_tied(): """The ``_NET_ACTIVE_WINDOW`` tie-break is a Linux/X11-only branch of ``_select_capture_target``; run it where ``sys.platform`` really is @@ -104,7 +104,7 @@ def test_default_capture_prefers_x11_active_window_when_z_index_tied(): assert target["window_id"] == 84043449 -@pytest.mark.linux_only +@pytest.mark.platforms("linux") def test_default_capture_skips_desktop_helper_when_active_window_unknown(): """Even without _NET_ACTIVE_WINDOW, ding/Desktop helpers must not win (#54173). diff --git a/tests/tools/test_credential_files.py b/tests/tools/test_credential_files.py index cb48d2a72179..11e04b8bcba6 100644 --- a/tests/tools/test_credential_files.py +++ b/tests/tools/test_credential_files.py @@ -87,6 +87,7 @@ def test_custom_container_base(self, tmp_path): assert mounts[0]["container_path"] == "/home/user/.hermes/skills" + @pytest.mark.require_symlinks def test_symlinks_are_sanitized(self, tmp_path): """Symlinks in skills dir should be excluded from the mount.""" hermes_home = tmp_path / ".hermes" @@ -126,6 +127,7 @@ def test_no_symlinks_returns_original_dir(self, tmp_path): class TestIterSkillsFiles: + @pytest.mark.require_symlinks def test_returns_files_skipping_symlinks(self, tmp_path): hermes_home = tmp_path / ".hermes" skills_dir = hermes_home / "skills" @@ -490,6 +492,7 @@ def test_enumerates_files(self, tmp_path, monkeypatch): assert "upload.zip" in names assert "report.pdf" in names + @pytest.mark.require_symlinks def test_skips_symlinks(self, tmp_path, monkeypatch): """Symlinks inside cache dirs are skipped.""" hermes_home = tmp_path / ".hermes" diff --git a/tests/tools/test_docker_environment.py b/tests/tools/test_docker_environment.py index 54082f6ed153..6998026f0f77 100644 --- a/tests/tools/test_docker_environment.py +++ b/tests/tools/test_docker_environment.py @@ -620,6 +620,7 @@ def _bind_mount_specs(run_args): ] +@pytest.mark.platforms("linux") def test_persistent_bind_mounts_survive_a_session_key_task_id(monkeypatch, tmp_path): """A gateway session key reaches the persistent sandbox path as-is, and it carries colons (``session:agent:main:telegram:dm:``). Docker reads diff --git a/tests/tools/test_execution_flag_detection.py b/tests/tools/test_execution_flag_detection.py index 65c1aeeb493f..27f28fc8ba27 100644 --- a/tests/tools/test_execution_flag_detection.py +++ b/tests/tools/test_execution_flag_detection.py @@ -19,6 +19,7 @@ (["rg", "--pre-glob", "--pre", "needle"], "needle\n", 0, "needle\n"), ], ) +@pytest.mark.platforms("linux") def test_real_read_tool_binaries_confirm_option_ownership( argv, stdin, expected_returncode, expected_output ): @@ -43,6 +44,7 @@ def test_real_read_tool_binaries_confirm_option_ownership( ("man", ["-P", "-payload-marker", "ls"], None, True), ], ) +@pytest.mark.platforms("linux") def test_real_binaries_execute_leading_dash_program_payload( tmp_path, tool, args, stdin, needs_tty ): diff --git a/tests/tools/test_file_operations.py b/tests/tools/test_file_operations.py index 9af6dcb73e1a..1da54771992d 100644 --- a/tests/tools/test_file_operations.py +++ b/tests/tools/test_file_operations.py @@ -286,7 +286,7 @@ def test_escape_shell_arg_simple(self, file_ops): assert file_ops._escape_shell_arg("hello") == "'hello'" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_escape_shell_arg_rewrites_forward_slash_native_paths(self, file_ops): """Windows-only: ``_bash_safe_path`` only rewrites drive paths to the Git Bash form on Windows, where the MSYS path mangling it works around @@ -295,7 +295,7 @@ def test_escape_shell_arg_rewrites_forward_slash_native_paths(self, file_ops): "C:/Users/alice/notes.txt" ) == "'/c/Users/alice/notes.txt'" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_read_file_uses_bash_safe_windows_paths(self, mock_env): """Windows-only: proves read_file's shell commands carry the MSYS path form Git Bash needs — a translation that is a no-op off Windows.""" @@ -439,6 +439,7 @@ class TestSearchFilesFallbackHiddenPaths: def _make_env(self): return make_real_subprocess_env("/") + @pytest.mark.platforms("linux") def test_hidden_root_with_hidden_ancestor_includes_files(self, tmp_path, monkeypatch): """Fallback find should include visible files when path is inside hidden root.""" root = tmp_path / ".hermes" / "logs" @@ -459,6 +460,7 @@ def test_hidden_root_with_hidden_ancestor_includes_files(self, tmp_path, monkeyp assert result.error is None assert set(result.files) == {str(visible_file), str(visible_nested_file)} + @pytest.mark.platforms("linux") def test_normal_root_still_excludes_hidden_descendants(self, tmp_path, monkeypatch): """Fallback find should still exclude hidden descendant paths for normal roots.""" root = tmp_path / "repo" @@ -590,6 +592,7 @@ class TestAtomicWriteNewFilePermissions: """_atomic_write should apply umask-default perms to new files (not 0600).""" @pytest.mark.parametrize("test_umask", [0o022, 0o002, 0o077]) + @pytest.mark.platforms("linux") def test_new_file_gets_umask_default_permissions(self, tmp_path, test_umask): """Newly created file should get umask-computed perms, not mktemp's 0600. @@ -614,6 +617,7 @@ def test_new_file_gets_umask_default_permissions(self, tmp_path, test_umask): f"got {actual_mode:04o}" ) + @pytest.mark.platforms("linux") def test_overwrite_still_preserves_existing_mode(self, tmp_path): """The new-file branch must not disturb the overwrite path's mode preservation (e.g. an executable script stays 0755).""" @@ -636,6 +640,7 @@ class TestAtomicWriteThroughSymlink: plain file, orphaning the real target and destroying the link (data-loss). """ + @pytest.mark.require_symlinks def test_write_follows_symlink_and_preserves_link(self, tmp_path): ops = ShellFileOperations(make_real_subprocess_env(str(tmp_path))) real = tmp_path / "real.txt" @@ -652,6 +657,7 @@ def test_write_follows_symlink_and_preserves_link(self, tmp_path): assert real.read_text() == "newcontent\n" assert os.path.realpath(link) == str(real) + @pytest.mark.require_symlinks def test_write_through_broken_symlink_falls_back(self, tmp_path): """A broken link resolves through readlink -f and creates the target.""" ops = ShellFileOperations(make_real_subprocess_env(str(tmp_path))) diff --git a/tests/tools/test_file_ops_cwd_tracking.py b/tests/tools/test_file_ops_cwd_tracking.py index 53ea596b7715..4436a15c1839 100644 --- a/tests/tools/test_file_ops_cwd_tracking.py +++ b/tests/tools/test_file_ops_cwd_tracking.py @@ -14,9 +14,10 @@ Fix: _exec() now prefers the LIVE ``env.cwd`` over the init-time ``self.cwd``. Explicit ``cwd`` arg to _exec still wins over both. """ - from __future__ import annotations +import pytest + from tools.file_operations import ShellFileOperations @@ -106,6 +107,7 @@ def execute(self, command, cwd=None, **kwargs): assert result.exit_code == 0 assert "fixed-content" in result.stdout + @pytest.mark.platforms("linux") def test_patch_returns_success_only_when_file_actually_written(self, tmp_path): """Safety rail: patch_replace success must reflect the real file state. diff --git a/tests/tools/test_file_read_guards.py b/tests/tools/test_file_read_guards.py index 1ee28f77892c..492f22f36ab3 100644 --- a/tests/tools/test_file_read_guards.py +++ b/tests/tools/test_file_read_guards.py @@ -6,6 +6,7 @@ Run with: python -m pytest tests/tools/test_file_read_guards.py -v """ +import pytest import json import os @@ -66,6 +67,7 @@ def _make_safe_tempdir(prefix: str) -> str: class TestDevicePathBlocking(unittest.TestCase): """Paths like /dev/zero should be rejected before any I/O.""" + @pytest.mark.platforms("linux") def test_blocked_device_detection(self): for dev in ("/dev/zero", "/dev/random", "/dev/urandom", "/dev/stdin", "/dev/tty", "/dev/console", "/dev/stdout", "/dev/stderr", @@ -76,6 +78,7 @@ def test_safe_device_not_blocked(self): self.assertFalse(_is_blocked_device("/dev/null")) self.assertFalse(_is_blocked_device("/dev/sda1")) + @pytest.mark.platforms("linux") def test_proc_fd_blocked(self): self.assertTrue(_is_blocked_device("/proc/self/fd/0")) self.assertTrue(_is_blocked_device("/proc/12345/fd/2")) @@ -92,6 +95,7 @@ def test_proc_fd_other_not_blocked(self): self.assertFalse(_is_blocked_device_path("/proc/self/fd/3")) + @pytest.mark.platforms("linux") def test_proc_sensitive_pseudo_files_blocked(self): """environ/cmdline/maps (and maps variants) under /proc/ must be blocked (issue #4427).""" for path in ( @@ -116,6 +120,7 @@ def test_proc_sensitive_pseudo_files_blocked(self): ): self.assertTrue(_is_blocked_device(path), f"{path} should be blocked") + @pytest.mark.platforms("linux") def test_proc_task_thread_sensitive_files_blocked(self): """Per-thread /proc//task// aliases leak the same data.""" for path in ( @@ -132,6 +137,7 @@ def test_proc_legitimate_files_not_blocked(self): for path in ("/proc/cpuinfo", "/proc/meminfo", "/proc/uptime", "/proc/version"): self.assertFalse(_is_blocked_device(path), f"{path} should not be blocked") + @pytest.mark.platforms("linux") def test_normpath_alias_to_blocked_device_is_blocked(self): self.assertTrue(_is_blocked_device("/dev/../dev/zero")) self.assertTrue(_is_blocked_device("/dev/./urandom")) @@ -162,6 +168,7 @@ def test_symlink_to_regular_file_not_blocked(self): self.assertFalse(_is_blocked_device(link_path)) + @pytest.mark.platforms("linux") def test_read_file_tool_rejects_device(self): """read_file_tool returns an error without any file I/O.""" result = json.loads(read_file_tool("/dev/zero", task_id="dev_test")) diff --git a/tests/tools/test_file_sync.py b/tests/tools/test_file_sync.py index 9ecabdb21bca..f6d737c13714 100644 --- a/tests/tools/test_file_sync.py +++ b/tests/tools/test_file_sync.py @@ -264,6 +264,7 @@ def test_file_disappears_between_list_and_upload(self, tmp_path): class TestConcurrency: + @pytest.mark.platforms("linux") def test_sync_back_waits_for_active_sync_transaction(self, tmp_path): initial_file = tmp_path / "initial.png" new_file = tmp_path / "new.png" @@ -317,6 +318,7 @@ def bulk_download(destination): class TestSyncBackSecurity: + @pytest.mark.platforms("linux") def test_sync_back_does_not_overwrite_uploaded_credential_files(self, tmp_path, monkeypatch): credential = tmp_path / "token.json" credential.write_text("host-token", encoding="utf-8") diff --git a/tests/tools/test_file_tools.py b/tests/tools/test_file_tools.py index 58fb91ac2b10..387737cf440d 100644 --- a/tests/tools/test_file_tools.py +++ b/tests/tools/test_file_tools.py @@ -44,6 +44,7 @@ def test_exception_returns_error_json(self, mock_get): class TestWriteFileHandler: @patch("tools.file_tools._get_file_ops") + @pytest.mark.platforms("linux") def test_writes_content(self, mock_get): mock_ops = MagicMock() result_obj = MagicMock() @@ -132,6 +133,7 @@ def test_non_string_content_returns_error(self): class TestPatchHandler: @patch("tools.file_tools._get_file_ops") + @pytest.mark.platforms("linux") def test_replace_mode_calls_patch_replace(self, mock_get): mock_ops = MagicMock() result_obj = MagicMock() @@ -225,6 +227,7 @@ class TestPatchSensitivePathExtraction: """ @patch("tools.file_tools._get_file_ops") + @pytest.mark.platforms("linux") def test_patch_move_to_sensitive_dst_blocked(self, mock_get): from tools.file_tools import patch_tool patch_text = ( @@ -239,6 +242,7 @@ def test_patch_move_to_sensitive_dst_blocked(self, mock_get): @patch("tools.file_tools._get_file_ops") + @pytest.mark.platforms("linux") def test_patch_update_no_space_after_asterisks_blocked(self, mock_get): """``***Update File:`` (no space after asterisks) must also be caught. @@ -312,7 +316,7 @@ def test_search_exception_returns_error(self, mock_get): class TestWindowsMsysPathResolution: """File tools must translate Git Bash drive paths before Path resolution.""" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_absolute_msys_path_normalized_before_windows_resolve(self, monkeypatch): """Windows-only: ``_resolve_path_for_task`` hands the translated path to ``ntpath``/``Path``, and only a real Windows ``Path`` renders @@ -325,7 +329,7 @@ def test_absolute_msys_path_normalized_before_windows_resolve(self, monkeypatch) assert str(resolved) == r"C:\Users\Mark\project\app.py" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_container_paths_skip_msys_translation(self, monkeypatch): """WSL/docker Linux paths must not be rewritten as Windows drives. @@ -455,6 +459,7 @@ def test_hermes_config_blocked_via_tilde_path(self, tmp_path, monkeypatch): assert "Hermes config" in result["error"] + @pytest.mark.platforms("linux") def test_system_path_still_blocked(self, monkeypatch): monkeypatch.setattr("tools.file_tools._hermes_config_resolved", "/some/other/path") monkeypatch.setattr("tools.file_tools._hermes_config_resolved_loaded", True) @@ -464,6 +469,7 @@ def test_system_path_still_blocked(self, monkeypatch): assert "error" in result assert "sensitive system path" in result["error"] + @pytest.mark.platforms("linux") def test_macos_private_var_carveouts(self): """macOS temp dirs under /private/var must not be blanket-blocked, while the genuinely-sensitive /private/var subtrees still are.""" diff --git a/tests/tools/test_file_tools_cwd_resolution.py b/tests/tools/test_file_tools_cwd_resolution.py index 28f7ac5f5d81..834f2500e8fc 100644 --- a/tests/tools/test_file_tools_cwd_resolution.py +++ b/tests/tools/test_file_tools_cwd_resolution.py @@ -89,6 +89,7 @@ def test_absolute_terminal_cwd_used_verbatim(_isolated_cwd, monkeypatch): assert resolved == (workspace / "target.py") +@pytest.mark.require_symlinks def test_container_absolute_input_path_does_not_follow_host_symlink(tmp_path, monkeypatch): """Docker paths are sandbox-local and must not be host-dereferenced. @@ -118,6 +119,7 @@ def test_container_path_normalization_uses_posix_path_syntax(): assert str(resolved) == "/workspace/projects/bar" +@pytest.mark.require_symlinks def test_container_relative_path_keeps_container_cwd_symlink(tmp_path, monkeypatch): """Relative Docker paths should stay under the container cwd textually.""" host_project = tmp_path / "host-project" @@ -166,8 +168,11 @@ def test_warning_fires_when_relative_path_escapes_workspace(_isolated_cwd, monke assert warn is not None assert "OUTSIDE the active workspace" in warn - assert str(decoy) in warn - assert str(workspace) in warn + # Paths are repr-embedded in the message, so backslashes arrive doubled + # on Windows; un-escape before the containment check (host-aware). + warn_flat = warn.replace("\\\\", "\\") + assert str(decoy) in warn_flat + assert str(workspace) in warn_flat # ── Fix C: sentinel TERMINAL_CWD + empty-registry worktree anchoring ───────── @@ -199,7 +204,7 @@ def test_warning_fires_from_terminal_cwd_when_registry_empty(_isolated_cwd, monk assert warn is not None assert "OUTSIDE the active workspace" in warn - assert str(workspace) in warn + assert str(workspace) in warn.replace("\\\\", "\\") # ── Fix A: write_file / patch report the resolved ABSOLUTE path ────────────── diff --git a/tests/tools/test_file_tools_live.py b/tests/tools/test_file_tools_live.py index 81a11c061ad3..42a51c03566d 100644 --- a/tests/tools/test_file_tools_live.py +++ b/tests/tools/test_file_tools_live.py @@ -98,6 +98,7 @@ def test_printf_no_trailing_newline(self, env): _assert_clean(result["output"]) + @pytest.mark.platforms("linux") def test_cat_deterministic_content(self, env, tmp_path): f = tmp_path / "det.txt" f.write_text(SIMPLE_CONTENT) @@ -225,6 +226,7 @@ def test_search_output_has_zero_noise(self, ops, populated_dir): # ── _expand_path ───────────────────────────────────────────────────────── class TestExpandPath: + @pytest.mark.platforms("linux") def test_tilde_exact(self, ops): result = ops._expand_path("~/test.txt") expected = f"{Path.home()}/test.txt" @@ -261,6 +263,7 @@ def test_echo(self, env): assert result["output"].strip() == "CLEAN_TEST" _assert_clean(result["output"]) + @pytest.mark.platforms("linux") def test_cat(self, env, tmp_path): f = tmp_path / "cat_test.txt" f.write_text("CAT_CONTENT_EXACT\n") diff --git a/tests/tools/test_file_tools_tilde_profile.py b/tests/tools/test_file_tools_tilde_profile.py index 23510b1f9ae0..93fb3cbb9106 100644 --- a/tests/tools/test_file_tools_tilde_profile.py +++ b/tests/tools/test_file_tools_tilde_profile.py @@ -31,6 +31,7 @@ class TestExpandTilde: """Verify the _expand_tilde() helper resolves ~ to the profile home.""" + @pytest.mark.platforms("linux") def test_tilde_expands_to_profile_home(self): """When get_subprocess_home returns a value, ~/path uses it.""" with patch("hermes_constants.get_subprocess_home", return_value="/opt/data/profiles/coder/home"): diff --git a/tests/tools/test_file_write_safety.py b/tests/tools/test_file_write_safety.py index fc92b55b09e5..e5b29972f754 100644 --- a/tests/tools/test_file_write_safety.py +++ b/tests/tools/test_file_write_safety.py @@ -232,22 +232,27 @@ def test_write_file_allowed_path_returns_no_error( class TestCheckSensitivePathMacOSBypass: """Verify _check_sensitive_path blocks /private/etc paths (issue #8734).""" + @pytest.mark.platforms("linux") def test_etc_hosts_blocked(self): from tools.file_tools import _check_sensitive_path assert _check_sensitive_path("/etc/hosts") is not None + @pytest.mark.platforms("linux") def test_private_etc_hosts_blocked(self): from tools.file_tools import _check_sensitive_path assert _check_sensitive_path("/private/etc/hosts") is not None + @pytest.mark.platforms("linux") def test_private_etc_ssh_config_blocked(self): from tools.file_tools import _check_sensitive_path assert _check_sensitive_path("/private/etc/ssh/sshd_config") is not None + @pytest.mark.platforms("linux") def test_private_var_blocked(self): from tools.file_tools import _check_sensitive_path assert _check_sensitive_path("/private/var/db/something") is not None + @pytest.mark.platforms("linux") def test_boot_still_blocked(self): from tools.file_tools import _check_sensitive_path assert _check_sensitive_path("/boot/grub/grub.cfg") is not None @@ -291,6 +296,7 @@ def test_no_temp_file_leaked_on_success(self, ops, tmp_path: Path): assert [p for p in os.listdir(tmp_path) if ".hermes-tmp" in p] == [] + @pytest.mark.platforms("linux") def test_patch_routes_through_atomic_write(self, ops, tmp_path: Path): target = tmp_path / "edit.py" target.write_text("a = 1\nb = 2\nc = 3\n", encoding="utf-8") @@ -508,6 +514,7 @@ def test_extra_patterns_from_config(self, tmp_path, approvals, monkeypatch): # ---- adversarial path shapes ---------------------------------------- + @pytest.mark.require_symlinks def test_symlink_to_protected_file_is_gated(self, tmp_path, approvals): """#41351 lesson: realpath first — innocent name, protected target.""" real = tmp_path / "AGENTS.md" diff --git a/tests/tools/test_find_shell.py b/tests/tools/test_find_shell.py index e84fdef0ce13..2947d4069f1d 100644 --- a/tests/tools/test_find_shell.py +++ b/tests/tools/test_find_shell.py @@ -18,6 +18,7 @@ class TestFindShellPrefersUserShell: """_find_shell should prefer $SHELL over bash on POSIX.""" + @pytest.mark.platforms("linux") def test_returns_shell_env_when_set_and_exists(self, tmp_path): """When $SHELL points to an existing allowlisted executable, _find_shell returns it.""" fake_zsh = tmp_path / "zsh" @@ -46,6 +47,7 @@ def test_falls_back_for_incompatible_shell_fish(self, tmp_path): assert _find_shell() == _find_bash() + @pytest.mark.platforms("linux") def test_honours_allowlisted_bash_and_dash(self, tmp_path): """Every allowlisted POSIX-sh-family shell is honoured.""" for name in ("bash", "dash", "sh", "ksh"): @@ -65,7 +67,7 @@ def test_falls_back_to_find_bash_when_shell_empty(self): class TestFindShellWindowsBehavior: """On Windows, _find_shell always delegates to _find_bash.""" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_ignores_shell_env(self): """On Windows, $SHELL is ignored — _find_shell delegates to _find_bash. @@ -105,7 +107,7 @@ def test_find_bash_still_prefers_bash(self): class TestFindBashSkipsBrokenCustomPath: """Stale HERMES_GIT_BASH_PATH must not brick Windows terminal startup.""" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_falls_through_to_portable_when_custom_fails_probe(self, tmp_path, monkeypatch): """Windows-only: the candidate ladder (HERMES_GIT_BASH_PATH → %LOCALAPPDATA%\\hermes\\git → Program Files) only exists in @@ -154,7 +156,7 @@ def fake_run(argv, **kwargs): assert local_mod._bash_starts(r"C:\Git\bin\bash.exe") is True assert calls[0][0][-1] == "/usr/bin/true; /usr/bin/cat --version >/dev/null" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_aslr_failure_surfaces_targeted_windows_command( self, tmp_path, monkeypatch ): @@ -193,7 +195,7 @@ def failed_probe(path: str) -> bool: assert str(tmp_path / "hermes" / "git") in message -@pytest.mark.macos_only +@pytest.mark.platforms("macos") @pytest.mark.skipif( not os.path.isfile("/bin/bash"), reason="reproduces the macOS system-bash-3.2 login-shell swallow", diff --git a/tests/tools/test_interrupted_command_cwd.py b/tests/tools/test_interrupted_command_cwd.py index ecf14f176ed6..630874b66fd1 100644 --- a/tests/tools/test_interrupted_command_cwd.py +++ b/tests/tools/test_interrupted_command_cwd.py @@ -58,6 +58,7 @@ def test_completed_command_reports_its_cwd(self, env, tmp_path): result, _ = _run(env, "sess", f"cd {target} && pwd") assert result["cwd_observed"] is True + @pytest.mark.platforms("linux") def test_interrupted_command_reports_no_cwd(self, env, tmp_path): target = tmp_path / "slow" target.mkdir() @@ -67,6 +68,7 @@ def test_interrupted_command_reports_no_cwd(self, env, tmp_path): class TestInterruptDoesNotStealAnotherSessionsCwd: + @pytest.mark.platforms("linux") def test_record_survives_an_interrupt(self, env, tmp_path): mine = tmp_path / "mine" theirs = tmp_path / "theirs" @@ -84,6 +86,7 @@ def test_record_survives_an_interrupt(self, env, tmp_path): _run(env, "mine", f"cd {mine} && sleep 20", timeout=2) assert tt.get_session_cwd("mine") == str(mine) + @pytest.mark.platforms("linux") def test_next_command_still_runs_in_my_directory(self, env, tmp_path): mine = tmp_path / "mine" theirs = tmp_path / "theirs" @@ -97,6 +100,7 @@ def test_next_command_still_runs_in_my_directory(self, env, tmp_path): result, _ = _run(env, "mine", "pwd") assert os.path.realpath(result["output"].strip()) == os.path.realpath(str(mine)) + @pytest.mark.platforms("linux") def test_single_session_keeps_its_own_prior_directory(self, env, tmp_path): """No second session needed: a lone session must not re-home either.""" first = tmp_path / "first" @@ -167,6 +171,7 @@ def _tool(self, monkeypatch, env, command, task_id, timeout=None): tt.terminal_tool(command=command, task_id=task_id, timeout=timeout) ) + @pytest.mark.platforms("linux") def test_interrupt_keeps_record_and_echoes_no_cwd(self, env, tmp_path, monkeypatch): mine = tmp_path / "mine" theirs = tmp_path / "theirs" @@ -194,6 +199,7 @@ def test_interrupt_keeps_record_and_echoes_no_cwd(self, env, tmp_path, monkeypat # my command_cwd (mine), so the echo would fire with THEIR directory. assert "cwd" not in result or result.get("cwd") is None + @pytest.mark.platforms("linux") def test_completed_command_still_records_and_echoes(self, env, tmp_path, monkeypatch): """The gate must not break the observed path: cd still round-trips.""" target = tmp_path / "target" diff --git a/tests/tools/test_lazy_deps.py b/tests/tools/test_lazy_deps.py index 74838a69a388..1da1d3072728 100644 --- a/tests/tools/test_lazy_deps.py +++ b/tests/tools/test_lazy_deps.py @@ -350,7 +350,7 @@ def test_windows_matrix_refresh_is_skipped_before_pip(self, monkeypatch): assert result["platform.matrix"].startswith("skipped:") assert "unsupported on Windows" in result["platform.matrix"] - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_matrix_probe_reports_unsupported_on_real_windows(self): # The probe itself keys off the real host: patching sys.platform only # proved the string, never that Windows actually hits this gate. diff --git a/tests/tools/test_lazy_deps_durable_target.py b/tests/tools/test_lazy_deps_durable_target.py index cb9b517c51a6..91ee19f8ff71 100644 --- a/tests/tools/test_lazy_deps_durable_target.py +++ b/tests/tools/test_lazy_deps_durable_target.py @@ -91,6 +91,7 @@ def test_creates_dir_and_stamp(self, tmp_path): assert stamp.read_text().strip() == ld._python_abi_tag() + @pytest.mark.platforms("linux") def test_readonly_target_reports_error(self, tmp_path): # A path under a non-writable parent should surface a clean error, # not raise. diff --git a/tests/tools/test_local_background_child_hang.py b/tests/tools/test_local_background_child_hang.py index a9251cc74003..539e2b111571 100644 --- a/tests/tools/test_local_background_child_hang.py +++ b/tests/tools/test_local_background_child_hang.py @@ -70,6 +70,7 @@ def test_setsid_disown_pattern_returns_promptly(self, local_env): _pkill("time.sleep(60)") + @pytest.mark.platforms("linux") def test_default_capture_is_full_fidelity_for_internal_consumers( self, local_env ): @@ -96,6 +97,7 @@ def test_default_capture_is_full_fidelity_for_internal_consumers( assert len(result["output"]) > 200000 + @pytest.mark.platforms("linux") def test_utf8_multibyte_across_read_boundary(self, local_env): """Multibyte UTF-8 characters straddling a 4096-byte ``os.read()`` boundary must be decoded correctly via the incremental decoder — not lost to a @@ -121,6 +123,7 @@ def test_utf8_multibyte_across_read_boundary(self, local_env): # And the "[binary output detected ...]" fallback must NOT fire assert "binary output detected" not in result["output"] + @pytest.mark.platforms("linux") def test_invalid_utf8_uses_replacement_not_fallback(self, local_env): """Truly invalid byte sequences must be substituted with U+FFFD (matching the pre-fix ``errors='replace'`` behaviour of the old ``TextIOWrapper`` diff --git a/tests/tools/test_local_env_blocklist.py b/tests/tools/test_local_env_blocklist.py index 5cae08330b5a..91dd7a0e184d 100644 --- a/tests/tools/test_local_env_blocklist.py +++ b/tests/tools/test_local_env_blocklist.py @@ -764,7 +764,7 @@ def test_windows_backslash_paths(self): assert hermes_win in entries assert user_win in entries - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_hermes_owned_paths_stripped(self): """On Windows, a Hermes venv site-packages entry written with backslashes is stripped by the same Hermes-owned check, while a @@ -1590,7 +1590,7 @@ def test_make_run_env_appends_homebrew_on_minimal_path(self, monkeypatch): assert entry in path_entries - @pytest.mark.macos_only + @pytest.mark.platforms("macos") def test_make_run_env_real_launchd_path_gains_homebrew(self): """The literal macOS launchd PATH is the production trigger for #35613. @@ -1608,7 +1608,7 @@ def test_make_run_env_real_launchd_path_gains_homebrew(self): assert path_entries[:4] == ["/usr/bin", "/bin", "/usr/sbin", "/sbin"] - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_make_run_env_preserves_windows_mixed_case_path_key(self, monkeypatch): """Windows-only: ``_path_env_key`` looks for a case-insensitive PATH key only on Windows, so the mixed-case ``Path`` preservation this diff --git a/tests/tools/test_local_env_cwd_recovery.py b/tests/tools/test_local_env_cwd_recovery.py index 41f101aaad4d..6e2069ce2dbd 100644 --- a/tests/tools/test_local_env_cwd_recovery.py +++ b/tests/tools/test_local_env_cwd_recovery.py @@ -7,6 +7,7 @@ Regression coverage for https://github.com/NousResearch/hermes-agent/issues/17558. """ +import pytest import os import shutil @@ -28,6 +29,7 @@ def test_returns_cwd_when_directory_exists(self, tmp_path): assert _resolve_safe_cwd(path) == path + @pytest.mark.platforms("linux") def test_returns_root_when_only_root_exists(self, monkeypatch): """If every ancestor except the filesystem root is gone, the root itself is still a valid recovery target — don't skip it just because diff --git a/tests/tools/test_local_env_relative_cwd.py b/tests/tools/test_local_env_relative_cwd.py index 46a9a563ecb0..f55e4c0d9f25 100644 --- a/tests/tools/test_local_env_relative_cwd.py +++ b/tests/tools/test_local_env_relative_cwd.py @@ -3,6 +3,7 @@ from pathlib import Path from tools.environments.local import LocalEnvironment, _resolve_local_initial_cwd +import pytest def test_relative_initial_cwd_resolves_from_parent(tmp_path, monkeypatch): @@ -13,6 +14,7 @@ def test_relative_initial_cwd_resolves_from_parent(tmp_path, monkeypatch): assert _resolve_local_initial_cwd("hermes-agent") == str(project) +@pytest.mark.platforms("linux") def test_local_environment_keeps_existing_relative_child_cwd(tmp_path, monkeypatch): project = tmp_path / "hermes-agent" project.mkdir() diff --git a/tests/tools/test_local_env_windows_msys.py b/tests/tools/test_local_env_windows_msys.py index 79a3edff0e7a..853d47e4d167 100644 --- a/tests/tools/test_local_env_windows_msys.py +++ b/tests/tools/test_local_env_windows_msys.py @@ -23,9 +23,9 @@ on a host where the path semantics, the drive letters, the path separator, and Git Bash itself are all absent. -So the Windows-behaviour tests are ``windows_only`` and run on the +So the Windows-behaviour tests are ``platforms("windows")`` and run on the Windows CI job against a real Git Bash layout. The "no-op off Windows" -cases assert genuine POSIX behaviour and are ``linux_only`` — on that +cases assert genuine POSIX behaviour and are ``platforms("linux")`` — on that host ``_IS_WINDOWS`` is already False, so no patching is needed at all. """ @@ -56,19 +56,19 @@ # --------------------------------------------------------------------------- class TestMsysToWindowsPath: - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_noop_on_non_windows(self): # On a non-Windows host the function must never rewrite the path # — POSIX-style paths are real paths there. assert _msys_to_windows_path("/c/Users/NVIDIA") == "/c/Users/NVIDIA" assert _msys_to_windows_path("/home/teknium") == "/home/teknium" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_translates_drive_path(self): assert _msys_to_windows_path("/c/Users/NVIDIA") == r"C:\Users\NVIDIA" assert _msys_to_windows_path("/d/Projects/foo bar") == r"D:\Projects\foo bar" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_empty_string(self): assert _msys_to_windows_path("") == "" @@ -78,11 +78,11 @@ def test_empty_string(self): # --------------------------------------------------------------------------- class TestWindowsToMsysPath: - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_noop_on_non_windows(self): assert _windows_to_msys_path(r"C:\Users\NVIDIA") == r"C:\Users\NVIDIA" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_does_not_translate_non_drive_path(self): assert _windows_to_msys_path("/tmp/foo") == "/tmp/foo" assert _windows_to_msys_path(r"\\server\share") == r"\\server\share" @@ -92,7 +92,7 @@ def test_does_not_translate_non_drive_path(self): # _bash_safe_path / _quote_bash_path — shell-script interpolation # --------------------------------------------------------------------------- -@pytest.mark.windows_only +@pytest.mark.platforms("windows") class TestBashSafePath: def test_native_windows_path_becomes_msys(self): assert _bash_safe_path(r"C:\Users\alice\notes.txt") == "/c/Users/alice/notes.txt" @@ -109,7 +109,7 @@ def test_quote_bash_path_quotes_mixed_windows_path(self): # _resolve_safe_cwd — Windows fast path # --------------------------------------------------------------------------- -@pytest.mark.windows_only +@pytest.mark.platforms("windows") class TestResolveSafeCwdWindows: def test_msys_path_resolves_to_native_when_native_exists(self, tmp_path): """The whole point of this fix: a Git Bash ``/c/Users/x`` value @@ -130,7 +130,7 @@ def test_msys_path_resolves_to_native_when_native_exists(self, tmp_path): # End-to-end: _update_cwd via stdout marker # --------------------------------------------------------------------------- -@pytest.mark.windows_only +@pytest.mark.platforms("windows") class TestUpdateCwdWindowsMsys: def test_marker_output_msys_path_stored_in_native_form(self, tmp_path): """When Git Bash emits ``/c/Users/x`` in the cwd marker on Windows, @@ -166,7 +166,7 @@ def test_marker_output_msys_path_stored_in_native_form(self, tmp_path): # End-to-end: _extract_cwd_from_output rollback when marker is invalid # --------------------------------------------------------------------------- -@pytest.mark.windows_only +@pytest.mark.platforms("windows") class TestExtractCwdFromOutputWindowsMsys: def test_stale_msys_marker_does_not_clobber_cwd(self, tmp_path): """When the cwd marker in stdout points at a non-existent path, @@ -218,7 +218,7 @@ def test_valid_msys_marker_normalized_to_native(self, tmp_path): # MSYS_NO_PATHCONV — native Windows command flags (#56700) # --------------------------------------------------------------------------- -@pytest.mark.windows_only +@pytest.mark.platforms("windows") class TestWindowsMsysPathconvDefaults: def test_make_run_env_sets_msys_no_pathconv_on_windows(self): run_env = _make_run_env({}) @@ -247,7 +247,7 @@ def _fake_isdir(self, existing): existing = {e.replace("\\", "/") for e in existing} return lambda p: p.replace("\\", "/") in existing - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_derives_dirs_from_portablegit_layout(self, monkeypatch): """The PortableGit layout probe, run on the real OS. @@ -277,12 +277,12 @@ def test_derives_dirs_from_portablegit_layout(self, monkeypatch): # Non-existent dirs (mingw32, usr/local/bin) are excluded. assert "/pg/mingw32/bin" not in norm - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_empty_off_windows(self, monkeypatch): monkeypatch.setattr(local_mod, "_git_bash_bin_dirs_cache", None) assert _git_bash_bin_dirs() == [] - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_make_run_env_noop_on_posix(self, monkeypatch): monkeypatch.setattr(local_mod, "_git_bash_bin_dirs_cache", None) run_env = _make_run_env({"PATH": "/usr/bin:/bin"}) @@ -294,7 +294,7 @@ def test_make_run_env_noop_on_posix(self, monkeypatch): # Command wrapping — native Windows cwd must be Git Bash-friendly for cd # --------------------------------------------------------------------------- -@pytest.mark.windows_only +@pytest.mark.platforms("windows") class TestWrapCommandWindowsNativeCwd: def test_wrap_command_converts_native_cwd_for_builtin_cd(self): with patch.object( diff --git a/tests/tools/test_local_interrupt_cleanup.py b/tests/tools/test_local_interrupt_cleanup.py index 74b1d55fd33d..9946406c6fc0 100644 --- a/tests/tools/test_local_interrupt_cleanup.py +++ b/tests/tools/test_local_interrupt_cleanup.py @@ -65,6 +65,7 @@ def _wait_for_pgid_exit(pgid: int, timeout: float = 60.0) -> bool: return not _pgid_still_alive(pgid) +@pytest.mark.platforms("linux") def test_kill_process_uses_cached_pgid_if_wrapper_already_exited(monkeypatch): """If the shell wrapper exits before cleanup, still kill its process group. @@ -97,6 +98,7 @@ def fake_killpg(pgid, sig): assert killpg_calls == [(67890, signal.SIGTERM), (67890, 0)] +@pytest.mark.platforms("linux") def test_wait_for_process_kills_subprocess_on_keyboardinterrupt(): """When KeyboardInterrupt arrives mid-poll, the subprocess group must be killed before the exception is re-raised.""" diff --git a/tests/tools/test_local_setsid_descendant_sweep.py b/tests/tools/test_local_setsid_descendant_sweep.py index 8ae35d68f746..0ad18ddd8963 100644 --- a/tests/tools/test_local_setsid_descendant_sweep.py +++ b/tests/tools/test_local_setsid_descendant_sweep.py @@ -17,6 +17,8 @@ import pytest +pytestmark = pytest.mark.platforms("linux") + from tools.environments.local import LocalEnvironment diff --git a/tests/tools/test_local_shell_init.py b/tests/tools/test_local_shell_init.py index 0f4f9af34aa9..335777aa19b0 100644 --- a/tests/tools/test_local_shell_init.py +++ b/tests/tools/test_local_shell_init.py @@ -19,6 +19,7 @@ class TestResolveShellInitFiles: + @pytest.mark.platforms("linux") def test_auto_sources_bashrc_when_present(self, tmp_path, monkeypatch): bashrc = tmp_path / ".bashrc" bashrc.write_text('export MARKER=seen\n') @@ -33,6 +34,7 @@ def test_auto_sources_bashrc_when_present(self, tmp_path, monkeypatch): assert resolved == [str(bashrc)] + @pytest.mark.platforms("linux") def test_auto_sources_profile_when_present(self, tmp_path, monkeypatch): """~/.profile is where ``n`` / ``nvm`` installers typically write their PATH export on Debian/Ubuntu, and it has no interactivity @@ -51,6 +53,7 @@ def test_auto_sources_profile_when_present(self, tmp_path, monkeypatch): assert resolved == [str(profile)] + @pytest.mark.platforms("linux") def test_auto_sources_profile_before_bashrc(self, tmp_path, monkeypatch): """Both files present: profile runs first so PATH exports in profile take effect even if bashrc short-circuits on the diff --git a/tests/tools/test_local_tempdir.py b/tests/tools/test_local_tempdir.py index b07b1b77ee46..93377ff90a29 100644 --- a/tests/tools/test_local_tempdir.py +++ b/tests/tools/test_local_tempdir.py @@ -1,9 +1,11 @@ from unittest.mock import patch from tools.environments.local import LocalEnvironment +import pytest class TestLocalTempDir: + @pytest.mark.platforms("linux") def test_uses_os_tmpdir_for_session_artifacts(self, monkeypatch): monkeypatch.setenv("TMPDIR", "/data/data/com.termux/files/usr/tmp") monkeypatch.delenv("TMP", raising=False) @@ -17,6 +19,7 @@ def test_uses_os_tmpdir_for_session_artifacts(self, monkeypatch): assert env._cwd_file == f"/data/data/com.termux/files/usr/tmp/hermes-cwd-{env._session_id}.txt" + @pytest.mark.platforms("linux") def test_falls_back_to_tempfile_when_tmp_missing(self, monkeypatch): monkeypatch.delenv("TMPDIR", raising=False) monkeypatch.delenv("TMP", raising=False) diff --git a/tests/tools/test_macos_protected_search.py b/tests/tools/test_macos_protected_search.py index 501280a85f73..c0e3ce44e41a 100644 --- a/tests/tools/test_macos_protected_search.py +++ b/tests/tools/test_macos_protected_search.py @@ -3,7 +3,7 @@ from pathlib import Path import tools.file_operations as file_operations -from tools.environments.local import LocalEnvironment +from tools.environments.local import LocalEnvironment, _bash_safe_path from tools.file_operations import ShellFileOperations, _macos_protected_search_exclusions @@ -128,7 +128,7 @@ def test_grep_fallback_prunes_by_path_not_basename(tmp_path, monkeypatch): for dirname in PROTECTED_NAMES: # Path-scoped pruning: full protected path present, no basename-wide # --exclude-dir for protected names. - assert str(home / dirname) in pruned_command + assert _bash_safe_path(str(home / dirname)) in pruned_command assert f"--exclude-dir={dirname}" not in pruned_command assert f"--exclude-dir='{dirname}'" not in pruned_command @@ -188,7 +188,7 @@ def test_find_fallback_prunes_protected_directories(tmp_path, monkeypatch): find_commands = [command for command in env.commands if command.startswith("find ")] assert find_commands for command in find_commands: - assert str(home / "Downloads") in command + assert _bash_safe_path(str(home / "Downloads")) in command assert "-prune" in command diff --git a/tests/tools/test_mcp_schema_cache.py b/tests/tools/test_mcp_schema_cache.py index cc6df9a6d29c..6f0c6e354770 100644 --- a/tests/tools/test_mcp_schema_cache.py +++ b/tests/tools/test_mcp_schema_cache.py @@ -4,6 +4,8 @@ fingerprint keying, read/write round-trip, and invalidation behavior. """ +import pytest + import tools.mcp_schema_cache as msc @@ -75,6 +77,7 @@ def test_malformed_entry_shapes_are_tolerated(self): class TestCacheFileLocation: + @pytest.mark.platforms("linux") def test_cache_lives_under_hermes_home_cache_dir_with_0600( self, monkeypatch, tmp_path ): diff --git a/tests/tools/test_mcp_tool.py b/tests/tools/test_mcp_tool.py index 5f05b0fc3432..98ded455a1d8 100644 --- a/tests/tools/test_mcp_tool.py +++ b/tests/tools/test_mcp_tool.py @@ -1278,9 +1278,9 @@ def test_windows_location_vars_passed_without_secrets(self): fake_env = { "PATH": r"C:\Windows\System32", - "ProgramFiles": r"C:\Program Files", - "ProgramData": r"C:\ProgramData", - "ProgramW6432": r"C:\Program Files", + "PROGRAMFILES": r"C:\Program Files", + "PROGRAMDATA": r"C:\ProgramData", + "PROGRAMW6432": r"C:\Program Files", "LOCALAPPDATA": r"C:\Users\alice\AppData\Local", "APPDATA": r"C:\Users\alice\AppData\Roaming", "USERPROFILE": r"C:\Users\alice", @@ -1290,9 +1290,9 @@ def test_windows_location_vars_passed_without_secrets(self): with patch.dict("os.environ", fake_env, clear=True): result = _build_safe_env(None) - assert result["ProgramFiles"] == r"C:\Program Files" - assert result["ProgramData"] == r"C:\ProgramData" - assert result["ProgramW6432"] == r"C:\Program Files" + assert result["PROGRAMFILES"] == r"C:\Program Files" + assert result["PROGRAMDATA"] == r"C:\ProgramData" + assert result["PROGRAMW6432"] == r"C:\Program Files" assert result["LOCALAPPDATA"].endswith("Local") assert result["APPDATA"].endswith("Roaming") assert result["USERPROFILE"] == r"C:\Users\alice" diff --git a/tests/tools/test_modal_sandbox_fixes.py b/tests/tools/test_modal_sandbox_fixes.py index 89878150dcf9..c0148923f5d3 100644 --- a/tests/tools/test_modal_sandbox_fixes.py +++ b/tests/tools/test_modal_sandbox_fixes.py @@ -85,6 +85,7 @@ def test_users_path_replaced_for_docker_by_default(self, monkeypatch): assert config["host_cwd"] is None assert config["docker_mount_cwd_to_workspace"] is False + @pytest.mark.platforms("linux") def test_users_path_maps_to_workspace_for_docker_when_enabled(self, monkeypatch): """Docker should map the host cwd into /workspace only when explicitly enabled.""" monkeypatch.setenv("TERMINAL_ENV", "docker") @@ -133,6 +134,7 @@ def test_default_cwd_is_root_for_container_backends(self, backend, monkeypatch): f"Backend {backend}: expected /root default, got {config['cwd']}" ) + @pytest.mark.platforms("linux") def test_docker_default_cwd_maps_current_directory_when_enabled(self, monkeypatch): """Docker should use /workspace when cwd mounting is explicitly enabled.""" monkeypatch.setattr("tools.terminal_tool.os.getcwd", lambda: "/home/user/project") diff --git a/tests/tools/test_plugin_guard.py b/tests/tools/test_plugin_guard.py index ac40f3ad1c94..35e70a507b93 100644 --- a/tests/tools/test_plugin_guard.py +++ b/tests/tools/test_plugin_guard.py @@ -115,6 +115,7 @@ def test_reverse_shell_is_dangerous(self, tmp_path): result = scan_plugin(plugin) assert result.verdict == "dangerous" + @pytest.mark.require_symlinks def test_symlink_escape_is_dangerous(self, tmp_path): plugin = _mk_plugin(tmp_path, BASE_FILES) outside = tmp_path / "outside-secret.txt" diff --git a/tests/tools/test_pr_6656_regressions.py b/tests/tools/test_pr_6656_regressions.py index c3f10a440620..c1f06da9f908 100644 --- a/tests/tools/test_pr_6656_regressions.py +++ b/tests/tools/test_pr_6656_regressions.py @@ -127,6 +127,7 @@ def test_absolute_path_rejected(self, hub_setup): assert ok is False assert victim.exists() + @pytest.mark.require_symlinks def test_symlink_escape_rejected(self, tmp_path, hub_setup): """Symlinks inside SKILLS_DIR that point outside must be refused after realpath resolution.""" diff --git a/tests/tools/test_process_registry.py b/tests/tools/test_process_registry.py index 8725a7f507bb..32e943d15cc8 100644 --- a/tests/tools/test_process_registry.py +++ b/tests/tools/test_process_registry.py @@ -144,7 +144,7 @@ def _wait_until(predicate, timeout: float = 5.0, interval: float = 0.05) -> bool return False -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_write_stdin_uses_str_for_windows_pty(registry): """pywinpty expects str input; bytes raises a PyString conversion error. @@ -168,7 +168,7 @@ def write(self, value): assert isinstance(written[0], str) -@pytest.mark.linux_only +@pytest.mark.platforms("linux") def test_write_stdin_uses_bytes_for_posix_pty(registry): """The POSIX counterpart: ptyprocess expects bytes, not str.""" written = [] @@ -187,7 +187,7 @@ def write(self, value): assert written == [b"hello\n"] -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_submit_stdin_uses_crlf_for_windows_pty(registry): """Enter on a Windows PTY is a carriage return, not a bare LF. @@ -213,7 +213,7 @@ def write(self, value): assert written == ["Y\r\n"] -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_submit_stdin_keeps_lf_for_windows_pipe(registry): """Non-PTY (Popen pipe) sessions keep the plain LF on Windows.""" session = _make_session(sid="pipe-win-submit") @@ -570,6 +570,7 @@ def test_close_stdin_pipe_mode(self, registry): proc.stdin.close.assert_called_once() assert result["status"] == "ok" + @pytest.mark.platforms("linux") def test_close_stdin_allows_eof_driven_process_to_finish(self, registry, tmp_path): """PTY mode: writing data + sending EOF lets an EOF-driven child finish. @@ -707,6 +708,7 @@ def test_prune_over_max_removes_oldest(self, registry): # ========================================================================= class TestSpawnEnvSanitization: + @pytest.mark.platforms("linux") def test_spawn_local_strips_blocked_vars_from_background_env(self, registry): captured = {} @@ -814,6 +816,7 @@ def execute(self, command, **kwargs): class TestPopenLeakOnSetupFailure: """Regression for issue #2749: subprocess orphaned when post-Popen setup raises.""" + @pytest.mark.platforms("linux") def test_popen_killed_when_thread_creation_fails(self, registry): """If Thread() raises after Popen, proc must be killed — not orphaned.""" killed = [] @@ -915,6 +918,7 @@ def fake_popen(args, **kwargs): # Simple background must remain as-is assert "sleep 5 &" in shell_cmd + @pytest.mark.platforms("linux") def test_pty_path_uses_rewritten_command(self, registry): """PTY spawn path must also use the rewritten command (issue #68915).""" mock_pty_proc = MagicMock() @@ -1061,6 +1065,7 @@ def test_kill_already_exited(self, registry): assert result["status"] == "already_exited" + @pytest.mark.platforms("linux") def test_kill_detached_session_uses_host_pid(self, registry): s = _make_session(sid="proc_detached", command="sleep 999") s.pid = 424242 @@ -1256,7 +1261,7 @@ class TestTerminateHostPidWindows: target handle only, not the tree. """ - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_invokes_taskkill_with_tree_and_force_flags(self, monkeypatch): """The Windows branch must shell out to ``taskkill /PID N /T /F``. @@ -1286,6 +1291,7 @@ def fake_run(args, **kwargs): class TestTerminateHostPidPosix: """POSIX branch walks the tree via psutil and SIGTERMs children first.""" + @pytest.mark.platforms("linux") def test_posix_walks_tree_and_terminates_children_then_parent(self, monkeypatch): from tools import process_registry as pr import psutil @@ -1323,6 +1329,7 @@ def terminate(self): "Children must be terminated before the parent" ) + @pytest.mark.platforms("linux") def test_posix_oserror_falls_back_to_os_kill(self, monkeypatch): from tools import process_registry as pr import psutil diff --git a/tests/tools/test_process_registry_write_stdin_surrogates.py b/tests/tools/test_process_registry_write_stdin_surrogates.py index 539d980caf68..1fbb547ead93 100644 --- a/tests/tools/test_process_registry_write_stdin_surrogates.py +++ b/tests/tools/test_process_registry_write_stdin_surrogates.py @@ -8,6 +8,7 @@ from tools.process_registry import ProcessRegistry +@pytest.mark.platforms("linux") def test_write_stdin_pty_surrogateescape_roundtrip(tmp_path): registry = ProcessRegistry() out = tmp_path / "out.bin" diff --git a/tests/tools/test_read_shell_line_clamp.py b/tests/tools/test_read_shell_line_clamp.py index 1d10b3b4d3c0..08a3f5de7d6d 100644 --- a/tests/tools/test_read_shell_line_clamp.py +++ b/tests/tools/test_read_shell_line_clamp.py @@ -120,7 +120,7 @@ def test_read_file_raw_not_clamped(tmp_path, ops): max_len = get_max_line_length() p = tmp_path / "long_raw.txt" long_line = "z" * (10 * max_len) - p.write_text(long_line + "\n") + p.write_bytes((long_line + "\n").encode("utf-8")) result = ops.read_file_raw(str(p)) assert result.error is None assert result.content == long_line + "\n" diff --git a/tests/tools/test_read_special_file_guard.py b/tests/tools/test_read_special_file_guard.py index 330f42edd783..9550a8f0d246 100644 --- a/tests/tools/test_read_special_file_guard.py +++ b/tests/tools/test_read_special_file_guard.py @@ -26,11 +26,13 @@ def test_directory(self, tmp_path): def test_missing_path(self, tmp_path): assert _special_file_kind(tmp_path / "nope") is None + @pytest.mark.platforms("linux") def test_fifo(self, tmp_path): fifo = tmp_path / "p.pipe" os.mkfifo(fifo) assert "FIFO" in (_special_file_kind(fifo) or "") + @pytest.mark.platforms("linux") def test_socket(self, tmp_path): sock_path = tmp_path / "s.sock" s = socket.socket(socket.AF_UNIX) @@ -40,6 +42,7 @@ def test_socket(self, tmp_path): finally: s.close() + @pytest.mark.platforms("linux") def test_symlink_to_fifo_followed(self, tmp_path): fifo = tmp_path / "p.pipe" os.mkfifo(fifo) @@ -54,6 +57,7 @@ def test_char_device(self): class TestReadFileToolFifoGuard: + @pytest.mark.platforms("linux") def test_fifo_read_returns_note_instantly(self, tmp_path, monkeypatch): import time diff --git a/tests/tools/test_search_auto_multiline.py b/tests/tools/test_search_auto_multiline.py index 618383243838..f1e46cab3234 100644 --- a/tests/tools/test_search_auto_multiline.py +++ b/tests/tools/test_search_auto_multiline.py @@ -14,7 +14,8 @@ def proj(tmp_path, monkeypatch): d.mkdir() (d / "mod.py").write_text( "def setup():\n init_db()\n return True\n\n" - "def teardown():\n close_db()\n" + "def teardown():\n close_db()\n", + newline="\n", ) return d @@ -40,7 +41,7 @@ def test_plain_pattern_unaffected(self, proj): def test_escaped_backslash_n_stays_literal(self, proj): # \\n = literal backslash+n search, not a newline: no multiline mode. - (proj / "strings.py").write_text('SEP = "a\\\\nb"\n') + (proj / "strings.py").write_text('SEP = "a\\\\nb"\n', newline="\n") r = json.loads(search_tool(r"a\\nb", path=str(proj), task_id="t-ml")) assert "error" not in r assert "multiline" not in (r.get("warning") or "") diff --git a/tests/tools/test_self_repo_guard.py b/tests/tools/test_self_repo_guard.py index 06d96b4beacd..2583bc2237da 100644 --- a/tests/tools/test_self_repo_guard.py +++ b/tests/tools/test_self_repo_guard.py @@ -149,6 +149,7 @@ def test_shell_heredoc_is_executed(self, repo): def test_tilde_dash_c_path(self, repo, monkeypatch, tmp_path): monkeypatch.setenv("HOME", str(repo.parent)) + monkeypatch.setenv("USERPROFILE", str(repo.parent)) hit, _ = _detect("git -C ~/hermes-agent checkout main", tmp_path, repo) assert hit is True @@ -374,8 +375,9 @@ def test_message_warns_against_tmp_for_dep_installs(self, repo): assert "tmpfs" in msg assert "Delete the clone" in msg - def test_scratch_hint_honors_hermes_home(self, repo, monkeypatch): - monkeypatch.setenv("HERMES_HOME", "/custom/hermes-home") + def test_scratch_hint_honors_hermes_home(self, repo, monkeypatch, tmp_path): + home = tmp_path / "custom" / "hermes-home" + monkeypatch.setenv("HERMES_HOME", str(home)) hit, msg = _detect("git rebase origin/main", repo, repo) assert hit is True - assert "/custom/hermes-home/scratch" in msg + assert str(home / "scratch") in msg diff --git a/tests/tools/test_skill_bundle_provenance.py b/tests/tools/test_skill_bundle_provenance.py index 1d5e6e02cf76..3ffe492d2f8b 100644 --- a/tests/tools/test_skill_bundle_provenance.py +++ b/tests/tools/test_skill_bundle_provenance.py @@ -1,6 +1,7 @@ """Multi-file third-party skill bundles and scanner provenance (#60598).""" import json +import os import subprocess import sys import threading @@ -511,7 +512,7 @@ def test_bundled_optional_source_still_includes_support_files(tmp_path, monkeypa bundle = source.fetch("official/category/official-demo") assert bundle is not None - assert set(bundle.files) == {"SKILL.md", "references/all.md"} + assert set(bundle.files) == {"SKILL.md", os.path.join("references", "all.md")} UPSTREAM_STUB_MD = """--- diff --git a/tests/tools/test_skill_manager_tool.py b/tests/tools/test_skill_manager_tool.py index 1d5bcb4629e9..e22f09a43759 100644 --- a/tests/tools/test_skill_manager_tool.py +++ b/tests/tools/test_skill_manager_tool.py @@ -981,6 +981,7 @@ def test_normal_delete_still_works(self, tmp_path): assert result["success"] is True, result assert not (tmp_path / "good-skill").exists() + @pytest.mark.require_symlinks def test_symlinked_skill_dir_refused(self, tmp_path): """A skill dir that is a symlink must not be rmtree'd — rmtree would otherwise follow it and delete the link target's contents.""" diff --git a/tests/tools/test_skill_view_path_check.py b/tests/tools/test_skill_view_path_check.py index c72504ca5474..376ee9d90d1e 100644 --- a/tests/tools/test_skill_view_path_check.py +++ b/tests/tools/test_skill_view_path_check.py @@ -85,10 +85,10 @@ def _old_path_escapes(self, resolved: Path, skill_dir_resolved: Path) -> bool: and resolved != skill_dir_resolved ) - # ``windows_only`` rather than ``skipif(os.sep == "/")``: the Windows CI + # ``platforms("windows")`` rather than ``skipif(os.sep == "/")``: the Windows CI # job greps for the marker to decide which files to import, so a bare # skipif leaves this running on no host at all. - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_old_check_false_positive_on_windows(self, tmp_path): """On Windows, the old check incorrectly blocks valid subpaths.""" skill_dir = tmp_path / "skills" / "axolotl" diff --git a/tests/tools/test_skills_guard.py b/tests/tools/test_skills_guard.py index 8a8b7eb350bf..c94c0782ed39 100644 --- a/tests/tools/test_skills_guard.py +++ b/tests/tools/test_skills_guard.py @@ -252,6 +252,7 @@ def test_structural_limits(self, tmp_path): ids = {fi.pattern_id for fi in _check_structure(tmp_path)} assert {"too_many_files", "oversized_file", "binary_file"} <= ids + @pytest.mark.require_symlinks def test_symlink_escape(self, tmp_path): target = tmp_path / "outside" target.mkdir() diff --git a/tests/tools/test_skills_hub.py b/tests/tools/test_skills_hub.py index 610ea2b34967..c807a1b1cba8 100644 --- a/tests/tools/test_skills_hub.py +++ b/tests/tools/test_skills_hub.py @@ -1,6 +1,7 @@ """Tests for tools/skills_hub.py — source adapters, lock file, taps, dedup logic.""" import json +import os import time from typing import List, Optional from unittest.mock import patch, MagicMock @@ -527,7 +528,7 @@ def test_bundle_content_hash_matches_installed_content_hash(self, tmp_path): skill_dir.mkdir() (skill_dir / "SKILL.md").write_text("same content") (skill_dir / "references").mkdir() - (skill_dir / "references" / "checklist.md").write_text("- [ ] security\n") + (skill_dir / "references" / "checklist.md").write_bytes(b"- [ ] security\n") assert bundle_content_hash(bundle) == content_hash(skill_dir) @@ -860,8 +861,8 @@ def test_fetch_preserves_binary_assets(self, tmp_path): (skill_dir / "assets" / "neutts-cli" / "samples" / "jo.wav").write_bytes( wav_bytes ) - (skill_dir / "assets" / "neutts-cli" / "samples" / "jo.txt").write_text( - "hello\n", encoding="utf-8" + (skill_dir / "assets" / "neutts-cli" / "samples" / "jo.txt").write_bytes( + b"hello\n" ) pycache_dir = skill_dir / "assets" / "neutts-cli" / "src" / "neutts_cli" / "__pycache__" pycache_dir.mkdir(parents=True) @@ -873,8 +874,8 @@ def test_fetch_preserves_binary_assets(self, tmp_path): bundle = src.fetch("official/mlops/models/neutts") assert bundle is not None - assert bundle.files["assets/neutts-cli/samples/jo.wav"] == wav_bytes - assert bundle.files["assets/neutts-cli/samples/jo.txt"] == b"hello\n" + assert bundle.files[os.path.join("assets", "neutts-cli", "samples", "jo.wav")] == wav_bytes + assert bundle.files[os.path.join("assets", "neutts-cli", "samples", "jo.txt")] == b"hello\n" assert "assets/neutts-cli/src/neutts_cli/__pycache__/cli.cpython-312.pyc" not in bundle.files def test_fetch_rejects_sibling_directory_traversal(self, tmp_path): diff --git a/tests/tools/test_skills_sync.py b/tests/tools/test_skills_sync.py index bb55469c4fce..d26a2dbd81d2 100644 --- a/tests/tools/test_skills_sync.py +++ b/tests/tools/test_skills_sync.py @@ -141,7 +141,7 @@ class TestComputeRelativeDest: def test_preserves_category_structure(self): bundled = Path("/repo/skills") dest = _compute_relative_dest(Path("/repo/skills/mlops/axolotl"), bundled) - assert str(dest).endswith("mlops/axolotl") + assert str(dest).endswith(os.path.join("mlops", "axolotl")) # Flat (uncategorized) skills keep their own name. assert _compute_relative_dest(Path("/repo/skills/simple"), bundled).name == "simple" diff --git a/tests/tools/test_skills_sync_client.py b/tests/tools/test_skills_sync_client.py index c967eef187a8..84855456eccd 100644 --- a/tests/tools/test_skills_sync_client.py +++ b/tests/tools/test_skills_sync_client.py @@ -353,6 +353,7 @@ def _raise(**kw): # --------------------------------------------------------------------------- class TestObjectBuilding: + @pytest.mark.platforms("linux") def test_build_tree_blob_and_exec(self, tmp_path): d = tmp_path / "skill" d.mkdir() diff --git a/tests/tools/test_skills_tool.py b/tests/tools/test_skills_tool.py index 99bcb4c8ac1e..56c8bb3f4a02 100644 --- a/tests/tools/test_skills_tool.py +++ b/tests/tools/test_skills_tool.py @@ -907,7 +907,7 @@ def test_nested_local_collides_with_top_level_external(self, tmp_path): assert "matches" in result assert len(result["matches"]) == 2 # Both paths surfaced - assert any("foundations/runtime" in p for p in result["matches"]) + assert any(os.path.join("foundations", "runtime") in p for p in result["matches"]) assert any("external" in p for p in result["matches"]) assert "hint" in result @@ -944,7 +944,7 @@ def test_support_markdown_does_not_collide_with_real_skill(self, tmp_path): result = json.loads(raw) assert result["success"] is True - assert result["path"] == "creative/sketch/SKILL.md" + assert result["path"] == os.path.join("creative", "sketch", "SKILL.md") assert "REAL SKETCH SKILL" in result["content"] diff --git a/tests/tools/test_spill_safety.py b/tests/tools/test_spill_safety.py index 82440df3997d..561d3d226d6f 100644 --- a/tests/tools/test_spill_safety.py +++ b/tests/tools/test_spill_safety.py @@ -43,6 +43,7 @@ def test_private_dir_is_0700_and_tightened(tmp_path): assert stat.S_IMODE(os.lstat(d).st_mode) == 0o700 +@pytest.mark.require_symlinks def test_ensure_spill_dir_refuses_symlinked_leaf(tmp_path): victim = tmp_path / "victim-dir" victim.mkdir() @@ -52,6 +53,7 @@ def test_ensure_spill_dir_refuses_symlinked_leaf(tmp_path): ensure_spill_dir(link) +@pytest.mark.require_symlinks def test_refuses_planted_symlink(tmp_path): """The core attack: symlink at the spill path must fail, not redirect.""" victim = tmp_path / "victim.txt" @@ -63,6 +65,7 @@ def test_refuses_planted_symlink(tmp_path): assert victim.read_text() == "original" +@pytest.mark.require_symlinks def test_refuses_dangling_symlink(tmp_path): target = tmp_path / "spill.txt" target.symlink_to(tmp_path / "does-not-exist.txt") @@ -71,6 +74,7 @@ def test_refuses_dangling_symlink(tmp_path): assert not (tmp_path / "does-not-exist.txt").exists() +@pytest.mark.require_symlinks def test_overwrite_removes_symlink_not_its_target(tmp_path): victim = tmp_path / "victim.txt" victim.write_text("original") diff --git a/tests/tools/test_ssh_bulk_upload.py b/tests/tools/test_ssh_bulk_upload.py index bf41a2fb6e6b..5b10c3523dfa 100644 --- a/tests/tools/test_ssh_bulk_upload.py +++ b/tests/tools/test_ssh_bulk_upload.py @@ -136,6 +136,7 @@ def capture_tar_cmd(cmd, **kwargs): assert len(staging_paths) == 1, "tar command should have been called" + @pytest.mark.require_symlinks def test_bulk_upload_never_stages_remote_home_prefix(self, mock_env, tmp_path): """Regression: do not archive /home/ path components.""" f1 = tmp_path / "nested.txt" diff --git a/tests/tools/test_stage2_hook_api_server_keygen.py b/tests/tools/test_stage2_hook_api_server_keygen.py index fdcf4cb22e36..34e53f497583 100644 --- a/tests/tools/test_stage2_hook_api_server_keygen.py +++ b/tests/tools/test_stage2_hook_api_server_keygen.py @@ -74,6 +74,7 @@ def _run_keygen( ) +@pytest.mark.platforms("linux") def test_keygen_creates_env_when_missing(stage2_text: str, tmp_path: Path) -> None: """No .env at all (failed/absent first-boot seed) must still yield a key.""" home = tmp_path / "home" @@ -116,6 +117,7 @@ def test_keygen_never_overwrites_operator_key( assert content == "API_SERVER_KEY=operator-provided-key-123\n" +@pytest.mark.require_symlinks def test_keygen_refuses_symlinked_env(stage2_text: str, tmp_path: Path) -> None: home = tmp_path / "home" home.mkdir() @@ -198,6 +200,7 @@ def _sed_is_gnu() -> bool: return probe.returncode == 0 and "GNU sed" in probe.stdout +@pytest.mark.platforms("linux") def test_keygen_readonly_env_degrades_to_warning_not_boot_abort( stage2_text: str, tmp_path: Path ) -> None: diff --git a/tests/tools/test_terminal_cwd_echo.py b/tests/tools/test_terminal_cwd_echo.py index 98acccb8c5f8..1f9e145c16db 100644 --- a/tests/tools/test_terminal_cwd_echo.py +++ b/tests/tools/test_terminal_cwd_echo.py @@ -22,6 +22,13 @@ def isolated_home(tmp_path, monkeypatch): class TestCwdEcho: + # `cd` with a *native* Windows path (``str(Path)`` → ``C:\\Users\\...``) + # doesn't survive the terminal's ``eval ''`` wrapper on Windows — + # bash re-parses the command and eats the backslash separators, so the + # ``cd`` no-ops and the cwd never changes. Native-path handling in the + # terminal is a separate Windows workstream; the echo feature itself is + # covered by the Linux lane here. + @pytest.mark.platforms("linux") def test_cd_reports_new_cwd(self, isolated_home, tmp_path): target = tmp_path / "projdir" target.mkdir() @@ -35,6 +42,7 @@ def test_non_cd_command_has_no_cwd_field(self, isolated_home): assert r["exit_code"] == 0 assert "cwd" not in r + @pytest.mark.platforms("linux") def test_cwd_persists_and_stops_reporting_when_stable(self, isolated_home, tmp_path): target = tmp_path / "stable" target.mkdir() @@ -45,6 +53,7 @@ def test_cwd_persists_and_stops_reporting_when_stable(self, isolated_home, tmp_p assert "cwd" not in r2 assert os.path.realpath(r2["output"].strip()) == os.path.realpath(str(target)) + @pytest.mark.platforms("linux") def test_cd_within_chain_reports_final_dir(self, isolated_home, tmp_path): a = tmp_path / "a" b = tmp_path / "b" diff --git a/tests/tools/test_terminal_output_transform_hook.py b/tests/tools/test_terminal_output_transform_hook.py index 9d9c75651603..7c49a1a1147e 100644 --- a/tests/tools/test_terminal_output_transform_hook.py +++ b/tests/tools/test_terminal_output_transform_hook.py @@ -3,6 +3,8 @@ from pathlib import Path from unittest.mock import MagicMock +import pytest + import hermes_cli.plugins as plugins_mod import tools.terminal_tool as terminal_tool_module from tools.environments.local import LocalEnvironment @@ -91,6 +93,7 @@ def test_terminal_output_transform_still_runs_strip_and_redact(monkeypatch, tmp_ assert "abc123def456" not in result["output"] # secret body is gone +@pytest.mark.platforms("linux") def test_large_process_output_is_bounded_before_sudo_and_plugin_hooks( monkeypatch, tmp_path ): diff --git a/tests/tools/test_terminal_truncation_spill.py b/tests/tools/test_terminal_truncation_spill.py index f14d0da5333e..8f7adbbd1527 100644 --- a/tests/tools/test_terminal_truncation_spill.py +++ b/tests/tools/test_terminal_truncation_spill.py @@ -20,6 +20,7 @@ def small_cap(tmp_path, monkeypatch): class TestTruncationSpill: + @pytest.mark.platforms("linux") def test_truncated_output_has_metadata_and_spill(self, small_cap): r = json.loads(terminal_tool( "python3 -c \"print('marker_head'); [print(f'row_{i}', 'x'*80) for i in range(200)]; print('marker_tail')\"", @@ -41,6 +42,7 @@ def test_small_output_has_no_metadata(self, small_cap): assert "full_output_path" not in r assert "output_total_chars" not in r + @pytest.mark.platforms("linux") def test_spill_is_redacted(self, small_cap): r = json.loads(terminal_tool( "python3 -c \"print('sk-proj-' + 'a1B2c3D4e5F6g7H8i9J0' * 3); [print('pad', 'y'*90) for i in range(200)]\"", @@ -59,6 +61,7 @@ def test_old_spills_cleaned(self, small_cap, tmp_path): "python3 -c \"[print('z'*90) for i in range(200)]\"", task_id="t-spill-4")) assert not stale.exists() + @pytest.mark.platforms("linux") def test_failed_command_still_gets_spill(self, small_cap): r = json.loads(terminal_tool( "python3 -c \"[print('e'*90) for i in range(200)]; import sys; sys.exit(3)\"", diff --git a/tests/tools/test_tirith_security.py b/tests/tools/test_tirith_security.py index ed5f8be92d26..2eb3c8fffe0c 100644 --- a/tests/tools/test_tirith_security.py +++ b/tests/tools/test_tirith_security.py @@ -54,6 +54,7 @@ def _json_stdout(findings=None, summary=""): # Exit code → action mapping # --------------------------------------------------------------------------- +@pytest.mark.platforms("linux") class TestExitCodeMapping: @patch("tools.tirith_security.subprocess.run") @patch("tools.tirith_security._load_security_config") @@ -94,6 +95,7 @@ def test_exit_2_warn_with_findings(self, mock_cfg, mock_run): # JSON parse failure (exit code still wins) # --------------------------------------------------------------------------- +@pytest.mark.platforms("linux") class TestJsonParseFailure: @patch("tools.tirith_security.subprocess.run") @patch("tools.tirith_security._load_security_config") @@ -119,6 +121,7 @@ def test_exit_0_invalid_json_allows(self, mock_cfg, mock_run): # Operational failures + fail_open # --------------------------------------------------------------------------- +@pytest.mark.platforms("linux") class TestOSErrorFailOpen: @patch("tools.tirith_security.subprocess.run") @patch("tools.tirith_security._load_security_config") @@ -141,6 +144,7 @@ def test_os_error_fail_closed(self, mock_cfg, mock_run): assert "fail-closed" in result["summary"] +@pytest.mark.platforms("linux") class TestTimeoutFailOpen: @patch("tools.tirith_security.subprocess.run") @patch("tools.tirith_security._load_security_config") @@ -153,6 +157,7 @@ def test_timeout_fail_closed(self, mock_cfg, mock_run): assert "fail-closed" in result["summary"] +@pytest.mark.platforms("linux") class TestUnknownExitCode: @patch("tools.tirith_security.subprocess.run") @patch("tools.tirith_security._load_security_config") @@ -182,6 +187,7 @@ def test_disabled_returns_allow(self, mock_cfg): # Findings cap + summary cap # --------------------------------------------------------------------------- +@pytest.mark.platforms("linux") class TestCaps: @patch("tools.tirith_security.subprocess.run") @patch("tools.tirith_security._load_security_config") @@ -199,6 +205,7 @@ def test_findings_and_summary_capped(self, mock_cfg, mock_run): # Programming errors propagate # --------------------------------------------------------------------------- +@pytest.mark.platforms("linux") class TestProgrammingErrors: @patch("tools.tirith_security.subprocess.run") @patch("tools.tirith_security._load_security_config") @@ -222,6 +229,7 @@ def test_disabled_returns_none(self, mock_cfg): _tirith_mod._resolved_path = None assert ensure_installed() is None + @pytest.mark.platforms("linux") @patch("tools.tirith_security.shutil.which", return_value="/usr/local/bin/tirith") @patch("tools.tirith_security._load_security_config") def test_found_on_path_returns_immediately(self, mock_cfg, mock_which): @@ -299,6 +307,7 @@ def test_explicit_path_still_honored_on_unsupported_platform(self, mock_cfg): # --------------------------------------------------------------------------- class TestFailedDownloadCaching: + @pytest.mark.platforms("linux") @patch("tools.tirith_security._mark_install_failed") @patch("tools.tirith_security._is_install_failed_on_disk", return_value=False) @patch("tools.tirith_security._install_tirith", return_value=(None, "download_failed")) @@ -341,6 +350,7 @@ def test_tilde_explicit_path_missing_no_download(self, mock_which, mock_install) _tirith_mod._resolved_path = None + @pytest.mark.platforms("linux") @patch("tools.tirith_security._mark_install_failed") @patch("tools.tirith_security._is_install_failed_on_disk", return_value=False) @patch("tools.tirith_security._install_tirith", return_value=("/auto/tirith", "")) @@ -490,6 +500,7 @@ def test_install_rejects_non_regular_tirith_member(self, mock_target, mock_which # --------------------------------------------------------------------------- class TestBackgroundInstall: + @pytest.mark.platforms("linux") def test_ensure_installed_non_blocking(self): """ensure_installed must return immediately when download needed.""" _tirith_mod._resolved_path = None @@ -586,6 +597,7 @@ def test_hermes_bin_dir_respects_hermes_home(self): # Warn-once dedupe (issue: tirith spawn failed spamming on Windows) # --------------------------------------------------------------------------- +@pytest.mark.platforms("linux") class TestSpawnWarningDedup: """When tirith isn't installed yet (background install in flight, or install marked failed), every terminal command spammed an identical @@ -636,6 +648,7 @@ def test_repeated_spawn_failure_logs_once(self, mock_cfg, mock_run, caplog): "tirith_timeout": 5, "tirith_fail_open": True} +@pytest.mark.platforms("linux") class TestAppTldSuppression: """warn verdicts whose only finding is lookalike_tld/.app are downgraded to allow.""" @@ -694,6 +707,7 @@ def test_app_tld_detection(self, finding, expected): # mkdtemp OSError → no_space (disk-full leak prevention) # --------------------------------------------------------------------------- +@pytest.mark.platforms("linux") class TestMkdtempOSErrorNoSpace: """When tempfile.mkdtemp raises OSError (e.g. disk full), _install_tirith must return (None, "no_space") instead of propagating the exception. diff --git a/tests/tools/test_tts_macos_output.py b/tests/tools/test_tts_macos_output.py index e890d7ad9544..83c0442c878d 100644 --- a/tests/tools/test_tts_macos_output.py +++ b/tests/tools/test_tts_macos_output.py @@ -74,12 +74,12 @@ def _spy_import_sd(): return sd_called["hit"] -@pytest.mark.macos_only +@pytest.mark.platforms("macos") def test_streaming_tts_skips_sounddevice_on_macos(monkeypatch): assert _run_stream(monkeypatch) is False -@pytest.mark.linux_only +@pytest.mark.platforms("linux") def test_streaming_tts_uses_sounddevice_off_macos(monkeypatch): # Off macOS the OutputStream setup runs; _import_sounddevice raising here # is caught by the function's own guard, so the call itself is what we assert. diff --git a/tests/tools/test_video_analyze.py b/tests/tools/test_video_analyze.py index 003511163022..280769fe05c1 100644 --- a/tests/tools/test_video_analyze.py +++ b/tests/tools/test_video_analyze.py @@ -6,6 +6,7 @@ from types import SimpleNamespace from unittest.mock import AsyncMock, MagicMock, patch +import pytest from tools.vision_tools import ( _detect_video_mime_type, @@ -145,6 +146,7 @@ def test_local_file_success(self, tmp_path, monkeypatch): assert data["success"] is True assert "demo" in data["analysis"].lower() + @pytest.mark.require_symlinks def test_local_file_read_guard_blocks_env_via_video_extension(self, tmp_path): """A .env file symlinked with a video extension must still be blocked. diff --git a/tests/tools/test_voice_mode.py b/tests/tools/test_voice_mode.py index be9f24f6de3c..b2ccac537973 100644 --- a/tests/tools/test_voice_mode.py +++ b/tests/tools/test_voice_mode.py @@ -120,6 +120,7 @@ def fake_clock(monkeypatch): # detect_audio_environment — WSL / SSH / Docker detection # ============================================================================ +@pytest.mark.platforms("linux") class TestPulseSocketReachable: def test_stale_socket_file_not_reachable(self, monkeypatch, tmp_path): """A socket file with no listener should not count as reachable.""" @@ -563,7 +564,7 @@ def test_real_speech_not_filtered(self): # ============================================================================ class TestPlayAudioFile: - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_play_wav_via_sounddevice(self, monkeypatch, sample_wav): np = pytest.importorskip("numpy") # Linux-gated rather than faking a non-macOS platform: on macOS WAV @@ -599,7 +600,7 @@ class TestMacOSAudioOutputPolicy: a TCC media-library prompt, which no faked platform on Linux reproduces — and `afplay` only resolves on a real macOS host.""" - @pytest.mark.macos_only + @pytest.mark.platforms("macos") def test_play_audio_file_skips_sounddevice_on_macos(self, monkeypatch, sample_wav): """On macOS, WAV playback must not import sounddevice; it routes to afplay.""" @@ -635,7 +636,7 @@ def _fake_popen(cmd, **kwargs): assert popen_cmds, "expected a system player to be invoked" assert popen_cmds[0][0] == "afplay" - @pytest.mark.macos_only + @pytest.mark.platforms("macos") def test_play_beep_routes_through_afplay_on_macos(self, monkeypatch): """On macOS, beeps synthesize with numpy but play via the tempfile/afplay path.""" pytest.importorskip("numpy") @@ -1391,6 +1392,7 @@ def _side_effect(cmd, **kwargs): return next(it) return _side_effect + @pytest.mark.platforms("linux") def test_powershell_pipeline_preserves_real_exit_status(self, sample_wav): """Regression (review of #63768): the shell pipeline must preserve the (ffmpeg && powershell) exit status past the unconditional @@ -1440,6 +1442,7 @@ def _capture_popen(cmd, **kw): "Shell pipeline must preserve the real exit status past cleanup: " + sh_script ) + @pytest.mark.platforms("linux") def test_wsl2_unique_temp_filename(self, monkeypatch, tmp_path, sample_wav): """Two concurrent calls must use different temp WAV filenames.""" from unittest.mock import patch, MagicMock diff --git a/tests/tools/test_wake_word.py b/tests/tools/test_wake_word.py index fc05fa11de45..a0aa7b31522c 100644 --- a/tests/tools/test_wake_word.py +++ b/tests/tools/test_wake_word.py @@ -251,7 +251,7 @@ def test_default_framework_tracks_the_macos_arm64_probe(): assert ww.default_inference_framework() == expected -@pytest.mark.macos_only +@pytest.mark.platforms("macos") def test_macos_arm64_prefers_tflite_on_this_host(): """On a real ARM64 Mac the default must be tflite (upstream #336). @@ -523,7 +523,7 @@ def _stream(**kwargs): det.stop() -@pytest.mark.windows_only +@pytest.mark.platforms("windows") def test_windows_silent_hint_names_selected_device(): hint = ww.silent_audio_hint( { @@ -537,7 +537,7 @@ def test_windows_silent_hint_names_selected_device(): assert "macOS" not in hint -@pytest.mark.macos_only +@pytest.mark.platforms("macos") def test_macos_silent_hint_points_at_privacy_settings(): """On macOS a silent stream is almost always the TCC mic permission, so the hint names System Settings rather than the device.""" @@ -548,7 +548,7 @@ def test_macos_silent_hint_points_at_privacy_settings(): assert "Microphone" in hint -@pytest.mark.linux_only +@pytest.mark.platforms("linux") def test_linux_silent_hint_names_selected_device(): hint = ww.silent_audio_hint( {"selector": 2, "name": "HD Audio Capture", "hostapi": "ALSA"} diff --git a/tests/tools/test_windows_agent_loop_papercuts.py b/tests/tools/test_windows_agent_loop_papercuts.py index 0b2c04c59042..ada9c7801c01 100644 --- a/tests/tools/test_windows_agent_loop_papercuts.py +++ b/tests/tools/test_windows_agent_loop_papercuts.py @@ -18,22 +18,22 @@ class TestSplitCommandLine: """#83934 / #78293 — backslashes in Windows paths must survive splitting.""" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_path_backslashes_preserved(self): argv = split_command_line(r"sessions export C:\Users\me\Desktop\out.jsonl") assert argv == ["sessions", "export", r"C:\Users\me\Desktop\out.jsonl"] - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_quoted_path_with_spaces(self): argv = split_command_line(r'run "C:\Program Files\App\tool.exe" --flag') assert argv == ["run", r"C:\Program Files\App\tool.exe", "--flag"] - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_bare_hook_command_path(self): argv = split_command_line(r"C:\Users\u\.local\bin\dcg.exe --hook pre") assert argv[0] == r"C:\Users\u\.local\bin\dcg.exe" - @pytest.mark.linux_only + @pytest.mark.platforms("linux") def test_posix_behavior_unchanged(self): assert split_command_line("echo 'a b' c") == ["echo", "a b", "c"] @@ -45,14 +45,14 @@ def test_unbalanced_quote_raises(self): class TestShellHooksWindowsPaths: """#78293 — hook script paths with backslashes resolve correctly.""" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_command_script_path_keeps_backslashes(self): from agent.shell_hooks import _command_script_path path = _command_script_path(r"C:\hooks\guard.py --strict") assert path == r"C:\hooks\guard.py" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_script_is_executable_finds_real_file(self, tmp_path): from agent.shell_hooks import script_is_executable @@ -68,7 +68,7 @@ def test_script_is_executable_finds_real_file(self, tmp_path): class TestWindowsMarketingVersion: """#51755 — Windows 11 must not be reported as Windows 10.""" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_matches_build_number(self): from agent.prompt_builder import _windows_marketing_version diff --git a/tests/tools/test_windows_native_support.py b/tests/tools/test_windows_native_support.py index 1a04a28f55aa..edd179eb0b3a 100644 --- a/tests/tools/test_windows_native_support.py +++ b/tests/tools/test_windows_native_support.py @@ -82,11 +82,11 @@ def test_reconfigure_stream_handles_missing_method(self, monkeypatch): # --------------------------------------------------------------------------- -@pytest.mark.windows_only +@pytest.mark.platforms("windows") class TestTerminatePidRoutingOnWindows: """``gateway.status.terminate_pid`` must use taskkill /T /F on Windows. - ``windows_only``: this used to patch the module-level ``_IS_WINDOWS`` + ``platforms("windows")``: this used to patch the module-level ``_IS_WINDOWS`` flag on Linux, which selected the taskkill branch on a host where ``taskkill`` does not exist and ``gateway/status`` cannot even import its ``msvcrt`` branch. On the Windows runner the flag is genuinely True, so @@ -255,7 +255,7 @@ def test_alive_pid_returns_true(self, monkeypatch): assert ProcessRegistry._is_host_pid_alive(os.getpid()) is True -@pytest.mark.linux_only +@pytest.mark.platforms("linux") class TestPidExistsOSErrorWidening: """gateway.status._pid_exists itself must widen Windows errors correctly. @@ -264,7 +264,7 @@ class TestPidExistsOSErrorWidening: gone PID instead of ``ProcessLookupError``. The function must catch the wider ``OSError`` to match POSIX semantics. - ``linux_only``: the subject is the POSIX fallback branch and its + ``platforms("linux")``: the subject is the POSIX fallback branch and its ``os.kill`` error handling, exercised with the errno values Windows produces. Gating to Linux is what makes ``_IS_WINDOWS`` genuinely False here instead of forced false by a patch. @@ -426,11 +426,11 @@ def test_resolve_node_command_returns_absolute_on_posix(self): # name (fallback) — both are acceptable behaviours. - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_detach_flags_exclude_detached_process(self): """DETACHED_PROCESS must stay OUT of every detach bundle. - ``windows_only`` (with ``IS_WINDOWS`` no longer patched): the helpers + ``platforms("windows")`` (with ``IS_WINDOWS`` no longer patched): the helpers return 0 off Windows, so on Linux the old flag patch was the only thing making the bit assertions reachable at all. @@ -454,7 +454,7 @@ def test_windows_detach_flags_exclude_detached_process(self): "DETACHED_PROCESS must not be in the no-breakaway fallback either." ) - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_detach_flags_includes_breakaway_from_job(self): """CREATE_BREAKAWAY_FROM_JOB is load-bearing for the GUI-driven update path. @@ -476,7 +476,7 @@ def test_windows_detach_flags_includes_breakaway_from_job(self): "can respawn the gateway after Electron exits." ) - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_detach_flags_without_breakaway_drops_only_that_bit(self): """Fallback retry payload for restrictive job objects. @@ -724,9 +724,9 @@ def test_source_has_windows_branch_using_hermes_home(self): class TestLocalEnvironmentPathInjectionGated: """Sane PATH completion must stay POSIX-only.""" - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_path_is_left_unchanged(self): - """``windows_only``: the assertion is that a real Windows ``PATH`` + """``platforms("windows")``: the assertion is that a real Windows ``PATH`` (``;``-separated, drive-lettered) comes back untouched. On Linux the old ``_IS_WINDOWS`` patch made the function return early without ever meeting a genuine Windows PATH.""" @@ -755,11 +755,11 @@ def test_posix_noop(self): assert _normalize_git_bash_path(None) is None - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_translation(self): """On native Windows, /c/Users/... becomes C:\\Users\\... - ``windows_only``: the function's whole job is producing native + ``platforms("windows")``: the function's whole job is producing native Windows paths, which is only meaningful where ``os.sep`` is ``\\``. """ import cli as cli_mod @@ -931,6 +931,7 @@ class TestWindowlessGatewayRestartSpec: hidden-console respawn spec (normalized interpreter + stable cwd + env overlay).""" + @pytest.mark.platforms("linux") def test_noop_on_non_windows(self): import hermes_cli.gateway_windows as gw @@ -948,13 +949,13 @@ def test_empty_argv_is_safe(self): assert cwd == "" assert env == {} - @pytest.mark.windows_only + @pytest.mark.platforms("windows") def test_windows_keeps_console_python_and_preserves_tail(self): """On Windows the console interpreter is kept (hidden-console launch, NOT a pythonw swap — #54220/#56747) while every subsequent argument is preserved verbatim. - ``windows_only``: faking this on Linux needed two more fakes to hold + ``platforms("windows")``: faking this on Linux needed two more fakes to hold it up — a pre-import so the lazy ``hermes_cli.gateway`` import didn't re-run ``gateway/status``'s ``import msvcrt`` branch, and a mock of ``get_hermes_home`` because the real one's ``Path.resolve()`` consults @@ -1000,7 +1001,7 @@ def test_windows_keeps_console_python_and_preserves_tail(self): # --------------------------------------------------------------------------- -@pytest.mark.windows_only +@pytest.mark.platforms("windows") class TestGatewayRunRestartWatcherOuterPopenFallback: """The Windows ``/restart`` watcher in ``gateway.run`` spawns an outer detached ``python -c `` process with @@ -1014,7 +1015,7 @@ class TestGatewayRunRestartWatcherOuterPopenFallback: Behavioral: drives the real coroutine with a mocked ``subprocess.Popen`` rather than asserting on source text. - ``windows_only``: this used to run on Linux behind a ``sys.platform`` + ``platforms("windows")``: this used to run on Linux behind a ``sys.platform`` patch, and the breakaway-bit assertions had to be skipped there anyway (``_subprocess_compat`` caches ``IS_WINDOWS`` at import, so the flags were all 0) — i.e. the most important assertions in the class never diff --git a/tests/tools/test_zombie_process_cleanup.py b/tests/tools/test_zombie_process_cleanup.py index ba3c20eebcea..87ed972671f2 100644 --- a/tests/tools/test_zombie_process_cleanup.py +++ b/tests/tools/test_zombie_process_cleanup.py @@ -11,6 +11,8 @@ import sys import threading +import pytest + def _spawn_sleep(seconds: float = 60) -> subprocess.Popen: @@ -32,6 +34,7 @@ def _pid_alive(pid: int) -> bool: class TestZombieReproduction: """Demonstrate that subprocesses survive when cleanup is not called.""" + @pytest.mark.platforms("linux") def test_orphaned_processes_survive_without_cleanup(self): """REPRODUCTION: processes spawned directly survive if no one kills them — this models the gap that causes zombie accumulation when diff --git a/tests/tui_gateway/test_bot_relay_methods.py b/tests/tui_gateway/test_bot_relay_methods.py index fb6d357744a4..ceca84663728 100644 --- a/tests/tui_gateway/test_bot_relay_methods.py +++ b/tests/tui_gateway/test_bot_relay_methods.py @@ -116,6 +116,7 @@ def test_reply_roundtrip_and_id_validation(home): assert "error" in err +@pytest.mark.platforms("linux") def test_deliver_write_failure_still_removes_tempfile(home, monkeypatch, tmp_path): """A failed payload write must not leak the relay DM tempfile.""" import glob diff --git a/tests/tui_gateway/test_compute_host.py b/tests/tui_gateway/test_compute_host.py index fa0019722f13..6b2249af6c8c 100644 --- a/tests/tui_gateway/test_compute_host.py +++ b/tests/tui_gateway/test_compute_host.py @@ -30,6 +30,7 @@ def _read_json_line(out: queue.Queue[dict], timeout: float = 2.0) -> dict: raise AssertionError("timed out waiting for compute host JSON") from exc +@pytest.mark.platforms("linux") def test_compute_host_line_json_seed_turn_interrupt(): repo = Path(__file__).resolve().parents[2] env = dict(os.environ) diff --git a/tests/tui_gateway/test_entry_import_off_main_thread.py b/tests/tui_gateway/test_entry_import_off_main_thread.py index d3a80c4dbdf5..136b7c3deb16 100644 --- a/tests/tui_gateway/test_entry_import_off_main_thread.py +++ b/tests/tui_gateway/test_entry_import_off_main_thread.py @@ -26,6 +26,8 @@ import subprocess import sys +import pytest + REPO_ROOT = "." @@ -65,6 +67,7 @@ def _spawn_worker_import_entry(): return proc.returncode, proc.stdout, proc.stderr +@pytest.mark.platforms("linux") def test_entry_imports_cleanly_from_worker_thread(): """First import of tui_gateway.entry from a worker thread must succeed.""" rc, out, err = _spawn_worker_import_entry() diff --git a/tests/verify/test_environment_and_runner.py b/tests/verify/test_environment_and_runner.py index 4c1d8885623c..e8d486aa2b46 100644 --- a/tests/verify/test_environment_and_runner.py +++ b/tests/verify/test_environment_and_runner.py @@ -5,6 +5,8 @@ import threading import time +import pytest + from agent.verify.environment import ( load_manifest, load_or_detect, @@ -128,6 +130,7 @@ def _free_port() -> int: class TestReadiness: + @pytest.mark.platforms("linux") def test_readiness_against_live_server(self, tmp_path): port = _free_port() recipe = Recipe( diff --git a/tests/website/test_generate_llms_txt.py b/tests/website/test_generate_llms_txt.py index 738ba55a0925..cf84acce415c 100644 --- a/tests/website/test_generate_llms_txt.py +++ b/tests/website/test_generate_llms_txt.py @@ -53,7 +53,7 @@ def _pages_on_disk(gen) -> set[str]: pages = set() for path in (*gen.DOCS.rglob("*.md"), *gen.DOCS.rglob("*.mdx")): rel = path.relative_to(gen.DOCS).with_suffix("") - slug = str(rel.parent) if rel.name == "index" else str(rel) + slug = rel.parent.as_posix() if rel.name == "index" else rel.as_posix() # The docs landing page is the index's subject; per-skill pages are # summarized by the two catalog reference pages. if slug == "." or slug.startswith(("user-guide/skills/bundled", "user-guide/skills/optional")): diff --git a/tools/approval.py b/tools/approval.py index 15867a188cf9..d0abe1ea592e 100644 --- a/tools/approval.py +++ b/tools/approval.py @@ -2168,6 +2168,21 @@ def _replace_simple_shell_expansions(word: str) -> str: return _PARAM_DEFAULT_RE.sub(lambda match: match.group("default"), word) +def _is_windows_drive_relative_backslash(word: str, i: int) -> bool: + """True when ``word[i] == '\\'`` is a path separator, not a shell escape. + + On Windows drive-relative paths (``C:\\Users\\x``, ``C:Users`` forms), a + backslash between word characters is a separator the shell leaves alone. + The codebase's Windows shell is still bash, so ``r\\m`` spells ``rm`` + exactly as on POSIX — the escape must keep being stripped everywhere + EXCEPT inside a drive-anchored path, where stripping would corrupt the + path (``C:\\Users`` → ``C:Users``) and open false-positive approvals. + """ + if i <= 0 or not word[i - 1].isalnum(): + return False + return bool(re.match(r"^[A-Za-z]:", word)) + + def _strip_shell_word_syntax(word: str) -> str: chars: list[str] = [] quote: str | None = None @@ -2175,7 +2190,12 @@ def _strip_shell_word_syntax(word: str) -> str: while i < len(word): ch = word[i] if quote: - if ch == "\\" and quote == '"' and i + 1 < len(word): + if ( + ch == "\\" + and quote == '"' + and i + 1 < len(word) + and not _is_windows_drive_relative_backslash(word, i) + ): chars.append(word[i + 1]) i += 2 continue @@ -2190,7 +2210,11 @@ def _strip_shell_word_syntax(word: str) -> str: quote = ch i += 1 continue - if ch == "\\" and i + 1 < len(word): + if ( + ch == "\\" + and i + 1 < len(word) + and not _is_windows_drive_relative_backslash(word, i) + ): chars.append(word[i + 1]) i += 2 continue @@ -3517,9 +3541,11 @@ def _get_approval_timeout() -> int: safe_cap = int(MAX_SAFE_TIMEOUT_S) except Exception: # Fail CLOSED: returning the raw value here would re-open the exact - # time_t overflow this clamp exists to prevent. ~1 year, matching - # agent.deadline.MAX_SAFE_TIMEOUT_S. - safe_cap = 365 * 24 * 3600 + # time_t overflow this clamp exists to prevent. Mirror the per-host + # ceiling of agent.deadline.MAX_SAFE_TIMEOUT_S — 1 year on POSIX, + # but the DWORD-millisecond ceiling on Windows (WaitForSingleObject + # caps at ~49.7 days; a year overflows it). + safe_cap = 4_294_907 if sys.platform == "win32" else 365 * 24 * 3600 if raw > safe_cap: logger.warning( "approvals.timeout=%s exceeds the platform-safe maximum; " diff --git a/tools/checkpoint_manager.py b/tools/checkpoint_manager.py index 1061874b4101..232a6e3effbd 100644 --- a/tools/checkpoint_manager.py +++ b/tools/checkpoint_manager.py @@ -2202,6 +2202,12 @@ def store_status(checkpoint_base: Optional[Path] = None) -> Dict: return out +def _rmtree_force(path: Path) -> None: + from hermes_cli.fs_utils import rmtree_force + + rmtree_force(path) + + def clear_all(checkpoint_base: Optional[Path] = None) -> Dict[str, int]: """Nuke the entire checkpoint base (store + legacy). Irreversible. @@ -2213,7 +2219,7 @@ def clear_all(checkpoint_base: Optional[Path] = None) -> Dict[str, int]: return out size = _dir_size_bytes(base) try: - shutil.rmtree(base) + _rmtree_force(base) out["bytes_freed"] = size out["deleted"] = True except OSError as exc: @@ -2235,7 +2241,7 @@ def clear_legacy(checkpoint_base: Optional[Path] = None) -> Dict[str, int]: continue try: size = _dir_size_bytes(child) - shutil.rmtree(child) + _rmtree_force(child) out["bytes_freed"] += size out["deleted"] += 1 except OSError as exc: diff --git a/tools/computer_use/cua_backend.py b/tools/computer_use/cua_backend.py index 2d3937787d21..3d8b93fa6ced 100644 --- a/tools/computer_use/cua_backend.py +++ b/tools/computer_use/cua_backend.py @@ -51,7 +51,7 @@ import threading import time import uuid -from pathlib import PureWindowsPath +from pathlib import PurePosixPath, PureWindowsPath from typing import Any, Dict, List, Optional, Tuple from hermes_cli._subprocess_compat import windows_hide_flags @@ -538,7 +538,10 @@ def _wsl_windows_path_to_posix(path: str) -> str: drive = (win.drive or "").rstrip(":").lower() if not drive: return path - return os.path.join("/mnt", drive, *(str(part) for part in win.parts[1:])) + # PurePosixPath (not os.path.join) so the result always uses forward + # slashes — os.path.join would emit backslashes on a native Windows host, + # producing the invalid ``/mnt\c\...`` the WSL side cannot read. + return str(PurePosixPath("/mnt", drive, *(str(part) for part in win.parts[1:]))) def _resolve_cua_driver_app_path(driver_cmd: str) -> Optional[str]: diff --git a/tools/environments/daytona.py b/tools/environments/daytona.py index 8aba71280410..c3ebda3eaa0c 100644 --- a/tools/environments/daytona.py +++ b/tools/environments/daytona.py @@ -10,7 +10,7 @@ import os import shlex import threading -from pathlib import Path +from pathlib import Path, PurePosixPath from tools.environments.base import ( BaseEnvironment, @@ -153,7 +153,9 @@ def __init__( def _daytona_upload(self, host_path: str, remote_path: str) -> None: """Upload a single file via Daytona SDK.""" - parent = str(Path(remote_path).parent) + # remote_path is a POSIX path on the sandbox; never run it through the + # host's Path (Windows would mangle the separators into backslashes). + parent = str(PurePosixPath(remote_path).parent) self._sandbox.process.exec(quoted_mkdir_command([parent])) self._sandbox.fs.upload_file(host_path, remote_path) diff --git a/tools/file_operations.py b/tools/file_operations.py index fbfd06cab8ef..cbef8fc1e260 100644 --- a/tools/file_operations.py +++ b/tools/file_operations.py @@ -1175,7 +1175,7 @@ def _expand_path(self, path: str) -> str: return path - def _escape_shell_arg(self, arg: str) -> str: + def _escape_shell_arg(self, arg: str, translate_path: bool = True) -> str: """Escape a string for safe use in shell commands. On Windows native drive paths (``C:\\Users\\x`` / ``C:/Users/x``) @@ -1185,10 +1185,20 @@ def _escape_shell_arg(self, arg: str) -> str: ``Directory \\drivers\\etc does not exist`` failure class. Reuses the env-layer translator so shell file ops and the terminal ``cd`` agree on the path form. No-op off Windows and for plain POSIX paths. + + ``translate_path=False`` skips that translation for non-path values + (regex patterns) whose backslashes are meaningful. On Windows those + backslashes are doubled first: the ``-c`` argument passes through + ``subprocess`` list-arg quoting (which wraps it in double quotes) and + ``bash`` then collapses ``\\\\`` → ``\\`` inside those double quotes, + so one level must survive the round trip. """ - from tools.environments.local import _bash_safe_path + from tools.environments.local import _IS_WINDOWS, _bash_safe_path - arg = _bash_safe_path(arg) + if translate_path: + arg = _bash_safe_path(arg) + elif _IS_WINDOWS: + arg = arg.replace("\\", "\\\\") # Use single quotes and escape any single quotes in the string return "'" + arg.replace("'", "'\"'\"'") + "'" @@ -1410,6 +1420,38 @@ def _not_regular_error(path: str) -> ReadResult: _UTF16_MAX_BYTES = 10 * 1024 * 1024 _UTF16_SAMPLE_BYTES = 512 + def _python_interpreter_cmd(self) -> str: + """Return a shell-safe Python interpreter for the terminal backend. + + On the local backend ``sys.executable`` is always a working + interpreter — and on Windows it dodges the Microsoft Store + ``python``/``python3`` alias stub (exit 49, "Python was not + found") that otherwise breaks inline ``-c`` snippets. Remote + backends (docker/ssh/...) don't have the agent's interpreter, so + they fall back to ``python3`` on their own PATH (the ``python`` + fallback is handled at the call site). + """ + if self._lsp_local_only(): + return self._escape_shell_arg(sys.executable) + return "python3" + + def _exec_python_snippet(self, snippet: str, py: str = None) -> ExecuteResult: + """Run a Python ``snippet`` in the terminal backend's interpreter. + + Base64-encodes the snippet so it survives every shell/quoting layer + as pure ASCII: Windows ``subprocess`` list-arg quoting and ``bash`` + double-quote processing both eat backslashes, which otherwise + corrupts Windows paths (``C:\\Users\\x``) and byte literals + (``b'\\xfe\\xff'``) embedded in a ``-c`` program. ``exec`` decodes + and runs it unchanged. + """ + encoded = base64.b64encode(snippet.encode("utf-8")).decode("ascii") + if py is None: + py = self._python_interpreter_cmd() + return self._exec( + f"{py} -c \"import base64; exec(base64.b64decode('{encoded}').decode())\"" + ) + def _try_read_utf16(self, path: str, offset: int, limit: int, file_size: int) -> "Optional[ReadResult]": """Attempt to read ``path`` as UTF-16 text, transcoded to UTF-8. @@ -1429,7 +1471,7 @@ def _try_read_utf16(self, path: str, offset: int, limit: int, snippet = ( "import sys, json, os\n" - f"p = {path!r}\n" + f"p = {json.dumps(path)}\n" f"offset = {int(offset)}\n" f"limit = {int(limit)}\n" f"MAX = {self._UTF16_MAX_BYTES}\n" @@ -1470,9 +1512,9 @@ def _try_read_utf16(self, path: str, offset: int, limit: int, " print('HERMES_UTF16:NO'); sys.exit(0)\n" ) - result = self._exec(f"python3 -c {self._escape_shell_arg(snippet)}") + result = self._exec_python_snippet(snippet) if result.exit_code != 0 and "python3" in (result.stdout or ""): - result = self._exec(f"python -c {self._escape_shell_arg(snippet)}") + result = self._exec_python_snippet(snippet, py="python") stdout = _strip_terminal_fence_leaks(result.stdout or "") marker = stdout.find("HERMES_UTF16:OK") @@ -1909,7 +1951,7 @@ def _python_delete(self, path: str, recursive: bool) -> WriteResult: # snippet via ``repr()`` so quoting is correct on every shell. snippet = ( "import shutil, pathlib, sys\n" - f"p = pathlib.Path({path!r})\n" + f"p = pathlib.Path({json.dumps(path)})\n" f"recursive = {bool(recursive)!r}\n" "try:\n" " if p.is_dir() and not p.is_symlink():\n" @@ -1929,12 +1971,12 @@ def _python_delete(self, path: str, recursive: bool) -> WriteResult: " print(str(exc), file=sys.stderr); sys.exit(1)\n" ) - result = self._exec(f"python3 -c {self._escape_shell_arg(snippet)}") + result = self._exec_python_snippet(snippet) - # Fall back to ``python`` (Windows / older systems where there's no + # Fall back to ``python`` (remote backends / older systems where there's no # ``python3`` symlink but a ``python`` binary is on PATH). if result.exit_code != 0 and "python3" in (result.stdout or ""): - result = self._exec(f"python -c {self._escape_shell_arg(snippet)}") + result = self._exec_python_snippet(snippet, py="python") if result.exit_code != 0: return WriteResult(error=f"Failed to delete {path}: {(result.stdout or '').strip() or 'unknown error'}") @@ -2987,10 +3029,10 @@ def _paths_note(per_file, cap: int = 5) -> str: extra = len(per_file) - cap return shown + (f" (+{extra} more)" if extra > 0 else "") - glob_expr = f" --glob {self._escape_shell_arg(file_glob)}" if file_glob else "" + glob_expr = f" --glob {self._escape_shell_arg(file_glob, translate_path=False)}" if file_glob else "" probe = self._exec( f"rg -i --count-matches{glob_expr} " - f"{self._escape_shell_arg(pattern)} {self._escape_native_tool_arg(path)} " + f"{self._escape_shell_arg(pattern, translate_path=False)} {self._escape_native_tool_arg(path)} " f"2>/dev/null | head -50", timeout=30, ) @@ -3007,7 +3049,7 @@ def _paths_note(per_file, cap: int = 5) -> str: # missing from results). hidden = self._exec( f"rg --hidden --no-ignore --count-matches{glob_expr} " - f"{self._escape_shell_arg(pattern)} {self._escape_native_tool_arg(path)} " + f"{self._escape_shell_arg(pattern, translate_path=False)} {self._escape_native_tool_arg(path)} " f"2>/dev/null | head -50", timeout=30, ) @@ -3021,7 +3063,7 @@ def _paths_note(per_file, cap: int = 5) -> str: if re.search(r"[.\[\](){}?*+^$\\|]", pattern): fixed = self._exec( f"rg -F --count-matches{glob_expr} " - f"{self._escape_shell_arg(pattern)} {self._escape_native_tool_arg(path)} " + f"{self._escape_shell_arg(pattern, translate_path=False)} {self._escape_native_tool_arg(path)} " f"2>/dev/null | head -50", timeout=30, ) @@ -3258,7 +3300,7 @@ def _search_with_rg(self, pattern: str, path: str, file_glob: Optional[str], cmd_parts.append("-c") # Count per file # Add pattern and path - cmd_parts.append(self._escape_shell_arg(pattern)) + cmd_parts.append(self._escape_shell_arg(pattern, translate_path=False)) # rg is a native Windows binary when installed via winget/cargo/choco: # it needs the C:/... path form, not the MSYS /c/... form (which # nothing converts back — Hermes sets MSYS_NO_PATHCONV for its bash). @@ -3418,7 +3460,7 @@ def _search_with_grep(self, pattern: str, path: str, file_glob: Optional[str], # ``.*`` to exclude the entire search. Anchor relative paths at the # shell's live cwd; quoting $PWD separately keeps user paths escaped # while working across local, container, and remote backends. - cmd_parts.append(self._escape_shell_arg(pattern)) + cmd_parts.append(self._escape_shell_arg(pattern, translate_path=False)) is_absolute = path.startswith(("/", "\\\\")) or bool( re.match(r"^[A-Za-z]:[\\/]", path) ) @@ -3467,7 +3509,7 @@ def _search_with_grep_pruned(self, pattern: str, path: str, file_glob: Optional[ grep_parts.append("-l") elif output_mode == "count": grep_parts.append("-c") - grep_parts.append(self._escape_shell_arg(pattern)) + grep_parts.append(self._escape_shell_arg(pattern, translate_path=False)) prune_terms = " -o ".join( f"-path {self._escape_shell_arg(item)}" for item in protected_paths diff --git a/tools/registry.py b/tools/registry.py index bf6d52f2ee63..23f3989a0022 100644 --- a/tools/registry.py +++ b/tools/registry.py @@ -25,7 +25,7 @@ from pathlib import Path from typing import Callable, Dict, List, Optional, Set -from hermes_constants import hermes_home_key +from hermes_constants import hermes_home_key, normalize_scope logger = logging.getLogger(__name__) @@ -550,6 +550,7 @@ def snapshot_registration( scope: Optional[str] = None, ) -> Optional[ToolEntry]: """Return the local slot state without following global fallback.""" + scope = normalize_scope(scope) with self._lock: target = self._tools if scope is None else self._scoped_tools.get(scope, {}) return target.get(name) @@ -607,6 +608,7 @@ def register_plugin_override_policy( The identity-bearing result lets plugin unload/reload revoke a stale authorization without losing durable module-to-profile attribution. """ + scope = normalize_scope(scope) with self._lock: policy = _PluginOverridePolicy(allowed) self._plugin_override_policy[(scope, module_namespace)] = policy @@ -620,6 +622,7 @@ def snapshot_plugin_override_policy( scope: Optional[str] = None, ) -> Optional[_PluginOverridePolicy]: """Return one local authorization generation without fallback.""" + scope = normalize_scope(scope) with self._lock: return self._plugin_override_policy.get((scope, module_namespace)) @@ -632,6 +635,7 @@ def restore_plugin_override_policy( scope: Optional[str] = None, ) -> bool: """CAS-restore policy state while retaining durable scope attribution.""" + scope = normalize_scope(scope) with self._lock: key = (scope, module_namespace) if self._plugin_override_policy.get(key) is not current: @@ -789,6 +793,7 @@ def register( owner = caller_owner or handler_owner if scope is None and owner is not None: scope = self._plugin_scope_of(owner) + scope = normalize_scope(scope) with self._lock: target = ( self._tools @@ -986,6 +991,7 @@ def restore_registration( newer entry under the same name, in which case unloading this entry must leave the newer entry untouched. """ + scope = normalize_scope(scope) with self._lock: target = ( self._tools diff --git a/tools/skills_tool.py b/tools/skills_tool.py index 53d7947aa8ed..9e4db9c54bb4 100644 --- a/tools/skills_tool.py +++ b/tools/skills_tool.py @@ -1646,7 +1646,7 @@ def _under_project(p: Path) -> bool: references_dir = skill_dir / "references" if references_dir.exists(): reference_files = [ - str(f.relative_to(skill_dir)) for f in references_dir.glob("*.md") + f.relative_to(skill_dir).as_posix() for f in references_dir.glob("*.md") ] templates_dir = skill_dir / "templates" @@ -1662,7 +1662,7 @@ def _under_project(p: Path) -> bool: ]: template_files.extend( [ - str(f.relative_to(skill_dir)) + f.relative_to(skill_dir).as_posix() for f in templates_dir.rglob(ext) ] ) @@ -1672,13 +1672,13 @@ def _under_project(p: Path) -> bool: if assets_dir.exists(): for f in assets_dir.rglob("*"): if f.is_file(): - asset_files.append(str(f.relative_to(skill_dir))) + asset_files.append(f.relative_to(skill_dir).as_posix()) scripts_dir = skill_dir / "scripts" if scripts_dir.exists(): for ext in ["*.py", "*.sh", "*.bash", "*.js", "*.ts", "*.rb"]: script_files.extend( - [str(f.relative_to(skill_dir)) for f in scripts_dir.glob(ext)] + [f.relative_to(skill_dir).as_posix() for f in scripts_dir.glob(ext)] ) # Read tags/related_skills with backward compat: diff --git a/uv.lock b/uv.lock index cadbe3f3f553..89047b226b92 100644 --- a/uv.lock +++ b/uv.lock @@ -44,13 +44,14 @@ pillow = false setuptools-rust = false tenacity = false pyjwt = false -slack-bolt = false +pytest-xdist = false fal-client = false asyncpg = false boto3 = false ruamel-yaml = false sherpa-onnx = false huggingface-hub = false +slack-bolt = false pytest-asyncio = false slack-sdk = false unpaddedbase64 = false @@ -1286,6 +1287,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/e2/bc/7a34e904a415040ba626948d0b0a36a08cd073f12b13342578a68331be3c/exa_py-2.10.2-py3-none-any.whl", hash = "sha256:ecb2a7581f4b7a8aeb6b434acce1bbc40f92ed1d4126b2aa6029913acd904a47", size = 72248, upload-time = "2026-03-26T20:29:37.306Z" }, ] +[[package]] +name = "execnet" +version = "2.1.2" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/bf/89/780e11f9588d9e7128a3f87788354c7946a9cbb1401ad38a48c4db9a4f07/execnet-2.1.2.tar.gz", hash = "sha256:63d83bfdd9a23e35b9c6a3261412324f964c2ec8dcd8d3c6916ee9373e0befcd", size = 166622, upload-time = "2025-11-12T09:56:37.75Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/ab/84/02fc1827e8cdded4aa65baef11296a9bbe595c474f0d6d758af082d849fd/execnet-2.1.2-py3-none-any.whl", hash = "sha256:67fba928dd5a544b783f6056f449e5e3931a5c378b128bc18501f7ea79e296ec", size = 40708, upload-time = "2025-11-12T09:56:36.333Z" }, +] + [[package]] name = "fal-client" version = "0.13.1" @@ -1753,6 +1763,7 @@ dev = [ { name = "mcp" }, { name = "pytest" }, { name = "pytest-asyncio" }, + { name = "pytest-xdist" }, { name = "ruff" }, { name = "setuptools" }, { name = "starlette" }, @@ -2011,6 +2022,7 @@ requires-dist = [ { name = "pyjwt", extras = ["crypto"], specifier = "==2.13.0" }, { name = "pytest", marker = "extra == 'dev'", specifier = "==9.1.1" }, { name = "pytest-asyncio", marker = "extra == 'dev'", specifier = "==1.3.0" }, + { name = "pytest-xdist", marker = "extra == 'dev'", specifier = "==3.8.0" }, { name = "python-dotenv", specifier = "==1.2.2" }, { name = "python-multipart", specifier = ">=0.0.9,<1" }, { name = "python-multipart", marker = "extra == 'web'", specifier = "==0.0.32" }, @@ -3701,6 +3713,19 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/e5/35/f8b19922b6a25bc0880171a2f1a003eaeb93657475193ab516fd87cac9da/pytest_asyncio-1.3.0-py3-none-any.whl", hash = "sha256:611e26147c7f77640e6d0a92a38ed17c3e9848063698d5c93d5aa7aa11cebff5", size = 15075, upload-time = "2025-11-10T16:07:45.537Z" }, ] +[[package]] +name = "pytest-xdist" +version = "3.8.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "execnet" }, + { name = "pytest" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/78/b4/439b179d1ff526791eb921115fca8e44e596a13efeda518b9d845a619450/pytest_xdist-3.8.0.tar.gz", hash = "sha256:7e578125ec9bc6050861aa93f2d59f1d8d085595d6551c2c90b6f4fad8d3a9f1", size = 88069, upload-time = "2025-07-01T13:30:59.346Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/ca/31/d4e37e9e550c2b92a9cbc2e4d0b7420a27224968580b5a447f420847c975/pytest_xdist-3.8.0-py3-none-any.whl", hash = "sha256:202ca578cfeb7370784a8c33d6d05bc6e13b4f25b5053c30a152269fd10f0b88", size = 46396, upload-time = "2025-07-01T13:30:56.632Z" }, +] + [[package]] name = "python-dateutil" version = "2.9.0.post0" @@ -4155,7 +4180,7 @@ resolution-markers = [ "python_full_version < '3.12'", ] dependencies = [ - { name = "numpy", marker = "python_full_version < '3.12'" }, + { name = "numpy" }, ] sdist = { url = "https://files.pythonhosted.org/packages/7a/97/5a3609c4f8d58b039179648e62dd220f89864f56f7357f5d4f45c29eb2cc/scipy-1.17.1.tar.gz", hash = "sha256:95d8e012d8cb8816c226aef832200b1d45109ed4464303e997c5b13122b297c0", size = 30573822, upload-time = "2026-02-23T00:26:24.851Z" } wheels = [ @@ -4210,7 +4235,7 @@ resolution-markers = [ "python_full_version == '3.12.*'", ] dependencies = [ - { name = "numpy", marker = "python_full_version >= '3.12'" }, + { name = "numpy" }, ] sdist = { url = "https://files.pythonhosted.org/packages/a7/25/c2700dfaf6442b4effaa91af24ebce5dc9d31bb4a69706313aae70d72cd0/scipy-1.18.0.tar.gz", hash = "sha256:67b2ad2ad54c72ca6d04975a9b2df8c3638c34ddd5b28738e94fc2b57929d378", size = 30774447, upload-time = "2026-06-19T15:01:43.456Z" } wheels = [ @@ -4858,11 +4883,11 @@ name = "vercel-workers" version = "0.0.25" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "anyio", marker = "python_full_version >= '3.12'" }, - { name = "httpx", marker = "python_full_version >= '3.12'" }, - { name = "pydantic", marker = "python_full_version >= '3.12'" }, - { name = "python-dotenv", marker = "python_full_version >= '3.12'" }, - { name = "vercel", marker = "python_full_version >= '3.12'" }, + { name = "anyio" }, + { name = "httpx" }, + { name = "pydantic" }, + { name = "python-dotenv" }, + { name = "vercel" }, ] sdist = { url = "https://files.pythonhosted.org/packages/30/df/04d37021ad7ca53b7599c313e411d91623c7a005c741f491d1eefb7a9f0c/vercel_workers-0.0.25.tar.gz", hash = "sha256:212ded01400b524be51d251df49f801caf115ad7d48cca7eb168cbeceda3def3", size = 64149, upload-time = "2026-06-20T19:26:27.177Z" } wheels = [ diff --git a/website/docs/user-guide/skills/bundled/software-development/software-development-python-debugpy.md b/website/docs/user-guide/skills/bundled/software-development/software-development-python-debugpy.md index e6f1120e0835..491afaf9d76e 100644 --- a/website/docs/user-guide/skills/bundled/software-development/software-development-python-debugpy.md +++ b/website/docs/user-guide/skills/bundled/software-development/software-development-python-debugpy.md @@ -125,7 +125,7 @@ scripts/run_tests.sh tests/path/to/test_file.py::test_name --trace scripts/run_tests.sh tests/path/to/test_file.py --showlocals --tb=long ``` -Note: `scripts/run_tests.sh` runs each test file in a captured subprocess via `run_tests_parallel.py` (no xdist), so interactive pdb does NOT work under the wrapper. Run pytest directly for `--pdb`: +Note: `scripts/run_tests.sh` runs pytest under xdist workers, so interactive pdb does NOT work under the wrapper. Run pytest directly for `--pdb`: ```bash source .venv/bin/activate diff --git a/website/scripts/generate-llms-txt.py b/website/scripts/generate-llms-txt.py index 91caeb4ad377..2dcfc3e61d9b 100644 --- a/website/scripts/generate-llms-txt.py +++ b/website/scripts/generate-llms-txt.py @@ -226,7 +226,7 @@ def slug_for(path: Path) -> str: rel = path.relative_to(DOCS).with_suffix("") if rel.name == "index": rel = rel.parent - return "" if str(rel) == "." else str(rel) + return "" if str(rel) == "." else rel.as_posix() def doc_path(slug: str) -> Path | None: