From 39f34a19905f214d882b136a792f5e2bb3a3659a Mon Sep 17 00:00:00 2001 From: Honglie Yi Date: Thu, 24 Sep 2026 13:42:39 +0000 Subject: [PATCH 1/6] ci(lmcache): publish LMCache wheels as Docker Hub images, not releases A GitHub Release needs a git tag. The lmcache-v...-rocm-torch210 tag created on main was picked up by setuptools-scm's git describe and broke `pip install -e .` in every Pre Checkin run. Push the wheel as a FROM scratch image, rocm/atom-dev:lmcache-v-g-rocm-torch210, instead; nothing appears under Releases or Tags, and the publish job no longer has contents: write. source_run_id publishes the wheel an earlier run built instead of rebuilding it. Co-Authored-By: Claude Opus 5.5 (1M context) --- .github/workflows/lmcache-rocm-wheel.yaml | 209 ++++++++++++++++------ 1 file changed, 157 insertions(+), 52 deletions(-) diff --git a/.github/workflows/lmcache-rocm-wheel.yaml b/.github/workflows/lmcache-rocm-wheel.yaml index 1ca4d27684..0d5a411d51 100644 --- a/.github/workflows/lmcache-rocm-wheel.yaml +++ b/.github/workflows/lmcache-rocm-wheel.yaml @@ -1,7 +1,7 @@ name: LMCache ROCm torch 2.10 Wheel # Builds an LMCache wheel for the ATOM image's torch ABI from any LMCache commit -# and, when asked, publishes it as a pinned GitHub Release on this repository. +# and, when asked, publishes it as a wheel-only image on Docker Hub. # # The ATOM image ships AMD's ROCm 7.2.4 / torch 2.10.0 build. LMCache only # publishes a wheel for that ABI with tagged releases; its rolling nightly-rocm @@ -21,10 +21,13 @@ name: LMCache ROCm torch 2.10 Wheel # # lmcache-0.5.6.dev98+g05fc77a0.rocm7.2.4.torch2.10.git3d3aa833.cxx11abi1-cp312-cp312-manylinux_2_39_x86_64.whl # -# Published releases are tagged lmcache-v-g-rocm-torch210, are -# never overwritten, and are marked pre-release and not-latest so they do not -# displace ATOM's own release on the repository page. The Dockerfiles pin one of -# them by URL and sha256. +# Published wheels are FROM scratch images holding only the wheel at /, pushed +# as rocm/atom-dev:lmcache-v-g-rocm-torch210 and never +# overwritten. They are deliberately not GitHub Releases: a release needs a git +# tag, and a tag reachable from main is what setuptools-scm's git describe picks +# up when computing ATOM's own version. The Dockerfiles pin the image by digest +# and the wheel by sha256. source_run_id publishes the wheel an earlier +# run built instead of rebuilding it, while that run's artifact is still kept. on: workflow_dispatch: @@ -34,10 +37,15 @@ on: type: string required: true publish: - description: "Publish the wheel as a GitHub Release on this repository. Off = build and validate only." + description: "Push the wheel to Docker Hub as rocm/atom-dev:lmcache-v...-rocm-torch210. Off = build and validate only." type: boolean required: false default: false + source_run_id: + description: "Optional: publish the wheel built by this earlier run of this workflow instead of building (its lmcache_commit must match)." + type: string + required: false + default: '' runner: description: "Runner to build on. A GPU runner also runs LMCache's HIP transfer kernels." type: choice @@ -73,10 +81,10 @@ env: jobs: build: name: Build and validate LMCache ${{ inputs.lmcache_commit }} + if: ${{ inputs.source_run_id == '' }} runs-on: ${{ inputs.runner || 'build-only-atom' }} outputs: version: ${{ steps.version.outputs.version }} - release_tag: ${{ steps.version.outputs.release_tag }} wheel: ${{ steps.wheel.outputs.wheel }} sha256: ${{ steps.wheel.outputs.sha256 }} steps: @@ -84,7 +92,7 @@ jobs: env: LMCACHE_COMMIT: ${{ inputs.lmcache_commit }} run: | - # A branch name would make the release say one thing and hold another + # A branch name would make the image tag say one thing and hold another # the next time someone reruns it. if ! [[ "${LMCACHE_COMMIT}" =~ ^[0-9a-f]{40}$ ]]; then echo "lmcache_commit must be a full 40-character lowercase SHA, got '${LMCACHE_COMMIT}'." @@ -127,7 +135,6 @@ jobs: short="${LMCACHE_COMMIT:0:8}" version="${base}+g${short}.${ROCM_TORCH210_LOCAL_VERSION}" echo "version=${version}" >> "$GITHUB_OUTPUT" - echo "release_tag=lmcache-v${base}-g${short}-rocm-torch210" >> "$GITHUB_OUTPUT" echo "LMCache ${LMCACHE_COMMIT} -> ${version}" - name: Build the wheel @@ -153,11 +160,18 @@ jobs: - name: Record the wheel id: wheel + env: + VERSION: ${{ steps.version.outputs.version }} run: | set -euo pipefail wheels=(lmcache-src/dist_rocm_torch210/*.whl) test "${#wheels[@]}" -eq 1 wheel=$(basename "${wheels[0]}") + # The publish job parses the version from this name. + if [ "${wheel}" != "lmcache-${VERSION}-${EXPECTED_WHEEL_TAG}.whl" ]; then + echo "Unexpected wheel name ${wheel}; the Dockerfiles expect lmcache-${VERSION}-${EXPECTED_WHEEL_TAG}.whl." + exit 1 + fi sha256=$(sha256sum "${wheels[0]}" | cut -d' ' -f1) echo "wheel=${wheel}" >> "$GITHUB_OUTPUT" echo "sha256=${sha256}" >> "$GITHUB_OUTPUT" @@ -226,67 +240,158 @@ jobs: fi publish: - name: Publish the wheel as a GitHub Release + name: Publish the wheel image to Docker Hub needs: build - if: ${{ inputs.publish }} + # With source_run_id the build job is skipped on purpose. + if: >- + ${{ !cancelled() && inputs.publish && + (needs.build.result == 'success' || + (inputs.source_run_id != '' && needs.build.result == 'skipped')) }} runs-on: ubuntu-latest permissions: - contents: write + contents: read + actions: read + outputs: + version: ${{ steps.wheel.outputs.version }} + image: ${{ steps.push.outputs.image }} + image_tag: ${{ steps.wheel.outputs.image_tag }} + wheel: ${{ steps.wheel.outputs.wheel }} + sha256: ${{ steps.wheel.outputs.sha256 }} + artifact: ${{ steps.wheel.outputs.artifact }} + source_run: ${{ steps.wheel.outputs.source_run }} env: - GH_TOKEN: ${{ github.token }} - GH_REPO: ${{ github.repository }} + IMAGE_REPO: rocm/atom-dev LMCACHE_COMMIT: ${{ inputs.lmcache_commit }} - VERSION: ${{ needs.build.outputs.version }} - RELEASE_TAG: ${{ needs.build.outputs.release_tag }} - WHEEL: ${{ needs.build.outputs.wheel }} - SHA256: ${{ needs.build.outputs.sha256 }} + SOURCE_RUN_ID: ${{ inputs.source_run_id }} + BUILT_SHA256: ${{ needs.build.outputs.sha256 }} + DOCKER_USERNAME: ${{ secrets.DOCKER_USERNAME }} + DOCKER_PASSWORD: ${{ secrets.DOCKER_PASSWORD }} steps: + - name: Checkout ATOM + uses: actions/checkout@v4 + with: + sparse-checkout: | + .github/actions + .github/scripts + - name: Download the wheel uses: actions/download-artifact@v4 with: - name: lmcache-rocm-torch210-${{ github.run_id }}-${{ github.run_attempt }} - path: release-wheel + # An earlier run may have several attempts; the next step insists on + # exactly one wheel rather than picking one. + pattern: >- + ${{ inputs.source_run_id != '' + && format('lmcache-rocm-torch210-{0}-*', inputs.source_run_id) + || format('lmcache-rocm-torch210-{0}-{1}', github.run_id, github.run_attempt) }} + run-id: ${{ inputs.source_run_id || github.run_id }} + github-token: ${{ github.token }} + path: ${{ runner.temp }}/wheel-artifacts - - name: Create the release + - name: Check the wheel + id: wheel run: | set -euo pipefail - echo "${SHA256} release-wheel/${WHEEL}" | sha256sum -c - - # A pinned Dockerfile trusts this URL and sha256 forever; never - # replace an asset under a tag that may already be referenced. - if gh release view "${RELEASE_TAG}" >/dev/null 2>&1; then - echo "Release ${RELEASE_TAG} already exists; not overwriting it." + if [ -n "${SOURCE_RUN_ID}" ] && ! [[ "${SOURCE_RUN_ID}" =~ ^[0-9]+$ ]]; then + echo "source_run_id must be a run ID, got '${SOURCE_RUN_ID}'." exit 1 fi - url="https://github.com/${GH_REPO}/releases/download/${RELEASE_TAG}/${WHEEL//+/%2B}" - cat > notes.md <+g.-.whl, as the build job names it. + suffix=".${ROCM_TORCH210_LOCAL_VERSION}-${EXPECTED_WHEEL_TAG}.whl" + short="${LMCACHE_COMMIT:0:8}" + if [[ "${wheel}" != lmcache-*"+g${short}${suffix}" ]]; then + echo "${wheel} is not an lmcache_commit ${LMCACHE_COMMIT} wheel for this ABI." + exit 1 + fi + version="${wheel#lmcache-}" + version="${version%-${EXPECTED_WHEEL_TAG}.whl}" + base="${version%%+*}" + sha256=$(sha256sum "${path}" | cut -d' ' -f1) + if [ -n "${BUILT_SHA256}" ] && [ "${sha256}" != "${BUILT_SHA256}" ]; then + echo "Downloaded wheel sha256 ${sha256} does not match the build's ${BUILT_SHA256}." + exit 1 + fi + mkdir -p "${RUNNER_TEMP}/wheel-image" + cp "${path}" "${RUNNER_TEMP}/wheel-image/" + { + echo "wheel=${wheel}" + echo "version=${version}" + echo "sha256=${sha256}" + echo "image_tag=lmcache-v${base}-g${short}-rocm-torch210" + echo "artifact=$(basename "$(dirname "${path}")")" + echo "source_run=${SOURCE_RUN_ID:-${GITHUB_RUN_ID}}" + } >> "$GITHUB_OUTPUT" + echo "${sha256} ${wheel}" - Pin in docker/Dockerfile: + - name: Log in to Docker Hub + uses: ./.github/actions/docker-auth + with: + username: ${{ secrets.DOCKER_USERNAME }} + password: ${{ secrets.DOCKER_PASSWORD }} - \`\`\` - ARG LMCACHE_WHEEL_NAME=${WHEEL} - ARG LMCACHE_WHEEL_URL=${url} - ARG LMCACHE_WHEEL_SHA256=${SHA256} - \`\`\` + - name: Build and push the wheel image + id: push + env: + VERSION: ${{ steps.wheel.outputs.version }} + IMAGE_TAG: ${{ steps.wheel.outputs.image_tag }} + SHA256: ${{ steps.wheel.outputs.sha256 }} + SOURCE_RUN: ${{ steps.wheel.outputs.source_run }} + run: | + set -euo pipefail + ref="${IMAGE_REPO}:${IMAGE_TAG}" + # The Dockerfiles pin the digest, so a moved tag would not change what + # they install, but it would make the tag lie about the pinned wheel. + if docker manifest inspect "${ref}" >/dev/null 2>&1; then + echo "${ref} already exists; not overwriting it." + exit 1 + fi + cat > "${RUNNER_TEMP}/wheel-image/Dockerfile" <<'EOF' + FROM scratch + COPY *.whl / EOF - gh release create "${RELEASE_TAG}" \ - --target "${GITHUB_SHA}" \ - --title "LMCache ${VERSION}" \ - --notes-file notes.md \ - --prerelease \ - --latest=false \ - "release-wheel/${WHEEL}" + docker build \ + --label "org.opencontainers.image.source=${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}" \ + --label "org.opencontainers.image.description=LMCache ${VERSION} wheel for ATOM's ROCm 7.2.4 / torch 2.10.0 image" \ + --label "com.amd.atom.lmcache.commit=${LMCACHE_COMMIT}" \ + --label "com.amd.atom.lmcache.version=${VERSION}" \ + --label "com.amd.atom.lmcache.wheel-sha256=${SHA256}" \ + --label "com.amd.atom.lmcache.build-run=${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${SOURCE_RUN}" \ + -t "${ref}" "${RUNNER_TEMP}/wheel-image" + .github/scripts/docker_push_retry.sh "${ref}" + digest=$(docker inspect --format '{{range .RepoDigests}}{{println .}}{{end}}' "${ref}" \ + | sed -n "s|^${IMAGE_REPO}@||p" | head -1) + if ! [[ "${digest}" =~ ^sha256:[0-9a-f]{64}$ ]]; then + echo "Could not read the pushed digest of ${ref}." + exit 1 + fi + echo "image=${ref}@${digest}" >> "$GITHUB_OUTPUT" - - name: Release summary + - name: Publish summary + env: + VERSION: ${{ steps.wheel.outputs.version }} + IMAGE: ${{ steps.push.outputs.image }} + WHEEL: ${{ steps.wheel.outputs.wheel }} + SHA256: ${{ steps.wheel.outputs.sha256 }} + SOURCE_RUN: ${{ steps.wheel.outputs.source_run }} run: | { - echo "## LMCache Release" - echo "- **Tag**: ${RELEASE_TAG}" - echo "- **Wheel**: ${WHEEL}" - echo "- **sha256**: ${SHA256}" + echo "## LMCache wheel image" + echo "- **LMCache commit**: [${LMCACHE_COMMIT}](https://github.com/LMCache/LMCache/commit/${LMCACHE_COMMIT})" + echo "- **Version**: \`${VERSION}\`" + echo "- **Wheel**: \`${WHEEL}\`" + echo "- **Built by**: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${SOURCE_RUN}" + echo "- **Build image**: \`${ROCM_PYTORCH_IMAGE}\`" + echo + echo "Pin in docker/Dockerfile and docker/atom_release.dockerfile:" + echo + echo '```' + echo "ARG LMCACHE_WHEEL_IMAGE=\"${IMAGE}\"" + echo "ARG LMCACHE_WHEEL_SHA256=${SHA256}" + echo '```' } >> "$GITHUB_STEP_SUMMARY" From 53717bd2ccd7be01e9c388f0ed51e8c690afb7ff Mon Sep 17 00:00:00 2001 From: Honglie Yi Date: Thu, 24 Sep 2026 12:21:25 +0000 Subject: [PATCH 2/6] fix(build): only use v* tags for setuptools_scm version The lmcache-v0.5.6.dev98-g05fc77a0-rocm-torch210 release tag is now the nearest tag reachable from main, and setuptools_scm cannot parse it as a PEP 440 version, so `pip install -e .` fails in the Pre Checkin CI. Restrict git describe to tags matching v[0-9]* so non-ATOM release tags are ignored while the ATOM version is still derived from vX.Y.Z tags. Co-Authored-By: Claude Opus 5.5 (1M context) --- pyproject.toml | 1 + 1 file changed, 1 insertion(+) diff --git a/pyproject.toml b/pyproject.toml index 98971cd71f..a581519095 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -28,6 +28,7 @@ Issues = "https://github.com/ROCm/ATOM/issues" write_to = "atom/_version.py" write_to_template = "__version__ = '{version}'\n" fallback_version = "0.1.0" +git_describe_command = ["git", "describe", "--dirty", "--tags", "--long", "--match", "v[0-9]*"] [tool.setuptools.packages.find] where = ["."] From 409206dc710ca2a4e94f88e7a9562f181aa800c5 Mon Sep 17 00:00:00 2001 From: Honglie Yi Date: Thu, 24 Sep 2026 13:58:12 +0000 Subject: [PATCH 3/6] ci: address review of the LMCache wheel publish path - atom-release.yaml: git describe only matches v[0-9]* tags, like setuptools-scm now does, so a stray tag cannot become the release version. - check-inputs job validates lmcache_commit (also on the source_run_id path), source_run_id's format, and rejects source_run_id without publish instead of finishing green having done nothing. - publish resolves exactly one artifact: the build's own attempt (so "Re-run failed jobs" finds it), or for source_run_id the latest unexpired attempt of a run of this workflow on the default or current branch whose build job succeeded. - Only "no such manifest" counts as the tag being free; rate limits, 5xx and network errors now fail instead of allowing an overwrite. - Docker Hub credentials are scoped to the push step. Co-Authored-By: Claude Opus 5.5 (1M context) --- .github/workflows/atom-release.yaml | 4 +- .github/workflows/lmcache-rocm-wheel.yaml | 128 +++++++++++++++++----- 2 files changed, 102 insertions(+), 30 deletions(-) diff --git a/.github/workflows/atom-release.yaml b/.github/workflows/atom-release.yaml index 0a179a7a3e..3d7cbfb649 100644 --- a/.github/workflows/atom-release.yaml +++ b/.github/workflows/atom-release.yaml @@ -260,7 +260,9 @@ jobs: # the runner host, whose python is whatever the host image ships, # and subprocess.run(capture_output=...) needs 3.7. The python below # only normalises a string, so it works on any python3. - DESCRIBE=$(git describe --tags --long --dirty) + # Only ATOM's own vX.Y.Z tags; others (e.g. lmcache-v...) would become + # the base version. Same filter as setuptools-scm in pyproject.toml. + DESCRIBE=$(git describe --tags --long --dirty --match 'v[0-9]*') VERSIONS=$(ROCM_VER="${ROCM_VER}" DATE_STAMP="${DATE_STAMP}" DESCRIBE="${DESCRIBE}" python3 - <<'PYEOF' import os, re diff --git a/.github/workflows/lmcache-rocm-wheel.yaml b/.github/workflows/lmcache-rocm-wheel.yaml index 0d5a411d51..16f9cb80e5 100644 --- a/.github/workflows/lmcache-rocm-wheel.yaml +++ b/.github/workflows/lmcache-rocm-wheel.yaml @@ -79,26 +79,48 @@ env: ROCM_TORCH210_LOCAL_VERSION: rocm7.2.4.torch2.10.git3d3aa833.cxx11abi1 jobs: - build: - name: Build and validate LMCache ${{ inputs.lmcache_commit }} - if: ${{ inputs.source_run_id == '' }} - runs-on: ${{ inputs.runner || 'build-only-atom' }} - outputs: - version: ${{ steps.version.outputs.version }} - wheel: ${{ steps.wheel.outputs.wheel }} - sha256: ${{ steps.wheel.outputs.sha256 }} + # Both the build and the source_run_id publish path rely on these, so they + # are checked once, before either runs. + check-inputs: + name: Check inputs + runs-on: ubuntu-latest + env: + LMCACHE_COMMIT: ${{ inputs.lmcache_commit }} + SOURCE_RUN_ID: ${{ inputs.source_run_id }} + PUBLISH: ${{ inputs.publish }} steps: - - name: Check the requested commit - env: - LMCACHE_COMMIT: ${{ inputs.lmcache_commit }} + - name: Check inputs run: | + set -euo pipefail # A branch name would make the image tag say one thing and hold another # the next time someone reruns it. if ! [[ "${LMCACHE_COMMIT}" =~ ^[0-9a-f]{40}$ ]]; then echo "lmcache_commit must be a full 40-character lowercase SHA, got '${LMCACHE_COMMIT}'." exit 1 fi + if [ -n "${SOURCE_RUN_ID}" ]; then + if ! [[ "${SOURCE_RUN_ID}" =~ ^[0-9]+$ ]]; then + echo "source_run_id must be the numeric run ID (.../actions/runs/), got '${SOURCE_RUN_ID}'." + exit 1 + fi + # Otherwise both jobs skip and the run is green having done nothing. + if [ "${PUBLISH}" != "true" ]; then + echo "source_run_id only selects the wheel to publish; set publish=true as well." + exit 1 + fi + fi + build: + name: Build and validate LMCache ${{ inputs.lmcache_commit }} + needs: check-inputs + if: ${{ inputs.source_run_id == '' }} + runs-on: ${{ inputs.runner || 'build-only-atom' }} + outputs: + version: ${{ steps.version.outputs.version }} + wheel: ${{ steps.wheel.outputs.wheel }} + sha256: ${{ steps.wheel.outputs.sha256 }} + artifact: ${{ steps.wheel.outputs.artifact }} + steps: - name: Checkout ATOM uses: actions/checkout@v4 with: @@ -175,6 +197,7 @@ jobs: sha256=$(sha256sum "${wheels[0]}" | cut -d' ' -f1) echo "wheel=${wheel}" >> "$GITHUB_OUTPUT" echo "sha256=${sha256}" >> "$GITHUB_OUTPUT" + echo "artifact=lmcache-rocm-torch210-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}" >> "$GITHUB_OUTPUT" echo "${sha256} ${wheel}" # A clean container of the build image, so nothing from the build tree or @@ -241,10 +264,10 @@ jobs: publish: name: Publish the wheel image to Docker Hub - needs: build + needs: [check-inputs, build] # With source_run_id the build job is skipped on purpose. if: >- - ${{ !cancelled() && inputs.publish && + ${{ !cancelled() && inputs.publish && needs.check-inputs.result == 'success' && (needs.build.result == 'success' || (inputs.source_run_id != '' && needs.build.result == 'skipped')) }} runs-on: ubuntu-latest @@ -264,8 +287,6 @@ jobs: LMCACHE_COMMIT: ${{ inputs.lmcache_commit }} SOURCE_RUN_ID: ${{ inputs.source_run_id }} BUILT_SHA256: ${{ needs.build.outputs.sha256 }} - DOCKER_USERNAME: ${{ secrets.DOCKER_USERNAME }} - DOCKER_PASSWORD: ${{ secrets.DOCKER_PASSWORD }} steps: - name: Checkout ATOM uses: actions/checkout@v4 @@ -274,27 +295,69 @@ jobs: .github/actions .github/scripts + - name: Resolve the wheel artifact + id: source + env: + GH_TOKEN: ${{ github.token }} + BUILT_ARTIFACT: ${{ needs.build.outputs.artifact }} + DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} + WORKFLOW_PATH: .github/workflows/lmcache-rocm-wheel.yaml + run: | + set -euo pipefail + if [ -z "${SOURCE_RUN_ID}" ]; then + # Named after the attempt that ran the build, which "Re-run failed + # jobs" reuses when only this job failed. + echo "run_id=${GITHUB_RUN_ID}" >> "$GITHUB_OUTPUT" + echo "artifact=${BUILT_ARTIFACT}" >> "$GITHUB_OUTPUT" + exit 0 + fi + api="repos/${GITHUB_REPOSITORY}/actions/runs/${SOURCE_RUN_ID}" + read -r path branch < <(gh api "${api}" --jq '"\(.path) \(.head_branch)"') + # Only this workflow's build and validation can vouch for the wheel, + # and only as defined on the default branch or the branch publishing it. + if [ "${path}" != "${WORKFLOW_PATH}" ]; then + echo "Run ${SOURCE_RUN_ID} is ${path}, not ${WORKFLOW_PATH}." + exit 1 + fi + if [ "${branch}" != "${DEFAULT_BRANCH}" ] && [ "${branch}" != "${GITHUB_REF_NAME}" ]; then + echo "Run ${SOURCE_RUN_ID} ran on ${branch}; publish it from ${branch} or ${DEFAULT_BRANCH}." + exit 1 + fi + # A run re-run in full has one artifact per attempt; take the latest. + artifact=$(gh api "${api}/artifacts?per_page=100" --jq " + [.artifacts[] | select(.expired | not) + | select(.name | startswith(\"lmcache-rocm-torch210-${SOURCE_RUN_ID}-\"))] + | sort_by(.name | split(\"-\") | last | tonumber) | last | .name // empty") + if [ -z "${artifact}" ]; then + echo "Run ${SOURCE_RUN_ID} has no unexpired wheel artifact." + exit 1 + fi + attempt="${artifact##*-}" + build=$(gh api "${api}/attempts/${attempt}/jobs?per_page=100" \ + --jq '[.jobs[] | select(.name | startswith("Build and validate"))][0].conclusion // empty') + if [ "${build}" != "success" ]; then + echo "Run ${SOURCE_RUN_ID} attempt ${attempt}: build job concluded '${build}', not success." + exit 1 + fi + echo "run_id=${SOURCE_RUN_ID}" >> "$GITHUB_OUTPUT" + echo "artifact=${artifact}" >> "$GITHUB_OUTPUT" + echo "Publishing ${artifact} from ${branch}" + - name: Download the wheel uses: actions/download-artifact@v4 with: - # An earlier run may have several attempts; the next step insists on - # exactly one wheel rather than picking one. - pattern: >- - ${{ inputs.source_run_id != '' - && format('lmcache-rocm-torch210-{0}-*', inputs.source_run_id) - || format('lmcache-rocm-torch210-{0}-{1}', github.run_id, github.run_attempt) }} - run-id: ${{ inputs.source_run_id || github.run_id }} + name: ${{ steps.source.outputs.artifact }} + run-id: ${{ steps.source.outputs.run_id }} github-token: ${{ github.token }} path: ${{ runner.temp }}/wheel-artifacts - name: Check the wheel id: wheel + env: + ARTIFACT: ${{ steps.source.outputs.artifact }} + SOURCE_RUN: ${{ steps.source.outputs.run_id }} run: | set -euo pipefail - if [ -n "${SOURCE_RUN_ID}" ] && ! [[ "${SOURCE_RUN_ID}" =~ ^[0-9]+$ ]]; then - echo "source_run_id must be a run ID, got '${SOURCE_RUN_ID}'." - exit 1 - fi mapfile -t wheels < <(find "${RUNNER_TEMP}/wheel-artifacts" -name '*.whl') if [ "${#wheels[@]}" -ne 1 ]; then echo "Expected exactly one wheel, found ${#wheels[@]}: ${wheels[*]}" @@ -324,8 +387,8 @@ jobs: echo "version=${version}" echo "sha256=${sha256}" echo "image_tag=lmcache-v${base}-g${short}-rocm-torch210" - echo "artifact=$(basename "$(dirname "${path}")")" - echo "source_run=${SOURCE_RUN_ID:-${GITHUB_RUN_ID}}" + echo "artifact=${ARTIFACT}" + echo "source_run=${SOURCE_RUN}" } >> "$GITHUB_OUTPUT" echo "${sha256} ${wheel}" @@ -342,14 +405,21 @@ jobs: IMAGE_TAG: ${{ steps.wheel.outputs.image_tag }} SHA256: ${{ steps.wheel.outputs.sha256 }} SOURCE_RUN: ${{ steps.wheel.outputs.source_run }} + DOCKER_USERNAME: ${{ secrets.DOCKER_USERNAME }} + DOCKER_PASSWORD: ${{ secrets.DOCKER_PASSWORD }} run: | set -euo pipefail ref="${IMAGE_REPO}:${IMAGE_TAG}" # The Dockerfiles pin the digest, so a moved tag would not change what # they install, but it would make the tag lie about the pinned wheel. - if docker manifest inspect "${ref}" >/dev/null 2>&1; then + # Anything but "no such manifest" (rate limit, 5xx, network) is not + # proof that the tag is free. + if err=$(docker manifest inspect "${ref}" 2>&1 >/dev/null); then echo "${ref} already exists; not overwriting it." exit 1 + elif [[ "${err}" != *"no such manifest"* ]]; then + echo "Could not check whether ${ref} exists: ${err}" + exit 1 fi cat > "${RUNNER_TEMP}/wheel-image/Dockerfile" <<'EOF' FROM scratch From 824ac16e3e5ba76b67d8a47ff514e08b09d50bf8 Mon Sep 17 00:00:00 2001 From: Honglie Yi Date: Thu, 24 Sep 2026 14:32:06 +0000 Subject: [PATCH 4/6] build(docker): install LMCache from the digest-pinned wheel image Move main's images from LMCache's 0.5.5rc3 release wheel to the 0.5.6.dev98 (dev@05fc77a) wheel published by lmcache-rocm-wheel.yaml, pulled from rocm/atom-dev by digest via a FROM scratch stage and checked by sha256. The workflow's bump-pr job and bump_lmcache_wheel_pin.py move this pin from now on. Adds grpcio and protobuf, which the newer wheel requires. Co-Authored-By: Claude Opus 5.5 (1M context) --- .github/scripts/bump_lmcache_wheel_pin.py | 61 +++++++++++++ .github/workflows/lmcache-rocm-wheel.yaml | 104 +++++++++++++++++++++- docker/Dockerfile | 39 +++++--- docker/atom_release.dockerfile | 40 ++++++--- 4 files changed, 220 insertions(+), 24 deletions(-) create mode 100755 .github/scripts/bump_lmcache_wheel_pin.py diff --git a/.github/scripts/bump_lmcache_wheel_pin.py b/.github/scripts/bump_lmcache_wheel_pin.py new file mode 100755 index 0000000000..7c3661444f --- /dev/null +++ b/.github/scripts/bump_lmcache_wheel_pin.py @@ -0,0 +1,61 @@ +#!/usr/bin/env python3 +"""Point the Dockerfiles' LMCache wheel pin at a new wheel image. + +The pin is LMCACHE_WHEEL_IMAGE (the wheel-only image, by digest) and +LMCACHE_WHEEL_SHA256; the Dockerfiles derive the expected lmcache.__version__ +from the wheel name inside the image, so nothing else changes. Every file must +carry each arg exactly once, otherwise the layout has drifted from what this +script knows and it refuses to guess. +""" + +from __future__ import annotations + +import argparse +import re +import sys +from pathlib import Path + +DEFAULT_FILES = ["docker/Dockerfile", "docker/atom_release.dockerfile"] + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument( + "--image", + required=True, + help="Wheel image, e.g. rocm/atom-dev:lmcache-v...-rocm-torch210@sha256:.", + ) + parser.add_argument("--sha256", required=True) + parser.add_argument("files", nargs="*", default=DEFAULT_FILES) + args = parser.parse_args() + + # A tag alone could be moved; the Dockerfiles must pin the digest. + if not re.fullmatch( + r"[a-z0-9./-]+:lmcache-v[^-]+-g[0-9a-f]{8}-rocm-torch210@sha256:[0-9a-f]{64}", + args.image, + ): + parser.error(f"not a digest-pinned LMCache wheel image: {args.image}") + if not re.fullmatch(r"[0-9a-f]{64}", args.sha256): + parser.error(f"not a sha256 digest: {args.sha256}") + + values = { + "LMCACHE_WHEEL_IMAGE": f'"{args.image}"', + "LMCACHE_WHEEL_SHA256": args.sha256, + } + for file in args.files: + path = Path(file) + text = path.read_text() + for key, value in values.items(): + matches = list(re.finditer(rf"^ARG {key}=.*$", text, re.MULTILINE)) + if len(matches) != 1: + print(f"{path}: expected one 'ARG {key}=', found {len(matches)}") + return 1 + start, end = matches[0].span() + text = f"{text[:start]}ARG {key}={value}{text[end:]}" + path.write_text(text) + print(f"{path}: pinned {args.image}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.github/workflows/lmcache-rocm-wheel.yaml b/.github/workflows/lmcache-rocm-wheel.yaml index 16f9cb80e5..928d90edea 100644 --- a/.github/workflows/lmcache-rocm-wheel.yaml +++ b/.github/workflows/lmcache-rocm-wheel.yaml @@ -26,7 +26,8 @@ name: LMCache ROCm torch 2.10 Wheel # overwritten. They are deliberately not GitHub Releases: a release needs a git # tag, and a tag reachable from main is what setuptools-scm's git describe picks # up when computing ATOM's own version. The Dockerfiles pin the image by digest -# and the wheel by sha256. source_run_id publishes the wheel an earlier +# and the wheel by sha256; after publishing, the workflow opens a draft PR that +# moves that pin to the new wheel. source_run_id publishes the wheel an earlier # run built instead of rebuilding it, while that run's artifact is still kept. on: @@ -46,6 +47,11 @@ on: type: string required: false default: '' + bump_pr: + description: "After publishing, open a draft PR that pins the new wheel in the Dockerfiles." + type: boolean + required: false + default: true runner: description: "Runner to build on. A GPU runner also runs LMCache's HIP transfer kernels." type: choice @@ -189,7 +195,7 @@ jobs: wheels=(lmcache-src/dist_rocm_torch210/*.whl) test "${#wheels[@]}" -eq 1 wheel=$(basename "${wheels[0]}") - # The publish job parses the version from this name. + # The publish job and the Dockerfiles parse the version from this name. if [ "${wheel}" != "lmcache-${VERSION}-${EXPECTED_WHEEL_TAG}.whl" ]; then echo "Unexpected wheel name ${wheel}; the Dockerfiles expect lmcache-${VERSION}-${EXPECTED_WHEEL_TAG}.whl." exit 1 @@ -465,3 +471,97 @@ jobs: echo "ARG LMCACHE_WHEEL_SHA256=${SHA256}" echo '```' } >> "$GITHUB_STEP_SUMMARY" + + bump-pr: + name: Open a PR pinning the new wheel + needs: publish + if: ${{ !cancelled() && needs.publish.result == 'success' && inputs.bump_pr }} + runs-on: ubuntu-latest + permissions: + contents: write + pull-requests: write + env: + GH_TOKEN: ${{ github.token }} + GH_REPO: ${{ github.repository }} + LMCACHE_COMMIT: ${{ inputs.lmcache_commit }} + VERSION: ${{ needs.publish.outputs.version }} + IMAGE: ${{ needs.publish.outputs.image }} + IMAGE_TAG: ${{ needs.publish.outputs.image_tag }} + WHEEL: ${{ needs.publish.outputs.wheel }} + SHA256: ${{ needs.publish.outputs.sha256 }} + SOURCE_RUN: ${{ needs.publish.outputs.source_run }} + steps: + - name: Checkout ATOM + uses: actions/checkout@v4 + with: + ref: ${{ github.event.repository.default_branch }} + + - name: Download the wheel + uses: actions/download-artifact@v4 + with: + name: ${{ needs.publish.outputs.artifact }} + run-id: ${{ needs.publish.outputs.source_run }} + github-token: ${{ github.token }} + path: ${{ runner.temp }}/new-wheel + + - name: Compare dependencies with the current pin + run: | + set -euo pipefail + # The Dockerfiles install LMCache with --no-deps and list its + # requirements by hand, so a new Requires-Dist has to be added there + # in the PR. Surface the difference instead of leaving it to the + # image build to find. + old_image="$(sed -n 's/^ARG LMCACHE_WHEEL_IMAGE="\(.*\)"$/\1/p' docker/Dockerfile)" + # A FROM scratch image has no command; create is enough for docker cp. + cid="$(docker create "${old_image}" /none)" + docker cp "${cid}:/" "${RUNNER_TEMP}/old-image" + docker rm "${cid}" >/dev/null + cp "${RUNNER_TEMP}"/old-image/*.whl "${RUNNER_TEMP}/old.whl" + requires() { + unzip -p "$1" '*.dist-info/METADATA' | sed -n 's/^Requires-Dist: //p' | sort + } + requires "${RUNNER_TEMP}/old.whl" > "${RUNNER_TEMP}/old-requires.txt" + requires "${RUNNER_TEMP}/new-wheel/${WHEEL}" > "${RUNNER_TEMP}/new-requires.txt" + diff -u "${RUNNER_TEMP}/old-requires.txt" "${RUNNER_TEMP}/new-requires.txt" \ + > "${RUNNER_TEMP}/requires.diff" || true + + - name: Open the PR + run: | + set -euo pipefail + .github/scripts/bump_lmcache_wheel_pin.py \ + --image "${IMAGE}" --sha256 "${SHA256}" + branch="ci/lmcache-wheel-${IMAGE_TAG#lmcache-}" + git config user.name "github-actions[bot]" + git config user.email "41898282+github-actions[bot]@users.noreply.github.com" + git switch -c "${branch}" + git commit -am "build(docker): pin LMCache ${VERSION} wheel" \ + -m "Built from LMCache/LMCache@${LMCACHE_COMMIT} by ${GITHUB_SERVER_URL}/${GH_REPO}/actions/runs/${SOURCE_RUN}." + git push origin "${branch}" + { + echo "Pins the LMCache wheel image \`${IMAGE}\` in docker/Dockerfile and docker/atom_release.dockerfile." + echo + echo "- LMCache commit: [${LMCACHE_COMMIT}](https://github.com/LMCache/LMCache/commit/${LMCACHE_COMMIT})" + echo "- Version: \`${VERSION}\`" + echo "- sha256: \`${SHA256}\`" + echo "- Built by: ${GITHUB_SERVER_URL}/${GH_REPO}/actions/runs/${SOURCE_RUN}" + echo + echo "## Requires-Dist changes" + echo + if [[ -s "${RUNNER_TEMP}/requires.diff" ]]; then + echo "The Dockerfiles install the wheel with \`--no-deps\`; add new requirements to the \`pip install\` before it." + echo + echo '```diff' + tail -n +3 "${RUNNER_TEMP}/requires.diff" + echo '```' + else + echo "None." + fi + echo + echo "Opened as a draft by the LMCache wheel workflow. PRs created with the workflow token do not start CI; mark it ready for review to run the checks." + } > "${RUNNER_TEMP}/pr.md" + pr_url="$(gh pr create --draft \ + --base "${{ github.event.repository.default_branch }}" \ + --head "${branch}" \ + --title "build(docker): pin LMCache ${VERSION} wheel" \ + --body-file "${RUNNER_TEMP}/pr.md")" + echo "- **Bump PR**: ${pr_url}" >> "$GITHUB_STEP_SUMMARY" diff --git a/docker/Dockerfile b/docker/Dockerfile index 3ae8887c47..0005f7cc3e 100644 --- a/docker/Dockerfile +++ b/docker/Dockerfile @@ -3,6 +3,9 @@ ARG BASE_IMAGE="rocm/pytorch:latest" ARG OOT_BASE_IMAGE="rocm/atom-dev:latest" ARG SGLANG_BASE_IMAGE="rocm/atom-dev:latest" ARG GPU_ARCH="gfx942;gfx950" +# LMCache wheel image (FROM scratch, one wheel at /), pinned by digest. See the +# LMCache section of atom_image. +ARG LMCACHE_WHEEL_IMAGE="rocm/atom-dev:lmcache-v0.5.6.dev98-g05fc77a0-rocm-torch210@sha256:d3cfe74f42d78a188992cae98efbe23053610216e7b247632d6be772e9d465d6" # OOT image extends an ATOM base image that already contains atom/aiter/mori. FROM ${OOT_BASE_IMAGE} AS atom_oot @@ -400,6 +403,9 @@ RUN echo "========== [Parallel] Building Aiter ==========" && \ MAX_JOBS=$MAX_JOBS PREBUILD_KERNELS=$PREBUILD_KERNELS \ GPU_ARCHS=$GPU_ARCH_LIST python3 setup.py develop +# The LMCache wheel, published by .github/workflows/lmcache-rocm-wheel.yaml. +FROM ${LMCACHE_WHEEL_IMAGE} AS lmcache_wheel + # -------------------------------------------------------------------- # Stage 3: Final merge — collect all build artifacts + install MORI/ATOM # -------------------------------------------------------------------- @@ -546,23 +552,34 @@ RUN echo "========== Install atomesh binary ==========" && \ atomesh --version # ========== LMCache (ROCm 7.2.4 / torch 2.10) for KV offload ========== -# Install the official wheel built for the image's exact PyTorch ABI. Keep -# --no-deps so pip cannot replace the preinstalled ROCm torch stack. -ARG LMCACHE_WHEEL_NAME=lmcache-0.5.5rc3+rocm7.2.4.torch2.10.git3d3aa833.cxx11abi1-cp312-cp312-manylinux_2_39_x86_64.whl -ARG LMCACHE_WHEEL_URL=https://github.com/LMCache/LMCache/releases/download/v0.5.5rc3-rocm-torch210/lmcache-0.5.5rc3%2Brocm7.2.4.torch2.10.git3d3aa833.cxx11abi1-cp312-cp312-manylinux_2_39_x86_64.whl -ARG LMCACHE_WHEEL_SHA256=06cda2fef1c2cf3926ffa59c4ba13b6029c6e40d3fba6db2bc2e7350a29c280f +# Install a wheel built for the image's exact PyTorch ABI. Keep --no-deps so +# pip cannot replace the preinstalled ROCm torch stack. The wheel comes from +# LMCACHE_WHEEL_IMAGE, a FROM scratch image holding only the wheel, which +# .github/workflows/lmcache-rocm-wheel.yaml pushes to Docker Hub; the same +# workflow opens the PR that moves the pin. The expected lmcache.__version__ +# follows from the wheel name; the ABI suffix and wheel tag are the workflow's +# ROCM_TORCH210_LOCAL_VERSION and EXPECTED_WHEEL_TAG, and change only with the +# image's torch. +ARG LMCACHE_WHEEL_SHA256=a5fe8f3f5b9dee602ac7d11241f65a1640cd3d26c101f2d1d0e0d8aee88b7aab +COPY --from=lmcache_wheel / /tmp/lmcache-wheel/ RUN echo "========== [ATOM] Install LMCache ROCm torch 2.10 wheel ==========" && \ - curl -fL "${LMCACHE_WHEEL_URL}" -o "/tmp/${LMCACHE_WHEEL_NAME}" && \ - echo "${LMCACHE_WHEEL_SHA256} /tmp/${LMCACHE_WHEEL_NAME}" | sha256sum -c - && \ + set -- /tmp/lmcache-wheel/*.whl && \ + { [ "$#" -eq 1 ] && [ -f "$1" ] || { echo "Expected one wheel in LMCACHE_WHEEL_IMAGE, found: $*"; exit 1; }; } && \ + lmcache_wheel="$1" && \ + lmcache_version="$(basename "${lmcache_wheel}" | sed -nE \ + 's/^lmcache-([^-]+\.rocm7\.2\.4\.torch2\.10\.git3d3aa833\.cxx11abi1)-cp312-cp312-manylinux_2_39_x86_64\.whl$/\1/p')" && \ + { [ -n "${lmcache_version}" ] || { echo "Unexpected LMCache wheel ${lmcache_wheel}"; exit 1; }; } && \ + echo "${LMCACHE_WHEEL_SHA256} ${lmcache_wheel}" | sha256sum -c - && \ "${VENV_PYTHON}" -m pip install \ prometheus_client==0.25.0 aiofile==3.11.1 aiofiles caio==0.9.25 \ blake3 redis sortedcontainers pyzmq cupy-rocm-7-0 \ cachetools cryptography numba openai \ opentelemetry-api==1.40.0 opentelemetry-sdk==1.40.0 \ opentelemetry-exporter-otlp==1.40.0 \ - opentelemetry-exporter-prometheus==0.61b0 && \ - "${VENV_PYTHON}" -m pip install --no-deps "/tmp/${LMCACHE_WHEEL_NAME}" && \ - rm -f "/tmp/${LMCACHE_WHEEL_NAME}" && \ + opentelemetry-exporter-prometheus==0.61b0 \ + "grpcio>=1.78.0" "protobuf>=6.31.1,<7" && \ + "${VENV_PYTHON}" -m pip install --no-deps "${lmcache_wheel}" && \ + rm -rf /tmp/lmcache-wheel && \ "${VENV_PYTHON}" -c "import torch; torch.cuda.is_available = lambda: True; import lmcache, lmcache.cuda_ops, lmcache.lmcache_native; \ from lmcache.v1.cache_engine import LMCacheEngineBuilder; \ from lmcache.v1.memory_management import MemoryFormat; \ @@ -574,7 +591,7 @@ from lmcache.utils import EngineType; \ from lmcache.v1.multiprocess.futures import DeviceMessagingFuture; \ from lmcache.v1.multiprocess.group_view import EngineGroupInfo; \ assert 'rocm' in torch.__version__, torch.__version__; \ -assert lmcache.__version__.startswith('0.5.5rc3+rocm7.2.4.torch2.10'), lmcache.__version__; \ +assert lmcache.__version__ == '${lmcache_version}', lmcache.__version__; \ assert lmcache.cuda_ops.__file__.endswith('.so'), lmcache.cuda_ops.__file__; \ assert lmcache.lmcache_native.__file__.endswith('.so'), lmcache.lmcache_native.__file__; \ assert hasattr(lmcache.cuda_ops, 'execute_object_group_transfer'), 'cuda_ops extension is incomplete'; \ diff --git a/docker/atom_release.dockerfile b/docker/atom_release.dockerfile index fc7efe0517..bef746a45a 100644 --- a/docker/atom_release.dockerfile +++ b/docker/atom_release.dockerfile @@ -2,6 +2,9 @@ # The digest prevents this historical tag from being moved underneath us. ARG BASE_IMAGE="rocm/pytorch:rocm7.2.4_ubuntu24.04_py3.12_pytorch_release_2.10.0@sha256:4449f856653602317e4101a76fce599c7fcd58ccec2e539951fce5f73083179e" ARG GPU_ARCH="gfx942;gfx950" +# LMCache wheel image (FROM scratch, one wheel at /), pinned by digest. See the +# LMCache section of atom_image. +ARG LMCACHE_WHEEL_IMAGE="rocm/atom-dev:lmcache-v0.5.6.dev98-g05fc77a0-rocm-torch210@sha256:d3cfe74f42d78a188992cae98efbe23053610216e7b247632d6be772e9d465d6" # ROCm 10 flavor: pass --build-arg BASE_IMAGE=rocm10-base to build the whole # image on the pip-installed ROCm 10 SDK below instead of the rocm/pytorch # apt image. All ROCm 10 component versions are ARGs so the 10.1 tracking @@ -430,6 +433,9 @@ else: print("torch_utils.py already carries the torch.Stream branch (upstream fix landed)") PY +# The LMCache wheel, published by .github/workflows/lmcache-rocm-wheel.yaml. +FROM ${LMCACHE_WHEEL_IMAGE} AS lmcache_wheel + # -------------------------------------------------------------------- # Stage 3: Final merge — collect all build artifacts + install MORI/ATOM # -------------------------------------------------------------------- @@ -637,11 +643,16 @@ RUN echo "========== Install atomesh binary ==========" && \ atomesh --version # ========== LMCache (ROCm 7.2.4 / torch 2.10) for KV offload ========== -# Install the official wheel built for the image's exact PyTorch ABI. Keep -# --no-deps so pip cannot replace the preinstalled ROCm torch stack. -ARG LMCACHE_WHEEL_NAME=lmcache-0.5.5rc3+rocm7.2.4.torch2.10.git3d3aa833.cxx11abi1-cp312-cp312-manylinux_2_39_x86_64.whl -ARG LMCACHE_WHEEL_URL=https://github.com/LMCache/LMCache/releases/download/v0.5.5rc3-rocm-torch210/lmcache-0.5.5rc3%2Brocm7.2.4.torch2.10.git3d3aa833.cxx11abi1-cp312-cp312-manylinux_2_39_x86_64.whl -ARG LMCACHE_WHEEL_SHA256=06cda2fef1c2cf3926ffa59c4ba13b6029c6e40d3fba6db2bc2e7350a29c280f +# Install a wheel built for the image's exact PyTorch ABI. Keep --no-deps so +# pip cannot replace the preinstalled ROCm torch stack. The wheel comes from +# LMCACHE_WHEEL_IMAGE, a FROM scratch image holding only the wheel, which +# .github/workflows/lmcache-rocm-wheel.yaml pushes to Docker Hub; the same +# workflow opens the PR that moves the pin. The expected lmcache.__version__ +# follows from the wheel name; the ABI suffix and wheel tag are the workflow's +# ROCM_TORCH210_LOCAL_VERSION and EXPECTED_WHEEL_TAG, and change only with the +# image's torch. +ARG LMCACHE_WHEEL_SHA256=a5fe8f3f5b9dee602ac7d11241f65a1640cd3d26c101f2d1d0e0d8aee88b7aab +COPY --from=lmcache_wheel / /tmp/lmcache-wheel/ # Docker builds do not expose a GPU, so LMCache's torch.cuda.is_available() # backend predicate is overridden only in the validation process below. # Two install paths, because the published wheel targets one ABI: its filename @@ -657,17 +668,23 @@ ARG LMCACHE_WHEEL_SHA256=06cda2fef1c2cf3926ffa59c4ba13b6029c6e40d3fba6db2bc2e735 ARG LMCACHE_TAG=v0.4.5 RUN if [ -z "${ROCM_HOME}" ]; then \ echo "========== [ATOM] Install LMCache ROCm torch 2.10 wheel ==========" && \ - curl -fL "${LMCACHE_WHEEL_URL}" -o "/tmp/${LMCACHE_WHEEL_NAME}" && \ - echo "${LMCACHE_WHEEL_SHA256} /tmp/${LMCACHE_WHEEL_NAME}" | sha256sum -c - && \ + set -- /tmp/lmcache-wheel/*.whl && \ + { [ "$#" -eq 1 ] && [ -f "$1" ] || { echo "Expected one wheel in LMCACHE_WHEEL_IMAGE, found: $*"; exit 1; }; } && \ + lmcache_wheel="$1" && \ + lmcache_version="$(basename "${lmcache_wheel}" | sed -nE \ + 's/^lmcache-([^-]+\.rocm7\.2\.4\.torch2\.10\.git3d3aa833\.cxx11abi1)-cp312-cp312-manylinux_2_39_x86_64\.whl$/\1/p')" && \ + { [ -n "${lmcache_version}" ] || { echo "Unexpected LMCache wheel ${lmcache_wheel}"; exit 1; }; } && \ + echo "${LMCACHE_WHEEL_SHA256} ${lmcache_wheel}" | sha256sum -c - && \ "${VENV_PYTHON}" -m pip install \ prometheus_client==0.25.0 aiofile==3.11.1 aiofiles caio==0.9.25 \ blake3 redis sortedcontainers pyzmq cupy-rocm-7-0 \ cachetools cryptography numba openai py-cpuinfo \ opentelemetry-api==1.40.0 opentelemetry-sdk==1.40.0 \ opentelemetry-exporter-otlp==1.40.0 \ - opentelemetry-exporter-prometheus==0.61b0 && \ - "${VENV_PYTHON}" -m pip install --no-deps "/tmp/${LMCACHE_WHEEL_NAME}" && \ - rm -f "/tmp/${LMCACHE_WHEEL_NAME}" && \ + opentelemetry-exporter-prometheus==0.61b0 \ + "grpcio>=1.78.0" "protobuf>=6.31.1,<7" && \ + "${VENV_PYTHON}" -m pip install --no-deps "${lmcache_wheel}" && \ + rm -rf /tmp/lmcache-wheel && \ "${VENV_PYTHON}" -c "import torch; torch.cuda.is_available = lambda: True; import lmcache, lmcache.cuda_ops, lmcache.lmcache_native; \ from lmcache.v1.cache_engine import LMCacheEngineBuilder; \ from lmcache.v1.memory_management import MemoryFormat; \ @@ -679,7 +696,7 @@ RUN if [ -z "${ROCM_HOME}" ]; then \ from lmcache.v1.multiprocess.futures import DeviceMessagingFuture; \ from lmcache.v1.multiprocess.group_view import EngineGroupInfo; \ assert 'rocm' in torch.__version__, torch.__version__; \ - assert lmcache.__version__.startswith('0.5.5rc3+rocm7.2.4.torch2.10'), lmcache.__version__; \ + assert lmcache.__version__ == '${lmcache_version}', lmcache.__version__; \ assert lmcache.cuda_ops.__file__.endswith('.so'), lmcache.cuda_ops.__file__; \ assert lmcache.lmcache_native.__file__.endswith('.so'), lmcache.lmcache_native.__file__; \ assert hasattr(lmcache.cuda_ops, 'execute_object_group_transfer'), 'cuda_ops extension is incomplete'; \ @@ -691,6 +708,7 @@ RUN if [ -z "${ROCM_HOME}" ]; then \ print('OK: lmcache', lmcache.__version__, 'HIP cuda_ops; torch', torch.__version__)" ; \ else \ echo "========== [ATOM] LMCache HIP c_ops (${LMCACHE_TAG}, arch=${PYTORCH_ROCM_ARCH}) ==========" && \ + rm -rf /tmp/lmcache-wheel && \ git clone https://github.com/LMCache/LMCache.git /opt/LMCache && \ cd /opt/LMCache && git checkout ${LMCACHE_TAG} && \ "${VENV_PYTHON}" -m pip install -r requirements/build.txt && \ From d843668299d41be908de7686e76d9ca2d92f8ac4 Mon Sep 17 00:00:00 2001 From: Honglie Yi Date: Thu, 24 Sep 2026 14:40:53 +0000 Subject: [PATCH 5/6] revert tag filters for setuptools-scm and the release version The LMCache wheel workflow no longer creates tags, and the only remaining tag creators are ATOM's own vX.Y.Z releases, so drop the --match 'v[0-9]*' filters from pyproject.toml and atom-release.yaml. Co-Authored-By: Claude Opus 5.5 (1M context) --- .github/workflows/atom-release.yaml | 4 +--- pyproject.toml | 1 - 2 files changed, 1 insertion(+), 4 deletions(-) diff --git a/.github/workflows/atom-release.yaml b/.github/workflows/atom-release.yaml index 3d7cbfb649..0a179a7a3e 100644 --- a/.github/workflows/atom-release.yaml +++ b/.github/workflows/atom-release.yaml @@ -260,9 +260,7 @@ jobs: # the runner host, whose python is whatever the host image ships, # and subprocess.run(capture_output=...) needs 3.7. The python below # only normalises a string, so it works on any python3. - # Only ATOM's own vX.Y.Z tags; others (e.g. lmcache-v...) would become - # the base version. Same filter as setuptools-scm in pyproject.toml. - DESCRIBE=$(git describe --tags --long --dirty --match 'v[0-9]*') + DESCRIBE=$(git describe --tags --long --dirty) VERSIONS=$(ROCM_VER="${ROCM_VER}" DATE_STAMP="${DATE_STAMP}" DESCRIBE="${DESCRIBE}" python3 - <<'PYEOF' import os, re diff --git a/pyproject.toml b/pyproject.toml index a581519095..98971cd71f 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -28,7 +28,6 @@ Issues = "https://github.com/ROCm/ATOM/issues" write_to = "atom/_version.py" write_to_template = "__version__ = '{version}'\n" fallback_version = "0.1.0" -git_describe_command = ["git", "describe", "--dirty", "--tags", "--long", "--match", "v[0-9]*"] [tool.setuptools.packages.find] where = ["."] From c1f046ab09c0821bdba09bb7e4754d1e41f2a33b Mon Sep 17 00:00:00 2001 From: Honglie Yi Date: Thu, 24 Sep 2026 14:47:25 +0000 Subject: [PATCH 6/6] ci(lmcache): address second review of the wheel image path - bump-pr gets actions: read; download-artifact with run-id reads the artifact through the Actions API and failed without it. - bump-pr can be re-run: it force-pushes its own per-wheel branch and reuses an open PR instead of failing on the existing branch. - source_run_id: the source run's build job name carries the full commit it was dispatched with; require it to equal lmcache_commit (the wheel name only has 8 hex digits). - Read the pushed digest from the registry (buildx imagetools) rather than RepoDigests, which differs under the containerd image store. - The summary only names the build image for a build in this run. - Dockerfiles: bind-mount the wheel stage in the install RUN instead of COPY + rm, so no layer keeps the wheel. Co-Authored-By: Claude Opus 5.5 (1M context) --- .github/workflows/lmcache-rocm-wheel.yaml | 42 ++++++++++++++++------- docker/Dockerfile | 5 ++- docker/atom_release.dockerfile | 6 ++-- 3 files changed, 34 insertions(+), 19 deletions(-) diff --git a/.github/workflows/lmcache-rocm-wheel.yaml b/.github/workflows/lmcache-rocm-wheel.yaml index 928d90edea..ce061a0646 100644 --- a/.github/workflows/lmcache-rocm-wheel.yaml +++ b/.github/workflows/lmcache-rocm-wheel.yaml @@ -339,12 +339,18 @@ jobs: exit 1 fi attempt="${artifact##*-}" - build=$(gh api "${api}/attempts/${attempt}/jobs?per_page=100" \ - --jq '[.jobs[] | select(.name | startswith("Build and validate"))][0].conclusion // empty') + IFS=$'\t' read -r build_name build < <(gh api "${api}/attempts/${attempt}/jobs?per_page=100" \ + --jq '[.jobs[] | select(.name | startswith("Build and validate"))][0] | "\(.name)\t\(.conclusion)"') if [ "${build}" != "success" ]; then echo "Run ${SOURCE_RUN_ID} attempt ${attempt}: build job concluded '${build}', not success." exit 1 fi + # The wheel name only carries 8 hex digits; the job name has the full + # commit the source run was dispatched with. + if [ "${build_name}" != "Build and validate LMCache ${LMCACHE_COMMIT}" ]; then + echo "Run ${SOURCE_RUN_ID} built '${build_name#Build and validate LMCache }', not ${LMCACHE_COMMIT}." + exit 1 + fi echo "run_id=${SOURCE_RUN_ID}" >> "$GITHUB_OUTPUT" echo "artifact=${artifact}" >> "$GITHUB_OUTPUT" echo "Publishing ${artifact} from ${branch}" @@ -440,8 +446,9 @@ jobs: --label "com.amd.atom.lmcache.build-run=${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${SOURCE_RUN}" \ -t "${ref}" "${RUNNER_TEMP}/wheel-image" .github/scripts/docker_push_retry.sh "${ref}" - digest=$(docker inspect --format '{{range .RepoDigests}}{{println .}}{{end}}' "${ref}" \ - | sed -n "s|^${IMAGE_REPO}@||p" | head -1) + # The registry's answer does not depend on the local image store + # (RepoDigests reads differently under containerd). + digest=$(docker buildx imagetools inspect "${ref}" --format '{{.Manifest.Digest}}') if ! [[ "${digest}" =~ ^sha256:[0-9a-f]{64}$ ]]; then echo "Could not read the pushed digest of ${ref}." exit 1 @@ -462,7 +469,11 @@ jobs: echo "- **Version**: \`${VERSION}\`" echo "- **Wheel**: \`${WHEEL}\`" echo "- **Built by**: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${SOURCE_RUN}" - echo "- **Build image**: \`${ROCM_PYTORCH_IMAGE}\`" + if [ "${SOURCE_RUN}" = "${GITHUB_RUN_ID}" ]; then + echo "- **Build image**: \`${ROCM_PYTORCH_IMAGE}\`" + else + echo "- **Build image**: see the build run's summary" + fi echo echo "Pin in docker/Dockerfile and docker/atom_release.dockerfile:" echo @@ -480,6 +491,8 @@ jobs: permissions: contents: write pull-requests: write + # download-artifact with run-id reads through the Actions API. + actions: read env: GH_TOKEN: ${{ github.token }} GH_REPO: ${{ github.repository }} @@ -533,10 +546,12 @@ jobs: branch="ci/lmcache-wheel-${IMAGE_TAG#lmcache-}" git config user.name "github-actions[bot]" git config user.email "41898282+github-actions[bot]@users.noreply.github.com" - git switch -c "${branch}" + git switch -C "${branch}" git commit -am "build(docker): pin LMCache ${VERSION} wheel" \ -m "Built from LMCache/LMCache@${LMCACHE_COMMIT} by ${GITHUB_SERVER_URL}/${GH_REPO}/actions/runs/${SOURCE_RUN}." - git push origin "${branch}" + # The branch name is fixed per wheel and owned by this job, so a re-run + # after a failure below replaces what the previous attempt pushed. + git push --force origin "${branch}" { echo "Pins the LMCache wheel image \`${IMAGE}\` in docker/Dockerfile and docker/atom_release.dockerfile." echo @@ -559,9 +574,12 @@ jobs: echo echo "Opened as a draft by the LMCache wheel workflow. PRs created with the workflow token do not start CI; mark it ready for review to run the checks." } > "${RUNNER_TEMP}/pr.md" - pr_url="$(gh pr create --draft \ - --base "${{ github.event.repository.default_branch }}" \ - --head "${branch}" \ - --title "build(docker): pin LMCache ${VERSION} wheel" \ - --body-file "${RUNNER_TEMP}/pr.md")" + pr_url="$(gh pr list --head "${branch}" --state open --json url --jq '.[0].url // empty')" + if [ -z "${pr_url}" ]; then + pr_url="$(gh pr create --draft \ + --base "${{ github.event.repository.default_branch }}" \ + --head "${branch}" \ + --title "build(docker): pin LMCache ${VERSION} wheel" \ + --body-file "${RUNNER_TEMP}/pr.md")" + fi echo "- **Bump PR**: ${pr_url}" >> "$GITHUB_STEP_SUMMARY" diff --git a/docker/Dockerfile b/docker/Dockerfile index 0005f7cc3e..33944f0a4f 100644 --- a/docker/Dockerfile +++ b/docker/Dockerfile @@ -561,8 +561,8 @@ RUN echo "========== Install atomesh binary ==========" && \ # ROCM_TORCH210_LOCAL_VERSION and EXPECTED_WHEEL_TAG, and change only with the # image's torch. ARG LMCACHE_WHEEL_SHA256=a5fe8f3f5b9dee602ac7d11241f65a1640cd3d26c101f2d1d0e0d8aee88b7aab -COPY --from=lmcache_wheel / /tmp/lmcache-wheel/ -RUN echo "========== [ATOM] Install LMCache ROCm torch 2.10 wheel ==========" && \ +RUN --mount=type=bind,from=lmcache_wheel,target=/tmp/lmcache-wheel \ + echo "========== [ATOM] Install LMCache ROCm torch 2.10 wheel ==========" && \ set -- /tmp/lmcache-wheel/*.whl && \ { [ "$#" -eq 1 ] && [ -f "$1" ] || { echo "Expected one wheel in LMCACHE_WHEEL_IMAGE, found: $*"; exit 1; }; } && \ lmcache_wheel="$1" && \ @@ -579,7 +579,6 @@ RUN echo "========== [ATOM] Install LMCache ROCm torch 2.10 wheel ==========" && opentelemetry-exporter-prometheus==0.61b0 \ "grpcio>=1.78.0" "protobuf>=6.31.1,<7" && \ "${VENV_PYTHON}" -m pip install --no-deps "${lmcache_wheel}" && \ - rm -rf /tmp/lmcache-wheel && \ "${VENV_PYTHON}" -c "import torch; torch.cuda.is_available = lambda: True; import lmcache, lmcache.cuda_ops, lmcache.lmcache_native; \ from lmcache.v1.cache_engine import LMCacheEngineBuilder; \ from lmcache.v1.memory_management import MemoryFormat; \ diff --git a/docker/atom_release.dockerfile b/docker/atom_release.dockerfile index bef746a45a..9f2f2542f2 100644 --- a/docker/atom_release.dockerfile +++ b/docker/atom_release.dockerfile @@ -652,7 +652,6 @@ RUN echo "========== Install atomesh binary ==========" && \ # ROCM_TORCH210_LOCAL_VERSION and EXPECTED_WHEEL_TAG, and change only with the # image's torch. ARG LMCACHE_WHEEL_SHA256=a5fe8f3f5b9dee602ac7d11241f65a1640cd3d26c101f2d1d0e0d8aee88b7aab -COPY --from=lmcache_wheel / /tmp/lmcache-wheel/ # Docker builds do not expose a GPU, so LMCache's torch.cuda.is_available() # backend predicate is overridden only in the validation process below. # Two install paths, because the published wheel targets one ABI: its filename @@ -666,7 +665,8 @@ COPY --from=lmcache_wheel / /tmp/lmcache-wheel/ # tells the two apart. Collapse this back to one path once a matching wheel # exists. ARG LMCACHE_TAG=v0.4.5 -RUN if [ -z "${ROCM_HOME}" ]; then \ +RUN --mount=type=bind,from=lmcache_wheel,target=/tmp/lmcache-wheel \ + if [ -z "${ROCM_HOME}" ]; then \ echo "========== [ATOM] Install LMCache ROCm torch 2.10 wheel ==========" && \ set -- /tmp/lmcache-wheel/*.whl && \ { [ "$#" -eq 1 ] && [ -f "$1" ] || { echo "Expected one wheel in LMCACHE_WHEEL_IMAGE, found: $*"; exit 1; }; } && \ @@ -684,7 +684,6 @@ RUN if [ -z "${ROCM_HOME}" ]; then \ opentelemetry-exporter-prometheus==0.61b0 \ "grpcio>=1.78.0" "protobuf>=6.31.1,<7" && \ "${VENV_PYTHON}" -m pip install --no-deps "${lmcache_wheel}" && \ - rm -rf /tmp/lmcache-wheel && \ "${VENV_PYTHON}" -c "import torch; torch.cuda.is_available = lambda: True; import lmcache, lmcache.cuda_ops, lmcache.lmcache_native; \ from lmcache.v1.cache_engine import LMCacheEngineBuilder; \ from lmcache.v1.memory_management import MemoryFormat; \ @@ -708,7 +707,6 @@ RUN if [ -z "${ROCM_HOME}" ]; then \ print('OK: lmcache', lmcache.__version__, 'HIP cuda_ops; torch', torch.__version__)" ; \ else \ echo "========== [ATOM] LMCache HIP c_ops (${LMCACHE_TAG}, arch=${PYTORCH_ROCM_ARCH}) ==========" && \ - rm -rf /tmp/lmcache-wheel && \ git clone https://github.com/LMCache/LMCache.git /opt/LMCache && \ cd /opt/LMCache && git checkout ${LMCACHE_TAG} && \ "${VENV_PYTHON}" -m pip install -r requirements/build.txt && \