diff --git a/.github/scripts/dsw-swe-verified/dispatch-release-benchmark.sh b/.github/scripts/dsw-swe-verified/dispatch-release-benchmark.sh index 256c8629cbe..8553a349ac7 100755 --- a/.github/scripts/dsw-swe-verified/dispatch-release-benchmark.sh +++ b/.github/scripts/dsw-swe-verified/dispatch-release-benchmark.sh @@ -6,6 +6,7 @@ set -euo pipefail : "${QWEN_REF:?QWEN_REF is required}" : "${QWEN_COMMIT:?QWEN_COMMIT is required}" : "${INSTANCE_LIMIT:?INSTANCE_LIMIT is required}" +: "${TERMINAL_BENCH_LIMIT:=89}" : "${BENCHMARK_IDEMPOTENCY_KEY:?BENCHMARK_IDEMPOTENCY_KEY is required}" : "${GITHUB_REPOSITORY:?GITHUB_REPOSITORY is required}" @@ -14,19 +15,44 @@ pool_root="${DSW_POOL_ROOT:-/mnt/workspace/qwen-benchmark-pool}" pool_bin="${POOL_BIN:-${pool_root}/venv/bin/qwen-benchmark-pool}" python_bin="${POOL_PYTHON:-${pool_root}/venv/bin/python}" dataset_root="${SWE_VERIFIED_DATASET_ROOT:-${pool_root}/datasets/swe-bench-verified}" +tb_task_cache="${TERMINAL_BENCH_TASK_CACHE:-/mnt/workspace/qwen-benchmark-eas-poc/cache/terminal-bench-2.0-harbor-tasks.tar.gz}" agent_cache_root="${QWEN_BENCHMARK_CACHE_ROOT:-/mnt/workspace/qwen-benchmark-cache}" agent_cache_prepare="${pool_root}/service/deploy/prepare-agent-cache.py" database_url="${BENCHMARK_POOL_DATABASE_URL:-postgresql://qwen_benchmark@127.0.0.1:55432/qwen_benchmark_dsw_release_v1}" +execution_backend="${BENCHMARK_EXECUTION_BACKEND:-harbor}" +model_env_file="${MODEL_ENV_FILE:-/mnt/workspace/qwen-benchmark-eas-poc/config/model.env}" +if [[ "${execution_backend}" == "eas-harbor" && -s "${model_env_file}" ]]; then + set -a + # This file contains only OPENAI_BASE_URL and OPENAI_MODEL; the API key is + # deliberately stored in a separate 0600 file consumed by the Executor. + source "${model_env_file}" + set +a +fi model_name="${OPENAI_MODEL:-qwen3.7-max}" dataset_revision="2" max_attempts="${BENCHMARK_MAX_ATTEMPTS:-4}" retry_backoff_seconds="${BENCHMARK_RETRY_BACKOFF_SECONDS:-60}" +eas_template_manifest="${EAS_TEMPLATE_MANIFEST:-${pool_root}/deploy/eas/templates.json}" +acr_image_manifest="${ACR_IMAGE_MANIFEST:-/mnt/workspace/qwen-benchmark-eas-poc/state/acr-manifest-c104f840.json}" +acr_image_state_dir="${ACR_IMAGE_STATE_DIR:-/mnt/data/qwen-benchmark/acr-prewarm/c104f840/state}" +eas_agent_cache_prepare="${EAS_AGENT_CACHE_PREPARE:-/mnt/workspace/qwen-benchmark-eas-poc/deploy/prepare-eas-agent-cache.py}" +eas_runtime_uploader="${EAS_RUNTIME_UPLOADER:-/mnt/workspace/qwen-benchmark-eas-poc/deploy/acr-upload-runtime-artifact.py}" +eas_node_bin="${EAS_NODE_BIN:-/mnt/workspace/qwen-benchmark-cache/node/runtime/bin}" +eas_docker_config="${EAS_DOCKER_CONFIG:-/mnt/workspace/.docker/config.json}" output_root="${GITHUB_WORKSPACE:-$(pwd)}/benchmark-output" if [[ ! "${INSTANCE_LIMIT}" =~ ^[0-9]+$ ]] || (( INSTANCE_LIMIT < 1 || INSTANCE_LIMIT > 500 )); then echo "INSTANCE_LIMIT must be between 1 and 500" >&2 exit 2 fi +if [[ ! "${TERMINAL_BENCH_LIMIT}" =~ ^[0-9]+$ ]] || (( TERMINAL_BENCH_LIMIT != 1 && TERMINAL_BENCH_LIMIT != 89 )); then + echo "TERMINAL_BENCH_LIMIT must be 1 or 89" >&2 + exit 2 +fi +if [[ -n "${TERMINAL_BENCH_INSTANCE_ID:-}" && "${TERMINAL_BENCH_LIMIT}" != "1" ]]; then + echo "TERMINAL_BENCH_INSTANCE_ID requires TERMINAL_BENCH_LIMIT=1" >&2 + exit 2 +fi if [[ ! "${max_attempts}" =~ ^[0-9]+$ ]] || (( max_attempts < 1 || max_attempts > 8 )); then echo "BENCHMARK_MAX_ATTEMPTS must be between 1 and 8" >&2 exit 2 @@ -35,29 +61,51 @@ if [[ ! "${retry_backoff_seconds}" =~ ^[0-9]+$ ]]; then echo "BENCHMARK_RETRY_BACKOFF_SECONDS must be a non-negative integer" >&2 exit 2 fi -for required_path in "${pool_bin}" "${python_bin}" "${dataset_root}" "${agent_cache_prepare}"; do +if [[ "${execution_backend}" != "harbor" && "${execution_backend}" != "eas-harbor" && "${execution_backend}" != "eas-smoke" ]]; then + echo "BENCHMARK_EXECUTION_BACKEND must be harbor, eas-harbor, or eas-smoke" >&2 + exit 2 +fi +required_paths=("${pool_bin}" "${python_bin}" "${dataset_root}" "${tb_task_cache}") +if [[ "${execution_backend}" == "harbor" ]]; then + required_paths+=("${agent_cache_prepare}") +elif [[ "${execution_backend}" == "eas-smoke" ]]; then + required_paths+=("${eas_template_manifest}") +elif [[ "${execution_backend}" == "eas-harbor" ]]; then + required_paths+=( + "${acr_image_manifest}" + "${acr_image_state_dir}" + "${eas_agent_cache_prepare}" + "${eas_runtime_uploader}" + "${eas_node_bin}/node" + "${eas_node_bin}/npm" + "${eas_docker_config}" + ) +fi +for required_path in "${required_paths[@]}"; do if [[ ! -e "${required_path}" ]]; then echo "Required DSW resource is missing: ${required_path}" >&2 exit 2 fi done -agent_cache_dirs=( - "${agent_cache_root}" - "${agent_cache_root}/node" - "${agent_cache_root}/nvm" - "${agent_cache_root}/npm" - "${agent_cache_root}/qwen-code" -) -for cache_dir in "${agent_cache_dirs[@]}"; do - if [[ ! -d "${cache_dir}" ]]; then - echo "::error::Benchmark cache directory is missing: ${cache_dir}" >&2 - exit 2 - fi - if [[ ! -w "${cache_dir}" ]]; then - echo "::error::Benchmark cache directory is not writable by $(id -un): ${cache_dir}" >&2 - exit 2 - fi -done +if [[ "${execution_backend}" == "harbor" ]]; then + agent_cache_dirs=( + "${agent_cache_root}" + "${agent_cache_root}/node" + "${agent_cache_root}/nvm" + "${agent_cache_root}/npm" + "${agent_cache_root}/qwen-code" + ) + for cache_dir in "${agent_cache_dirs[@]}"; do + if [[ ! -d "${cache_dir}" ]]; then + echo "::error::Benchmark cache directory is missing: ${cache_dir}" >&2 + exit 2 + fi + if [[ ! -w "${cache_dir}" ]]; then + echo "::error::Benchmark cache directory is not writable by $(id -un): ${cache_dir}" >&2 + exit 2 + fi + done +fi mkdir -p "${output_root}" manifest_path="${output_root}/manifest.json" @@ -76,34 +124,57 @@ fi # tasks become claimable. This normally takes seconds on a warm DSW cache and # does not wait for the benchmark itself. qwen_version="${QWEN_REF#v}" -"${python_bin}" "${agent_cache_prepare}" \ - --cache-root "${agent_cache_root}" \ - --node-version "${QWEN_BENCHMARK_NODE_VERSION:-v22.23.1}" \ - --nvm-version "${QWEN_BENCHMARK_NVM_VERSION:-v0.40.2}" \ - --qwen-version "${qwen_version}" \ - --npm-registry "${NPM_CONFIG_REGISTRY:-https://registry.npmjs.org}" \ - > "${output_root}/agent-cache-manifest-path.txt" +if [[ "${execution_backend}" == "harbor" ]]; then + "${python_bin}" "${agent_cache_prepare}" \ + --cache-root "${agent_cache_root}" \ + --node-version "${QWEN_BENCHMARK_NODE_VERSION:-v22.23.1}" \ + --nvm-version "${QWEN_BENCHMARK_NVM_VERSION:-v0.40.2}" \ + --qwen-version "${qwen_version}" \ + --npm-registry "${NPM_CONFIG_REGISTRY:-https://registry.npmjs.org}" \ + > "${output_root}/agent-cache-manifest-path.txt" +elif [[ "${execution_backend}" == "eas-smoke" ]]; then + "${pool_bin}" validate-eas-templates \ + --task-manifest "${manifest_path}" \ + --template-manifest "${eas_template_manifest}" >/dev/null +elif [[ "${execution_backend}" == "eas-harbor" ]]; then + "${python_bin}" "${eas_agent_cache_prepare}" \ + --version "${qwen_version}" \ + --tag "qwen-code-cache-${qwen_version}-nodegzip-v2" \ + --node-bin "${eas_node_bin}" \ + --docker-config "${eas_docker_config}" \ + --uploader "${eas_runtime_uploader}" \ + --output-root "/mnt/workspace/qwen-benchmark-eas-poc/cache/agent-releases" \ + > "${output_root}/eas-agent-cache.json" +fi export BENCHMARK_POOL_DATABASE_URL="${database_url}" "${pool_bin}" init-db >/dev/null +submit_args=( + --idempotency-key "${BENCHMARK_IDEMPOTENCY_KEY}" + --suite "dsw_release_swe_verified_v1" + --dataset "swe-bench/swe-bench-verified" + --dataset-revision "${dataset_revision}" + --task-prefix "swe-bench/" + --qwen-ref "${QWEN_REF}" + --qwen-commit "${QWEN_COMMIT}" + --model "${model_name}" + --manifest "${manifest_path}" + --max-attempts "${max_attempts}" + --retry-backoff-seconds "${retry_backoff_seconds}" + --infra-failure-threshold 0 + --repository "${GITHUB_REPOSITORY}" + --release-id "${RELEASE_ID}" + --release-tag "${RELEASE_TAG}" + --github-run-url "${GITHUB_RUN_URL:-}" +) +if [[ "${execution_backend}" == "eas-harbor" ]]; then + submit_args+=( + --acr-manifest "${acr_image_manifest}" + --acr-state-dir "${acr_image_state_dir}" + ) +fi submit_json="$( - "${pool_bin}" submit \ - --idempotency-key "${BENCHMARK_IDEMPOTENCY_KEY}" \ - --suite "dsw_release_swe_verified_v1" \ - --dataset "swe-bench/swe-bench-verified" \ - --dataset-revision "${dataset_revision}" \ - --task-prefix "swe-bench/" \ - --qwen-ref "${QWEN_REF}" \ - --qwen-commit "${QWEN_COMMIT}" \ - --model "${model_name}" \ - --manifest "${manifest_path}" \ - --max-attempts "${max_attempts}" \ - --retry-backoff-seconds "${retry_backoff_seconds}" \ - --infra-failure-threshold 0 \ - --repository "${GITHUB_REPOSITORY}" \ - --release-id "${RELEASE_ID}" \ - --release-tag "${RELEASE_TAG}" \ - --github-run-url "${GITHUB_RUN_URL:-}" + "${pool_bin}" submit "${submit_args[@]}" )" run_id="$( "${python_bin}" -c ' @@ -124,12 +195,38 @@ print(run_id) ' <<< "${submit_json}" )" +# The release worker must not remain alive for either benchmark. Persist the +# exact TB 2.0 task set now; the DSW Publisher dispatches it only after the SWE +# result and trajectory bundle have been written successfully to the Release. +# SWE scoreability is independent: a published QUARANTINED result still starts +# the TB follow-up. +tb_manifest_path="${output_root}/terminal-bench-2.0-manifest.json" +tb_manifest_args=( + --archive "${tb_task_cache}" + --limit "${TERMINAL_BENCH_LIMIT}" + --output "${tb_manifest_path}" +) +if [[ -n "${TERMINAL_BENCH_INSTANCE_ID:-}" ]]; then + tb_manifest_args+=(--instance-id "${TERMINAL_BENCH_INSTANCE_ID}") +fi +"${python_bin}" "${script_root}/make-terminal-bench-manifest.py" "${tb_manifest_args[@]}" +"${pool_bin}" create-release-chain \ + --swe-run-id "${run_id}" \ + --tb-idempotency-key "${BENCHMARK_IDEMPOTENCY_KEY}-terminal-bench-2.0" \ + --tb-manifest "${tb_manifest_path}" \ + --max-attempts "${max_attempts}" \ + --retry-backoff-seconds "${retry_backoff_seconds}" \ + > "${output_root}/terminal-bench-chain.json" + jq -n \ --arg status "QUEUED" \ --arg run_id "${run_id}" \ --arg release_tag "${RELEASE_TAG}" \ --arg qwen_ref "${QWEN_REF}" \ --arg qwen_commit "${QWEN_COMMIT}" \ + --arg execution_backend "${execution_backend}" \ + --arg terminal_bench_status "PENDING_SWE_PUBLICATION" \ + --argjson terminal_bench_expected_instances "${TERMINAL_BENCH_LIMIT}" \ --argjson expected_instances "${INSTANCE_LIMIT}" \ '{ status: $status, @@ -137,6 +234,9 @@ jq -n \ release_tag: $release_tag, qwen_ref: $qwen_ref, qwen_commit: $qwen_commit, + execution_backend: $execution_backend, + terminal_bench_status: $terminal_bench_status, + terminal_bench_expected_instances: $terminal_bench_expected_instances, expected_instances: $expected_instances }' > "${output_root}/dispatch-receipt.json" diff --git a/.github/scripts/dsw-swe-verified/make-terminal-bench-manifest.py b/.github/scripts/dsw-swe-verified/make-terminal-bench-manifest.py new file mode 100755 index 00000000000..6019cb96bbf --- /dev/null +++ b/.github/scripts/dsw-swe-verified/make-terminal-bench-manifest.py @@ -0,0 +1,48 @@ +#!/usr/bin/env python3 +"""Build a frozen Terminal-Bench task manifest from the versioned ACR cache.""" +from __future__ import annotations + +import argparse +import json +import re +import tarfile +from pathlib import PurePosixPath + +parser = argparse.ArgumentParser() +parser.add_argument("--archive", required=True) +parser.add_argument("--limit", type=int, choices=(1, 89), default=89) +parser.add_argument("--instance-id") +parser.add_argument("--output", required=True) +args = parser.parse_args() + +task_names: set[str] = set() +with tarfile.open(args.archive, "r:gz") as bundle: + for member in bundle.getmembers(): + parts = PurePosixPath(member.name).parts + if ( + len(parts) >= 4 + and parts[0] == "tasks" + and parts[1] != "packages" + and parts[-1] == "instruction.md" + ): + task_names.add(parts[-2]) +if len(task_names) != 89: + raise SystemExit(f"expected 89 Terminal-Bench 2.0 tasks, found {len(task_names)}") +if args.instance_id and args.limit != 1: + raise SystemExit("--instance-id requires --limit 1") +if args.instance_id: + if args.instance_id not in task_names: + raise SystemExit(f"Unknown Terminal-Bench 2.0 task: {args.instance_id}") + selected = [args.instance_id] +else: + selected = sorted(task_names)[: args.limit] +payload = { + "schema_version": "qwen-code-terminal-bench-2.0-manifest/v1", + "dataset": "terminal-bench", + "dataset_revision": "2.0", + "expected_instances": len(selected), + "instance_ids": selected, +} +with open(args.output, "w", encoding="utf-8") as stream: + json.dump(payload, stream, indent=2) + stream.write("\n") diff --git a/.github/scripts/dsw-swe-verified/make-terminal-bench-manifest.test.mjs b/.github/scripts/dsw-swe-verified/make-terminal-bench-manifest.test.mjs new file mode 100644 index 00000000000..1db0c9a08d5 --- /dev/null +++ b/.github/scripts/dsw-swe-verified/make-terminal-bench-manifest.test.mjs @@ -0,0 +1,58 @@ +import assert from 'node:assert/strict'; +import { spawnSync } from 'node:child_process'; +import { mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { dirname, join } from 'node:path'; +import { after, before, describe, it } from 'node:test'; +import { fileURLToPath } from 'node:url'; + +const root = mkdtempSync(join(tmpdir(), 'dsw-tb-manifest-')); +const script = join(dirname(fileURLToPath(import.meta.url)), 'make-terminal-bench-manifest.py'); +let archive; + +before(() => { + const tasks = join(root, 'tasks'); + mkdirSync(tasks); + for (let i = 0; i < 89; i += 1) { + const task = join( + tasks, + `frozen-id-${String(i).padStart(2, '0')}`, + `task-${String(i).padStart(2, '0')}`, + ); + mkdirSync(task, { recursive: true }); + writeFileSync(join(task, 'instruction.md'), 'test\n'); + } + archive = join(root, 'tasks.tar.gz'); + const result = spawnSync('tar', ['-czf', archive, '-C', root, 'tasks'], { encoding: 'utf8' }); + assert.equal(result.status, 0, result.stderr); +}); + +after(() => rmSync(root, { recursive: true, force: true })); + +const run = (...args) => spawnSync('python3', [script, '--archive', archive, ...args], { encoding: 'utf8' }); + +describe('make-terminal-bench-manifest', () => { + it('selects one exact task for an end-to-end smoke', () => { + const output = join(root, 'one.json'); + const result = run('--limit', '1', '--instance-id', 'task-42', '--output', output); + assert.equal(result.status, 0, result.stderr); + const manifest = JSON.parse(readFileSync(output, 'utf8')); + assert.equal(manifest.expected_instances, 1); + assert.deepEqual(manifest.instance_ids, ['task-42']); + }); + + it('keeps all 89 tasks for a full release chain', () => { + const output = join(root, 'full.json'); + const result = run('--limit', '89', '--output', output); + assert.equal(result.status, 0, result.stderr); + const manifest = JSON.parse(readFileSync(output, 'utf8')); + assert.equal(manifest.expected_instances, 89); + assert.equal(manifest.instance_ids.length, 89); + }); + + it('rejects an unknown exact task', () => { + const result = run('--limit', '1', '--instance-id', 'missing', '--output', join(root, 'bad.json')); + assert.equal(result.status, 1); + assert.match(result.stderr, /Unknown Terminal-Bench/); + }); +}); diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 7b9fa2678f9..3457c4d251a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -50,7 +50,7 @@ env: # BOTH the github_ci_only helper step and the full-profile Test step, so a # new helper test can't be added to one path and silently dropped from the # other. - HELPER_TESTS: '.github/scripts/pr-safety-precheck.test.mjs .github/scripts/cap-release-notes.test.mjs .github/scripts/ci/classify-profile.test.mjs .github/scripts/ci/classify-pr-profile.test.mjs .github/scripts/upsert-bot-comment.test.mjs .github/scripts/ci/main-failure-signature.test.mjs .github/scripts/classify-release-notes.test.mjs .github/scripts/dsw-swe-verified/make-manifest.test.mjs .github/scripts/resolve-sandbox-image.test.mjs .github/scripts/web-shell-visuals-publish.test.mjs .github/scripts/web-shell-visuals-compose.test.mjs .github/scripts/serve-ab-diff.test.mjs .github/scripts/qwen-triage-workflow.test.mjs .github/scripts/assign-issue-owner.test.mjs .github/scripts/auto-minimize-spam.test.mjs .github/scripts/ci-runner-routing.test.mjs' + HELPER_TESTS: '.github/scripts/pr-safety-precheck.test.mjs .github/scripts/cap-release-notes.test.mjs .github/scripts/ci/classify-profile.test.mjs .github/scripts/ci/classify-pr-profile.test.mjs .github/scripts/upsert-bot-comment.test.mjs .github/scripts/ci/main-failure-signature.test.mjs .github/scripts/classify-release-notes.test.mjs .github/scripts/dsw-swe-verified/make-manifest.test.mjs .github/scripts/dsw-swe-verified/make-terminal-bench-manifest.test.mjs .github/scripts/resolve-sandbox-image.test.mjs .github/scripts/web-shell-visuals-publish.test.mjs .github/scripts/web-shell-visuals-compose.test.mjs .github/scripts/serve-ab-diff.test.mjs .github/scripts/qwen-triage-workflow.test.mjs .github/scripts/assign-issue-owner.test.mjs .github/scripts/auto-minimize-spam.test.mjs .github/scripts/ci-runner-routing.test.mjs' jobs: classify_pr: diff --git a/.github/workflows/dsw-swe-verified-release.yml b/.github/workflows/dsw-swe-verified-release.yml index 7a2f41dcf73..91f14aec17f 100644 --- a/.github/workflows/dsw-swe-verified-release.yml +++ b/.github/workflows/dsw-swe-verified-release.yml @@ -1,4 +1,4 @@ -name: 'DSW SWE-bench Verified Release' +name: 'DSW Harbor Benchmark Release' on: release: @@ -27,6 +27,27 @@ on: required: false default: 'sympy__sympy-20590' type: 'string' + terminal_bench_limit: + description: 'Number of Terminal-Bench 2.0 tasks (89 is the full suite)' + required: true + default: '1' + type: 'choice' + options: + - '1' + - '89' + terminal_bench_instance_id: + description: 'Exact Terminal-Bench 2.0 task when terminal_bench_limit is 1' + required: false + default: 'regex-log' + type: 'string' + execution_backend: + description: 'Execution backend; eas-smoke validates infrastructure only' + required: true + default: 'eas-smoke' + type: 'choice' + options: + - 'eas-smoke' + - 'eas-harbor' permissions: contents: 'read' @@ -43,6 +64,11 @@ jobs: runs-on: 'ubuntu-latest' outputs: should_run: '${{ steps.gate.outputs.should_run }}' + execution_backend: '${{ steps.gate.outputs.execution_backend }}' + instance_limit: '${{ steps.gate.outputs.instance_limit }}' + instance_id: '${{ steps.gate.outputs.instance_id }}' + terminal_bench_limit: '${{ steps.gate.outputs.terminal_bench_limit }}' + terminal_bench_instance_id: '${{ steps.gate.outputs.terminal_bench_instance_id }}' steps: - name: 'Check release version' id: 'gate' @@ -50,31 +76,88 @@ jobs: EVENT_NAME: '${{ github.event_name }}' RELEASE_TAG: '${{ github.event.release.tag_name || '''' }}' RELEASE_PRERELEASE: '${{ github.event.release.prerelease || false }}' + INPUT_EXECUTION_BACKEND: '${{ inputs.execution_backend }}' + INPUT_INSTANCE_LIMIT: '${{ inputs.instance_limit }}' + INPUT_INSTANCE_ID: '${{ inputs.instance_id }}' + INPUT_TERMINAL_BENCH_LIMIT: '${{ inputs.terminal_bench_limit }}' + INPUT_TERMINAL_BENCH_INSTANCE_ID: '${{ inputs.terminal_bench_instance_id }}' run: |- should_run='false' + execution_backend='eas-harbor' + instance_limit='500' + instance_id='' + terminal_bench_limit='89' + terminal_bench_instance_id='' if [[ "${EVENT_NAME}" == 'workflow_dispatch' ]]; then should_run='true' + execution_backend="${INPUT_EXECUTION_BACKEND}" + instance_limit="${INPUT_INSTANCE_LIMIT}" + if [[ "${instance_limit}" == '1' ]]; then + instance_id="${INPUT_INSTANCE_ID}" + fi + terminal_bench_limit="${INPUT_TERMINAL_BENCH_LIMIT}" + if [[ "${terminal_bench_limit}" == '1' ]]; then + terminal_bench_instance_id="${INPUT_TERMINAL_BENCH_INSTANCE_ID}" + fi + elif [[ "${RELEASE_PRERELEASE}" == 'true' && "${RELEASE_TAG}" =~ ^dsw-eas-smoke-[0-9A-Za-z._-]+$ ]]; then + # This tag intentionally exercises the production-sized EAS chain. + should_run='true' + execution_backend='eas-harbor' + instance_limit='500' + elif [[ "${RELEASE_PRERELEASE}" == 'true' && "${RELEASE_TAG}" =~ ^dsw-eas-tb-smoke-[0-9A-Za-z._-]+$ ]]; then + # A release-event smoke must exercise the same persistent EAS + # Harbor and Publisher chain as the full 500 + 89 benchmark. + should_run='true' + execution_backend='eas-harbor' + instance_limit='1' + instance_id='sympy__sympy-20590' + terminal_bench_limit='1' + terminal_bench_instance_id='regex-log' + elif [[ "${RELEASE_PRERELEASE}" == 'true' && "${RELEASE_TAG}" =~ ^dsw-eas-full-[0-9A-Za-z._-]+$ ]]; then + should_run='true' + execution_backend='eas-harbor' + instance_limit='500' + instance_id='' elif [[ "${RELEASE_PRERELEASE}" == 'false' && "${RELEASE_TAG}" =~ ^v[0-9]+\.[0-9]+\.0$ ]]; then should_run='true' fi - echo "should_run=${should_run}" >> "${GITHUB_OUTPUT}" + { + echo "should_run=${should_run}" + echo "execution_backend=${execution_backend}" + echo "instance_limit=${instance_limit}" + echo "instance_id=${instance_id}" + echo "terminal_bench_limit=${terminal_bench_limit}" + echo "terminal_bench_instance_id=${terminal_bench_instance_id}" + } >> "${GITHUB_OUTPUT}" if [[ "${should_run}" == 'false' ]]; then echo "::notice::Skipping benchmark for ${RELEASE_TAG:-non-release event}" fi benchmark: - name: 'Dispatch SWE-bench Verified to DSW' + name: 'Dispatch SWE-bench and Terminal-Bench to DSW' needs: 'release_gate' if: '${{ needs.release_gate.outputs.should_run == ''true'' }}' - runs-on: ['self-hosted', 'Linux', 'X64', 'qwen-benchmark-dsw'] + runs-on: ['self-hosted', 'Linux', 'X64', 'qwen-benchmark-dsw-hk-eas'] timeout-minutes: 15 env: RELEASE_TAG: '${{ github.event_name == ''release'' && github.event.release.tag_name || inputs.release_tag }}' QWEN_REF: '${{ github.event_name == ''release'' && github.event.release.tag_name || inputs.qwen_release_tag || inputs.release_tag }}' - INSTANCE_LIMIT: '${{ github.event_name == ''release'' && ''500'' || inputs.instance_limit }}' - BENCHMARK_INSTANCE_ID: '${{ (github.event_name != ''release'' && inputs.instance_limit == ''1'') && inputs.instance_id || '''' }}' + INSTANCE_LIMIT: '${{ needs.release_gate.outputs.instance_limit }}' + BENCHMARK_INSTANCE_ID: '${{ needs.release_gate.outputs.instance_id }}' + TERMINAL_BENCH_LIMIT: '${{ needs.release_gate.outputs.terminal_bench_limit }}' + TERMINAL_BENCH_INSTANCE_ID: '${{ needs.release_gate.outputs.terminal_bench_instance_id }}' BENCHMARK_IDEMPOTENCY_KEY: 'dsw-release-v1-${{ github.run_id }}' GITHUB_RUN_URL: '${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}' + BENCHMARK_EXECUTION_BACKEND: '${{ needs.release_gate.outputs.execution_backend }}' + DSW_POOL_ROOT: '/mnt/workspace/qwen-benchmark-acr-prewarm' + POOL_BIN: '/mnt/workspace/qwen-benchmark-eas-poc/venv/bin/qwen-benchmark-pool' + POOL_PYTHON: '/mnt/workspace/qwen-benchmark-eas-poc/venv/bin/python' + PYTHONPATH: '/mnt/workspace/qwen-benchmark-eas-poc' + SWE_VERIFIED_DATASET_ROOT: '/mnt/workspace/qwen-benchmark-acr-prewarm/datasets/swe-bench-verified' + EAS_TEMPLATE_MANIFEST: '/mnt/workspace/qwen-benchmark-acr-prewarm/deploy/eas/templates.json' + ACR_IMAGE_MANIFEST: '/mnt/workspace/qwen-benchmark-eas-poc/state/acr-manifest-c104f840.json' + ACR_IMAGE_STATE_DIR: '/mnt/data/qwen-benchmark/acr-prewarm/c104f840/state' + BENCHMARK_POOL_DATABASE_URL: 'postgresql://postgres@/qwen_benchmark_eas_long_pool_utf8?host=/mnt/dynamic/qwen-benchmark-pg/run&port=55432' steps: - name: 'Checkout pipeline' @@ -130,10 +213,12 @@ jobs: POOL_RUN_ID: '${{ steps.dispatch.outputs.run_id }}' run: |- { - echo "### DSW SWE-bench Verified dispatch" + echo "### DSW Harbor benchmark dispatch" echo echo "- Pool run: \`${POOL_RUN_ID}\`" echo "- State: queued" - echo "- Cases: ${INSTANCE_LIMIT}" - echo "- The persistent DSW publisher will update the Release after the run reaches a validated terminal state." + echo "- SWE-bench cases: ${INSTANCE_LIMIT}" + echo "- Terminal-Bench 2.0: ${TERMINAL_BENCH_LIMIT} frozen tasks recorded as a durable follow-up." + echo "- TB is dispatched only after the SWE Publisher successfully updates this Release." + echo "- The persistent DSW publisher writes separate, verifier-backed SWE and TB result sections and assets to this Release." } >> "${GITHUB_STEP_SUMMARY}"