diff --git a/.github/workflows/opencode-review.yml b/.github/workflows/opencode-review.yml index 1b779a7e4..e7bfd8184 100644 --- a/.github/workflows/opencode-review.yml +++ b/.github/workflows/opencode-review.yml @@ -583,7 +583,7 @@ jobs: export R_LIBS_USER="${RUNNER_TEMP}/R-library" mkdir -p "$R_LIBS_USER" run_and_capture "R coverage tooling (covr/testthat)" \ - bash -c 'timeout 780 Rscript -e '\''user_repo <- "https://cloud.r-project.org"; codename <- tryCatch(sub("[[:space:]]+$", "", system2("lsb_release", "-cs", stdout = TRUE, stderr = FALSE)[1]), error = function(e) ""); if (length(codename) == 1 && !is.na(codename) && nzchar(codename)) { options(HTTPUserAgent = sprintf("R/%s R (%s)", getRversion(), paste(getRversion(), R.version$platform, R.version$arch, R.version$os))); repos <- c(sprintf("https://p3m.dev/cran/__linux__/%s/latest", codename), user_repo) } else { repos <- user_repo }; lib <- Sys.getenv("R_LIBS_USER"); install_deps <- c("Depends", "Imports", "LinkingTo"); dir.create(lib, recursive = TRUE, showWarnings = FALSE); .libPaths(c(lib, .libPaths())); required <- c("covr", "testthat"); if (file.exists("DESCRIPTION")) { desc <- read.dcf("DESCRIPTION")[1, , drop = FALSE]; fields <- intersect(c("Depends", "Imports", "LinkingTo"), colnames(desc)); values <- as.character(desc[, fields, drop = TRUE]); values <- values[!is.na(values)]; package_deps <- trimws(gsub("\\s*\\([^)]*\\)", "", unlist(strsplit(paste(values, collapse = ","), ","), use.names = FALSE))); package_deps <- setdiff(package_deps[nzchar(package_deps)], "R"); required <- unique(c(required, package_deps)); }; for (pkg in required) if (!requireNamespace(pkg, quietly = TRUE)) install.packages(pkg, repos = repos, lib = lib, dependencies = install_deps); missing <- required[!vapply(required, requireNamespace, logical(1), quietly = TRUE)]; if (length(missing)) stop("R coverage tooling packages unavailable after install: ", paste(missing, collapse = ", "))'\'' || { echo "R coverage tooling install unavailable or exceeded the runner time budget; deferring to required peer R CMD check evidence."; exit 0; }' + bash -c 'Rscript -e '\''repos <- "https://cloud.r-project.org"; lib <- Sys.getenv("R_LIBS_USER"); install_deps <- c("Depends", "Imports", "LinkingTo"); dir.create(lib, recursive = TRUE, showWarnings = FALSE); .libPaths(c(lib, .libPaths())); required <- c("covr", "testthat"); if (file.exists("DESCRIPTION")) { desc <- read.dcf("DESCRIPTION")[1, , drop = FALSE]; fields <- intersect(c("Depends", "Imports", "LinkingTo", "Suggests"), colnames(desc)); values <- as.character(desc[, fields, drop = TRUE]); values <- values[!is.na(values)]; package_deps <- trimws(gsub("\\s*\\([^)]*\\)", "", unlist(strsplit(paste(values, collapse = ","), ","), use.names = FALSE))); package_deps <- setdiff(package_deps[nzchar(package_deps)], "R"); required <- unique(c(required, package_deps)); }; for (pkg in required) if (!requireNamespace(pkg, quietly = TRUE)) install.packages(pkg, repos = repos, lib = lib, dependencies = install_deps); missing <- required[!vapply(required, requireNamespace, logical(1), quietly = TRUE)]; if (length(missing)) stop("R coverage tooling packages unavailable after install: ", paste(missing, collapse = ", "))'\'' || { echo "R coverage tooling install unavailable in coverage runner; deferring to required peer R CMD check evidence."; exit 0; }' if [ -f DESCRIPTION ]; then if [ -d tests/testthat ]; then run_and_capture "R package testthat suite" \ @@ -834,7 +834,6 @@ jobs: needs: [coverage-evidence] if: >- always() - && needs.coverage-evidence.result != 'cancelled' && ( github.event_name == 'workflow_dispatch' || ( @@ -2007,7 +2006,7 @@ jobs: "$schema": "https://opencode.ai/config.json", "model": "github-models/deepseek/deepseek-r1-0528", "small_model": "github-models/deepseek/deepseek-v3-0324", - "enabled_providers": ["openai", "github-models"], + "enabled_providers": ["github-models"], "lsp": true, "mcp": { "codegraph": { @@ -2133,50 +2132,6 @@ jobs: } }, "provider": { - "openai": { - "npm": "@ai-sdk/openai", - "name": "OpenAI (direct)", - "options": { - "baseURL": "https://api.openai.com/v1", - "apiKey": "{env:OPENAI_API_KEY}" - }, - "models": { - "gpt-5": { - "name": "OpenAI GPT-5 (direct)", - "tool_call": true, - "reasoning": true, - "options": { - "reasoningEffort": "high" - }, - "variants": { - "high": { - "reasoningEffort": "high" - } - }, - "limit": { - "context": 400000, - "output": 128000 - } - }, - "gpt-5-mini": { - "name": "OpenAI GPT-5 Mini (direct)", - "tool_call": true, - "reasoning": true, - "options": { - "reasoningEffort": "high" - }, - "variants": { - "high": { - "reasoningEffort": "high" - } - }, - "limit": { - "context": 400000, - "output": 128000 - } - } - } - }, "github-models": { "npm": "@ai-sdk/openai-compatible", "name": "GitHub Models", @@ -2384,44 +2339,31 @@ jobs: env: STRIX_GITHUB_MODELS_TOKEN: ${{ secrets.STRIX_GITHUB_MODELS_TOKEN || github.token }} GITHUB_TOKEN: ${{ secrets.STRIX_GITHUB_MODELS_TOKEN || github.token }} - # Native OpenAI backend for the lead review model. GitHub Models - # rate-limits every request and caps bodies at ~4000 tokens, so the - # rate-starved shared pool never returned a verdict; hitting - # api.openai.com directly with the org OPENAI_API_KEY gives the lead - # model a working, un-throttled backend. Resolves {env:OPENAI_API_KEY} - # in the opencode.jsonc "openai" provider block. - OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} USE_GITHUB_TOKEN: "true" SHARE: "false" NPM_CONFIG_IGNORE_SCRIPTS: "true" NO_COLOR: "1" - # Lead with the NATIVE OpenAI backend (openai/gpt-5-mini, openai/gpt-5 - # via api.openai.com with the org OPENAI_API_KEY). GitHub Models - # rate-limited ("Too many requests") and 4000-token-capped - # (413 tokens_limit_reached) EVERY model in the shared pool, so the - # reviewer never produced a verdict and every run hung to the 350-min - # timeout — a 100% org-wide failure. The native provider is not subject - # to those limits, so it can actually complete and approve. The - # existing github-models entries stay as fallbacks (tried only if the - # native key is missing or the direct call fails). - # github-models ordering rationale (unchanged): contract-reliable mini - # reasoning models first, high-quota non-reasoning models next, and the - # rate-starved github-models flagships (gpt-5/o3, 8-12 req/day) last so - # a throttled/hung leader always falls back instead of eating the step. - OPENCODE_MODEL_CANDIDATES: "openai/gpt-5-mini openai/gpt-5 github-models/openai/o4-mini github-models/openai/o3-mini github-models/openai/gpt-5-mini github-models/openai/gpt-5-nano github-models/openai/gpt-5-chat github-models/deepseek/deepseek-r1-0528 github-models/deepseek/deepseek-r1 github-models/deepseek/deepseek-v3-0324 github-models/mistral-ai/mistral-medium-2505 github-models/meta/llama-4-maverick-17b-128e-instruct-fp8 github-models/meta/llama-4-scout-17b-16e-instruct github-models/openai/o3 github-models/openai/gpt-5" + # Ordered contract-reliability first, then quota, with the rate-starved + # flagships last. The mini reasoning models (o4-mini, o3-mini, gpt-5- + # mini/nano/chat) reliably emit the strict review contract — every + # required label and only source-backed findings — so they lead. The + # high-quota non-reasoning models (deepseek-v3, mistral, llama-4) emit + # bare or hallucinated reviews the publish/approve gates reject, so + # they are fallbacks only. gpt-5/o3 ("Reasoning" tier, 8-12 req/day) + # stay last: first-placing them stalled every review until timeout + # because a rate-limited/hung flagship never fell back. + OPENCODE_MODEL_CANDIDATES: "github-models/openai/o4-mini github-models/openai/o3-mini github-models/openai/gpt-5-mini github-models/openai/gpt-5-nano github-models/openai/gpt-5-chat github-models/deepseek/deepseek-r1-0528 github-models/deepseek/deepseek-r1 github-models/deepseek/deepseek-v3-0324 github-models/mistral-ai/mistral-medium-2505 github-models/meta/llama-4-maverick-17b-128e-instruct-fp8 github-models/meta/llama-4-scout-17b-16e-instruct github-models/openai/o3 github-models/openai/gpt-5" # One attempt per model, then fall through to the next model. Retrying # the SAME model 5x let a rate-limited/hung leader consume the whole # step, so the pool never reached a healthy fallback model. OPENCODE_MODEL_ATTEMPTS: "1" - # 15 min per model — enough for a bounded review attempt, but short - # enough that a hung provider yields to the next candidate before it - # freezes the review queue. - OPENCODE_RUN_TIMEOUT_SECONDS: "900" + # 90 min per model — generous for a deep tool-using review, but bounded + # so a rate-limited model yields to the next one instead of eating the + # 350-min step. (20400s = 340min gave one model the entire budget with + # no fallback; 600s was too short for a proper review.) + OPENCODE_RUN_TIMEOUT_SECONDS: "5400" OPENCODE_EXPORT_TIMEOUT_SECONDS: "120" - # Bound provider/model-pool outages before the 350-min job timeout. A - # zero budget disables the script deadline and caused org-wide hangs. - OPENCODE_TOTAL_RETRY_BUDGET_SECONDS: "2700" - OPENCODE_POOL_MAX_CYCLES: "1" + OPENCODE_TOTAL_RETRY_BUDGET_SECONDS: "0" OPENCODE_BACKOFF_INITIAL_SECONDS: "30" OPENCODE_BACKOFF_MAX_SECONDS: "30" OPENCODE_FIRST_ATTEMPT_AGENT: ci-review @@ -2790,9 +2732,6 @@ jobs: CHECK_LOOKUP_GH_TOKEN: ${{ github.token }} GH_REPOSITORY: ${{ github.event.pull_request.base.repo.full_name || github.event.inputs.target_repository || github.repository }} STRIX_GITHUB_MODELS_TOKEN: ${{ secrets.STRIX_GITHUB_MODELS_TOKEN || github.token }} - # Exposed so the "openai" provider in opencode.jsonc resolves during the - # failed-check diagnosis opencode run that shares this config. - OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} OPENCODE_APP_TOKEN: ${{ steps.opencode_app_token.outputs.token }} OPENCODE_EVIDENCE_FILE: ${{ runner.temp }}/opencode-review-evidence.md OPENCODE_FAILED_CHECK_EVIDENCE_FILE: ${{ runner.temp }}/opencode-failed-check-evidence.md @@ -2869,6 +2808,12 @@ jobs: fi } + gh_error_is_rate_limited() { + local error_file="$1" + [ -s "$error_file" ] || return 1 + grep -Eiq '(API rate limit exceeded|rate limit exceeded|secondary rate limit)' "$error_file" + } + emit_change_flow_mermaid_graph() { local merge_state="${1:-UNKNOWN}" local changed_files_file surfaces_file idx next_node diff --git a/.github/workflows/osv-scanner-pr.yml b/.github/workflows/osv-scanner-pr.yml index 335161db8..4d5c4475f 100644 --- a/.github/workflows/osv-scanner-pr.yml +++ b/.github/workflows/osv-scanner-pr.yml @@ -29,12 +29,7 @@ jobs: osv-scan: if: github.event.action != 'closed' # ponytail: use upstream reusable PR workflow, don't hand-roll the diff scan - # Pinned to v2.3.8 + 1 commit (3a7550f) which gates the JSON job outputs - # behind the new `export-results` input (default false). v2.3.8 dumped the - # full old/new osv-scanner JSON into job outputs unconditionally, tripping - # GitHub's 1,048,576-byte job-outputs cap and failing the run. Same nested - # action pins as v2.3.8; only the Export step is now conditional. - uses: google/osv-scanner-action/.github/workflows/osv-scanner-reusable-pr.yml@3a7550f43ba5b58905a821ce3a0ed24c4858b3f4 # v2.3.8 + export-results gate + uses: google/osv-scanner-action/.github/workflows/osv-scanner-reusable-pr.yml@9a498708959aeaef5ef730655706c5a1df1edbc2 # v2.3.8 permissions: actions: read contents: read @@ -44,13 +39,3 @@ jobs: # (medium_or_higher), not by failing this check. Keep the check green # so it only supplies the analysis; the ruleset decides blocking. fail-on-vuln: false - # RELIABILITY: resolve Maven parent POMs through Google's byte-identical - # Maven Central mirror instead of repo.maven.apache.org, which - # intermittently 429s during transitive resolution (e.g. - # spring-boot-starter-parent). Same maven2 layout, same bytes -> no - # coverage loss; transitive scanning stays fully enabled. See - # security-scan.yml for the full rationale. - scan-args: |- - --maven-registry=https://maven-central.storage-download.googleapis.com/maven2 - -r - ./ diff --git a/.github/workflows/security-scan.yml b/.github/workflows/security-scan.yml index 35fbf751b..7a077b05c 100644 --- a/.github/workflows/security-scan.yml +++ b/.github/workflows/security-scan.yml @@ -57,23 +57,6 @@ jobs: security-events: write with: fail-on-vuln: true - # RELIABILITY: point Maven transitive (parent-POM) resolution at Google's - # byte-identical Maven Central mirror instead of repo.maven.apache.org. - # osv-scanner resolves parent POMs (e.g. spring-boot-starter-parent) over - # its own HTTP client (osv-scalibr pomxmlnet -> defaultRegistry.URL); the - # canonical Central host intermittently returns HTTP 429, which failed this - # required gate on APPROVED Maven PRs with "No issues found" (a network - # flake, not a real vuln). The mirror serves the same maven2 layout and the - # same bytes, so coverage is unchanged -- this ONLY swaps the default - # registry host. Repos' own pom.xml are still added on top - # (scalibr AddRegistry), and transitive scanning stays fully enabled (no - # --no-resolve). Caching ~/.m2 or a settings.xml would NOT help: - # osv-scanner never reads ~/.m2/repository and parses settings.xml only for - # auth, not . --maven-registry is the only effective lever. - scan-args: |- - --maven-registry=https://maven-central.storage-download.googleapis.com/maven2 - -r - ./ dependency-review: if: github.event.action != 'closed' @@ -151,10 +134,6 @@ jobs: format: sarif output: trivy-results.sarif exit-code: "1" - # Without this, trivy-action rebuilds the SARIF scan with ALL - # severities and exit-code applies to any LOW/MEDIUM finding, - # contradicting the documented CRITICAL/HIGH-only gate above. - limit-severities-for-sarif: true - name: Upload Trivy SARIF to code scanning if: always() && hashFiles('trivy-results.sarif') != '' uses: github/codeql-action/upload-sarif@8aad20d150bbac5944a9f9d289da16a4b0d87c1e # v4.36.2 diff --git a/scripts/ci/noema_review_gate.py b/scripts/ci/noema_review_gate.py index cb039dba1..17337ee63 100644 --- a/scripts/ci/noema_review_gate.py +++ b/scripts/ci/noema_review_gate.py @@ -279,7 +279,7 @@ def call_llm(repo: str, number: int, pr: dict[str, Any], diff: str, truncated: b return None parsed = urllib.parse.urlparse(api_url) if parsed.scheme.lower() not in {"http", "https"}: - raise ValueError("URL scheme must be http or https") + raise ValueError("URL must start with http:// or https://") hostname = (parsed.hostname or "").lower() if not hostname: raise ValueError("URL must have a valid hostname") @@ -299,6 +299,9 @@ def call_llm(repo: str, number: int, pr: dict[str, Any], diff: str, truncated: b if ip.is_private or ip.is_loopback or ip.is_link_local or ip.is_multicast or ip.is_unspecified: raise ValueError("URL cannot target internal IP addresses") + if not (api_url.lower().startswith("http://") or api_url.lower().startswith("https://")): + raise ValueError(f"NOEMA_LLM_API_URL must start with http:// or https:// to prevent SSRF vulnerabilities, got: {api_url}") + prompt = { "role": "user", "content": "\n".join( @@ -335,6 +338,8 @@ def call_llm(repo: str, number: int, pr: dict[str, Any], diff: str, truncated: b }, method="POST", ) + if parsed.scheme.lower() not in {"http", "https"}: + raise ValueError("URL scheme http or https required") with urllib.request.urlopen(request, timeout=120) as response: # nosec B310 raw = response.read().decode("utf-8") data = json.loads(raw) diff --git a/scripts/ci/opencode_review_normalize_output.py b/scripts/ci/opencode_review_normalize_output.py index 88b931872..d8a984381 100755 --- a/scripts/ci/opencode_review_normalize_output.py +++ b/scripts/ci/opencode_review_normalize_output.py @@ -228,6 +228,8 @@ PREFERRED_REVIEW_LANGUAGE_RE = re.compile( r"Preferred review language:\s*`?([A-Za-z]+)`?", re.IGNORECASE ) +# Performance optimization: Pre-compile regex at module level to avoid redundant parsing inside changed_files_from_evidence loops. +LIST_MARKER_RE = re.compile(r"^[-*+]\s+") def admits_missing_structural_review(reason: str, summary: str) -> bool: @@ -530,7 +532,7 @@ def changed_files_from_evidence(text: str) -> list[str]: line = raw_line.strip() if not line or line.startswith("#"): continue - line = re.sub(r"^[-*+]\s+", "", line) + line = LIST_MARKER_RE.sub("", line) parts = line.split("\t") path = parts[-1].strip() if not path or path.startswith("["): diff --git a/scripts/ci/run_opencode_review_model_pool.sh b/scripts/ci/run_opencode_review_model_pool.sh index 91e1d700c..6cdf1b85f 100644 --- a/scripts/ci/run_opencode_review_model_pool.sh +++ b/scripts/ci/run_opencode_review_model_pool.sh @@ -106,23 +106,6 @@ is_context_overflow_failure() { grep -Eiq 'ContextOverflowError|tokens_limit_reached|Request body too large|context window' "$opencode_json_file" } -is_direct_openai_candidate() { - case "$1" in - openai/*) return 0 ;; - *) return 1 ;; - esac -} - -should_skip_model_candidate() { - local model_candidate="$1" - - if is_direct_openai_candidate "$model_candidate" && [ -z "${OPENAI_API_KEY:-}" ]; then - printf 'Skipping OpenCode %s because OPENAI_API_KEY is not configured; falling back to the next provider-qualified candidate.\n' "$model_candidate" - return 0 - fi - return 1 -} - run_one_model_attempt() { local model_candidate="$1" local attempt="$2" @@ -186,13 +169,12 @@ run_one_model_attempt() { main() { local attempts budget_seconds deadline now remaining model_candidate attempt safe_model prompt_file candidate_output_file - local opencode_json_file opencode_export_file agent retry_sleep original_run_timeout run_status cycle_sleep cycle max_cycles + local opencode_json_file opencode_export_file agent retry_sleep original_run_timeout run_status cycle_sleep cycle local -a model_candidates attempts="${OPENCODE_MODEL_ATTEMPTS:-3}" original_run_timeout="${OPENCODE_RUN_TIMEOUT_SECONDS:-900}" budget_seconds="${OPENCODE_TOTAL_RETRY_BUDGET_SECONDS:-18000}" - max_cycles="${OPENCODE_POOL_MAX_CYCLES:-0}" deadline=0 if [ "$budget_seconds" -gt 0 ]; then deadline=$((SECONDS + budget_seconds)) @@ -210,9 +192,6 @@ main() { while :; do printf 'Starting OpenCode model pool cycle %s.\n' "$cycle" for model_candidate in "${model_candidates[@]}"; do - if should_skip_model_candidate "$model_candidate"; then - continue - fi assert_reasoning_effort_for_candidate "$model_candidate" safe_model="${model_candidate//\//-}" prompt_file="${RUNNER_TEMP}/opencode-review-${safe_model}-prompt.md" @@ -264,11 +243,6 @@ main() { done printf 'OpenCode completed a full model-candidate cycle without a valid control conclusion; continuing until a model succeeds or the GitHub Actions job timeout is reached.\n' - if [ "$max_cycles" -gt 0 ] && [ "$cycle" -ge "$max_cycles" ]; then - printf 'OpenCode model pool reached configured max cycle count %s without a valid control conclusion.\n' "$max_cycles" - record_review_model "" - exit 1 - fi cycle_sleep="${OPENCODE_POOL_CYCLE_SLEEP_SECONDS:-60}" if [ "$deadline" -gt 0 ] && [ $((SECONDS + cycle_sleep)) -gt "$deadline" ]; then cycle_sleep=$((deadline - SECONDS)) diff --git a/scripts/ci/test_strix_quick_gate.sh b/scripts/ci/test_strix_quick_gate.sh index 54ad0faef..f51cfff59 100755 --- a/scripts/ci/test_strix_quick_gate.sh +++ b/scripts/ci/test_strix_quick_gate.sh @@ -384,7 +384,6 @@ assert_opencode_review_uses_codegraph_and_gpt5_fallback() { assert_file_contains "$workflow_file" 'cancel-in-progress: true' "opencode review cancels stale in-progress review attempts when a newer PR event arrives" assert_file_contains "$workflow_file" "github.event.pull_request.head.repo.full_name == github.repository" "opencode pull_request_target coverage execution is limited to same-repository PR heads" assert_file_contains "$workflow_file" "if: always() && (github.event_name == 'workflow_dispatch' || github.event_name == 'pull_request_target')" "opencode review side effects are limited to manual or required PR events" - assert_file_contains "$workflow_file" "needs.coverage-evidence.result != 'cancelled'" "opencode review does not enqueue stale side-effect jobs after coverage evidence cancellation" assert_file_contains "$workflow_file" "opencode-review-target:" "opencode trusted review job owns the required check surface" assert_file_contains "$workflow_file" "Initialize CodeGraph index for OpenCode" "opencode review workflow initializes CodeGraph before review" assert_file_contains "$workflow_file" "actions: write" "opencode review workflow can read failed Actions logs and dispatch the merge scheduler after approval" @@ -522,17 +521,16 @@ assert_opencode_review_uses_codegraph_and_gpt5_fallback() { assert_file_contains "$REPO_ROOT/scripts/ci/run_opencode_review_model_pool.sh" "tokens_limit_reached" "opencode review detects provider context-window overflow" assert_file_contains "$REPO_ROOT/scripts/ci/run_opencode_review_model_pool.sh" "skipping remaining attempts for this model" "opencode review skips same-model retries after context-window overflow" assert_file_contains "$workflow_file" 'timeout-minutes: 360' "opencode review target uses the maximum GitHub-hosted runner timeout" - assert_file_contains "$workflow_file" 'timeout-minutes: 350' "opencode model pool keeps a bounded job timeout while leaving approval headroom" - assert_file_contains "$workflow_file" 'OPENCODE_RUN_TIMEOUT_SECONDS: "2400"' "opencode primary review has a bounded per-model timeout before trying fallback models" - assert_file_contains "$workflow_file" 'OPENCODE_TOTAL_RETRY_BUDGET_SECONDS: "7200"' "opencode model pool has a script-level retry budget below the job timeout" - assert_file_contains "$workflow_file" 'OPENCODE_POOL_MAX_CYCLES: "1"' "opencode model pool stops after one full candidate pass instead of looping to the job timeout" + assert_file_contains "$workflow_file" 'timeout-minutes: 285' "opencode model pool keeps retrying for most of the job budget while leaving approval headroom" + assert_file_contains "$workflow_file" 'OPENCODE_RUN_TIMEOUT_SECONDS: "600"' "opencode primary review has a bounded per-model timeout before trying fallback models" + assert_file_contains "$workflow_file" 'OPENCODE_TOTAL_RETRY_BUDGET_SECONDS: "0"' "opencode model pool has no script-level retry budget" assert_file_contains "$workflow_file" "needs.coverage-evidence.result == 'success'" "opencode model pool only runs after coverage evidence passed" assert_file_contains "$workflow_file" "id: opencode_review_model_pool" "opencode DeepSeek V3 fallback still runs after a primary model timeout or step failure when coverage evidence passed" assert_file_contains "$workflow_file" "always()" "opencode fallback chain uses always() so failed model steps cannot skip every fallback" assert_file_contains "$workflow_file" 'OPENCODE_MODEL_ATTEMPTS: "1"' "opencode fallback tries the catalog promptly instead of spending the entire review on one model" assert_file_contains "$workflow_file" "Run OpenCode PR Review model pool" "opencode review includes a broad catalog fallback pool" assert_file_contains "$workflow_file" "steps.opencode_review_model_pool.outcome == 'success'" "opencode model step must succeed before review publication" - assert_file_contains "$workflow_file" "openai/gpt-5-mini openai/gpt-5 github-models/openai/o4-mini github-models/openai/o3-mini github-models/openai/gpt-5-mini" "opencode review tries native OpenAI before GitHub Models fallbacks" + assert_file_contains "$workflow_file" "github-models/openai/o4-mini github-models/openai/o3-mini github-models/openai/gpt-5-mini github-models/openai/gpt-5-chat github-models/openai/o3 github-models/mistral-ai/mistral-medium-2505 github-models/openai/gpt-5-nano github-models/deepseek/deepseek-r1-0528 github-models/deepseek/deepseek-r1 github-models/deepseek/deepseek-v3-0324 github-models/meta/llama-4-maverick-17b-128e-instruct-fp8 github-models/meta/llama-4-scout-17b-16e-instruct" "opencode review tries high-effort reasoning fallbacks before broader catalog models" assert_file_contains "$workflow_file" "The publish gate re-runs source-backed validation against PR-head data" "opencode review publish gate validates model output against the PR-head worktree" assert_file_contains "$workflow_file" '"openai/o3"' "opencode config declares OpenAI o3 fallback" assert_file_contains "$workflow_file" '"openai/o4-mini"' "opencode config declares OpenAI o4-mini fallback" @@ -630,17 +628,15 @@ assert_opencode_review_uses_codegraph_and_gpt5_fallback() { assert_file_not_contains "$workflow_file" '[ "$changed_count" -gt 0 ] && [ "$changed_count" -le 2 ]' "opencode model-exhaustion fallback must not cap deterministic approval scope" assert_file_contains "$REPO_ROOT/scripts/ci/run_opencode_review_model_pool.sh" "completed a full model-candidate cycle without a valid control conclusion" "opencode model-output failures keep retrying instead of publishing a review" assert_file_contains "$REPO_ROOT/scripts/ci/run_opencode_review_model_pool.sh" "OpenCode model pool has no configured model candidates." "opencode model pool fails fast when no candidates are configured" - assert_file_contains "$REPO_ROOT/scripts/ci/run_opencode_review_model_pool.sh" "OPENAI_API_KEY is not configured" "opencode model pool skips native OpenAI candidates when the org secret is absent" - assert_file_contains "$REPO_ROOT/scripts/ci/run_opencode_review_model_pool.sh" "configured max cycle count" "opencode model pool exits before the job timeout after configured cycles" assert_file_contains "$REPO_ROOT/scripts/ci/run_opencode_review_model_pool.sh" 'OPENCODE_TOTAL_RETRY_BUDGET_SECONDS:-18000' "opencode model pool keeps a safe default retry budget unless the workflow explicitly disables it" assert_file_not_contains "$workflow_file" "no model produced a valid review control block" "opencode model-failure path no longer documents a final exhausted state" assert_file_contains "$workflow_file" 'OPENCODE_MODEL_ATTEMPTS: "1"' "opencode primary and fallback paths avoid multi-attempt stalls on one model" assert_file_contains "$workflow_file" 'OPENCODE_MODEL_ATTEMPTS: "1"' "opencode catalog fallback tries each model once before moving on" - assert_file_contains "$workflow_file" 'OPENCODE_RUN_TIMEOUT_SECONDS: "2400"' "opencode catalog fallback has a bounded model review timeout before step timeout" + assert_file_contains "$workflow_file" 'OPENCODE_RUN_TIMEOUT_SECONDS: "600"' "opencode catalog fallback has a bounded model review timeout before step timeout" assert_file_contains "$REPO_ROOT/scripts/ci/run_opencode_review_model_pool.sh" "OpenCode %s attempt %s/%s failed" "opencode catalog fallback records per-model retry failures" assert_file_contains "$REPO_ROOT/scripts/ci/run_opencode_review_model_pool.sh" "exponential backoff" "opencode model retry paths use exponential backoff instead of fixed sleeps" assert_file_contains "$workflow_file" "github-models/openai/o4-mini github-models/openai/o3-mini" "opencode review tries compact OpenAI reasoning model fallbacks early" - assert_file_contains "$workflow_file" "github-models/openai/gpt-5-chat github-models/deepseek/deepseek-r1-0528 github-models/deepseek/deepseek-r1" "opencode review keeps reasoning catalog fallbacks after compact attempts" + assert_file_contains "$workflow_file" "github-models/openai/gpt-5-chat github-models/openai/o3 github-models/mistral-ai" "opencode review keeps full OpenAI and non-OpenAI catalog fallbacks after compact reasoning attempts" assert_file_contains "$workflow_file" "coverage-evidence:" "opencode workflow measures coverage before review" assert_file_contains "$workflow_file" "github.event_name == 'workflow_dispatch' || github.event_name == 'pull_request_target'" "manual and required OpenCode reviews measure coverage instead of approving skipped coverage evidence" assert_file_contains "$workflow_file" "Exchange OpenCode app token for target repository coverage reads" "coverage evidence can read private target repositories through the OpenCode app token" diff --git a/scripts/ci/validate_opencode_failed_check_review.sh b/scripts/ci/validate_opencode_failed_check_review.sh index a500e5fbc..948c76648 100755 --- a/scripts/ci/validate_opencode_failed_check_review.sh +++ b/scripts/ci/validate_opencode_failed_check_review.sh @@ -213,26 +213,40 @@ location_re = re.compile( re.IGNORECASE, ) +# Performance optimization: Pre-compile regex at module level to avoid redundant parsing inside loop processing functions. +clean_prefix_re = re.compile(r"^.*?│\s*") +clean_suffix_re = re.compile(r"\s*│.*$") +clean_timestamp_re = re.compile(r"^.*?[0-9]Z\s+") +whitespace_re = re.compile(r"\s+") +new_field_re = re.compile( + r"^(Title|Severity|CVSS Score|CVSS Vector|Target|Endpoint|Method|Description|Impact|Technical Analysis|PoC Description|PoC Code|Code Locations|Remediation)\b", + re.IGNORECASE, +) +window_model_re = re.compile( + r"(?:model|for model)\s+((?:github[-_]models|openai|deepseek|vertex_ai)/[A-Za-z0-9._/-]+)", + re.IGNORECASE, +) +box_chars_re = re.compile(r"^[╭╰─]+$") +title_re = re.compile(r"^Title:\s+(.+)", re.IGNORECASE) +severity_re = re.compile(r"^Severity:\s+(CRITICAL|HIGH|MEDIUM|LOW|NONE)\b", re.IGNORECASE) +endpoint_re = re.compile(r"^Endpoint:\s+(.+)", re.IGNORECASE) +method_re = re.compile(r"^Method:\s+(.+)", re.IGNORECASE) +target_re = re.compile(r"^Target:\s+(.+)", re.IGNORECASE) + def clean(raw_line: str) -> str: line = ansi_re.sub("", raw_line).replace("\r", "") if "│" in line: - line = re.sub(r"^.*?│\s*", "", line) - line = re.sub(r"\s*│.*$", "", line) + line = clean_prefix_re.sub("", line) + line = clean_suffix_re.sub("", line) else: - line = re.sub(r"^.*?[0-9]Z\s+", "", line) - line = re.sub(r"\s+", " ", line).strip() + line = clean_timestamp_re.sub("", line) + line = whitespace_re.sub(" ", line).strip() return line def starts_new_field(line: str) -> bool: - return bool( - re.match( - r"^(Title|Severity|CVSS Score|CVSS Vector|Target|Endpoint|Method|Description|Impact|Technical Analysis|PoC Description|PoC Code|Code Locations|Remediation)\b", - line, - re.IGNORECASE, - ) - ) + return bool(new_field_re.match(line)) class ReportParser: @@ -271,11 +285,7 @@ class ReportParser: self.finish_report() self.in_window = True self.window_model = "" - match = re.search( - r"(?:model|for model)\s+((?:github[-_]models|openai|deepseek|vertex_ai)/[A-Za-z0-9._/-]+)", - line, - re.IGNORECASE, - ) + match = window_model_re.search(line) if match: self.window_model = match.group(1) self.current_model = match.group(1) @@ -296,7 +306,7 @@ class ReportParser: return False if not line: self.continuation = "" - elif not starts_new_field(line) and not re.match(r"^[╭╰─]+$", line) and line.lower() != "vulnerability report": + elif not starts_new_field(line) and not box_chars_re.match(line) and line.lower() != "vulnerability report": if self.continuation == "title": self.title = f"{self.title} {line}".strip() elif self.continuation == "endpoint": @@ -309,28 +319,28 @@ class ReportParser: return False def _parse_field(self, line: str) -> None: - field_match = re.match(r"^Title:\s+(.+)", line, re.IGNORECASE) + field_match = title_re.match(line) if field_match: self.finish_report() self.title = field_match.group(1) self.report_model = self.window_model self.continuation = "title" return - field_match = re.match(r"^Severity:\s+(CRITICAL|HIGH|MEDIUM|LOW|NONE)\b", line, re.IGNORECASE) + field_match = severity_re.match(line) if field_match: self.severity = field_match.group(1).upper() return - field_match = re.match(r"^Endpoint:\s+(.+)", line, re.IGNORECASE) + field_match = endpoint_re.match(line) if field_match: self.endpoint = field_match.group(1) self.continuation = "endpoint" return - field_match = re.match(r"^Method:\s+(.+)", line, re.IGNORECASE) + field_match = method_re.match(line) if field_match: self.method = field_match.group(1) self.continuation = "" return - field_match = re.match(r"^Target:\s+(.+)", line, re.IGNORECASE) + field_match = target_re.match(line) if field_match: self.target = field_match.group(1) self.continuation = "target" diff --git a/tests/test_noema_review_gate.py b/tests/test_noema_review_gate.py index 8285fceb5..bd063eb0d 100644 --- a/tests/test_noema_review_gate.py +++ b/tests/test_noema_review_gate.py @@ -206,7 +206,7 @@ def test_call_llm_handles_configuration_and_verdicts(monkeypatch): monkeypatch.setenv("NOEMA_LLM_API_URL", "file:///etc/passwd") monkeypatch.setenv("NOEMA_LLM_API_KEY", "secret") - with pytest.raises(ValueError, match="URL scheme must be http or https"): + with pytest.raises(ValueError, match="must start with http:// or https://"): noema.call_llm("owner/repo", 1, pr, "diff", False) monkeypatch.setenv("NOEMA_LLM_API_URL", "https://llm.example.test/chat") @@ -240,7 +240,7 @@ def fake_urlopen(request, timeout): # Test invalid scheme (and no original URL in error) monkeypatch.setenv("NOEMA_LLM_API_URL", "file:///etc/passwd") - with pytest.raises(ValueError, match="URL scheme must be http or https"): + with pytest.raises(ValueError, match="must start with http:// or https://"): noema.call_llm("owner/repo", 1, pr, "diff", False) # Test localhost rejection diff --git a/tests/test_opencode_agent_contract.py b/tests/test_opencode_agent_contract.py index 4777ef8bf..ece0a6bc1 100644 --- a/tests/test_opencode_agent_contract.py +++ b/tests/test_opencode_agent_contract.py @@ -69,28 +69,16 @@ def test_opencode_model_pool_sets_high_effort_for_capable_candidates(): """Guard every review-pool candidate against silent reasoning-effort drift.""" config = json.loads(Path("opencode.jsonc").read_text(encoding="utf-8")) workflow = Path(".github/workflows/opencode-review.yml").read_text(encoding="utf-8") - github_models = config["provider"]["github-models"]["models"] + models = config["provider"]["github-models"]["models"] candidates_match = re.search(r'OPENCODE_MODEL_CANDIDATES: "([^"]+)"', workflow) assert candidates_match is not None candidates = candidates_match.group(1).split() - candidate_pairs = [candidate.split("/", 1) for candidate in candidates] - direct_openai_models = [ - model_name for provider, model_name in candidate_pairs if provider == "openai" - ] - github_candidate_models = [ - model_name for provider, model_name in candidate_pairs if provider == "github-models" - ] + candidate_models = [candidate.removeprefix("github-models/") for candidate in candidates] - assert candidate_pairs - assert candidate_pairs[:3] == [ - ["openai", "gpt-5-mini"], - ["openai", "gpt-5"], - ["github-models", "openai/o4-mini"], - ] - assert direct_openai_models == ["gpt-5-mini", "gpt-5"] - assert set(github_candidate_models).issubset(set(github_models)) - assert github_candidate_models[:3] == [ + assert candidate_models + assert set(candidate_models).issubset(set(models)) + assert candidate_models[:3] == [ "openai/o4-mini", "openai/o3-mini", "openai/gpt-5-mini", @@ -108,23 +96,20 @@ def test_opencode_model_pool_sets_high_effort_for_capable_candidates(): "mistral-ai/mistral-medium-2505", "meta/llama-4-maverick-17b-128e-instruct-fp8", "meta/llama-4-scout-17b-16e-instruct", - }.issubset(set(github_candidate_models)) - assert '"openai": {' in workflow - assert '"apiKey": "{env:OPENAI_API_KEY}"' in workflow - for model_name in direct_openai_models + github_candidate_models: + }.issubset(set(candidate_models)) + for model_name in candidate_models: assert f'"{model_name}": {{' in workflow def is_reasoning_capable(model_name: str) -> bool: return ( - model_name.startswith("gpt-5") - or model_name.startswith("openai/gpt-5") + model_name.startswith("openai/gpt-5") or model_name.startswith("openai/o3") or model_name.startswith("openai/o4") or model_name.startswith("deepseek/deepseek-r1") ) - for model_name in github_candidate_models: - model_config = github_models[model_name] + for model_name in candidate_models: + model_config = models[model_name] if is_reasoning_capable(model_name): assert model_config["reasoning"] is True, model_name assert model_config["options"]["reasoningEffort"] == "high", model_name @@ -291,20 +276,17 @@ def test_workflow_provisions_sandbox_tool_and_reviewer_agent(): assert "model pool was intentionally skipped" not in workflow assert "deterministic fallback" not in workflow assert "production source 또는 package manifest 변경이 없습니다" not in workflow - assert "needs.coverage-evidence.result != 'cancelled'" in workflow assert "request_changes_for_coverage_evidence_failure" in workflow assert '"## Review outcome"' in workflow assert '"## Check outcome"' not in workflow assert "publish REQUEST_CHANGES when coverage-evidence blocker states" in workflow - assert re.search(r"opencode-review-target:[\s\S]{0,520}timeout-minutes: 360", workflow) + assert re.search(r"opencode-review-target:[\s\S]{0,240}timeout-minutes: 360", workflow) assert 'timeout-minutes: 75' in workflow assert re.search(r"Run OpenCode PR Review model pool[\s\S]{0,240}timeout-minutes: 350", workflow) assert 'APPROVAL_CHECK_WAIT_ATTEMPTS: "81"' in workflow assert 'APPROVAL_CHECK_WAIT_SLEEP_SECONDS: "30"' in workflow assert ( - 'OPENCODE_MODEL_CANDIDATES: "openai/gpt-5-mini ' - "openai/gpt-5 " - "github-models/openai/o4-mini " + 'OPENCODE_MODEL_CANDIDATES: "github-models/openai/o4-mini ' "github-models/openai/o3-mini " "github-models/openai/gpt-5-mini " "github-models/openai/gpt-5-nano " @@ -319,15 +301,11 @@ def test_workflow_provisions_sandbox_tool_and_reviewer_agent(): 'github-models/openai/gpt-5"' ) in workflow assert 'OPENCODE_MODEL_ATTEMPTS: "1"' in workflow - assert 'OPENCODE_RUN_TIMEOUT_SECONDS: "900"' in workflow + assert 'OPENCODE_RUN_TIMEOUT_SECONDS: "5400"' in workflow assert 'OPENCODE_EXPORT_TIMEOUT_SECONDS: "120"' in workflow - assert 'OPENCODE_TOTAL_RETRY_BUDGET_SECONDS: "2700"' in workflow - assert 'OPENCODE_POOL_MAX_CYCLES: "1"' in workflow + assert 'OPENCODE_TOTAL_RETRY_BUDGET_SECONDS: "0"' in workflow assert 'OPENCODE_BACKOFF_MAX_SECONDS: "30"' in workflow assert "while :" in model_pool_runner - assert "should_skip_model_candidate" in model_pool_runner - assert "OPENAI_API_KEY is not configured" in model_pool_runner - assert "configured max cycle count" in model_pool_runner assert "OpenCode model pool has no configured model candidates." in model_pool_runner assert 'OPENCODE_TOTAL_RETRY_BUDGET_SECONDS:-18000' in model_pool_runner assert "completed a full model-candidate cycle without a valid control conclusion" in model_pool_runner @@ -434,7 +412,7 @@ def test_merge_scheduler_uses_escalating_mutation_credentials(): assert "steps.scheduler_app_token.outputs.token" in workflow assert "SCHEDULER_READ_TOKEN: ${{ github.token }}" in workflow assert "SCHEDULER_MUTATION_TOKEN_SOURCE" in workflow - assert 'default: "1"' in workflow + assert 'default: "-1"' in workflow assert 'review_dispatch_limit="-1"' in workflow