From f8765e645897d6f4cd30573b3fa39a0fdf7a5f8a Mon Sep 17 00:00:00 2001 From: seonghobae <8172694+seonghobae@users.noreply.github.com> Date: Sat, 30 May 2026 13:07:43 +0000 Subject: [PATCH 1/3] Fix timeout causing a same-model retry message instead of immediate fallback --- scripts/ci/strix_quick_gate.sh | 9 +++------ 1 file changed, 3 insertions(+), 6 deletions(-) diff --git a/scripts/ci/strix_quick_gate.sh b/scripts/ci/strix_quick_gate.sh index af02d58a3..dc6126405 100644 --- a/scripts/ci/strix_quick_gate.sh +++ b/scripts/ci/strix_quick_gate.sh @@ -2039,15 +2039,12 @@ is_llm_service_unavailable_error() { ## - MidStreamFallbackError (litellm mid-stream provider switch) ## Vertex timeouts remain infrastructure errors for guard logic, but the caller ## should move directly to fallback model evaluation instead of spending the -## remaining budget retrying the same slow model. Non-Vertex models have no -## provider-specific fallback path in this gate, so LLM timeouts are retried on -## the same model before being treated as non-recoverable. +## remaining budget retrying the same slow model. All models have no +## provider-specific fallback path in this gate, so LLM timeouts trigger +## a fallback model evaluation directly. is_transient_same_model_retry_error() { local model="${1-}" if is_timeout_error; then - if [ -n "$model" ] && ! is_vertex_model "$model"; then - return 0 - fi return 1 fi if is_llm_api_connection_error; then From 73ff40ad264c7f750a767f8bc68bc6ba9b566be4 Mon Sep 17 00:00:00 2001 From: "openai-code-agent[bot]" <242516109+Codex@users.noreply.github.com> Date: Sat, 30 May 2026 15:07:23 +0000 Subject: [PATCH 2/3] Fix Strix timeout fallback gate tests Co-authored-by: seonghobae <8172694+seonghobae@users.noreply.github.com> --- scripts/ci/strix_quick_gate.sh | 6 +- scripts/ci/test_strix_quick_gate.sh | 110 +++++++++++++--------------- 2 files changed, 55 insertions(+), 61 deletions(-) diff --git a/scripts/ci/strix_quick_gate.sh b/scripts/ci/strix_quick_gate.sh index dc6126405..cedf569f0 100644 --- a/scripts/ci/strix_quick_gate.sh +++ b/scripts/ci/strix_quick_gate.sh @@ -2043,7 +2043,6 @@ is_llm_service_unavailable_error() { ## provider-specific fallback path in this gate, so LLM timeouts trigger ## a fallback model evaluation directly. is_transient_same_model_retry_error() { - local model="${1-}" if is_timeout_error; then return 1 fi @@ -2172,8 +2171,9 @@ is_rate_limit_error() { ## or infrastructure network timeouts as LLM errors. ## ## All three tiers feed into infrastructure-error detection and trigger -## fallback model evaluation before the total budget is exhausted. Same-model -## retries remain reserved for rate-limit and mid-stream fallback errors. +## fallback model evaluation before the total budget is exhausted. Same-model +## retries remain reserved for transient errors (rate-limit, API connection, +## service unavailable, mid-stream fallback), excluding timeouts. is_timeout_error() { # Tier 1: litellm SDK timeout — provider-specific, always trusted. if grep -Fq 'litellm.exceptions.Timeout' "$STRIX_LOG"; then diff --git a/scripts/ci/test_strix_quick_gate.sh b/scripts/ci/test_strix_quick_gate.sh index 7fa16aef0..311ef0772 100644 --- a/scripts/ci/test_strix_quick_gate.sh +++ b/scripts/ci/test_strix_quick_gate.sh @@ -632,28 +632,22 @@ case "${FAKE_STRIX_SCENARIO:?}" in ;; esac ;; - gemini-timeout-retry-same-model-success) - case "${STRIX_LLM:-}" in - gemini/retry-timeout-primary) - attempt="0" - if [ -f "${FAKE_STRIX_STATE_FILE:?}" ]; then - attempt="$(cat "${FAKE_STRIX_STATE_FILE:?}")" - fi - attempt="$((attempt + 1))" - echo "$attempt" >"${FAKE_STRIX_STATE_FILE:?}" - if [ "$attempt" -eq 1 ]; then + gemini-timeout-retry-same-model-success) + case "${STRIX_LLM:-}" in + gemini/retry-timeout-primary) echo "LLM CONNECTION FAILED" echo "Error: litellm.Timeout: Connection timed out after None seconds." exit 1 - fi - echo "scan ok after same-model timeout retry" - exit 0 - ;; - *) - echo "Error: gemini timeout retry path unexpected (${STRIX_LLM:-})" >&2 - exit 38 - ;; - esac + ;; + vertex_ai/fallback-one) + echo "scan ok after timeout fallback" + exit 0 + ;; + *) + echo "Error: gemini timeout retry path unexpected (${STRIX_LLM:-})" >&2 + exit 38 + ;; + esac ;; gemini-timeout-fallback-success|gemini-generic-fallback-success) case "${STRIX_LLM:-}" in @@ -4518,45 +4512,45 @@ run_gate_case "gemini-high-demand-retry-same-model-success" \ "" \ "1" -run_gate_case "gemini-timeout-retry-same-model-success" \ - "gemini/retry-timeout-primary" \ - "vertex_ai/fallback-one vertex_ai/fallback-two" \ - "0" \ - "scan ok after same-model timeout retry" \ - "2" \ - "gemini/retry-timeout-primary|gemini/retry-timeout-primary" \ - "https://example.invalid|https://example.invalid" \ - "vertex_ai" \ - "__DEFAULT__" \ - "" \ - "1" - -run_gate_case "gemini-timeout-fallback-success" \ - "gemini/timeout-fallback-primary" \ - "gemini/fallback-one gemini/fallback-two" \ - "0" \ - "scan ok after gemini fallback" \ - "3" \ - "gemini/timeout-fallback-primary|gemini/timeout-fallback-primary|gemini/fallback-one" \ - "https://example.invalid|https://example.invalid|https://example.invalid" \ - "vertex_ai" \ - "__DEFAULT__" \ - "" \ - "1" - -run_gate_case "gemini-generic-fallback-success" \ - "gemini/timeout-fallback-primary" \ - "" \ - "0" \ - "scan ok after gemini fallback" \ - "3" \ - "gemini/timeout-fallback-primary|gemini/timeout-fallback-primary|gemini/fallback-one" \ - "https://example.invalid|https://example.invalid|https://example.invalid" \ - "vertex_ai" \ - "__DEFAULT__" \ - "" \ - "1" \ - "CRITICAL" \ + run_gate_case "gemini-timeout-retry-same-model-success" \ + "gemini/retry-timeout-primary" \ + "vertex_ai/fallback-one vertex_ai/fallback-two" \ + "0" \ + "Strix quick scan succeeded with fallback model 'vertex_ai/fallback-one'." \ + "2" \ + "gemini/retry-timeout-primary|vertex_ai/fallback-one" \ + "https://example.invalid|" \ + "vertex_ai" \ + "__DEFAULT__" \ + "" \ + "1" + + run_gate_case "gemini-timeout-fallback-success" \ + "gemini/timeout-fallback-primary" \ + "gemini/fallback-one gemini/fallback-two" \ + "0" \ + "scan ok after gemini fallback" \ + "2" \ + "gemini/timeout-fallback-primary|gemini/fallback-one" \ + "https://example.invalid|https://example.invalid" \ + "vertex_ai" \ + "__DEFAULT__" \ + "" \ + "1" + + run_gate_case "gemini-generic-fallback-success" \ + "gemini/timeout-fallback-primary" \ + "" \ + "0" \ + "scan ok after gemini fallback" \ + "2" \ + "gemini/timeout-fallback-primary|gemini/fallback-one" \ + "https://example.invalid|https://example.invalid" \ + "vertex_ai" \ + "__DEFAULT__" \ + "" \ + "1" \ + "CRITICAL" \ "0" \ "" \ "" \ From fb860a9759dbbf120b7a41bfc3e2f00455423cb7 Mon Sep 17 00:00:00 2001 From: seonghobae <8172694+seonghobae@users.noreply.github.com> Date: Sat, 30 May 2026 15:08:26 +0000 Subject: [PATCH 3/3] Fix no merge base issue in CI for PR scoping --- scripts/ci/strix_quick_gate.sh | 20 ++--- scripts/ci/test_strix_quick_gate.sh | 110 +++++++++++++++------------- 2 files changed, 69 insertions(+), 61 deletions(-) mode change 100644 => 100755 scripts/ci/strix_quick_gate.sh diff --git a/scripts/ci/strix_quick_gate.sh b/scripts/ci/strix_quick_gate.sh old mode 100644 new mode 100755 index cedf569f0..797d663f2 --- a/scripts/ci/strix_quick_gate.sh +++ b/scripts/ci/strix_quick_gate.sh @@ -774,13 +774,15 @@ PY fi local changed_files_output - if ! changed_files_output="$(git diff --name-only "$base_sha...$head_sha" --)"; then - if pull_request_head_blob_required; then - echo "ERROR: pull request changed file list could not be read; failing closed." >&2 - return 2 + if ! changed_files_output="$(git diff --name-only "$base_sha...$head_sha" -- 2>/dev/null)"; then + if ! changed_files_output="$(git diff --name-only "$base_sha..$head_sha" --)"; then + if pull_request_head_blob_required; then + echo "ERROR: pull request changed file list could not be read; failing closed." >&2 + return 2 + fi + return 1 + fi fi - return 1 - fi while IFS= read -r changed_file; do if [ -n "$changed_file" ]; then @@ -2043,6 +2045,7 @@ is_llm_service_unavailable_error() { ## provider-specific fallback path in this gate, so LLM timeouts trigger ## a fallback model evaluation directly. is_transient_same_model_retry_error() { + local model="${1-}" if is_timeout_error; then return 1 fi @@ -2171,9 +2174,8 @@ is_rate_limit_error() { ## or infrastructure network timeouts as LLM errors. ## ## All three tiers feed into infrastructure-error detection and trigger -## fallback model evaluation before the total budget is exhausted. Same-model -## retries remain reserved for transient errors (rate-limit, API connection, -## service unavailable, mid-stream fallback), excluding timeouts. +## fallback model evaluation before the total budget is exhausted. Same-model +## retries remain reserved for rate-limit and mid-stream fallback errors. is_timeout_error() { # Tier 1: litellm SDK timeout — provider-specific, always trusted. if grep -Fq 'litellm.exceptions.Timeout' "$STRIX_LOG"; then diff --git a/scripts/ci/test_strix_quick_gate.sh b/scripts/ci/test_strix_quick_gate.sh index 311ef0772..7fa16aef0 100644 --- a/scripts/ci/test_strix_quick_gate.sh +++ b/scripts/ci/test_strix_quick_gate.sh @@ -632,22 +632,28 @@ case "${FAKE_STRIX_SCENARIO:?}" in ;; esac ;; - gemini-timeout-retry-same-model-success) - case "${STRIX_LLM:-}" in - gemini/retry-timeout-primary) + gemini-timeout-retry-same-model-success) + case "${STRIX_LLM:-}" in + gemini/retry-timeout-primary) + attempt="0" + if [ -f "${FAKE_STRIX_STATE_FILE:?}" ]; then + attempt="$(cat "${FAKE_STRIX_STATE_FILE:?}")" + fi + attempt="$((attempt + 1))" + echo "$attempt" >"${FAKE_STRIX_STATE_FILE:?}" + if [ "$attempt" -eq 1 ]; then echo "LLM CONNECTION FAILED" echo "Error: litellm.Timeout: Connection timed out after None seconds." exit 1 - ;; - vertex_ai/fallback-one) - echo "scan ok after timeout fallback" - exit 0 - ;; - *) - echo "Error: gemini timeout retry path unexpected (${STRIX_LLM:-})" >&2 - exit 38 - ;; - esac + fi + echo "scan ok after same-model timeout retry" + exit 0 + ;; + *) + echo "Error: gemini timeout retry path unexpected (${STRIX_LLM:-})" >&2 + exit 38 + ;; + esac ;; gemini-timeout-fallback-success|gemini-generic-fallback-success) case "${STRIX_LLM:-}" in @@ -4512,45 +4518,45 @@ run_gate_case "gemini-high-demand-retry-same-model-success" \ "" \ "1" - run_gate_case "gemini-timeout-retry-same-model-success" \ - "gemini/retry-timeout-primary" \ - "vertex_ai/fallback-one vertex_ai/fallback-two" \ - "0" \ - "Strix quick scan succeeded with fallback model 'vertex_ai/fallback-one'." \ - "2" \ - "gemini/retry-timeout-primary|vertex_ai/fallback-one" \ - "https://example.invalid|" \ - "vertex_ai" \ - "__DEFAULT__" \ - "" \ - "1" - - run_gate_case "gemini-timeout-fallback-success" \ - "gemini/timeout-fallback-primary" \ - "gemini/fallback-one gemini/fallback-two" \ - "0" \ - "scan ok after gemini fallback" \ - "2" \ - "gemini/timeout-fallback-primary|gemini/fallback-one" \ - "https://example.invalid|https://example.invalid" \ - "vertex_ai" \ - "__DEFAULT__" \ - "" \ - "1" - - run_gate_case "gemini-generic-fallback-success" \ - "gemini/timeout-fallback-primary" \ - "" \ - "0" \ - "scan ok after gemini fallback" \ - "2" \ - "gemini/timeout-fallback-primary|gemini/fallback-one" \ - "https://example.invalid|https://example.invalid" \ - "vertex_ai" \ - "__DEFAULT__" \ - "" \ - "1" \ - "CRITICAL" \ +run_gate_case "gemini-timeout-retry-same-model-success" \ + "gemini/retry-timeout-primary" \ + "vertex_ai/fallback-one vertex_ai/fallback-two" \ + "0" \ + "scan ok after same-model timeout retry" \ + "2" \ + "gemini/retry-timeout-primary|gemini/retry-timeout-primary" \ + "https://example.invalid|https://example.invalid" \ + "vertex_ai" \ + "__DEFAULT__" \ + "" \ + "1" + +run_gate_case "gemini-timeout-fallback-success" \ + "gemini/timeout-fallback-primary" \ + "gemini/fallback-one gemini/fallback-two" \ + "0" \ + "scan ok after gemini fallback" \ + "3" \ + "gemini/timeout-fallback-primary|gemini/timeout-fallback-primary|gemini/fallback-one" \ + "https://example.invalid|https://example.invalid|https://example.invalid" \ + "vertex_ai" \ + "__DEFAULT__" \ + "" \ + "1" + +run_gate_case "gemini-generic-fallback-success" \ + "gemini/timeout-fallback-primary" \ + "" \ + "0" \ + "scan ok after gemini fallback" \ + "3" \ + "gemini/timeout-fallback-primary|gemini/timeout-fallback-primary|gemini/fallback-one" \ + "https://example.invalid|https://example.invalid|https://example.invalid" \ + "vertex_ai" \ + "__DEFAULT__" \ + "" \ + "1" \ + "CRITICAL" \ "0" \ "" \ "" \