From 3a733cd08924efddbfbf0734abc44ee7655d5785 Mon Sep 17 00:00:00 2001 From: Tomer Keshet Date: Thu, 16 Oct 2025 13:25:12 +0300 Subject: [PATCH 1/8] wip --- .../test_case.yaml | 2 +- .../bidder-app.yaml | 396 ++++++++++++++++++ .../prometheus-config.yaml | 17 + .../run-bidder-check.sh | 153 +++++++ .../test_case.yaml | 49 +++ .../toolsets.yaml | 9 + 6 files changed, 625 insertions(+), 1 deletion(-) create mode 100644 tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/bidder-app.yaml create mode 100644 tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/prometheus-config.yaml create mode 100755 tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/run-bidder-check.sh create mode 100644 tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/test_case.yaml create mode 100644 tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/toolsets.yaml diff --git a/tests/llm/fixtures/test_ask_holmes/124_checkout_latency_prometheus/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/124_checkout_latency_prometheus/test_case.yaml index d4488b54c9..fd9756bbe8 100644 --- a/tests/llm/fixtures/test_ask_holmes/124_checkout_latency_prometheus/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/124_checkout_latency_prometheus/test_case.yaml @@ -1,5 +1,5 @@ user_prompt: -- "The checkout service in namespace app-124 is experiencing high latency. Investigate why." +- "The /checkout endpoint in namespace app-124 is experiencing high latency. Investigate why." # Note telling holmes to focus on http_request_duration_seconds_bucket metric passes with sonnet - 'The checkout service in namespace app-124 is experiencing high latency. Investigate why. USE prometheus and http_request_duration_seconds_bucket' expected_output: diff --git a/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/bidder-app.yaml b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/bidder-app.yaml new file mode 100644 index 0000000000..3861452e76 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/bidder-app.yaml @@ -0,0 +1,396 @@ +# ------------------------------------------------------------------------------ +# Energy Market Bidding Service Test Environment +# +# This manifest sets up a test to validate that Holmes can detect when +# a specific energy exchange (NordPool) starts accepting 100% of bids +# instead of the normal 10% rate, combined with a traffic surge. +# +# Components: +# - Namespace: app-160 +# - Bidder Service: Python/Flask app that processes bid requests +# * Normal behavior: 10% bid acceptance rate (realistic for energy markets) +# * Bug: After 100 requests from NordPool, always accepts bids from NordPool +# * Exports Prometheus metrics with exchange and decision labels +# - k6 Job: Simulates traffic from 5 energy exchanges +# * Phase 1 (0-15s): Normal traffic distribution +# * Phase 2 (15s-45s): NordPool traffic increases 10x +# ------------------------------------------------------------------------------ +apiVersion: v1 +kind: Namespace +metadata: + name: app-160 + +--- +apiVersion: v1 +kind: Secret +metadata: + name: bidder-app + namespace: app-160 +type: Opaque +stringData: + app.py: | + import os, time, json, random, logging + from flask import Flask, request, jsonify + from prometheus_client import Counter, Gauge, Histogram, generate_latest, CONTENT_TYPE_LATEST + from threading import Lock + + app = Flask(__name__) + + # Disable logging for cleaner test + log = logging.getLogger('werkzeug') + log.setLevel(logging.ERROR) + app.logger.disabled = True + + # Business constants + NORMAL_BID_RATE = 0.10 # 10% bid rate is typical for energy markets + + # Bug tracking + nordpool_counter = 0 + nordpool_lock = Lock() + + # Prometheus metrics + bid_requests = Counter( + "bid_requests_total", + "Total bid requests received", + ["exchange", "decision"] + ) + + request_duration = Histogram( + "bid_request_duration_seconds", + "Duration of bid request processing", + ["exchange"], + buckets=[0.001, 0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1.0] + ) + + current_bid_rate = Gauge( + "bid_rate_current", + "Current bid acceptance rate by exchange", + ["exchange"] + ) + + # Track bid rates for gauge updates + exchange_stats = {} + stats_lock = Lock() + + def update_bid_rate(exchange, decision): + with stats_lock: + if exchange not in exchange_stats: + exchange_stats[exchange] = {"bid": 0, "no_bid": 0} + exchange_stats[exchange][decision] += 1 + total = exchange_stats[exchange]["bid"] + exchange_stats[exchange]["no_bid"] + if total > 0: + rate = exchange_stats[exchange]["bid"] / total + current_bid_rate.labels(exchange=exchange).set(rate) + + def should_bid(exchange): + global nordpool_counter + + # Bug logic for NordPool + if exchange == "NordPool": + with nordpool_lock: + nordpool_counter += 1 + if nordpool_counter > 100: + # Bug: Always bid after 100 requests + return True + + # Normal logic: 10% bid rate + return random.random() < NORMAL_BID_RATE + + @app.route("/metrics") + def metrics(): + return generate_latest(), 200, {"Content-Type": CONTENT_TYPE_LATEST} + + @app.route("/healthz") + def healthz(): + return "ok", 200 + + @app.route("/bid", methods=["POST"]) + def process_bid(): + start_time = time.perf_counter() + + # Extract request data + data = request.get_json() or {} + exchange = data.get("exchange", "unknown") + energy_id = data.get("energy_id", "") + + # Validate exchange + valid_exchanges = ["NordPool", "EPEX_SPOT", "EEX", "OMIE", "GME"] + if exchange not in valid_exchanges: + return jsonify({"error": "Invalid exchange"}), 400 + + # Make bid decision + decision = "bid" if should_bid(exchange) else "no_bid" + + # Update metrics + bid_requests.labels(exchange=exchange, decision=decision).inc() + update_bid_rate(exchange, decision) + + # Simulate processing time + process_time = random.uniform(0.005, 0.025) + time.sleep(process_time) + + # Record request duration + duration = time.perf_counter() - start_time + request_duration.labels(exchange=exchange).observe(duration) + + # Return response + response = { + "exchange": exchange, + "energy_id": energy_id, + "decision": decision, + "timestamp": int(time.time() * 1000) + } + + return jsonify(response), 200 + + if __name__ == "__main__": + app.run(host="0.0.0.0", port=8080) + +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: bidder + namespace: app-160 +spec: + replicas: 1 + selector: + matchLabels: + app: bidder + template: + metadata: + labels: + app: bidder + annotations: + prometheus.io/scrape: "true" + prometheus.io/path: "/metrics" + prometheus.io/port: "8080" + spec: + containers: + - name: app + image: python:3.11-slim + imagePullPolicy: IfNotPresent + command: ["/bin/sh", "-c"] + args: + - pip install --no-cache-dir flask prometheus_client && python /app/app.py + ports: + - containerPort: 8080 + name: http + volumeMounts: + - name: app-code + mountPath: /app + readinessProbe: + httpGet: + path: /healthz + port: 8080 + initialDelaySeconds: 5 + periodSeconds: 5 + timeoutSeconds: 2 + failureThreshold: 3 + livenessProbe: + httpGet: + path: /healthz + port: 8080 + initialDelaySeconds: 10 + periodSeconds: 10 + timeoutSeconds: 2 + failureThreshold: 3 + resources: + requests: + memory: "128Mi" + cpu: "100m" + limits: + memory: "256Mi" + cpu: "500m" + volumes: + - name: app-code + secret: + secretName: bidder-app + +--- +apiVersion: v1 +kind: Service +metadata: + name: bidder + namespace: app-160 + labels: + app: bidder +spec: + selector: + app: bidder + ports: + - name: http + port: 80 + targetPort: 8080 + +--- +apiVersion: v1 +kind: Secret +metadata: + name: k6-script + namespace: app-160 +type: Opaque +stringData: + test.js: | + import http from 'k6/http'; + import { check, sleep } from 'k6'; + import { Rate } from 'k6/metrics'; + + const bidAcceptanceRate = new Rate('bid_acceptance_rate'); + + export const options = { + scenarios: { + // Normal traffic for first 15 seconds (enough for bug to trigger at ~6 req/s for NordPool) + normal_traffic: { + executor: 'constant-arrival-rate', + rate: 30, // 30 requests per second total + timeUnit: '1s', + duration: '15s', + preAllocatedVUs: 30, + exec: 'normalTraffic', + }, + // After 15 seconds, surge NordPool traffic + nordpool_surge: { + executor: 'constant-arrival-rate', + rate: 60, // 60 requests per second from NordPool alone + timeUnit: '1s', + startTime: '15s', + duration: '30s', + preAllocatedVUs: 60, + exec: 'nordpoolSurge', + }, + // Continue normal traffic for other exchanges + continued_normal: { + executor: 'constant-arrival-rate', + rate: 20, // Reduced rate for other exchanges + timeUnit: '1s', + startTime: '15s', + duration: '30s', + preAllocatedVUs: 20, + exec: 'otherExchanges', + }, + }, + thresholds: { + 'http_req_duration': ['p(95)<100'], + 'http_req_failed': ['rate<0.01'], + }, + }; + + const exchanges = ['NordPool', 'EPEX_SPOT', 'EEX', 'OMIE', 'GME']; + const exchangeWeights = { + 'NordPool': 0.20, + 'EPEX_SPOT': 0.25, + 'EEX': 0.25, + 'OMIE': 0.15, + 'GME': 0.15, + }; + + function selectExchange() { + const rand = Math.random(); + let cumulative = 0; + for (const [exchange, weight] of Object.entries(exchangeWeights)) { + cumulative += weight; + if (rand < cumulative) return exchange; + } + return 'NordPool'; + } + + function makeBidRequest(exchange) { + const url = 'http://bidder.app-160.svc.cluster.local/bid'; + const payload = JSON.stringify({ + exchange: exchange, + energy_id: `ENERGY_${Date.now()}_${Math.random().toString(36).substr(2, 9)}`, + quantity_mwh: Math.floor(Math.random() * 100) + 10, + price_per_mwh: Math.floor(Math.random() * 50) + 30, + }); + + const params = { + headers: { 'Content-Type': 'application/json' }, + tags: { exchange: exchange }, + }; + + const res = http.post(url, payload, params); + + check(res, { + 'status is 200': (r) => r.status === 200, + 'has decision': (r) => { + const body = JSON.parse(r.body); + return body.decision === 'bid' || body.decision === 'no_bid'; + }, + }); + + // Track bid acceptance + if (res.status === 200) { + const body = JSON.parse(res.body); + bidAcceptanceRate.add(body.decision === 'bid'); + } + + return res; + } + + export function normalTraffic() { + const exchange = selectExchange(); + makeBidRequest(exchange); + sleep(0.01); + } + + export function nordpoolSurge() { + makeBidRequest('NordPool'); + sleep(0.01); + } + + export function otherExchanges() { + const otherExchanges = exchanges.filter(e => e !== 'NordPool'); + const exchange = otherExchanges[Math.floor(Math.random() * otherExchanges.length)]; + makeBidRequest(exchange); + sleep(0.01); + } + +--- +apiVersion: batch/v1 +kind: Job +metadata: + name: k6-energy-market + namespace: app-160 +spec: + ttlSecondsAfterFinished: 300 + template: + spec: + restartPolicy: Never + initContainers: + - name: wait-for-bidder + image: curlimages/curl:8.8.0 + command: ["sh", "-c"] + args: + - | + echo "Waiting for bidder service to be ready..." + until curl -fsS http://bidder.app-160.svc.cluster.local/healthz; do + echo "Bidder not ready, waiting..." + sleep 2 + done + echo "Bidder service is ready!" + - name: wait-for-prometheus + image: curlimages/curl:8.8.0 + command: ["sh", "-c"] + args: + - | + echo "Waiting for Prometheus to be ready..." + until curl -fsS http://prometheus.app-160.svc.cluster.local:9090/-/ready; do + echo "Prometheus not ready, waiting..." + sleep 2 + done + echo "Prometheus is ready!" + containers: + - name: k6 + image: grafana/k6:0.49.0 + args: ["run", "/scripts/test.js"] + volumeMounts: + - name: script + mountPath: /scripts + volumes: + - name: script + secret: + secretName: k6-script + items: + - key: test.js + path: test.js diff --git a/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/prometheus-config.yaml b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/prometheus-config.yaml new file mode 100644 index 0000000000..75cf244067 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/prometheus-config.yaml @@ -0,0 +1,17 @@ +# Prometheus configuration for test 160 +--- +apiVersion: v1 +kind: ConfigMap +metadata: + name: prometheus-config + namespace: app-160 +data: + prometheus.yml: | + global: + scrape_interval: 5s + evaluation_interval: 5s + scrape_configs: + - job_name: 'bidder' + metrics_path: /metrics + static_configs: + - targets: ['bidder.app-160.svc.cluster.local:80'] diff --git a/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/run-bidder-check.sh b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/run-bidder-check.sh new file mode 100755 index 0000000000..0c7b126855 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/run-bidder-check.sh @@ -0,0 +1,153 @@ +#!/usr/bin/env bash + +set -euo pipefail + +# Energy Market Bidding Bug Validation Script +# Validates that: +# 1. NordPool bid rate changes from ~10% to ~100% +# 2. Other exchanges maintain ~10% bid rate +# 3. NordPool traffic increases by ~10x +# 4. Combined effect results in ~100x increase in NordPool bid volume + +NS="${NS:-app-160}" +JOB="${JOB:-k6-energy-market}" +WAIT_TIMEOUT="${WAIT_TIMEOUT:-90s}" +PROM_URL="${PROM_URL:-http://prometheus.${NS}.svc.cluster.local:9090}" + +# Prometheus queries - using 30s windows for faster test +PROMQL_NORDPOOL_BID_RATE='sum(rate(bid_requests_total{exchange="NordPool",decision="bid"}[30s])) / sum(rate(bid_requests_total{exchange="NordPool"}[30s]))' +PROMQL_OTHER_BID_RATE='sum(rate(bid_requests_total{exchange!="NordPool",decision="bid"}[30s])) / sum(rate(bid_requests_total{exchange!="NordPool"}[30s]))' +PROMQL_NORDPOOL_RPS='sum(rate(bid_requests_total{exchange="NordPool"}[30s]))' +PROMQL_OTHER_RPS='sum(rate(bid_requests_total{exchange!="NordPool"}[30s]))' +PROMQL_NORDPOOL_BIDS_PER_SEC='sum(rate(bid_requests_total{exchange="NordPool",decision="bid"}[30s]))' + +echo ">>> Waiting for Job/${JOB} to complete (timeout: ${WAIT_TIMEOUT})" +if ! kubectl -n "${NS}" wait --for=condition=complete "job/${JOB}" --timeout="${WAIT_TIMEOUT}"; then + echo "ERROR: Job/${JOB} did not complete in time." >&2 + kubectl -n "${NS}" get job "${JOB}" -o yaml + kubectl -n "${NS}" logs -l job-name="${JOB}" --tail=50 + exit 1 +fi + +# Run curl inside the cluster to query Prometheus +curl_query() { + local q="$1" + kubectl run curl-$$ \ + --image=curlimages/curl:8.8.0 \ + --restart=Never --rm -i --quiet -- \ + sh -c "curl -sS --get '${PROM_URL}/api/v1/query' --data-urlencode 'query=${q}'" 2>&1 | \ + awk 'match($0, /^[[:space:]]*{/) { print; exit }' +} + +# Retry helper for Prometheus queries +fetch_json() { + local q="$1" json="" attempt=1 + + while [ $attempt -le 15 ]; do + json="$(curl_query "$q" 2>/dev/null || echo "")" + + # Check if we got valid JSON with success status + if printf '%s' "$json" | jq -e '.status == "success"' >/dev/null 2>&1; then + echo "$json" + return 0 + fi + + echo "Waiting for Prometheus data (attempt $attempt/15)..." >&2 + sleep 2 + attempt=$((attempt+1)) + done + + # Return last attempt even if failed + echo "$json" +} + +# Extract numeric value from Prometheus response +extract_val() { + local json="$1" + local val + val="$(printf '%s' "$json" | jq -r '.data.result[0].value[1] // empty' 2>/dev/null)" || true + if [ -z "${val}" ] || [ "${val}" = "null" ] || [ "${val}" = "NaN" ]; then + echo "0" + return + fi + printf '%s' "$val" +} + +echo ">>> Fetching metrics from Prometheus" + +# Get all metrics +echo "Fetching NordPool bid rate..." +nordpool_bid_rate_json="$(fetch_json "$PROMQL_NORDPOOL_BID_RATE")" +nordpool_bid_rate="$(extract_val "$nordpool_bid_rate_json")" + +echo "Fetching other exchanges bid rate..." +other_bid_rate_json="$(fetch_json "$PROMQL_OTHER_BID_RATE")" +other_bid_rate="$(extract_val "$other_bid_rate_json")" + +echo "Fetching NordPool request rate..." +nordpool_rps_json="$(fetch_json "$PROMQL_NORDPOOL_RPS")" +nordpool_rps="$(extract_val "$nordpool_rps_json")" + +echo "Fetching other exchanges request rate..." +other_rps_json="$(fetch_json "$PROMQL_OTHER_RPS")" +other_rps="$(extract_val "$other_rps_json")" + +echo "Fetching NordPool bids per second..." +nordpool_bids_per_sec_json="$(fetch_json "$PROMQL_NORDPOOL_BIDS_PER_SEC")" +nordpool_bids_per_sec="$(extract_val "$nordpool_bids_per_sec_json")" + +# Display results +echo "" +echo "=== METRICS SUMMARY ===" +echo "NordPool bid rate: ${nordpool_bid_rate} (expected: ~1.0 after bug)" +echo "Other exchanges bid rate: ${other_bid_rate} (expected: ~0.1)" +echo "NordPool request rate: ${nordpool_rps} req/s" +echo "Other exchanges request rate: ${other_rps} req/s" +echo "NordPool bids per second: ${nordpool_bids_per_sec}" + +# Validate the bug is present +echo "" +echo ">>> Validating test conditions..." + +# Check 1: NordPool bid rate should be close to 100% +if awk -v rate="$nordpool_bid_rate" 'BEGIN{exit !(rate >= 0.95)}'; then + echo "✓ NordPool bid rate is ~100% (actual: ${nordpool_bid_rate})" +else + echo "✗ ERROR: NordPool bid rate is not ~100% (actual: ${nordpool_bid_rate})" >&2 + echo "Debug info - NordPool bid rate response:" + echo "$nordpool_bid_rate_json" | jq '.' 2>/dev/null || echo "$nordpool_bid_rate_json" + exit 2 +fi + +# Check 2: Other exchanges should maintain ~10% bid rate +if awk -v rate="$other_bid_rate" 'BEGIN{exit !(rate >= 0.08 && rate <= 0.15)}'; then + echo "✓ Other exchanges bid rate is ~10% (actual: ${other_bid_rate})" +else + echo "✗ ERROR: Other exchanges bid rate is not ~10% (actual: ${other_bid_rate})" >&2 + echo "Debug info - Other bid rate response:" + echo "$other_bid_rate_json" | jq '.' 2>/dev/null || echo "$other_bid_rate_json" + exit 3 +fi + +# Check 3: NordPool should have significant traffic (at least 10 req/s) +if awk -v rate="$nordpool_rps" 'BEGIN{exit !(rate >= 10)}'; then + echo "✓ NordPool has significant traffic (${nordpool_rps} req/s)" +else + echo "✗ ERROR: NordPool traffic too low (${nordpool_rps} req/s)" >&2 + exit 4 +fi + +# Check 4: Calculate bid volume increase +# Expected: ~10x traffic increase * 10x bid rate increase = ~100x bid volume increase +expected_baseline_bids="0.1" # Rough estimate: baseline would be ~10% of baseline traffic +bid_volume_factor=$(awk -v bps="$nordpool_bids_per_sec" -v base="$expected_baseline_bids" 'BEGIN{printf "%.1f", bps/base}') + +echo "✓ NordPool bid volume increased by approximately ${bid_volume_factor}x" + +echo "" +echo ">>> PASS: Energy market bidding bug successfully reproduced!" +echo " - NordPool bid rate jumped from ~10% to ~100%" +echo " - Other exchanges maintained normal ~10% bid rate" +echo " - Combined with traffic surge, bid volume increased dramatically" + +exit 0 diff --git a/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/test_case.yaml new file mode 100644 index 0000000000..25cc53ea62 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/test_case.yaml @@ -0,0 +1,49 @@ +user_prompt: +# - "The energy bidding service in namespace app-160 is experiencing unusual bid patterns. Investigate why bid commitments have increased dramatically." +- "Is there issues with the bidding rate in the bidding endpoint in namespace app-160? Note: Normal rate should be around 10%, and we are working with a very small resolution of seconds" + +expected_output: + - The answer must identify that NordPool exchange has an abnormally high bid acceptance rate (close to 100%) compared to other exchanges (~10%) + - The answer should mention that this behavior started after a certain point in time or after a certain number of requests + - The answer should note the correlation with increased traffic from NordPool + +tags: + - kubernetes + - prometheus + - hard + - chain-of-causation + +port_forwards: + - namespace: app-160 + service: prometheus + local_port: 9090 + remote_port: 9090 + +before_test: | + set -e # Exit immediately if any command fails + + echo "📁 Creating namespace..." + kubectl create namespace app-160 --save-config --dry-run=client -o yaml | kubectl apply -f - || true + + echo "⚙️ Setting up Prometheus..." + kubectl apply -f prometheus-config.yaml -n app-160 + kubectl apply -f ../../shared/prometheus.yaml -n app-160 + + echo "🚀 Deploying bidder service and k6 test..." + kubectl apply -f bidder-app.yaml -n app-160 + + echo "⏳ Waiting for pods to be ready..." + kubectl wait --for=condition=ready pod -l app=prometheus -n app-160 --timeout=60s + kubectl wait --for=condition=ready pod -l app=bidder -n app-160 --timeout=60s + + echo "🔍 Running validation..." + ./run-bidder-check.sh + + echo "🧹 Cleaning up k6 job and pods to prevent AI detection" + kubectl delete job k6-energy-market -n app-160 --ignore-not-found=true + kubectl delete pods -n app-160 -l job-name=k6-energy-market --ignore-not-found=true + + echo "✅ Test setup completed successfully" + +after_test: | + kubectl delete namespace app-160 || true diff --git a/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/toolsets.yaml b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/toolsets.yaml new file mode 100644 index 0000000000..4a571a90c2 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/toolsets.yaml @@ -0,0 +1,9 @@ +toolsets: + kubernetes/core: + enabled: true + kubernetes/logs: + enabled: true + prometheus/metrics: + enabled: true + config: + prometheus_url: http://localhost:9090 From e9838ae26d8c868601e7220211735e3db4a74229 Mon Sep 17 00:00:00 2001 From: Tomer Keshet Date: Thu, 16 Oct 2025 14:50:06 +0300 Subject: [PATCH 2/8] emphasizing test data is limited and adding a runbook for 80% pass rate --- .../160_electricity_market_bidding_bug/bidding_system.md | 3 +++ .../160_electricity_market_bidding_bug/test_case.yaml | 5 ++++- 2 files changed, 7 insertions(+), 1 deletion(-) create mode 100644 tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/bidding_system.md diff --git a/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/bidding_system.md b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/bidding_system.md new file mode 100644 index 0000000000..85a1509fa9 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/bidding_system.md @@ -0,0 +1,3 @@ +- Normal bid rate should be around 10% +- Problems with bid rate and traffic can change in a matter of seconds, use the highest resolution possible +- The total bid request traffic should usually be consistent. diff --git a/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/test_case.yaml index 25cc53ea62..fbd6068980 100644 --- a/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/test_case.yaml @@ -1,6 +1,6 @@ user_prompt: # - "The energy bidding service in namespace app-160 is experiencing unusual bid patterns. Investigate why bid commitments have increased dramatically." -- "Is there issues with the bidding rate in the bidding endpoint in namespace app-160? Note: Normal rate should be around 10%, and we are working with a very small resolution of seconds" +- "Is there issues with the bidding rate in the bidding endpoint in namespace app-160? Check the recent metrics - note that the traffic test has already completed, so you'll see historical data that eventually drops to zero, which is expected." expected_output: - The answer must identify that NordPool exchange has an abnormally high bid acceptance rate (close to 100%) compared to other exchanges (~10%) @@ -47,3 +47,6 @@ before_test: | after_test: | kubectl delete namespace app-160 || true + +include_files: + - bidding_system.md From 44ab558e5e72c5e006f634cc828b91d5493b3773 Mon Sep 17 00:00:00 2001 From: Tomer Keshet Date: Tue, 21 Oct 2025 13:09:44 +0300 Subject: [PATCH 3/8] add a test of slowness and scaling --- .../bidder-v1.yaml | 150 +++++++++++++ .../bidder-v2.yaml | 139 ++++++++++++ .../161_bidding_version_performance/hpa.yaml | 38 ++++ .../k6-v1-traffic.yaml | 103 +++++++++ .../k6-v2-traffic.yaml | 103 +++++++++ .../prometheus-config.yaml | 15 ++ .../run-performance-check.sh | 207 ++++++++++++++++++ .../test_case.yaml | 68 ++++++ .../toolsets.yaml | 9 + .../wait-for-scaling.sh | 42 ++++ 10 files changed, 874 insertions(+) create mode 100644 tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v1.yaml create mode 100644 tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml create mode 100644 tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/hpa.yaml create mode 100644 tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/k6-v1-traffic.yaml create mode 100644 tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/k6-v2-traffic.yaml create mode 100644 tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/prometheus-config.yaml create mode 100755 tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/run-performance-check.sh create mode 100644 tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/test_case.yaml create mode 100644 tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/toolsets.yaml create mode 100755 tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/wait-for-scaling.sh diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v1.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v1.yaml new file mode 100644 index 0000000000..d334907692 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v1.yaml @@ -0,0 +1,150 @@ +apiVersion: v1 +kind: Secret +metadata: + name: bidder-app-v1 + namespace: app-161 +type: Opaque +stringData: + app.py: | + from flask import Flask, Response, jsonify, request + from prometheus_client import Counter, Histogram, Gauge, Info, generate_latest, CONTENT_TYPE_LATEST + import time + import random + import logging + + app = Flask(__name__) + + # Disable logging for cleaner test + log = logging.getLogger('werkzeug') + log.setLevel(logging.ERROR) + app.logger.disabled = True + + # Version info + VERSION = "v1.0" + + # Prometheus metrics + bid_requests = Counter( + 'bid_requests_total', + 'Total bid requests processed', + ['exchange', 'decision', 'version'] + ) + + request_duration = Histogram( + 'bid_request_duration_seconds', + 'Bid request processing duration', + ['exchange', 'version'], + buckets=[0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1.0, 2.5, 5.0] + ) + + build_info = Info('bidder_build', 'Build information') + build_info.info({'version': VERSION, 'commit': 'abc123', 'branch': 'main'}) + + @app.route('/healthz') + def health(): + return 'ok' + + @app.route('/bid', methods=['POST']) + def process_bid(): + # v1.0 - Fast processing + processing_time = 0.04 + random.uniform(0, 0.02) # 40-60ms + time.sleep(processing_time) + + data = request.get_json() or {} + exchange = data.get('exchange', 'NordPool') + + # 10% bid rate (realistic for energy markets) + decision = 'bid' if random.random() < 0.1 else 'no_bid' + + bid_requests.labels(exchange=exchange, decision=decision, version=VERSION).inc() + request_duration.labels(exchange=exchange, version=VERSION).observe(processing_time) + + return jsonify({ + 'decision': decision, + 'version': VERSION, + 'processing_time': processing_time + }) + + @app.route('/metrics') + def metrics(): + return Response(generate_latest(), mimetype=CONTENT_TYPE_LATEST) + + if __name__ == '__main__': + app.run(host='0.0.0.0', port=8080) + +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: bidder + namespace: app-161 + labels: + app: bidder + version: v1.0 + annotations: + kubernetes.io/change-cause: "Initial deployment of energy bidding service v1.0" +spec: + replicas: 2 + selector: + matchLabels: + app: bidder + template: + metadata: + labels: + app: bidder + version: v1.0 + annotations: + prometheus.io/scrape: "true" + prometheus.io/port: "8080" + prometheus.io/path: "/metrics" + spec: + containers: + - name: bidder + image: python:3.11-slim + command: ["/bin/sh", "-c"] + args: + - pip install flask prometheus_client && python /app/app.py + ports: + - containerPort: 8080 + name: http + volumeMounts: + - name: app-code + mountPath: /app + resources: + requests: + memory: "128Mi" + cpu: "100m" + limits: + memory: "256Mi" + cpu: "500m" + readinessProbe: + httpGet: + path: /healthz + port: 8080 + initialDelaySeconds: 10 + periodSeconds: 5 + livenessProbe: + httpGet: + path: /healthz + port: 8080 + initialDelaySeconds: 15 + periodSeconds: 10 + volumes: + - name: app-code + secret: + secretName: bidder-app-v1 + +--- +apiVersion: v1 +kind: Service +metadata: + name: bidder + namespace: app-161 + labels: + app: bidder +spec: + selector: + app: bidder + ports: + - port: 80 + targetPort: 8080 + name: http diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml new file mode 100644 index 0000000000..900c2b13a9 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml @@ -0,0 +1,139 @@ +apiVersion: v1 +kind: Secret +metadata: + name: bidder-app-v2 + namespace: app-161 +type: Opaque +stringData: + app.py: | + from flask import Flask, Response, jsonify, request + from prometheus_client import Counter, Histogram, Gauge, Info, generate_latest, CONTENT_TYPE_LATEST + import time + import random + import logging + + app = Flask(__name__) + + # Disable logging for cleaner test + log = logging.getLogger('werkzeug') + log.setLevel(logging.ERROR) + app.logger.disabled = True + + # Version info + VERSION = "v2.0" + + # Prometheus metrics + bid_requests = Counter( + 'bid_requests_total', + 'Total bid requests processed', + ['exchange', 'decision', 'version'] + ) + + request_duration = Histogram( + 'bid_request_duration_seconds', + 'Bid request processing duration', + ['exchange', 'version'], + buckets=[0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1.0, 2.5, 5.0] + ) + + build_info = Info('bidder_build', 'Build information') + build_info.info({'version': VERSION, 'commit': 'def456', 'branch': 'main'}) + + @app.route('/healthz') + def health(): + return 'ok' + + @app.route('/bid', methods=['POST']) + def process_bid(): + # v2.0 - Slow processing due to new "enhanced" algorithm + processing_time = 1.8 + random.uniform(0, 0.4) # 1.8-2.2s + time.sleep(processing_time) + + data = request.get_json() or {} + exchange = data.get('exchange', 'NordPool') + + # 10% bid rate (realistic for energy markets) + decision = 'bid' if random.random() < 0.1 else 'no_bid' + + bid_requests.labels(exchange=exchange, decision=decision, version=VERSION).inc() + request_duration.labels(exchange=exchange, version=VERSION).observe(processing_time) + + return jsonify({ + 'decision': decision, + 'version': VERSION, + 'processing_time': processing_time + }) + + @app.route('/metrics') + def metrics(): + return Response(generate_latest(), mimetype=CONTENT_TYPE_LATEST) + + if __name__ == '__main__': + app.run(host='0.0.0.0', port=8080) + +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: bidder + namespace: app-161 + labels: + app: bidder + version: v2.0 + annotations: + kubernetes.io/change-cause: "Updated to v2.0 with enhanced bid processing algorithm" +spec: + replicas: 2 # Start with same replica count + selector: + matchLabels: + app: bidder + strategy: + type: RollingUpdate + rollingUpdate: + maxSurge: 1 + maxUnavailable: 0 + template: + metadata: + labels: + app: bidder + version: v2.0 + annotations: + prometheus.io/scrape: "true" + prometheus.io/port: "8080" + prometheus.io/path: "/metrics" + spec: + containers: + - name: bidder + image: python:3.11-slim + command: ["/bin/sh", "-c"] + args: + - pip install flask prometheus_client && python /app/app.py + ports: + - containerPort: 8080 + name: http + volumeMounts: + - name: app-code + mountPath: /app + resources: + requests: + memory: "128Mi" + cpu: "100m" + limits: + memory: "256Mi" + cpu: "500m" + readinessProbe: + httpGet: + path: /healthz + port: 8080 + initialDelaySeconds: 10 + periodSeconds: 5 + livenessProbe: + httpGet: + path: /healthz + port: 8080 + initialDelaySeconds: 15 + periodSeconds: 10 + volumes: + - name: app-code + secret: + secretName: bidder-app-v2 diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/hpa.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/hpa.yaml new file mode 100644 index 0000000000..71dd8debdd --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/hpa.yaml @@ -0,0 +1,38 @@ +apiVersion: autoscaling/v2 +kind: HorizontalPodAutoscaler +metadata: + name: bidder-hpa + namespace: app-161 + labels: + app: bidder +spec: + scaleTargetRef: + apiVersion: apps/v1 + kind: Deployment + name: bidder + minReplicas: 2 + maxReplicas: 10 + metrics: + - type: Resource + resource: + name: cpu + target: + type: Utilization + averageUtilization: 50 + behavior: + scaleUp: + stabilizationWindowSeconds: 30 + policies: + - type: Percent + value: 100 + periodSeconds: 15 + - type: Pods + value: 2 + periodSeconds: 15 + selectPolicy: Max + scaleDown: + stabilizationWindowSeconds: 300 + policies: + - type: Percent + value: 10 + periodSeconds: 60 diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/k6-v1-traffic.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/k6-v1-traffic.yaml new file mode 100644 index 0000000000..e7b4632bc9 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/k6-v1-traffic.yaml @@ -0,0 +1,103 @@ +apiVersion: v1 +kind: Secret +metadata: + name: k6-v1-script + namespace: app-161 +type: Opaque +stringData: + test.js: | + import http from 'k6/http'; + import { check, sleep } from 'k6'; + import { Rate } from 'k6/metrics'; + + const bidAcceptanceRate = new Rate('bid_acceptance_rate'); + + export const options = { + scenarios: { + constant_load: { + executor: 'constant-arrival-rate', + rate: 100, // 100 requests per second + timeUnit: '1s', + duration: '2m', // 2 minutes to build up history + preAllocatedVUs: 150, + }, + }, + thresholds: { + 'http_req_duration': ['p(95)<200'], // v1.0 should be fast + 'http_req_failed': ['rate<0.01'], + }, + }; + + const exchanges = ['NordPool', 'EPEX_SPOT', 'EEX', 'OMIE', 'GME']; + + function selectExchange() { + return exchanges[Math.floor(Math.random() * exchanges.length)]; + } + + export default function() { + const exchange = selectExchange(); + const url = 'http://bidder.app-161.svc.cluster.local/bid'; + + const payload = JSON.stringify({ + exchange: exchange, + amount: Math.floor(Math.random() * 1000) + 100, + price: Math.random() * 100 + 20, + }); + + const params = { + headers: { + 'Content-Type': 'application/json', + }, + timeout: '5s', + }; + + const res = http.post(url, payload, params); + + check(res, { + 'status is 200': (r) => r.status === 200, + 'has version': (r) => r.json('version') !== undefined, + 'version is v1.0': (r) => r.json('version') === 'v1.0', + }); + + if (res.status === 200 && res.json('decision') === 'bid') { + bidAcceptanceRate.add(1); + } else { + bidAcceptanceRate.add(0); + } + } + +--- +apiVersion: batch/v1 +kind: Job +metadata: + name: k6-v1-traffic + namespace: app-161 + labels: + app: k6 + phase: v1 +spec: + template: + metadata: + labels: + app: k6 + phase: v1 + spec: + restartPolicy: Never + containers: + - name: k6 + image: loadimpact/k6:0.47.0 + args: ["run", "/scripts/test.js"] + volumeMounts: + - name: script + mountPath: /scripts + resources: + requests: + memory: "256Mi" + cpu: "500m" + limits: + memory: "512Mi" + cpu: "1000m" + volumes: + - name: script + secret: + secretName: k6-v1-script diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/k6-v2-traffic.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/k6-v2-traffic.yaml new file mode 100644 index 0000000000..189150e220 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/k6-v2-traffic.yaml @@ -0,0 +1,103 @@ +apiVersion: v1 +kind: Secret +metadata: + name: k6-v2-script + namespace: app-161 +type: Opaque +stringData: + test.js: | + import http from 'k6/http'; + import { check, sleep } from 'k6'; + import { Rate } from 'k6/metrics'; + + const bidAcceptanceRate = new Rate('bid_acceptance_rate'); + + export const options = { + scenarios: { + constant_load: { + executor: 'constant-arrival-rate', + rate: 100, // Same 100 requests per second + timeUnit: '1s', + duration: '2m', // 2 minutes to trigger scaling + preAllocatedVUs: 150, + }, + }, + thresholds: { + 'http_req_duration': ['p(95)<3000'], // v2.0 is slower + 'http_req_failed': ['rate<0.01'], + }, + }; + + const exchanges = ['NordPool', 'EPEX_SPOT', 'EEX', 'OMIE', 'GME']; + + function selectExchange() { + return exchanges[Math.floor(Math.random() * exchanges.length)]; + } + + export default function() { + const exchange = selectExchange(); + const url = 'http://bidder.app-161.svc.cluster.local/bid'; + + const payload = JSON.stringify({ + exchange: exchange, + amount: Math.floor(Math.random() * 1000) + 100, + price: Math.random() * 100 + 20, + }); + + const params = { + headers: { + 'Content-Type': 'application/json', + }, + timeout: '10s', // Increased timeout for slow v2.0 + }; + + const res = http.post(url, payload, params); + + check(res, { + 'status is 200': (r) => r.status === 200, + 'has version': (r) => r.json('version') !== undefined, + 'version is v2.0': (r) => r.json('version') === 'v2.0', + }); + + if (res.status === 200 && res.json('decision') === 'bid') { + bidAcceptanceRate.add(1); + } else { + bidAcceptanceRate.add(0); + } + } + +--- +apiVersion: batch/v1 +kind: Job +metadata: + name: k6-v2-traffic + namespace: app-161 + labels: + app: k6 + phase: v2 +spec: + template: + metadata: + labels: + app: k6 + phase: v2 + spec: + restartPolicy: Never + containers: + - name: k6 + image: loadimpact/k6:0.47.0 + args: ["run", "/scripts/test.js"] + volumeMounts: + - name: script + mountPath: /scripts + resources: + requests: + memory: "256Mi" + cpu: "500m" + limits: + memory: "512Mi" + cpu: "1000m" + volumes: + - name: script + secret: + secretName: k6-v2-script diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/prometheus-config.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/prometheus-config.yaml new file mode 100644 index 0000000000..b8fdec814c --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/prometheus-config.yaml @@ -0,0 +1,15 @@ +apiVersion: v1 +kind: ConfigMap +metadata: + name: prometheus-config + namespace: app-161 +data: + prometheus.yml: | + global: + scrape_interval: 5s + evaluation_interval: 5s + scrape_configs: + - job_name: 'bidder' + metrics_path: /metrics + static_configs: + - targets: ['bidder.app-161.svc.cluster.local:80'] diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/run-performance-check.sh b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/run-performance-check.sh new file mode 100755 index 0000000000..c34c1dbb8a --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/run-performance-check.sh @@ -0,0 +1,207 @@ +#!/usr/bin/env bash + +set -euo pipefail + +# Energy Market Bidding Performance Validation Script +# Validates that: +# 1. v1.0 metrics exist showing ~50ms latency +# 2. v2.0 metrics exist showing ~2s latency +# 3. Service has scaled from 2 to ~10 pods +# 4. Prometheus has captured the version transition + +NS="${NS:-app-161}" +PROM_URL="${PROM_URL:-http://prometheus.${NS}.svc.cluster.local:9090}" + +# Track validation failures +VALIDATION_FAILURES=0 + +# Prometheus queries +# Query over a longer time range to capture both v1.0 and v2.0 +PROMQL_V1_LATENCY='avg(avg_over_time(bid_request_duration_seconds_sum{version="v1.0"}[30m]) / avg_over_time(bid_request_duration_seconds_count{version="v1.0"}[30m]))' +PROMQL_V2_LATENCY='avg(avg_over_time(bid_request_duration_seconds_sum{version="v2.0"}[30m]) / avg_over_time(bid_request_duration_seconds_count{version="v2.0"}[30m]))' +# Use last_over_time to get the most recent value even for stale series +PROMQL_V1_REQUESTS='last_over_time(bid_requests_total{version="v1.0"}[30m])' +PROMQL_V2_REQUESTS='last_over_time(bid_requests_total{version="v2.0"}[30m])' +PROMQL_BUILD_INFO='last_over_time(bidder_build_info[30m])' + +echo ">>> Checking deployment status" +POD_COUNT=$(kubectl get deployment bidder -n "${NS}" -o jsonpath='{.status.readyReplicas}') +DEPLOYMENT_VERSION=$(kubectl get deployment bidder -n "${NS}" -o jsonpath='{.metadata.labels.version}') + +echo "Current pod count: ${POD_COUNT}" +echo "Current deployment version: ${DEPLOYMENT_VERSION}" + +if [ "$DEPLOYMENT_VERSION" != "v2.0" ]; then + echo "✗ ERROR: Expected version v2.0, found ${DEPLOYMENT_VERSION}" >&2 + exit 1 +fi + +if [ "$POD_COUNT" -lt 3 ]; then + echo "⚠️ WARNING: Expected at least 3 pods after scaling, found ${POD_COUNT}" + VALIDATION_FAILURES=$((VALIDATION_FAILURES + 1)) +fi + +echo "✓ Deployment is running v2.0 with ${POD_COUNT} pods" + +# Run curl inside the cluster to query Prometheus +curl_query() { + local q="$1" + # Run kubectl and extract JSON from the output + local output + output=$(kubectl run curl-$$ \ + --image=curlimages/curl:8.8.0 \ + --restart=Never --rm -i --quiet -- \ + sh -c "curl -sS --get '${PROM_URL}/api/v1/query' --data-urlencode 'query=${q}'" 2>&1) + + # Extract JSON - more robust extraction that handles kubectl output + # First remove any non-JSON prefix, then extract the JSON object + echo "$output" | grep -o '{"status":.*}' | head -1 +} + +# Retry helper for Prometheus queries +fetch_json() { + local q="$1" json="" attempt=1 + + while [ $attempt -le 15 ]; do + json="$(curl_query "$q" 2>/dev/null || echo "")" + + # Check if we got valid JSON with success status + if printf '%s' "$json" | jq -e '.status == "success"' >/dev/null 2>&1; then + echo "$json" + return 0 + fi + + echo "Waiting for Prometheus data (attempt $attempt/15)..." >&2 + sleep 2 + attempt=$((attempt+1)) + done + + # Return last attempt even if failed + echo "$json" +} + +# Extract numeric value from Prometheus response +extract_val() { + local json="$1" + local val + val="$(printf '%s' "$json" | jq -r '.data.result[0].value[1] // empty' 2>/dev/null)" || true + if [ -z "${val}" ] || [ "${val}" = "null" ] || [ "${val}" = "NaN" ]; then + echo "0" + return + fi + printf '%s' "$val" +} + +echo "" +echo ">>> Fetching metrics from Prometheus" + +# Check build info +echo "Checking build info metrics..." +build_info_json="$(fetch_json "$PROMQL_BUILD_INFO")" +v1_exists="$(printf '%s' "$build_info_json" | jq -r '.data.result[] | select(.metric.version == "v1.0") | .value[1]' 2>/dev/null)" || true +v2_exists="$(printf '%s' "$build_info_json" | jq -r '.data.result[] | select(.metric.version == "v2.0") | .value[1]' 2>/dev/null)" || true + +if [ -n "$v1_exists" ] && [ -n "$v2_exists" ]; then + echo "✓ Found metrics for both v1.0 and v2.0" +else + echo "⚠️ WARNING: Missing metrics for some versions (v1.0: ${v1_exists:-missing}, v2.0: ${v2_exists:-missing})" + # Mark as failure if v1.0 build info is missing + if [ -z "$v1_exists" ]; then + VALIDATION_FAILURES=$((VALIDATION_FAILURES + 1)) + fi +fi + +# Get v1.0 metrics +echo "Fetching v1.0 latency metrics..." +v1_latency_json="$(fetch_json "$PROMQL_V1_LATENCY")" +v1_latency="$(extract_val "$v1_latency_json")" + +echo "Fetching v1.0 request count..." +v1_requests_json="$(fetch_json "$PROMQL_V1_REQUESTS")" +v1_requests="$(extract_val "$v1_requests_json")" + +# Get v2.0 metrics +echo "Fetching v2.0 latency metrics..." +v2_latency_json="$(fetch_json "$PROMQL_V2_LATENCY")" +v2_latency="$(extract_val "$v2_latency_json")" + +echo "Fetching v2.0 request count..." +v2_requests_json="$(fetch_json "$PROMQL_V2_REQUESTS")" +v2_requests="$(extract_val "$v2_requests_json")" + +# Display results +echo "" +echo "=== METRICS SUMMARY ===" +echo "v1.0 average latency: ${v1_latency}s (expected: ~0.05s)" +echo "v1.0 total requests: ${v1_requests}" +echo "v2.0 average latency: ${v2_latency}s (expected: ~2.0s)" +echo "v2.0 total requests: ${v2_requests}" +echo "Current pod count: ${POD_COUNT} (expected: 8-10)" + +# Validate the test conditions +echo "" +echo ">>> Validating test conditions..." + +# Check 1: v1.0 should have had fast latency (~50ms) +if [ "$v1_requests" = "0" ]; then + echo "⚠️ WARNING: No v1.0 requests found - v1.0 metrics may have been lost during upgrade" + echo "This is expected if Prometheus hasn't retained the metrics long enough" + VALIDATION_FAILURES=$((VALIDATION_FAILURES + 1)) +else + if awk -v lat="$v1_latency" 'BEGIN{exit !(lat > 0.03 && lat < 0.08)}'; then + echo "✓ v1.0 latency was fast (~50ms): ${v1_latency}s" + else + echo "⚠️ WARNING: v1.0 latency outside expected range: ${v1_latency}s (expected 0.03-0.08s)" + VALIDATION_FAILURES=$((VALIDATION_FAILURES + 1)) + fi +fi + +# Check 2: v2.0 should have slow latency (~2s) +if [ "$v2_requests" = "0" ]; then + echo "✗ ERROR: No v2.0 requests found - v2.0 metrics missing" >&2 + # Keep critical early exit for v2.0 missing + exit 3 +fi + +if awk -v lat="$v2_latency" 'BEGIN{exit !(lat > 1.5 && lat < 2.5)}'; then + echo "✓ v2.0 latency is slow (~2s): ${v2_latency}s" +else + echo "✗ ERROR: v2.0 latency not in expected range: ${v2_latency}s (expected 1.5-2.5s)" >&2 + # Keep critical early exit for v2.0 latency out of range + exit 4 +fi + +# Check 3: Performance degradation factor +if [ "$v1_latency" != "0" ] && [ "$v1_requests" != "0" ]; then + degradation_factor=$(awk -v v2="$v2_latency" -v v1="$v1_latency" 'BEGIN{printf "%.1f", v2/v1}') + echo "✓ Performance degraded by ${degradation_factor}x (v2.0 is ${degradation_factor}x slower than v1.0)" +else + echo "⚠️ Cannot calculate degradation factor without v1.0 metrics" +fi + +# Check 4: Scaling occurred +if [ "$POD_COUNT" -ge 3 ]; then + echo "✓ Service scaled to ${POD_COUNT} pods (from initial 2)" +else + echo "⚠️ WARNING: Service only has ${POD_COUNT} pods (expected scaling to >2)" + VALIDATION_FAILURES=$((VALIDATION_FAILURES + 1)) +fi + +# Check for any validation failures +if [ "$VALIDATION_FAILURES" -gt 0 ]; then + echo "" + echo ">>> FAIL: Energy market bidding setup had ${VALIDATION_FAILURES} validation failure(s)!" + exit 1 +else + echo "" + echo ">>> PASS: Energy market bidding performance degradation successfully set up!" + if [ "$v1_requests" != "0" ]; then + echo " - v1.0 had fast performance (~50ms)" + else + echo " - v1.0 metrics not available (lost during upgrade)" + fi + echo " - v2.0 has slow performance (~2s)" + echo " - Service scaled from 2 to ${POD_COUNT} pods" + echo " - Current state shows performance degradation" + exit 0 +fi diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/test_case.yaml new file mode 100644 index 0000000000..981d340b2a --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/test_case.yaml @@ -0,0 +1,68 @@ +user_prompt: +- "Users are reporting slow response times from the energy bidding service in namespace app-161. The service was working fine earlier today. Please investigate the cause of the latency." + +expected_output: + - Must identify that the service is running version v2.0 which has degraded performance + - Should compare metrics showing v2.0 processes requests in ~2 seconds vs v1.0's ~50ms + - Should note that the service has scaled to 10 pods (as a consequence, not the main issue) + - May mention high CPU usage across all pods + - Should conclude the root cause is the version change, not a capacity issue + +tags: + - kubernetes + - prometheus + - medium + +port_forwards: + - namespace: app-161 + service: prometheus + local_port: 9090 + remote_port: 9090 + +setup_timeout: 360 + +before_test: | + set -e + + echo "📁 Creating namespace..." + kubectl create namespace app-161 --save-config --dry-run=client -o yaml | kubectl apply -f - + + echo "⚙️ Setting up monitoring..." + kubectl apply -f prometheus-config.yaml -n app-161 + kubectl apply -f ../../shared/prometheus.yaml -n app-161 + + echo "🚀 Phase 1: Deploy v1.0 (fast version)..." + kubectl apply -f bidder-v1.yaml -n app-161 + kubectl apply -f hpa.yaml -n app-161 + + echo "⏳ Waiting for v1.0 pods..." + kubectl wait --for=condition=ready pod -l app=prometheus -n app-161 --timeout=60s + kubectl wait --for=condition=ready pod -l app=bidder -n app-161 --timeout=60s + + echo "📊 Generate traffic for v1.0 (2 minutes)..." + kubectl apply -f k6-v1-traffic.yaml -n app-161 + kubectl wait --for=condition=complete job/k6-v1-traffic -n app-161 --timeout=150s + + echo "🔄 Phase 2: Upgrade to v2.0 (slow version)..." + kubectl apply -f bidder-v2.yaml -n app-161 + + echo "⏳ Waiting for v2.0 rollout..." + kubectl rollout status deployment/bidder -n app-161 --timeout=60s + + echo "📊 Generate traffic for v2.0 (2 minutes)..." + kubectl apply -f k6-v2-traffic.yaml -n app-161 + kubectl wait --for=condition=complete job/k6-v2-traffic -n app-161 --timeout=150s + + echo "⏳ Waiting for HPA to scale up..." + ./wait-for-scaling.sh + + echo "🔍 Running validation..." + ./run-performance-check.sh + + echo "🧹 Cleaning up k6 jobs to prevent AI detection" + kubectl delete jobs -n app-161 -l app=k6 --ignore-not-found=true + + echo "✅ Test setup completed" + +after_test: | + kubectl delete namespace app-161 || true diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/toolsets.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/toolsets.yaml new file mode 100644 index 0000000000..4a571a90c2 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/toolsets.yaml @@ -0,0 +1,9 @@ +toolsets: + kubernetes/core: + enabled: true + kubernetes/logs: + enabled: true + prometheus/metrics: + enabled: true + config: + prometheus_url: http://localhost:9090 diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/wait-for-scaling.sh b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/wait-for-scaling.sh new file mode 100755 index 0000000000..824e08ebcf --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/wait-for-scaling.sh @@ -0,0 +1,42 @@ +#!/usr/bin/env bash + +set -euo pipefail + +NS="${NS:-app-161}" +TARGET_REPLICAS=3 # We expect at least 8 replicas after scaling +MAX_WAIT=120 # Maximum 2 minutes to wait for scaling + +echo ">>> Waiting for HPA to scale up the deployment..." + +start_time=$(date +%s) +while true; do + current_replicas=$(kubectl get deployment bidder -n "${NS}" -o jsonpath='{.status.replicas}') + + echo "Current replicas: ${current_replicas}" + + if [ "${current_replicas}" -ge "${TARGET_REPLICAS}" ]; then + echo "✓ Deployment has scaled to ${current_replicas} replicas" + break + fi + + elapsed=$(($(date +%s) - start_time)) + if [ "${elapsed}" -gt "${MAX_WAIT}" ]; then + echo "⚠️ WARNING: Deployment hasn't scaled to ${TARGET_REPLICAS} replicas after ${MAX_WAIT} seconds" + echo "Current replicas: ${current_replicas}" + echo "Continuing anyway..." + break + fi + + echo "Waiting for scaling... (${elapsed}s elapsed)" + sleep 5 +done + +# Show HPA status +echo "" +echo ">>> HPA Status:" +kubectl get hpa bidder-hpa -n "${NS}" +kubectl describe hpa bidder-hpa -n "${NS}" | grep -A 5 "Metrics:" + +echo "" +echo ">>> Deployment Status:" +kubectl get deployment bidder -n "${NS}" -o wide From 340bee44745dc234d86300a292ba98c3126db847 Mon Sep 17 00:00:00 2001 From: Tomer Keshet Date: Tue, 21 Oct 2025 13:27:30 +0300 Subject: [PATCH 4/8] pass 70%. failing because false positive on startup issues --- .../161_bidding_version_performance/bidder-v2.yaml | 4 ++-- .../161_bidding_version_performance/test_case.yaml | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml index 900c2b13a9..2215c8689c 100644 --- a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml @@ -80,8 +80,8 @@ metadata: labels: app: bidder version: v2.0 - annotations: - kubernetes.io/change-cause: "Updated to v2.0 with enhanced bid processing algorithm" +# annotations: +# kubernetes.io/change-cause: "Updated to v2.0 with enhanced bid processing algorithm" spec: replicas: 2 # Start with same replica count selector: diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/test_case.yaml index 981d340b2a..16ffab45ed 100644 --- a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/test_case.yaml @@ -4,8 +4,8 @@ user_prompt: expected_output: - Must identify that the service is running version v2.0 which has degraded performance - Should compare metrics showing v2.0 processes requests in ~2 seconds vs v1.0's ~50ms - - Should note that the service has scaled to 10 pods (as a consequence, not the main issue) - - May mention high CPU usage across all pods +# - Should note that the service has more pods in v2.0 compared to v1.0 (as a consequence, not the main issue) +# - May mention high CPU usage across all pods - Should conclude the root cause is the version change, not a capacity issue tags: From 5e833f469962dcc9d088a329248e7af455973087 Mon Sep 17 00:00:00 2001 From: Tomer Keshet Date: Tue, 21 Oct 2025 14:11:11 +0300 Subject: [PATCH 5/8] use pre-built image w/python libs for fast startup+ less false positives --- .../shared/python-flask-otel/Dockerfile | 4 +++- .../fixtures/shared/python-flask-otel/build.sh | 3 ++- .../bidder-v1.yaml | 18 ++++++++++++------ .../bidder-v2.yaml | 18 ++++++++++++------ 4 files changed, 29 insertions(+), 14 deletions(-) diff --git a/tests/llm/fixtures/shared/python-flask-otel/Dockerfile b/tests/llm/fixtures/shared/python-flask-otel/Dockerfile index 5385e07cde..879331c19d 100644 --- a/tests/llm/fixtures/shared/python-flask-otel/Dockerfile +++ b/tests/llm/fixtures/shared/python-flask-otel/Dockerfile @@ -2,6 +2,7 @@ FROM python:3.11-slim # Install Flask and OpenTelemetry packages (for Tempo-based tests) # Plus additional packages needed for New Relic-based tests +# Plus prometheus_client for Prometheus-based tests RUN pip install --no-cache-dir \ flask==3.1.2 \ opentelemetry-api==1.37.0 \ @@ -12,7 +13,8 @@ RUN pip install --no-cache-dir \ opentelemetry-exporter-otlp-proto-grpc==1.37.0 \ opentelemetry-proto==1.37.0 \ opentelemetry-semantic-conventions==0.58b0 \ - requests==2.32.3 + requests==2.32.3 \ + prometheus-client==0.21.1 # Create app directory RUN mkdir /app diff --git a/tests/llm/fixtures/shared/python-flask-otel/build.sh b/tests/llm/fixtures/shared/python-flask-otel/build.sh index 884487698b..87aac8316a 100755 --- a/tests/llm/fixtures/shared/python-flask-otel/build.sh +++ b/tests/llm/fixtures/shared/python-flask-otel/build.sh @@ -6,7 +6,7 @@ set -e REGISTRY="me-west1-docker.pkg.dev/robusta-development/development" IMAGE_NAME="python-flask-otel" -IMAGE_TAG="2.1" +IMAGE_TAG="2.2" FULL_IMAGE="${REGISTRY}/${IMAGE_NAME}:${IMAGE_TAG}" LOCAL_IMAGE="holmes-test/${IMAGE_NAME}:${IMAGE_TAG}" @@ -33,3 +33,4 @@ echo " - OpenTelemetry API & SDK 1.37.0" echo " - OpenTelemetry Flask instrumentation 0.58b0" echo " - OpenTelemetry OTLP gRPC exporter 1.37.0" echo " - Requests 2.32.3" +echo " - Prometheus Client 0.21.1" diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v1.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v1.yaml index d334907692..ed317af41c 100644 --- a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v1.yaml +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v1.yaml @@ -99,10 +99,9 @@ spec: spec: containers: - name: bidder - image: python:3.11-slim - command: ["/bin/sh", "-c"] - args: - - pip install flask prometheus_client && python /app/app.py + image: me-west1-docker.pkg.dev/robusta-development/development/python-flask-otel:2.2 + imagePullPolicy: IfNotPresent + command: ["python", "/app/app.py"] ports: - containerPort: 8080 name: http @@ -116,17 +115,24 @@ spec: limits: memory: "256Mi" cpu: "500m" + startupProbe: + httpGet: + path: /healthz + port: 8080 + initialDelaySeconds: 5 + periodSeconds: 5 + failureThreshold: 30 # 5 + 5*30 = 155 seconds max startup time readinessProbe: httpGet: path: /healthz port: 8080 - initialDelaySeconds: 10 + initialDelaySeconds: 5 periodSeconds: 5 livenessProbe: httpGet: path: /healthz port: 8080 - initialDelaySeconds: 15 + initialDelaySeconds: 5 periodSeconds: 10 volumes: - name: app-code diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml index 2215c8689c..7fe6682c19 100644 --- a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml @@ -104,10 +104,9 @@ spec: spec: containers: - name: bidder - image: python:3.11-slim - command: ["/bin/sh", "-c"] - args: - - pip install flask prometheus_client && python /app/app.py + image: me-west1-docker.pkg.dev/robusta-development/development/python-flask-otel:2.2 + imagePullPolicy: IfNotPresent + command: ["python", "/app/app.py"] ports: - containerPort: 8080 name: http @@ -121,17 +120,24 @@ spec: limits: memory: "256Mi" cpu: "500m" + startupProbe: + httpGet: + path: /healthz + port: 8080 + initialDelaySeconds: 5 + periodSeconds: 5 + failureThreshold: 30 # 5 + 5*30 = 155 seconds max startup time readinessProbe: httpGet: path: /healthz port: 8080 - initialDelaySeconds: 10 + initialDelaySeconds: 5 periodSeconds: 5 livenessProbe: httpGet: path: /healthz port: 8080 - initialDelaySeconds: 15 + initialDelaySeconds: 5 periodSeconds: 10 volumes: - name: app-code From 8899c0db34c7c7b3ce2a21862119270cf23dbc87 Mon Sep 17 00:00:00 2001 From: Tomer Keshet Date: Tue, 21 Oct 2025 17:34:52 +0300 Subject: [PATCH 6/8] dont include the version explicitly but include pod information --- .../bidder-v1.yaml | 40 ++++++++++++++++--- .../bidder-v2.yaml | 40 ++++++++++++++++--- .../run-performance-check.sh | 34 ++++++---------- .../test_case.yaml | 3 -- 4 files changed, 82 insertions(+), 35 deletions(-) diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v1.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v1.yaml index ed317af41c..3ee1898d3a 100644 --- a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v1.yaml +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v1.yaml @@ -11,6 +11,7 @@ stringData: import time import random import logging + import os app = Flask(__name__) @@ -22,17 +23,22 @@ stringData: # Version info VERSION = "v1.0" - # Prometheus metrics + # Get metadata from environment + NAMESPACE = os.environ.get('NAMESPACE', 'app-161') + POD_NAME = os.environ.get('POD_NAME', os.environ.get('HOSTNAME', 'unknown')) + SERVICE = os.environ.get('SERVICE_NAME', 'bidder') + + # Prometheus metrics with standard k8s labels bid_requests = Counter( 'bid_requests_total', 'Total bid requests processed', - ['exchange', 'decision', 'version'] + ['namespace', 'pod', 'service', 'exchange', 'decision'] ) request_duration = Histogram( 'bid_request_duration_seconds', 'Bid request processing duration', - ['exchange', 'version'], + ['namespace', 'pod', 'service', 'exchange'], buckets=[0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1.0, 2.5, 5.0] ) @@ -55,8 +61,19 @@ stringData: # 10% bid rate (realistic for energy markets) decision = 'bid' if random.random() < 0.1 else 'no_bid' - bid_requests.labels(exchange=exchange, decision=decision, version=VERSION).inc() - request_duration.labels(exchange=exchange, version=VERSION).observe(processing_time) + bid_requests.labels( + namespace=NAMESPACE, + pod=POD_NAME, + service=SERVICE, + exchange=exchange, + decision=decision + ).inc() + request_duration.labels( + namespace=NAMESPACE, + pod=POD_NAME, + service=SERVICE, + exchange=exchange + ).observe(processing_time) return jsonify({ 'decision': decision, @@ -105,6 +122,19 @@ spec: ports: - containerPort: 8080 name: http + env: + - name: NAMESPACE + valueFrom: + fieldRef: + fieldPath: metadata.namespace + - name: POD_NAME + valueFrom: + fieldRef: + fieldPath: metadata.name + - name: SERVICE_NAME + valueFrom: + fieldRef: + fieldPath: metadata.labels['app'] volumeMounts: - name: app-code mountPath: /app diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml index 7fe6682c19..dc20fa667e 100644 --- a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml @@ -11,6 +11,7 @@ stringData: import time import random import logging + import os app = Flask(__name__) @@ -22,17 +23,22 @@ stringData: # Version info VERSION = "v2.0" - # Prometheus metrics + # Get metadata from environment + NAMESPACE = os.environ.get('NAMESPACE', 'app-161') + POD_NAME = os.environ.get('POD_NAME', os.environ.get('HOSTNAME', 'unknown')) + SERVICE = os.environ.get('SERVICE_NAME', 'bidder') + + # Prometheus metrics with standard k8s labels bid_requests = Counter( 'bid_requests_total', 'Total bid requests processed', - ['exchange', 'decision', 'version'] + ['namespace', 'pod', 'service', 'exchange', 'decision'] ) request_duration = Histogram( 'bid_request_duration_seconds', 'Bid request processing duration', - ['exchange', 'version'], + ['namespace', 'pod', 'service', 'exchange'], buckets=[0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1.0, 2.5, 5.0] ) @@ -55,8 +61,19 @@ stringData: # 10% bid rate (realistic for energy markets) decision = 'bid' if random.random() < 0.1 else 'no_bid' - bid_requests.labels(exchange=exchange, decision=decision, version=VERSION).inc() - request_duration.labels(exchange=exchange, version=VERSION).observe(processing_time) + bid_requests.labels( + namespace=NAMESPACE, + pod=POD_NAME, + service=SERVICE, + exchange=exchange, + decision=decision + ).inc() + request_duration.labels( + namespace=NAMESPACE, + pod=POD_NAME, + service=SERVICE, + exchange=exchange + ).observe(processing_time) return jsonify({ 'decision': decision, @@ -110,6 +127,19 @@ spec: ports: - containerPort: 8080 name: http + env: + - name: NAMESPACE + valueFrom: + fieldRef: + fieldPath: metadata.namespace + - name: POD_NAME + valueFrom: + fieldRef: + fieldPath: metadata.name + - name: SERVICE_NAME + valueFrom: + fieldRef: + fieldPath: metadata.labels['app'] volumeMounts: - name: app-code mountPath: /app diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/run-performance-check.sh b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/run-performance-check.sh index c34c1dbb8a..7810be6bab 100755 --- a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/run-performance-check.sh +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/run-performance-check.sh @@ -16,13 +16,16 @@ PROM_URL="${PROM_URL:-http://prometheus.${NS}.svc.cluster.local:9090}" VALIDATION_FAILURES=0 # Prometheus queries -# Query over a longer time range to capture both v1.0 and v2.0 -PROMQL_V1_LATENCY='avg(avg_over_time(bid_request_duration_seconds_sum{version="v1.0"}[30m]) / avg_over_time(bid_request_duration_seconds_count{version="v1.0"}[30m]))' -PROMQL_V2_LATENCY='avg(avg_over_time(bid_request_duration_seconds_sum{version="v2.0"}[30m]) / avg_over_time(bid_request_duration_seconds_count{version="v2.0"}[30m]))' -# Use last_over_time to get the most recent value even for stale series -PROMQL_V1_REQUESTS='last_over_time(bid_requests_total{version="v1.0"}[30m])' -PROMQL_V2_REQUESTS='last_over_time(bid_requests_total{version="v2.0"}[30m])' -PROMQL_BUILD_INFO='last_over_time(bidder_build_info[30m])' +# Since we can't distinguish v1 from v2 by labels, we'll use time-based queries +# v1.0 metrics: older data (offset by 5m to look at historical data) +# v2.0 metrics: recent data (last 5m) +PROMQL_V1_LATENCY='avg(avg_over_time(bid_request_duration_seconds_sum{namespace="app-161"}[5m] offset 5m) / avg_over_time(bid_request_duration_seconds_count{namespace="app-161"}[5m] offset 5m))' +PROMQL_V2_LATENCY='avg(avg_over_time(bid_request_duration_seconds_sum{namespace="app-161"}[5m]) / avg_over_time(bid_request_duration_seconds_count{namespace="app-161"}[5m]))' +# Use offset to get v1.0 request count from earlier time period +PROMQL_V1_REQUESTS='sum(increase(bid_requests_total{namespace="app-161"}[5m] offset 5m))' +PROMQL_V2_REQUESTS='sum(increase(bid_requests_total{namespace="app-161"}[5m]))' +# Build info query - filter by namespace +PROMQL_BUILD_INFO='last_over_time(bidder_build_info{namespace="app-161"}[30m])' echo ">>> Checking deployment status" POD_COUNT=$(kubectl get deployment bidder -n "${NS}" -o jsonpath='{.status.readyReplicas}') @@ -95,21 +98,8 @@ extract_val() { echo "" echo ">>> Fetching metrics from Prometheus" -# Check build info -echo "Checking build info metrics..." -build_info_json="$(fetch_json "$PROMQL_BUILD_INFO")" -v1_exists="$(printf '%s' "$build_info_json" | jq -r '.data.result[] | select(.metric.version == "v1.0") | .value[1]' 2>/dev/null)" || true -v2_exists="$(printf '%s' "$build_info_json" | jq -r '.data.result[] | select(.metric.version == "v2.0") | .value[1]' 2>/dev/null)" || true - -if [ -n "$v1_exists" ] && [ -n "$v2_exists" ]; then - echo "✓ Found metrics for both v1.0 and v2.0" -else - echo "⚠️ WARNING: Missing metrics for some versions (v1.0: ${v1_exists:-missing}, v2.0: ${v2_exists:-missing})" - # Mark as failure if v1.0 build info is missing - if [ -z "$v1_exists" ]; then - VALIDATION_FAILURES=$((VALIDATION_FAILURES + 1)) - fi -fi +# Skip build info check since we're using time-based queries +echo "Using time-based queries to distinguish v1.0 (older) from v2.0 (recent) metrics..." # Get v1.0 metrics echo "Fetching v1.0 latency metrics..." diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/test_case.yaml index 16ffab45ed..50eccba43e 100644 --- a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/test_case.yaml @@ -56,9 +56,6 @@ before_test: | echo "⏳ Waiting for HPA to scale up..." ./wait-for-scaling.sh - echo "🔍 Running validation..." - ./run-performance-check.sh - echo "🧹 Cleaning up k6 jobs to prevent AI detection" kubectl delete jobs -n app-161 -l app=k6 --ignore-not-found=true From de292aec084470d6f51374acef8b98d75c3bc211 Mon Sep 17 00:00:00 2001 From: Tomer Keshet Date: Tue, 21 Oct 2025 18:18:57 +0300 Subject: [PATCH 7/8] removed the check file because it is not correct at the moment --- .../run-performance-check.sh | 197 ------------------ 1 file changed, 197 deletions(-) delete mode 100755 tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/run-performance-check.sh diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/run-performance-check.sh b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/run-performance-check.sh deleted file mode 100755 index 7810be6bab..0000000000 --- a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/run-performance-check.sh +++ /dev/null @@ -1,197 +0,0 @@ -#!/usr/bin/env bash - -set -euo pipefail - -# Energy Market Bidding Performance Validation Script -# Validates that: -# 1. v1.0 metrics exist showing ~50ms latency -# 2. v2.0 metrics exist showing ~2s latency -# 3. Service has scaled from 2 to ~10 pods -# 4. Prometheus has captured the version transition - -NS="${NS:-app-161}" -PROM_URL="${PROM_URL:-http://prometheus.${NS}.svc.cluster.local:9090}" - -# Track validation failures -VALIDATION_FAILURES=0 - -# Prometheus queries -# Since we can't distinguish v1 from v2 by labels, we'll use time-based queries -# v1.0 metrics: older data (offset by 5m to look at historical data) -# v2.0 metrics: recent data (last 5m) -PROMQL_V1_LATENCY='avg(avg_over_time(bid_request_duration_seconds_sum{namespace="app-161"}[5m] offset 5m) / avg_over_time(bid_request_duration_seconds_count{namespace="app-161"}[5m] offset 5m))' -PROMQL_V2_LATENCY='avg(avg_over_time(bid_request_duration_seconds_sum{namespace="app-161"}[5m]) / avg_over_time(bid_request_duration_seconds_count{namespace="app-161"}[5m]))' -# Use offset to get v1.0 request count from earlier time period -PROMQL_V1_REQUESTS='sum(increase(bid_requests_total{namespace="app-161"}[5m] offset 5m))' -PROMQL_V2_REQUESTS='sum(increase(bid_requests_total{namespace="app-161"}[5m]))' -# Build info query - filter by namespace -PROMQL_BUILD_INFO='last_over_time(bidder_build_info{namespace="app-161"}[30m])' - -echo ">>> Checking deployment status" -POD_COUNT=$(kubectl get deployment bidder -n "${NS}" -o jsonpath='{.status.readyReplicas}') -DEPLOYMENT_VERSION=$(kubectl get deployment bidder -n "${NS}" -o jsonpath='{.metadata.labels.version}') - -echo "Current pod count: ${POD_COUNT}" -echo "Current deployment version: ${DEPLOYMENT_VERSION}" - -if [ "$DEPLOYMENT_VERSION" != "v2.0" ]; then - echo "✗ ERROR: Expected version v2.0, found ${DEPLOYMENT_VERSION}" >&2 - exit 1 -fi - -if [ "$POD_COUNT" -lt 3 ]; then - echo "⚠️ WARNING: Expected at least 3 pods after scaling, found ${POD_COUNT}" - VALIDATION_FAILURES=$((VALIDATION_FAILURES + 1)) -fi - -echo "✓ Deployment is running v2.0 with ${POD_COUNT} pods" - -# Run curl inside the cluster to query Prometheus -curl_query() { - local q="$1" - # Run kubectl and extract JSON from the output - local output - output=$(kubectl run curl-$$ \ - --image=curlimages/curl:8.8.0 \ - --restart=Never --rm -i --quiet -- \ - sh -c "curl -sS --get '${PROM_URL}/api/v1/query' --data-urlencode 'query=${q}'" 2>&1) - - # Extract JSON - more robust extraction that handles kubectl output - # First remove any non-JSON prefix, then extract the JSON object - echo "$output" | grep -o '{"status":.*}' | head -1 -} - -# Retry helper for Prometheus queries -fetch_json() { - local q="$1" json="" attempt=1 - - while [ $attempt -le 15 ]; do - json="$(curl_query "$q" 2>/dev/null || echo "")" - - # Check if we got valid JSON with success status - if printf '%s' "$json" | jq -e '.status == "success"' >/dev/null 2>&1; then - echo "$json" - return 0 - fi - - echo "Waiting for Prometheus data (attempt $attempt/15)..." >&2 - sleep 2 - attempt=$((attempt+1)) - done - - # Return last attempt even if failed - echo "$json" -} - -# Extract numeric value from Prometheus response -extract_val() { - local json="$1" - local val - val="$(printf '%s' "$json" | jq -r '.data.result[0].value[1] // empty' 2>/dev/null)" || true - if [ -z "${val}" ] || [ "${val}" = "null" ] || [ "${val}" = "NaN" ]; then - echo "0" - return - fi - printf '%s' "$val" -} - -echo "" -echo ">>> Fetching metrics from Prometheus" - -# Skip build info check since we're using time-based queries -echo "Using time-based queries to distinguish v1.0 (older) from v2.0 (recent) metrics..." - -# Get v1.0 metrics -echo "Fetching v1.0 latency metrics..." -v1_latency_json="$(fetch_json "$PROMQL_V1_LATENCY")" -v1_latency="$(extract_val "$v1_latency_json")" - -echo "Fetching v1.0 request count..." -v1_requests_json="$(fetch_json "$PROMQL_V1_REQUESTS")" -v1_requests="$(extract_val "$v1_requests_json")" - -# Get v2.0 metrics -echo "Fetching v2.0 latency metrics..." -v2_latency_json="$(fetch_json "$PROMQL_V2_LATENCY")" -v2_latency="$(extract_val "$v2_latency_json")" - -echo "Fetching v2.0 request count..." -v2_requests_json="$(fetch_json "$PROMQL_V2_REQUESTS")" -v2_requests="$(extract_val "$v2_requests_json")" - -# Display results -echo "" -echo "=== METRICS SUMMARY ===" -echo "v1.0 average latency: ${v1_latency}s (expected: ~0.05s)" -echo "v1.0 total requests: ${v1_requests}" -echo "v2.0 average latency: ${v2_latency}s (expected: ~2.0s)" -echo "v2.0 total requests: ${v2_requests}" -echo "Current pod count: ${POD_COUNT} (expected: 8-10)" - -# Validate the test conditions -echo "" -echo ">>> Validating test conditions..." - -# Check 1: v1.0 should have had fast latency (~50ms) -if [ "$v1_requests" = "0" ]; then - echo "⚠️ WARNING: No v1.0 requests found - v1.0 metrics may have been lost during upgrade" - echo "This is expected if Prometheus hasn't retained the metrics long enough" - VALIDATION_FAILURES=$((VALIDATION_FAILURES + 1)) -else - if awk -v lat="$v1_latency" 'BEGIN{exit !(lat > 0.03 && lat < 0.08)}'; then - echo "✓ v1.0 latency was fast (~50ms): ${v1_latency}s" - else - echo "⚠️ WARNING: v1.0 latency outside expected range: ${v1_latency}s (expected 0.03-0.08s)" - VALIDATION_FAILURES=$((VALIDATION_FAILURES + 1)) - fi -fi - -# Check 2: v2.0 should have slow latency (~2s) -if [ "$v2_requests" = "0" ]; then - echo "✗ ERROR: No v2.0 requests found - v2.0 metrics missing" >&2 - # Keep critical early exit for v2.0 missing - exit 3 -fi - -if awk -v lat="$v2_latency" 'BEGIN{exit !(lat > 1.5 && lat < 2.5)}'; then - echo "✓ v2.0 latency is slow (~2s): ${v2_latency}s" -else - echo "✗ ERROR: v2.0 latency not in expected range: ${v2_latency}s (expected 1.5-2.5s)" >&2 - # Keep critical early exit for v2.0 latency out of range - exit 4 -fi - -# Check 3: Performance degradation factor -if [ "$v1_latency" != "0" ] && [ "$v1_requests" != "0" ]; then - degradation_factor=$(awk -v v2="$v2_latency" -v v1="$v1_latency" 'BEGIN{printf "%.1f", v2/v1}') - echo "✓ Performance degraded by ${degradation_factor}x (v2.0 is ${degradation_factor}x slower than v1.0)" -else - echo "⚠️ Cannot calculate degradation factor without v1.0 metrics" -fi - -# Check 4: Scaling occurred -if [ "$POD_COUNT" -ge 3 ]; then - echo "✓ Service scaled to ${POD_COUNT} pods (from initial 2)" -else - echo "⚠️ WARNING: Service only has ${POD_COUNT} pods (expected scaling to >2)" - VALIDATION_FAILURES=$((VALIDATION_FAILURES + 1)) -fi - -# Check for any validation failures -if [ "$VALIDATION_FAILURES" -gt 0 ]; then - echo "" - echo ">>> FAIL: Energy market bidding setup had ${VALIDATION_FAILURES} validation failure(s)!" - exit 1 -else - echo "" - echo ">>> PASS: Energy market bidding performance degradation successfully set up!" - if [ "$v1_requests" != "0" ]; then - echo " - v1.0 had fast performance (~50ms)" - else - echo " - v1.0 metrics not available (lost during upgrade)" - fi - echo " - v2.0 has slow performance (~2s)" - echo " - Service scaled from 2 to ${POD_COUNT} pods" - echo " - Current state shows performance degradation" - exit 0 -fi From 1fea326c2c3ab9dde5783d4b3a78e9950a6e0616 Mon Sep 17 00:00:00 2001 From: Tomer Keshet Date: Thu, 23 Oct 2025 10:09:35 +0300 Subject: [PATCH 8/8] fixed comment --- .../161_bidding_version_performance/wait-for-scaling.sh | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/wait-for-scaling.sh b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/wait-for-scaling.sh index 824e08ebcf..ff8f116feb 100755 --- a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/wait-for-scaling.sh +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/wait-for-scaling.sh @@ -3,7 +3,7 @@ set -euo pipefail NS="${NS:-app-161}" -TARGET_REPLICAS=3 # We expect at least 8 replicas after scaling +TARGET_REPLICAS=3 # We expect at least 3 replicas after scaling MAX_WAIT=120 # Maximum 2 minutes to wait for scaling echo ">>> Waiting for HPA to scale up the deployment..."