diff --git a/tests/llm/fixtures/shared/python-flask-otel/Dockerfile b/tests/llm/fixtures/shared/python-flask-otel/Dockerfile index 5385e07cd..879331c19 100644 --- a/tests/llm/fixtures/shared/python-flask-otel/Dockerfile +++ b/tests/llm/fixtures/shared/python-flask-otel/Dockerfile @@ -2,6 +2,7 @@ FROM python:3.11-slim # Install Flask and OpenTelemetry packages (for Tempo-based tests) # Plus additional packages needed for New Relic-based tests +# Plus prometheus_client for Prometheus-based tests RUN pip install --no-cache-dir \ flask==3.1.2 \ opentelemetry-api==1.37.0 \ @@ -12,7 +13,8 @@ RUN pip install --no-cache-dir \ opentelemetry-exporter-otlp-proto-grpc==1.37.0 \ opentelemetry-proto==1.37.0 \ opentelemetry-semantic-conventions==0.58b0 \ - requests==2.32.3 + requests==2.32.3 \ + prometheus-client==0.21.1 # Create app directory RUN mkdir /app diff --git a/tests/llm/fixtures/shared/python-flask-otel/build.sh b/tests/llm/fixtures/shared/python-flask-otel/build.sh index 884487698..87aac8316 100755 --- a/tests/llm/fixtures/shared/python-flask-otel/build.sh +++ b/tests/llm/fixtures/shared/python-flask-otel/build.sh @@ -6,7 +6,7 @@ set -e REGISTRY="me-west1-docker.pkg.dev/robusta-development/development" IMAGE_NAME="python-flask-otel" -IMAGE_TAG="2.1" +IMAGE_TAG="2.2" FULL_IMAGE="${REGISTRY}/${IMAGE_NAME}:${IMAGE_TAG}" LOCAL_IMAGE="holmes-test/${IMAGE_NAME}:${IMAGE_TAG}" @@ -33,3 +33,4 @@ echo " - OpenTelemetry API & SDK 1.37.0" echo " - OpenTelemetry Flask instrumentation 0.58b0" echo " - OpenTelemetry OTLP gRPC exporter 1.37.0" echo " - Requests 2.32.3" +echo " - Prometheus Client 0.21.1" diff --git a/tests/llm/fixtures/test_ask_holmes/124_checkout_latency_prometheus/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/124_checkout_latency_prometheus/test_case.yaml index d4488b54c..fd9756bbe 100644 --- a/tests/llm/fixtures/test_ask_holmes/124_checkout_latency_prometheus/test_case.yaml +++ b/tests/llm/fixtures/test_ask_holmes/124_checkout_latency_prometheus/test_case.yaml @@ -1,5 +1,5 @@ user_prompt: -- "The checkout service in namespace app-124 is experiencing high latency. Investigate why." +- "The /checkout endpoint in namespace app-124 is experiencing high latency. Investigate why." # Note telling holmes to focus on http_request_duration_seconds_bucket metric passes with sonnet - 'The checkout service in namespace app-124 is experiencing high latency. Investigate why. USE prometheus and http_request_duration_seconds_bucket' expected_output: diff --git a/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/bidder-app.yaml b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/bidder-app.yaml new file mode 100644 index 000000000..3861452e7 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/bidder-app.yaml @@ -0,0 +1,396 @@ +# ------------------------------------------------------------------------------ +# Energy Market Bidding Service Test Environment +# +# This manifest sets up a test to validate that Holmes can detect when +# a specific energy exchange (NordPool) starts accepting 100% of bids +# instead of the normal 10% rate, combined with a traffic surge. +# +# Components: +# - Namespace: app-160 +# - Bidder Service: Python/Flask app that processes bid requests +# * Normal behavior: 10% bid acceptance rate (realistic for energy markets) +# * Bug: After 100 requests from NordPool, always accepts bids from NordPool +# * Exports Prometheus metrics with exchange and decision labels +# - k6 Job: Simulates traffic from 5 energy exchanges +# * Phase 1 (0-15s): Normal traffic distribution +# * Phase 2 (15s-45s): NordPool traffic increases 10x +# ------------------------------------------------------------------------------ +apiVersion: v1 +kind: Namespace +metadata: + name: app-160 + +--- +apiVersion: v1 +kind: Secret +metadata: + name: bidder-app + namespace: app-160 +type: Opaque +stringData: + app.py: | + import os, time, json, random, logging + from flask import Flask, request, jsonify + from prometheus_client import Counter, Gauge, Histogram, generate_latest, CONTENT_TYPE_LATEST + from threading import Lock + + app = Flask(__name__) + + # Disable logging for cleaner test + log = logging.getLogger('werkzeug') + log.setLevel(logging.ERROR) + app.logger.disabled = True + + # Business constants + NORMAL_BID_RATE = 0.10 # 10% bid rate is typical for energy markets + + # Bug tracking + nordpool_counter = 0 + nordpool_lock = Lock() + + # Prometheus metrics + bid_requests = Counter( + "bid_requests_total", + "Total bid requests received", + ["exchange", "decision"] + ) + + request_duration = Histogram( + "bid_request_duration_seconds", + "Duration of bid request processing", + ["exchange"], + buckets=[0.001, 0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1.0] + ) + + current_bid_rate = Gauge( + "bid_rate_current", + "Current bid acceptance rate by exchange", + ["exchange"] + ) + + # Track bid rates for gauge updates + exchange_stats = {} + stats_lock = Lock() + + def update_bid_rate(exchange, decision): + with stats_lock: + if exchange not in exchange_stats: + exchange_stats[exchange] = {"bid": 0, "no_bid": 0} + exchange_stats[exchange][decision] += 1 + total = exchange_stats[exchange]["bid"] + exchange_stats[exchange]["no_bid"] + if total > 0: + rate = exchange_stats[exchange]["bid"] / total + current_bid_rate.labels(exchange=exchange).set(rate) + + def should_bid(exchange): + global nordpool_counter + + # Bug logic for NordPool + if exchange == "NordPool": + with nordpool_lock: + nordpool_counter += 1 + if nordpool_counter > 100: + # Bug: Always bid after 100 requests + return True + + # Normal logic: 10% bid rate + return random.random() < NORMAL_BID_RATE + + @app.route("/metrics") + def metrics(): + return generate_latest(), 200, {"Content-Type": CONTENT_TYPE_LATEST} + + @app.route("/healthz") + def healthz(): + return "ok", 200 + + @app.route("/bid", methods=["POST"]) + def process_bid(): + start_time = time.perf_counter() + + # Extract request data + data = request.get_json() or {} + exchange = data.get("exchange", "unknown") + energy_id = data.get("energy_id", "") + + # Validate exchange + valid_exchanges = ["NordPool", "EPEX_SPOT", "EEX", "OMIE", "GME"] + if exchange not in valid_exchanges: + return jsonify({"error": "Invalid exchange"}), 400 + + # Make bid decision + decision = "bid" if should_bid(exchange) else "no_bid" + + # Update metrics + bid_requests.labels(exchange=exchange, decision=decision).inc() + update_bid_rate(exchange, decision) + + # Simulate processing time + process_time = random.uniform(0.005, 0.025) + time.sleep(process_time) + + # Record request duration + duration = time.perf_counter() - start_time + request_duration.labels(exchange=exchange).observe(duration) + + # Return response + response = { + "exchange": exchange, + "energy_id": energy_id, + "decision": decision, + "timestamp": int(time.time() * 1000) + } + + return jsonify(response), 200 + + if __name__ == "__main__": + app.run(host="0.0.0.0", port=8080) + +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: bidder + namespace: app-160 +spec: + replicas: 1 + selector: + matchLabels: + app: bidder + template: + metadata: + labels: + app: bidder + annotations: + prometheus.io/scrape: "true" + prometheus.io/path: "/metrics" + prometheus.io/port: "8080" + spec: + containers: + - name: app + image: python:3.11-slim + imagePullPolicy: IfNotPresent + command: ["/bin/sh", "-c"] + args: + - pip install --no-cache-dir flask prometheus_client && python /app/app.py + ports: + - containerPort: 8080 + name: http + volumeMounts: + - name: app-code + mountPath: /app + readinessProbe: + httpGet: + path: /healthz + port: 8080 + initialDelaySeconds: 5 + periodSeconds: 5 + timeoutSeconds: 2 + failureThreshold: 3 + livenessProbe: + httpGet: + path: /healthz + port: 8080 + initialDelaySeconds: 10 + periodSeconds: 10 + timeoutSeconds: 2 + failureThreshold: 3 + resources: + requests: + memory: "128Mi" + cpu: "100m" + limits: + memory: "256Mi" + cpu: "500m" + volumes: + - name: app-code + secret: + secretName: bidder-app + +--- +apiVersion: v1 +kind: Service +metadata: + name: bidder + namespace: app-160 + labels: + app: bidder +spec: + selector: + app: bidder + ports: + - name: http + port: 80 + targetPort: 8080 + +--- +apiVersion: v1 +kind: Secret +metadata: + name: k6-script + namespace: app-160 +type: Opaque +stringData: + test.js: | + import http from 'k6/http'; + import { check, sleep } from 'k6'; + import { Rate } from 'k6/metrics'; + + const bidAcceptanceRate = new Rate('bid_acceptance_rate'); + + export const options = { + scenarios: { + // Normal traffic for first 15 seconds (enough for bug to trigger at ~6 req/s for NordPool) + normal_traffic: { + executor: 'constant-arrival-rate', + rate: 30, // 30 requests per second total + timeUnit: '1s', + duration: '15s', + preAllocatedVUs: 30, + exec: 'normalTraffic', + }, + // After 15 seconds, surge NordPool traffic + nordpool_surge: { + executor: 'constant-arrival-rate', + rate: 60, // 60 requests per second from NordPool alone + timeUnit: '1s', + startTime: '15s', + duration: '30s', + preAllocatedVUs: 60, + exec: 'nordpoolSurge', + }, + // Continue normal traffic for other exchanges + continued_normal: { + executor: 'constant-arrival-rate', + rate: 20, // Reduced rate for other exchanges + timeUnit: '1s', + startTime: '15s', + duration: '30s', + preAllocatedVUs: 20, + exec: 'otherExchanges', + }, + }, + thresholds: { + 'http_req_duration': ['p(95)<100'], + 'http_req_failed': ['rate<0.01'], + }, + }; + + const exchanges = ['NordPool', 'EPEX_SPOT', 'EEX', 'OMIE', 'GME']; + const exchangeWeights = { + 'NordPool': 0.20, + 'EPEX_SPOT': 0.25, + 'EEX': 0.25, + 'OMIE': 0.15, + 'GME': 0.15, + }; + + function selectExchange() { + const rand = Math.random(); + let cumulative = 0; + for (const [exchange, weight] of Object.entries(exchangeWeights)) { + cumulative += weight; + if (rand < cumulative) return exchange; + } + return 'NordPool'; + } + + function makeBidRequest(exchange) { + const url = 'http://bidder.app-160.svc.cluster.local/bid'; + const payload = JSON.stringify({ + exchange: exchange, + energy_id: `ENERGY_${Date.now()}_${Math.random().toString(36).substr(2, 9)}`, + quantity_mwh: Math.floor(Math.random() * 100) + 10, + price_per_mwh: Math.floor(Math.random() * 50) + 30, + }); + + const params = { + headers: { 'Content-Type': 'application/json' }, + tags: { exchange: exchange }, + }; + + const res = http.post(url, payload, params); + + check(res, { + 'status is 200': (r) => r.status === 200, + 'has decision': (r) => { + const body = JSON.parse(r.body); + return body.decision === 'bid' || body.decision === 'no_bid'; + }, + }); + + // Track bid acceptance + if (res.status === 200) { + const body = JSON.parse(res.body); + bidAcceptanceRate.add(body.decision === 'bid'); + } + + return res; + } + + export function normalTraffic() { + const exchange = selectExchange(); + makeBidRequest(exchange); + sleep(0.01); + } + + export function nordpoolSurge() { + makeBidRequest('NordPool'); + sleep(0.01); + } + + export function otherExchanges() { + const otherExchanges = exchanges.filter(e => e !== 'NordPool'); + const exchange = otherExchanges[Math.floor(Math.random() * otherExchanges.length)]; + makeBidRequest(exchange); + sleep(0.01); + } + +--- +apiVersion: batch/v1 +kind: Job +metadata: + name: k6-energy-market + namespace: app-160 +spec: + ttlSecondsAfterFinished: 300 + template: + spec: + restartPolicy: Never + initContainers: + - name: wait-for-bidder + image: curlimages/curl:8.8.0 + command: ["sh", "-c"] + args: + - | + echo "Waiting for bidder service to be ready..." + until curl -fsS http://bidder.app-160.svc.cluster.local/healthz; do + echo "Bidder not ready, waiting..." + sleep 2 + done + echo "Bidder service is ready!" + - name: wait-for-prometheus + image: curlimages/curl:8.8.0 + command: ["sh", "-c"] + args: + - | + echo "Waiting for Prometheus to be ready..." + until curl -fsS http://prometheus.app-160.svc.cluster.local:9090/-/ready; do + echo "Prometheus not ready, waiting..." + sleep 2 + done + echo "Prometheus is ready!" + containers: + - name: k6 + image: grafana/k6:0.49.0 + args: ["run", "/scripts/test.js"] + volumeMounts: + - name: script + mountPath: /scripts + volumes: + - name: script + secret: + secretName: k6-script + items: + - key: test.js + path: test.js diff --git a/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/bidding_system.md b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/bidding_system.md new file mode 100644 index 000000000..85a1509fa --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/bidding_system.md @@ -0,0 +1,3 @@ +- Normal bid rate should be around 10% +- Problems with bid rate and traffic can change in a matter of seconds, use the highest resolution possible +- The total bid request traffic should usually be consistent. diff --git a/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/prometheus-config.yaml b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/prometheus-config.yaml new file mode 100644 index 000000000..75cf24406 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/prometheus-config.yaml @@ -0,0 +1,17 @@ +# Prometheus configuration for test 160 +--- +apiVersion: v1 +kind: ConfigMap +metadata: + name: prometheus-config + namespace: app-160 +data: + prometheus.yml: | + global: + scrape_interval: 5s + evaluation_interval: 5s + scrape_configs: + - job_name: 'bidder' + metrics_path: /metrics + static_configs: + - targets: ['bidder.app-160.svc.cluster.local:80'] diff --git a/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/run-bidder-check.sh b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/run-bidder-check.sh new file mode 100755 index 000000000..0c7b12685 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/run-bidder-check.sh @@ -0,0 +1,153 @@ +#!/usr/bin/env bash + +set -euo pipefail + +# Energy Market Bidding Bug Validation Script +# Validates that: +# 1. NordPool bid rate changes from ~10% to ~100% +# 2. Other exchanges maintain ~10% bid rate +# 3. NordPool traffic increases by ~10x +# 4. Combined effect results in ~100x increase in NordPool bid volume + +NS="${NS:-app-160}" +JOB="${JOB:-k6-energy-market}" +WAIT_TIMEOUT="${WAIT_TIMEOUT:-90s}" +PROM_URL="${PROM_URL:-http://prometheus.${NS}.svc.cluster.local:9090}" + +# Prometheus queries - using 30s windows for faster test +PROMQL_NORDPOOL_BID_RATE='sum(rate(bid_requests_total{exchange="NordPool",decision="bid"}[30s])) / sum(rate(bid_requests_total{exchange="NordPool"}[30s]))' +PROMQL_OTHER_BID_RATE='sum(rate(bid_requests_total{exchange!="NordPool",decision="bid"}[30s])) / sum(rate(bid_requests_total{exchange!="NordPool"}[30s]))' +PROMQL_NORDPOOL_RPS='sum(rate(bid_requests_total{exchange="NordPool"}[30s]))' +PROMQL_OTHER_RPS='sum(rate(bid_requests_total{exchange!="NordPool"}[30s]))' +PROMQL_NORDPOOL_BIDS_PER_SEC='sum(rate(bid_requests_total{exchange="NordPool",decision="bid"}[30s]))' + +echo ">>> Waiting for Job/${JOB} to complete (timeout: ${WAIT_TIMEOUT})" +if ! kubectl -n "${NS}" wait --for=condition=complete "job/${JOB}" --timeout="${WAIT_TIMEOUT}"; then + echo "ERROR: Job/${JOB} did not complete in time." >&2 + kubectl -n "${NS}" get job "${JOB}" -o yaml + kubectl -n "${NS}" logs -l job-name="${JOB}" --tail=50 + exit 1 +fi + +# Run curl inside the cluster to query Prometheus +curl_query() { + local q="$1" + kubectl run curl-$$ \ + --image=curlimages/curl:8.8.0 \ + --restart=Never --rm -i --quiet -- \ + sh -c "curl -sS --get '${PROM_URL}/api/v1/query' --data-urlencode 'query=${q}'" 2>&1 | \ + awk 'match($0, /^[[:space:]]*{/) { print; exit }' +} + +# Retry helper for Prometheus queries +fetch_json() { + local q="$1" json="" attempt=1 + + while [ $attempt -le 15 ]; do + json="$(curl_query "$q" 2>/dev/null || echo "")" + + # Check if we got valid JSON with success status + if printf '%s' "$json" | jq -e '.status == "success"' >/dev/null 2>&1; then + echo "$json" + return 0 + fi + + echo "Waiting for Prometheus data (attempt $attempt/15)..." >&2 + sleep 2 + attempt=$((attempt+1)) + done + + # Return last attempt even if failed + echo "$json" +} + +# Extract numeric value from Prometheus response +extract_val() { + local json="$1" + local val + val="$(printf '%s' "$json" | jq -r '.data.result[0].value[1] // empty' 2>/dev/null)" || true + if [ -z "${val}" ] || [ "${val}" = "null" ] || [ "${val}" = "NaN" ]; then + echo "0" + return + fi + printf '%s' "$val" +} + +echo ">>> Fetching metrics from Prometheus" + +# Get all metrics +echo "Fetching NordPool bid rate..." +nordpool_bid_rate_json="$(fetch_json "$PROMQL_NORDPOOL_BID_RATE")" +nordpool_bid_rate="$(extract_val "$nordpool_bid_rate_json")" + +echo "Fetching other exchanges bid rate..." +other_bid_rate_json="$(fetch_json "$PROMQL_OTHER_BID_RATE")" +other_bid_rate="$(extract_val "$other_bid_rate_json")" + +echo "Fetching NordPool request rate..." +nordpool_rps_json="$(fetch_json "$PROMQL_NORDPOOL_RPS")" +nordpool_rps="$(extract_val "$nordpool_rps_json")" + +echo "Fetching other exchanges request rate..." +other_rps_json="$(fetch_json "$PROMQL_OTHER_RPS")" +other_rps="$(extract_val "$other_rps_json")" + +echo "Fetching NordPool bids per second..." +nordpool_bids_per_sec_json="$(fetch_json "$PROMQL_NORDPOOL_BIDS_PER_SEC")" +nordpool_bids_per_sec="$(extract_val "$nordpool_bids_per_sec_json")" + +# Display results +echo "" +echo "=== METRICS SUMMARY ===" +echo "NordPool bid rate: ${nordpool_bid_rate} (expected: ~1.0 after bug)" +echo "Other exchanges bid rate: ${other_bid_rate} (expected: ~0.1)" +echo "NordPool request rate: ${nordpool_rps} req/s" +echo "Other exchanges request rate: ${other_rps} req/s" +echo "NordPool bids per second: ${nordpool_bids_per_sec}" + +# Validate the bug is present +echo "" +echo ">>> Validating test conditions..." + +# Check 1: NordPool bid rate should be close to 100% +if awk -v rate="$nordpool_bid_rate" 'BEGIN{exit !(rate >= 0.95)}'; then + echo "✓ NordPool bid rate is ~100% (actual: ${nordpool_bid_rate})" +else + echo "✗ ERROR: NordPool bid rate is not ~100% (actual: ${nordpool_bid_rate})" >&2 + echo "Debug info - NordPool bid rate response:" + echo "$nordpool_bid_rate_json" | jq '.' 2>/dev/null || echo "$nordpool_bid_rate_json" + exit 2 +fi + +# Check 2: Other exchanges should maintain ~10% bid rate +if awk -v rate="$other_bid_rate" 'BEGIN{exit !(rate >= 0.08 && rate <= 0.15)}'; then + echo "✓ Other exchanges bid rate is ~10% (actual: ${other_bid_rate})" +else + echo "✗ ERROR: Other exchanges bid rate is not ~10% (actual: ${other_bid_rate})" >&2 + echo "Debug info - Other bid rate response:" + echo "$other_bid_rate_json" | jq '.' 2>/dev/null || echo "$other_bid_rate_json" + exit 3 +fi + +# Check 3: NordPool should have significant traffic (at least 10 req/s) +if awk -v rate="$nordpool_rps" 'BEGIN{exit !(rate >= 10)}'; then + echo "✓ NordPool has significant traffic (${nordpool_rps} req/s)" +else + echo "✗ ERROR: NordPool traffic too low (${nordpool_rps} req/s)" >&2 + exit 4 +fi + +# Check 4: Calculate bid volume increase +# Expected: ~10x traffic increase * 10x bid rate increase = ~100x bid volume increase +expected_baseline_bids="0.1" # Rough estimate: baseline would be ~10% of baseline traffic +bid_volume_factor=$(awk -v bps="$nordpool_bids_per_sec" -v base="$expected_baseline_bids" 'BEGIN{printf "%.1f", bps/base}') + +echo "✓ NordPool bid volume increased by approximately ${bid_volume_factor}x" + +echo "" +echo ">>> PASS: Energy market bidding bug successfully reproduced!" +echo " - NordPool bid rate jumped from ~10% to ~100%" +echo " - Other exchanges maintained normal ~10% bid rate" +echo " - Combined with traffic surge, bid volume increased dramatically" + +exit 0 diff --git a/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/test_case.yaml new file mode 100644 index 000000000..fbd606898 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/test_case.yaml @@ -0,0 +1,52 @@ +user_prompt: +# - "The energy bidding service in namespace app-160 is experiencing unusual bid patterns. Investigate why bid commitments have increased dramatically." +- "Is there issues with the bidding rate in the bidding endpoint in namespace app-160? Check the recent metrics - note that the traffic test has already completed, so you'll see historical data that eventually drops to zero, which is expected." + +expected_output: + - The answer must identify that NordPool exchange has an abnormally high bid acceptance rate (close to 100%) compared to other exchanges (~10%) + - The answer should mention that this behavior started after a certain point in time or after a certain number of requests + - The answer should note the correlation with increased traffic from NordPool + +tags: + - kubernetes + - prometheus + - hard + - chain-of-causation + +port_forwards: + - namespace: app-160 + service: prometheus + local_port: 9090 + remote_port: 9090 + +before_test: | + set -e # Exit immediately if any command fails + + echo "📁 Creating namespace..." + kubectl create namespace app-160 --save-config --dry-run=client -o yaml | kubectl apply -f - || true + + echo "⚙️ Setting up Prometheus..." + kubectl apply -f prometheus-config.yaml -n app-160 + kubectl apply -f ../../shared/prometheus.yaml -n app-160 + + echo "🚀 Deploying bidder service and k6 test..." + kubectl apply -f bidder-app.yaml -n app-160 + + echo "⏳ Waiting for pods to be ready..." + kubectl wait --for=condition=ready pod -l app=prometheus -n app-160 --timeout=60s + kubectl wait --for=condition=ready pod -l app=bidder -n app-160 --timeout=60s + + echo "🔍 Running validation..." + ./run-bidder-check.sh + + echo "🧹 Cleaning up k6 job and pods to prevent AI detection" + kubectl delete job k6-energy-market -n app-160 --ignore-not-found=true + kubectl delete pods -n app-160 -l job-name=k6-energy-market --ignore-not-found=true + + echo "✅ Test setup completed successfully" + +after_test: | + kubectl delete namespace app-160 || true + +include_files: + - bidding_system.md diff --git a/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/toolsets.yaml b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/toolsets.yaml new file mode 100644 index 000000000..4a571a90c --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/160_electricity_market_bidding_bug/toolsets.yaml @@ -0,0 +1,9 @@ +toolsets: + kubernetes/core: + enabled: true + kubernetes/logs: + enabled: true + prometheus/metrics: + enabled: true + config: + prometheus_url: http://localhost:9090 diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v1.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v1.yaml new file mode 100644 index 000000000..3ee1898d3 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v1.yaml @@ -0,0 +1,186 @@ +apiVersion: v1 +kind: Secret +metadata: + name: bidder-app-v1 + namespace: app-161 +type: Opaque +stringData: + app.py: | + from flask import Flask, Response, jsonify, request + from prometheus_client import Counter, Histogram, Gauge, Info, generate_latest, CONTENT_TYPE_LATEST + import time + import random + import logging + import os + + app = Flask(__name__) + + # Disable logging for cleaner test + log = logging.getLogger('werkzeug') + log.setLevel(logging.ERROR) + app.logger.disabled = True + + # Version info + VERSION = "v1.0" + + # Get metadata from environment + NAMESPACE = os.environ.get('NAMESPACE', 'app-161') + POD_NAME = os.environ.get('POD_NAME', os.environ.get('HOSTNAME', 'unknown')) + SERVICE = os.environ.get('SERVICE_NAME', 'bidder') + + # Prometheus metrics with standard k8s labels + bid_requests = Counter( + 'bid_requests_total', + 'Total bid requests processed', + ['namespace', 'pod', 'service', 'exchange', 'decision'] + ) + + request_duration = Histogram( + 'bid_request_duration_seconds', + 'Bid request processing duration', + ['namespace', 'pod', 'service', 'exchange'], + buckets=[0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1.0, 2.5, 5.0] + ) + + build_info = Info('bidder_build', 'Build information') + build_info.info({'version': VERSION, 'commit': 'abc123', 'branch': 'main'}) + + @app.route('/healthz') + def health(): + return 'ok' + + @app.route('/bid', methods=['POST']) + def process_bid(): + # v1.0 - Fast processing + processing_time = 0.04 + random.uniform(0, 0.02) # 40-60ms + time.sleep(processing_time) + + data = request.get_json() or {} + exchange = data.get('exchange', 'NordPool') + + # 10% bid rate (realistic for energy markets) + decision = 'bid' if random.random() < 0.1 else 'no_bid' + + bid_requests.labels( + namespace=NAMESPACE, + pod=POD_NAME, + service=SERVICE, + exchange=exchange, + decision=decision + ).inc() + request_duration.labels( + namespace=NAMESPACE, + pod=POD_NAME, + service=SERVICE, + exchange=exchange + ).observe(processing_time) + + return jsonify({ + 'decision': decision, + 'version': VERSION, + 'processing_time': processing_time + }) + + @app.route('/metrics') + def metrics(): + return Response(generate_latest(), mimetype=CONTENT_TYPE_LATEST) + + if __name__ == '__main__': + app.run(host='0.0.0.0', port=8080) + +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: bidder + namespace: app-161 + labels: + app: bidder + version: v1.0 + annotations: + kubernetes.io/change-cause: "Initial deployment of energy bidding service v1.0" +spec: + replicas: 2 + selector: + matchLabels: + app: bidder + template: + metadata: + labels: + app: bidder + version: v1.0 + annotations: + prometheus.io/scrape: "true" + prometheus.io/port: "8080" + prometheus.io/path: "/metrics" + spec: + containers: + - name: bidder + image: me-west1-docker.pkg.dev/robusta-development/development/python-flask-otel:2.2 + imagePullPolicy: IfNotPresent + command: ["python", "/app/app.py"] + ports: + - containerPort: 8080 + name: http + env: + - name: NAMESPACE + valueFrom: + fieldRef: + fieldPath: metadata.namespace + - name: POD_NAME + valueFrom: + fieldRef: + fieldPath: metadata.name + - name: SERVICE_NAME + valueFrom: + fieldRef: + fieldPath: metadata.labels['app'] + volumeMounts: + - name: app-code + mountPath: /app + resources: + requests: + memory: "128Mi" + cpu: "100m" + limits: + memory: "256Mi" + cpu: "500m" + startupProbe: + httpGet: + path: /healthz + port: 8080 + initialDelaySeconds: 5 + periodSeconds: 5 + failureThreshold: 30 # 5 + 5*30 = 155 seconds max startup time + readinessProbe: + httpGet: + path: /healthz + port: 8080 + initialDelaySeconds: 5 + periodSeconds: 5 + livenessProbe: + httpGet: + path: /healthz + port: 8080 + initialDelaySeconds: 5 + periodSeconds: 10 + volumes: + - name: app-code + secret: + secretName: bidder-app-v1 + +--- +apiVersion: v1 +kind: Service +metadata: + name: bidder + namespace: app-161 + labels: + app: bidder +spec: + selector: + app: bidder + ports: + - port: 80 + targetPort: 8080 + name: http diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml new file mode 100644 index 000000000..dc20fa667 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/bidder-v2.yaml @@ -0,0 +1,175 @@ +apiVersion: v1 +kind: Secret +metadata: + name: bidder-app-v2 + namespace: app-161 +type: Opaque +stringData: + app.py: | + from flask import Flask, Response, jsonify, request + from prometheus_client import Counter, Histogram, Gauge, Info, generate_latest, CONTENT_TYPE_LATEST + import time + import random + import logging + import os + + app = Flask(__name__) + + # Disable logging for cleaner test + log = logging.getLogger('werkzeug') + log.setLevel(logging.ERROR) + app.logger.disabled = True + + # Version info + VERSION = "v2.0" + + # Get metadata from environment + NAMESPACE = os.environ.get('NAMESPACE', 'app-161') + POD_NAME = os.environ.get('POD_NAME', os.environ.get('HOSTNAME', 'unknown')) + SERVICE = os.environ.get('SERVICE_NAME', 'bidder') + + # Prometheus metrics with standard k8s labels + bid_requests = Counter( + 'bid_requests_total', + 'Total bid requests processed', + ['namespace', 'pod', 'service', 'exchange', 'decision'] + ) + + request_duration = Histogram( + 'bid_request_duration_seconds', + 'Bid request processing duration', + ['namespace', 'pod', 'service', 'exchange'], + buckets=[0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1.0, 2.5, 5.0] + ) + + build_info = Info('bidder_build', 'Build information') + build_info.info({'version': VERSION, 'commit': 'def456', 'branch': 'main'}) + + @app.route('/healthz') + def health(): + return 'ok' + + @app.route('/bid', methods=['POST']) + def process_bid(): + # v2.0 - Slow processing due to new "enhanced" algorithm + processing_time = 1.8 + random.uniform(0, 0.4) # 1.8-2.2s + time.sleep(processing_time) + + data = request.get_json() or {} + exchange = data.get('exchange', 'NordPool') + + # 10% bid rate (realistic for energy markets) + decision = 'bid' if random.random() < 0.1 else 'no_bid' + + bid_requests.labels( + namespace=NAMESPACE, + pod=POD_NAME, + service=SERVICE, + exchange=exchange, + decision=decision + ).inc() + request_duration.labels( + namespace=NAMESPACE, + pod=POD_NAME, + service=SERVICE, + exchange=exchange + ).observe(processing_time) + + return jsonify({ + 'decision': decision, + 'version': VERSION, + 'processing_time': processing_time + }) + + @app.route('/metrics') + def metrics(): + return Response(generate_latest(), mimetype=CONTENT_TYPE_LATEST) + + if __name__ == '__main__': + app.run(host='0.0.0.0', port=8080) + +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: bidder + namespace: app-161 + labels: + app: bidder + version: v2.0 +# annotations: +# kubernetes.io/change-cause: "Updated to v2.0 with enhanced bid processing algorithm" +spec: + replicas: 2 # Start with same replica count + selector: + matchLabels: + app: bidder + strategy: + type: RollingUpdate + rollingUpdate: + maxSurge: 1 + maxUnavailable: 0 + template: + metadata: + labels: + app: bidder + version: v2.0 + annotations: + prometheus.io/scrape: "true" + prometheus.io/port: "8080" + prometheus.io/path: "/metrics" + spec: + containers: + - name: bidder + image: me-west1-docker.pkg.dev/robusta-development/development/python-flask-otel:2.2 + imagePullPolicy: IfNotPresent + command: ["python", "/app/app.py"] + ports: + - containerPort: 8080 + name: http + env: + - name: NAMESPACE + valueFrom: + fieldRef: + fieldPath: metadata.namespace + - name: POD_NAME + valueFrom: + fieldRef: + fieldPath: metadata.name + - name: SERVICE_NAME + valueFrom: + fieldRef: + fieldPath: metadata.labels['app'] + volumeMounts: + - name: app-code + mountPath: /app + resources: + requests: + memory: "128Mi" + cpu: "100m" + limits: + memory: "256Mi" + cpu: "500m" + startupProbe: + httpGet: + path: /healthz + port: 8080 + initialDelaySeconds: 5 + periodSeconds: 5 + failureThreshold: 30 # 5 + 5*30 = 155 seconds max startup time + readinessProbe: + httpGet: + path: /healthz + port: 8080 + initialDelaySeconds: 5 + periodSeconds: 5 + livenessProbe: + httpGet: + path: /healthz + port: 8080 + initialDelaySeconds: 5 + periodSeconds: 10 + volumes: + - name: app-code + secret: + secretName: bidder-app-v2 diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/hpa.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/hpa.yaml new file mode 100644 index 000000000..71dd8debd --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/hpa.yaml @@ -0,0 +1,38 @@ +apiVersion: autoscaling/v2 +kind: HorizontalPodAutoscaler +metadata: + name: bidder-hpa + namespace: app-161 + labels: + app: bidder +spec: + scaleTargetRef: + apiVersion: apps/v1 + kind: Deployment + name: bidder + minReplicas: 2 + maxReplicas: 10 + metrics: + - type: Resource + resource: + name: cpu + target: + type: Utilization + averageUtilization: 50 + behavior: + scaleUp: + stabilizationWindowSeconds: 30 + policies: + - type: Percent + value: 100 + periodSeconds: 15 + - type: Pods + value: 2 + periodSeconds: 15 + selectPolicy: Max + scaleDown: + stabilizationWindowSeconds: 300 + policies: + - type: Percent + value: 10 + periodSeconds: 60 diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/k6-v1-traffic.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/k6-v1-traffic.yaml new file mode 100644 index 000000000..e7b4632bc --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/k6-v1-traffic.yaml @@ -0,0 +1,103 @@ +apiVersion: v1 +kind: Secret +metadata: + name: k6-v1-script + namespace: app-161 +type: Opaque +stringData: + test.js: | + import http from 'k6/http'; + import { check, sleep } from 'k6'; + import { Rate } from 'k6/metrics'; + + const bidAcceptanceRate = new Rate('bid_acceptance_rate'); + + export const options = { + scenarios: { + constant_load: { + executor: 'constant-arrival-rate', + rate: 100, // 100 requests per second + timeUnit: '1s', + duration: '2m', // 2 minutes to build up history + preAllocatedVUs: 150, + }, + }, + thresholds: { + 'http_req_duration': ['p(95)<200'], // v1.0 should be fast + 'http_req_failed': ['rate<0.01'], + }, + }; + + const exchanges = ['NordPool', 'EPEX_SPOT', 'EEX', 'OMIE', 'GME']; + + function selectExchange() { + return exchanges[Math.floor(Math.random() * exchanges.length)]; + } + + export default function() { + const exchange = selectExchange(); + const url = 'http://bidder.app-161.svc.cluster.local/bid'; + + const payload = JSON.stringify({ + exchange: exchange, + amount: Math.floor(Math.random() * 1000) + 100, + price: Math.random() * 100 + 20, + }); + + const params = { + headers: { + 'Content-Type': 'application/json', + }, + timeout: '5s', + }; + + const res = http.post(url, payload, params); + + check(res, { + 'status is 200': (r) => r.status === 200, + 'has version': (r) => r.json('version') !== undefined, + 'version is v1.0': (r) => r.json('version') === 'v1.0', + }); + + if (res.status === 200 && res.json('decision') === 'bid') { + bidAcceptanceRate.add(1); + } else { + bidAcceptanceRate.add(0); + } + } + +--- +apiVersion: batch/v1 +kind: Job +metadata: + name: k6-v1-traffic + namespace: app-161 + labels: + app: k6 + phase: v1 +spec: + template: + metadata: + labels: + app: k6 + phase: v1 + spec: + restartPolicy: Never + containers: + - name: k6 + image: loadimpact/k6:0.47.0 + args: ["run", "/scripts/test.js"] + volumeMounts: + - name: script + mountPath: /scripts + resources: + requests: + memory: "256Mi" + cpu: "500m" + limits: + memory: "512Mi" + cpu: "1000m" + volumes: + - name: script + secret: + secretName: k6-v1-script diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/k6-v2-traffic.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/k6-v2-traffic.yaml new file mode 100644 index 000000000..189150e22 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/k6-v2-traffic.yaml @@ -0,0 +1,103 @@ +apiVersion: v1 +kind: Secret +metadata: + name: k6-v2-script + namespace: app-161 +type: Opaque +stringData: + test.js: | + import http from 'k6/http'; + import { check, sleep } from 'k6'; + import { Rate } from 'k6/metrics'; + + const bidAcceptanceRate = new Rate('bid_acceptance_rate'); + + export const options = { + scenarios: { + constant_load: { + executor: 'constant-arrival-rate', + rate: 100, // Same 100 requests per second + timeUnit: '1s', + duration: '2m', // 2 minutes to trigger scaling + preAllocatedVUs: 150, + }, + }, + thresholds: { + 'http_req_duration': ['p(95)<3000'], // v2.0 is slower + 'http_req_failed': ['rate<0.01'], + }, + }; + + const exchanges = ['NordPool', 'EPEX_SPOT', 'EEX', 'OMIE', 'GME']; + + function selectExchange() { + return exchanges[Math.floor(Math.random() * exchanges.length)]; + } + + export default function() { + const exchange = selectExchange(); + const url = 'http://bidder.app-161.svc.cluster.local/bid'; + + const payload = JSON.stringify({ + exchange: exchange, + amount: Math.floor(Math.random() * 1000) + 100, + price: Math.random() * 100 + 20, + }); + + const params = { + headers: { + 'Content-Type': 'application/json', + }, + timeout: '10s', // Increased timeout for slow v2.0 + }; + + const res = http.post(url, payload, params); + + check(res, { + 'status is 200': (r) => r.status === 200, + 'has version': (r) => r.json('version') !== undefined, + 'version is v2.0': (r) => r.json('version') === 'v2.0', + }); + + if (res.status === 200 && res.json('decision') === 'bid') { + bidAcceptanceRate.add(1); + } else { + bidAcceptanceRate.add(0); + } + } + +--- +apiVersion: batch/v1 +kind: Job +metadata: + name: k6-v2-traffic + namespace: app-161 + labels: + app: k6 + phase: v2 +spec: + template: + metadata: + labels: + app: k6 + phase: v2 + spec: + restartPolicy: Never + containers: + - name: k6 + image: loadimpact/k6:0.47.0 + args: ["run", "/scripts/test.js"] + volumeMounts: + - name: script + mountPath: /scripts + resources: + requests: + memory: "256Mi" + cpu: "500m" + limits: + memory: "512Mi" + cpu: "1000m" + volumes: + - name: script + secret: + secretName: k6-v2-script diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/prometheus-config.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/prometheus-config.yaml new file mode 100644 index 000000000..b8fdec814 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/prometheus-config.yaml @@ -0,0 +1,15 @@ +apiVersion: v1 +kind: ConfigMap +metadata: + name: prometheus-config + namespace: app-161 +data: + prometheus.yml: | + global: + scrape_interval: 5s + evaluation_interval: 5s + scrape_configs: + - job_name: 'bidder' + metrics_path: /metrics + static_configs: + - targets: ['bidder.app-161.svc.cluster.local:80'] diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/test_case.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/test_case.yaml new file mode 100644 index 000000000..50eccba43 --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/test_case.yaml @@ -0,0 +1,65 @@ +user_prompt: +- "Users are reporting slow response times from the energy bidding service in namespace app-161. The service was working fine earlier today. Please investigate the cause of the latency." + +expected_output: + - Must identify that the service is running version v2.0 which has degraded performance + - Should compare metrics showing v2.0 processes requests in ~2 seconds vs v1.0's ~50ms +# - Should note that the service has more pods in v2.0 compared to v1.0 (as a consequence, not the main issue) +# - May mention high CPU usage across all pods + - Should conclude the root cause is the version change, not a capacity issue + +tags: + - kubernetes + - prometheus + - medium + +port_forwards: + - namespace: app-161 + service: prometheus + local_port: 9090 + remote_port: 9090 + +setup_timeout: 360 + +before_test: | + set -e + + echo "📁 Creating namespace..." + kubectl create namespace app-161 --save-config --dry-run=client -o yaml | kubectl apply -f - + + echo "⚙️ Setting up monitoring..." + kubectl apply -f prometheus-config.yaml -n app-161 + kubectl apply -f ../../shared/prometheus.yaml -n app-161 + + echo "🚀 Phase 1: Deploy v1.0 (fast version)..." + kubectl apply -f bidder-v1.yaml -n app-161 + kubectl apply -f hpa.yaml -n app-161 + + echo "⏳ Waiting for v1.0 pods..." + kubectl wait --for=condition=ready pod -l app=prometheus -n app-161 --timeout=60s + kubectl wait --for=condition=ready pod -l app=bidder -n app-161 --timeout=60s + + echo "📊 Generate traffic for v1.0 (2 minutes)..." + kubectl apply -f k6-v1-traffic.yaml -n app-161 + kubectl wait --for=condition=complete job/k6-v1-traffic -n app-161 --timeout=150s + + echo "🔄 Phase 2: Upgrade to v2.0 (slow version)..." + kubectl apply -f bidder-v2.yaml -n app-161 + + echo "⏳ Waiting for v2.0 rollout..." + kubectl rollout status deployment/bidder -n app-161 --timeout=60s + + echo "📊 Generate traffic for v2.0 (2 minutes)..." + kubectl apply -f k6-v2-traffic.yaml -n app-161 + kubectl wait --for=condition=complete job/k6-v2-traffic -n app-161 --timeout=150s + + echo "⏳ Waiting for HPA to scale up..." + ./wait-for-scaling.sh + + echo "🧹 Cleaning up k6 jobs to prevent AI detection" + kubectl delete jobs -n app-161 -l app=k6 --ignore-not-found=true + + echo "✅ Test setup completed" + +after_test: | + kubectl delete namespace app-161 || true diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/toolsets.yaml b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/toolsets.yaml new file mode 100644 index 000000000..4a571a90c --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/toolsets.yaml @@ -0,0 +1,9 @@ +toolsets: + kubernetes/core: + enabled: true + kubernetes/logs: + enabled: true + prometheus/metrics: + enabled: true + config: + prometheus_url: http://localhost:9090 diff --git a/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/wait-for-scaling.sh b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/wait-for-scaling.sh new file mode 100755 index 000000000..ff8f116fe --- /dev/null +++ b/tests/llm/fixtures/test_ask_holmes/161_bidding_version_performance/wait-for-scaling.sh @@ -0,0 +1,42 @@ +#!/usr/bin/env bash + +set -euo pipefail + +NS="${NS:-app-161}" +TARGET_REPLICAS=3 # We expect at least 3 replicas after scaling +MAX_WAIT=120 # Maximum 2 minutes to wait for scaling + +echo ">>> Waiting for HPA to scale up the deployment..." + +start_time=$(date +%s) +while true; do + current_replicas=$(kubectl get deployment bidder -n "${NS}" -o jsonpath='{.status.replicas}') + + echo "Current replicas: ${current_replicas}" + + if [ "${current_replicas}" -ge "${TARGET_REPLICAS}" ]; then + echo "✓ Deployment has scaled to ${current_replicas} replicas" + break + fi + + elapsed=$(($(date +%s) - start_time)) + if [ "${elapsed}" -gt "${MAX_WAIT}" ]; then + echo "⚠️ WARNING: Deployment hasn't scaled to ${TARGET_REPLICAS} replicas after ${MAX_WAIT} seconds" + echo "Current replicas: ${current_replicas}" + echo "Continuing anyway..." + break + fi + + echo "Waiting for scaling... (${elapsed}s elapsed)" + sleep 5 +done + +# Show HPA status +echo "" +echo ">>> HPA Status:" +kubectl get hpa bidder-hpa -n "${NS}" +kubectl describe hpa bidder-hpa -n "${NS}" | grep -A 5 "Metrics:" + +echo "" +echo ">>> Deployment Status:" +kubectl get deployment bidder -n "${NS}" -o wide