diff --git a/ci-operator/config/openshift/distributed-tracing-console-plugin/openshift-distributed-tracing-console-plugin-main__upstream-amd64-aws.yaml b/ci-operator/config/openshift/distributed-tracing-console-plugin/openshift-distributed-tracing-console-plugin-main__upstream-amd64-aws.yaml index b49c05ef03da2..392482b677f75 100644 --- a/ci-operator/config/openshift/distributed-tracing-console-plugin/openshift-distributed-tracing-console-plugin-main__upstream-amd64-aws.yaml +++ b/ci-operator/config/openshift/distributed-tracing-console-plugin/openshift-distributed-tracing-console-plugin-main__upstream-amd64-aws.yaml @@ -53,11 +53,7 @@ tests: steps: cluster_profile: azure-observability env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md BASE_DOMAIN: observability.azure.devcluster.openshift.com - post: - - ref: openshift-observability-qe-agent - - chain: cucushift-installer-rehearse-azure-ipi-deprovision test: - ref: distributed-tracing-tests-tracing-ui-upstream workflow: cucushift-installer-rehearse-azure-ipi diff --git a/ci-operator/config/openshift/distributed-tracing-qe/openshift-distributed-tracing-qe-main__ocp-4.16-disconnected.yaml b/ci-operator/config/openshift/distributed-tracing-qe/openshift-distributed-tracing-qe-main__ocp-4.16-disconnected.yaml index 664d1232df7f2..d64ac7e11d9fa 100644 --- a/ci-operator/config/openshift/distributed-tracing-qe/openshift-distributed-tracing-qe-main__ocp-4.16-disconnected.yaml +++ b/ci-operator/config/openshift/distributed-tracing-qe/openshift-distributed-tracing-qe-main__ocp-4.16-disconnected.yaml @@ -47,7 +47,6 @@ tests: steps: cluster_profile: gcp-observability env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md MULTISTAGE_PARAM_OVERRIDE_OTEL_INDEX_IMAGE: brew.registry.redhat.io/rh-osbs/iib:1155560 MULTISTAGE_PARAM_OVERRIDE_TEMPO_INDEX_IMAGE: brew.registry.redhat.io/rh-osbs/iib:1157120 OPERATORS: | @@ -58,9 +57,6 @@ tests: ] SELF_MANAGED_ADDITIONAL_CA: "true" SELF_MANAGED_REGISTRY_CERT: "true" - post: - - ref: openshift-observability-qe-agent - - chain: cucushift-installer-rehearse-gcp-ipi-disconnected-deprovision test: - ref: distributed-tracing-install-disconnected - ref: install-operators diff --git a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.12-stage.yaml b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.12-stage.yaml index fd9162dcff3f1..75ed4b1a69975 100644 --- a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.12-stage.yaml +++ b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.12-stage.yaml @@ -44,7 +44,6 @@ tests: steps: cluster_profile: gcp-observability env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md MULTISTAGE_PARAM_OVERRIDE_TEMPO_INDEX_IMAGE: brew.registry.redhat.io/rh-osbs/iib:1157117 MULTISTAGE_PARAM_OVERRIDE_TEMPO_TESTS_BRANCH: rhosdt-3.10 OPERATORS: | @@ -56,9 +55,6 @@ tests: {"name": "serverless-operator", "source": "redhat-operators", "channel": "stable", "install_namespace": "openshift-serverless", "target_namespaces": "", "operator_group": "openshift-serverless"} ] SKIP_TESTS: tests/e2e-openshift-object-stores/* tests/e2e-openshift-tls-profile/03-tls-profile - post: - - ref: openshift-observability-qe-agent - - chain: cucushift-installer-rehearse-gcp-ipi-deprovision test: - ref: distributed-tracing-install-tempo-konflux-catalogsource - ref: install-operators diff --git a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.14-arm-stage.yaml b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.14-arm-stage.yaml index ec4458bfa83db..4ed8740209015 100644 --- a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.14-arm-stage.yaml +++ b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.14-arm-stage.yaml @@ -57,7 +57,6 @@ tests: dependencies: OPENSHIFT_INSTALL_RELEASE_IMAGE_OVERRIDE: release:arm64-latest env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md BASE_DOMAIN: observability.azure.devcluster.openshift.com COMPUTE_NODE_TYPE: Standard_D4ps_v5 MULTISTAGE_PARAM_OVERRIDE_TEMPO_INDEX_IMAGE: quay.io/redhat-user-workloads/rhosdt-tenant/tempo/tempo-bundle@sha256:90cfc733bf3815568738cf089ed13fa232561b0c24ce0c99b71154753254b1be @@ -72,9 +71,6 @@ tests: ] SKIP_TESTS: tests/e2e/generate tests/e2e-openshift/monolithic-multitenancy-static tests/e2e-openshift-serverless/* tests/e2e-openshift-object-stores/* tests/e2e-openshift-tls-profile/03-tls-profile - post: - - ref: openshift-observability-qe-agent - - chain: cucushift-installer-rehearse-azure-ipi-deprovision test: - ref: idp-htpasswd - as: install-tempo-operator diff --git a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.16-ibm-z-stage.yaml b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.16-ibm-z-stage.yaml index 625b266b8f0b8..f917ad5e01c9c 100644 --- a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.16-ibm-z-stage.yaml +++ b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.16-ibm-z-stage.yaml @@ -43,7 +43,6 @@ tests: {{end}}' steps: env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md MULTISTAGE_PARAM_OVERRIDE_TEMPO_INDEX_IMAGE: quay.io/redhat-user-workloads/rhosdt-tenant/tempo/tempo-bundle@sha256:90cfc733bf3815568738cf089ed13fa232561b0c24ce0c99b71154753254b1be MULTISTAGE_PARAM_OVERRIDE_TEMPO_TESTS_BRANCH: rhosdt-3.10 OPERATORS: | @@ -57,9 +56,6 @@ tests: ] SKIP_TESTS: tests/e2e/generate tests/e2e/gateway tests/e2e-openshift/monolithic-multitenancy-static tests/e2e-openshift-ossm/* tests/e2e-openshift-object-stores/* tests/e2e-openshift-serverless/* - post: - - ref: openshift-observability-qe-agent - - ref: openshift-observability-ibm-z-cluster-destroy test: - ref: idp-htpasswd - as: install-tempo-operator diff --git a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.17-fips-stage.yaml b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.17-fips-stage.yaml index 71d6b26c748ff..9e1dc5d63dbc9 100644 --- a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.17-fips-stage.yaml +++ b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.17-fips-stage.yaml @@ -44,7 +44,6 @@ tests: steps: cluster_profile: gcp-observability env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md FIPS_ENABLED: "true" MULTISTAGE_PARAM_OVERRIDE_TEMPO_INDEX_IMAGE: brew.registry.redhat.io/rh-osbs/iib:1157124 MULTISTAGE_PARAM_OVERRIDE_TEMPO_TESTS_BRANCH: rhosdt-3.10 @@ -59,9 +58,6 @@ tests: {"name": "serverless-operator", "source": "redhat-operators", "channel": "stable", "install_namespace": "openshift-serverless", "target_namespaces": "", "operator_group": "openshift-serverless"} ] SKIP_TESTS: tests/e2e-openshift-object-stores/* tests/e2e-openshift-tls-profile/03-tls-profile - post: - - ref: openshift-observability-qe-agent - - chain: cucushift-installer-rehearse-gcp-ipi-deprovision test: - ref: distributed-tracing-install-tempo-konflux-catalogsource - ref: install-operators diff --git a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.17-ibm-p-stage.yaml b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.17-ibm-p-stage.yaml index 75c11e71d79e7..61fc3b6ab9667 100644 --- a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.17-ibm-p-stage.yaml +++ b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.17-ibm-p-stage.yaml @@ -43,7 +43,6 @@ tests: {{end}}' steps: env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md FIPS_ENABLED: "false" MULTISTAGE_PARAM_OVERRIDE_TEMPO_INDEX_IMAGE: quay.io/redhat-user-workloads/rhosdt-tenant/tempo/tempo-bundle@sha256:90cfc733bf3815568738cf089ed13fa232561b0c24ce0c99b71154753254b1be MULTISTAGE_PARAM_OVERRIDE_TEMPO_TESTS_BRANCH: rhosdt-3.10 @@ -59,9 +58,6 @@ tests: ] SKIP_TESTS: tests/e2e/generate tests/e2e/gateway tests/e2e-openshift/monolithic-multitenancy-static tests/e2e-openshift-ossm/* tests/e2e-openshift-object-stores/* tests/e2e-openshift-serverless/* - post: - - ref: openshift-observability-qe-agent - - ref: openshift-observability-ibm-p-cluster-destroy test: - ref: idp-htpasswd - as: install-tempo-operator diff --git a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.19-stage.yaml b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.19-stage.yaml index 44bda7528541e..e7ba68c2a5acc 100644 --- a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.19-stage.yaml +++ b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.19-stage.yaml @@ -48,7 +48,6 @@ tests: {{end}}' steps: env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md MULTISTAGE_PARAM_OVERRIDE_TEMPO_INDEX_IMAGE: brew.registry.redhat.io/rh-osbs/iib:1157126 MULTISTAGE_PARAM_OVERRIDE_TEMPO_TESTS_BRANCH: rhosdt-3.10 OPERATORS: | @@ -62,9 +61,6 @@ tests: {"name": "serverless-operator", "source": "redhat-operators", "channel": "stable", "install_namespace": "openshift-serverless", "target_namespaces": "", "operator_group": "openshift-serverless"} ] SKIP_TESTS: tests/e2e-openshift-object-stores/* - post: - - ref: openshift-observability-qe-agent - - chain: gather test: - ref: distributed-tracing-install-tempo-konflux-catalogsource - ref: install-operators diff --git a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.20-stage.yaml b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.20-stage.yaml index 18dd22e96e21d..edb6cbc3b9c66 100644 --- a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.20-stage.yaml +++ b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.20-stage.yaml @@ -44,7 +44,6 @@ tests: steps: cluster_profile: gcp-observability env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md MULTISTAGE_PARAM_OVERRIDE_TEMPO_INDEX_IMAGE: brew.registry.redhat.io/rh-osbs/iib:1157128 MULTISTAGE_PARAM_OVERRIDE_TEMPO_TESTS_BRANCH: rhosdt-3.10 OPERATORS: | @@ -58,9 +57,6 @@ tests: {"name": "serverless-operator", "source": "redhat-operators", "channel": "stable", "install_namespace": "openshift-serverless", "target_namespaces": "", "operator_group": "openshift-serverless"} ] SKIP_TESTS: tests/e2e-openshift-object-stores/* - post: - - ref: openshift-observability-qe-agent - - chain: cucushift-installer-rehearse-gcp-ipi-deprovision test: - ref: distributed-tracing-install-tempo-konflux-catalogsource - ref: install-operators diff --git a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.21-stage.yaml b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.21-stage.yaml index 95a8c5024a93d..f4c29240047bd 100644 --- a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.21-stage.yaml +++ b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__tempo-product-ocp-4.21-stage.yaml @@ -40,7 +40,6 @@ tests: steps: cluster_profile: gcp-observability env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md MULTISTAGE_PARAM_OVERRIDE_TEMPO_INDEX_IMAGE: brew.registry.redhat.io/rh-osbs/iib:1157127 MULTISTAGE_PARAM_OVERRIDE_TEMPO_TESTS_BRANCH: rhosdt-3.10 OPERATORS: | @@ -54,9 +53,6 @@ tests: {"name": "serverless-operator", "source": "redhat-operators", "channel": "stable", "install_namespace": "openshift-serverless", "target_namespaces": "", "operator_group": "openshift-serverless"} ] SKIP_TESTS: tests/e2e-openshift-object-stores/* - post: - - ref: openshift-observability-qe-agent - - chain: cucushift-installer-rehearse-gcp-ipi-deprovision test: - ref: distributed-tracing-install-tempo-konflux-catalogsource - ref: install-operators diff --git a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__upstream-ocp-4.12-amd64.yaml b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__upstream-ocp-4.12-amd64.yaml index 408e05e6800a9..8a05c6cb7d7a9 100644 --- a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__upstream-ocp-4.12-amd64.yaml +++ b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__upstream-ocp-4.12-amd64.yaml @@ -49,7 +49,6 @@ tests: steps: cluster_profile: aws-observability env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md BASE_DOMAIN: devobscluster.devcluster.openshift.com OPERATORS: | [ @@ -59,9 +58,6 @@ tests: {"name": "serverless-operator", "source": "redhat-operators", "channel": "stable", "install_namespace": "openshift-serverless", "target_namespaces": "", "operator_group": "openshift-serverless"} ] SKIP_TESTS: tests/e2e-openshift-object-stores/* - post: - - ref: openshift-observability-qe-agent - - chain: cucushift-installer-rehearse-aws-ipi-deprovision test: - as: install cli: latest diff --git a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__upstream-ocp-4.21-amd64.yaml b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__upstream-ocp-4.21-amd64.yaml index fc897aa5dece9..578b08714f96c 100644 --- a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__upstream-ocp-4.21-amd64.yaml +++ b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__upstream-ocp-4.21-amd64.yaml @@ -50,7 +50,6 @@ tests: steps: cluster_profile: azure-observability env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md BASE_DOMAIN: observability.azure.devcluster.openshift.com COMPUTE_NODE_TYPE: Standard_D4s_v3 CONTROL_PLANE_INSTANCE_TYPE: Standard_D8s_v3 @@ -64,9 +63,6 @@ tests: {"name": "serverless-operator", "source": "redhat-operators", "channel": "stable", "install_namespace": "openshift-serverless", "target_namespaces": "", "operator_group": "openshift-serverless"} ] SKIP_TESTS: tests/e2e-openshift-object-stores/* - post: - - ref: openshift-observability-qe-agent - - chain: cucushift-installer-rehearse-azure-ipi-deprovision test: - as: install cli: latest diff --git a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__upstream-ocp-4.22-amd64.yaml b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__upstream-ocp-4.22-amd64.yaml index 1136f51355f1c..b40704e293627 100644 --- a/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__upstream-ocp-4.22-amd64.yaml +++ b/ci-operator/config/openshift/grafana-tempo-operator/openshift-grafana-tempo-operator-main__upstream-ocp-4.22-amd64.yaml @@ -46,7 +46,7 @@ tests: steps: cluster_profile: gcp-observability env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md + AGENT_SKILL: RHOSDT OPERATORS: | [ {"name": "opentelemetry-product", "source": "redhat-operators", "channel": "stable", "install_namespace": "openshift-opentelemetry-operator", "target_namespaces": "", "operator_group": "openshift-opentelemetry-operator"}, diff --git a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.12-stage.yaml b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.12-stage.yaml index 840f3ad8a8f99..f04ab5cb8c7b8 100644 --- a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.12-stage.yaml +++ b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.12-stage.yaml @@ -44,7 +44,6 @@ tests: steps: cluster_profile: gcp-observability env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md MULTISTAGE_PARAM_OVERRIDE_OTEL_INDEX_IMAGE: brew.registry.redhat.io/rh-osbs/iib:1155563 MULTISTAGE_PARAM_OVERRIDE_OTEL_TESTS_BRANCH: rhosdt-3.10 OPERATORS: | @@ -56,9 +55,6 @@ tests: SKIP_TESTS: tests/e2e/smoke-ip-families tests/e2e-openshift/export-to-cluster-logging-lokistack tests/e2e-otel/*aws* tests/e2e-openshift/kafka tests/e2e-otel/google* tests/e2e-targetallocator/targetallocator-prometheuscr tests/e2e-instrumentation/instrumentation-go tests/e2e/httpRoute - post: - - ref: openshift-observability-qe-agent - - chain: cucushift-installer-rehearse-gcp-ipi-deprovision test: - ref: distributed-tracing-install-otel-konflux-catalogsource - ref: install-operators diff --git a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.14-arm-stage.yaml b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.14-arm-stage.yaml index 77677a18eba5b..270c99f045c5c 100644 --- a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.14-arm-stage.yaml +++ b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.14-arm-stage.yaml @@ -57,7 +57,6 @@ tests: dependencies: OPENSHIFT_INSTALL_RELEASE_IMAGE_OVERRIDE: release:arm64-latest env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md BASE_DOMAIN: observability.azure.devcluster.openshift.com COMPUTE_NODE_TYPE: Standard_D4ps_v5 MULTISTAGE_PARAM_OVERRIDE_OTEL_INDEX_IMAGE: quay.io/redhat-user-workloads/rhosdt-tenant/otel/opentelemetry-bundle@sha256:35ea32f1d12e6abfce4be19e3a69589ed8d777f6a606aa29a898a6c9c634b581 @@ -74,9 +73,6 @@ tests: tests/e2e/smoke-ip-families tests/e2e-targetallocator/targetallocator-prometheuscr tests/e2e-otel/*aws* tests/e2e-openshift/must-gather tests/e2e-otel/google* tests/e2e-instrumentation/instrumentation-go tests/e2e/httpRoute - post: - - ref: openshift-observability-qe-agent - - chain: cucushift-installer-rehearse-azure-ipi-deprovision test: - ref: idp-htpasswd - as: install-opentelemetry-operator diff --git a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.16-ibm-z-stage.yaml b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.16-ibm-z-stage.yaml index d6e160c44f183..5059b345b47c8 100644 --- a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.16-ibm-z-stage.yaml +++ b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.16-ibm-z-stage.yaml @@ -43,7 +43,6 @@ tests: {{end}}' steps: env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md MULTISTAGE_PARAM_OVERRIDE_OTEL_INDEX_IMAGE: quay.io/redhat-user-workloads/rhosdt-tenant/otel/opentelemetry-bundle@sha256:35ea32f1d12e6abfce4be19e3a69589ed8d777f6a606aa29a898a6c9c634b581 MULTISTAGE_PARAM_OVERRIDE_OTEL_TESTS_BRANCH: rhosdt-3.10 OPERATORS: | @@ -62,9 +61,6 @@ tests: tests/e2e-openshift/route tests/e2e/smoke-ip-families tests/e2e-openshift/export-to-cluster-logging-lokistack tests/e2e-targetallocator/targetallocator-prometheuscr tests/e2e-otel/*aws* tests/e2e-openshift/must-gather tests/e2e-otel/google* - post: - - ref: openshift-observability-qe-agent - - ref: openshift-observability-ibm-z-cluster-destroy test: - as: install-opentelemetry-operator cli: latest diff --git a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.17-fips-stage.yaml b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.17-fips-stage.yaml index 0636e45c7f02f..810f4156860da 100644 --- a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.17-fips-stage.yaml +++ b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.17-fips-stage.yaml @@ -44,7 +44,6 @@ tests: steps: cluster_profile: gcp-observability env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md FIPS_ENABLED: "true" MULTISTAGE_PARAM_OVERRIDE_OTEL_INDEX_IMAGE: brew.registry.redhat.io/rh-osbs/iib:1155558 MULTISTAGE_PARAM_OVERRIDE_OTEL_TESTS_BRANCH: rhosdt-3.10 @@ -60,9 +59,6 @@ tests: SKIP_TESTS: tests/e2e/smoke-ip-families tests/e2e-otel/*aws* tests/e2e-otel/oidcauthextension tests/e2e-openshift/export-to-cluster-logging-lokistack tests/e2e-otel/google* tests/e2e/smoke-ports tests/e2e-instrumentation/instrumentation-go - post: - - ref: openshift-observability-qe-agent - - chain: cucushift-installer-rehearse-gcp-ipi-deprovision test: - ref: distributed-tracing-install-otel-konflux-catalogsource - ref: install-operators diff --git a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.17-ibm-p-stage.yaml b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.17-ibm-p-stage.yaml index 904fa2b804ce6..122ea45ac40e9 100644 --- a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.17-ibm-p-stage.yaml +++ b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.17-ibm-p-stage.yaml @@ -43,7 +43,6 @@ tests: {{end}}' steps: env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md FIPS_ENABLED: "false" MULTISTAGE_PARAM_OVERRIDE_OTEL_INDEX_IMAGE: quay.io/redhat-user-workloads/rhosdt-tenant/otel/opentelemetry-bundle@sha256:35ea32f1d12e6abfce4be19e3a69589ed8d777f6a606aa29a898a6c9c634b581 MULTISTAGE_PARAM_OVERRIDE_OTEL_TESTS_BRANCH: rhosdt-3.10 @@ -59,9 +58,6 @@ tests: SKIP_TESTS: tests/e2e-openshift/must-gather tests/e2e-targetallocator/* tests/e2e-targetallocator-cr/* tests/e2e-multi-instrumentation/* tests/e2e-otel/* tests/e2e-opampbridge/* tests/e2e-pdb/* tests/e2e-instrumentation/* tests/e2e-prometheuscr/* tests/e2e-otel/google* - post: - - ref: openshift-observability-qe-agent - - ref: openshift-observability-ibm-p-cluster-destroy test: - as: install-opentelemetry-operator cli: latest diff --git a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.19-stage.yaml b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.19-stage.yaml index 9d140ed76828a..d193747df5659 100644 --- a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.19-stage.yaml +++ b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.19-stage.yaml @@ -44,7 +44,6 @@ tests: steps: cluster_profile: azure-observability env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md BASE_DOMAIN: observability.azure.devcluster.openshift.com COMPUTE_NODE_TYPE: Standard_D4s_v3 CONTROL_PLANE_INSTANCE_TYPE: Standard_D8s_v3 @@ -61,9 +60,6 @@ tests: ] SKIP_TESTS: tests/e2e/smoke-ip-families tests/e2e-otel/*aws* tests/e2e-otel/google* tests/e2e-instrumentation/instrumentation-go - post: - - ref: openshift-observability-qe-agent - - chain: cucushift-installer-rehearse-azure-ipi-deprovision test: - ref: distributed-tracing-install-otel-konflux-catalogsource - ref: install-operators diff --git a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.20-stage.yaml b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.20-stage.yaml index 9186d4a2e609c..88e261c57e76b 100644 --- a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.20-stage.yaml +++ b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.20-stage.yaml @@ -48,7 +48,6 @@ tests: {{end}}' steps: env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md MULTISTAGE_PARAM_OVERRIDE_OTEL_INDEX_IMAGE: brew.registry.redhat.io/rh-osbs/iib:1155556 MULTISTAGE_PARAM_OVERRIDE_OTEL_TESTS_BRANCH: rhosdt-3.10 OPERATORS: | @@ -62,9 +61,6 @@ tests: ] SKIP_TESTS: tests/e2e/smoke-ip-families tests/e2e-otel/*aws* tests/e2e-otel/google* tests/e2e-instrumentation/instrumentation-go - post: - - ref: openshift-observability-qe-agent - - chain: gather test: - ref: distributed-tracing-install-otel-konflux-catalogsource - ref: install-operators diff --git a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.21-stage.yaml b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.21-stage.yaml index 340db26140787..2a172b5a249c5 100644 --- a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.21-stage.yaml +++ b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__opentelemetry-product-ocp-4.21-stage.yaml @@ -40,7 +40,6 @@ tests: steps: cluster_profile: gcp-observability env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md MULTISTAGE_PARAM_OVERRIDE_OTEL_INDEX_IMAGE: brew.registry.redhat.io/rh-osbs/iib:1155557 MULTISTAGE_PARAM_OVERRIDE_OTEL_TESTS_BRANCH: rhosdt-3.10 OPERATORS: | @@ -54,9 +53,6 @@ tests: ] SKIP_TESTS: tests/e2e/smoke-ip-families tests/e2e-otel/*aws* tests/e2e-otel/google* tests/e2e-instrumentation/instrumentation-go - post: - - ref: openshift-observability-qe-agent - - chain: cucushift-installer-rehearse-gcp-ipi-deprovision test: - ref: distributed-tracing-install-otel-konflux-catalogsource - ref: install-operators diff --git a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__upstream-ocp-4.12-amd64.yaml b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__upstream-ocp-4.12-amd64.yaml index 7fc1ba7779d27..11b15e0125e46 100644 --- a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__upstream-ocp-4.12-amd64.yaml +++ b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__upstream-ocp-4.12-amd64.yaml @@ -107,7 +107,6 @@ tests: steps: cluster_profile: aws-observability env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md BASE_DOMAIN: devobscluster.devcluster.openshift.com OPERATORS: | [ @@ -117,9 +116,6 @@ tests: SKIP_TESTS: tests/e2e/smoke-ip-families tests/e2e-openshift/export-to-cluster-logging-lokistack tests/e2e-otel/*aws* tests/e2e-openshift/kafka tests/e2e-otel/google* tests/e2e-targetallocator/targetallocator-prometheuscr tests/e2e-instrumentation/instrumentation-go tests/e2e/httpRoute - post: - - ref: openshift-observability-qe-agent - - chain: cucushift-installer-rehearse-aws-ipi-deprovision test: - as: install cli: latest diff --git a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__upstream-ocp-4.21-amd64.yaml b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__upstream-ocp-4.21-amd64.yaml index ec38f9f54be2a..10be887a10130 100644 --- a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__upstream-ocp-4.21-amd64.yaml +++ b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__upstream-ocp-4.21-amd64.yaml @@ -108,7 +108,6 @@ tests: steps: cluster_profile: azure-observability env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md BASE_DOMAIN: observability.azure.devcluster.openshift.com COMPUTE_NODE_TYPE: Standard_D4s_v3 CONTROL_PLANE_INSTANCE_TYPE: Standard_D8s_v3 @@ -122,9 +121,6 @@ tests: ] SKIP_TESTS: tests/e2e/smoke-ip-families tests/e2e-otel/*aws* tests/e2e-otel/google* tests/e2e-instrumentation/instrumentation-go - post: - - ref: openshift-observability-qe-agent - - chain: cucushift-installer-rehearse-azure-ipi-deprovision test: - as: install cli: latest diff --git a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__upstream-ocp-4.22-amd64.yaml b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__upstream-ocp-4.22-amd64.yaml index 0ba2f1f660baa..797294770ef3b 100644 --- a/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__upstream-ocp-4.22-amd64.yaml +++ b/ci-operator/config/openshift/open-telemetry-opentelemetry-operator/openshift-open-telemetry-opentelemetry-operator-main__upstream-ocp-4.22-amd64.yaml @@ -104,7 +104,7 @@ tests: steps: cluster_profile: gcp-observability env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md + AGENT_SKILL: RHOSDT OPERATORS: | [ {"name": "tempo-product", "source": "redhat-operators", "channel": "stable", "install_namespace": "openshift-tempo-operator", "target_namespaces": "", "operator_group": "openshift-tempo-operator"}, diff --git a/ci-operator/step-registry/openshift-observability/qe-agent/README.md b/ci-operator/step-registry/openshift-observability/qe-agent/README.md index 366e64d8b9574..d8ca435d8177c 100644 --- a/ci-operator/step-registry/openshift-observability/qe-agent/README.md +++ b/ci-operator/step-registry/openshift-observability/qe-agent/README.md @@ -5,11 +5,35 @@ Agentic post-step for OpenShift Observability teams that autonomously triages e2 When a test step reports failures, the agent reads the JUnit XML reports and a context file from `SHARED_DIR`, then runs Claude with a team-provided skill to diagnose the root cause and — depending on the skill — attempt remediation (re-run failing tests, propose a fix, write a bug report, etc.). The step is a **no-op** (exits immediately with no cost) when: -- `AGENT_SKILL_URL` is not set +- `AGENT_SKILL` is not set - `SHARED_DIR/qe-agent-context.json` is absent (test step did not run or skipped the trap) - `has_test_failures` in the context file is `false` (all tests passed) -Full Claude output is streamed to CI logs in real-time and saved to `ARTIFACT_DIR/qe-agent-output.json` for post-run inspection. +Claude's full session output (stream-json) is captured to a temporary file and never written to the CI build-log or uploaded to GCS. Two derived artifacts are extracted from it and saved to `ARTIFACT_DIR` after the session completes: a cost/usage record and a Bash command audit log (see [Audit artifacts](#audit-artifacts)). + +--- + +## Why use this step + +The traditional workflow for debugging a CI test failure is: + +1. Provision a new cluster (~30–45 minutes) +2. Set up the test environment (operators, dependencies, configuration) +3. Run the failing test suite +4. Inspect logs and iterate + +End-to-end this typically takes **1–2 hours of cluster time**, which at OpenShift CI cloud rates is substantially more expensive than the AI API cost — and requires an engineer's attention throughout. + +The qe-agent step runs **inside the already-provisioned CI job**, against the cluster that just ran the tests. The environment is already set up. The failing test artifacts are already present. Claude reruns the exact failing tests, collects diagnostics, and produces a root-cause analysis and proposed fix — all autonomously, at a cost of **$3–5 in AI API spend** per run based on observed production runs. + +The cost can be reduced further by: +- Implementing cross-run pattern awareness using Sippy to skip analysis for known failures +- Reducing the rerun count in the flakiness confirmation loop +- Switching to a smaller model for simpler triage tasks + +Even at the current ceiling of $5, a single qe-agent run replaces most of the value of a 1–2 hour manual debug session on a freshly provisioned cluster. + +> **Where to add this step**: Target upstream testing CI jobs first, not the full job matrix. Upstream jobs run against the latest operator and test code, so they surface both product regressions and test issues at the earliest point in the development cycle — exactly where autonomous triage adds the most value. Adding the step to every job (stage, product, multi-arch, disconnected) spreads AI budget across runs where failures are often already understood, and increases noise before the step has proven itself in your environment. Until cross-run pattern awareness via Sippy is implemented — preventing the agent from running against failures that are already known and tracked — keep the step limited to upstream jobs where each failure is more likely to be novel and worth investigating. --- @@ -98,9 +122,9 @@ Replace `step_script_ref` with the path to your commands script relative to the > **Note on JUnit XML size**: `SHARED_DIR` is backed by a Kubernetes Secret with a 1 MiB limit shared across all files. JUnit XMLs are safe to copy; do **not** copy binary artifacts such as Cypress screenshots. -### 4. Add the step and set `AGENT_SKILL_URL` in your CI config +### 4. Add the step and set `AGENT_SKILL` in your CI config -Add `openshift-observability-qe-agent` to the `post:` phase of your test and set `AGENT_SKILL_URL` in the `env:` block of the same test to the raw URL of your team's `SKILL.md`: +Add `openshift-observability-qe-agent` to the `post:` phase of your test and set `AGENT_SKILL` in the `env:` block to your team's skill name: ```yaml tests: @@ -108,7 +132,7 @@ tests: steps: cluster_profile: env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift//main/plugins/qe-agent/skills/SKILL.md + AGENT_SKILL: RHOSDT # ... other env vars post: - ref: openshift-observability-qe-agent @@ -121,17 +145,81 @@ tests: ## Agent Skill -The skill is a `SKILL.md` Markdown file hosted in your team's QE repository. It is fetched at runtime by the step and passed to Claude as the system prompt. The skill defines how Claude should approach the failure: what tests to re-run, what logs to collect, how to classify the root cause, and what output to produce. +Skills are Markdown files hosted in this step's `skills/` directory within the `openshift/release` repository: + +```text +ci-operator/step-registry/openshift-observability/qe-agent/skills/ +├── RHOSDT.md ← Red Hat OpenShift Distributed Tracing (Tempo, OTel, Tracing UI) +└── OWNERS +``` + +Each skill is fetched at runtime from `https://raw.githubusercontent.com/openshift/release/main/...` and passed to Claude as its system prompt. The skill defines how Claude should approach the failure: what tests to re-run, what logs to collect, how to classify the root cause, and what output to produce. + +### Adding a new team skill -See the [Distributed Tracing QE skill](https://github.com/openshift/distributed-tracing-qe/blob/main/plugins/qe-agent/skills/SKILL.md) as a reference implementation. +1. Create `ci-operator/step-registry/openshift-observability/qe-agent/skills/.md` +2. Add your team identifier to the `OWNERS` file in `skills/` +3. Open a PR to `openshift/release` — the step OWNERS review and approve it +4. Set `AGENT_SKILL: ` in your CI config -### AGENT_SKILL_URL +Skill names must be alphanumeric (hyphens and underscores allowed). The step rejects any value that does not match `^[A-Za-z0-9_-]+$` to prevent path traversal. + +### AGENT_SKILL | Property | Value | |---|---| | Required | Yes | -| Allowlist | Must start with `https://raw.githubusercontent.com/openshift/` — any other URL is rejected at runtime | -| Example | `https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md` | +| Format | Alphanumeric, hyphens, underscores only | +| Resolves to | `ci-operator/step-registry/openshift-observability/qe-agent/skills/.md` in `openshift/release` | +| Example | `RHOSDT` | + +--- + +## Blast Radius and Risk Profile + +This step grants Claude Code CLI unrestricted Bash access inside a CI pod that holds live cluster credentials. Before adopting it, understand what Claude can and cannot do. + +### What Claude can do + +| Capability | Scope | +|---|---| +| Run shell commands | Full pod OS access (non-root). No explicit sandbox — the skill file is the constraint. | +| `kubectl` / `oc` | Any operation allowed by the pod's service account RBAC: create, delete, patch namespaces, pods, ClusterRoles, CSVs, CRDs, and more. | +| Read files | Any file visible to the pod process, including mounted secrets, `KUBECONFIG`, `SHARED_DIR`, and `ARTIFACT_DIR`. | +| Write files | Anything writable by the pod: `ARTIFACT_DIR` (uploaded to GCS), `SHARED_DIR` (shared with other post-steps), `/tmp`. | +| Clone git repos | Network access to GitHub is available; the skill instructs `git clone` during test environment setup. | + +### What Claude cannot do + +- **No outbound HTTP** — `WebFetch` is not in `allowedTools`; Claude cannot call arbitrary external URLs. +- **No git push** — no git credentials are mounted; file changes are confined to the pod. +- **No cross-tenant cluster access** — RBAC bounds apply; Claude cannot reach other teams' clusters or namespaces. +- **No persistence beyond the job** — the test cluster is ephemeral and torn down by the deprovision chain after every job. + +### Worst-case scenarios + +**Unexpected cluster mutation** — Claude misinterprets a skill step and deletes a namespace or patches a resource it should not touch. Not a concern in practice: the step runs in the `post:` phase against a test-only cluster, and the deprovision chain that follows destroys the entire cluster regardless. No production or shared infrastructure is reachable. + +**Runaway session** — Claude enters a reasoning loop, consuming turns and Vertex AI budget without making progress. Hard-bounded on two axes: `--max-budget-usd 5` stops the session the moment spend reaches $5 — preventing runaway cost from exhausting the Vertex AI cost center budget — and the 90-minute wall-clock timeout is the outer limit. `best_effort: true` ensures neither bound can block the pipeline. + +**ARTIFACT_DIR pollution** — Claude writes unexpected files to `ARTIFACT_DIR`, which are then uploaded to GCS. The skill scopes Claude to documented output paths, but this is a prompt-level control, not a technical one. + +**SHARED_DIR pollution** — Claude writes unexpected files to `SHARED_DIR`, which is shared with the deprovision post-step. The deprovision step ignores unexpected files in practice, but this is not formally enforced. + +### Audit artifacts + +After every run, two files are written to `ARTIFACT_DIR` for post-incident review: + +| File | Contents | +|---|---| +| `qe-agent-usage.json` | Token counts (input, output, cache), USD cost, turn count, and wall-clock duration. | +| `qe-agent-commands.log` | Every Bash command Claude executed — the command strings only, not their output. Cluster data (pod logs, events, API responses) is deliberately excluded. | + +The full stream-json session output (which includes cluster logs, API responses, and `--verbose` traces) is captured to a temporary file in the pod and deleted on exit — it never reaches the CI build-log or GCS. Only these two derived files, which contain no cluster data, are written to `ARTIFACT_DIR`. + +### Required: private Prow deck only + +This step **must only be used with jobs backed by a private Prow deck** (login required to view artifacts). Claude reads cluster diagnostics — pod logs, events, resource specs — that may contain internal IP addresses, service URLs, error messages, and configuration details that must not be world-readable. Do not attach this step to any job whose artifacts are publicly accessible without authentication. --- @@ -139,7 +227,7 @@ See the [Distributed Tracing QE skill](https://github.com/openshift/distributed- | Variable | Default | Description | |---|---|---| -| `AGENT_SKILL_URL` | — | Raw URL to the team's `SKILL.md`. Must start with `https://raw.githubusercontent.com/openshift/`. Step is skipped if unset or URL fails the allowlist check. | +| `AGENT_SKILL` | — | Name of the skill to load from the `skills/` directory. Step is skipped if unset or the file does not exist. | | `CLAUDE_MODEL` | `claude-opus-4-6` | Claude model used for analysis. | | `CLAUDE_CODE_USE_VERTEX` | `1` | Enable Google Vertex AI backend for Claude Code. | | `CLOUD_ML_REGION` | `global` | Google Cloud region for Vertex AI. | @@ -155,7 +243,7 @@ The Distributed Tracing QE team (Tempo Operator, OpenTelemetry Operator, Tracing **CI config snippet** (`openshift-grafana-tempo-operator-main__upstream-ocp-4.22-amd64.yaml`): ```yaml env: - AGENT_SKILL_URL: https://raw.githubusercontent.com/openshift/distributed-tracing-qe/main/plugins/qe-agent/skills/SKILL.md + AGENT_SKILL: RHOSDT post: - ref: openshift-observability-qe-agent - chain: cucushift-installer-rehearse-azure-ipi-deprovision diff --git a/ci-operator/step-registry/openshift-observability/qe-agent/openshift-observability-qe-agent-commands.sh b/ci-operator/step-registry/openshift-observability/qe-agent/openshift-observability-qe-agent-commands.sh index 259b6ba1f59d3..74863a3e5689a 100755 --- a/ci-operator/step-registry/openshift-observability/qe-agent/openshift-observability-qe-agent-commands.sh +++ b/ci-operator/step-registry/openshift-observability/qe-agent/openshift-observability-qe-agent-commands.sh @@ -37,45 +37,112 @@ fi echo "Claude Code CLI: $(claude --version 2>/dev/null || echo 'unknown')" # --------------------------------------------------------------------------- -# 4. Validate and load the qe-agent skill from the URL provided by the job. -# Only URLs under https://raw.githubusercontent.com/openshift/ are trusted. +# 4. Validate and load the qe-agent skill by name. +# Skills are hosted in the openshift/release step registry alongside this +# step, under ci-operator/step-registry/openshift-observability/qe-agent/skills/. +# Each team sets AGENT_SKILL to the name of their skill file (without .md). # --------------------------------------------------------------------------- -if [[ -z "${AGENT_SKILL_URL:-}" ]]; then - echo "ERROR: AGENT_SKILL_URL is not set — skipping qe-agent." +if [[ -z "${AGENT_SKILL:-}" ]]; then + echo "ERROR: AGENT_SKILL is not set — skipping qe-agent." exit 0 fi -readonly SKILL_URL_ALLOWLIST="https://raw.githubusercontent.com/openshift/" -if [[ "${AGENT_SKILL_URL}" != "${SKILL_URL_ALLOWLIST}"* ]]; then - echo "ERROR: AGENT_SKILL_URL '${AGENT_SKILL_URL}' is not from an allowed host." - echo " Skill URLs must start with: ${SKILL_URL_ALLOWLIST}" +# Reject names with path traversal or special characters — only allow +# alphanumeric, hyphens, and underscores. +if [[ ! "${AGENT_SKILL}" =~ ^[A-Za-z0-9_-]+$ ]]; then + echo "ERROR: AGENT_SKILL '${AGENT_SKILL}' contains invalid characters." + echo " Only alphanumeric characters, hyphens, and underscores are allowed." exit 0 fi -echo "Fetching qe-agent skill from ${AGENT_SKILL_URL}..." -SKILL_CONTENT=$(curl -fsSL --connect-timeout 10 --max-time 30 --retry 3 "${AGENT_SKILL_URL}") || true +readonly SKILL_BASE_URL="https://raw.githubusercontent.com/openshift/release/main/ci-operator/step-registry/openshift-observability/qe-agent/skills" +readonly SKILL_URL="${SKILL_BASE_URL}/${AGENT_SKILL}.md" + +echo "Fetching qe-agent skill '${AGENT_SKILL}' from ${SKILL_URL}..." + +# --max-redirs 0: do not follow redirects — the allowlist check is on the +# constructed URL only, and a redirect could bypass it. +SKILL_CONTENT=$(curl -fsS --max-redirs 0 --connect-timeout 10 --max-time 30 --retry 3 "${SKILL_URL}") || true if [[ -z "${SKILL_CONTENT}" ]]; then - echo "ERROR: Failed to fetch skill from ${AGENT_SKILL_URL} — skipping qe-agent." + echo "ERROR: Failed to fetch skill '${AGENT_SKILL}' — check that the file exists at:" + echo " ci-operator/step-registry/openshift-observability/qe-agent/skills/${AGENT_SKILL}.md" + exit 0 +fi + +# Guard against unexpectedly large payloads (100 KB limit). +# Use wc -c for a true byte count; ${#var} counts characters and would allow +# multi-byte UTF-8 content to bypass the limit. +readonly MAX_SKILL_BYTES=102400 +SKILL_BYTE_COUNT=$(printf '%s' "${SKILL_CONTENT}" | wc -c) +if [[ ${SKILL_BYTE_COUNT} -gt ${MAX_SKILL_BYTES} ]]; then + echo "ERROR: Skill content is ${SKILL_BYTE_COUNT} bytes, exceeds the ${MAX_SKILL_BYTES}-byte limit — skipping qe-agent." exit 0 fi -echo "Skill loaded." +echo "Skill '${AGENT_SKILL}' loaded (${SKILL_BYTE_COUNT} bytes)." # --------------------------------------------------------------------------- -# 5. Run Claude non-interactively with the skill as system prompt +# 5. Run Claude non-interactively with the skill as system prompt. +# +# The full stream-json output is captured to a temp file so we can extract +# cost/usage and a command audit log after Claude exits. The temp file is +# NOT in ARTIFACT_DIR — it may contain cluster logs and API responses that +# should not be uploaded to GCS. Only the derived extracts are saved there. # --------------------------------------------------------------------------- echo "Running qe-agent..." +_QE_STREAM=$(mktemp) +trap 'rm -f "${_QE_STREAM}"' EXIT + claude --print \ --dangerously-skip-permissions \ - --allowedTools "Bash,Read,Write,Grep,Glob,WebFetch" \ - --model "${CLAUDE_MODEL}" \ + --allowedTools "Bash,Read,Write,Grep,Glob" \ + --model "${CLAUDE_MODEL:-claude-opus-4-6}" \ + --max-budget-usd 5 \ --verbose \ --output-format stream-json \ --system-prompt "${SKILL_CONTENT}" \ "SHARED_DIR=${SHARED_DIR} ARTIFACT_DIR=${ARTIFACT_DIR}. The test step context is in ${SHARED_DIR}/qe-agent-context.json and JUnit XML files are at ${SHARED_DIR}/qe-agent-junit-*.xml. Execute the skill starting with Step 0: read ${SHARED_DIR}/qe-agent-context.json." \ - 2>&1 | tee "${ARTIFACT_DIR}/qe-agent-output.json" || true + > "${_QE_STREAM}" 2>&1 \ + || true + +# --------------------------------------------------------------------------- +# Cost tracking — the terminal "result" record contains token counts and USD +# cost but no cluster data, so saving it to ARTIFACT_DIR is safe. +# --------------------------------------------------------------------------- +grep '"type":"result"' "${_QE_STREAM}" 2>/dev/null | head -1 \ + > "${ARTIFACT_DIR}/qe-agent-usage.json" || true + +if [[ -s "${ARTIFACT_DIR}/qe-agent-usage.json" ]]; then + _COST=$(jq -r '.total_cost_usd // 0' "${ARTIFACT_DIR}/qe-agent-usage.json" 2>/dev/null || echo 0) + _TURNS=$(jq -r '.num_turns // 0' "${ARTIFACT_DIR}/qe-agent-usage.json" 2>/dev/null || echo 0) + _DUR_S=$(( $(jq -r '.duration_ms // 0' "${ARTIFACT_DIR}/qe-agent-usage.json" 2>/dev/null || echo 0) / 1000 )) + _IN=$(jq -r '.usage.input_tokens // 0' "${ARTIFACT_DIR}/qe-agent-usage.json" 2>/dev/null || echo 0) + _OUT=$(jq -r '.usage.output_tokens // 0' "${ARTIFACT_DIR}/qe-agent-usage.json" 2>/dev/null || echo 0) + echo "Cost: \$${_COST} | Turns: ${_TURNS} | Duration: ${_DUR_S}s | Tokens in: ${_IN} out: ${_OUT}" +fi + +# --------------------------------------------------------------------------- +# Command audit log — every Bash tool call Claude made, command strings only. +# Cluster output (pod logs, events, API responses) is not captured here — only +# the command text, which contains no sensitive data and enables post-incident +# review of what Claude actually executed on the cluster. +# --------------------------------------------------------------------------- +if command -v jq &>/dev/null; then + jq -r ' + select(.type == "assistant") + | .message.content[]? + | select(.type == "tool_use" and .name == "Bash") + | "---\n" + (.input.command // "") + ' "${_QE_STREAM}" 2>/dev/null \ + > "${ARTIFACT_DIR}/qe-agent-commands.log" || true + + if [[ -s "${ARTIFACT_DIR}/qe-agent-commands.log" ]]; then + _CMD_COUNT=$(grep -c '^---$' "${ARTIFACT_DIR}/qe-agent-commands.log" 2>/dev/null || echo 0) + echo "Audit log: ${_CMD_COUNT} Bash commands → ${ARTIFACT_DIR}/qe-agent-commands.log" + fi +fi echo "=== QE Agent Complete ===" diff --git a/ci-operator/step-registry/openshift-observability/qe-agent/openshift-observability-qe-agent-ref.yaml b/ci-operator/step-registry/openshift-observability/qe-agent/openshift-observability-qe-agent-ref.yaml index 470d4a4d50735..15536a6cac8b2 100644 --- a/ci-operator/step-registry/openshift-observability/qe-agent/openshift-observability-qe-agent-ref.yaml +++ b/ci-operator/step-registry/openshift-observability/qe-agent/openshift-observability-qe-agent-ref.yaml @@ -6,11 +6,13 @@ ref: timeout: 1h30m0s grace_period: 2m0s env: - - name: AGENT_SKILL_URL + - name: AGENT_SKILL documentation: |- - Raw URL to the SKILL.md file that the qe-agent will use as its system prompt. - Each team sets this to their own skill URL, e.g.: - https://raw.githubusercontent.com/openshift//main/plugins/qe-agent/skills/SKILL.md + Name of the skill file to load from the step's skills/ directory. + Must match a file at ci-operator/step-registry/openshift-observability/qe-agent/skills/.md + in the openshift/release repository. Only alphanumeric characters, hyphens, and underscores + are allowed. The step is skipped if this variable is unset or the skill file does not exist. + Example: RHOSDT - name: CLAUDE_CODE_USE_VERTEX default: "1" documentation: |- diff --git a/ci-operator/step-registry/openshift-observability/qe-agent/skills/OWNERS b/ci-operator/step-registry/openshift-observability/qe-agent/skills/OWNERS new file mode 100644 index 0000000000000..273601469cb07 --- /dev/null +++ b/ci-operator/step-registry/openshift-observability/qe-agent/skills/OWNERS @@ -0,0 +1,5 @@ +approvers: +- IshwarKanse +options: {} +reviewers: +- IshwarKanse diff --git a/ci-operator/step-registry/openshift-observability/qe-agent/skills/RHOSDT.md b/ci-operator/step-registry/openshift-observability/qe-agent/skills/RHOSDT.md new file mode 100644 index 0000000000000..09b9f34c10718 --- /dev/null +++ b/ci-operator/step-registry/openshift-observability/qe-agent/skills/RHOSDT.md @@ -0,0 +1,849 @@ +--- +name: qe-agent +description: Use this skill to analyze failing CI tests for Red Hat OpenShift Distributed Tracing (OpenTelemetry Operator, Tempo Operator, Tracing UI console plugin), rerun the specific failing tests, diagnose whether the failure is a product bug or a test that needs fixing, apply fixes to test source files when needed, and export results to the artifact directory. Trigger whenever $SHARED_DIR/qe-agent-context.json is present with has_test_failures=true or when an engineer asks to debug, rerun, or fix failing distributed tracing QE tests. +--- + +# RHOSDT QE Agent — Test Failure Triage and Fix + +This skill drives an agentic loop that takes failing CI test results, reruns the failing tests, determines root cause (product bug vs broken test), and either fixes the test or writes a structured bug report. + +## Test Infrastructure Overview + +Three test suites are supported. The JUnit report name prefix tells you which suite failed: + +| JUnit prefix | Suite | Framework | Repo | +|---|---|---|---| +| `junit_otel_*` | OpenTelemetry Operator | chainsaw | `https://github.com/openshift/opentelemetry-operator` | +| `junit_tempo_*` | Tempo Operator | chainsaw | `https://github.com/grafana/tempo-operator` | +| `junit_distributed-tracing-console-plugin*` | Tracing UI (Cypress) | Cypress/npm | `https://github.com/openshift/distributed-tracing-console-plugin` | +| `junit_distributed_tracing_disconnected` | Disconnected (distributed-tracing-qe) | chainsaw | `https://github.com/openshift/distributed-tracing-qe` | + +--- + +## Step 0 — Read Setup Context and Fetch the Step Script + +Read `${SHARED_DIR}/qe-agent-context.json`. The test step writes it at exit time: + +```json +{ + "step_script_ref": "distributed-tracing/tests/opentelemetry/downstream/distributed-tracing-tests-opentelemetry-downstream-commands.sh", + "has_test_failures": true, + "env": { + "MULTISTAGE_PARAM_OVERRIDE_OTEL_TESTS_BRANCH": "rhosdt-3.9" + } +} +``` + +- `step_script_ref` — path relative to `ci-operator/step-registry/` in the openshift/release repo +- `env` — runtime env var values that were injected at job time and are needed to reproduce setup (e.g. branch names, image refs); most steps have an empty `env` + +Construct the raw GitHub URL and fetch the script: + +```text +https://raw.githubusercontent.com/openshift/release/main/ci-operator/step-registry/ +``` + +Read the script carefully. It is divided into two logical sections: +1. **Setup** — everything before the `chainsaw test` or `npx cypress run` commands: cloning repos, `oc apply`, `kubectl create`, CSV patches, `make build`, env variable setup +2. **Test execution** — the `chainsaw test` / `npx cypress run` invocations themselves + +## Step 0a — Verify Cluster Stability + +Before running any prerequisites setup or test reruns, confirm the cluster is stable. The original CI test step may have applied resources that triggered MachineConfig updates — running tests while nodes are updating causes spurious failures. + +```bash +oc get machineconfigpools.machineconfiguration.openshift.io +``` + +For each MachineConfigPool, all of the following must be true before proceeding: +- `UPDATED` = `True` +- `UPDATING` = `False` +- `DEGRADED` = `False` +- `READYMACHINECOUNT` = `MACHINECOUNT` (all machines ready) + +**If any pool is not ready**, wait and recheck every 60 seconds: + +```bash +# Wait until all MCPs are updated, not updating, and not degraded (20-minute timeout) +deadline=$((SECONDS + 1200)) +while oc get machineconfigpools.machineconfiguration.openshift.io \ + -o jsonpath='{range .items[*]}{.status.conditions[?(@.type=="Updated")].status}{" "}{.status.conditions[?(@.type=="Updating")].status}{" "}{.status.conditions[?(@.type=="Degraded")].status}{"\n"}{end}' \ + | grep -qvE '^True False False$'; do + echo "MCPs not ready yet, waiting 60s..." + if (( SECONDS >= deadline )); then + echo "ERROR: MCPs still not ready after 20 minutes — cluster is unhealthy." + oc get machineconfigpools.machineconfiguration.openshift.io + exit 1 + fi + sleep 60 + oc get machineconfigpools.machineconfiguration.openshift.io +done +echo "All MCPs ready — proceeding." +``` + +If the timeout fires, the loop prints the current MCP status and exits — the cluster is unhealthy and test results would be unreliable. + +## Step 0b — Re-establish the Test Environment + +Export any env vars from the `env` field, then **run the setup section of the fetched script** — the commands up to (but not including) the first `chainsaw test` or `npx cypress run` invocation. + +Key adaptations when running setup commands from the script: + +- **Image-mount `cp -R` → `git clone`**: Several step scripts copy test repos from image mounts that don't exist in the qe-agent pod. Replace each with a `git clone`: + - `cp -R /tmp/opentelemetry-operator /tmp/opentelemetry-tests` → `git clone https://github.com/openshift/opentelemetry-operator.git /tmp/opentelemetry-tests` + - `cp -R /tmp/distributed-tracing-qe /tmp/distributed-tracing-tests` → `git clone https://github.com/openshift/distributed-tracing-qe.git /tmp/distributed-tracing-tests` + - For any other `cp -R /tmp/` pattern, check `oc get csv -o yaml | grep github.com` or the CSV annotations to identify the source repo and clone from there. + +- **`kubectl create -f ` for CRDs** — change to `kubectl apply -f ` since `create` fails if the CRD already exists from the original test run; `apply` is idempotent. + +- **CSV patches (`oc patch csv ...`)** — the operator is already installed and already patched from the original test step. Skip these unless the test you are rerunning specifically requires a freshly patched CSV. Check `oc get csv -n ` to verify env vars are already set. + +- **`unset NAMESPACE`** — always run this before chainsaw to avoid conflicts. + +- **`SKIP_TESTS` processing** — the setup sections of several scripts contain a block that reads `$SKIP_TESTS` and removes test directories. Skip this block entirely: `$SKIP_TESTS` is not set in the qe-agent pod, and for reruns you want all test directories present so you can target the specific failing one. + +- **GOPATH / `make build` (Tempo stage and downstream only)** — the Tempo stage and downstream scripts export `GOPATH=/tmp/go`, `GOBIN=/tmp/go/bin`, `GOCACHE=/tmp/.cache/go-build` and run `make build` to compile test helpers. Run these as part of setup. Because each bash invocation starts a fresh shell, you must re-export `GOPATH`, `GOBIN`, `GOCACHE`, and `PATH` (with `GOBIN` prepended) at the start of any bash call in Steps 3–6 that runs chainsaw for Tempo stage or downstream tests — without `PATH` updated, helper binaries compiled into `/tmp/go/bin` cannot be found by name. + +- **IDP / htpasswd setup (Tracing UI only)** — the Tracing UI setup creates an htpasswd secret and patches the cluster oauth. Check if the secret already exists (`oc get secret htpass-secret -n openshift-config`) before running `oc create secret` — skip creation if it does. Similarly, only patch oauth if the htpasswd IDP is not already configured. + +After setup, `cd` into the repo directory and proceed with Steps 1–6. + +If `qe-agent-context.json` does not exist, infer the suite from the JUnit file name prefix (`junit_otel_*` → OpenTelemetry, `junit_tempo_*` → Tempo, `junit_distributed-tracing-console-plugin*` → Tracing UI, `junit_distributed_tracing_disconnected` → Disconnected) and skip the rerun — proceed directly to diagnosis from the JUnit content and cluster state. + +## Step 1 — Parse JUnit XMLs and Identify Failures + +Read all JUnit XML files from `${SHARED_DIR}/qe-agent-junit-*.xml` (flat files copied by the test step trap function). + +For each XML file, extract: +- **Suite name** (`name` attribute on ``) +- **Failed test cases**: `` elements that contain a `` or `` child +- **Failure message**: the `message` attribute and text body of ``/`` +- **Stack trace / details**: the full text content of the failure element + +Group failures by suite so you process each operator's failures together. + +If no `${SHARED_DIR}/qe-agent-junit-*.xml` files are found, exit with a clear message — the test steps did not run or produced no results. + +### High-failure triage: more than 5 failures total + +When the total number of failing test cases across all suites is more than 5, it is very likely that all failures share a single root cause (operator crash, missing CRD, network partition, install failure) rather than being independent bugs. Debugging all of them individually wastes time and produces redundant output. + +**What to do:** + +1. **Look for a common pattern** across the failure messages. Common indicators: + - All messages contain the same error string (e.g., `connection refused`, `resource not ready`, `no such host`, `image pull failed`, `CRD not found`) + - All tests fail at the same chainsaw step name (e.g., `step-01-apply`, `assert`) + - All failures reference the same namespace, resource kind, or operator condition + - Failure times are clustered tightly (within seconds of each other) — indicating the cluster state changed once and all tests hit it + +2. **If a clear pattern exists**: pick the **simplest failing test** (fewest steps in `chainsaw-test.yaml`, or shortest failure message) as the representative case. Record the pattern and the chosen representative in the analysis summary. Proceed with Steps 2–5 for that one test only, skipping the rest. + +3. **If no clear pattern**: the failures are likely independent. Fall back to processing each failure individually (standard flow) but cap at 3 tests to stay within time budget — note in the summary that only the first 3 were investigated. + +Write the pattern conclusion near the top of `${ARTIFACT_DIR}/qe-agent-analysis.md` so it is visible immediately. + +--- + +## Step 2 — Locate Test Source Files + +For **chainsaw** suites (OpenTelemetry, Tempo): + +The JUnit test case name usually matches the folder name under the test directory. For example, a failing test named `e2e/targetallocator` corresponds to `tests/e2e/targetallocator/`. Inside that folder look for: +- `chainsaw-test.yaml` — the test definition (steps, assertions) +- `*.yaml` resource manifests applied during the test +- `assert.yaml` / `error.yaml` — explicit assertion files + +To find the right folder when the name mapping is unclear, use `find /tests -type d -path "*/"`. +The `-path` flag matches the full directory path, so nested test folders like `e2e/targetallocator` are found correctly; `-name` only matches the final path component and will miss them. + +Once located, record this as `TEST_DIR` (e.g. `tests/e2e/targetallocator`). The rerun commands in Step 3 reference `${TEST_DIR}` directly. + +For **Cypress** suites (Tracing UI): + +The failing test name maps to a `describe` + `it` block inside `.cy.js` or `.cy.ts` files under `tests/cypress/e2e/`. Use `grep -r ""` to locate the spec file. + +The repo location and how it was set up is determined by the fetched step script (Step 0b). By the time you reach Step 2, the setup commands from that script have already been run: +- **Upstream tests**: the step script does `cp -R /tmp/ /tmp/` — since the source is an image mount that does not exist in the qe-agent pod, Step 0b substitutes this with a `git clone`. The repo is at the destination path shown in the script (e.g., `/tmp/opentelemetry-tests`, `/tmp/tempo-tests`). +- **Downstream / stage tests**: the step script does `git clone /tmp/` directly. Step 0b runs this clone. The repo is at the path shown in the script. + +Use the destination path from the step script as your repo root — do not guess or check `/tmp/` broadly. + +--- + +## Step 3 — Rerun the Failing Tests + +Rerun only the specific failing tests, not the entire suite, to save time and keep the rerun focused. + +### Cleaning up test resources before each rerun + +Chainsaw reruns use `--skip-delete` so resources remain on the cluster after the test finishes — this lets you inspect them and understand why a test failed. However, because resources persist, **you must clean up before running the same test again**, otherwise the next run will collide with leftover state. + +`kubectl delete -f /` is **not sufficient** — chainsaw tests create resources in multiple ways beyond static YAML files: +- Script steps that run `kubectl apply` / `oc apply` dynamically +- Resources created by the operator itself in response to CRs (e.g., a `TempoStack` CR triggers the operator to create Deployments, Services, ConfigMaps) +- Cluster-scoped resources (ClusterRoles, ClusterRoleBindings, CRDs) created by test setup scripts +- Chainsaw's own test namespaces — chainsaw automatically creates a namespace per test with a `chainsaw-` prefix (e.g., `chainsaw-targetallocator`, `chainsaw-tls-profile`) + +**Reliable cleanup approach:** + +```bash +# 1. Find and delete the chainsaw test namespace(s) for this test +# Chainsaw prefixes namespaces with "chainsaw-" followed by the test name +kubectl get namespace | grep "chainsaw-" +kubectl delete namespace chainsaw- --ignore-not-found=true +kubectl wait --for=delete namespace/chainsaw- --timeout=5m 2>/dev/null || true +``` + +Deleting the namespace cascades and removes all namespaced resources the test created — CRs, operator-managed Deployments, Services, ConfigMaps — regardless of how they were created (YAML, script, or operator reconciliation). + +```bash +# 2. Read chainsaw-test.yaml to identify cluster-scoped resources created by the test +# (ClusterRoles, ClusterRoleBindings, CRDs, etc.) and delete them explicitly. +# +# IMPORTANT: use a test-specific label selector to avoid deleting cluster-scoped +# resources that belong to other concurrent tests or parallel job runs. +# First inspect what labels chainsaw has set on the resources: +kubectl get clusterrole,clusterrolebinding -l app.kubernetes.io/managed-by=chainsaw --show-labels 2>/dev/null | head -20 +# Chainsaw sets a per-namespace label — use it to scope the delete to this test only: +kubectl delete clusterrole,clusterrolebinding \ + -l app.kubernetes.io/managed-by=chainsaw \ + -l chainsaw.kyverno.io/test-namespace=chainsaw- \ + --ignore-not-found=true +# If that label is absent in the output above, fall back to the test-name label instead: +# kubectl delete clusterrole,clusterrolebinding \ +# -l app.kubernetes.io/managed-by=chainsaw \ +# -l chainsaw.kyverno.io/test-name= \ +# --ignore-not-found=true + +# 3. If the test's script steps created additional resources (visible in chainsaw-test.yaml +# script blocks), identify and delete those resources manually +``` + +```bash +# 4. Verify the namespace is gone before rerunning +kubectl get namespace | grep "chainsaw-" && echo "WARNING: namespace still exists" || echo "Clean" +``` + +Read `chainsaw-test.yaml` for the failing test before cleanup — it tells you what namespaces, CRs, and cluster-scoped resources the test creates, which guides what to delete. + +### OpenTelemetry Operator +```bash +# Use TEST_DIR resolved in Step 2 (e.g. tests/e2e/targetallocator) +# Declare TEST_DIR explicitly — each bash invocation starts a fresh shell. +TEST_DIR="" +# Read the fetched step script (from Step 0) to check whether --selector is used in the chainsaw invocation. +# Example: grep -o '\-\-selector [^ ]*' /tmp/fetched-step-script.sh | awk '{print "--selector", $2}' +OTEL_SELECTOR="" # set to "--selector " if the original script uses one, otherwise leave empty +CHAINSAW_CMD="chainsaw test --skip-delete --quiet --report-name junit_rerun_otel --report-path ${ARTIFACT_DIR} --report-format XML" +CHAINSAW_CMD+=" --test-dir ${TEST_DIR}" +[[ -n "${OTEL_SELECTOR}" ]] && CHAINSAW_CMD+=" ${OTEL_SELECTOR}" +eval "$CHAINSAW_CMD" +``` + +### Tempo Operator +```bash +# Use TEST_DIR resolved in Step 2 (e.g. tests/e2e-openshift/tls-profile) +# Declare TEST_DIR explicitly — each bash invocation starts a fresh shell. +TEST_DIR="" +# Re-export GOPATH so test helpers compiled by make build (Step 0b) are on PATH. +export GOPATH=/tmp/go GOBIN=/tmp/go/bin GOCACHE=/tmp/.cache/go-build +export PATH="/tmp/go/bin:${PATH}" +# Tempo always uses --config .chainsaw-openshift.yaml (visible in the fetched step script) +chainsaw test \ + --skip-delete \ + --config .chainsaw-openshift.yaml \ + --quiet \ + --report-name "junit_rerun_tempo" \ + --report-path "${ARTIFACT_DIR}" \ + --report-format XML \ + --test-dir "${TEST_DIR}" +``` + +### Disconnected (distributed-tracing-qe) +```bash +# Use TEST_DIR resolved in Step 2 (e.g. tests/e2e/disconnected-smoke) +# Declare TEST_DIR explicitly — each bash invocation starts a fresh shell. +TEST_DIR="" +CHAINSAW_CMD="chainsaw test --skip-delete --quiet --report-name junit_rerun_disconnected --report-path ${ARTIFACT_DIR} --report-format XML" +CHAINSAW_CMD+=" --test-dir ${TEST_DIR}" +eval "$CHAINSAW_CMD" +``` + +### Tracing UI (Cypress) +```bash +export NO_COLOR=1 +export CYPRESS_CACHE_FOLDER=/tmp/Cypress +npx cypress run \ + --browser chrome \ + --headless \ + --spec "tests/cypress/e2e/" \ + --reporter junit \ + --reporter-options "mochaFile=${ARTIFACT_DIR}/junit_rerun_cypress_run1.xml" +``` + +After the rerun, read the fresh JUnit XML (saved to `$ARTIFACT_DIR`) to check whether the test is: +- **Consistently failing** — same failure, same message → proceed to Step 4 (diagnose) +- **Passed on first rerun** — possible flakiness → do not stop here; run the test 3 more times (4 total reruns) to confirm and locate where the flakiness occurs (see below) +- **Fixed by environment reset** — only relevant if the test setup was stale + +### Flakiness confirmation loop + +If the test passes on the first rerun, run it 3 more times sequentially. Clean up test resources before each run (see above). Use a unique `--report-name` per run so the XMLs don't overwrite each other: + +**OpenTelemetry Operator:** +```bash +# Each bash invocation starts a fresh shell — re-declare variables from the first rerun. +TEST_DIR="" +OTEL_SELECTOR="" # empty string if no --selector was used +for i in 2 3 4; do + kubectl delete namespace chainsaw- --ignore-not-found=true + kubectl wait --for=delete namespace/chainsaw- --timeout=5m 2>/dev/null || true + kubectl delete clusterrole,clusterrolebinding \ + -l app.kubernetes.io/managed-by=chainsaw \ + -l chainsaw.kyverno.io/test-namespace=chainsaw- \ + --ignore-not-found=true + + CHAINSAW_CMD="chainsaw test --skip-delete --quiet --report-name junit_rerun_otel_run${i} --report-path ${ARTIFACT_DIR} --report-format XML" + CHAINSAW_CMD+=" --test-dir ${TEST_DIR}" + [[ -n "${OTEL_SELECTOR}" ]] && CHAINSAW_CMD+=" ${OTEL_SELECTOR}" + eval "$CHAINSAW_CMD" +done +``` + +**Tempo Operator** (always include `--config .chainsaw-openshift.yaml`; re-export GOPATH for stage or downstream tests): +```bash +# Each bash invocation starts a fresh shell — re-declare TEST_DIR from Step 2. +TEST_DIR="" +for i in 2 3 4; do + kubectl delete namespace chainsaw- --ignore-not-found=true + kubectl wait --for=delete namespace/chainsaw- --timeout=5m 2>/dev/null || true + kubectl delete clusterrole,clusterrolebinding \ + -l app.kubernetes.io/managed-by=chainsaw \ + -l chainsaw.kyverno.io/test-namespace=chainsaw- \ + --ignore-not-found=true + + # Re-export GOPATH for Tempo stage/downstream so test helpers on /tmp/go/bin are accessible + export GOPATH=/tmp/go GOBIN=/tmp/go/bin GOCACHE=/tmp/.cache/go-build + export PATH="/tmp/go/bin:${PATH}" + + chainsaw test \ + --skip-delete \ + --config .chainsaw-openshift.yaml \ + --quiet \ + --report-name "junit_rerun_tempo_run${i}" \ + --report-path "${ARTIFACT_DIR}" \ + --report-format XML \ + --test-dir "${TEST_DIR}" +done +``` + +For Cypress: +```bash +for i in 2 3 4; do + npx cypress run \ + --browser chrome \ + --headless \ + --spec "tests/cypress/e2e/" \ + --reporter junit \ + --reporter-options "mochaFile=${ARTIFACT_DIR}/junit_rerun_cypress_run${i}.xml" +done +``` + +**Disconnected (distributed-tracing-qe):** +```bash +# Each bash invocation starts a fresh shell — re-declare TEST_DIR from Step 2. +TEST_DIR="" +for i in 2 3 4; do + kubectl delete namespace chainsaw- --ignore-not-found=true + kubectl wait --for=delete namespace/chainsaw- --timeout=5m 2>/dev/null || true + kubectl delete clusterrole,clusterrolebinding \ + -l app.kubernetes.io/managed-by=chainsaw \ + -l chainsaw.kyverno.io/test-namespace=chainsaw- \ + --ignore-not-found=true + + CHAINSAW_CMD="chainsaw test --skip-delete --quiet --report-name junit_rerun_disconnected_run${i} --report-path ${ARTIFACT_DIR} --report-format XML" + CHAINSAW_CMD+=" --test-dir ${TEST_DIR}" + eval "$CHAINSAW_CMD" +done +``` + +After all 4 runs, count how many passed vs failed. Record the pass/fail pattern (e.g., `PFPP`, `PPFP`). Then inspect the test source: +- Look for missing `wait` blocks between an action and an assertion +- Look for very short `timeout` values in chainsaw steps (e.g., `timeout: 30s` where the operator may take longer) +- Look for assertions that depend on ordering of concurrent resources +- For Cypress: look for missing `cy.wait()` or `cy.intercept()` before asserting UI state + +If the failure is reproducible even 1 out of 4 runs, classify as `FLAKY` and proceed to Step 5c to fix it. + +--- + +## Step 4 — Diagnose: Product Bug vs Test Issue + +Read the failure message, rerun output, and test source files together. Then run the full operator diagnostics below before making any classification decision — the logs and resource status are the primary evidence. + +### Operator Diagnostics + +Run all of the following. Capture output that is relevant to the failure (errors, warnings, crash reasons, unexpected conditions) and include it in the bug report or analysis summary. + +#### OpenTelemetry Operator + +```bash +# Operator pod status and logs +oc get pods -n opentelemetry-operator-system +oc logs -n opentelemetry-operator-system deploy/opentelemetry-operator-controller-manager --tail=150 +oc logs -n opentelemetry-operator-system deploy/opentelemetry-operator-controller-manager --previous --tail=50 2>/dev/null || true + +# OpenTelemetryCollector instances across all namespaces +oc get opentelemetrycollectors --all-namespaces -o wide +oc get opentelemetrycollectors --all-namespaces -o jsonpath='{range .items[*]}{.metadata.namespace}/{.metadata.name}: {.status.conditions[*].type}={.status.conditions[*].status} {.status.conditions[*].message}{"\n"}{end}' + +# Instrumentation, OpAMPBridge, TargetAllocator CRs +oc get instrumentations --all-namespaces -o wide 2>/dev/null || true +oc get opampbridges --all-namespaces -o wide 2>/dev/null || true + +# Collector and sidecar pods in the test namespace +oc get pods -n -o wide +oc describe pods -n | grep -A10 -E 'Events:|Reason:|State:|Exit Code:' + +# Events in operator namespace and test namespace +oc get events -n opentelemetry-operator-system --sort-by='.lastTimestamp' | tail -20 +oc get events -n --sort-by='.lastTimestamp' | tail -30 + +# CSV and subscription status +oc get csv -n opentelemetry-operator-system -o jsonpath='{range .items[*]}{.metadata.name}: {.status.phase} — {.status.message}{"\n"}{end}' +oc get subscription -n opentelemetry-operator-system -o jsonpath='{range .items[*]}{.metadata.name}: {.status.currentCSV} state={.status.state}{"\n"}{end}' 2>/dev/null || true +``` + +#### Tempo Operator + +```bash +# Operator pod status and logs +oc get pods -n openshift-tempo-operator +oc logs -n openshift-tempo-operator deploy/tempo-operator-controller --tail=150 +oc logs -n openshift-tempo-operator deploy/tempo-operator-controller --previous --tail=50 2>/dev/null || true + +# TempoStack and TempoMonolithic instances across all namespaces +oc get tempostacks --all-namespaces -o wide 2>/dev/null || true +oc get tempostacks --all-namespaces -o jsonpath='{range .items[*]}{.metadata.namespace}/{.metadata.name}: {.status.conditions[*].type}={.status.conditions[*].status} {.status.conditions[*].message}{"\n"}{end}' 2>/dev/null || true +oc get tempomonolithics --all-namespaces -o wide 2>/dev/null || true + +# Operand pods in the test namespace (query gateway, distributor, ingester, compactor, querier) +oc get pods -n -o wide +oc describe pods -n | grep -A10 -E 'Events:|Reason:|State:|Exit Code:|OOMKilled' + +# Events in operator namespace and test namespace +oc get events -n openshift-tempo-operator --sort-by='.lastTimestamp' | tail -20 +oc get events -n --sort-by='.lastTimestamp' | tail -30 + +# Storage secret and object store config (common Tempo failure cause) +oc get secret -n | grep -E 'minio|s3|gcs|azure|storage' + +# CSV and subscription status +oc get csv -n openshift-tempo-operator -o jsonpath='{range .items[*]}{.metadata.name}: {.status.phase} — {.status.message}{"\n"}{end}' +oc get subscription -n openshift-tempo-operator -o jsonpath='{range .items[*]}{.metadata.name}: {.status.currentCSV} state={.status.state}{"\n"}{end}' 2>/dev/null || true +``` + +#### Cluster Observability Operator (COO) + +```bash +# Operator pod status and logs (namespace depends on install mode) +COO_NS="$(oc get pods --all-namespaces -l app.kubernetes.io/name=observability-operator -o jsonpath='{.items[0].metadata.namespace}' 2>/dev/null)" +if [ -z "${COO_NS}" ]; then + COO_NS="openshift-cluster-observability-operator" +fi +oc get pods -n "${COO_NS}" +oc logs -n "${COO_NS}" deploy/observability-operator --tail=150 2>/dev/null || \ + oc logs -n "${COO_NS}" $(oc get pods -n "${COO_NS}" -l app.kubernetes.io/name=observability-operator -o name | head -1) --tail=150 2>/dev/null || true +oc logs -n "${COO_NS}" deploy/observability-operator --previous --tail=50 2>/dev/null || true + +# UIPlugin CRs (controls Tracing UI console plugin registration) +oc get uiplugins --all-namespaces -o wide 2>/dev/null || true +oc get uiplugins --all-namespaces -o jsonpath='{range .items[*]}{.metadata.namespace}/{.metadata.name}: {.status.conditions[*].type}={.status.conditions[*].status} {.status.conditions[*].message}{"\n"}{end}' 2>/dev/null || true + +# MonitoringStack CRs +oc get monitoringstacks --all-namespaces -o wide 2>/dev/null || true + +# Console plugin registration status (Tracing UI) +oc get consoleplugin distributed-tracing-plugin -o jsonpath='{.status}{"\n"}' 2>/dev/null || true +oc get consoles.operator.openshift.io cluster -o jsonpath='{.spec.plugins}{"\n"}' 2>/dev/null || true + +# Events in COO namespace +oc get events -n "${COO_NS}" --sort-by='.lastTimestamp' | tail -20 + +# CSV and subscription status +oc get csv -n "${COO_NS}" -o jsonpath='{range .items[*]}{.metadata.name}: {.status.phase} — {.status.message}{"\n"}{end}' +``` + +#### Disconnected (distributed-tracing-qe) + +The disconnected suite tests OTel and Tempo operators in a cluster with no direct internet access — images must be mirrored and catalog sources must reference an internal registry. + +```bash +# Check both operator namespaces +oc get pods -n opentelemetry-operator-system +oc logs -n opentelemetry-operator-system deploy/opentelemetry-operator-controller-manager --tail=100 +oc get pods -n openshift-tempo-operator +oc logs -n openshift-tempo-operator deploy/tempo-operator-controller --tail=100 + +# Disconnected-specific: look for image pull failures (most common failure cause) +oc get pods --all-namespaces | grep -E 'ImagePullBackOff|ErrImagePull' 2>/dev/null || true + +# Mirror configuration +oc get imagecontentsourcepolicies 2>/dev/null || true # OCP 4.12 and earlier +oc get idms 2>/dev/null || true # ImageDigestMirrorSet (OCP 4.13+) +oc get itms 2>/dev/null || true # ImageTagMirrorSet + +# Catalog sources (must be READY in disconnected clusters) +oc get catalogsource -n openshift-marketplace -o wide 2>/dev/null || true + +# Events across all test namespaces (image pull errors, operator errors) +oc get events --all-namespaces --sort-by='.lastTimestamp' | grep -E 'Failed|Error|Warning' | tail -30 +``` + +#### CRD and API availability check + +```bash +# Verify all expected CRDs are present — missing CRDs cause many test failures +oc get crd | grep -E 'opentelemetry|tempo|observability|uiplugin|monitoringstack' + +# Check operator API groups are registered +oc api-resources | grep -E 'opentelemetry|tempo|observability' +``` + +### Product Bug indicators +Classify as `PRODUCT_BUG` when the evidence shows the operator or operand itself misbehaved: +- Operator pod in `CrashLoopBackOff` or `OOMKilled` +- Operand resource (e.g., `TempoStack`, `OpenTelemetryCollector`) stuck in an error state not caused by the test YAML +- API object that the operator should have created is missing +- Image pull failure for an operand image referenced in the CSV +- CRD validation error rejecting a valid CR that worked in a prior release +- Timeout waiting for operator reconciliation when the operator logs show no activity + +### Test Issue indicators +Classify as `TEST_ISSUE` when the test itself is wrong or stale: +- Hardcoded version string or image tag in the test YAML that doesn't match the currently installed operator version +- Wrong namespace name in an assertion (namespace changed between releases) +- Race condition: the test asserts a resource state before the operator has had time to act — look for very short `timeout` values in chainsaw steps or missing `wait` blocks +- Missing prerequisite in the test setup (e.g., a CRD that must be installed before the test runs but isn't part of the test's `setup` steps) +- Assertion checks a field or value that changed in the operator API (e.g., a renamed status condition) +- Cypress test references a UI element selector that changed in the console plugin + +### Cluster Instability indicators +When MCP updating or operator restarts are present and tests pass cleanly on rerun, **do not classify as `CLUSTER_INSTABILITY` yet** — first rule out the operator itself as the source of API server pressure using the steps below. Only after confirming the operator is not looping can you attribute the instability to external infra churn. + +Classify as `CLUSTER_INSTABILITY` when **all four** conditions hold: +- MCPs were updating (`UPDATING=True`, `UPDATED=False`) at the time of the original test run, **or** the operator pod shows `RESTARTS > 0` with liveness/readiness probe failures or leader election loss events correlated with MCP rollout timing +- Tests pass cleanly and quickly on all reruns (rerun duration significantly less than the original failing run) +- No code defect identified in either the operator or the test +- Operator debug logs (collected below) show **no tight reconciliation loops** driving excessive API server calls + +`CLUSTER_INSTABILITY` takes precedence over `FLAKY`: if all 4 reruns pass cleanly AND the MCP or operator restart evidence points to infrastructure churn during the original run (and the operator is not looping), classify as `CLUSTER_INSTABILITY`, not `FLAKY`. Proceed to Step 5d instead of Step 5c. + +#### Ruling out operator-caused API server pressure + +Operator reconciliation loops — where the operator continuously re-queues the same object without back-off — can themselves drive enough API server load to cause liveness probe timeouts and leader election flaps, which looks identical to external infra churn. Check this before concluding the infrastructure is at fault. + +**Step 1 — Enable debug logging on the operator CSV** + +Patch the operator CSV to add `--zap-log-level=debug`. This causes OLM to restart the operator pod with verbose reconciliation logging. + +For the **OpenTelemetry Operator**: +```bash +# Filter by name AND Succeeded phase — during upgrades both old and new CSVs coexist in the namespace +CSV_NAME=$(oc get csv -n opentelemetry-operator-system --no-headers \ + | awk '/opentelemetry-operator/ && /Succeeded/ {print $1}' | head -1) +# Show current args to confirm --zap-log-level is present (it is set to 'info' by default) +oc get csv "${CSV_NAME}" -n opentelemetry-operator-system \ + -o jsonpath='{range .spec.install.spec.deployments[0].spec.template.spec.containers[0].args[*]}{.}{"\n"}{end}' + +# Find the 0-based index of the existing --zap-log-level arg and replace its value with debug +ZAP_IDX=$(oc get csv "${CSV_NAME}" -n opentelemetry-operator-system \ + -o jsonpath='{range .spec.install.spec.deployments[0].spec.template.spec.containers[0].args[*]}{.}{"\n"}{end}' \ + | awk '/--zap-log-level/{print NR-1; exit}') +if [[ -z "${ZAP_IDX}" ]]; then echo "ERROR: --zap-log-level not found in CSV args"; exit 1; fi +oc patch csv "${CSV_NAME}" -n opentelemetry-operator-system --type=json \ + -p="[{\"op\":\"replace\",\"path\":\"/spec/install/spec/deployments/0/spec/template/spec/containers/0/args/${ZAP_IDX}\",\"value\":\"--zap-log-level=debug\"}]" +``` + +For the **Tempo Operator**: +```bash +CSV_NAME=$(oc get csv -n openshift-tempo-operator --no-headers \ + | awk '/tempo-operator/ && /Succeeded/ {print $1}' | head -1) +oc get csv "${CSV_NAME}" -n openshift-tempo-operator \ + -o jsonpath='{range .spec.install.spec.deployments[0].spec.template.spec.containers[0].args[*]}{.}{"\n"}{end}' + +ZAP_IDX=$(oc get csv "${CSV_NAME}" -n openshift-tempo-operator \ + -o jsonpath='{range .spec.install.spec.deployments[0].spec.template.spec.containers[0].args[*]}{.}{"\n"}{end}' \ + | awk '/--zap-log-level/{print NR-1; exit}') +if [[ -z "${ZAP_IDX}" ]]; then echo "ERROR: --zap-log-level not found in CSV args"; exit 1; fi +oc patch csv "${CSV_NAME}" -n openshift-tempo-operator --type=json \ + -p="[{\"op\":\"replace\",\"path\":\"/spec/install/spec/deployments/0/spec/template/spec/containers/0/args/${ZAP_IDX}\",\"value\":\"--zap-log-level=debug\"}]" +``` + +Wait for the operator pod to restart with the new flag: +```bash +# OpenTelemetry +oc rollout status deploy/opentelemetry-operator-controller-manager -n opentelemetry-operator-system --timeout=3m + +# Tempo +oc rollout status deploy/tempo-operator-controller -n openshift-tempo-operator --timeout=3m +``` + +**Step 2 — Collect debug logs and check for reconciliation loops** + +Let the operator run for 2–3 minutes, then collect logs: + +```bash +# OpenTelemetry — capture last 500 lines of debug output +oc logs -n opentelemetry-operator-system deploy/opentelemetry-operator-controller-manager --tail=500 \ + | grep -E '"reconcileID"|"Reconciling"|"requeue"|"error"' \ + | head -100 + +# Tempo +oc logs -n openshift-tempo-operator deploy/tempo-operator-controller --tail=500 \ + | grep -E '"reconcileID"|"Reconciling"|"requeue"|"error"' \ + | head -100 +``` + +**Indicators of a reconciliation loop causing pressure:** +- The same resource name (`"name":""`) appears in `Reconciling` log lines many times within seconds (more than once every 2–3 seconds is a strong signal) +- `"requeue"` entries at very short intervals (sub-second) with no intervening `"Reconciling finished"` or success message +- `controller-runtime` lines reporting queue depth growing (`"queue depth": N` increasing over time) +- Rate-limiting warnings: `"controller-runtime/controller"` messages like `"Reconciler error"` followed immediately by rapid requeue + +**Indicators that the operator is healthy (no loop):** +- `Reconciling` log lines appear infrequently (one reconcile per object per event, with gaps of 10+ seconds between repeated reconciles for the same object) +- No `requeue` entries on short intervals +- Log volume is low and stable + +If a reconciliation loop is found, reclassify as `PRODUCT_BUG` (the operator is mis-behaving under load) and write a bug report in Step 5b. Include the relevant log lines as evidence. + +When genuinely ambiguous, gather more cluster evidence before deciding. Explain your reasoning explicitly in the output. + +--- + +## Step 5a — If TEST_ISSUE: Fix and Export + +Apply the **minimal** change to make the test correct. Avoid refactoring or improving unrelated parts of the test — a focused, small diff is easier to review and merge. + +**For chainsaw tests:** +- Edit `chainsaw-test.yaml`, `assert.yaml`, resource manifests, or other YAML files in the test folder +- Common fixes: update image/version references, fix namespace, add a `wait` step before an assertion, correct a changed field name in assertions + +**For Cypress tests:** +- Edit the `.cy.js` / `.cy.ts` spec file +- Common fixes: update CSS selector, fix a changed route or API endpoint, add a `cy.wait()` for async operations + +After editing, copy only the changed files to `${ARTIFACT_DIR}/test-fixes/` **preserving the directory path relative to the repo root**: + +```bash +# Example: tests/e2e/targetallocator/chainsaw-test.yaml was fixed +dest="${ARTIFACT_DIR}/test-fixes/tests/e2e/targetallocator" +mkdir -p "${dest}" +cp tests/e2e/targetallocator/chainsaw-test.yaml "${dest}/" +``` + +Write a `${ARTIFACT_DIR}/test-fixes/CHANGES.md` using this structure: + +```markdown +# Test Fix Summary + +## Failing test + / + +## Root cause + + +## Fix applied + + +## Files changed +- `tests/e2e//chainsaw-test.yaml` + +## Verification +Rerun result after fix: [PASS / FAIL / not re-verified] +``` + +--- + +## Step 5b — If PRODUCT_BUG: Write Bug Report + +Do not attempt to fix the operator code. Instead, write `${ARTIFACT_DIR}/bug-report.md`: + +````markdown +# Product Bug Report + +## Summary + + +## Affected component +- Operator: +- Namespace: +- Failing test: + +## Reproduction +1. + +## Observed behavior + + +## Expected behavior + + +## Evidence +### Operator logs +```text + +``` + +### Cluster events +```text + +``` + +### JUnit failure message +```text + +``` + +## Suggested severity + +```` + +--- + +## Step 5c — If FLAKY: Fix and Export + +Apply the minimal change that eliminates the race or timing condition. Do not suppress flakiness with blanket retries — find and fix the root cause. + +**For chainsaw tests:** +- If a step asserts state immediately after a resource is applied, add an explicit `wait` step using `chainsaw wait` or a `sleep` in a script step before the assertion +- If a `timeout` is too short, increase it to give the operator time to reconcile (common fix: `30s` → `2m`) +- If two resources are created concurrently and one depends on the other, reorder the steps to create the dependency first + +Example — adding a wait step in `chainsaw-test.yaml`: +```yaml +- name: Wait for collector to be ready before asserting + wait: + apiVersion: opentelemetry.io/v1alpha1 + kind: OpenTelemetryCollector + name: otel-collector + timeout: 2m + for: + condition: + name: Ready + value: "True" +``` + +**For Cypress tests:** +- Add `cy.intercept()` to wait for the relevant API call before asserting +- Use `cy.findByText(...).should('be.visible')` with a custom timeout rather than asserting immediately +- Replace `cy.wait()` (fixed-time sleep) with a condition-based wait when possible + +After editing, copy changed files to `${ARTIFACT_DIR}/test-fixes/` with the same structure as Step 5a. Write `CHANGES.md` using the same template, and include the pass/fail pattern from the 4 rerun runs as evidence. + +--- + +## Step 5d — If CLUSTER_INSTABILITY: Write Incident Note + +Do not attempt to fix the operator or test code. Write `${ARTIFACT_DIR}/cluster-instability-report.md`: + +````markdown +# Cluster Instability Report + +## Summary + + +## Affected Tests +| Suite | Test Case | Original Duration | Rerun Duration | +|---|---|---|---| +| | | | | + +## Root Cause + + +## Evidence + +### MachineConfigPool status at triage time +```text + +``` + +### Operator pod restarts and events +```text + +``` + +## Rerun Results +All reruns passed cleanly — failures are not reproducible outside the original cluster instability window. + +## Recommendation +Rerun the CI job. The failures are caused by cluster infrastructure churn, not by a product or test defect. +```` + +--- + +## Step 6 — Write Analysis Summary + +Write `${ARTIFACT_DIR}/qe-agent-analysis.md` **immediately after each test is diagnosed** (after Step 4/5a/5b/5c/5d is complete for that test), not at the very end. If multiple tests are being investigated sequentially, write an initial draft after the first test is diagnosed and overwrite it after each subsequent test is diagnosed. This ensures the report is present in `${ARTIFACT_DIR}` — and therefore uploaded by the sidecar — even if the session is interrupted before all tests are fully processed. + +Do not wait for background flakiness confirmation runs to finish before writing the first draft. Write a partial entry for in-progress tests (e.g. "Rerun 1: PASS — flakiness confirmation in progress") and overwrite with final results when the runs complete. + +Throughout the run, note any place where a skill step was wrong, incomplete, or had to be adapted — commands that failed, assumptions that did not hold, diagnostics that were decisive but not mentioned in the skill, or steps that wasted time. Record all of these in the **Skill Improvement Recommendations** section of the analysis. This feedback is used to improve the skill so future runs are faster, cheaper, and more accurate. If the skill worked as written with no deviations, write "None." + +````markdown +# QE Agent Analysis + +## Failed Tests +| Suite | Test Case | JUnit File | +|---|---|---| +| | | | + +## Rerun Result + + +## Diagnosis +**** + + + +## Rerun Summary +| Run | Result | +|---|---| +| Original CI run | FAIL | +| Rerun 1 | PASS / FAIL | +| Rerun 2 | PASS / FAIL | +| Rerun 3 | PASS / FAIL | +| Rerun 4 | PASS / FAIL | + +## Outcome +: Test fix applied. Changed files in `${ARTIFACT_DIR}/test-fixes/`. See `CHANGES.md` for details. +: Bug report written to `${ARTIFACT_DIR}/bug-report.md`. +: Flaky test confirmed (pattern: ). Fix applied to `${ARTIFACT_DIR}/test-fixes/`. See `CHANGES.md` for root cause and fix details. +: Incident note written to `${ARTIFACT_DIR}/cluster-instability-report.md`. Recommendation: rerun the CI job. + +## Skill Improvement Recommendations + +: None. +: +- **Step **: . Suggested fix: . + +Examples of what belongs here: +- A command in the skill failed and had to be adapted (wrong flag, missing argument, changed API) +- A diagnostic the skill did not mention turned out to be the decisive evidence +- A step the skill prescribed was unnecessary or wasted significant time +- The cleanup approach did not work and a different method had to be used +- An assumption in the skill (namespace, resource name, container index) did not hold for this operator version +```` + +--- + +## Notes for CI context + +- The cluster is already provisioned and the operator is already installed — do not reinstall the operator +- `$KUBECONFIG` is set and points to the test cluster +- `oc` and `kubectl` are available +- `chainsaw` is available in PATH +- The test repo is set up by Step 0b using commands from the fetched step script — the repo path is the destination shown in the script (e.g., `/tmp/opentelemetry-tests`, `/tmp/tempo-tests`). The qe-agent runs in a fresh pod so `/tmp/` is always empty at start; Step 0b populates it +- All output files must go to `$ARTIFACT_DIR` (uploaded to GCS by the sidecar) or `$SHARED_DIR` (accessible to other steps) +- This step runs `best_effort: true` — always exit 0 even if analysis is incomplete