diff --git a/build/evals.mk b/build/evals.mk index ab6734aa2..d50d3d975 100644 --- a/build/evals.mk +++ b/build/evals.mk @@ -74,7 +74,15 @@ claude-agent-acp: ## Install the claude-agent-acp adapter for the acp-anthropic .PHONY: run-evals run-evals: mcpchecker jq $(if $(filter acp-anthropic,$(AGENT)),claude-agent-acp) ## Run mcpchecker evals (knobs: SUITE, AGENT, MODEL; see evals/README.md) - $(if $(MODEL),ANTHROPIC_MODEL=$(MODEL) )PATH="$(shell pwd)/_output/tools/node_modules/.bin:$(PATH)" $(MCPCHECKER) check $(EVAL_CONFIG) \ + @# Prefer MCP_EVAL_KUBECONFIG when KUBECONFIG is unset so setup/verify kubectl + @# targets the same cluster as make run-server. + @# TODO: mcpchecker kubernetes extension should respect KUBECONFIG directly + @# (upstream issue: mcpchecker doesn't properly propagate KUBECONFIG to extensions). + @if [ -z "$${KUBECONFIG:-}" ] && [ -n "$(MCP_EVAL_KUBECONFIG)" ]; then \ + export KUBECONFIG="$(MCP_EVAL_KUBECONFIG)"; \ + fi; \ + $(if $(MODEL),ANTHROPIC_MODEL=$(MODEL) )PATH="$(shell pwd)/_output/tools/bin:$(shell pwd)/_output/tools/node_modules/.bin:$${PATH}" \ + $(MCPCHECKER) check $(EVAL_CONFIG) \ $(if $(EVAL_LABEL_SELECTOR),--label-selector $(EVAL_LABEL_SELECTOR),) \ $(if $(EVAL_TASK_FILTER),--run "$(EVAL_TASK_FILTER)",) \ $(if $(filter true,$(EVAL_VERBOSE)),--verbose,) \ diff --git a/dev/config/mcp-configs/000-config-defaults.toml b/dev/config/mcp-configs/000-config-defaults.toml index 53284e039..d6dd02afc 100644 --- a/dev/config/mcp-configs/000-config-defaults.toml +++ b/dev/config/mcp-configs/000-config-defaults.toml @@ -1,3 +1,7 @@ # !CAUTION! Destructive mode enabled by default for development. read_only = false disable_destructive = false + +# Evals run the server out-of-cluster; pin the kubeconfig provider so an +# in-cluster environment cannot take over. Override in a later drop-in if needed. +cluster_provider_strategy = "kubeconfig" diff --git a/evals/tasks/core/create-pod-mount-configmaps/verify.sh b/evals/tasks/core/create-pod-mount-configmaps/verify.sh index dfa053b49..f919d0dd7 100755 --- a/evals/tasks/core/create-pod-mount-configmaps/verify.sh +++ b/evals/tasks/core/create-pod-mount-configmaps/verify.sh @@ -46,16 +46,23 @@ if [[ "$POD_IMAGE" != "quay.io/nginx/nginx-unprivileged:alpine" ]]; then exit 1 fi -# Verify the values are accessible in the pod +# Verify the values are accessible in the pod. +# The prompt never specifies a container name, so resolve the actual first +# container name instead of letting kubectl pick a default -- if the pod ends +# up with a kubectl.kubernetes.io/default-container annotation that doesn't +# match the container name the model chose, a bare `kubectl exec pod1` fails +# with "container not found" even though there's exactly one container. +CONTAINER_NAME=$(kubectl get pod pod1 -n $NAMESPACE -o jsonpath='{.spec.containers[0].name}') + echo "Verifying environment variable in pod..." -ENV_TEST=$(kubectl exec pod1 -n $NAMESPACE -- sh -c 'echo $COLOR') +ENV_TEST=$(kubectl exec pod1 -n $NAMESPACE -c "$CONTAINER_NAME" -- sh -c 'echo $COLOR') if [[ "$ENV_TEST" != "blue" ]]; then echo "Environment variable 'COLOR' is not accessible in the pod or has incorrect value: '$ENV_TEST'" exit 1 fi echo "Verifying volume mount in pod..." -VOLUME_TEST=$(kubectl exec pod1 -n $NAMESPACE -- cat /etc/sizes/size) +VOLUME_TEST=$(kubectl exec pod1 -n $NAMESPACE -c "$CONTAINER_NAME" -- cat /etc/sizes/size) if [[ "$VOLUME_TEST" != "medium" ]]; then echo "Volume mount is not accessible in the pod or file has incorrect content: '$VOLUME_TEST'" exit 1 diff --git a/evals/tasks/core/create-pod-resources-limits/verify.sh b/evals/tasks/core/create-pod-resources-limits/verify.sh index b8e5417a9..d80d9fb1d 100755 --- a/evals/tasks/core/create-pod-resources-limits/verify.sh +++ b/evals/tasks/core/create-pod-resources-limits/verify.sh @@ -6,8 +6,10 @@ if ! kubectl get namespace limits-test &>/dev/null; then exit 1 fi -# Wait for pod to be ready -TIMEOUT="120s" +# Wait for pod to be ready. Default matches generic K8s; environments +# with other needs can override via VERIFY_TIMEOUT without +# changing the default for everyone else. +TIMEOUT="${VERIFY_TIMEOUT:-120s}" if ! kubectl wait --for=condition=Ready pod/resource-limits-pod -n limits-test --timeout=$TIMEOUT; then echo "Pod 'resource-limits-pod' is not ready in namespace 'limits-test'" exit 1 diff --git a/evals/tasks/core/fix-service-routing/cleanup.sh b/evals/tasks/core/fix-service-routing/cleanup.sh index e791fc73d..97b6fbee7 100755 --- a/evals/tasks/core/fix-service-routing/cleanup.sh +++ b/evals/tasks/core/fix-service-routing/cleanup.sh @@ -1,2 +1,3 @@ #!/usr/bin/env bash +kubectl delete pod -n web test-connection --ignore-not-found kubectl delete namespace web diff --git a/evals/tasks/core/fix-service-routing/setup.sh b/evals/tasks/core/fix-service-routing/setup.sh index bfc35afa0..c4a370ec7 100755 --- a/evals/tasks/core/fix-service-routing/setup.sh +++ b/evals/tasks/core/fix-service-routing/setup.sh @@ -17,7 +17,8 @@ metadata: spec: ports: - port: 80 - targetPort: 80 + # nginx-unprivileged listens on 8080; only the selector is intentionally wrong + targetPort: 8080 selector: app: web # Mismatched label - deployment has app=nginx EOF diff --git a/evals/tasks/core/fix-service-routing/verify.sh b/evals/tasks/core/fix-service-routing/verify.sh index 65805c596..19266cbd9 100755 --- a/evals/tasks/core/fix-service-routing/verify.sh +++ b/evals/tasks/core/fix-service-routing/verify.sh @@ -1,13 +1,41 @@ #!/usr/bin/env bash -# Check if service has endpoints -endpoints=$(kubectl get endpoints nginx -n web -o jsonpath='{.subsets[0].addresses}') -if [[ ! -z "$endpoints" ]]; then - # Verify service can access the pod - if kubectl run -n web test-connection --image=quay.io/prometheus/busybox --restart=Never --rm -i --wait --timeout=180s \ - -- wget -qO- nginx; then - exit 0 - fi +set -euo pipefail + +# Check if service has endpoints. EndpointSlice/Endpoints propagation is +# asynchronous, so poll for a bounded period instead of failing on the first +# empty read -- otherwise a correct fix can still be reported as failed if it +# hasn't converged yet. +endpoints="" +for i in $(seq 1 15); do + endpoints=$(kubectl get endpoints nginx -n web -o jsonpath='{.subsets[0].addresses}' 2>/dev/null || true) + if [[ -n "$endpoints" ]]; then + break + fi + sleep 2 +done +if [[ -z "$endpoints" ]]; then + echo "Service nginx in namespace web has no endpoints" + exit 1 fi -# If we get here, service connection failed -exit 1 +# Verify service can access the pod with a compatible probe pod. +cat <<'EOF' | kubectl apply -f - +apiVersion: v1 +kind: Pod +metadata: + name: test-connection + namespace: web +spec: + restartPolicy: Never + containers: + - name: test-connection + image: quay.io/curl/curl:8.11.1 + command: ["curl", "-sf", "--max-time", "15", "http://nginx"] +EOF + +if ! kubectl wait -n web --for=jsonpath='{.status.phase}'=Succeeded pod/test-connection --timeout=180s; then + echo "Service connection probe did not succeed" + kubectl get pod -n web test-connection -o yaml || true + exit 1 +fi +exit 0 diff --git a/evals/tasks/core/fix-service-with-no-endpoints/verify.sh b/evals/tasks/core/fix-service-with-no-endpoints/verify.sh index d6d6d3872..1e74112ef 100755 --- a/evals/tasks/core/fix-service-with-no-endpoints/verify.sh +++ b/evals/tasks/core/fix-service-with-no-endpoints/verify.sh @@ -7,9 +7,10 @@ if ! kubectl get deployment web-app-deployment -n webshop-frontend &>/dev/null; exit 1 fi -# Check if pods are being created successfully +# Check if pods are being created successfully. +# Environments can override via VERIFY_TIMEOUT echo "Waiting for pods to become ready..." -TIMEOUT="120s" +TIMEOUT="${VERIFY_TIMEOUT:-120s}" if ! kubectl wait --for=condition=Ready pods -l app=web-app -n webshop-frontend --timeout=$TIMEOUT; then echo "Pods are not reaching Ready state after fixing the node selector" exit 1 diff --git a/evals/tasks/core/list-images-for-pods/list-images-for-pods.yaml b/evals/tasks/core/list-images-for-pods/list-images-for-pods.yaml index f32242a69..aee37684e 100644 --- a/evals/tasks/core/list-images-for-pods/list-images-for-pods.yaml +++ b/evals/tasks/core/list-images-for-pods/list-images-for-pods.yaml @@ -16,4 +16,4 @@ steps: cleanup: file: cleanup.sh prompt: - inline: "What images are all pods running in the cluster?" + inline: "What images are pods running in the cluster? Query the cluster now and report the images you find directly in your response -- do not ask for clarification or preferences about output format." diff --git a/evals/tasks/core/multi-container-pod-communication/multi-container-pod-communication.yaml b/evals/tasks/core/multi-container-pod-communication/multi-container-pod-communication.yaml index 0a9446516..7fd8d8e43 100644 --- a/evals/tasks/core/multi-container-pod-communication/multi-container-pod-communication.yaml +++ b/evals/tasks/core/multi-container-pod-communication/multi-container-pod-communication.yaml @@ -19,7 +19,7 @@ steps: inline: | In the multi-container-logging namespace, run a pod called communication-pod with two containers: 1. A 'web-server' nginx instance (quay.io/nginx/nginx-unprivileged image) that serves traffic - 2. A 'logger' busybox instance (quay.io/prometheus/busybox image) that processes those logs from a shared volume, 'logs-volume' with 'tail -f /var/log/nginx/access.log' + 2. A 'logger' busybox instance (busybox:latest image) that processes those logs from a shared volume, 'logs-volume' with 'tail -f /var/log/nginx/access.log' Both containers should be defined in spec.containers (not as init containers or sidecars). Both containers should mount logs-volume at '/var/log/nginx'. diff --git a/evals/tasks/core/multi-container-pod-communication/verify.sh b/evals/tasks/core/multi-container-pod-communication/verify.sh index 70e3bd445..33671237e 100755 --- a/evals/tasks/core/multi-container-pod-communication/verify.sh +++ b/evals/tasks/core/multi-container-pod-communication/verify.sh @@ -3,7 +3,9 @@ set -euo pipefail NAMESPACE="multi-container-logging" POD_NAME="communication-pod" -TIMEOUT="120s" +# Default matches generic K8s; environments with other needs can +# override via VERIFY_TIMEOUT without changing the default for everyone else. +TIMEOUT="${VERIFY_TIMEOUT:-120s}" # Wait for pod to be running echo "Waiting for pod '$POD_NAME' to be ready..." diff --git a/evals/tasks/core/setup-dev-cluster/verify.sh b/evals/tasks/core/setup-dev-cluster/verify.sh index 0cc98583c..c36973305 100755 --- a/evals/tasks/core/setup-dev-cluster/verify.sh +++ b/evals/tasks/core/setup-dev-cluster/verify.sh @@ -6,6 +6,9 @@ readonly DEVELOPERS=("alice" "bob" "charlie") readonly DEV_NAMESPACES=("dev-alice" "dev-bob" "dev-charlie") readonly ALL_NAMESPACES=("${DEV_NAMESPACES[@]}" "dev-shared" "staging" "prod") readonly TEST_LABEL="app=verification-test" +# Default matches generic K8s; environments with other needs can +# override via VERIFY_TIMEOUT without changing the default for everyone else. +readonly TEST_POD_TIMEOUT="${VERIFY_TIMEOUT:-60s}" # --- Cleanup Function --- cleanup() { @@ -192,7 +195,7 @@ metadata: spec: containers: - name: curl - image: quay.io/curl/curl:latest + image: quay.io/curl/curl:8.11.1 command: ["sleep", "3600"] resources: limits: @@ -223,7 +226,7 @@ EOF echo " - Waiting for test pods to be ready..." for dev in "${DEVELOPERS[@]}"; do - kubectl wait --for=condition=Ready pod/test-pod-${dev} -n "dev-${dev}" --timeout=60s + kubectl wait --for=condition=Ready pod/test-pod-${dev} -n "dev-${dev}" --timeout="${TEST_POD_TIMEOUT}" done # Test that alice cannot reach bob's service diff --git a/evals/tasks/core/statefulset-lifecycle/verify.sh b/evals/tasks/core/statefulset-lifecycle/verify.sh index d6a4b1b8a..45f83d6eb 100755 --- a/evals/tasks/core/statefulset-lifecycle/verify.sh +++ b/evals/tasks/core/statefulset-lifecycle/verify.sh @@ -5,10 +5,28 @@ set -euo pipefail NAMESPACE="statefulset-test" STS_NAME="db" EXPECTED_CONTENT="initial_data" +# Defaults match generic K8s; environments with other needs can +# override via VERIFY_TIMEOUT without changing the default +# for everyone else. +DELETE_TIMEOUT="${VERIFY_TIMEOUT:-120s}" +READY_TIMEOUT="${VERIFY_TIMEOUT:-120s}" echo "Verifying old pods are deleted" -# Wait for scale-down: 1 ready pods and deletion of old pods -kubectl wait pod/db-1 pod/db-2 -n statefulset-test --for=delete --timeout=120s +# Wait for scale-down: deletion of db-1/db-2 (may already be gone). +# Fail closed: --ignore-not-found only suppresses the "not found" case, so any +# other API/auth/transport error still returns non-zero and is treated as a +# verification failure instead of being silently read as "pod is gone". +for pod in db-1 db-2; do + kubectl wait "pod/${pod}" -n "${NAMESPACE}" --for=delete --timeout="${DELETE_TIMEOUT}" 2>/dev/null || true + if ! out=$(kubectl get pod "$pod" -n "${NAMESPACE}" --ignore-not-found -o name 2>&1); then + echo "Unable to verify pod $pod was deleted: $out" + exit 1 + fi + if [[ -n "$out" ]]; then + echo "Pod $pod still exists after scale-down" + exit 1 + fi +done echo "Old pods are deleted" # Verify correct number of replicas @@ -20,7 +38,16 @@ if [[ "${replicas}" -ne 1 ]]; then fi echo "StatefulSet is running with 1 replicas" -# Verify db-0 exists and have the correct data +# Wait for db-0 to become Ready before reading data; uses READY_TIMEOUT +# (derived from VERIFY_TIMEOUT) so environments can adjust as needed. +echo "Waiting for pod db-0 to become Ready" +if ! kubectl wait --for=condition=Ready "pod/db-0" -n "${NAMESPACE}" --timeout="${READY_TIMEOUT}"; then + echo "Pod db-0 not Ready in time" + kubectl get pvc,pod -n "${NAMESPACE}" -o wide || true + exit 1 +fi + +# Verify db-0 has the correct data for pod in db-0; do if ! kubectl get pod "$pod" -n "${NAMESPACE}" &> /dev/null; then echo "Pod $pod not found in namespace $NAMESPACE" @@ -34,4 +61,4 @@ for pod in db-0; do fi done -exit 0 \ No newline at end of file +exit 0