diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml new file mode 100644 index 00000000..f60474dc --- /dev/null +++ b/.ci/release.cloudbuild.yaml @@ -0,0 +1,523 @@ +# Weekly release pipeline. +# +# Gates the image that the Build trigger pushed for this commit, tags it, +# deploys it to GKE with automatic rollback, moves :latest, and deploys it +# to Cloud Run. +# +# Each run prints one RELEASE_RESULT line (lib.sh), and alert_policy.yaml +# emails it. Cloud Build has no on-failure hook, so each failed step prints +# its own line. A step timeout or a cancel prints no line, so +# build_status_policy.yaml emails every run that fails. +# +# Use $$ for shell variables and $$(...). Cloud Build expands a single $. + +steps: + + # Waits for the image that the Build trigger pushes after its tests pass. + - id: resolve-release + name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' + timeout: '2400s' + entrypoint: 'bash' + args: + - '-c' + - | + set -uo pipefail + . /workspace/.ci/release/lib.sh + STAGE=resolve-release + + if [[ -z "$COMMIT_SHA" || -z "$SHORT_SHA" ]]; then + release_fail $$STAGE "COMMIT_SHA or SHORT_SHA is empty." + fi + + # Stop if another run is active, because two runs race on GKE and + # :latest. If the list call fails, continue. + if [[ -n "$TRIGGER_NAME" ]]; then + OTHERS=$$(gcloud builds list --ongoing --region="$LOCATION" \ + --filter="substitutions.TRIGGER_NAME=$TRIGGER_NAME AND id!=$BUILD_ID" \ + --format='value(id)' 2>/dev/null) + if [[ -n "$$OTHERS" ]]; then + release_fail $$STAGE "Another release run is active: $$OTHERS" + fi + fi + + REPO="us-central1-docker.pkg.dev/$PROJECT_ID/evalbench/eval_server" + DEADLINE=$$(( $$(date +%s) + ${_IMAGE_WAIT_SECONDS} )) + echo "Resolving $$REPO:$COMMIT_SHA" + while true; do + DIGEST=$$(gcloud artifacts docker images describe \ + "$$REPO:$COMMIT_SHA" \ + --format='value(image_summary.digest)' 2>/dev/null) + [[ -n "$$DIGEST" ]] && break + if (( $$(date +%s) >= DEADLINE )); then + release_fail $$STAGE "No image at $$REPO:$COMMIT_SHA after ${_IMAGE_WAIT_SECONDS} s. The Build trigger failed or is still running for this commit." + fi + echo "No image yet. Checking again in 5 min." + sleep 300 + done + + TAG="v$$(date -u +%Y.%m.%d).$SHORT_SHA" + + echo "$$REPO" > /workspace/image_repo.txt + echo "$$DIGEST" > /workspace/image_digest.txt + echo "$$TAG" > /workspace/release_tag.txt + echo "digest = $$DIGEST" + echo "tag = $$TAG" + + # Lists the PRs merged since the release that GKE serves now, for the + # result line. Every failure writes a reason and exits 0, so this step + # never fails the release. + - id: release-notes + name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' + waitFor: ['resolve-release'] + entrypoint: 'bash' + args: + - '-c' + - | + set -uo pipefail + NOTES=/workspace/release_prs.txt + GITHUB=https://github.com/GoogleCloudPlatform/evalbench + + unavailable() { + echo "PR list unavailable: $$1" | tee "$$NOTES" + exit 0 + } + + if ! gcloud container clusters get-credentials \ + evalbench-directpath-cluster \ + --zone us-central1-c --project "$PROJECT_ID" > /dev/null 2>&1; then + unavailable "no credentials for evalbench-directpath-cluster." + fi + OLD_DIGEST=$$(kubectl get pods -n evalbench-namespace \ + -l app=evalbench-eval-server \ + -o jsonpath="{range .items[*]}{.status.containerStatuses[?(@.name=='evalbench-eval')].imageID}{'\n'}{end}" \ + | sed -e 's#^.*://##' -e 's#^.*@##' \ + | grep -v '^$$' | sort -u) + if [[ $$(echo "$$OLD_DIGEST" | grep -c .) -ne 1 ]]; then + unavailable "GKE does not serve exactly one digest." + fi + + # The Build trigger tags each image with its full commit SHA. The + # version filter needs the full resource name. + OLD_SHA=$$(gcloud artifacts tags list --package=eval_server \ + --repository=evalbench --location=us-central1 \ + --project="$PROJECT_ID" \ + --filter="version=projects/$PROJECT_ID/locations/us-central1/repositories/evalbench/packages/eval_server/versions/$$OLD_DIGEST" \ + --format='value(name.basename())' 2> /dev/null \ + | grep -E '^[0-9a-f]{40}$$' | head -1) + if [[ -z "$$OLD_SHA" ]]; then + unavailable "the previous commit is unknown. GKE serves $$OLD_DIGEST, which has no commit SHA tag." + fi + echo "$$GITHUB/compare/$$OLD_SHA...$COMMIT_SHA" \ + > /workspace/release_compare.txt + + # A treeless clone has every commit message and no file contents. + # A step timeout fails the release, so limit the clone instead. + if ! timeout 300 git clone --quiet --filter=tree:0 --no-checkout \ + "$$GITHUB.git" /tmp/evalbench; then + unavailable "git clone of $$GITHUB failed or took more than 300 s." + fi + if ! python3 /workspace/.ci/release/release_notes.py \ + --repo-dir /tmp/evalbench --old "$$OLD_SHA" --new "$COMMIT_SHA" \ + > /tmp/prs.txt; then + unavailable "git log $$OLD_SHA..$COMMIT_SHA failed." + fi + mv /tmp/prs.txt "$$NOTES" + echo "PRs since $$OLD_SHA:" + cat "$$NOTES" + + # Serves the candidate image over gRPC and runs three checks on it: + # 1. The smoke eval: the server generates SQL with a model. eval_client + # exits 0 for any completed run, so the verifier grades the CSVs. + # 2. client_sql_smoke.py: the client-generated SQL path, with fixed scores. + # 3. The log scan: any ERROR, CRITICAL, or traceback in the server log + # fails the gate. + - id: smoke-gate + name: 'us-central1-docker.pkg.dev/$PROJECT_ID/evalbench/eval_server:$COMMIT_SHA' + dir: '/evalbench' + waitFor: ['resolve-release'] + timeout: '1200s' + entrypoint: 'bash' + args: + - '-c' + - | + set -uo pipefail + . /workspace/.ci/release/lib.sh + STAGE=smoke-gate + CONFIG=.ci/release/release_smoke_config.yaml + # CsvReporter forces this directory on the gRPC path. + RESULTS=/tmp_session_files/results + SERVER_LOG=/tmp/server.log + + # Add both module roots so eval_client and proto imports resolve. + export PYTHONPATH=/evalbench/evalbench:/evalbench/evalbench/evalproto + + echo "===== start eval_server =====" + python evalbench/eval_server.py --localhost > "$$SERVER_LOG" 2>&1 & + SERVER_PID=$$! + trap 'kill $$SERVER_PID 2>/dev/null || true' EXIT + + dump_log() { + echo "----- eval_server log, last 60 lines -----" + tail -60 "$$SERVER_LOG" + } + + if ! python .ci/release/wait_for_grpc.py --timeout 300 --insecure; then + dump_log + release_fail $$STAGE "The eval server did not answer Ping within 300 s." + fi + + # The final scan starts at the passing attempt, so check startup here. + PROBLEMS=$$(log_problems < "$$SERVER_LOG") + if [[ -n "$$PROBLEMS" ]]; then + echo "$$PROBLEMS" + dump_log + release_fail $$STAGE "The eval server logged errors or exceptions at startup." + fi + + # Retry once if the client crashes or the verifier exits 2 (model). + for ATTEMPT in 1 2; do + echo "===== smoke eval, attempt $$ATTEMPT =====" + LOG_OFFSET=$$(stat -c %s "$$SERVER_LOG") + STARTED=$$(date +%s) + CLIENT_RC=0 + VERIFY_RC=0 + EVALBENCH_HOST=localhost PORT=50051 EVALBENCH_INSECURE=true \ + python evalbench/client/eval_client.py \ + --endpoint=local --experiment="$$CONFIG" || CLIENT_RC=$$? + if [[ $$CLIENT_RC -ne 0 ]]; then + echo "eval_client exited with code $$CLIENT_RC." + else + python .ci/release/verify_release_smoke.py --config "$$CONFIG" \ + --results-dir "$$RESULTS" --since "$$STARTED" || VERIFY_RC=$$? + if [[ $$VERIFY_RC -eq 0 ]]; then + break + fi + fi + if [[ $$ATTEMPT -eq 1 ]] && [[ $$CLIENT_RC -ne 0 || $$VERIFY_RC -eq 2 ]]; then + echo "The failure can come from the model. Retrying once." + continue + fi + dump_log + release_fail $$STAGE "The smoke eval failed on attempt $$ATTEMPT. The build log has the verifier output." + done + + # Scores are fixed, so there is no retry. + echo "===== client SQL smoke =====" + if ! EVALBENCH_INSECURE=true python .ci/release/client_sql_smoke.py \ + --results-dir "$$RESULTS"; then + dump_log + release_fail $$STAGE "The client SQL smoke failed. The build log lists the problems." + fi + + # Skip a failed model attempt. Handled exceptions still fail the gate. + PROBLEMS=$$(tail -c +$$((LOG_OFFSET + 1)) "$$SERVER_LOG" | log_problems) + if [[ -n "$$PROBLEMS" ]]; then + echo "$$PROBLEMS" + release_fail $$STAGE "The eval server logged errors or exceptions during the smoke evals." + fi + + echo "===== smoke gate passed =====" + env: + - 'EVAL_GCP_PROJECT_ID=$PROJECT_ID' + - 'EVAL_GCP_PROJECT_REGION=${_EVAL_REGION}' + - 'GOOGLE_CLOUD_PROJECT=$PROJECT_ID' + - 'UV_CACHE_DIR=/tmp/uv-cache' + + # Tags the tested digest, not :$COMMIT_SHA. + - id: tag-release + name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' + waitFor: ['smoke-gate'] + timeout: '120s' + entrypoint: 'bash' + args: + - '-c' + - | + set -uo pipefail + . /workspace/.ci/release/lib.sh + STAGE=tag-release + REPO=$$(cat /workspace/image_repo.txt) + DIGEST=$$(cat /workspace/image_digest.txt) + TAG=$$(cat /workspace/release_tag.txt) + + # If a Build re-run moved :$COMMIT_SHA, the gate tested another image. + CURRENT=$$(gcloud artifacts docker images describe \ + "$$REPO:$COMMIT_SHA" --format='value(image_summary.digest)') + if [[ "$$CURRENT" != "$$DIGEST" ]]; then + release_fail $$STAGE ":$COMMIT_SHA moved from $$DIGEST to $$CURRENT during the run. Start the release again." + fi + + if ! move_tag "$$REPO" "$$TAG" "$$DIGEST"; then + release_fail $$STAGE "Could not apply the release tag $$TAG." + fi + echo "Tagged $$REPO@$$DIGEST as :$$TAG" + + # Records the digest that GKE serves now. The pod imageID is exact, but + # the Deployment spec can hold a tag. + - id: capture-rollback-target + name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' + waitFor: ['tag-release'] + timeout: '120s' + entrypoint: 'bash' + args: + - '-c' + - | + set -uo pipefail + . /workspace/.ci/release/lib.sh + STAGE=capture-rollback-target + NAMESPACE=evalbench-namespace + SELECTOR=app=evalbench-eval-server + CONTAINER=evalbench-eval + + if ! gcloud container clusters get-credentials \ + evalbench-directpath-cluster \ + --zone us-central1-c --project "$PROJECT_ID"; then + release_fail $$STAGE "Cannot get credentials for evalbench-directpath-cluster." + fi + + RUNNING=$$(kubectl get pods -n "$$NAMESPACE" -l "$$SELECTOR" \ + -o jsonpath="{range .items[*]}{.status.containerStatuses[?(@.name=='$$CONTAINER')].imageID}{'\n'}{end}" \ + | sed -e 's#^.*://##' -e 's#^.*@##' \ + | grep -v '^$$' | sort -u) + COUNT=$$(echo "$$RUNNING" | grep -c .) + + if [[ "$$COUNT" -eq 0 ]]; then + release_fail $$STAGE "No running $$CONTAINER container, so there is no rollback target." + fi + if [[ "$$COUNT" -gt 1 ]]; then + echo "$$RUNNING" + release_fail $$STAGE "Pods serve more than one digest. A rollout is already in progress." + fi + + echo "$$RUNNING" > /workspace/rollback_digest.txt + echo "rollback target = $$RUNNING" + + # Deploy, check, and rollback share one step, because Cloud Build has no + # on-failure hook. + - id: deploy-gke + name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' + waitFor: ['capture-rollback-target'] + # 2 rollouts (600 s), confirm (60 s), and Ping (120 s). + timeout: '900s' + entrypoint: 'bash' + args: + - '-c' + - | + set -uo pipefail + . /workspace/.ci/release/lib.sh + STAGE=deploy-gke + NAMESPACE=evalbench-namespace + SELECTOR=app=evalbench-eval-server + CONTAINER=evalbench-eval + DEPLOYMENT=deployment/evalbench-eval-server-deploy + ROLLOUT_TIMEOUT=300s + + REPO=$$(cat /workspace/image_repo.txt) + NEW_DIGEST=$$(cat /workspace/image_digest.txt) + OLD_DIGEST=$$(cat /workspace/rollback_digest.txt) + TAG=$$(cat /workspace/release_tag.txt) + + if ! gcloud container clusters get-credentials \ + evalbench-directpath-cluster \ + --zone us-central1-c --project "$PROJECT_ID"; then + release_fail $$STAGE "Cannot get credentials for evalbench-directpath-cluster." + fi + + if [[ "$$NEW_DIGEST" == "$$OLD_DIGEST" ]]; then + echo "GKE already serves $$NEW_DIGEST. No rollout is needed." + exit 0 + fi + + running_digests() { + kubectl get pods -n "$$NAMESPACE" -l "$$SELECTOR" \ + -o jsonpath="{range .items[*]}{.status.containerStatuses[?(@.name=='$$CONTAINER')].imageID}{'\n'}{end}" \ + | sed -e 's#^.*://##' -e 's#^.*@##' \ + | grep -v '^$$' | sort -u + } + + # Pin by digest, so a tag move cannot change the running image. + pin() { + kubectl set image "$$DEPLOYMENT" "$$CONTAINER=$$REPO@$$1" \ + -n "$$NAMESPACE" \ + && kubectl rollout status "$$DEPLOYMENT" -n "$$NAMESPACE" \ + --timeout="$$ROLLOUT_TIMEOUT" + } + + rollback() { + echo "===== rolling back to $$OLD_DIGEST =====" + if pin "$$OLD_DIGEST"; then + release_result ROLLED_BACK $$STAGE "$$1 GKE serves the previous digest $$OLD_DIGEST again." + else + kubectl get pods -n "$$NAMESPACE" -l "$$SELECTOR" -o wide + release_result ROLLBACK_FAILED $$STAGE "$$1 The rollback did not complete. Last known-good image: $$REPO@$$OLD_DIGEST" + fi + exit 1 + } + + echo "===== deploy $$NEW_DIGEST =====" + if ! pin "$$NEW_DIGEST"; then + kubectl get pods -n "$$NAMESPACE" -l "$$SELECTOR" -o wide + rollback "The rollout did not complete within $$ROLLOUT_TIMEOUT." + fi + + # Old pods can stay Terminating for a few seconds after the rollout. + echo "===== confirm the pods serve the new digest =====" + for ATTEMPT in {1..20}; do + SERVING=$$(running_digests) + [[ "$$SERVING" == "$$NEW_DIGEST" ]] && break + sleep 3 + done + if [[ "$$SERVING" != "$$NEW_DIGEST" ]]; then + echo "Pods report: $$SERVING" + rollback "The pods do not serve the release digest." + fi + + # The Deployment has no readiness probe, so Ready means only that the + # process started. Ping is the real health check. + POD=$$(kubectl get pods -n "$$NAMESPACE" -l "$$SELECTOR" \ + -o jsonpath='{.items[0].metadata.name}') + echo "===== gRPC health check on $$POD =====" + if ! kubectl exec -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" -- \ + python /evalbench/.ci/release/wait_for_grpc.py --timeout 120; then + kubectl logs -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" --tail=150 + rollback "The new pod $$POD did not answer Ping." + fi + + echo "===== scan the new pod startup log =====" + # Client requests can log an ERROR after startup, so stop at + # "Server started". + PROBLEMS=$$(kubectl logs -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" \ + | sed '/Server started/q' | log_problems) + if [[ -n "$$PROBLEMS" ]]; then + echo "$$PROBLEMS" + rollback "The new pod $$POD logged errors or exceptions at startup." + fi + + # An annotation does not restart the pod. + kubectl annotate "$$DEPLOYMENT" -n "$$NAMESPACE" --overwrite \ + kubernetes.io/change-cause="release $$TAG ($$NEW_DIGEST)" || true + + echo "===== GKE serves $$REPO@$$NEW_DIGEST as $$TAG =====" + + # Moves :latest only after GKE passes, so manual deploys get a verified + # image. + - id: promote-latest + name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' + waitFor: ['deploy-gke'] + timeout: '120s' + entrypoint: 'bash' + args: + - '-c' + - | + set -uo pipefail + . /workspace/.ci/release/lib.sh + REPO=$$(cat /workspace/image_repo.txt) + DIGEST=$$(cat /workspace/image_digest.txt) + if ! move_tag "$$REPO" latest "$$DIGEST"; then + release_fail promote-latest "GKE serves $$DIGEST, but :latest did not move. Break-glass deploys still use the previous release." + fi + echo "$$REPO:latest -> $$DIGEST" + + # Cloud Run pulls from evalbench-dev. The copy keeps the digest. + - id: copy-cloudrun-image + name: 'gcr.io/go-containerregistry/gcrane:debug' + waitFor: ['promote-latest'] + timeout: '300s' + entrypoint: 'sh' + args: + - '-c' + - | + . /workspace/.ci/release/lib.sh + CR_REPO=us-central1-docker.pkg.dev/evalbench-dev/cr-images/eval_server + REPO=$$(cat /workspace/image_repo.txt) + DIGEST=$$(cat /workspace/image_digest.txt) + TAG=$$(cat /workspace/release_tag.txt) + if ! gcrane copy "$$REPO@$$DIGEST" "$$CR_REPO:$$TAG"; then + release_fail copy-cloudrun-image "Could not copy the release to $$CR_REPO. Cloud Run still serves the previous release." + fi + + # The ingress is internal, so Cloud Build cannot send HTTP probes. gcloud + # waits for the revision startup check, and pre-merge CI tests the pages. + - id: deploy-cloudrun + name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' + waitFor: ['copy-cloudrun-image'] + timeout: '600s' + entrypoint: 'bash' + args: + - '-c' + - | + set -uo pipefail + . /workspace/.ci/release/lib.sh + STAGE=deploy-cloudrun + SERVICE=evalbench + JOBS=(precompute-job recompute-job) + RUN=(--project=evalbench-dev --region=us-central1 --quiet) + CR_REPO=us-central1-docker.pkg.dev/evalbench-dev/cr-images/eval_server + DIGEST=$$(cat /workspace/image_digest.txt) + + OLD_REVISION=$$(gcloud run services describe "$$SERVICE" "$${RUN[@]}" \ + --format='value(status.latestReadyRevisionName)') + echo "current revision = $$OLD_REVISION" + + echo "===== deploy a revision with no traffic =====" + if ! gcloud run deploy "$$SERVICE" "$${RUN[@]}" \ + --image="$$CR_REPO@$$DIGEST" --no-traffic; then + release_fail $$STAGE "The new revision did not become ready. Traffic stays on $$OLD_REVISION." + fi + + NEW_REVISION=$$(gcloud run services describe "$$SERVICE" "$${RUN[@]}" \ + --format='value(status.latestCreatedRevisionName)') + SERVED=$$(gcloud run revisions describe "$$NEW_REVISION" "$${RUN[@]}" \ + --format='value(status.imageDigest)') + if [[ "$$SERVED" != *"@$$DIGEST" ]]; then + release_fail $$STAGE "$$NEW_REVISION runs $$SERVED, not $$DIGEST. Traffic stays on $$OLD_REVISION." + fi + + echo "===== shift traffic to $$NEW_REVISION =====" + if ! gcloud run services update-traffic "$$SERVICE" "$${RUN[@]}" \ + --to-latest; then + release_fail $$STAGE "Could not shift traffic to $$NEW_REVISION. Traffic stays on $$OLD_REVISION." + fi + + FAILED_JOBS="" + for JOB in "$${JOBS[@]}"; do + gcloud run jobs update "$$JOB" "$${RUN[@]}" \ + --image="$$CR_REPO@$$DIGEST" || FAILED_JOBS="$$FAILED_JOBS $$JOB" + done + if [[ -n "$$FAILED_JOBS" ]]; then + release_fail $$STAGE "The UI serves the release, but these jobs did not update:$$FAILED_JOBS" + fi + + # 'make deploy-corprun' uses :latest. + if ! move_tag "$$CR_REPO" latest "$$DIGEST"; then + release_fail $$STAGE "The UI and jobs serve the release, but cr-images :latest did not move." + fi + echo "===== Cloud Run: $$NEW_REVISION serves $$CR_REPO@$$DIGEST =====" + + - id: notify-success + name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' + waitFor: ['deploy-cloudrun', 'release-notes'] + timeout: '60s' + entrypoint: 'bash' + args: + - '-c' + - | + . /workspace/.ci/release/lib.sh + release_result SUCCESS notify-success "GKE and Cloud Run serve $$(cat /workspace/release_tag.txt) ($$(cat /workspace/image_digest.txt))." + +substitutions: + _EVAL_REGION: 'global' + # The Build trigger took 15 to 26 min in 59 runs. + _IMAGE_WAIT_SECONDS: '1800' + +# Sum of the step timeouts. +timeout: '5820s' + +options: + logging: CLOUD_LOGGING_ONLY + machineType: 'E2_HIGHCPU_8' + # lib.sh reads these to build the RELEASE_RESULT line. + env: + - 'BUILD_ID=$BUILD_ID' + - 'COMMIT_SHA=$COMMIT_SHA' + - 'LOCATION=$LOCATION' + - 'PROJECT_ID=$PROJECT_ID' diff --git a/.ci/release/alert_policy.yaml b/.ci/release/alert_policy.yaml new file mode 100644 index 00000000..f05bde85 --- /dev/null +++ b/.ci/release/alert_policy.yaml @@ -0,0 +1,54 @@ +# Log-based alert that emails the RELEASE_RESULT line (lib.sh) of each +# weekly release run. +# +# TRIGGER_ID and CHANNEL_NAME are placeholders, so this public file has no +# project resource IDs. The live policy already has the real values: +# TRIGGER_ID id of the trigger 'weekly-evalbench-release' (us-central1) +# CHANNEL_NAME name of the notification channel 'EvalBench release' +# To change the policy, fill in a copy of this file and run +# 'gcloud alpha monitoring policies update'. Do not create a duplicate. +displayName: 'EvalBench weekly release' +combiner: OR +conditions: + - displayName: 'Weekly release result' + conditionMatchedLog: + filter: >- + resource.type="build" + AND resource.labels.build_trigger_id="TRIGGER_ID" + AND textPayload:"RELEASE_RESULT status=" + labelExtractors: + status: 'REGEXP_EXTRACT(textPayload, "status=(\\S+)")' + stage: 'REGEXP_EXTRACT(textPayload, "stage=(\\S+)")' + tag: 'REGEXP_EXTRACT(textPayload, "tag=(\\S+)")' + commit: 'REGEXP_EXTRACT(textPayload, "commit=(\\S+)")' + # 'log' is a reserved label name. + build_log: 'REGEXP_EXTRACT(textPayload, "log=(\\S+)")' + compare: 'REGEXP_EXTRACT(textPayload, "compare=(\\S+)")' + prs: 'REGEXP_EXTRACT(textPayload, "prs=\"([^\"]*)\"")' + detail: 'REGEXP_EXTRACT(textPayload, "detail=\"(.*)\"")' +alertStrategy: + # 5 minutes is the minimum. + notificationRateLimit: + period: 300s + autoClose: 1800s +documentation: + subject: 'EvalBench release ${log.extracted_label.status}: ${log.extracted_label.tag}' + mimeType: text/markdown + content: | + **Status:** ${log.extracted_label.status} + + **Detail:** ${log.extracted_label.detail} + + **Stage:** ${log.extracted_label.stage} + + **Tag:** ${log.extracted_label.tag} + + **PRs since the last release:** ${log.extracted_label.prs} + + **Compare:** ${log.extracted_label.compare} + + **Commit:** ${log.extracted_label.commit} + + **Build log:** ${log.extracted_label.build_log} +notificationChannels: + - CHANNEL_NAME diff --git a/.ci/release/build_status_policy.yaml b/.ci/release/build_status_policy.yaml new file mode 100644 index 00000000..6c945813 --- /dev/null +++ b/.ci/release/build_status_policy.yaml @@ -0,0 +1,49 @@ +# Log-based alert that emails when a weekly release run fails. +# +# Cloud Build writes an audit entry when a build ends. Its status code is 1 +# for CANCELLED, 4 for TIMEOUT, 9 for FAILURE, and 13 for INTERNAL_ERROR. A +# step timeout or a step image that does not pull gives code 9 and prints no +# RELEASE_RESULT line, so this policy also emails code 9. A failed step that +# prints its RELEASE_RESULT line sends a second email from alert_policy.yaml. +# +# Not covered: a build that never starts, for example an invalid config. +# +# A log-based policy has only one condition, so this alert is a separate +# policy. TRIGGER_ID and CHANNEL_NAME are placeholders, as in +# alert_policy.yaml. To create the policy, fill in a copy of this file and +# run 'gcloud alpha monitoring policies create'. +displayName: 'EvalBench weekly release build status' +combiner: OR +conditions: + - displayName: 'Weekly release build failed' + conditionMatchedLog: + filter: >- + logName:"cloudaudit.googleapis.com%2Factivity" + AND resource.type="build" + AND resource.labels.build_trigger_id="TRIGGER_ID" + AND protoPayload.methodName="google.devtools.cloudbuild.v1.CloudBuild.CreateBuild" + AND operation.last=true + AND protoPayload.status.code>0 + labelExtractors: + code: 'EXTRACT(protoPayload.status.code)' +alertStrategy: + # 5 minutes is the minimum. + notificationRateLimit: + period: 300s + autoClose: 1800s +documentation: + subject: 'EvalBench release build failed (code ${log.extracted_label.code})' + mimeType: text/markdown + content: | + **Status code:** ${log.extracted_label.code} (1 = CANCELLED, 4 = TIMEOUT, 9 = FAILURE, 13 = INTERNAL_ERROR) + + The release run failed. + + 1. Open the build log to see where the run stopped. + 2. Check that all GKE pods and Cloud Run serve the same image digest. If + they do not, the deploy is incomplete. + 3. If necessary, start the release again. + + **Build log:** https://console.cloud.google.com/cloud-build/builds;region=us-central1/${resource.label.build_id}?project=${resource.label.project_id} +notificationChannels: + - CHANNEL_NAME diff --git a/.ci/release/client_sql_smoke.py b/.ci/release/client_sql_smoke.py new file mode 100644 index 00000000..f16150c8 --- /dev/null +++ b/.ci/release/client_sql_smoke.py @@ -0,0 +1,186 @@ +#!/usr/bin/env python3 +"""Runs the client-generated SQL path on the local eval server. + +Most production clients generate SQL themselves and send it back. The server +uses the noop generator, reads its configs from the EvalConfig resources, and +only executes and scores the SQL. This script does the same with fixed SQL, +so every score is known before the run: + + batch pass golden SQL on every item. Every scorer gives 100. + streaming pass wrong SQL on the last item. exact_match and set_match give + 0 for it, which proves that the scorers can fail. + +No model is called, so a failure is never model variance. + +Exit codes: 0 every score matches, 1 otherwise. + +Local test, with the server from 'python evalbench/eval_server.py --localhost': + EVALBENCH_INSECURE=true python .ci/release/client_sql_smoke.py +""" +import argparse +import asyncio +import os +import sys + +import yaml + +_HERE = os.path.dirname(os.path.abspath(__file__)) +_REPO = os.path.dirname(os.path.dirname(_HERE)) +# Add both module roots so eval_client and generated proto imports resolve. +for _path in (os.path.join(_REPO, "evalbench"), + os.path.join(_REPO, "evalbench", "evalproto")): + if _path not in sys.path: + sys.path.insert(0, _path) + +from client.eval_client import EvalbenchClient # noqa: E402 +from evalproto import eval_config_pb2, eval_connect_pb2 # noqa: E402 +from evalproto import eval_request_pb2 # noqa: E402 +from verify_release_smoke import ( # noqa: E402 + HARD_ERROR_COLUMNS, as_prompt_id, as_score, is_empty, read_rows) + +# The server replaces a config value that equals a resource address with the +# path of that resource in its session directory. +ROOT = "release_smoke" +DIALECT = "sqlite" +# Resource address -> repo file. None means the content is in RESOURCES_TEXT. +RESOURCES = { + f"{ROOT}/release_smoke.evalset.json": + ".ci/release/release_smoke.evalset.json", + f"{ROOT}/sqlite.yaml": "datasets/bat/db_configs/sqlite.yaml", + f"{ROOT}/noop_model.yaml": None, +} +RESOURCES_TEXT = {f"{ROOT}/noop_model.yaml": "generator: noop\n"} +CONFIG = { + "dataset_config": f"{ROOT}/release_smoke.evalset.json", + "dataset_format": "evalbench-standard-format", + "database_configs": [f"{ROOT}/sqlite.yaml"], + "dialects": [DIALECT], + "dialect": DIALECT, + # The orchestrator sends read queries only. DML goldens also write + # CURRENT_TIMESTAMP, so their scores are not fixed. + "query_types": ["dql"], + "setup_directory": "datasets/bat/setup", + "model_config": f"{ROOT}/noop_model.yaml", + "prompt_generator": "NOOPGenerator", + "num_trials": 1, + "scorers": {"exact_match": None, "set_match": None, + "executable_sql": None, "returned_sql": None}, + "reporting": {"csv": {"output_directory": "results"}}, +} +WRONG_SQL = "SELECT -1 AS wrong_answer;" +# Expected scores for the item with WRONG_SQL. Golden SQL gives 100 on all. +WRONG_SCORES = {"exact_match": 0, "set_match": 0, + "executable_sql": 100, "returned_sql": 100} + + +def config_request(): + request = eval_config_pb2.EvalConfigRequest( + yaml_config=yaml.safe_dump(CONFIG).encode("utf-8")) + for address, path in RESOURCES.items(): + if path is None: + content = RESOURCES_TEXT[address].encode("utf-8") + else: + with open(os.path.join(_REPO, path), "rb") as f: + content = f.read() + request.resources.add(address=address, content=content) + return request + + +async def run_pass(streaming, wrong_last): + """Returns (job_id, {item id: sent SQL}) for one Eval call.""" + client = EvalbenchClient("local") + metadata = client.metadata + try: + await client.stub.Ping(eval_request_pb2.PingRequest(), + metadata=metadata) + await client.stub.Connect( + eval_connect_pb2.EvalConnectRequest( + client_id="release-smoke", streaming_eval=streaming), + metadata=metadata) + await client.stub.EvalConfig(config_request(), metadata=metadata) + items = [item async for item in client.stub.ListEvalInputs( + eval_request_pb2.EvalInputRequest(), metadata=metadata)] + if not items: + raise RuntimeError("ListEvalInputs returned no items") + sent = {} + for i, item in enumerate(items): + golden = item.golden_sql[DIALECT].sql_statements[0] + wrong = wrong_last and i == len(items) - 1 + item.generated_sql = WRONG_SQL if wrong else golden + sent[str(item.id)] = item.generated_sql + response = await client.stub.Eval(iter(items), metadata=metadata) + return response.response, sent + finally: + await client.channel.close() + + +def check(name, results_dir, job_id, sent, wrong_id): + """Returns the problems in the reports of one pass.""" + job_dir = os.path.join(results_dir, job_id) + if not job_id or not os.path.isdir(job_dir): + return [f"no report directory {job_dir}"] + problems = [] + # prompt_id is the dataset id. id is the eval id that scores.csv uses. + evals = {as_prompt_id(r.get("prompt_id")): r + for r in read_rows(job_dir, "evals.csv")} + eval_ids = {} + for item_id, sql in sorted(sent.items()): + row = evals.get(item_id) + if row is None: + problems.append(f"{item_id}: no eval row") + continue + eval_ids[item_id] = as_prompt_id(row.get("id")) + for column in HARD_ERROR_COLUMNS + ("sql_generator_error", + "generated_error"): + if not is_empty(row.get(column)): + problems.append(f"{item_id}: {column}: " + f"{row[column].strip()[:200]}") + # The noop generator must keep the client SQL. + if (row.get("generated_sql") or "").strip() != sql.strip(): + problems.append(f"{item_id}: generated_sql is not the SQL that " + f"the client sent") + + scores = {(r.get("comparator"), as_prompt_id(r.get("id"))): r + for r in read_rows(job_dir, "scores.csv")} + for scorer in sorted(CONFIG["scorers"]): + for item_id, eval_id in sorted(eval_ids.items()): + want = WRONG_SCORES[scorer] if item_id == wrong_id else 100 + row = scores.get((scorer, eval_id)) + got = as_score(row.get("score")) if row else None + if got != want: + problems.append(f"{scorer}: {item_id} scored {got}, " + f"expected {want}") + status = "[PASS]" if not problems else "[FAIL]" + print(f"{status} {name}: job {job_id}, {len(sent)} items, " + f"{len(problems)} problem(s)") + for problem in problems: + print(f" {problem}") + return problems + + +async def run(results_dir): + failed = False + for name, streaming, wrong_last in (("batch pass", False, False), + ("streaming pass", True, True)): + try: + job_id, sent = await run_pass(streaming, wrong_last) + except Exception as e: + print(f"[FAIL] {name}: {type(e).__name__}: {e}") + failed = True + continue + wrong_id = list(sent)[-1] if wrong_last else None + failed |= bool(check(name, results_dir, job_id, sent, wrong_id)) + return 1 if failed else 0 + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + # CsvReporter forces this directory on the gRPC path. + parser.add_argument("--results-dir", default="/tmp_session_files/results", + help="Directory that holds /scores.csv.") + args = parser.parse_args() + return asyncio.run(run(args.results_dir)) + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.ci/release/lib.sh b/.ci/release/lib.sh new file mode 100644 index 00000000..35387fef --- /dev/null +++ b/.ci/release/lib.sh @@ -0,0 +1,64 @@ +# Helpers for the steps in release.cloudbuild.yaml: +# . /workspace/.ci/release/lib.sh +# +# POSIX sh, so the busybox shell in the gcrane image can source it. Needs +# BUILD_ID, COMMIT_SHA, LOCATION and PROJECT_ID from options.env. + +# alert_policy.yaml matches this prefix and parses the line format. If you +# change either, update the policy. +RELEASE_RESULT_PREFIX="RELEASE_RESULT" + +# Prints the result line of the run. +# +# Usage: release_result STATUS STAGE DETAIL +# STATUS SUCCESS, FAILED, ROLLED_BACK, or ROLLBACK_FAILED +# STAGE the id of the step that ended the run +# DETAIL one sentence for the email +# +# Reads the tag, PR list, and compare link from /workspace. Keep detail +# last, because the alert policy reads it with a greedy regex. +release_result() { + _tag=$(cat /workspace/release_tag.txt 2>/dev/null || echo "none") + _detail=$(printf '%s' "$3" | tr '"\n' "' ") + _prs=$(cat /workspace/release_prs.txt 2>/dev/null \ + || echo "PR list not computed before this step ended the run.") + _prs=$(printf '%s' "$_prs" | tr '"\n' "' ") + _compare=$(cat /workspace/release_compare.txt 2>/dev/null || echo "unavailable") + _log="https://console.cloud.google.com/cloud-build/builds;region=${LOCATION}/${BUILD_ID}?project=${PROJECT_ID}" + _commit="https://github.com/GoogleCloudPlatform/evalbench/commit/${COMMIT_SHA}" + echo "${RELEASE_RESULT_PREFIX} status=$1 stage=$2 tag=${_tag} commit=${_commit} log=${_log} compare=${_compare} prs=\"${_prs}\" detail=\"${_detail}\"" +} + +release_fail() { + release_result FAILED "$1" "$2" + exit 1 +} + +# Prints the ERROR and CRITICAL records and tracebacks in an eval_server log. +# Prints nothing if the log is clean. Matches only at the start of a line, +# because eval payloads can quote a traceback. +# +# Usage: log_problems < LOG +log_problems() { + _log=$(cat) + printf '%s\n' "$_log" \ + | grep -nE '^[0-9-]{10} [0-9:,]+ \[[^]]*\] (ERROR|CRITICAL) ' | head -40 + printf '%s\n' "$_log" \ + | grep -n -A8 '^Traceback (most recent call last)' | head -80 +} + +# Points TAG at DIGEST. 'tags update' moves an existing tag without the +# tags.delete permission. 'docker tags add' creates a new tag. +# +# Usage: move_tag IMAGE TAG DIGEST +# IMAGE the image path, e.g. us-central1-docker.pkg.dev/PROJECT/REPO/PKG +move_tag() { + _host=$(echo "$1" | cut -d/ -f1) + _project=$(echo "$1" | cut -d/ -f2) + _repository=$(echo "$1" | cut -d/ -f3) + _package=$(echo "$1" | cut -d/ -f4-) + gcloud artifacts tags update "$2" --location="${_host%-docker.pkg.dev}" \ + --project="$_project" --repository="$_repository" \ + --package="$_package" --version="$3" 2>/dev/null \ + || gcloud artifacts docker tags add "$1@$3" "$1:$2" +} diff --git a/.ci/release/release_notes.py b/.ci/release/release_notes.py new file mode 100644 index 00000000..e7342b0e --- /dev/null +++ b/.ci/release/release_notes.py @@ -0,0 +1,128 @@ +#!/usr/bin/env python3 +"""Prints the PRs merged between two commits as one line for the release email. + +The release email comes from a log-based alert, so the list must fit on the +RELEASE_RESULT line. The script groups PRs by the conventional-commit type of +the title, for example "feat(agy): ..." or "ci: ...". + +Usage: + release_notes.py --repo-dir DIR --old OLD_SHA --new NEW_SHA + +Exit codes: 0 with the list on stdout, 1 if git fails. +""" +import argparse +import re +import subprocess +import sys + +# Group order in the email. Other types go to "other". +TYPE_ORDER = ("feat", "fix", "perf", "refactor", "ci", "test", "docs", + "build", "chore") +# The alert label keeps about 1,024 characters. Leave room for the suffix. +MAX_CHARS = 900 +MAX_TITLE = 70 + +SQUASH_RE = re.compile(r"^(?P.*?)\s*\(#(?P<pr>\d+)\)$") +MERGE_RE = re.compile(r"^Merge pull request #(?P<pr>\d+) from \S+") +TYPE_RE = re.compile( + r"^(?P<type>[a-zA-Z]+)(?P<scope>\([^)]*\))?!?:\s*(?P<desc>.+)$") + + +def parse_commit(subject, body): + """Returns (pr, title) for a PR merge commit, or None.""" + match = SQUASH_RE.match(subject) + if match: + return int(match.group("pr")), match.group("title") + match = MERGE_RE.match(subject) + if match: + lines = [line.strip() for line in body.splitlines() if line.strip()] + return int(match.group("pr")), (lines[0] if lines else subject) + return None + + +def group_of(title): + """Returns (type, short title) for a PR title. + + "feat(agy): add x" becomes ("feat", "agy: add x"). A title without a + known type goes to ("other", title). + """ + match = TYPE_RE.match(title) + if not match or match.group("type").lower() not in TYPE_ORDER: + return "other", title + scope = (match.group("scope") or "").strip("()") + desc = match.group("desc") + return match.group("type").lower(), f"{scope}: {desc}" if scope else desc + + +def shorten(title): + title = title.replace('"', "'").strip() + if len(title) > MAX_TITLE: + title = title[:MAX_TITLE - 3].rstrip(" .,") + "..." + return title + + +def format_prs(prs): + """Formats [(pr, title)] as "3 PRs. feat (1): #1 a. fix (2): #2 b, #3 c." + + Stops at MAX_CHARS on a whole PR and adds "+N more". + """ + if not prs: + return "No new PRs." + groups = {} + for pr, title in prs: + kind, short = group_of(title) + groups.setdefault(kind, []).append(f"#{pr} {shorten(short)}") + + text = f"{len(prs)} PRs." + shown = 0 + for kind in TYPE_ORDER + ("other",): + items = groups.get(kind, []) + if not items: + continue + head = f" {kind} ({len(items)}): " + for i, item in enumerate(items): + piece = (head if i == 0 else ", ") + item + if len(text) + len(piece) > MAX_CHARS: + return f"{text} ... +{len(prs) - shown} more in the compare link." + text += piece + shown += 1 + text += "." + return text + + +def merged_prs(repo_dir, old, new): + """Returns [(pr, title)] for first-parent commits in old..new, newest first.""" + out = subprocess.run( + ["git", "-C", repo_dir, "log", "--first-parent", + "--format=%s%x1f%b%x1e", f"{old}..{new}"], + check=True, capture_output=True, text=True).stdout + prs = [] + for record in out.split("\x1e"): + if not record.strip(): + continue + subject, _, body = record.strip("\n").partition("\x1f") + parsed = parse_commit(subject.strip(), body) + if parsed: + prs.append(parsed) + return prs + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--repo-dir", required=True, + help="Git checkout that contains both commits.") + parser.add_argument("--old", required=True, + help="Commit that GKE served before the release.") + parser.add_argument("--new", required=True, help="Release commit.") + args = parser.parse_args() + try: + prs = merged_prs(args.repo_dir, args.old, args.new) + except subprocess.CalledProcessError as e: + print(f"git log failed: {e.stderr.strip()}", file=sys.stderr) + return 1 + print(format_prs(prs)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.ci/release/release_smoke.evalset.json b/.ci/release/release_smoke.evalset.json new file mode 100644 index 00000000..98571150 --- /dev/null +++ b/.ci/release/release_smoke.evalset.json @@ -0,0 +1,235 @@ +[ + { + "id": 3, + "nl_prompt": "How many bloggers who have posted at least 1 blog in 2024 and have linked twitter account", + "query_type": "dql", + "database": "db_blog", + "dialects": [ + "sqlite" + ], + "golden_sql": { + "sqlite": [ + "SELECT COUNT(DISTINCT u.user_id) AS num_bloggers FROM tbl_users u INNER JOIN tbl_posts p ON u.user_id = p.user_id WHERE p.created_at BETWEEN DATE('2024-01-01') AND DATE('2024-12-31') AND json_extract(u.social_media_links, '$.twitter') IS NOT NULL;" + ] + }, + "eval_query": { + "sqlite": [ + null + ] + }, + "setup_sql": {}, + "cleanup_sql": {}, + "tags": [ + "DQL", + "difficulty: moderate", + "SELECT", + "JOIN", + "JSON", + "JSON_EXTRACT", + "datalinking" + ], + "other": { + "Comment": "LGTM", + "nl_prompt_base": "How many bloggers who have posted at least 1 blog in 2024 and have linked twitter account", + "nl_prompt_extra_context": "", + "public": true + } + }, + { + "id": 5, + "nl_prompt": "What is the number of comments per post for each user? -- Use user_id to identify users and post_id to identify posts.", + "query_type": "dql", + "database": "db_blog", + "dialects": [ + "sqlite" + ], + "golden_sql": { + "sqlite": [ + "SELECT u.user_id, p.post_id, COUNT(c.comment_id) AS comment_count FROM tbl_users u INNER JOIN tbl_posts p ON u.user_id = p.user_id LEFT JOIN tbl_comments c ON p.post_id = c.post_id GROUP BY u.user_id, p.post_id;" + ] + }, + "eval_query": { + "sqlite": [ + null + ] + }, + "setup_sql": {}, + "cleanup_sql": {}, + "tags": [ + "DQL", + "difficulty: simple", + "SELECT", + "JOIN", + "AGGREGATE", + "CASE", + "LEFT_JOIN", + "IS_NULL" + ], + "other": { + "Comment": "LGTM", + "nl_prompt_base": "What is the number of comments per post for each user?", + "nl_prompt_extra_context": "Use user_id to identify users and post_id to identify posts.", + "public": true + } + }, + { + "id": 42, + "nl_prompt": "Can you update my username to 'magic_one' -- user_id = 9;", + "query_type": "dml", + "database": "db_blog", + "dialects": [ + "sqlite" + ], + "golden_sql": { + "sqlite": [ + "UPDATE tbl_users SET username = 'magic_one', updated_at = CURRENT_TIMESTAMP WHERE user_id = 9;" + ] + }, + "eval_query": { + "sqlite": [ + "SELECT username, updated_at, updated_at > '2024-02-27 07:27:14' AS is_updated FROM tbl_users WHERE user_id = 9;" + ] + }, + "setup_sql": { + "sqlite": [] + }, + "cleanup_sql": { + "sqlite": [ + "UPDATE tbl_users SET username = 'sophia.anderson', updated_at = '2024-02-27 07:26:14' WHERE user_id = 9;" + ] + }, + "tags": [ + "DML", + "difficulty: moderate", + "UPDATE" + ], + "other": { + "Comment": "LGTM. Instructions do not include updated_at, but that is implied.", + "nl_prompt_base": "Can you update my username to 'magic_one'", + "nl_prompt_extra_context": "user_id = 9;", + "public": true + } + }, + { + "id": 184, + "nl_prompt": "Update the post content in tbl_posts -- post_id = 3, new content should be 'Updated content'", + "query_type": "dml", + "database": "db_blog", + "dialects": [ + "sqlite" + ], + "golden_sql": { + "sqlite": [ + "UPDATE tbl_posts SET content = 'Updated content' WHERE post_id = 3;" + ] + }, + "eval_query": { + "sqlite": [ + "SELECT content FROM tbl_posts WHERE post_id = 3;" + ] + }, + "setup_sql": { + "sqlite": [ + "UPDATE tbl_posts SET content = 'Tips and tricks for efficient web development.' WHERE post_id = 3;" + ] + }, + "cleanup_sql": { + "sqlite": [ + "UPDATE tbl_posts SET content = 'Tips and tricks for efficient web development.' WHERE post_id = 3;" + ] + }, + "tags": [ + "DML", + "difficulty: simple", + "UPDATE" + ], + "other": { + "nl_prompt_base": "Update the post content in tbl_posts", + "nl_prompt_extra_context": "post_id = 3, new content should be 'Updated content'", + "public": true + } + }, + { + "id": 43, + "nl_prompt": "Update the labels table to include a json field called comments_description.", + "query_type": "ddl", + "database": "db_blog", + "dialects": [ + "sqlite" + ], + "golden_sql": { + "sqlite": [ + "ALTER TABLE tbl_labels ADD COLUMN comments_description TEXT CHECK (json_valid(comments_description));" + ] + }, + "eval_query": { + "sqlite": [ + "SELECT COUNT(*) FROM tbl_labels WHERE json_valid(comments_description);" + ] + }, + "setup_sql": { + "sqlite": [ + "BEGIN TRANSACTION; CREATE TABLE tbl_labels_new (label_id INTEGER PRIMARY KEY, creator_id INTEGER DEFAULT NULL, updated_by INTEGER DEFAULT NULL, name TEXT DEFAULT NULL, description TEXT DEFAULT NULL, created_at TEXT DEFAULT NULL, updated_at TEXT DEFAULT NULL, post_count INTEGER DEFAULT NULL, visibility_status TEXT DEFAULT NULL, is_active INTEGER DEFAULT NULL, usage_frequency INTEGER DEFAULT NULL, parent_label_id INTEGER DEFAULT NULL, FOREIGN KEY (creator_id) REFERENCES tbl_users(user_id), FOREIGN KEY (updated_by) REFERENCES tbl_users(user_id), FOREIGN KEY (parent_label_id) REFERENCES tbl_labels(label_id)); INSERT INTO tbl_labels_new (label_id, creator_id, updated_by, name, description, created_at, updated_at, post_count, visibility_status, is_active, usage_frequency, parent_label_id) SELECT label_id, creator_id, updated_by, name, description, created_at, updated_at, post_count, visibility_status, is_active, usage_frequency, parent_label_id FROM tbl_labels; DROP TABLE tbl_labels; ALTER TABLE tbl_labels_new RENAME TO tbl_labels; COMMIT;" + ] + }, + "cleanup_sql": { + "sqlite": [ + "BEGIN TRANSACTION; CREATE TABLE tbl_labels_new (label_id INTEGER PRIMARY KEY, creator_id INTEGER DEFAULT NULL, updated_by INTEGER DEFAULT NULL, name TEXT DEFAULT NULL, description TEXT DEFAULT NULL, created_at TEXT DEFAULT NULL, updated_at TEXT DEFAULT NULL, post_count INTEGER DEFAULT NULL, visibility_status TEXT DEFAULT NULL, is_active INTEGER DEFAULT NULL, usage_frequency INTEGER DEFAULT NULL, parent_label_id INTEGER DEFAULT NULL, FOREIGN KEY (creator_id) REFERENCES tbl_users(user_id), FOREIGN KEY (updated_by) REFERENCES tbl_users(user_id), FOREIGN KEY (parent_label_id) REFERENCES tbl_labels(label_id)); INSERT INTO tbl_labels_new (label_id, creator_id, updated_by, name, description, created_at, updated_at, post_count, visibility_status, is_active, usage_frequency, parent_label_id) SELECT label_id, creator_id, updated_by, name, description, created_at, updated_at, post_count, visibility_status, is_active, usage_frequency, parent_label_id FROM tbl_labels; DROP TABLE tbl_labels; ALTER TABLE tbl_labels_new RENAME TO tbl_labels; COMMIT;" + ] + }, + "tags": [ + "DDL", + "difficulty: simple", + "ALTER", + "ADD_COLUMN", + "JSON" + ], + "other": { + "Comment": "LGTM", + "nl_prompt_base": "Update the labels table to include a json type field called comments_description.", + "nl_prompt_extra_context": null, + "public": true + } + }, + { + "id": 185, + "nl_prompt": "Rename the content column in the tbl_posts table to post_content", + "query_type": "ddl", + "database": "db_blog", + "dialects": [ + "sqlite" + ], + "golden_sql": { + "sqlite": [ + "ALTER TABLE tbl_posts RENAME COLUMN content TO post_content;" + ] + }, + "eval_query": { + "sqlite": [ + "SELECT post_content FROM tbl_posts;" + ] + }, + "setup_sql": { + "sqlite": [ + "ALTER TABLE tbl_posts RENAME COLUMN post_content TO content;" + ] + }, + "cleanup_sql": { + "sqlite": [ + "ALTER TABLE tbl_posts RENAME COLUMN post_content TO content;" + ] + }, + "tags": [ + "DDL", + "difficulty: simple", + "ALTER", + "RENAME" + ], + "other": { + "comment": "LGTM", + "nl_prompt_base": "Rename the content column in the tbl_posts table to post_content", + "nl_prompt_extra_context": null, + "public": true + } + } +] diff --git a/.ci/release/release_smoke_config.yaml b/.ci/release/release_smoke_config.yaml new file mode 100644 index 00000000..c44464dd --- /dev/null +++ b/.ci/release/release_smoke_config.yaml @@ -0,0 +1,34 @@ +# Release smoke eval. The release pipeline sends it to the gRPC server in +# the candidate image. SQLite only, so it needs no external database. Paths +# are relative to /evalbench. +dataset_config: .ci/release/release_smoke.evalset.json +# One row per prompt, so the verifier can check coverage. +num_trials: 1 + +database_configs: + - datasets/bat/db_configs/sqlite.yaml +dialects: + - sqlite +query_types: + - dql + - dml + - ddl + +# Rebuild db_blog on every run, because DML and DDL prompts change it. +setup_directory: datasets/bat/setup + +model_config: datasets/model_configs/gemini_2.5_pro_model.yaml +prompt_generator: 'SQLGenBasePromptGenerator' + +# The verifier needs a score from every scorer, not a high score. +scorers: + exact_match: null + returned_sql: null + executable_sql: null + set_match: null + llmrater: + model_config: datasets/model_configs/gemini_2.5_pro_model.yaml + +reporting: + csv: + output_directory: 'results/release' diff --git a/.ci/release/verify_release_smoke.py b/.ci/release/verify_release_smoke.py new file mode 100644 index 00000000..a0575a93 --- /dev/null +++ b/.ci/release/verify_release_smoke.py @@ -0,0 +1,290 @@ +#!/usr/bin/env python3 +"""Grades the release smoke eval from its CSV reports. + +A clean run has: + 1. All four report files, each with rows. + 2. Exactly one eval row for each (dialect, prompt). + 3. No prompt generator error and no golden query error. + 4. Generated SQL and no SQL generator error on every row. + 5. A numeric score with no comparison_error from every scorer on every row. + 6. At least one executed query for each dialect and query type. +Accuracy is printed but not gated, because model variance makes it flaky. + +Exit codes: 0 clean, 1 hard failure, 2 only model-dependent checks failed +(4, 6, or LLM judge errors), so a retry can help. + +Local test (the CLI runner writes to results/release): + START=$(date +%s) + EVAL_CONFIG=.ci/release/release_smoke_config.yaml ./evalbench/run.sh + uv run python .ci/release/verify_release_smoke.py --since "$START" +""" +import argparse +import ast +import csv +import json +import math +import os +import sys + +from pyaml_env import parse_config + +DEFAULT_CONFIG = ".ci/release/release_smoke_config.yaml" +REPORT_FILES = ("configs.csv", "evals.csv", "scores.csv", "summary.csv") +HARD_ERROR_COLUMNS = ("prompt_generator_error", "golden_error") +# Scorers that call a model. Their errors are model-dependent. +LLM_SCORERS = {"llmrater"} +EMPTY = {"", "nan", "none", "null"} + +HARD = 1 +RETRYABLE = 2 + + +def is_empty(raw): + return (raw or "").strip().lower() in EMPTY + + +def as_score(raw): + try: + score = float(raw) + except (TypeError, ValueError): + return None + return score if math.isfinite(score) else None + + +def as_prompt_id(raw): + """pandas can write id 9 as "9.0".""" + text = (raw or "").strip() + try: + return str(int(float(text))) + except ValueError: + return text + + +def as_dialect(raw): + """The CSV stores the dialects list as its repr, e.g. "['sqlite']".""" + try: + parsed = ast.literal_eval(raw or "") + except (ValueError, SyntaxError): + parsed = raw + if isinstance(parsed, (list, tuple)): + return ",".join(str(d) for d in parsed) + return str(parsed or "") + + +def expected_prompts(config): + """Returns {(dialect, prompt id): query type} that the run must cover. + + Filters like dataset.py: by the run config dialects and query types, if + the lists are not empty. + """ + with open(config["dataset_config"]) as f: + data = json.load(f) + items = data["scenarios"] if isinstance(data, dict) else data + dialects = config.get("dialects") or [] + query_types = [q.lower() for q in config.get("query_types") or []] + expected = {} + for item in items: + query_type = item["query_type"].lower() + if query_types and query_type not in query_types: + continue + for dialect in item.get("dialects") or []: + if not dialects or dialect in dialects: + expected[(dialect, str(item["id"]))] = query_type + return expected + + +def job_dirs(results_dir, since): + """Returns job directories changed after `since`, newest first.""" + if not os.path.isdir(results_dir): + return [] + dirs = [os.path.join(results_dir, d) for d in os.listdir(results_dir)] + dirs = [d for d in dirs + if os.path.isdir(d) and os.path.getmtime(d) >= since] + return sorted(dirs, key=os.path.getmtime, reverse=True) + + +def read_rows(job_dir, name): + with open(os.path.join(job_dir, name), newline="") as f: + return list(csv.DictReader(f)) + + +def eval_key(row): + return (as_dialect(row.get("dialects")), + as_prompt_id(row.get("prompt_id") or row.get("id"))) + + +def find_job(results_dir, since, expected): + """Returns the newest job whose evals.csv covers an expected prompt.""" + for job_dir in job_dirs(results_dir, since): + path = os.path.join(job_dir, "evals.csv") + if not os.path.exists(path): + continue + keys = {eval_key(row) for row in read_rows(job_dir, "evals.csv")} + if keys & expected.keys(): + return job_dir + return None + + +class Problems: + """Collects failures and remembers whether any of them is hard.""" + + def __init__(self): + self.items = [] + self.hard = False + + def add(self, message, retryable=False): + self.items.append(message + (" [model]" if retryable else "")) + self.hard = self.hard or not retryable + + +def check_reports(job_dir, problems): + for name in REPORT_FILES: + path = os.path.join(job_dir, name) + if not os.path.exists(path): + problems.add(f"{name} is missing") + elif not read_rows(job_dir, name): + problems.add(f"{name} has no rows") + + +def check_coverage(expected, evals, problems): + counts = {} + for row in evals: + counts[eval_key(row)] = counts.get(eval_key(row), 0) + 1 + for key in sorted(expected.keys() - counts.keys()): + problems.add(f"no eval row for prompt {key[1]} [{key[0]}]") + for key in sorted(counts.keys() - expected.keys()): + problems.add(f"eval row for unexpected prompt {key[1]} [{key[0]}]") + for key, count in sorted(counts.items()): + if count > 1: + problems.add(f"{count} eval rows for prompt {key[1]} [{key[0]}]") + + +def check_errors(evals, problems): + for row in evals: + target = "{1} [{0}]".format(*eval_key(row)) + for column in HARD_ERROR_COLUMNS: + if not is_empty(row.get(column)): + problems.add( + f"{target}: {column}: {row[column].strip()[:200]}") + if not is_empty(row.get("sql_generator_error")): + problems.add(f"{target}: sql_generator_error: " + f"{row['sql_generator_error'].strip()[:200]}", + retryable=True) + elif is_empty(row.get("generated_sql")): + problems.add(f"{target}: the model returned no SQL", + retryable=True) + + +def check_scores(scorers, evals, scores, problems): + by_key = {} + for row in scores: + by_key[(row.get("comparator"), row.get("id"), + as_dialect(row.get("dialects")))] = row + for scorer in scorers: + retryable = scorer in LLM_SCORERS + for eval_row in evals: + dialect, prompt_id = eval_key(eval_row) + target = f"{prompt_id} [{dialect}]" + row = by_key.get((scorer, eval_row.get("id"), dialect)) + if row is None: + problems.add(f"{scorer}: no score row for {target}") + continue + error = row.get("comparison_error") + if not is_empty(error): + problems.add(f"{scorer}: errored on {target}: " + f"{error.strip()[:200]}", retryable=retryable) + elif as_score(row.get("score")) is None: + problems.add(f"{scorer}: non-numeric score for {target}", + retryable=retryable) + + +def check_execution(expected, evals, scores, problems): + """Requires one executed generated query per dialect and query type.""" + executed = {(r.get("id"), as_dialect(r.get("dialects"))) + for r in scores + if r.get("comparator") == "executable_sql" + and (as_score(r.get("score")) or 0) > 0} + groups = {} + for row in evals: + key = eval_key(row) + query_type = expected.get(key) + if query_type is None: + continue + group = groups.setdefault((key[0], query_type), []) + group.append((row.get("id"), key[0]) in executed) + for (dialect, query_type), results in sorted(groups.items()): + if not any(results): + problems.add(f"executable_sql: no generated {query_type} query " + f"executed on {dialect}", retryable=True) + + +def print_accuracy(scorers, scores): + print("Accuracy (reported, not gated):") + for scorer in scorers: + values = [as_score(r.get("score")) for r in scores + if r.get("comparator") == scorer] + values = [v for v in values if v is not None] + if values: + print(f" {scorer:<16} mean {sum(values) / len(values):6.2f} " + f"over {len(values)} rows") + + +def verify(config_path, results_dir, since): + config = parse_config(config_path) + scorers = sorted(config.get("scorers") or {}) + results_dir = results_dir or config["reporting"]["csv"]["output_directory"] + expected = expected_prompts(config) + if not expected: + print(f"[FAIL] {config['dataset_config']} has no prompts for " + f"dialects {config.get('dialects')} and query types " + f"{config.get('query_types')}") + return HARD + + job_dir = find_job(results_dir, since, expected) + if job_dir is None: + print(f"[FAIL] no evals.csv under {results_dir} covers the dataset") + return HARD + print(f"Job: {job_dir}") + + problems = Problems() + check_reports(job_dir, problems) + if not problems.items: + evals = read_rows(job_dir, "evals.csv") + scores = read_rows(job_dir, "scores.csv") + print(f"Eval rows: {len(evals)}, scorers: {', '.join(scorers)}\n") + check_coverage(expected, evals, problems) + check_errors(evals, problems) + check_scores(scorers, evals, scores, problems) + check_execution(expected, evals, scores, problems) + print_accuracy(scorers, scores) + print() + + if not problems.items: + print(f"[PASS] clean run: {len(expected)} prompts, " + f"{len(scorers)} scorers") + return 0 + kind = "hard" if problems.hard else "model-dependent" + print(f"[FAIL] {len(problems.items)} problem(s), {kind}") + for problem in problems.items: + print(f" {problem}") + return HARD if problems.hard else RETRYABLE + + +def main(): + parser = argparse.ArgumentParser( + description=__doc__, formatter_class=argparse.RawTextHelpFormatter) + parser.add_argument("--config", default=DEFAULT_CONFIG, + help="Run config of the smoke eval.") + parser.add_argument("--results-dir", + help="Overrides reporting.csv.output_directory. " + "The gRPC server writes to " + "/tmp_session_files/results.") + parser.add_argument("--since", type=float, default=0.0, + help="Ignore job directories older than this Unix " + "time, so a retry does not grade an old job.") + args = parser.parse_args() + return verify(args.config, args.results_dir, args.since) + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.ci/release/wait_for_grpc.py b/.ci/release/wait_for_grpc.py new file mode 100644 index 00000000..14c94241 --- /dev/null +++ b/.ci/release/wait_for_grpc.py @@ -0,0 +1,83 @@ +#!/usr/bin/env python3 +"""Blocks until the eval server answers a gRPC Ping, or the deadline passes. + +Exit codes: + 0 the server answered Ping + 1 the deadline passed, or ALTS handshakes failed repeatedly +""" +import argparse +import asyncio +import os +import sys +import time + +_HERE = os.path.dirname(os.path.abspath(__file__)) +_REPO = os.path.dirname(os.path.dirname(_HERE)) +# Add both module roots so eval_client and generated proto imports resolve. +for _path in (os.path.join(_REPO, "evalbench"), + os.path.join(_REPO, "evalbench", "evalproto")): + if _path not in sys.path: + sys.path.insert(0, _path) + +from client.eval_client import EvalbenchClient # noqa: E402 + +_ALTS_FAILURE_LIMIT = 5 + + +async def wait(timeout: float, interval: float) -> int: + host = os.getenv("EVALBENCH_HOST", "localhost") + port = os.getenv("PORT", "50051") + insecure = os.getenv("EVALBENCH_INSECURE", "").lower() == "true" + deadline = time.monotonic() + timeout + attempt = 0 + last_error = "no attempt completed" + alts_failures = 0 + + mode = "insecure" if insecure else "ALTS" + print(f"Waiting up to {timeout:.0f}s for Ping on {host}:{port} ({mode})") + while time.monotonic() < deadline: + attempt += 1 + # Rebuild client per attempt to avoid stale TRANSIENT_FAILURE backoff. + client = EvalbenchClient("local") + try: + response = await asyncio.wait_for(client.ping(), timeout=interval) + print(f"Ping answered after {attempt} attempts: " + f"{response.response}") + return 0 + except Exception as e: + last_error = f"{type(e).__name__}: {e}" + if "Alts handshake failed" in last_error: + alts_failures += 1 + if alts_failures >= _ALTS_FAILURE_LIMIT: + print(f"Aborting after {alts_failures} consecutive ALTS " + f"handshake failures.") + print("Pass --insecure, or set EVALBENCH_INSECURE=true.") + return 1 + else: + alts_failures = 0 + finally: + await client.channel.close() + await asyncio.sleep(interval) + + print(f"No Ping within {timeout:.0f}s after {attempt} attempts.") + print(f"Last error: {last_error}") + return 1 + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--timeout", type=float, default=300.0, + help="Seconds to wait before giving up.") + parser.add_argument("--interval", type=float, default=3.0, + help="Seconds between attempts and per-attempt " + "deadline.") + parser.add_argument("--insecure", action="store_true", + help="Use an insecure channel to match --localhost.") + args = parser.parse_args() + if args.insecure: + os.environ["EVALBENCH_INSECURE"] = "true" + return asyncio.run(wait(args.timeout, args.interval)) + + +if __name__ == "__main__": + sys.exit(main())