From 258a235868f96a7aa042d6f43f723457c08a9e6b Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad Date: Thu, 17 Sep 2026 06:55:57 +0000 Subject: [PATCH 01/27] ci: add a release smoke pipeline that runs a real eval against the deployed service Deploys the release image to the cluster, waits for the gRPC server, runs a small NL2SQL evalset end to end, and fails the build if the results regress. Switches the eval-server deployment to the Recreate strategy so the rollout does not deadlock on the ReadWriteOnce PVC. --- .ci/release.cloudbuild.yaml | 132 +++++++++++++++++++++ .ci/release/release_smoke.evalset.json | 63 ++++++++++ .ci/release/release_smoke_config.yaml | 29 +++++ .ci/release/verify_nl2sql_smoke.py | 158 +++++++++++++++++++++++++ .ci/release/wait_for_grpc.py | 84 +++++++++++++ evalbench_service/k8s/evalbench.yaml | 4 + 6 files changed, 470 insertions(+) create mode 100644 .ci/release.cloudbuild.yaml create mode 100644 .ci/release/release_smoke.evalset.json create mode 100644 .ci/release/release_smoke_config.yaml create mode 100644 .ci/release/verify_nl2sql_smoke.py create mode 100644 .ci/release/wait_for_grpc.py diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml new file mode 100644 index 00000000..f452d60a --- /dev/null +++ b/.ci/release.cloudbuild.yaml @@ -0,0 +1,132 @@ +# Weekly release pipeline -- Stages 0 to 5. +# +# Resolves the tested image from the Build trigger, runs the gRPC smoke gate +# inside that image, and tags the immutable release. +# +# Shell variables use $$NAME and command substitutions use $$(...) so Cloud +# Build passes them to bash unexpanded. + +steps: + + # Stage 0: Verify that the Build trigger published the commit image. + - id: resolve-release + name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' + entrypoint: 'bash' + args: + - '-c' + - | + set -uo pipefail + + if [[ -z "$COMMIT_SHA" || -z "$SHORT_SHA" ]]; then + echo "FAIL: COMMIT_SHA or SHORT_SHA is empty." + exit 1 + fi + + IMAGE="${_IMAGE_REPO}:$COMMIT_SHA" + echo "Resolving $$IMAGE" + + DIGEST=$$(gcloud artifacts docker images describe "$$IMAGE" \ + --format='value(image_summary.digest)' 2>/dev/null) + if [[ -z "$$DIGEST" ]]; then + echo "FAIL: no image at $$IMAGE" + exit 1 + fi + + # Use ISO week-year (%G) and ISO week (%V) to avoid year-boundary drift. + TAG="v$$(date -u +%G.%V).$SHORT_SHA" + + echo "$$DIGEST" > /workspace/image_digest.txt + echo "$$TAG" > /workspace/release_tag.txt + echo "digest = $$DIGEST" + echo "tag = $$TAG" + + # Stages 1-4: Start eval_server, run gRPC smoke eval, and verify scores.csv. + - id: smoke-gate + name: '${_IMAGE_REPO}:$COMMIT_SHA' + dir: '/evalbench' + waitFor: ['resolve-release'] + entrypoint: 'bash' + args: + - '-c' + - | + set -uo pipefail + + # Add both module roots so eval_client and proto imports resolve. + export PYTHONPATH=/evalbench/evalbench:/evalbench/evalbench/evalproto + + echo "===== start eval_server =====" + python evalbench/eval_server.py --localhost > /tmp/server.log 2>&1 & + SERVER_PID=$$! + trap 'kill $$SERVER_PID 2>/dev/null || true' EXIT + + dump_log() { + echo "----- eval_server log, last 40 lines -----" + tail -40 /tmp/server.log + } + + echo "===== wait for gRPC =====" + if ! python .ci/release/wait_for_grpc.py --timeout 300 --insecure; then + dump_log + exit 1 + fi + + echo "===== run the smoke eval over gRPC =====" + if ! EVALBENCH_HOST=localhost PORT=50051 EVALBENCH_INSECURE=true \ + python evalbench/client/eval_client.py \ + --endpoint=local \ + --experiment=.ci/release/release_smoke_config.yaml; then + dump_log + exit 1 + fi + + # CsvReporter forces /tmp_session_files/results on the gRPC path. + echo "===== audit scores.csv =====" + if ! python .ci/release/verify_nl2sql_smoke.py \ + --results-dir /tmp_session_files/results; then + dump_log + exit 1 + fi + + echo "===== smoke gate passed =====" + env: + - 'EVAL_GCP_PROJECT_ID=${_EVAL_PROJECT}' + - 'EVAL_GCP_PROJECT_REGION=${_EVAL_REGION}' + - 'GOOGLE_CLOUD_PROJECT=${_EVAL_PROJECT}' + - 'UV_CACHE_DIR=/tmp/uv-cache' + + # Stage 5: Create the immutable release tag and promote test tag. + - id: tag-release + name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' + waitFor: ['smoke-gate'] + entrypoint: 'bash' + args: + - '-c' + - | + set -euo pipefail + DIGEST=$$(cat /workspace/image_digest.txt) + TAG=$$(cat /workspace/release_tag.txt) + echo "Tagging ${_IMAGE_REPO}@$$DIGEST as :$$TAG" + gcloud artifacts docker tags add \ + "${_IMAGE_REPO}@$$DIGEST" \ + "${_IMAGE_REPO}:$$TAG" + echo "Release tag created: ${_IMAGE_REPO}:$$TAG" + + if [[ -n "${_PROMOTE_TAG}" ]]; then + echo "Tagging ${_IMAGE_REPO}@$$DIGEST as :${_PROMOTE_TAG}" + gcloud artifacts docker tags add \ + "${_IMAGE_REPO}@$$DIGEST" \ + "${_IMAGE_REPO}:${_PROMOTE_TAG}" + echo "Promotion tag created: ${_IMAGE_REPO}:${_PROMOTE_TAG}" + fi + +substitutions: + _IMAGE_REPO: 'us-central1-docker.pkg.dev/cloud-db-nl2sql/evalbench/eval_server' + _EVAL_PROJECT: 'cloud-db-nl2sql' + _EVAL_REGION: 'global' + _PROMOTE_TAG: 'latest-test' + +timeout: '1800s' + +options: + logging: CLOUD_LOGGING_ONLY + machineType: 'E2_HIGHCPU_8' diff --git a/.ci/release/release_smoke.evalset.json b/.ci/release/release_smoke.evalset.json new file mode 100644 index 00000000..455cd0a0 --- /dev/null +++ b/.ci/release/release_smoke.evalset.json @@ -0,0 +1,63 @@ +[ + { + "id": 3, + "nl_prompt": "How many bloggers who have posted at least 1 blog in 2024 and have linked twitter account", + "query_type": "dql", + "database": "db_blog", + "dialects": [ + "sqlite" + ], + "golden_sql": { + "sqlite": [ + "SELECT COUNT(DISTINCT u.user_id) AS num_bloggers FROM tbl_users u INNER JOIN tbl_posts p ON u.user_id = p.user_id WHERE p.created_at BETWEEN DATE('2024-01-01') AND DATE('2024-12-31') AND json_extract(u.social_media_links, '$.twitter') IS NOT NULL;" + ] + }, + "eval_query": { + "sqlite": [ + null + ] + }, + "setup_sql": {}, + "cleanup_sql": {}, + "tags": [ + "DQL", + "difficulty: moderate", + "SELECT", + "JOIN", + "JSON", + "JSON_EXTRACT", + "datalinking" + ] + }, + { + "id": 5, + "nl_prompt": "What is the number of comments per post for each user? -- Use user_id to identify users and post_id to identify posts.", + "query_type": "dql", + "database": "db_blog", + "dialects": [ + "sqlite" + ], + "golden_sql": { + "sqlite": [ + "SELECT u.user_id, p.post_id, COUNT(c.comment_id) AS comment_count FROM tbl_users u INNER JOIN tbl_posts p ON u.user_id = p.user_id LEFT JOIN tbl_comments c ON p.post_id = c.post_id GROUP BY u.user_id, p.post_id;" + ] + }, + "eval_query": { + "sqlite": [ + null + ] + }, + "setup_sql": {}, + "cleanup_sql": {}, + "tags": [ + "DQL", + "difficulty: simple", + "SELECT", + "JOIN", + "AGGREGATE", + "CASE", + "LEFT_JOIN", + "IS_NULL" + ] + } +] diff --git a/.ci/release/release_smoke_config.yaml b/.ci/release/release_smoke_config.yaml new file mode 100644 index 00000000..41472115 --- /dev/null +++ b/.ci/release/release_smoke_config.yaml @@ -0,0 +1,29 @@ +# Release smoke gate -- one-shot NL2SQL leg. +# Uses sqlite only to avoid external database dependencies. +# Paths are relative to /evalbench, the container WORKDIR. +dataset_config: .ci/release/release_smoke.evalset.json + +database_configs: + - datasets/bat/db_configs/sqlite.yaml +dialects: + - sqlite +query_types: + - dql + +# Build db_blog from datasets/bat/setup/db_blog/sqlite/*.sql on every run. +setup_directory: datasets/bat/setup + +model_config: datasets/model_configs/gemini_2.5_pro_model.yaml +prompt_generator: 'SQLGenBasePromptGenerator' + +# verify_nl2sql_smoke.py enforces score > 0 for returned_sql and executable_sql. +# exact_match and set_match check liveness only. +scorers: + returned_sql: null + executable_sql: null + exact_match: null + set_match: null + +reporting: + csv: + output_directory: 'results/release/oneshot' diff --git a/.ci/release/verify_nl2sql_smoke.py b/.ci/release/verify_nl2sql_smoke.py new file mode 100644 index 00000000..89667501 --- /dev/null +++ b/.ci/release/verify_nl2sql_smoke.py @@ -0,0 +1,158 @@ +#!/usr/bin/env python3 +"""Gates the weekly release build on the smoke run's scores.csv. + +evalbench exits 0 when a run completes, so exit code alone cannot gate a +release. This script audits scores.csv in two tiers: + - Tier 1 (score > 0): structural scorers (returned_sql, executable_sql). + - Tier 2 (liveness): scorer produced a finite number with no comparison_error. + +When eval runs via eval_server.py, CsvReporter writes to +/tmp_session_files/results. Pass --results-dir to override the config path. +""" +import argparse +import csv +import json +import math +import os +import re +import sys + +from pyaml_env import parse_config + +# CsvReporter names each trial row "_trial_". +_TRIAL_SUFFIX = re.compile(r"^(?P.+)_trial_\d+$") + +LEGS = { + "oneshot": { + "config": ".ci/release/release_smoke_config.yaml", + "positive": {"returned_sql", "executable_sql"}, + }, +} + + +def base_id(row_id): + """Strips the _trial_N suffix from a row id.""" + match = _TRIAL_SUFFIX.match(row_id or "") + return match.group("base") if match else (row_id or "") + + +def expected_ids(dataset_config): + """Returns scenario ids from a flat list or a dict with 'scenarios'.""" + with open(dataset_config) as f: + data = json.load(f) + items = data["scenarios"] if isinstance(data, dict) else data + return {str(item["id"]) for item in items} + + +def job_dirs(results_dir): + """Returns job directories under results_dir, newest first.""" + if not os.path.isdir(results_dir): + return [] + dirs = [os.path.join(results_dir, d) for d in os.listdir(results_dir)] + dirs = [d for d in dirs if os.path.isdir(d)] + return sorted(dirs, key=os.path.getmtime, reverse=True) + + +def load_rows(job_dir): + """Returns {(comparator, base_id): (score_or_None, error)}, or None.""" + path = os.path.join(job_dir, "scores.csv") + if not os.path.exists(path): + return None + rows = {} + with open(path, newline="") as f: + for row in csv.DictReader(f): + try: + score = float(row["score"]) + if not math.isfinite(score): + score = None + except (KeyError, TypeError, ValueError): + score = None + key = (row.get("comparator"), base_id(row.get("id"))) + rows[key] = (score, (row.get("comparison_error") or "").strip()) + return rows + + +def find_job(results_dir, ids): + """Returns (job_dir, rows) for the newest job that contains all ids.""" + for job_dir in job_dirs(results_dir): + rows = load_rows(job_dir) + if rows is None: + continue + if ids <= {row_id for _, row_id in rows}: + return job_dir, rows + return None, None + + +def check(leg_name, leg, results_dir_override): + config = parse_config(leg["config"]) + scorers = sorted(config.get("scorers") or {}) + results_dir = results_dir_override or config["reporting"]["csv"][ + "output_directory"] + ids = expected_ids(config["dataset_config"]) + + job_dir, rows = find_job(results_dir, ids) + if job_dir is None: + return [f"no scores.csv under {results_dir} covering ids " + f"{sorted(ids)}"], 0, scorers, None + + problems = [] + checked = 0 + for scorer in scorers: + for scenario_id in sorted(ids): + entry = rows.get((scorer, scenario_id)) + if entry is None: + problems.append(f"{scorer}: no row for {scenario_id}") + continue + score, error = entry + checked += 1 + if error: + problems.append( + f"{scorer}: errored on {scenario_id} -- {error[:120]}") + elif score is None: + problems.append( + f"{scorer}: non-numeric score for {scenario_id}") + elif scorer in leg["positive"] and score <= 0: + problems.append( + f"{scorer}: {scenario_id} reported 0 -- the released " + f"image could not complete this step") + return problems, checked, scorers, job_dir + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--leg", action="append", choices=sorted(LEGS), + help="Leg to verify. Repeatable. Defaults to all.") + parser.add_argument("--results-dir", + help="Overrides reporting.csv.output_directory. " + "Required when the run went through " + "eval_server.py.") + args = parser.parse_args() + + legs = args.leg or sorted(LEGS) + failed = [] + for leg_name in legs: + leg = LEGS[leg_name] + problems, checked, scorers, job_dir = check( + leg_name, leg, args.results_dir) + print(f"{leg_name} ({leg['config']})") + print(f" must be > 0: {', '.join(sorted(leg['positive']))}") + if job_dir: + print(f" job: {job_dir}") + if problems: + failed.append(leg_name) + print(" [FAIL]") + for problem in problems: + print(f" {problem}") + else: + print(f" [PASS] {checked} checks across {len(scorers)} scorers") + print() + + if failed: + print(f"FAILED: {', '.join(failed)}") + return 1 + print("Release smoke gate passed.") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.ci/release/wait_for_grpc.py b/.ci/release/wait_for_grpc.py new file mode 100644 index 00000000..e91ea6d7 --- /dev/null +++ b/.ci/release/wait_for_grpc.py @@ -0,0 +1,84 @@ +#!/usr/bin/env python3 +"""Blocks until the eval server answers a gRPC Ping, or the deadline passes. + +Exit codes: + 0 the server answered Ping + 1 the deadline passed or a permanent security mismatch occurred +""" +import argparse +import asyncio +import os +import sys +import time + +_HERE = os.path.dirname(os.path.abspath(__file__)) +_REPO = os.path.dirname(os.path.dirname(_HERE)) +# Add both module roots so eval_client and generated proto imports resolve. +for _path in (os.path.join(_REPO, "evalbench"), + os.path.join(_REPO, "evalbench", "evalproto")): + if _path not in sys.path: + sys.path.insert(0, _path) + +from client.eval_client import EvalbenchClient # noqa: E402 + +# Consecutive ALTS handshake failures before aborting. +_ALTS_FAILURE_LIMIT = 5 + + +async def wait(timeout: float, interval: float) -> int: + host = os.getenv("EVALBENCH_HOST", "localhost") + port = os.getenv("PORT", "50051") + insecure = os.getenv("EVALBENCH_INSECURE", "").lower() == "true" + deadline = time.monotonic() + timeout + attempt = 0 + last_error = "no attempt completed" + alts_failures = 0 + + mode = "insecure" if insecure else "ALTS" + print(f"Waiting up to {timeout:.0f}s for Ping on {host}:{port} ({mode})") + while time.monotonic() < deadline: + attempt += 1 + # Rebuild client per attempt to avoid stale TRANSIENT_FAILURE backoff. + client = EvalbenchClient("local") + try: + response = await asyncio.wait_for(client.ping(), timeout=interval) + print(f"Ping answered after {attempt} attempts: " + f"{response.response}") + return 0 + except Exception as e: + last_error = f"{type(e).__name__}: {e}" + if "Alts handshake failed" in last_error: + alts_failures += 1 + if alts_failures >= _ALTS_FAILURE_LIMIT: + print(f"\nAborting after {alts_failures} consecutive ALTS " + f"handshake failures.") + print("Pass --insecure, or set EVALBENCH_INSECURE=true.") + return 1 + else: + alts_failures = 0 + finally: + await client.channel.close() + await asyncio.sleep(interval) + + print(f"No Ping within {timeout:.0f}s after {attempt} attempts.") + print(f"Last error -- {last_error}") + return 1 + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--timeout", type=float, default=300.0, + help="Seconds to wait before giving up.") + parser.add_argument("--interval", type=float, default=3.0, + help="Seconds between attempts and per-attempt " + "deadline.") + parser.add_argument("--insecure", action="store_true", + help="Use an insecure channel to match --localhost.") + args = parser.parse_args() + if args.insecure: + os.environ["EVALBENCH_INSECURE"] = "true" + return asyncio.run(wait(args.timeout, args.interval)) + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/evalbench_service/k8s/evalbench.yaml b/evalbench_service/k8s/evalbench.yaml index 536e9ecc..272aae50 100644 --- a/evalbench_service/k8s/evalbench.yaml +++ b/evalbench_service/k8s/evalbench.yaml @@ -7,6 +7,10 @@ metadata: app: evalbench-eval-server spec: replicas: 1 + # Use Recreate strategy because evalbench-ssd-pvc is ReadWriteOnce. + # RollingUpdate deadlocks when the new pod waits for the volume mount. + strategy: + type: Recreate selector: matchLabels: app: evalbench-eval-server From 6c3a5a0a586404606b48507235e3bb55ef37f616 Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad Date: Wed, 23 Sep 2026 12:04:19 +0000 Subject: [PATCH 02/27] refactor: update NL2SQL verification to include dialect-specific scenario keys --- .ci/release.cloudbuild.yaml | 41 ++++++++++--------- .ci/release/verify_nl2sql_smoke.py | 64 +++++++++++++++++++++--------- 2 files changed, 69 insertions(+), 36 deletions(-) diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index f452d60a..7ff8c158 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -1,4 +1,4 @@ -# Weekly release pipeline -- Stages 0 to 5. +# Weekly release pipeline. # # Resolves the tested image from the Build trigger, runs the gRPC smoke gate # inside that image, and tags the immutable release. @@ -8,7 +8,8 @@ steps: - # Stage 0: Verify that the Build trigger published the commit image. + # Nothing is built here -- fail unless the Build trigger already pushed an + # image for this commit, then derive the release tag from its digest. - id: resolve-release name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' entrypoint: 'bash' @@ -22,7 +23,8 @@ steps: exit 1 fi - IMAGE="${_IMAGE_REPO}:$COMMIT_SHA" + REPO="us-central1-docker.pkg.dev/$PROJECT_ID/evalbench/eval_server" + IMAGE="$$REPO:$COMMIT_SHA" echo "Resolving $$IMAGE" DIGEST=$$(gcloud artifacts docker images describe "$$IMAGE" \ @@ -35,14 +37,17 @@ steps: # Use ISO week-year (%G) and ISO week (%V) to avoid year-boundary drift. TAG="v$$(date -u +%G.%V).$SHORT_SHA" + echo "$$REPO" > /workspace/image_repo.txt echo "$$DIGEST" > /workspace/image_digest.txt echo "$$TAG" > /workspace/release_tag.txt echo "digest = $$DIGEST" echo "tag = $$TAG" - # Stages 1-4: Start eval_server, run gRPC smoke eval, and verify scores.csv. + # Exercise the candidate image itself: serve, evaluate, and audit the scores + # it produced. eval_client exits 0 on a completed run, so scores.csv is the + # only real gate. - id: smoke-gate - name: '${_IMAGE_REPO}:$COMMIT_SHA' + name: 'us-central1-docker.pkg.dev/$PROJECT_ID/evalbench/eval_server:$COMMIT_SHA' dir: '/evalbench' waitFor: ['resolve-release'] entrypoint: 'bash' @@ -89,12 +94,13 @@ steps: echo "===== smoke gate passed =====" env: - - 'EVAL_GCP_PROJECT_ID=${_EVAL_PROJECT}' + - 'EVAL_GCP_PROJECT_ID=$PROJECT_ID' - 'EVAL_GCP_PROJECT_REGION=${_EVAL_REGION}' - - 'GOOGLE_CLOUD_PROJECT=${_EVAL_PROJECT}' + - 'GOOGLE_CLOUD_PROJECT=$PROJECT_ID' - 'UV_CACHE_DIR=/tmp/uv-cache' - # Stage 5: Create the immutable release tag and promote test tag. + # Tag by digest, not by :$COMMIT_SHA, so the release points at the exact + # image the gate just passed. - id: tag-release name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' waitFor: ['smoke-gate'] @@ -103,25 +109,24 @@ steps: - '-c' - | set -euo pipefail + REPO=$$(cat /workspace/image_repo.txt) DIGEST=$$(cat /workspace/image_digest.txt) TAG=$$(cat /workspace/release_tag.txt) - echo "Tagging ${_IMAGE_REPO}@$$DIGEST as :$$TAG" + echo "Tagging $$REPO@$$DIGEST as :$$TAG" gcloud artifacts docker tags add \ - "${_IMAGE_REPO}@$$DIGEST" \ - "${_IMAGE_REPO}:$$TAG" - echo "Release tag created: ${_IMAGE_REPO}:$$TAG" + "$$REPO@$$DIGEST" \ + "$$REPO:$$TAG" + echo "Release tag created: $$REPO:$$TAG" if [[ -n "${_PROMOTE_TAG}" ]]; then - echo "Tagging ${_IMAGE_REPO}@$$DIGEST as :${_PROMOTE_TAG}" + echo "Tagging $$REPO@$$DIGEST as :${_PROMOTE_TAG}" gcloud artifacts docker tags add \ - "${_IMAGE_REPO}@$$DIGEST" \ - "${_IMAGE_REPO}:${_PROMOTE_TAG}" - echo "Promotion tag created: ${_IMAGE_REPO}:${_PROMOTE_TAG}" + "$$REPO@$$DIGEST" \ + "$$REPO:${_PROMOTE_TAG}" + echo "Promotion tag created: $$REPO:${_PROMOTE_TAG}" fi substitutions: - _IMAGE_REPO: 'us-central1-docker.pkg.dev/cloud-db-nl2sql/evalbench/eval_server' - _EVAL_PROJECT: 'cloud-db-nl2sql' _EVAL_REGION: 'global' _PROMOTE_TAG: 'latest-test' diff --git a/.ci/release/verify_nl2sql_smoke.py b/.ci/release/verify_nl2sql_smoke.py index 89667501..6d6de334 100644 --- a/.ci/release/verify_nl2sql_smoke.py +++ b/.ci/release/verify_nl2sql_smoke.py @@ -10,6 +10,7 @@ /tmp_session_files/results. Pass --results-dir to override the config path. """ import argparse +import ast import csv import json import math @@ -36,12 +37,33 @@ def base_id(row_id): return match.group("base") if match else (row_id or "") -def expected_ids(dataset_config): - """Returns scenario ids from a flat list or a dict with 'scenarios'.""" +def row_dialect(raw): + """scores.csv stores the dialects list as its repr, e.g. "['sqlite']".""" + try: + parsed = ast.literal_eval(raw or "") + except (ValueError, SyntaxError): + parsed = raw + if isinstance(parsed, (list, tuple)): + return ",".join(str(dialect) for dialect in parsed) + return str(parsed or "") + + +def expected_keys(dataset_config, config_dialects): + """Returns the {(dialect, scenario id)} pairs the run should score. + + A scenario runs once per dialect, and dataset.py intersects the scenario's + own dialects with config['dialects'] when that list is non-empty. + """ with open(dataset_config) as f: data = json.load(f) items = data["scenarios"] if isinstance(data, dict) else data - return {str(item["id"]) for item in items} + keys = set() + for item in items: + dialects = item.get("dialects") or [] + if config_dialects: + dialects = [d for d in dialects if d in config_dialects] + keys.update((dialect, str(item["id"])) for dialect in dialects) + return keys def job_dirs(results_dir): @@ -54,7 +76,7 @@ def job_dirs(results_dir): def load_rows(job_dir): - """Returns {(comparator, base_id): (score_or_None, error)}, or None.""" + """Returns {(dialect, comparator, base_id): (score_or_None, error)}, or None.""" path = os.path.join(job_dir, "scores.csv") if not os.path.exists(path): return None @@ -67,18 +89,20 @@ def load_rows(job_dir): score = None except (KeyError, TypeError, ValueError): score = None - key = (row.get("comparator"), base_id(row.get("id"))) + key = (row_dialect(row.get("dialects")), + row.get("comparator"), + base_id(row.get("id"))) rows[key] = (score, (row.get("comparison_error") or "").strip()) return rows -def find_job(results_dir, ids): - """Returns (job_dir, rows) for the newest job that contains all ids.""" +def find_job(results_dir, keys): + """Returns (job_dir, rows) for the newest job covering every expected key.""" for job_dir in job_dirs(results_dir): rows = load_rows(job_dir) if rows is None: continue - if ids <= {row_id for _, row_id in rows}: + if keys <= {(dialect, row_id) for dialect, _, row_id in rows}: return job_dir, rows return None, None @@ -88,32 +112,36 @@ def check(leg_name, leg, results_dir_override): scorers = sorted(config.get("scorers") or {}) results_dir = results_dir_override or config["reporting"]["csv"][ "output_directory"] - ids = expected_ids(config["dataset_config"]) + keys = expected_keys(config["dataset_config"], config.get("dialects") or []) + if not keys: + return [f"{config['dataset_config']} has no scenarios matching " + f"dialects {config.get('dialects')}"], 0, scorers, None - job_dir, rows = find_job(results_dir, ids) + job_dir, rows = find_job(results_dir, keys) if job_dir is None: - return [f"no scores.csv under {results_dir} covering ids " - f"{sorted(ids)}"], 0, scorers, None + return [f"no scores.csv under {results_dir} covering " + f"{sorted(keys)}"], 0, scorers, None problems = [] checked = 0 for scorer in scorers: - for scenario_id in sorted(ids): - entry = rows.get((scorer, scenario_id)) + for dialect, scenario_id in sorted(keys): + target = f"{scenario_id} [{dialect}]" + entry = rows.get((dialect, scorer, scenario_id)) if entry is None: - problems.append(f"{scorer}: no row for {scenario_id}") + problems.append(f"{scorer}: no row for {target}") continue score, error = entry checked += 1 if error: problems.append( - f"{scorer}: errored on {scenario_id} -- {error[:120]}") + f"{scorer}: errored on {target} -- {error[:120]}") elif score is None: problems.append( - f"{scorer}: non-numeric score for {scenario_id}") + f"{scorer}: non-numeric score for {target}") elif scorer in leg["positive"] and score <= 0: problems.append( - f"{scorer}: {scenario_id} reported 0 -- the released " + f"{scorer}: {target} reported 0 -- the released " f"image could not complete this step") return problems, checked, scorers, job_dir From 119001bdfcecd0c1075ca08a410c431687b934ec Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad Date: Thu, 24 Sep 2026 05:43:16 +0000 Subject: [PATCH 03/27] refactor: remove Recreate deployment strategy in evalbench service yaml --- evalbench_service/k8s/evalbench.yaml | 4 ---- 1 file changed, 4 deletions(-) diff --git a/evalbench_service/k8s/evalbench.yaml b/evalbench_service/k8s/evalbench.yaml index 272aae50..536e9ecc 100644 --- a/evalbench_service/k8s/evalbench.yaml +++ b/evalbench_service/k8s/evalbench.yaml @@ -7,10 +7,6 @@ metadata: app: evalbench-eval-server spec: replicas: 1 - # Use Recreate strategy because evalbench-ssd-pvc is ReadWriteOnce. - # RollingUpdate deadlocks when the new pod waits for the volume mount. - strategy: - type: Recreate selector: matchLabels: app: evalbench-eval-server From 71ae37ecfaea9d9282fcdd153d596d20b15a0e02 Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad Date: Thu, 24 Sep 2026 08:32:31 +0000 Subject: [PATCH 04/27] ci(release): capture the running digest before a release can deploy Drops the movable promotion tag from tag-release so an aborted run cannot touch any deployment reference, and adds a step that records the digest the cluster is currently serving. The Deployment spec only carries the mutable :latest tag, so the kubelet's imageID is the only usable rollback target. The step refuses to proceed when nothing is running or when pods report more than one digest, since neither state yields a target worth restoring. --- .ci/release.cloudbuild.yaml | 50 +++++++++++++++++++++++++++++++------ 1 file changed, 43 insertions(+), 7 deletions(-) diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index 7ff8c158..0706d32d 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -118,17 +118,53 @@ steps: "$$REPO:$$TAG" echo "Release tag created: $$REPO:$$TAG" - if [[ -n "${_PROMOTE_TAG}" ]]; then - echo "Tagging $$REPO@$$DIGEST as :${_PROMOTE_TAG}" - gcloud artifacts docker tags add \ - "$$REPO@$$DIGEST" \ - "$$REPO:${_PROMOTE_TAG}" - echo "Promotion tag created: $$REPO:${_PROMOTE_TAG}" + # The Deployment spec only carries the mutable :latest tag, so the kubelet's + # imageID is the only record of which digest is actually serving. + - id: capture-rollback-target + name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' + waitFor: ['tag-release'] + entrypoint: 'bash' + args: + - '-c' + - | + set -uo pipefail + + CLUSTER=evalbench-directpath-cluster + ZONE=us-central1-c + NAMESPACE=evalbench-namespace + SELECTOR=app=evalbench-eval-server + CONTAINER=evalbench-eval + + if ! gcloud container clusters get-credentials "$$CLUSTER" \ + --zone "$$ZONE" --project "$PROJECT_ID"; then + echo "FAIL: cannot get credentials for $$CLUSTER" + exit 1 fi + RUNNING=$$(kubectl get pods -n "$$NAMESPACE" -l "$$SELECTOR" \ + -o jsonpath="{range .items[*]}{.status.containerStatuses[?(@.name=='$$CONTAINER')].imageID}{'\n'}{end}" \ + | sed -e 's#^.*://##' -e 's#^.*@##' \ + | grep -v '^$$' | sort -u) + COUNT=$$(echo "$$RUNNING" | grep -c .) + + if [[ "$$COUNT" -eq 0 ]]; then + echo "FAIL: no running $$CONTAINER container in $$NAMESPACE." + echo "Refusing to deploy without a rollback target." + exit 1 + fi + + if [[ "$$COUNT" -gt 1 ]]; then + echo "FAIL: pods are serving more than one digest:" + echo "$$RUNNING" + echo "A rollout is already in progress. Refusing to deploy." + exit 1 + fi + + echo "$$RUNNING" > /workspace/rollback_digest.txt + echo "rollback target = $$RUNNING" + substitutions: _EVAL_REGION: 'global' - _PROMOTE_TAG: 'latest-test' timeout: '1800s' From d1446a62592473d59c8bedf0de3c523fe82af2e8 Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad Date: Thu, 24 Sep 2026 08:38:41 +0000 Subject: [PATCH 05/27] ci(release): deploy the gated digest with automatic rollback Promotes :latest to the smoke-gated digest, restarts the Deployment, and confirms the pods answer a gRPC Ping before the build passes. Any failure repoints :latest at the previously captured digest and fails the build. Deploy, verify, and rollback share one step because Cloud Build has no on-failure hook -- a separate rollback step would never run. --- .ci/release.cloudbuild.yaml | 95 ++++++++++++++++++++++++++++++++++++- 1 file changed, 94 insertions(+), 1 deletion(-) diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index 0706d32d..d4788464 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -163,10 +163,103 @@ steps: echo "$$RUNNING" > /workspace/rollback_digest.txt echo "rollback target = $$RUNNING" + # Deploy, verify, and roll back live in one step: Cloud Build has no + # on-failure hook, so a separate rollback step would never run. + - id: deploy + name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' + waitFor: ['capture-rollback-target'] + entrypoint: 'bash' + args: + - '-c' + - | + set -uo pipefail + + NAMESPACE=evalbench-namespace + SELECTOR=app=evalbench-eval-server + CONTAINER=evalbench-eval + DEPLOYMENT=deployment/evalbench-eval-server-deploy + ROLLOUT_TIMEOUT=600s + + REPO=$$(cat /workspace/image_repo.txt) + NEW_DIGEST=$$(cat /workspace/image_digest.txt) + OLD_DIGEST=$$(cat /workspace/rollback_digest.txt) + + if ! gcloud container clusters get-credentials \ + evalbench-directpath-cluster \ + --zone us-central1-c --project "$PROJECT_ID"; then + echo "FAIL: cannot get cluster credentials." + exit 1 + fi + + running_digests() { + kubectl get pods -n "$$NAMESPACE" -l "$$SELECTOR" \ + -o jsonpath="{range .items[*]}{.status.containerStatuses[?(@.name=='$$CONTAINER')].imageID}{'\n'}{end}" \ + | sed -e 's#^.*://##' -e 's#^.*@##' \ + | grep -v '^$$' | sort -u + } + + # The Deployment tracks :latest, so moving the tag is what selects the + # image; the restart is what makes the kubelet pull it. + roll_to() { + echo "Pointing :latest at $$1" + gcloud artifacts docker tags add "$$REPO@$$1" "$$REPO:latest" || return 1 + kubectl rollout restart "$$DEPLOYMENT" -n "$$NAMESPACE" || return 1 + kubectl rollout status "$$DEPLOYMENT" -n "$$NAMESPACE" \ + --timeout="$$ROLLOUT_TIMEOUT" || return 1 + } + + rollback() { + echo "===== rolling back to $$OLD_DIGEST =====" + if ! roll_to "$$OLD_DIGEST"; then + echo "FAIL: rollback did not complete. Not retrying." + echo "Last known-good digest: $$REPO@$$OLD_DIGEST" + kubectl get pods -n "$$NAMESPACE" -l "$$SELECTOR" -o wide + exit 1 + fi + echo "Rolled back to $$OLD_DIGEST. Failing the build." + exit 1 + } + + echo "===== deploy $$NEW_DIGEST =====" + if ! roll_to "$$NEW_DIGEST"; then + echo "FAIL: rollout did not complete within $$ROLLOUT_TIMEOUT." + kubectl get pods -n "$$NAMESPACE" -l "$$SELECTOR" -o wide + rollback + fi + + # Pods that lost the rollout can linger in Terminating for a few + # seconds after rollout status returns, so retry before believing it. + echo "===== confirm the pods are serving $$NEW_DIGEST =====" + for _ in $$(seq 20); do + SERVING=$$(running_digests) + [[ "$$SERVING" == "$$NEW_DIGEST" ]] && break + sleep 3 + done + if [[ "$$SERVING" != "$$NEW_DIGEST" ]]; then + echo "FAIL: expected $$NEW_DIGEST, pods report:" + echo "$$SERVING" + rollback + fi + + # There is no readiness probe on this Deployment, so "Ready" only means + # the process started. A gRPC Ping is the first real health signal. + echo "===== gRPC health check =====" + POD=$$(kubectl get pods -n "$$NAMESPACE" -l "$$SELECTOR" \ + -o jsonpath='{.items[0].metadata.name}') + if ! kubectl exec -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" -- \ + python /evalbench/.ci/release/wait_for_grpc.py --timeout 300; then + echo "FAIL: $$POD never answered Ping." + kubectl logs -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" --tail=40 + rollback + fi + + echo "===== release deployed =====" + echo "serving $$REPO@$$NEW_DIGEST as $$(cat /workspace/release_tag.txt)" + substitutions: _EVAL_REGION: 'global' -timeout: '1800s' +timeout: '3600s' options: logging: CLOUD_LOGGING_ONLY From 86132f3c1baef4a36bd2d07ec970e32ba8d60c71 Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad Date: Thu, 24 Sep 2026 08:48:55 +0000 Subject: [PATCH 06/27] ci(release): deploy the Mesop UI to Corp Run after the GKE gate Copies the digest GKE accepted into the evalbench-dev registry and deploys it to the Cloud Run service, so both surfaces serve identical bits. The copy is registry-side: both repos share a host, so gcrane mounts existing blobs. Deploying by digest doubles as proof the copy landed the right image. Cloud Run keeps a revision with no traffic until it is ready, so a failed deploy already leaves the previous revision serving; traffic is pinned back to it explicitly in case the deploy fails partway. Renames deploy to deploy-gke now that there are two deploy targets. --- .ci/release.cloudbuild.yaml | 66 +++++++++++++++++++++++++++++++++++-- 1 file changed, 64 insertions(+), 2 deletions(-) diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index d4788464..9ef9fd93 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -165,7 +165,7 @@ steps: # Deploy, verify, and roll back live in one step: Cloud Build has no # on-failure hook, so a separate rollback step would never run. - - id: deploy + - id: deploy-gke name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' waitFor: ['capture-rollback-target'] entrypoint: 'bash' @@ -249,13 +249,75 @@ steps: if ! kubectl exec -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" -- \ python /evalbench/.ci/release/wait_for_grpc.py --timeout 300; then echo "FAIL: $$POD never answered Ping." - kubectl logs -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" --tail=40 + kubectl logs -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" --tail=150 rollback fi echo "===== release deployed =====" echo "serving $$REPO@$$NEW_DIGEST as $$(cat /workspace/release_tag.txt)" + # The Mesop UI runs the same image on Cloud Run in evalbench-dev. Both repos + # live on us-central1-docker.pkg.dev, so gcrane mounts the existing blobs + # instead of transferring the image. :latest keeps the manual + # 'make deploy-corprun' path pointing at the same bits. + - id: copy-corprun-image + name: 'gcr.io/go-containerregistry/gcrane' + waitFor: ['deploy-gke'] + args: + - 'copy' + - 'us-central1-docker.pkg.dev/$PROJECT_ID/evalbench/eval_server:$COMMIT_SHA' + - 'us-central1-docker.pkg.dev/evalbench-dev/cr-images/eval_server:latest' + + - id: deploy-corprun + name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' + waitFor: ['copy-corprun-image'] + entrypoint: 'bash' + args: + - '-c' + - | + set -uo pipefail + + SERVICE=evalbench + CR_PROJECT=evalbench-dev + CR_REGION=us-central1 + CR_REPO=us-central1-docker.pkg.dev/evalbench-dev/cr-images/eval_server + + DIGEST=$$(cat /workspace/image_digest.txt) + + # Cloud Run pins each revision to a digest, so the previous revision is + # a complete rollback target on its own. + OLD_REVISION=$$(gcloud run services describe "$$SERVICE" \ + --project="$$CR_PROJECT" --region="$$CR_REGION" \ + --format='value(status.latestReadyRevisionName)') + if [[ -z "$$OLD_REVISION" ]]; then + echo "FAIL: $$SERVICE has no ready revision to roll back to." + exit 1 + fi + echo "rollback revision = $$OLD_REVISION" + + # Deploying by digest also proves the copy landed the exact image the + # GKE gate accepted; a mismatched copy cannot resolve and fails here. + # A revision that never becomes ready serves no traffic, so a failed + # deploy leaves OLD_REVISION in place by itself. + echo "===== deploy to Cloud Run =====" + if ! gcloud run deploy "$$SERVICE" \ + --project="$$CR_PROJECT" \ + --region="$$CR_REGION" \ + --image="$$CR_REPO@$$DIGEST"; then + echo "FAIL: Cloud Run deploy did not succeed." + echo "===== pinning traffic back to $$OLD_REVISION =====" + gcloud run services update-traffic "$$SERVICE" \ + --project="$$CR_PROJECT" --region="$$CR_REGION" \ + --to-revisions="$$OLD_REVISION=100" + exit 1 + fi + + NEW_REVISION=$$(gcloud run services describe "$$SERVICE" \ + --project="$$CR_PROJECT" --region="$$CR_REGION" \ + --format='value(status.latestReadyRevisionName)') + echo "===== Corp Run deployed =====" + echo "$$NEW_REVISION serving $$CR_REPO@$$DIGEST" + substitutions: _EVAL_REGION: 'global' From 635713b8e553176b6de0329f1f22aae3b9bc775c Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad Date: Thu, 24 Sep 2026 10:48:49 +0000 Subject: [PATCH 07/27] ci(release): correct and tighten the pipeline comments The file header still described a pipeline that stopped at tagging, which has not been true since the deploy steps landed. Trims the remaining comments to the reason behind each choice and drops the parts that restated the code. --- .ci/release.cloudbuild.yaml | 31 ++++++++++++++----------------- 1 file changed, 14 insertions(+), 17 deletions(-) diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index 9ef9fd93..933c0cbf 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -1,15 +1,15 @@ # Weekly release pipeline. # -# Resolves the tested image from the Build trigger, runs the gRPC smoke gate -# inside that image, and tags the immutable release. +# Gates the image the Build trigger already pushed for this commit, tags it as +# an immutable release, then rolls it out to GKE and Corp Run. # # Shell variables use $$NAME and command substitutions use $$(...) so Cloud # Build passes them to bash unexpanded. steps: - # Nothing is built here -- fail unless the Build trigger already pushed an - # image for this commit, then derive the release tag from its digest. + # Nothing is built here: the Build trigger must have already pushed an image + # for this commit. - id: resolve-release name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' entrypoint: 'bash' @@ -43,9 +43,8 @@ steps: echo "digest = $$DIGEST" echo "tag = $$TAG" - # Exercise the candidate image itself: serve, evaluate, and audit the scores - # it produced. eval_client exits 0 on a completed run, so scores.csv is the - # only real gate. + # eval_client exits 0 on any completed run, so scores.csv is the only real + # gate on whether the candidate image evaluates correctly. - id: smoke-gate name: 'us-central1-docker.pkg.dev/$PROJECT_ID/evalbench/eval_server:$COMMIT_SHA' dir: '/evalbench' @@ -198,8 +197,8 @@ steps: | grep -v '^$$' | sort -u } - # The Deployment tracks :latest, so moving the tag is what selects the - # image; the restart is what makes the kubelet pull it. + # The Deployment tracks :latest, so the tag selects the image and the + # restart is what makes the kubelet pull it. roll_to() { echo "Pointing :latest at $$1" gcloud artifacts docker tags add "$$REPO@$$1" "$$REPO:latest" || return 1 @@ -256,10 +255,8 @@ steps: echo "===== release deployed =====" echo "serving $$REPO@$$NEW_DIGEST as $$(cat /workspace/release_tag.txt)" - # The Mesop UI runs the same image on Cloud Run in evalbench-dev. Both repos - # live on us-central1-docker.pkg.dev, so gcrane mounts the existing blobs - # instead of transferring the image. :latest keeps the manual - # 'make deploy-corprun' path pointing at the same bits. + # The Mesop UI serves this same image from Cloud Run in evalbench-dev. + # Writing :latest keeps the manual 'make deploy-corprun' path in sync. - id: copy-corprun-image name: 'gcr.io/go-containerregistry/gcrane' waitFor: ['deploy-gke'] @@ -295,16 +292,16 @@ steps: fi echo "rollback revision = $$OLD_REVISION" - # Deploying by digest also proves the copy landed the exact image the - # GKE gate accepted; a mismatched copy cannot resolve and fails here. - # A revision that never becomes ready serves no traffic, so a failed - # deploy leaves OLD_REVISION in place by itself. + # Deploying by digest proves the copy landed the image the GKE gate + # accepted; a mismatched copy cannot resolve and fails here. echo "===== deploy to Cloud Run =====" if ! gcloud run deploy "$$SERVICE" \ --project="$$CR_PROJECT" \ --region="$$CR_REGION" \ --image="$$CR_REPO@$$DIGEST"; then echo "FAIL: Cloud Run deploy did not succeed." + # A revision that never goes ready takes no traffic, so this only + # matters if the deploy failed after the traffic split moved. echo "===== pinning traffic back to $$OLD_REVISION =====" gcloud run services update-traffic "$$SERVICE" \ --project="$$CR_PROJECT" --region="$$CR_REGION" \ From 6f05b65dbc7d24c7f270be774a41f4ded5a7c62f Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad Date: Thu, 24 Sep 2026 10:51:18 +0000 Subject: [PATCH 08/27] ci(release): tighten roll_to and the convergence loop roll_to chains its three commands instead of guarding each with an explicit return, and the retry loop uses brace expansion rather than spawning seq. --- .ci/release.cloudbuild.yaml | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index 933c0cbf..5ab7e341 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -201,10 +201,10 @@ steps: # restart is what makes the kubelet pull it. roll_to() { echo "Pointing :latest at $$1" - gcloud artifacts docker tags add "$$REPO@$$1" "$$REPO:latest" || return 1 - kubectl rollout restart "$$DEPLOYMENT" -n "$$NAMESPACE" || return 1 - kubectl rollout status "$$DEPLOYMENT" -n "$$NAMESPACE" \ - --timeout="$$ROLLOUT_TIMEOUT" || return 1 + gcloud artifacts docker tags add "$$REPO@$$1" "$$REPO:latest" \ + && kubectl rollout restart "$$DEPLOYMENT" -n "$$NAMESPACE" \ + && kubectl rollout status "$$DEPLOYMENT" -n "$$NAMESPACE" \ + --timeout="$$ROLLOUT_TIMEOUT" } rollback() { @@ -229,7 +229,7 @@ steps: # Pods that lost the rollout can linger in Terminating for a few # seconds after rollout status returns, so retry before believing it. echo "===== confirm the pods are serving $$NEW_DIGEST =====" - for _ in $$(seq 20); do + for _ in {1..20}; do SERVING=$$(running_digests) [[ "$$SERVING" == "$$NEW_DIGEST" ]] && break sleep 3 From 76fa1d1943094f8066cd5125b5d7a766a37d3e2f Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad Date: Thu, 24 Sep 2026 10:53:45 +0000 Subject: [PATCH 09/27] ci(release): report a failed Cloud Run traffic pin The rollback ignored update-traffic's exit code, so a release that failed to restore the previous revision reported the same output as one that restored it cleanly. Names the last known-good revision when the pin fails, matching how the GKE path reports an unrecoverable rollback. Adds --quiet so a prompt can never stall the build to its timeout. --- .ci/release.cloudbuild.yaml | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index 5ab7e341..c6899699 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -295,7 +295,7 @@ steps: # Deploying by digest proves the copy landed the image the GKE gate # accepted; a mismatched copy cannot resolve and fails here. echo "===== deploy to Cloud Run =====" - if ! gcloud run deploy "$$SERVICE" \ + if ! gcloud run deploy "$$SERVICE" --quiet \ --project="$$CR_PROJECT" \ --region="$$CR_REGION" \ --image="$$CR_REPO@$$DIGEST"; then @@ -303,9 +303,13 @@ steps: # A revision that never goes ready takes no traffic, so this only # matters if the deploy failed after the traffic split moved. echo "===== pinning traffic back to $$OLD_REVISION =====" - gcloud run services update-traffic "$$SERVICE" \ + if ! gcloud run services update-traffic "$$SERVICE" --quiet \ --project="$$CR_PROJECT" --region="$$CR_REGION" \ - --to-revisions="$$OLD_REVISION=100" + --to-revisions="$$OLD_REVISION=100"; then + echo "FAIL: could not pin traffic back. $$SERVICE may be serving" + echo "an unverified revision." + echo "Last known-good revision: $$OLD_REVISION" + fi exit 1 fi From 677bcaf120412f12e5f7678ab124eb0852c8cecd Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad Date: Fri, 25 Sep 2026 06:21:15 +0000 Subject: [PATCH 10/27] ci(release): move tags in place instead of recreating them gcloud artifacts docker tags add deletes and recreates a tag that already exists, which needs artifactregistry.tags.delete. The build account holds artifactregistry.writer, which stops at tags.update. :latest always exists, so roll_to could never have succeeded. --- .ci/release.cloudbuild.yaml | 15 ++++++++++----- 1 file changed, 10 insertions(+), 5 deletions(-) diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index c6899699..7ee668b7 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -112,10 +112,14 @@ steps: DIGEST=$$(cat /workspace/image_digest.txt) TAG=$$(cat /workspace/release_tag.txt) echo "Tagging $$REPO@$$DIGEST as :$$TAG" - gcloud artifacts docker tags add \ - "$$REPO@$$DIGEST" \ - "$$REPO:$$TAG" - echo "Release tag created: $$REPO:$$TAG" + # 'docker tags add' deletes and recreates an existing tag, which needs + # tags.delete. 'tags update' repoints it in place, but cannot create + # it, so fall back to add for a tag this commit has not had before. + gcloud artifacts tags update "$$TAG" --location=us-central1 \ + --repository=evalbench --package=eval_server \ + --version="$$DIGEST" 2>/dev/null \ + || gcloud artifacts docker tags add "$$REPO@$$DIGEST" "$$REPO:$$TAG" + echo "Release tag: $$REPO:$$TAG" # The Deployment spec only carries the mutable :latest tag, so the kubelet's # imageID is the only record of which digest is actually serving. @@ -201,7 +205,8 @@ steps: # restart is what makes the kubelet pull it. roll_to() { echo "Pointing :latest at $$1" - gcloud artifacts docker tags add "$$REPO@$$1" "$$REPO:latest" \ + gcloud artifacts tags update latest --location=us-central1 \ + --repository=evalbench --package=eval_server --version="$$1" \ && kubectl rollout restart "$$DEPLOYMENT" -n "$$NAMESPACE" \ && kubectl rollout status "$$DEPLOYMENT" -n "$$NAMESPACE" \ --timeout="$$ROLLOUT_TIMEOUT" From 6e513d95cba0f10b7d31bcc0374ceee68970bb16 Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad Date: Fri, 25 Sep 2026 07:01:58 +0000 Subject: [PATCH 11/27] chore: update loop variable name and add progress logging to release verification script --- .ci/release.cloudbuild.yaml | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index 7ee668b7..634743b0 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -234,7 +234,7 @@ steps: # Pods that lost the rollout can linger in Terminating for a few # seconds after rollout status returns, so retry before believing it. echo "===== confirm the pods are serving $$NEW_DIGEST =====" - for _ in {1..20}; do + for ATTEMPT in {1..20}; do SERVING=$$(running_digests) [[ "$$SERVING" == "$$NEW_DIGEST" ]] && break sleep 3 @@ -244,6 +244,7 @@ steps: echo "$$SERVING" rollback fi + echo "Every pod reports the release digest (attempt $$ATTEMPT)." # There is no readiness probe on this Deployment, so "Ready" only means # the process started. A gRPC Ping is the first real health signal. From 1da7338ab6129cd3549a87e6744779e9f9d239c2 Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad Date: Fri, 25 Sep 2026 07:21:49 +0000 Subject: [PATCH 12/27] chore: update image release tag format to use YYYY.MM.DD date convention --- .ci/release.cloudbuild.yaml | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index 634743b0..09a474e6 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -34,8 +34,7 @@ steps: exit 1 fi - # Use ISO week-year (%G) and ISO week (%V) to avoid year-boundary drift. - TAG="v$$(date -u +%G.%V).$SHORT_SHA" + TAG="v$$(date -u +%Y.%m.%d).$SHORT_SHA" echo "$$REPO" > /workspace/image_repo.txt echo "$$DIGEST" > /workspace/image_digest.txt From 60d9fca09379bd072b9b55b65126a53ace83335b Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad <74178334+omkargaikwad23@users.noreply.github.com> Date: Tue, 29 Sep 2026 10:40:18 +0530 Subject: [PATCH 13/27] ci(nl2sql): gate pull requests on a QueryData NL2SQL check (#611) * ci(nl2sql): add a QueryData NL2SQL smoke eval and its verifier Add the configs and the verifier for an NL2SQL presubmit check that runs QueryData (query_data_api generator) against db_blog on the Cloud SQL Postgres CI instance, which the skills harness already uses. - nl2sql_run_config.yaml: 8 DQL prompts, NOOPGenerator, scorers returned_sql, executable_sql, set_match and llmrater. - db_configs/nl2sql_ci_postgres.yaml and model_configs/querydata_ci.yaml: all connection values come from env and the password from a secret. - nl2sql_seed_config.yaml: seeds db_blog once with setup_databases.py. CI does not re-seed per run, because a re-seed drops the public schema and concurrent builds would break each other. - verify_nl2sql.py: requires a clean run. All report files, one row per prompt, no generator or golden SQL error, SQL on every row, a numeric score from every scorer, at least one executed query, and pipeline_debug_info in the report. Accuracy is printed but not gated. cloudbuild.yaml and verifier/verify.py are unchanged. A separate Cloud Build config will run this check without an Artifact Registry push. * ci(nl2sql): run the NL2SQL smoke eval as the verify-nl2sql check Add .ci/nl2sql.cloudbuild.yaml for a verify-nl2sql Cloud Build trigger in evalbench-testing. Steps: preflight, pull-cache, build-image, nl2sql-eval and verify. - The image stays local to the build. Nothing goes to Artifact Registry, so PR builds need no registry write access. - build-image reads the evalbench-harness-ci layer cache but never pushes it. - preflight fails in seconds if a trigger substitution or the database secret is empty. - cloudbuild.yaml and the test/build triggers in cloud-db-nl2sql are unchanged. * fix(query_data_api): make REST token refresh thread-safe Runner threads share one QueryDataAPIGenerator. When several threads made the first REST call at the same time, the ones that did not finish refresh() sent "Authorization: Bearer None" and QueryData returned 401. A lock now serializes credential creation and refresh. The token is read while the lock is held. _get_credentials() is renamed to _get_access_token() because it now returns the token. The new test sends 8 concurrent first calls. All 8 must send the refreshed token, and refresh() must run once. --- .ci/db_configs/nl2sql_ci_postgres.yaml | 9 + .ci/model_configs/querydata_ci.yaml | 18 ++ .ci/nl2sql.cloudbuild.yaml | 84 +++++++++ .ci/nl2sql_run_config.yaml | 29 +++ .ci/nl2sql_seed_config.yaml | 15 ++ .ci/nl2sql_smoke.evalset.json | 233 +++++++++++++++++++++++++ .ci/verify_nl2sql.py | 212 ++++++++++++++++++++++ 7 files changed, 600 insertions(+) create mode 100644 .ci/db_configs/nl2sql_ci_postgres.yaml create mode 100644 .ci/model_configs/querydata_ci.yaml create mode 100644 .ci/nl2sql.cloudbuild.yaml create mode 100644 .ci/nl2sql_run_config.yaml create mode 100644 .ci/nl2sql_seed_config.yaml create mode 100644 .ci/nl2sql_smoke.evalset.json create mode 100644 .ci/verify_nl2sql.py diff --git a/.ci/db_configs/nl2sql_ci_postgres.yaml b/.ci/db_configs/nl2sql_ci_postgres.yaml new file mode 100644 index 00000000..55442fe3 --- /dev/null +++ b/.ci/db_configs/nl2sql_ci_postgres.yaml @@ -0,0 +1,9 @@ +# Cloud SQL Postgres CI instance, shared with the verify-harnesses check. +# The dataset's "database" field (db_blog) replaces database_name. +db_type: postgres +dialect: postgres +database_name: db_blog +database_path: !ENV ${EVAL_GCP_PROJECT_ID}:${CLOUD_SQL_POSTGRES_REGION}:${CLOUD_SQL_POSTGRES_INSTANCE} +max_executions_per_minute: 180 +user_name: !ENV ${CLOUD_SQL_POSTGRES_USER} +password: !ENV ${CLOUD_SQL_POSTGRES_PASSWORD} diff --git a/.ci/model_configs/querydata_ci.yaml b/.ci/model_configs/querydata_ci.yaml new file mode 100644 index 00000000..6d67077a --- /dev/null +++ b/.ci/model_configs/querydata_ci.yaml @@ -0,0 +1,18 @@ +# QueryData (Gemini Data Analytics API) generator. +generator: query_data_api +project_id: !ENV ${EVAL_GCP_PROJECT_ID} +location: global +# Returns pipeline_debug_info, which verify_nl2sql.py requires. +use_rest_api: true +execs_per_minute: 30 + +# Must name the same database as .ci/db_configs/nl2sql_ci_postgres.yaml. +context: + datasource_references: + cloud_sql_reference: + database_reference: + engine: POSTGRESQL + project_id: !ENV ${EVAL_GCP_PROJECT_ID} + region: !ENV ${CLOUD_SQL_POSTGRES_REGION} + instance_id: !ENV ${CLOUD_SQL_POSTGRES_INSTANCE} + database_id: db_blog diff --git a/.ci/nl2sql.cloudbuild.yaml b/.ci/nl2sql.cloudbuild.yaml new file mode 100644 index 00000000..4477493e --- /dev/null +++ b/.ci/nl2sql.cloudbuild.yaml @@ -0,0 +1,84 @@ +# verify-nl2sql: NL2SQL presubmit check (trigger in evalbench-testing). +# Builds the image locally (no Artifact Registry push), runs the QueryData +# smoke eval (.ci/nl2sql_run_config.yaml) and grades it with +# .ci/verify_nl2sql.py. +# +# Trigger substitutions: _EVAL_PROJECT, _CLOUD_SQL_POSTGRES_REGION, +# _CLOUD_SQL_POSTGRES_INSTANCE, _CLOUD_SQL_POSTGRES_USER. +# Prerequisite: db_blog is seeded (.ci/nl2sql_seed_config.yaml). +steps: + # Fails fast if a substitution or the secret is empty. + - id: preflight + name: 'bash' + entrypoint: 'bash' + waitFor: ['-'] + args: + - '-c' + - | + rc=0 + for var in EVAL_GCP_PROJECT_ID CLOUD_SQL_POSTGRES_REGION \ + CLOUD_SQL_POSTGRES_INSTANCE CLOUD_SQL_POSTGRES_USER \ + CLOUD_SQL_POSTGRES_PASSWORD; do + if [ -z "$${!var}" ]; then + echo "[FAIL] $$var is empty. Set the trigger substitution or secret." >&2 + rc=1 + fi + done + exit $$rc + env: &eval_env + - 'EVAL_GCP_PROJECT_ID=${_EVAL_PROJECT}' + - 'EVAL_GCP_PROJECT_REGION=global' + - 'UV_CACHE_DIR=/tmp/uv-cache' + - 'CLOUD_SQL_POSTGRES_REGION=${_CLOUD_SQL_POSTGRES_REGION}' + - 'CLOUD_SQL_POSTGRES_INSTANCE=${_CLOUD_SQL_POSTGRES_INSTANCE}' + - 'CLOUD_SQL_POSTGRES_USER=${_CLOUD_SQL_POSTGRES_USER}' + secretEnv: &eval_secrets ['CLOUD_SQL_POSTGRES_PASSWORD'] + + # Read-only reuse of the verify-harnesses layer cache. + - id: pull-cache + name: 'gcr.io/cloud-builders/docker' + entrypoint: 'bash' + waitFor: ['-'] + args: ['-c', 'docker pull us-central1-docker.pkg.dev/$PROJECT_ID/evalbench/evalbench-harness-ci:cache || exit 0'] + + - id: build-image + name: 'gcr.io/cloud-builders/docker' + waitFor: ['pull-cache'] + args: ['build', + '--cache-from', 'us-central1-docker.pkg.dev/$PROJECT_ID/evalbench/evalbench-harness-ci:cache', + '-t', 'evalbench-nl2sql-ci', + '-f', 'evalbench_service/Dockerfile', '.'] + + # Exits 0 when the run completes. The verify step decides pass or fail. + - id: nl2sql-eval + name: 'evalbench-nl2sql-ci' + dir: '/evalbench' + waitFor: ['preflight', 'build-image'] + entrypoint: 'bash' + args: ['-c', 'uv run --no-sync evalbench/evalbench.py --experiment_config=.ci/nl2sql_run_config.yaml'] + env: *eval_env + secretEnv: *eval_secrets + volumes: &results_volume + - name: 'eval_results' + path: '/evalbench/results' + + - id: verify + name: 'evalbench-nl2sql-ci' + dir: '/evalbench' + waitFor: ['nl2sql-eval'] + entrypoint: 'bash' + args: ['-c', 'uv run --no-sync python3 .ci/verify_nl2sql.py'] + volumes: *results_volume + +availableSecrets: + secretManager: + - versionName: projects/${_EVAL_PROJECT}/secrets/ci-smoke-postgres-password/versions/latest + env: 'CLOUD_SQL_POSTGRES_PASSWORD' + +# A cold image build plus the eval can exceed the 10 minute default. +timeout: '1800s' + +options: + logging: CLOUD_LOGGING_ONLY + # Sized for the image build. + machineType: 'E2_HIGHCPU_8' diff --git a/.ci/nl2sql_run_config.yaml b/.ci/nl2sql_run_config.yaml new file mode 100644 index 00000000..4bdaf40f --- /dev/null +++ b/.ci/nl2sql_run_config.yaml @@ -0,0 +1,29 @@ +# NL2SQL smoke eval for the verify-nl2sql check (.ci/nl2sql.cloudbuild.yaml). +# QueryData answers DQL prompts against db_blog on the CI instance. +# +# No setup_directory: db_blog is seeded once (.ci/nl2sql_seed_config.yaml). +# A per-run seed would drop tables that concurrent builds still query. +dataset_config: .ci/nl2sql_smoke.evalset.json +dataset_format: evalbench-standard-format +database_configs: + - .ci/db_configs/nl2sql_ci_postgres.yaml +dialects: + - postgres +# QueryData generates read queries only. +query_types: + - dql + +model_config: .ci/model_configs/querydata_ci.yaml +# QueryData receives the raw question, with no prompt template. +prompt_generator: 'NOOPGenerator' + +scorers: + returned_sql: null + executable_sql: null + set_match: null + llmrater: + model_config: datasets/model_configs/gemini_3.1_pro_model.yaml + +reporting: + csv: + output_directory: 'results' diff --git a/.ci/nl2sql_seed_config.yaml b/.ci/nl2sql_seed_config.yaml new file mode 100644 index 00000000..f0c80038 --- /dev/null +++ b/.ci/nl2sql_seed_config.yaml @@ -0,0 +1,15 @@ +# One-time seed of db_blog on the CI instance. CI does not run this. +# Re-run only when the db_blog schema or data changes. It drops and +# re-creates the public schema. +# +# Set EVAL_GCP_PROJECT_ID and CLOUD_SQL_POSTGRES_{REGION,INSTANCE,USER,PASSWORD} +# to the verify-nl2sql trigger values, then run: +# PYTHONPATH=.:evalbench uv run python evalbench/util/setup_databases.py \ +# --experiment_config=.ci/nl2sql_seed_config.yaml +dataset_config: .ci/nl2sql_smoke.evalset.json +dataset_format: evalbench-standard-format +database_configs: + - .ci/db_configs/nl2sql_ci_postgres.yaml +dialects: + - postgres +setup_directory: datasets/bat/setup diff --git a/.ci/nl2sql_smoke.evalset.json b/.ci/nl2sql_smoke.evalset.json new file mode 100644 index 00000000..fa891950 --- /dev/null +++ b/.ci/nl2sql_smoke.evalset.json @@ -0,0 +1,233 @@ +[ + { + "id": 9, + "nl_prompt": "Can you find all the users who have not made any posts?", + "query_type": "dql", + "database": "db_blog", + "dialects": [ + "postgres" + ], + "golden_sql": { + "postgres": [ + "SELECT u.user_id FROM tbl_users u LEFT JOIN tbl_posts p ON u.user_id = p.user_id WHERE p.post_id IS NULL;" + ] + }, + "eval_query": { + "postgres": [ + null + ] + }, + "setup_sql": {}, + "cleanup_sql": {}, + "tags": [ + "DQL", + "difficulty: simple", + "SELECT", + "LEFT_JOIN" + ] + }, + { + "id": 16, + "nl_prompt": "How many followers does each user have?", + "query_type": "dql", + "database": "db_blog", + "dialects": [ + "postgres" + ], + "golden_sql": { + "postgres": [ + "SELECT tbl_users.user_id, COUNT(tbl_followers.follower_id) AS follower_count FROM tbl_users LEFT JOIN tbl_followers ON tbl_users.user_id = tbl_followers.following_id GROUP BY tbl_users.user_id;" + ] + }, + "eval_query": { + "postgres": [ + null + ] + }, + "setup_sql": {}, + "cleanup_sql": {}, + "tags": [ + "DQL", + "difficulty: simple", + "SELECT", + "JOIN", + "AGGREGATE", + "LEFT_JOIN" + ] + }, + { + "id": 18, + "nl_prompt": "What posts have been reported and are pending review? What is the reason for the review? -- Identify posts by including post_id", + "query_type": "dql", + "database": "db_blog", + "dialects": [ + "postgres" + ], + "golden_sql": { + "postgres": [ + "SELECT p.post_id, r.report_reason FROM tbl_posts p INNER JOIN tbl_reports r ON p.post_id = r.post_id WHERE r.is_resolved = FALSE;" + ] + }, + "eval_query": { + "postgres": [ + null + ] + }, + "setup_sql": {}, + "cleanup_sql": {}, + "tags": [ + "DQL", + "difficulty: simple", + "SELECT", + "JOIN" + ] + }, + { + "id": 19, + "nl_prompt": "What is the most common reason for reporting a post, and what is the average time to resolve the report in minutes?", + "query_type": "dql", + "database": "db_blog", + "dialects": [ + "postgres" + ], + "golden_sql": { + "postgres": [ + "WITH rc AS (SELECT report_reason, COUNT(*) AS report_count, AVG(EXTRACT(EPOCH FROM (resolution_timestamp - report_timestamp)) / 60) AS avg_resolution_time FROM tbl_reports GROUP BY report_reason) SELECT report_reason, report_count, avg_resolution_time from rc WHERE report_count = (SELECT MAX(report_count) FROM rc);" + ] + }, + "eval_query": { + "postgres": [ + null + ] + }, + "setup_sql": {}, + "cleanup_sql": {}, + "tags": [ + "DQL", + "difficulty: moderate", + "SELECT", + "AGGREGATE", + "DATE", + "time" + ] + }, + { + "id": 100, + "nl_prompt": "Identify which is the most active hour of the day across all users", + "query_type": "dql", + "database": "db_blog", + "dialects": [ + "postgres" + ], + "golden_sql": { + "postgres": [ + "SELECT EXTRACT(HOUR FROM activity_timestamp) AS active_hour, COUNT(*) AS activity_count FROM tbl_user_activity GROUP BY active_hour ORDER BY activity_count DESC LIMIT 1;" + ] + }, + "eval_query": { + "postgres": [ + null + ] + }, + "setup_sql": {}, + "cleanup_sql": {}, + "tags": [ + "DQL", + "difficulty: moderate", + "SELECT", + "EXTRACT", + "AGGREGATE", + "SORT", + "LIMIT", + "HOUR", + "time" + ] + }, + { + "id": 180, + "nl_prompt": "Which posts have tags related to both 'programming' and 'database' ?", + "query_type": "dql", + "database": "db_blog", + "dialects": [ + "postgres" + ], + "golden_sql": { + "postgres": [ + "SELECT * FROM tbl_posts WHERE tags::jsonb @> '[ \"programming\", \"database\" ]'::jsonb;" + ] + }, + "eval_query": { + "postgres": [ + null + ] + }, + "setup_sql": {}, + "cleanup_sql": {}, + "tags": [ + "DQL", + "difficulty: simple", + "SELECT", + "JSON", + "difficulty: moderate", + "JSON_CONTAINS" + ] + }, + { + "id": 156, + "nl_prompt": "List the latest uploaded media file for each user -- include user_id, user name and media url", + "query_type": "dql", + "database": "db_blog", + "dialects": [ + "postgres" + ], + "golden_sql": { + "postgres": [ + "SELECT m1.uploader_id, u.username, m1.url FROM tbl_media m1 INNER JOIN ( SELECT uploader_id, MAX(upload_date) AS max_upload_date FROM tbl_media GROUP BY uploader_id ) m2 ON m1.uploader_id = m2.uploader_id AND m1.upload_date = m2.max_upload_date INNER JOIN tbl_users u ON m1.uploader_id = u.user_id;" + ] + }, + "eval_query": { + "postgres": [ + null + ] + }, + "setup_sql": {}, + "cleanup_sql": {}, + "tags": [ + "DQL", + "SELECT", + "AGGREGATE", + "JOIN", + "difficulty: moderate" + ] + }, + { + "id": 3, + "nl_prompt": "How many bloggers who have posted at least 1 blog in 2024 and have linked twitter account", + "query_type": "dql", + "database": "db_blog", + "dialects": [ + "postgres" + ], + "golden_sql": { + "postgres": [ + "SELECT COUNT(DISTINCT u.user_id) AS num_bloggers FROM tbl_users u INNER JOIN tbl_posts p ON u.user_id = p.user_id WHERE p.created_at >= DATE('2024-01-01') AND p.created_at <= DATE('2024-12-31') AND u.social_media_links ->> 'twitter' IS NOT NULL;" + ] + }, + "eval_query": { + "postgres": [ + null + ] + }, + "setup_sql": {}, + "cleanup_sql": {}, + "tags": [ + "DQL", + "difficulty: moderate", + "SELECT", + "JOIN", + "JSON", + "JSON_EXTRACT", + "datalinking" + ] + } +] diff --git a/.ci/verify_nl2sql.py b/.ci/verify_nl2sql.py new file mode 100644 index 00000000..5ab55ea6 --- /dev/null +++ b/.ci/verify_nl2sql.py @@ -0,0 +1,212 @@ +#!/usr/bin/env python3 +"""Grades the NL2SQL smoke eval for the verify-nl2sql check. + +evalbench.py exits 0 whenever a run completes, so this script reads the CSV +reports and fails the build unless the run is clean: + 1. configs.csv, evals.csv, scores.csv and summary.csv exist and have rows. + 2. Each dataset prompt has exactly one row in evals.csv. + 3. No prompt generator, SQL generator or golden SQL error on any row. + 4. Every row has generated SQL, and returned_sql scores above 0. + 5. Each configured scorer has a numeric score on every row, with no + comparison_error. + 6. At least one generated query executed (executable_sql above 0). + 7. The report contains pipeline_debug_info from QueryData. + +Not gated: generated_error, which executable_sql already scores, and +accuracy, which is printed only. Model variance would make the build flaky. +""" +import csv +import json +import math +import os +import sys + +from pyaml_env import parse_config + +RUN_CONFIG = os.environ.get("NL2SQL_RUN_CONFIG", ".ci/nl2sql_run_config.yaml") +REPORT_FILES = ("configs.csv", "evals.csv", "scores.csv", "summary.csv") +ERROR_COLUMNS = ( + "prompt_generator_error", + "sql_generator_error", + "golden_error", +) +EMPTY = {"", "nan", "none", "null"} + + +def is_empty(raw): + return (raw or "").strip().lower() in EMPTY + + +def as_score(raw): + try: + score = float(raw) + except (TypeError, ValueError): + return None + # nan is not a valid score. + return score if math.isfinite(score) else None + + +def as_prompt_id(raw): + # pandas may write id 9 as "9.0". + text = (raw or "").strip() + try: + return str(int(float(text))) + except ValueError: + return text + + +def latest_job_dir(output_dir): + if not os.path.isdir(output_dir): + return None + jobs = [os.path.join(output_dir, d) for d in os.listdir(output_dir) + if os.path.isdir(os.path.join(output_dir, d))] + return max(jobs, key=os.path.getmtime) if jobs else None + + +def read_rows(job_dir, name): + with open(os.path.join(job_dir, name), newline="") as f: + return list(csv.DictReader(f)) + + +def check_reports(job_dir): + problems = [] + for name in REPORT_FILES: + path = os.path.join(job_dir, name) + if not os.path.exists(path): + problems.append(f"{name} is missing") + elif not read_rows(job_dir, name): + problems.append(f"{name} has no rows") + return problems + + +def check_coverage(evalset, evals): + with open(evalset) as f: + expected = {str(item["id"]) for item in json.load(f)} + counts = {} + for row in evals: + prompt_id = as_prompt_id(row.get("prompt_id") or row.get("id")) + counts[prompt_id] = counts.get(prompt_id, 0) + 1 + + problems = [] + missing = sorted(expected - counts.keys()) + if missing: + problems.append(f"no eval row for prompt(s) {', '.join(missing)}") + extra = sorted(counts.keys() - expected) + if extra: + problems.append(f"eval rows for unknown prompt(s) {', '.join(extra)}") + duplicated = sorted(p for p, n in counts.items() if n > 1) + if duplicated: + problems.append( + f"more than one eval row for prompt(s) {', '.join(duplicated)}") + return problems + + +def check_errors(evals): + problems = [] + for row in evals: + eval_id = row.get("id") + for column in ERROR_COLUMNS: + value = row.get(column) + if not is_empty(value): + problems.append(f"{eval_id}: {column}: {value.strip()[:200]}") + if is_empty(row.get("generated_sql")): + problems.append(f"{eval_id}: QueryData returned no SQL") + return problems + + +def check_querydata_output(evals): + problems = [] + for row in evals: + if "pipeline_debug_info" not in (row.get("other") or ""): + problems.append( + f"{row.get('id')}: no pipeline_debug_info in the 'other' " + "column, so the QueryData output was not reported" + ) + return problems + + +def check_scores(scorers, evals, scores): + by_key = {} + for row in scores: + by_key[(row.get("comparator"), row.get("id"))] = row + + problems = [] + for scorer in scorers: + for eval_row in evals: + eval_id = eval_row.get("id") + row = by_key.get((scorer, eval_id)) + if row is None: + problems.append(f"{scorer}: no score row for {eval_id}") + continue + error = (row.get("comparison_error") or "").strip() + score = as_score(row.get("score")) + if not is_empty(error): + problems.append( + f"{scorer}: errored on {eval_id}: {error[:200]}") + elif score is None: + problems.append(f"{scorer}: non-numeric score for {eval_id}") + elif scorer == "returned_sql" and score <= 0: + problems.append(f"returned_sql: scored 0 for {eval_id}") + + if "executable_sql" in scorers: + executed = [as_score(r.get("score")) for r in scores + if r.get("comparator") == "executable_sql"] + if not any(s is not None and s > 0 for s in executed): + problems.append( + "executable_sql: no generated query executed on the CI " + "database, so the run is broken, not just inaccurate" + ) + return problems + + +def print_accuracy(scorers, scores): + print("Accuracy (reported, not gated):") + for scorer in scorers: + values = [as_score(r.get("score")) for r in scores + if r.get("comparator") == scorer] + values = [v for v in values if v is not None] + if values: + mean = sum(values) / len(values) + print(f" {scorer:<16} mean {mean:6.2f} over {len(values)} rows") + + +def main(): + config = parse_config(RUN_CONFIG) + scorers = sorted(config.get("scorers") or {}) + output_dir = config["reporting"]["csv"]["output_directory"] + + job_dir = latest_job_dir(output_dir) + if job_dir is None: + print(f"[FAIL] no run output under {output_dir}") + return 1 + print(f"Job: {job_dir}") + + problems = check_reports(job_dir) + if problems: + print("[FAIL]") + for problem in problems: + print(f" {problem}") + return 1 + + evals = read_rows(job_dir, "evals.csv") + scores = read_rows(job_dir, "scores.csv") + print(f"Eval rows: {len(evals)}, scorers: {', '.join(scorers)}\n") + + problems = check_coverage(config["dataset_config"], evals) + problems += check_errors(evals) + problems += check_querydata_output(evals) + problems += check_scores(scorers, evals, scores) + + print_accuracy(scorers, scores) + print() + if problems: + print(f"[FAIL] {len(problems)} problem(s)") + for problem in problems: + print(f" {problem}") + return 1 + print(f"[PASS] clean run: {len(evals)} prompts, {len(scorers)} scorers") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) From e0f238db153482f7923a23081472463260614151 Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad Date: Tue, 29 Sep 2026 09:01:04 +0000 Subject: [PATCH 14/27] ci(release): gate, tag, and deploy the weekly release by digest - Gate the image that the Build trigger pushed for the commit with a SQLite smoke eval over gRPC. Retry once on model failures. - Tag the tested digest, deploy it to GKE with automatic rollback, move :latest, and deploy it to Cloud Run. - Print one RELEASE_RESULT line per run and add a log-based email alert. - Rename verify_nl2sql_smoke.py to verify_release_smoke.py. --- .ci/release.cloudbuild.yaml | 367 +++++++++++++++---------- .ci/release/alert_policy.yaml | 44 +++ .ci/release/lib.sh | 46 ++++ .ci/release/release_smoke.evalset.json | 176 +++++++++++- .ci/release/release_smoke_config.yaml | 20 +- .ci/release/verify_nl2sql_smoke.py | 186 ------------- .ci/release/verify_release_smoke.py | 290 +++++++++++++++++++ 7 files changed, 795 insertions(+), 334 deletions(-) create mode 100644 .ci/release/alert_policy.yaml create mode 100644 .ci/release/lib.sh delete mode 100644 .ci/release/verify_nl2sql_smoke.py create mode 100644 .ci/release/verify_release_smoke.py diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index 09a474e6..5470c649 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -1,15 +1,18 @@ # Weekly release pipeline. # -# Gates the image the Build trigger already pushed for this commit, tags it as -# an immutable release, then rolls it out to GKE and Corp Run. +# Gates the image that the Build trigger pushed for this commit, tags it, +# deploys it to GKE with automatic rollback, moves :latest, and deploys it +# to Corp Run. # -# Shell variables use $$NAME and command substitutions use $$(...) so Cloud -# Build passes them to bash unexpanded. +# Each run prints one RELEASE_RESULT line (lib.sh), and alert_policy.yaml +# emails it. Cloud Build has no on-failure hook, so a failed step prints its +# own line. A timeout or a cancel prints no line. +# +# Use $$ for shell variables and $$(...). Cloud Build expands a single $. steps: - # Nothing is built here: the Build trigger must have already pushed an image - # for this commit. + # The Build trigger must have pushed an image for this commit. - id: resolve-release name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' entrypoint: 'bash' @@ -17,21 +20,30 @@ steps: - '-c' - | set -uo pipefail + . /workspace/.ci/release/lib.sh + STAGE=resolve-release if [[ -z "$COMMIT_SHA" || -z "$SHORT_SHA" ]]; then - echo "FAIL: COMMIT_SHA or SHORT_SHA is empty." - exit 1 + release_fail $$STAGE "COMMIT_SHA or SHORT_SHA is empty." fi - REPO="us-central1-docker.pkg.dev/$PROJECT_ID/evalbench/eval_server" - IMAGE="$$REPO:$COMMIT_SHA" - echo "Resolving $$IMAGE" + # Two runs race on GKE and :latest. A failed list call does not block. + if [[ -n "$TRIGGER_ID" ]]; then + OTHERS=$$(gcloud builds list --ongoing --region="$LOCATION" \ + --filter="buildTriggerId=$TRIGGER_ID AND id!=$BUILD_ID" \ + --format='value(id)' 2>/dev/null) + if [[ -n "$$OTHERS" ]]; then + release_fail $$STAGE "Another release run is active: $$OTHERS" + fi + fi - DIGEST=$$(gcloud artifacts docker images describe "$$IMAGE" \ + REPO="us-central1-docker.pkg.dev/$PROJECT_ID/evalbench/eval_server" + echo "Resolving $$REPO:$COMMIT_SHA" + DIGEST=$$(gcloud artifacts docker images describe \ + "$$REPO:$COMMIT_SHA" \ --format='value(image_summary.digest)' 2>/dev/null) if [[ -z "$$DIGEST" ]]; then - echo "FAIL: no image at $$IMAGE" - exit 1 + release_fail $$STAGE "No image at $$REPO:$COMMIT_SHA. The Build trigger has not published this commit." fi TAG="v$$(date -u +%Y.%m.%d).$SHORT_SHA" @@ -42,8 +54,8 @@ steps: echo "digest = $$DIGEST" echo "tag = $$TAG" - # eval_client exits 0 on any completed run, so scores.csv is the only real - # gate on whether the candidate image evaluates correctly. + # Serves the candidate image over gRPC and runs the smoke eval on it. + # eval_client exits 0 for any completed run, so the verifier grades the CSVs. - id: smoke-gate name: 'us-central1-docker.pkg.dev/$PROJECT_ID/evalbench/eval_server:$COMMIT_SHA' dir: '/evalbench' @@ -53,41 +65,66 @@ steps: - '-c' - | set -uo pipefail + . /workspace/.ci/release/lib.sh + STAGE=smoke-gate + CONFIG=.ci/release/release_smoke_config.yaml + # CsvReporter forces this directory on the gRPC path. + RESULTS=/tmp_session_files/results + SERVER_LOG=/tmp/server.log # Add both module roots so eval_client and proto imports resolve. export PYTHONPATH=/evalbench/evalbench:/evalbench/evalbench/evalproto echo "===== start eval_server =====" - python evalbench/eval_server.py --localhost > /tmp/server.log 2>&1 & + python evalbench/eval_server.py --localhost > "$$SERVER_LOG" 2>&1 & SERVER_PID=$$! trap 'kill $$SERVER_PID 2>/dev/null || true' EXIT dump_log() { - echo "----- eval_server log, last 40 lines -----" - tail -40 /tmp/server.log + echo "----- eval_server log, last 60 lines -----" + tail -60 "$$SERVER_LOG" } echo "===== wait for gRPC =====" if ! python .ci/release/wait_for_grpc.py --timeout 300 --insecure; then dump_log - exit 1 + release_fail $$STAGE "The eval server did not answer Ping within 300 s." fi - echo "===== run the smoke eval over gRPC =====" - if ! EVALBENCH_HOST=localhost PORT=50051 EVALBENCH_INSECURE=true \ - python evalbench/client/eval_client.py \ - --endpoint=local \ - --experiment=.ci/release/release_smoke_config.yaml; then + # Retry once if the client crashes or the verifier exits 2 (model). + for ATTEMPT in 1 2; do + echo "===== smoke eval, attempt $$ATTEMPT =====" + LOG_OFFSET=$$(stat -c %s "$$SERVER_LOG") + STARTED=$$(date +%s) + CLIENT_RC=0 + VERIFY_RC=0 + EVALBENCH_HOST=localhost PORT=50051 EVALBENCH_INSECURE=true \ + python evalbench/client/eval_client.py \ + --endpoint=local --experiment="$$CONFIG" || CLIENT_RC=$$? + if [[ $$CLIENT_RC -ne 0 ]]; then + echo "eval_client exited with code $$CLIENT_RC." + else + python .ci/release/verify_release_smoke.py --config "$$CONFIG" \ + --results-dir "$$RESULTS" --since "$$STARTED" || VERIFY_RC=$$? + if [[ $$VERIFY_RC -eq 0 ]]; then + break + fi + fi + if [[ $$ATTEMPT -eq 1 ]] && [[ $$CLIENT_RC -ne 0 || $$VERIFY_RC -eq 2 ]]; then + echo "The failure can come from the model. Retrying once." + continue + fi dump_log - exit 1 - fi + release_fail $$STAGE "The smoke eval failed on attempt $$ATTEMPT. The build log has the verifier output." + done - # CsvReporter forces /tmp_session_files/results on the gRPC path. - echo "===== audit scores.csv =====" - if ! python .ci/release/verify_nl2sql_smoke.py \ - --results-dir /tmp_session_files/results; then - dump_log - exit 1 + # Scan only the passing attempt. Any traceback fails the gate, also + # one from a handled exception. + EXCEPTIONS=$$(tail -c +$$((LOG_OFFSET + 1)) "$$SERVER_LOG" \ + | grep -n -A8 'Traceback (most recent call last)' | head -80) + if [[ -n "$$EXCEPTIONS" ]]; then + echo "$$EXCEPTIONS" + release_fail $$STAGE "The eval server logged exceptions during the smoke eval." fi echo "===== smoke gate passed =====" @@ -97,8 +134,7 @@ steps: - 'GOOGLE_CLOUD_PROJECT=$PROJECT_ID' - 'UV_CACHE_DIR=/tmp/uv-cache' - # Tag by digest, not by :$COMMIT_SHA, so the release points at the exact - # image the gate just passed. + # Tags the tested digest, not :$COMMIT_SHA. - id: tag-release name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' waitFor: ['smoke-gate'] @@ -106,22 +142,28 @@ steps: args: - '-c' - | - set -euo pipefail + set -uo pipefail + . /workspace/.ci/release/lib.sh + STAGE=tag-release REPO=$$(cat /workspace/image_repo.txt) DIGEST=$$(cat /workspace/image_digest.txt) TAG=$$(cat /workspace/release_tag.txt) + + # If a Build re-run moved :$COMMIT_SHA, the gate tested another image. + CURRENT=$$(gcloud artifacts docker images describe \ + "$$REPO:$COMMIT_SHA" --format='value(image_summary.digest)') + if [[ "$$CURRENT" != "$$DIGEST" ]]; then + release_fail $$STAGE ":$COMMIT_SHA moved from $$DIGEST to $$CURRENT during the run. Start the release again." + fi + echo "Tagging $$REPO@$$DIGEST as :$$TAG" - # 'docker tags add' deletes and recreates an existing tag, which needs - # tags.delete. 'tags update' repoints it in place, but cannot create - # it, so fall back to add for a tag this commit has not had before. - gcloud artifacts tags update "$$TAG" --location=us-central1 \ - --repository=evalbench --package=eval_server \ - --version="$$DIGEST" 2>/dev/null \ - || gcloud artifacts docker tags add "$$REPO@$$DIGEST" "$$REPO:$$TAG" + if ! move_tag "$$REPO" "$$TAG" "$$DIGEST"; then + release_fail $$STAGE "Could not apply the release tag $$TAG." + fi echo "Release tag: $$REPO:$$TAG" - # The Deployment spec only carries the mutable :latest tag, so the kubelet's - # imageID is the only record of which digest is actually serving. + # Records the digest that GKE serves now. The pod imageID is exact, but + # the Deployment spec can hold a tag. - id: capture-rollback-target name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' waitFor: ['tag-release'] @@ -130,17 +172,16 @@ steps: - '-c' - | set -uo pipefail - - CLUSTER=evalbench-directpath-cluster - ZONE=us-central1-c + . /workspace/.ci/release/lib.sh + STAGE=capture-rollback-target NAMESPACE=evalbench-namespace SELECTOR=app=evalbench-eval-server CONTAINER=evalbench-eval - if ! gcloud container clusters get-credentials "$$CLUSTER" \ - --zone "$$ZONE" --project "$PROJECT_ID"; then - echo "FAIL: cannot get credentials for $$CLUSTER" - exit 1 + if ! gcloud container clusters get-credentials \ + evalbench-directpath-cluster \ + --zone us-central1-c --project "$PROJECT_ID"; then + release_fail $$STAGE "Cannot get credentials for evalbench-directpath-cluster." fi RUNNING=$$(kubectl get pods -n "$$NAMESPACE" -l "$$SELECTOR" \ @@ -150,23 +191,18 @@ steps: COUNT=$$(echo "$$RUNNING" | grep -c .) if [[ "$$COUNT" -eq 0 ]]; then - echo "FAIL: no running $$CONTAINER container in $$NAMESPACE." - echo "Refusing to deploy without a rollback target." - exit 1 + release_fail $$STAGE "No running $$CONTAINER container, so there is no rollback target." fi - if [[ "$$COUNT" -gt 1 ]]; then - echo "FAIL: pods are serving more than one digest:" echo "$$RUNNING" - echo "A rollout is already in progress. Refusing to deploy." - exit 1 + release_fail $$STAGE "Pods serve more than one digest. A rollout is already in progress." fi echo "$$RUNNING" > /workspace/rollback_digest.txt echo "rollback target = $$RUNNING" - # Deploy, verify, and roll back live in one step: Cloud Build has no - # on-failure hook, so a separate rollback step would never run. + # Deploy, check, and rollback share one step. Cloud Build has no + # on-failure hook, so a separate rollback step never runs. - id: deploy-gke name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' waitFor: ['capture-rollback-target'] @@ -175,7 +211,8 @@ steps: - '-c' - | set -uo pipefail - + . /workspace/.ci/release/lib.sh + STAGE=deploy-gke NAMESPACE=evalbench-namespace SELECTOR=app=evalbench-eval-server CONTAINER=evalbench-eval @@ -185,12 +222,17 @@ steps: REPO=$$(cat /workspace/image_repo.txt) NEW_DIGEST=$$(cat /workspace/image_digest.txt) OLD_DIGEST=$$(cat /workspace/rollback_digest.txt) + TAG=$$(cat /workspace/release_tag.txt) if ! gcloud container clusters get-credentials \ evalbench-directpath-cluster \ --zone us-central1-c --project "$PROJECT_ID"; then - echo "FAIL: cannot get cluster credentials." - exit 1 + release_fail $$STAGE "Cannot get credentials for evalbench-directpath-cluster." + fi + + if [[ "$$NEW_DIGEST" == "$$OLD_DIGEST" ]]; then + echo "GKE already serves $$NEW_DIGEST. No rollout is needed." + exit 0 fi running_digests() { @@ -200,129 +242,172 @@ steps: | grep -v '^$$' | sort -u } - # The Deployment tracks :latest, so the tag selects the image and the - # restart is what makes the kubelet pull it. - roll_to() { - echo "Pointing :latest at $$1" - gcloud artifacts tags update latest --location=us-central1 \ - --repository=evalbench --package=eval_server --version="$$1" \ - && kubectl rollout restart "$$DEPLOYMENT" -n "$$NAMESPACE" \ + # Pin by digest, so a tag move cannot change the running image. + pin() { + echo "Pinning $$CONTAINER to $$REPO@$$1" + kubectl set image "$$DEPLOYMENT" "$$CONTAINER=$$REPO@$$1" \ + -n "$$NAMESPACE" \ && kubectl rollout status "$$DEPLOYMENT" -n "$$NAMESPACE" \ --timeout="$$ROLLOUT_TIMEOUT" } + # Restore the previous digest and print the result line. rollback() { echo "===== rolling back to $$OLD_DIGEST =====" - if ! roll_to "$$OLD_DIGEST"; then - echo "FAIL: rollback did not complete. Not retrying." - echo "Last known-good digest: $$REPO@$$OLD_DIGEST" + if pin "$$OLD_DIGEST"; then + release_result ROLLED_BACK $$STAGE "$$1 GKE serves the previous digest $$OLD_DIGEST again." + else kubectl get pods -n "$$NAMESPACE" -l "$$SELECTOR" -o wide - exit 1 + release_result ROLLBACK_FAILED $$STAGE "$$1 The rollback did not complete. Last known-good image: $$REPO@$$OLD_DIGEST" fi - echo "Rolled back to $$OLD_DIGEST. Failing the build." exit 1 } echo "===== deploy $$NEW_DIGEST =====" - if ! roll_to "$$NEW_DIGEST"; then - echo "FAIL: rollout did not complete within $$ROLLOUT_TIMEOUT." + if ! pin "$$NEW_DIGEST"; then kubectl get pods -n "$$NAMESPACE" -l "$$SELECTOR" -o wide - rollback + rollback "The rollout did not complete within $$ROLLOUT_TIMEOUT." fi - # Pods that lost the rollout can linger in Terminating for a few - # seconds after rollout status returns, so retry before believing it. - echo "===== confirm the pods are serving $$NEW_DIGEST =====" + # Old pods can stay Terminating for a few seconds after the rollout. + echo "===== confirm the pods serve $$NEW_DIGEST =====" for ATTEMPT in {1..20}; do SERVING=$$(running_digests) [[ "$$SERVING" == "$$NEW_DIGEST" ]] && break sleep 3 done if [[ "$$SERVING" != "$$NEW_DIGEST" ]]; then - echo "FAIL: expected $$NEW_DIGEST, pods report:" - echo "$$SERVING" - rollback + echo "Pods report: $$SERVING" + rollback "The pods do not serve the release digest." fi - echo "Every pod reports the release digest (attempt $$ATTEMPT)." - # There is no readiness probe on this Deployment, so "Ready" only means - # the process started. A gRPC Ping is the first real health signal. + # The Deployment has no readiness probe, so Ready means only that the + # process started. Ping is the real health check. echo "===== gRPC health check =====" POD=$$(kubectl get pods -n "$$NAMESPACE" -l "$$SELECTOR" \ -o jsonpath='{.items[0].metadata.name}') if ! kubectl exec -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" -- \ python /evalbench/.ci/release/wait_for_grpc.py --timeout 300; then - echo "FAIL: $$POD never answered Ping." kubectl logs -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" --tail=150 - rollback + rollback "The new pod $$POD did not answer Ping." fi - echo "===== release deployed =====" - echo "serving $$REPO@$$NEW_DIGEST as $$(cat /workspace/release_tag.txt)" + echo "===== scan the new pod logs =====" + EXCEPTIONS=$$(kubectl logs -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" \ + | grep -n -A8 'Traceback (most recent call last)' | head -80) + if [[ -n "$$EXCEPTIONS" ]]; then + echo "$$EXCEPTIONS" + rollback "The new pod $$POD logged server exceptions." + fi + + # Record the release. An annotation does not restart the pod. + kubectl annotate "$$DEPLOYMENT" -n "$$NAMESPACE" --overwrite \ + kubernetes.io/change-cause="release $$TAG ($$NEW_DIGEST)" || true - # The Mesop UI serves this same image from Cloud Run in evalbench-dev. - # Writing :latest keeps the manual 'make deploy-corprun' path in sync. - - id: copy-corprun-image - name: 'gcr.io/go-containerregistry/gcrane' + echo "===== GKE serves $$REPO@$$NEW_DIGEST as $$TAG =====" + + # Moves :latest only after GKE passes, so manual deploys get a verified + # image. + - id: promote-latest + name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' waitFor: ['deploy-gke'] + entrypoint: 'bash' + args: + - '-c' + - | + set -uo pipefail + . /workspace/.ci/release/lib.sh + REPO=$$(cat /workspace/image_repo.txt) + DIGEST=$$(cat /workspace/image_digest.txt) + if ! move_tag "$$REPO" latest "$$DIGEST"; then + release_fail promote-latest "GKE serves $$DIGEST, but :latest did not move. Break-glass deploys still use the previous release." + fi + echo "$$REPO:latest -> $$DIGEST" + + # Cloud Run pulls from evalbench-dev. The copy keeps the digest. + - id: copy-cloudrun-image + name: 'gcr.io/go-containerregistry/gcrane:debug' + waitFor: ['promote-latest'] + entrypoint: 'sh' args: - - 'copy' - - 'us-central1-docker.pkg.dev/$PROJECT_ID/evalbench/eval_server:$COMMIT_SHA' - - 'us-central1-docker.pkg.dev/evalbench-dev/cr-images/eval_server:latest' + - '-c' + - | + . /workspace/.ci/release/lib.sh + CR_REPO=us-central1-docker.pkg.dev/evalbench-dev/cr-images/eval_server + REPO=$$(cat /workspace/image_repo.txt) + DIGEST=$$(cat /workspace/image_digest.txt) + TAG=$$(cat /workspace/release_tag.txt) + if ! gcrane copy "$$REPO@$$DIGEST" "$$CR_REPO:$$TAG"; then + release_fail copy-cloudrun-image "Could not copy the release to $$CR_REPO. Cloud Run still serves the previous release." + fi - - id: deploy-corprun + # The ingress is internal, so Cloud Build cannot send HTTP probes. gcloud + # waits for the revision startup check, and pre-merge CI tests the pages. + - id: deploy-cloudrun name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' - waitFor: ['copy-corprun-image'] + waitFor: ['copy-cloudrun-image'] entrypoint: 'bash' args: - '-c' - | set -uo pipefail - + . /workspace/.ci/release/lib.sh + STAGE=deploy-cloudrun SERVICE=evalbench - CR_PROJECT=evalbench-dev - CR_REGION=us-central1 + JOBS=(precompute-job recompute-job) + RUN=(--project=evalbench-dev --region=us-central1 --quiet) CR_REPO=us-central1-docker.pkg.dev/evalbench-dev/cr-images/eval_server - DIGEST=$$(cat /workspace/image_digest.txt) - # Cloud Run pins each revision to a digest, so the previous revision is - # a complete rollback target on its own. - OLD_REVISION=$$(gcloud run services describe "$$SERVICE" \ - --project="$$CR_PROJECT" --region="$$CR_REGION" \ + OLD_REVISION=$$(gcloud run services describe "$$SERVICE" "$${RUN[@]}" \ --format='value(status.latestReadyRevisionName)') - if [[ -z "$$OLD_REVISION" ]]; then - echo "FAIL: $$SERVICE has no ready revision to roll back to." - exit 1 + echo "current revision = $$OLD_REVISION" + + # Keep traffic on the current revision until the digest check passes. + echo "===== deploy a revision with no traffic =====" + if ! gcloud run deploy "$$SERVICE" "$${RUN[@]}" \ + --image="$$CR_REPO@$$DIGEST" --no-traffic; then + release_fail $$STAGE "The new revision did not become ready. Traffic stays on $$OLD_REVISION." fi - echo "rollback revision = $$OLD_REVISION" - - # Deploying by digest proves the copy landed the image the GKE gate - # accepted; a mismatched copy cannot resolve and fails here. - echo "===== deploy to Cloud Run =====" - if ! gcloud run deploy "$$SERVICE" --quiet \ - --project="$$CR_PROJECT" \ - --region="$$CR_REGION" \ - --image="$$CR_REPO@$$DIGEST"; then - echo "FAIL: Cloud Run deploy did not succeed." - # A revision that never goes ready takes no traffic, so this only - # matters if the deploy failed after the traffic split moved. - echo "===== pinning traffic back to $$OLD_REVISION =====" - if ! gcloud run services update-traffic "$$SERVICE" --quiet \ - --project="$$CR_PROJECT" --region="$$CR_REGION" \ - --to-revisions="$$OLD_REVISION=100"; then - echo "FAIL: could not pin traffic back. $$SERVICE may be serving" - echo "an unverified revision." - echo "Last known-good revision: $$OLD_REVISION" - fi - exit 1 + + NEW_REVISION=$$(gcloud run services describe "$$SERVICE" "$${RUN[@]}" \ + --format='value(status.latestCreatedRevisionName)') + SERVED=$$(gcloud run revisions describe "$$NEW_REVISION" "$${RUN[@]}" \ + --format='value(status.imageDigest)') + if [[ "$$SERVED" != *"@$$DIGEST" ]]; then + release_fail $$STAGE "$$NEW_REVISION runs $$SERVED, not $$DIGEST. Traffic stays on $$OLD_REVISION." fi - NEW_REVISION=$$(gcloud run services describe "$$SERVICE" \ - --project="$$CR_PROJECT" --region="$$CR_REGION" \ - --format='value(status.latestReadyRevisionName)') - echo "===== Corp Run deployed =====" - echo "$$NEW_REVISION serving $$CR_REPO@$$DIGEST" + echo "===== shift traffic to $$NEW_REVISION =====" + if ! gcloud run services update-traffic "$$SERVICE" "$${RUN[@]}" \ + --to-latest; then + release_fail $$STAGE "Could not shift traffic to $$NEW_REVISION. Traffic stays on $$OLD_REVISION." + fi + + FAILED_JOBS="" + for JOB in "$${JOBS[@]}"; do + gcloud run jobs update "$$JOB" "$${RUN[@]}" \ + --image="$$CR_REPO@$$DIGEST" || FAILED_JOBS="$$FAILED_JOBS $$JOB" + done + if [[ -n "$$FAILED_JOBS" ]]; then + release_fail $$STAGE "The UI serves the release, but these jobs did not update:$$FAILED_JOBS" + fi + + # 'make deploy-corprun' uses :latest. + if ! move_tag "$$CR_REPO" latest "$$DIGEST"; then + release_fail $$STAGE "The UI and jobs serve the release, but cr-images :latest did not move." + fi + echo "===== Cloud Run: $$NEW_REVISION serves $$CR_REPO@$$DIGEST =====" + + - id: notify-success + name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' + waitFor: ['deploy-cloudrun'] + entrypoint: 'bash' + args: + - '-c' + - | + . /workspace/.ci/release/lib.sh + release_result SUCCESS release "GKE and Cloud Run serve $$(cat /workspace/image_digest.txt)." substitutions: _EVAL_REGION: 'global' @@ -332,3 +417,9 @@ timeout: '3600s' options: logging: CLOUD_LOGGING_ONLY machineType: 'E2_HIGHCPU_8' + # lib.sh reads these to build the RELEASE_RESULT line. + env: + - 'BUILD_ID=$BUILD_ID' + - 'COMMIT_SHA=$COMMIT_SHA' + - 'LOCATION=$LOCATION' + - 'PROJECT_ID=$PROJECT_ID' diff --git a/.ci/release/alert_policy.yaml b/.ci/release/alert_policy.yaml new file mode 100644 index 00000000..2798ede1 --- /dev/null +++ b/.ci/release/alert_policy.yaml @@ -0,0 +1,44 @@ +# Log-based alert that emails the RELEASE_RESULT line (lib.sh) of each +# weekly release run. +# +# Replace TRIGGER_ID and CHANNEL_ID, then create the policy: +# gcloud alpha monitoring policies create --project=cloud-db-nl2sql \ +# --policy-from-file=.ci/release/alert_policy.yaml +displayName: 'EvalBench weekly release' +combiner: OR +conditions: + - displayName: 'RELEASE_RESULT line in the weekly release build' + conditionMatchedLog: + filter: >- + resource.type="build" + AND resource.labels.build_trigger_id="TRIGGER_ID" + AND textPayload:"RELEASE_RESULT status=" + labelExtractors: + status: 'REGEXP_EXTRACT(textPayload, "status=(\\S+)")' + stage: 'REGEXP_EXTRACT(textPayload, "stage=(\\S+)")' + tag: 'REGEXP_EXTRACT(textPayload, "tag=(\\S+)")' + commit: 'REGEXP_EXTRACT(textPayload, "commit=(\\S+)")' + log: 'REGEXP_EXTRACT(textPayload, "log=(\\S+)")' + detail: 'REGEXP_EXTRACT(textPayload, "detail=\"(.*)\"")' +alertStrategy: + # 5 minutes is the minimum. A run prints one line, so no result is lost. + notificationRateLimit: + period: 300s + autoClose: 1800s +documentation: + subject: 'EvalBench release ${log.extracted_label.status}: ${log.extracted_label.tag}' + mimeType: text/markdown + content: | + **Status:** ${log.extracted_label.status} + + **Stage:** ${log.extracted_label.stage} + + **Tag:** ${log.extracted_label.tag} + + **Detail:** ${log.extracted_label.detail} + + **Commit:** ${log.extracted_label.commit} + + **Build log:** ${log.extracted_label.log} +notificationChannels: + - projects/cloud-db-nl2sql/notificationChannels/CHANNEL_ID diff --git a/.ci/release/lib.sh b/.ci/release/lib.sh new file mode 100644 index 00000000..8e38f5e4 --- /dev/null +++ b/.ci/release/lib.sh @@ -0,0 +1,46 @@ +# Helpers for the steps in release.cloudbuild.yaml: +# . /workspace/.ci/release/lib.sh +# +# POSIX sh, so the busybox shell in the gcrane image can source it. Needs +# BUILD_ID, COMMIT_SHA, LOCATION and PROJECT_ID from options.env. + +# alert_policy.yaml parses this line. Update it if the format changes. +RELEASE_RESULT_PREFIX="RELEASE_RESULT" + +# Prints the result line of the run. +# +# Usage: release_result STATUS STAGE DETAIL +# STATUS SUCCESS, FAILED, ROLLED_BACK, or ROLLBACK_FAILED +# STAGE the id of the step that ended the run +# DETAIL one sentence for the email. Double quotes become single quotes. +release_result() { + _tag=$(cat /workspace/release_tag.txt 2>/dev/null || echo "none") + _detail=$(printf '%s' "$3" | tr '"\n' "' ") + _log="https://console.cloud.google.com/cloud-build/builds;region=${LOCATION}/${BUILD_ID}?project=${PROJECT_ID}" + _commit="https://github.com/GoogleCloudPlatform/evalbench/commit/${COMMIT_SHA}" + echo "${RELEASE_RESULT_PREFIX} status=$1 stage=$2 tag=${_tag} commit=${_commit} log=${_log} detail=\"${_detail}\"" +} + +# Prints a FAILED result line and ends the step with an error. +# +# Usage: release_fail STAGE DETAIL +release_fail() { + release_result FAILED "$1" "$2" + exit 1 +} + +# Points TAG at DIGEST. 'tags update' moves an existing tag without the +# tags.delete permission. 'docker tags add' creates a new tag. +# +# Usage: move_tag IMAGE TAG DIGEST +# IMAGE the image path, e.g. us-central1-docker.pkg.dev/PROJECT/REPO/PKG +move_tag() { + _host=$(echo "$1" | cut -d/ -f1) + _project=$(echo "$1" | cut -d/ -f2) + _repository=$(echo "$1" | cut -d/ -f3) + _package=$(echo "$1" | cut -d/ -f4-) + gcloud artifacts tags update "$2" --location="${_host%-docker.pkg.dev}" \ + --project="$_project" --repository="$_repository" \ + --package="$_package" --version="$3" 2>/dev/null \ + || gcloud artifacts docker tags add "$1@$3" "$1:$2" +} diff --git a/.ci/release/release_smoke.evalset.json b/.ci/release/release_smoke.evalset.json index 455cd0a0..98571150 100644 --- a/.ci/release/release_smoke.evalset.json +++ b/.ci/release/release_smoke.evalset.json @@ -27,7 +27,13 @@ "JSON", "JSON_EXTRACT", "datalinking" - ] + ], + "other": { + "Comment": "LGTM", + "nl_prompt_base": "How many bloggers who have posted at least 1 blog in 2024 and have linked twitter account", + "nl_prompt_extra_context": "", + "public": true + } }, { "id": 5, @@ -58,6 +64,172 @@ "CASE", "LEFT_JOIN", "IS_NULL" - ] + ], + "other": { + "Comment": "LGTM", + "nl_prompt_base": "What is the number of comments per post for each user?", + "nl_prompt_extra_context": "Use user_id to identify users and post_id to identify posts.", + "public": true + } + }, + { + "id": 42, + "nl_prompt": "Can you update my username to 'magic_one' -- user_id = 9;", + "query_type": "dml", + "database": "db_blog", + "dialects": [ + "sqlite" + ], + "golden_sql": { + "sqlite": [ + "UPDATE tbl_users SET username = 'magic_one', updated_at = CURRENT_TIMESTAMP WHERE user_id = 9;" + ] + }, + "eval_query": { + "sqlite": [ + "SELECT username, updated_at, updated_at > '2024-02-27 07:27:14' AS is_updated FROM tbl_users WHERE user_id = 9;" + ] + }, + "setup_sql": { + "sqlite": [] + }, + "cleanup_sql": { + "sqlite": [ + "UPDATE tbl_users SET username = 'sophia.anderson', updated_at = '2024-02-27 07:26:14' WHERE user_id = 9;" + ] + }, + "tags": [ + "DML", + "difficulty: moderate", + "UPDATE" + ], + "other": { + "Comment": "LGTM. Instructions do not include updated_at, but that is implied.", + "nl_prompt_base": "Can you update my username to 'magic_one'", + "nl_prompt_extra_context": "user_id = 9;", + "public": true + } + }, + { + "id": 184, + "nl_prompt": "Update the post content in tbl_posts -- post_id = 3, new content should be 'Updated content'", + "query_type": "dml", + "database": "db_blog", + "dialects": [ + "sqlite" + ], + "golden_sql": { + "sqlite": [ + "UPDATE tbl_posts SET content = 'Updated content' WHERE post_id = 3;" + ] + }, + "eval_query": { + "sqlite": [ + "SELECT content FROM tbl_posts WHERE post_id = 3;" + ] + }, + "setup_sql": { + "sqlite": [ + "UPDATE tbl_posts SET content = 'Tips and tricks for efficient web development.' WHERE post_id = 3;" + ] + }, + "cleanup_sql": { + "sqlite": [ + "UPDATE tbl_posts SET content = 'Tips and tricks for efficient web development.' WHERE post_id = 3;" + ] + }, + "tags": [ + "DML", + "difficulty: simple", + "UPDATE" + ], + "other": { + "nl_prompt_base": "Update the post content in tbl_posts", + "nl_prompt_extra_context": "post_id = 3, new content should be 'Updated content'", + "public": true + } + }, + { + "id": 43, + "nl_prompt": "Update the labels table to include a json field called comments_description.", + "query_type": "ddl", + "database": "db_blog", + "dialects": [ + "sqlite" + ], + "golden_sql": { + "sqlite": [ + "ALTER TABLE tbl_labels ADD COLUMN comments_description TEXT CHECK (json_valid(comments_description));" + ] + }, + "eval_query": { + "sqlite": [ + "SELECT COUNT(*) FROM tbl_labels WHERE json_valid(comments_description);" + ] + }, + "setup_sql": { + "sqlite": [ + "BEGIN TRANSACTION; CREATE TABLE tbl_labels_new (label_id INTEGER PRIMARY KEY, creator_id INTEGER DEFAULT NULL, updated_by INTEGER DEFAULT NULL, name TEXT DEFAULT NULL, description TEXT DEFAULT NULL, created_at TEXT DEFAULT NULL, updated_at TEXT DEFAULT NULL, post_count INTEGER DEFAULT NULL, visibility_status TEXT DEFAULT NULL, is_active INTEGER DEFAULT NULL, usage_frequency INTEGER DEFAULT NULL, parent_label_id INTEGER DEFAULT NULL, FOREIGN KEY (creator_id) REFERENCES tbl_users(user_id), FOREIGN KEY (updated_by) REFERENCES tbl_users(user_id), FOREIGN KEY (parent_label_id) REFERENCES tbl_labels(label_id)); INSERT INTO tbl_labels_new (label_id, creator_id, updated_by, name, description, created_at, updated_at, post_count, visibility_status, is_active, usage_frequency, parent_label_id) SELECT label_id, creator_id, updated_by, name, description, created_at, updated_at, post_count, visibility_status, is_active, usage_frequency, parent_label_id FROM tbl_labels; DROP TABLE tbl_labels; ALTER TABLE tbl_labels_new RENAME TO tbl_labels; COMMIT;" + ] + }, + "cleanup_sql": { + "sqlite": [ + "BEGIN TRANSACTION; CREATE TABLE tbl_labels_new (label_id INTEGER PRIMARY KEY, creator_id INTEGER DEFAULT NULL, updated_by INTEGER DEFAULT NULL, name TEXT DEFAULT NULL, description TEXT DEFAULT NULL, created_at TEXT DEFAULT NULL, updated_at TEXT DEFAULT NULL, post_count INTEGER DEFAULT NULL, visibility_status TEXT DEFAULT NULL, is_active INTEGER DEFAULT NULL, usage_frequency INTEGER DEFAULT NULL, parent_label_id INTEGER DEFAULT NULL, FOREIGN KEY (creator_id) REFERENCES tbl_users(user_id), FOREIGN KEY (updated_by) REFERENCES tbl_users(user_id), FOREIGN KEY (parent_label_id) REFERENCES tbl_labels(label_id)); INSERT INTO tbl_labels_new (label_id, creator_id, updated_by, name, description, created_at, updated_at, post_count, visibility_status, is_active, usage_frequency, parent_label_id) SELECT label_id, creator_id, updated_by, name, description, created_at, updated_at, post_count, visibility_status, is_active, usage_frequency, parent_label_id FROM tbl_labels; DROP TABLE tbl_labels; ALTER TABLE tbl_labels_new RENAME TO tbl_labels; COMMIT;" + ] + }, + "tags": [ + "DDL", + "difficulty: simple", + "ALTER", + "ADD_COLUMN", + "JSON" + ], + "other": { + "Comment": "LGTM", + "nl_prompt_base": "Update the labels table to include a json type field called comments_description.", + "nl_prompt_extra_context": null, + "public": true + } + }, + { + "id": 185, + "nl_prompt": "Rename the content column in the tbl_posts table to post_content", + "query_type": "ddl", + "database": "db_blog", + "dialects": [ + "sqlite" + ], + "golden_sql": { + "sqlite": [ + "ALTER TABLE tbl_posts RENAME COLUMN content TO post_content;" + ] + }, + "eval_query": { + "sqlite": [ + "SELECT post_content FROM tbl_posts;" + ] + }, + "setup_sql": { + "sqlite": [ + "ALTER TABLE tbl_posts RENAME COLUMN post_content TO content;" + ] + }, + "cleanup_sql": { + "sqlite": [ + "ALTER TABLE tbl_posts RENAME COLUMN post_content TO content;" + ] + }, + "tags": [ + "DDL", + "difficulty: simple", + "ALTER", + "RENAME" + ], + "other": { + "comment": "LGTM", + "nl_prompt_base": "Rename the content column in the tbl_posts table to post_content", + "nl_prompt_extra_context": null, + "public": true + } } ] diff --git a/.ci/release/release_smoke_config.yaml b/.ci/release/release_smoke_config.yaml index 41472115..e5f621c3 100644 --- a/.ci/release/release_smoke_config.yaml +++ b/.ci/release/release_smoke_config.yaml @@ -1,7 +1,9 @@ -# Release smoke gate -- one-shot NL2SQL leg. -# Uses sqlite only to avoid external database dependencies. -# Paths are relative to /evalbench, the container WORKDIR. +# Release smoke eval. The release pipeline sends it to the gRPC server in +# the candidate image. SQLite only, so it needs no external database. Paths +# are relative to /evalbench. dataset_config: .ci/release/release_smoke.evalset.json +# One row per prompt, so the verifier can check coverage. +num_trials: 1 database_configs: - datasets/bat/db_configs/sqlite.yaml @@ -9,21 +11,23 @@ dialects: - sqlite query_types: - dql + - dml + - ddl -# Build db_blog from datasets/bat/setup/db_blog/sqlite/*.sql on every run. +# Rebuild db_blog on every run, because DML and DDL prompts change it. setup_directory: datasets/bat/setup model_config: datasets/model_configs/gemini_2.5_pro_model.yaml prompt_generator: 'SQLGenBasePromptGenerator' -# verify_nl2sql_smoke.py enforces score > 0 for returned_sql and executable_sql. -# exact_match and set_match check liveness only. +# The verifier needs a score from every scorer, not a high score. scorers: returned_sql: null executable_sql: null - exact_match: null set_match: null + llmrater: + model_config: datasets/model_configs/gemini_2.5_pro_model.yaml reporting: csv: - output_directory: 'results/release/oneshot' + output_directory: 'results/release' diff --git a/.ci/release/verify_nl2sql_smoke.py b/.ci/release/verify_nl2sql_smoke.py deleted file mode 100644 index 6d6de334..00000000 --- a/.ci/release/verify_nl2sql_smoke.py +++ /dev/null @@ -1,186 +0,0 @@ -#!/usr/bin/env python3 -"""Gates the weekly release build on the smoke run's scores.csv. - -evalbench exits 0 when a run completes, so exit code alone cannot gate a -release. This script audits scores.csv in two tiers: - - Tier 1 (score > 0): structural scorers (returned_sql, executable_sql). - - Tier 2 (liveness): scorer produced a finite number with no comparison_error. - -When eval runs via eval_server.py, CsvReporter writes to -/tmp_session_files/results. Pass --results-dir to override the config path. -""" -import argparse -import ast -import csv -import json -import math -import os -import re -import sys - -from pyaml_env import parse_config - -# CsvReporter names each trial row "_trial_". -_TRIAL_SUFFIX = re.compile(r"^(?P.+)_trial_\d+$") - -LEGS = { - "oneshot": { - "config": ".ci/release/release_smoke_config.yaml", - "positive": {"returned_sql", "executable_sql"}, - }, -} - - -def base_id(row_id): - """Strips the _trial_N suffix from a row id.""" - match = _TRIAL_SUFFIX.match(row_id or "") - return match.group("base") if match else (row_id or "") - - -def row_dialect(raw): - """scores.csv stores the dialects list as its repr, e.g. "['sqlite']".""" - try: - parsed = ast.literal_eval(raw or "") - except (ValueError, SyntaxError): - parsed = raw - if isinstance(parsed, (list, tuple)): - return ",".join(str(dialect) for dialect in parsed) - return str(parsed or "") - - -def expected_keys(dataset_config, config_dialects): - """Returns the {(dialect, scenario id)} pairs the run should score. - - A scenario runs once per dialect, and dataset.py intersects the scenario's - own dialects with config['dialects'] when that list is non-empty. - """ - with open(dataset_config) as f: - data = json.load(f) - items = data["scenarios"] if isinstance(data, dict) else data - keys = set() - for item in items: - dialects = item.get("dialects") or [] - if config_dialects: - dialects = [d for d in dialects if d in config_dialects] - keys.update((dialect, str(item["id"])) for dialect in dialects) - return keys - - -def job_dirs(results_dir): - """Returns job directories under results_dir, newest first.""" - if not os.path.isdir(results_dir): - return [] - dirs = [os.path.join(results_dir, d) for d in os.listdir(results_dir)] - dirs = [d for d in dirs if os.path.isdir(d)] - return sorted(dirs, key=os.path.getmtime, reverse=True) - - -def load_rows(job_dir): - """Returns {(dialect, comparator, base_id): (score_or_None, error)}, or None.""" - path = os.path.join(job_dir, "scores.csv") - if not os.path.exists(path): - return None - rows = {} - with open(path, newline="") as f: - for row in csv.DictReader(f): - try: - score = float(row["score"]) - if not math.isfinite(score): - score = None - except (KeyError, TypeError, ValueError): - score = None - key = (row_dialect(row.get("dialects")), - row.get("comparator"), - base_id(row.get("id"))) - rows[key] = (score, (row.get("comparison_error") or "").strip()) - return rows - - -def find_job(results_dir, keys): - """Returns (job_dir, rows) for the newest job covering every expected key.""" - for job_dir in job_dirs(results_dir): - rows = load_rows(job_dir) - if rows is None: - continue - if keys <= {(dialect, row_id) for dialect, _, row_id in rows}: - return job_dir, rows - return None, None - - -def check(leg_name, leg, results_dir_override): - config = parse_config(leg["config"]) - scorers = sorted(config.get("scorers") or {}) - results_dir = results_dir_override or config["reporting"]["csv"][ - "output_directory"] - keys = expected_keys(config["dataset_config"], config.get("dialects") or []) - if not keys: - return [f"{config['dataset_config']} has no scenarios matching " - f"dialects {config.get('dialects')}"], 0, scorers, None - - job_dir, rows = find_job(results_dir, keys) - if job_dir is None: - return [f"no scores.csv under {results_dir} covering " - f"{sorted(keys)}"], 0, scorers, None - - problems = [] - checked = 0 - for scorer in scorers: - for dialect, scenario_id in sorted(keys): - target = f"{scenario_id} [{dialect}]" - entry = rows.get((dialect, scorer, scenario_id)) - if entry is None: - problems.append(f"{scorer}: no row for {target}") - continue - score, error = entry - checked += 1 - if error: - problems.append( - f"{scorer}: errored on {target} -- {error[:120]}") - elif score is None: - problems.append( - f"{scorer}: non-numeric score for {target}") - elif scorer in leg["positive"] and score <= 0: - problems.append( - f"{scorer}: {target} reported 0 -- the released " - f"image could not complete this step") - return problems, checked, scorers, job_dir - - -def main(): - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--leg", action="append", choices=sorted(LEGS), - help="Leg to verify. Repeatable. Defaults to all.") - parser.add_argument("--results-dir", - help="Overrides reporting.csv.output_directory. " - "Required when the run went through " - "eval_server.py.") - args = parser.parse_args() - - legs = args.leg or sorted(LEGS) - failed = [] - for leg_name in legs: - leg = LEGS[leg_name] - problems, checked, scorers, job_dir = check( - leg_name, leg, args.results_dir) - print(f"{leg_name} ({leg['config']})") - print(f" must be > 0: {', '.join(sorted(leg['positive']))}") - if job_dir: - print(f" job: {job_dir}") - if problems: - failed.append(leg_name) - print(" [FAIL]") - for problem in problems: - print(f" {problem}") - else: - print(f" [PASS] {checked} checks across {len(scorers)} scorers") - print() - - if failed: - print(f"FAILED: {', '.join(failed)}") - return 1 - print("Release smoke gate passed.") - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/.ci/release/verify_release_smoke.py b/.ci/release/verify_release_smoke.py new file mode 100644 index 00000000..a0575a93 --- /dev/null +++ b/.ci/release/verify_release_smoke.py @@ -0,0 +1,290 @@ +#!/usr/bin/env python3 +"""Grades the release smoke eval from its CSV reports. + +A clean run has: + 1. All four report files, each with rows. + 2. Exactly one eval row for each (dialect, prompt). + 3. No prompt generator error and no golden query error. + 4. Generated SQL and no SQL generator error on every row. + 5. A numeric score with no comparison_error from every scorer on every row. + 6. At least one executed query for each dialect and query type. +Accuracy is printed but not gated, because model variance makes it flaky. + +Exit codes: 0 clean, 1 hard failure, 2 only model-dependent checks failed +(4, 6, or LLM judge errors), so a retry can help. + +Local test (the CLI runner writes to results/release): + START=$(date +%s) + EVAL_CONFIG=.ci/release/release_smoke_config.yaml ./evalbench/run.sh + uv run python .ci/release/verify_release_smoke.py --since "$START" +""" +import argparse +import ast +import csv +import json +import math +import os +import sys + +from pyaml_env import parse_config + +DEFAULT_CONFIG = ".ci/release/release_smoke_config.yaml" +REPORT_FILES = ("configs.csv", "evals.csv", "scores.csv", "summary.csv") +HARD_ERROR_COLUMNS = ("prompt_generator_error", "golden_error") +# Scorers that call a model. Their errors are model-dependent. +LLM_SCORERS = {"llmrater"} +EMPTY = {"", "nan", "none", "null"} + +HARD = 1 +RETRYABLE = 2 + + +def is_empty(raw): + return (raw or "").strip().lower() in EMPTY + + +def as_score(raw): + try: + score = float(raw) + except (TypeError, ValueError): + return None + return score if math.isfinite(score) else None + + +def as_prompt_id(raw): + """pandas can write id 9 as "9.0".""" + text = (raw or "").strip() + try: + return str(int(float(text))) + except ValueError: + return text + + +def as_dialect(raw): + """The CSV stores the dialects list as its repr, e.g. "['sqlite']".""" + try: + parsed = ast.literal_eval(raw or "") + except (ValueError, SyntaxError): + parsed = raw + if isinstance(parsed, (list, tuple)): + return ",".join(str(d) for d in parsed) + return str(parsed or "") + + +def expected_prompts(config): + """Returns {(dialect, prompt id): query type} that the run must cover. + + Filters like dataset.py: by the run config dialects and query types, if + the lists are not empty. + """ + with open(config["dataset_config"]) as f: + data = json.load(f) + items = data["scenarios"] if isinstance(data, dict) else data + dialects = config.get("dialects") or [] + query_types = [q.lower() for q in config.get("query_types") or []] + expected = {} + for item in items: + query_type = item["query_type"].lower() + if query_types and query_type not in query_types: + continue + for dialect in item.get("dialects") or []: + if not dialects or dialect in dialects: + expected[(dialect, str(item["id"]))] = query_type + return expected + + +def job_dirs(results_dir, since): + """Returns job directories changed after `since`, newest first.""" + if not os.path.isdir(results_dir): + return [] + dirs = [os.path.join(results_dir, d) for d in os.listdir(results_dir)] + dirs = [d for d in dirs + if os.path.isdir(d) and os.path.getmtime(d) >= since] + return sorted(dirs, key=os.path.getmtime, reverse=True) + + +def read_rows(job_dir, name): + with open(os.path.join(job_dir, name), newline="") as f: + return list(csv.DictReader(f)) + + +def eval_key(row): + return (as_dialect(row.get("dialects")), + as_prompt_id(row.get("prompt_id") or row.get("id"))) + + +def find_job(results_dir, since, expected): + """Returns the newest job whose evals.csv covers an expected prompt.""" + for job_dir in job_dirs(results_dir, since): + path = os.path.join(job_dir, "evals.csv") + if not os.path.exists(path): + continue + keys = {eval_key(row) for row in read_rows(job_dir, "evals.csv")} + if keys & expected.keys(): + return job_dir + return None + + +class Problems: + """Collects failures and remembers whether any of them is hard.""" + + def __init__(self): + self.items = [] + self.hard = False + + def add(self, message, retryable=False): + self.items.append(message + (" [model]" if retryable else "")) + self.hard = self.hard or not retryable + + +def check_reports(job_dir, problems): + for name in REPORT_FILES: + path = os.path.join(job_dir, name) + if not os.path.exists(path): + problems.add(f"{name} is missing") + elif not read_rows(job_dir, name): + problems.add(f"{name} has no rows") + + +def check_coverage(expected, evals, problems): + counts = {} + for row in evals: + counts[eval_key(row)] = counts.get(eval_key(row), 0) + 1 + for key in sorted(expected.keys() - counts.keys()): + problems.add(f"no eval row for prompt {key[1]} [{key[0]}]") + for key in sorted(counts.keys() - expected.keys()): + problems.add(f"eval row for unexpected prompt {key[1]} [{key[0]}]") + for key, count in sorted(counts.items()): + if count > 1: + problems.add(f"{count} eval rows for prompt {key[1]} [{key[0]}]") + + +def check_errors(evals, problems): + for row in evals: + target = "{1} [{0}]".format(*eval_key(row)) + for column in HARD_ERROR_COLUMNS: + if not is_empty(row.get(column)): + problems.add( + f"{target}: {column}: {row[column].strip()[:200]}") + if not is_empty(row.get("sql_generator_error")): + problems.add(f"{target}: sql_generator_error: " + f"{row['sql_generator_error'].strip()[:200]}", + retryable=True) + elif is_empty(row.get("generated_sql")): + problems.add(f"{target}: the model returned no SQL", + retryable=True) + + +def check_scores(scorers, evals, scores, problems): + by_key = {} + for row in scores: + by_key[(row.get("comparator"), row.get("id"), + as_dialect(row.get("dialects")))] = row + for scorer in scorers: + retryable = scorer in LLM_SCORERS + for eval_row in evals: + dialect, prompt_id = eval_key(eval_row) + target = f"{prompt_id} [{dialect}]" + row = by_key.get((scorer, eval_row.get("id"), dialect)) + if row is None: + problems.add(f"{scorer}: no score row for {target}") + continue + error = row.get("comparison_error") + if not is_empty(error): + problems.add(f"{scorer}: errored on {target}: " + f"{error.strip()[:200]}", retryable=retryable) + elif as_score(row.get("score")) is None: + problems.add(f"{scorer}: non-numeric score for {target}", + retryable=retryable) + + +def check_execution(expected, evals, scores, problems): + """Requires one executed generated query per dialect and query type.""" + executed = {(r.get("id"), as_dialect(r.get("dialects"))) + for r in scores + if r.get("comparator") == "executable_sql" + and (as_score(r.get("score")) or 0) > 0} + groups = {} + for row in evals: + key = eval_key(row) + query_type = expected.get(key) + if query_type is None: + continue + group = groups.setdefault((key[0], query_type), []) + group.append((row.get("id"), key[0]) in executed) + for (dialect, query_type), results in sorted(groups.items()): + if not any(results): + problems.add(f"executable_sql: no generated {query_type} query " + f"executed on {dialect}", retryable=True) + + +def print_accuracy(scorers, scores): + print("Accuracy (reported, not gated):") + for scorer in scorers: + values = [as_score(r.get("score")) for r in scores + if r.get("comparator") == scorer] + values = [v for v in values if v is not None] + if values: + print(f" {scorer:<16} mean {sum(values) / len(values):6.2f} " + f"over {len(values)} rows") + + +def verify(config_path, results_dir, since): + config = parse_config(config_path) + scorers = sorted(config.get("scorers") or {}) + results_dir = results_dir or config["reporting"]["csv"]["output_directory"] + expected = expected_prompts(config) + if not expected: + print(f"[FAIL] {config['dataset_config']} has no prompts for " + f"dialects {config.get('dialects')} and query types " + f"{config.get('query_types')}") + return HARD + + job_dir = find_job(results_dir, since, expected) + if job_dir is None: + print(f"[FAIL] no evals.csv under {results_dir} covers the dataset") + return HARD + print(f"Job: {job_dir}") + + problems = Problems() + check_reports(job_dir, problems) + if not problems.items: + evals = read_rows(job_dir, "evals.csv") + scores = read_rows(job_dir, "scores.csv") + print(f"Eval rows: {len(evals)}, scorers: {', '.join(scorers)}\n") + check_coverage(expected, evals, problems) + check_errors(evals, problems) + check_scores(scorers, evals, scores, problems) + check_execution(expected, evals, scores, problems) + print_accuracy(scorers, scores) + print() + + if not problems.items: + print(f"[PASS] clean run: {len(expected)} prompts, " + f"{len(scorers)} scorers") + return 0 + kind = "hard" if problems.hard else "model-dependent" + print(f"[FAIL] {len(problems.items)} problem(s), {kind}") + for problem in problems.items: + print(f" {problem}") + return HARD if problems.hard else RETRYABLE + + +def main(): + parser = argparse.ArgumentParser( + description=__doc__, formatter_class=argparse.RawTextHelpFormatter) + parser.add_argument("--config", default=DEFAULT_CONFIG, + help="Run config of the smoke eval.") + parser.add_argument("--results-dir", + help="Overrides reporting.csv.output_directory. " + "The gRPC server writes to " + "/tmp_session_files/results.") + parser.add_argument("--since", type=float, default=0.0, + help="Ignore job directories older than this Unix " + "time, so a retry does not grade an old job.") + args = parser.parse_args() + return verify(args.config, args.results_dir, args.since) + + +if __name__ == "__main__": + sys.exit(main()) From 78a611b8d449d7bb95d25af8b5f14b99963d4ceb Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad Date: Wed, 30 Sep 2026 06:02:41 +0000 Subject: [PATCH 15/27] feat: add automated release notes generation and update alert policy to include PR comparisons --- .ci/release.cloudbuild.yaml | 77 +++++++++++++++++--- .ci/release/alert_policy.yaml | 22 ++++-- .ci/release/lib.sh | 10 ++- .ci/release/release_notes.py | 128 ++++++++++++++++++++++++++++++++++ .ci/release/wait_for_grpc.py | 6 +- 5 files changed, 223 insertions(+), 20 deletions(-) create mode 100644 .ci/release/release_notes.py diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index 5470c649..d1090f8a 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -27,7 +27,8 @@ steps: release_fail $$STAGE "COMMIT_SHA or SHORT_SHA is empty." fi - # Two runs race on GKE and :latest. A failed list call does not block. + # Stop if another run is active, because two runs race on GKE and + # :latest. If the list call fails, continue. if [[ -n "$TRIGGER_ID" ]]; then OTHERS=$$(gcloud builds list --ongoing --region="$LOCATION" \ --filter="buildTriggerId=$TRIGGER_ID AND id!=$BUILD_ID" \ @@ -54,6 +55,65 @@ steps: echo "digest = $$DIGEST" echo "tag = $$TAG" + # Lists the PRs merged since the release that GKE serves now. lib.sh adds + # the list to the result line. Runs in parallel with the smoke gate. Every + # failure writes a reason and exits 0, so this step never fails the release. + - id: release-notes + name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' + waitFor: ['resolve-release'] + entrypoint: 'bash' + args: + - '-c' + - | + set -uo pipefail + NOTES=/workspace/release_prs.txt + GITHUB=https://github.com/GoogleCloudPlatform/evalbench + + unavailable() { + echo "PR list unavailable: $$1" | tee "$$NOTES" + exit 0 + } + + if ! gcloud container clusters get-credentials \ + evalbench-directpath-cluster \ + --zone us-central1-c --project "$PROJECT_ID" > /dev/null 2>&1; then + unavailable "no credentials for evalbench-directpath-cluster." + fi + OLD_DIGEST=$$(kubectl get pods -n evalbench-namespace \ + -l app=evalbench-eval-server \ + -o jsonpath="{range .items[*]}{.status.containerStatuses[?(@.name=='evalbench-eval')].imageID}{'\n'}{end}" \ + | sed -e 's#^.*://##' -e 's#^.*@##' \ + | grep -v '^$$' | sort -u) + if [[ $$(echo "$$OLD_DIGEST" | grep -c .) -ne 1 ]]; then + unavailable "GKE does not serve exactly one digest." + fi + + # The Build trigger tags each image with its full commit SHA. + OLD_SHA=$$(gcloud artifacts tags list --package=eval_server \ + --repository=evalbench --location=us-central1 \ + --project="$PROJECT_ID" --filter="version~$$OLD_DIGEST" \ + --format='value(name.basename())' 2> /dev/null \ + | grep -E '^[0-9a-f]{40}$$' | head -1) + if [[ -z "$$OLD_SHA" ]]; then + unavailable "the previous commit is unknown. GKE serves $$OLD_DIGEST, which has no commit SHA tag." + fi + echo "$$GITHUB/compare/$$OLD_SHA...$COMMIT_SHA" \ + > /workspace/release_compare.txt + + # A treeless clone has every commit message and no file contents. + if ! git clone --quiet --filter=tree:0 --no-checkout \ + "$$GITHUB.git" /tmp/evalbench; then + unavailable "git clone of $$GITHUB failed." + fi + if ! python3 /workspace/.ci/release/release_notes.py \ + --repo-dir /tmp/evalbench --old "$$OLD_SHA" --new "$COMMIT_SHA" \ + > /tmp/prs.txt; then + unavailable "git log $$OLD_SHA..$COMMIT_SHA failed." + fi + mv /tmp/prs.txt "$$NOTES" + echo "PRs since $$OLD_SHA:" + cat "$$NOTES" + # Serves the candidate image over gRPC and runs the smoke eval on it. # eval_client exits 0 for any completed run, so the verifier grades the CSVs. - id: smoke-gate @@ -85,7 +145,6 @@ steps: tail -60 "$$SERVER_LOG" } - echo "===== wait for gRPC =====" if ! python .ci/release/wait_for_grpc.py --timeout 300 --insecure; then dump_log release_fail $$STAGE "The eval server did not answer Ping within 300 s." @@ -156,11 +215,10 @@ steps: release_fail $$STAGE ":$COMMIT_SHA moved from $$DIGEST to $$CURRENT during the run. Start the release again." fi - echo "Tagging $$REPO@$$DIGEST as :$$TAG" if ! move_tag "$$REPO" "$$TAG" "$$DIGEST"; then release_fail $$STAGE "Could not apply the release tag $$TAG." fi - echo "Release tag: $$REPO:$$TAG" + echo "Tagged $$REPO@$$DIGEST as :$$TAG" # Records the digest that GKE serves now. The pod imageID is exact, but # the Deployment spec can hold a tag. @@ -201,8 +259,8 @@ steps: echo "$$RUNNING" > /workspace/rollback_digest.txt echo "rollback target = $$RUNNING" - # Deploy, check, and rollback share one step. Cloud Build has no - # on-failure hook, so a separate rollback step never runs. + # Deploy, check, and rollback share one step, so a failed check can roll + # back. See the header. - id: deploy-gke name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' waitFor: ['capture-rollback-target'] @@ -244,7 +302,6 @@ steps: # Pin by digest, so a tag move cannot change the running image. pin() { - echo "Pinning $$CONTAINER to $$REPO@$$1" kubectl set image "$$DEPLOYMENT" "$$CONTAINER=$$REPO@$$1" \ -n "$$NAMESPACE" \ && kubectl rollout status "$$DEPLOYMENT" -n "$$NAMESPACE" \ @@ -270,7 +327,7 @@ steps: fi # Old pods can stay Terminating for a few seconds after the rollout. - echo "===== confirm the pods serve $$NEW_DIGEST =====" + echo "===== confirm the pods serve the new digest =====" for ATTEMPT in {1..20}; do SERVING=$$(running_digests) [[ "$$SERVING" == "$$NEW_DIGEST" ]] && break @@ -283,9 +340,9 @@ steps: # The Deployment has no readiness probe, so Ready means only that the # process started. Ping is the real health check. - echo "===== gRPC health check =====" POD=$$(kubectl get pods -n "$$NAMESPACE" -l "$$SELECTOR" \ -o jsonpath='{.items[0].metadata.name}') + echo "===== gRPC health check on $$POD =====" if ! kubectl exec -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" -- \ python /evalbench/.ci/release/wait_for_grpc.py --timeout 300; then kubectl logs -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" --tail=150 @@ -401,7 +458,7 @@ steps: - id: notify-success name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' - waitFor: ['deploy-cloudrun'] + waitFor: ['deploy-cloudrun', 'release-notes'] entrypoint: 'bash' args: - '-c' diff --git a/.ci/release/alert_policy.yaml b/.ci/release/alert_policy.yaml index 2798ede1..3dfd09a7 100644 --- a/.ci/release/alert_policy.yaml +++ b/.ci/release/alert_policy.yaml @@ -1,9 +1,12 @@ # Log-based alert that emails the RELEASE_RESULT line (lib.sh) of each # weekly release run. # -# Replace TRIGGER_ID and CHANNEL_ID, then create the policy: -# gcloud alpha monitoring policies create --project=cloud-db-nl2sql \ -# --policy-from-file=.ci/release/alert_policy.yaml +# TRIGGER_ID and CHANNEL_NAME are placeholders, so this public file has no +# project resource IDs. The live policy already has the real values: +# TRIGGER_ID id of the trigger 'weekly-evalbench-release' (us-central1) +# CHANNEL_NAME name of the notification channel 'EvalBench release' +# To change the policy, fill in a copy of this file and run +# 'gcloud alpha monitoring policies update'. Do not create a second policy. displayName: 'EvalBench weekly release' combiner: OR conditions: @@ -18,7 +21,10 @@ conditions: stage: 'REGEXP_EXTRACT(textPayload, "stage=(\\S+)")' tag: 'REGEXP_EXTRACT(textPayload, "tag=(\\S+)")' commit: 'REGEXP_EXTRACT(textPayload, "commit=(\\S+)")' - log: 'REGEXP_EXTRACT(textPayload, "log=(\\S+)")' + # 'log' is a reserved label name. + build_log: 'REGEXP_EXTRACT(textPayload, "log=(\\S+)")' + compare: 'REGEXP_EXTRACT(textPayload, "compare=(\\S+)")' + prs: 'REGEXP_EXTRACT(textPayload, "prs=\"([^\"]*)\"")' detail: 'REGEXP_EXTRACT(textPayload, "detail=\"(.*)\"")' alertStrategy: # 5 minutes is the minimum. A run prints one line, so no result is lost. @@ -37,8 +43,12 @@ documentation: **Detail:** ${log.extracted_label.detail} + **PRs since the previous release:** ${log.extracted_label.prs} + + **Compare:** ${log.extracted_label.compare} + **Commit:** ${log.extracted_label.commit} - **Build log:** ${log.extracted_label.log} + **Build log:** ${log.extracted_label.build_log} notificationChannels: - - projects/cloud-db-nl2sql/notificationChannels/CHANNEL_ID + - CHANNEL_NAME diff --git a/.ci/release/lib.sh b/.ci/release/lib.sh index 8e38f5e4..d6598bd0 100644 --- a/.ci/release/lib.sh +++ b/.ci/release/lib.sh @@ -13,12 +13,20 @@ RELEASE_RESULT_PREFIX="RELEASE_RESULT" # STATUS SUCCESS, FAILED, ROLLED_BACK, or ROLLBACK_FAILED # STAGE the id of the step that ended the run # DETAIL one sentence for the email. Double quotes become single quotes. +# +# The release-notes step writes the PR list and the compare link. If the step +# did not finish, the line says so. detail stays last, because the alert +# policy reads it with a greedy regex. release_result() { _tag=$(cat /workspace/release_tag.txt 2>/dev/null || echo "none") _detail=$(printf '%s' "$3" | tr '"\n' "' ") + _prs=$(cat /workspace/release_prs.txt 2>/dev/null \ + || echo "PR list not computed before this step ended the run.") + _prs=$(printf '%s' "$_prs" | tr '"\n' "' ") + _compare=$(cat /workspace/release_compare.txt 2>/dev/null || echo "none") _log="https://console.cloud.google.com/cloud-build/builds;region=${LOCATION}/${BUILD_ID}?project=${PROJECT_ID}" _commit="https://github.com/GoogleCloudPlatform/evalbench/commit/${COMMIT_SHA}" - echo "${RELEASE_RESULT_PREFIX} status=$1 stage=$2 tag=${_tag} commit=${_commit} log=${_log} detail=\"${_detail}\"" + echo "${RELEASE_RESULT_PREFIX} status=$1 stage=$2 tag=${_tag} commit=${_commit} log=${_log} compare=${_compare} prs=\"${_prs}\" detail=\"${_detail}\"" } # Prints a FAILED result line and ends the step with an error. diff --git a/.ci/release/release_notes.py b/.ci/release/release_notes.py new file mode 100644 index 00000000..e7342b0e --- /dev/null +++ b/.ci/release/release_notes.py @@ -0,0 +1,128 @@ +#!/usr/bin/env python3 +"""Prints the PRs merged between two commits as one line for the release email. + +The release email comes from a log-based alert, so the list must fit on the +RELEASE_RESULT line. The script groups PRs by the conventional-commit type of +the title, for example "feat(agy): ..." or "ci: ...". + +Usage: + release_notes.py --repo-dir DIR --old OLD_SHA --new NEW_SHA + +Exit codes: 0 with the list on stdout, 1 if git fails. +""" +import argparse +import re +import subprocess +import sys + +# Group order in the email. Other types go to "other". +TYPE_ORDER = ("feat", "fix", "perf", "refactor", "ci", "test", "docs", + "build", "chore") +# The alert label keeps about 1,024 characters. Leave room for the suffix. +MAX_CHARS = 900 +MAX_TITLE = 70 + +SQUASH_RE = re.compile(r"^(?P.*?)\s*\(#(?P<pr>\d+)\)$") +MERGE_RE = re.compile(r"^Merge pull request #(?P<pr>\d+) from \S+") +TYPE_RE = re.compile( + r"^(?P<type>[a-zA-Z]+)(?P<scope>\([^)]*\))?!?:\s*(?P<desc>.+)$") + + +def parse_commit(subject, body): + """Returns (pr, title) for a PR merge commit, or None.""" + match = SQUASH_RE.match(subject) + if match: + return int(match.group("pr")), match.group("title") + match = MERGE_RE.match(subject) + if match: + lines = [line.strip() for line in body.splitlines() if line.strip()] + return int(match.group("pr")), (lines[0] if lines else subject) + return None + + +def group_of(title): + """Returns (type, short title) for a PR title. + + "feat(agy): add x" becomes ("feat", "agy: add x"). A title without a + known type goes to ("other", title). + """ + match = TYPE_RE.match(title) + if not match or match.group("type").lower() not in TYPE_ORDER: + return "other", title + scope = (match.group("scope") or "").strip("()") + desc = match.group("desc") + return match.group("type").lower(), f"{scope}: {desc}" if scope else desc + + +def shorten(title): + title = title.replace('"', "'").strip() + if len(title) > MAX_TITLE: + title = title[:MAX_TITLE - 3].rstrip(" .,") + "..." + return title + + +def format_prs(prs): + """Formats [(pr, title)] as "3 PRs. feat (1): #1 a. fix (2): #2 b, #3 c." + + Stops at MAX_CHARS on a whole PR and adds "+N more". + """ + if not prs: + return "No new PRs." + groups = {} + for pr, title in prs: + kind, short = group_of(title) + groups.setdefault(kind, []).append(f"#{pr} {shorten(short)}") + + text = f"{len(prs)} PRs." + shown = 0 + for kind in TYPE_ORDER + ("other",): + items = groups.get(kind, []) + if not items: + continue + head = f" {kind} ({len(items)}): " + for i, item in enumerate(items): + piece = (head if i == 0 else ", ") + item + if len(text) + len(piece) > MAX_CHARS: + return f"{text} ... +{len(prs) - shown} more in the compare link." + text += piece + shown += 1 + text += "." + return text + + +def merged_prs(repo_dir, old, new): + """Returns [(pr, title)] for first-parent commits in old..new, newest first.""" + out = subprocess.run( + ["git", "-C", repo_dir, "log", "--first-parent", + "--format=%s%x1f%b%x1e", f"{old}..{new}"], + check=True, capture_output=True, text=True).stdout + prs = [] + for record in out.split("\x1e"): + if not record.strip(): + continue + subject, _, body = record.strip("\n").partition("\x1f") + parsed = parse_commit(subject.strip(), body) + if parsed: + prs.append(parsed) + return prs + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--repo-dir", required=True, + help="Git checkout that contains both commits.") + parser.add_argument("--old", required=True, + help="Commit that GKE served before the release.") + parser.add_argument("--new", required=True, help="Release commit.") + args = parser.parse_args() + try: + prs = merged_prs(args.repo_dir, args.old, args.new) + except subprocess.CalledProcessError as e: + print(f"git log failed: {e.stderr.strip()}", file=sys.stderr) + return 1 + print(format_prs(prs)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.ci/release/wait_for_grpc.py b/.ci/release/wait_for_grpc.py index e91ea6d7..4725185c 100644 --- a/.ci/release/wait_for_grpc.py +++ b/.ci/release/wait_for_grpc.py @@ -3,7 +3,7 @@ Exit codes: 0 the server answered Ping - 1 the deadline passed or a permanent security mismatch occurred + 1 the deadline passed, or ALTS handshakes failed repeatedly """ import argparse import asyncio @@ -50,7 +50,7 @@ async def wait(timeout: float, interval: float) -> int: if "Alts handshake failed" in last_error: alts_failures += 1 if alts_failures >= _ALTS_FAILURE_LIMIT: - print(f"\nAborting after {alts_failures} consecutive ALTS " + print(f"Aborting after {alts_failures} consecutive ALTS " f"handshake failures.") print("Pass --insecure, or set EVALBENCH_INSECURE=true.") return 1 @@ -61,7 +61,7 @@ async def wait(timeout: float, interval: float) -> int: await asyncio.sleep(interval) print(f"No Ping within {timeout:.0f}s after {attempt} attempts.") - print(f"Last error -- {last_error}") + print(f"Last error: {last_error}") return 1 From 0129e11424cac255f4294b4aacd70f428fcbd378 Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad <omkargaikwad@google.com> Date: Wed, 30 Sep 2026 06:41:12 +0000 Subject: [PATCH 16/27] ci(release): gate the client-generated SQL path in the smoke gate Most production clients generate SQL themselves and send it back. The server uses the noop generator and reads its configs from EvalConfig resources. The smoke gate tested only server-side generation. - Add client_sql_smoke.py. It sends fixed SQL through EvalConfig resources, the noop generator, and batch and streaming Eval. Golden SQL must score 100. Wrong SQL must score 0 on exact_match and set_match. No model is called, so a failure blocks the release with no retry. - Fail the gate on ERROR and CRITICAL server log lines, as well as on tracebacks. - Add the exact_match scorer to the model smoke eval. --- .ci/release.cloudbuild.yaml | 28 +++- .ci/release/client_sql_smoke.py | 186 ++++++++++++++++++++++++++ .ci/release/release_smoke_config.yaml | 1 + 3 files changed, 208 insertions(+), 7 deletions(-) create mode 100644 .ci/release/client_sql_smoke.py diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index d1090f8a..181374c2 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -114,8 +114,10 @@ steps: echo "PRs since $$OLD_SHA:" cat "$$NOTES" - # Serves the candidate image over gRPC and runs the smoke eval on it. - # eval_client exits 0 for any completed run, so the verifier grades the CSVs. + # Serves the candidate image over gRPC and runs two checks on it: + # 1. The smoke eval: the server generates SQL with a model. eval_client + # exits 0 for any completed run, so the verifier grades the CSVs. + # 2. client_sql_smoke.py: the client-generated SQL path, with fixed scores. - id: smoke-gate name: 'us-central1-docker.pkg.dev/$PROJECT_ID/evalbench/eval_server:$COMMIT_SHA' dir: '/evalbench' @@ -177,13 +179,25 @@ steps: release_fail $$STAGE "The smoke eval failed on attempt $$ATTEMPT. The build log has the verifier output." done - # Scan only the passing attempt. Any traceback fails the gate, also - # one from a handled exception. - EXCEPTIONS=$$(tail -c +$$((LOG_OFFSET + 1)) "$$SERVER_LOG" \ + # Client SQL, noop generator, EvalConfig resources, batch and + # streaming. Scores are fixed, so there is no retry. + echo "===== client SQL smoke =====" + if ! EVALBENCH_INSECURE=true python .ci/release/client_sql_smoke.py \ + --results-dir "$$RESULTS"; then + dump_log + release_fail $$STAGE "The client SQL smoke failed. The build log lists the problems." + fi + + # Scan from the passing attempt to the end. Any traceback or ERROR + # line fails the gate, also one from a handled exception. + SCAN=$$(tail -c +$$((LOG_OFFSET + 1)) "$$SERVER_LOG") + EXCEPTIONS=$$(echo "$$SCAN" \ | grep -n -A8 'Traceback (most recent call last)' | head -80) - if [[ -n "$$EXCEPTIONS" ]]; then + ERRORS=$$(echo "$$SCAN" | grep -nE '\] (ERROR|CRITICAL) ' | head -40) + if [[ -n "$$EXCEPTIONS$$ERRORS" ]]; then + echo "$$ERRORS" echo "$$EXCEPTIONS" - release_fail $$STAGE "The eval server logged exceptions during the smoke eval." + release_fail $$STAGE "The eval server logged errors or exceptions during the smoke evals." fi echo "===== smoke gate passed =====" diff --git a/.ci/release/client_sql_smoke.py b/.ci/release/client_sql_smoke.py new file mode 100644 index 00000000..f16150c8 --- /dev/null +++ b/.ci/release/client_sql_smoke.py @@ -0,0 +1,186 @@ +#!/usr/bin/env python3 +"""Runs the client-generated SQL path on the local eval server. + +Most production clients generate SQL themselves and send it back. The server +uses the noop generator, reads its configs from the EvalConfig resources, and +only executes and scores the SQL. This script does the same with fixed SQL, +so every score is known before the run: + + batch pass golden SQL on every item. Every scorer gives 100. + streaming pass wrong SQL on the last item. exact_match and set_match give + 0 for it, which proves that the scorers can fail. + +No model is called, so a failure is never model variance. + +Exit codes: 0 every score matches, 1 otherwise. + +Local test, with the server from 'python evalbench/eval_server.py --localhost': + EVALBENCH_INSECURE=true python .ci/release/client_sql_smoke.py +""" +import argparse +import asyncio +import os +import sys + +import yaml + +_HERE = os.path.dirname(os.path.abspath(__file__)) +_REPO = os.path.dirname(os.path.dirname(_HERE)) +# Add both module roots so eval_client and generated proto imports resolve. +for _path in (os.path.join(_REPO, "evalbench"), + os.path.join(_REPO, "evalbench", "evalproto")): + if _path not in sys.path: + sys.path.insert(0, _path) + +from client.eval_client import EvalbenchClient # noqa: E402 +from evalproto import eval_config_pb2, eval_connect_pb2 # noqa: E402 +from evalproto import eval_request_pb2 # noqa: E402 +from verify_release_smoke import ( # noqa: E402 + HARD_ERROR_COLUMNS, as_prompt_id, as_score, is_empty, read_rows) + +# The server replaces a config value that equals a resource address with the +# path of that resource in its session directory. +ROOT = "release_smoke" +DIALECT = "sqlite" +# Resource address -> repo file. None means the content is in RESOURCES_TEXT. +RESOURCES = { + f"{ROOT}/release_smoke.evalset.json": + ".ci/release/release_smoke.evalset.json", + f"{ROOT}/sqlite.yaml": "datasets/bat/db_configs/sqlite.yaml", + f"{ROOT}/noop_model.yaml": None, +} +RESOURCES_TEXT = {f"{ROOT}/noop_model.yaml": "generator: noop\n"} +CONFIG = { + "dataset_config": f"{ROOT}/release_smoke.evalset.json", + "dataset_format": "evalbench-standard-format", + "database_configs": [f"{ROOT}/sqlite.yaml"], + "dialects": [DIALECT], + "dialect": DIALECT, + # The orchestrator sends read queries only. DML goldens also write + # CURRENT_TIMESTAMP, so their scores are not fixed. + "query_types": ["dql"], + "setup_directory": "datasets/bat/setup", + "model_config": f"{ROOT}/noop_model.yaml", + "prompt_generator": "NOOPGenerator", + "num_trials": 1, + "scorers": {"exact_match": None, "set_match": None, + "executable_sql": None, "returned_sql": None}, + "reporting": {"csv": {"output_directory": "results"}}, +} +WRONG_SQL = "SELECT -1 AS wrong_answer;" +# Expected scores for the item with WRONG_SQL. Golden SQL gives 100 on all. +WRONG_SCORES = {"exact_match": 0, "set_match": 0, + "executable_sql": 100, "returned_sql": 100} + + +def config_request(): + request = eval_config_pb2.EvalConfigRequest( + yaml_config=yaml.safe_dump(CONFIG).encode("utf-8")) + for address, path in RESOURCES.items(): + if path is None: + content = RESOURCES_TEXT[address].encode("utf-8") + else: + with open(os.path.join(_REPO, path), "rb") as f: + content = f.read() + request.resources.add(address=address, content=content) + return request + + +async def run_pass(streaming, wrong_last): + """Returns (job_id, {item id: sent SQL}) for one Eval call.""" + client = EvalbenchClient("local") + metadata = client.metadata + try: + await client.stub.Ping(eval_request_pb2.PingRequest(), + metadata=metadata) + await client.stub.Connect( + eval_connect_pb2.EvalConnectRequest( + client_id="release-smoke", streaming_eval=streaming), + metadata=metadata) + await client.stub.EvalConfig(config_request(), metadata=metadata) + items = [item async for item in client.stub.ListEvalInputs( + eval_request_pb2.EvalInputRequest(), metadata=metadata)] + if not items: + raise RuntimeError("ListEvalInputs returned no items") + sent = {} + for i, item in enumerate(items): + golden = item.golden_sql[DIALECT].sql_statements[0] + wrong = wrong_last and i == len(items) - 1 + item.generated_sql = WRONG_SQL if wrong else golden + sent[str(item.id)] = item.generated_sql + response = await client.stub.Eval(iter(items), metadata=metadata) + return response.response, sent + finally: + await client.channel.close() + + +def check(name, results_dir, job_id, sent, wrong_id): + """Returns the problems in the reports of one pass.""" + job_dir = os.path.join(results_dir, job_id) + if not job_id or not os.path.isdir(job_dir): + return [f"no report directory {job_dir}"] + problems = [] + # prompt_id is the dataset id. id is the eval id that scores.csv uses. + evals = {as_prompt_id(r.get("prompt_id")): r + for r in read_rows(job_dir, "evals.csv")} + eval_ids = {} + for item_id, sql in sorted(sent.items()): + row = evals.get(item_id) + if row is None: + problems.append(f"{item_id}: no eval row") + continue + eval_ids[item_id] = as_prompt_id(row.get("id")) + for column in HARD_ERROR_COLUMNS + ("sql_generator_error", + "generated_error"): + if not is_empty(row.get(column)): + problems.append(f"{item_id}: {column}: " + f"{row[column].strip()[:200]}") + # The noop generator must keep the client SQL. + if (row.get("generated_sql") or "").strip() != sql.strip(): + problems.append(f"{item_id}: generated_sql is not the SQL that " + f"the client sent") + + scores = {(r.get("comparator"), as_prompt_id(r.get("id"))): r + for r in read_rows(job_dir, "scores.csv")} + for scorer in sorted(CONFIG["scorers"]): + for item_id, eval_id in sorted(eval_ids.items()): + want = WRONG_SCORES[scorer] if item_id == wrong_id else 100 + row = scores.get((scorer, eval_id)) + got = as_score(row.get("score")) if row else None + if got != want: + problems.append(f"{scorer}: {item_id} scored {got}, " + f"expected {want}") + status = "[PASS]" if not problems else "[FAIL]" + print(f"{status} {name}: job {job_id}, {len(sent)} items, " + f"{len(problems)} problem(s)") + for problem in problems: + print(f" {problem}") + return problems + + +async def run(results_dir): + failed = False + for name, streaming, wrong_last in (("batch pass", False, False), + ("streaming pass", True, True)): + try: + job_id, sent = await run_pass(streaming, wrong_last) + except Exception as e: + print(f"[FAIL] {name}: {type(e).__name__}: {e}") + failed = True + continue + wrong_id = list(sent)[-1] if wrong_last else None + failed |= bool(check(name, results_dir, job_id, sent, wrong_id)) + return 1 if failed else 0 + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + # CsvReporter forces this directory on the gRPC path. + parser.add_argument("--results-dir", default="/tmp_session_files/results", + help="Directory that holds <job_id>/scores.csv.") + args = parser.parse_args() + return asyncio.run(run(args.results_dir)) + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.ci/release/release_smoke_config.yaml b/.ci/release/release_smoke_config.yaml index e5f621c3..c44464dd 100644 --- a/.ci/release/release_smoke_config.yaml +++ b/.ci/release/release_smoke_config.yaml @@ -22,6 +22,7 @@ prompt_generator: 'SQLGenBasePromptGenerator' # The verifier needs a score from every scorer, not a high score. scorers: + exact_match: null returned_sql: null executable_sql: null set_match: null From 33cd12601fe4b318b428727acdf492dda234a0e1 Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad <omkargaikwad@google.com> Date: Wed, 30 Sep 2026 06:53:56 +0000 Subject: [PATCH 17/27] ci(release): scan only real error records in server logs Eval payloads can quote a traceback. The live GKE pod log had 114 such quoted tracebacks and 0 real ones. The old scan matched them, so one client request could roll back a healthy release. - Add log_problems to lib.sh. It matches ERROR and CRITICAL log records and tracebacks only at the start of a line. - Use it in smoke-gate and deploy-gke. deploy-gke now also fails on ERROR records, not only on tracebacks. - Clarify the release_result comment. --- .ci/release.cloudbuild.yaml | 22 +++++++++------------- .ci/release/lib.sh | 21 +++++++++++++++++---- 2 files changed, 26 insertions(+), 17 deletions(-) diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index 181374c2..34c52e39 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -189,14 +189,10 @@ steps: fi # Scan from the passing attempt to the end. Any traceback or ERROR - # line fails the gate, also one from a handled exception. - SCAN=$$(tail -c +$$((LOG_OFFSET + 1)) "$$SERVER_LOG") - EXCEPTIONS=$$(echo "$$SCAN" \ - | grep -n -A8 'Traceback (most recent call last)' | head -80) - ERRORS=$$(echo "$$SCAN" | grep -nE '\] (ERROR|CRITICAL) ' | head -40) - if [[ -n "$$EXCEPTIONS$$ERRORS" ]]; then - echo "$$ERRORS" - echo "$$EXCEPTIONS" + # record fails the gate, also one from a handled exception. + PROBLEMS=$$(tail -c +$$((LOG_OFFSET + 1)) "$$SERVER_LOG" | log_problems) + if [[ -n "$$PROBLEMS" ]]; then + echo "$$PROBLEMS" release_fail $$STAGE "The eval server logged errors or exceptions during the smoke evals." fi @@ -364,11 +360,11 @@ steps: fi echo "===== scan the new pod logs =====" - EXCEPTIONS=$$(kubectl logs -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" \ - | grep -n -A8 'Traceback (most recent call last)' | head -80) - if [[ -n "$$EXCEPTIONS" ]]; then - echo "$$EXCEPTIONS" - rollback "The new pod $$POD logged server exceptions." + PROBLEMS=$$(kubectl logs -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" \ + | log_problems) + if [[ -n "$$PROBLEMS" ]]; then + echo "$$PROBLEMS" + rollback "The new pod $$POD logged server errors or exceptions." fi # Record the release. An annotation does not restart the pod. diff --git a/.ci/release/lib.sh b/.ci/release/lib.sh index d6598bd0..bfd5d146 100644 --- a/.ci/release/lib.sh +++ b/.ci/release/lib.sh @@ -12,11 +12,11 @@ RELEASE_RESULT_PREFIX="RELEASE_RESULT" # Usage: release_result STATUS STAGE DETAIL # STATUS SUCCESS, FAILED, ROLLED_BACK, or ROLLBACK_FAILED # STAGE the id of the step that ended the run -# DETAIL one sentence for the email. Double quotes become single quotes. +# DETAIL one sentence for the email # -# The release-notes step writes the PR list and the compare link. If the step -# did not finish, the line says so. detail stays last, because the alert -# policy reads it with a greedy regex. +# Reads the tag, PR list, and compare link from /workspace. If a step did not +# write them, it prints a fallback. Double quotes become single quotes. Keep +# detail last, because the alert policy reads it with a greedy regex. release_result() { _tag=$(cat /workspace/release_tag.txt 2>/dev/null || echo "none") _detail=$(printf '%s' "$3" | tr '"\n' "' ") @@ -37,6 +37,19 @@ release_fail() { exit 1 } +# Prints the ERROR and CRITICAL records and tracebacks in an eval_server log. +# Prints nothing if the log is clean. Matches only at the start of a line, +# because eval payloads can quote a traceback. +# +# Usage: log_problems < LOG +log_problems() { + _log=$(cat) + printf '%s\n' "$_log" \ + | grep -nE '^[0-9-]{10} [0-9:,]+ \[[^]]*\] (ERROR|CRITICAL) ' | head -40 + printf '%s\n' "$_log" \ + | grep -n -A8 '^Traceback (most recent call last)' | head -80 +} + # Points TAG at DIGEST. 'tags update' moves an existing tag without the # tags.delete permission. 'docker tags add' creates a new tag. # From ed4d7bf474f5daf16f9d9f4443a157f17ac3e53e Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad <omkargaikwad@google.com> Date: Wed, 30 Sep 2026 10:17:35 +0000 Subject: [PATCH 18/27] ci(release): use TRIGGER_NAME in the concurrency guard Cloud Build rejects $TRIGGER_ID as an unknown substitution, so the trigger did not start. Match other active runs on the built-in TRIGGER_NAME, which triggered builds store in their substitutions. Also list the log scan as the third smoke-gate check, and remove an overstated claim from the alert policy comment. --- .ci/release.cloudbuild.yaml | 8 +++++--- .ci/release/alert_policy.yaml | 2 +- 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index 34c52e39..230a2b86 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -29,9 +29,9 @@ steps: # Stop if another run is active, because two runs race on GKE and # :latest. If the list call fails, continue. - if [[ -n "$TRIGGER_ID" ]]; then + if [[ -n "$TRIGGER_NAME" ]]; then OTHERS=$$(gcloud builds list --ongoing --region="$LOCATION" \ - --filter="buildTriggerId=$TRIGGER_ID AND id!=$BUILD_ID" \ + --filter="substitutions.TRIGGER_NAME=$TRIGGER_NAME AND id!=$BUILD_ID" \ --format='value(id)' 2>/dev/null) if [[ -n "$$OTHERS" ]]; then release_fail $$STAGE "Another release run is active: $$OTHERS" @@ -114,10 +114,12 @@ steps: echo "PRs since $$OLD_SHA:" cat "$$NOTES" - # Serves the candidate image over gRPC and runs two checks on it: + # Serves the candidate image over gRPC and runs three checks on it: # 1. The smoke eval: the server generates SQL with a model. eval_client # exits 0 for any completed run, so the verifier grades the CSVs. # 2. client_sql_smoke.py: the client-generated SQL path, with fixed scores. + # 3. The log scan: any ERROR, CRITICAL, or traceback in the server log + # fails the gate. - id: smoke-gate name: 'us-central1-docker.pkg.dev/$PROJECT_ID/evalbench/eval_server:$COMMIT_SHA' dir: '/evalbench' diff --git a/.ci/release/alert_policy.yaml b/.ci/release/alert_policy.yaml index 3dfd09a7..620d6dde 100644 --- a/.ci/release/alert_policy.yaml +++ b/.ci/release/alert_policy.yaml @@ -27,7 +27,7 @@ conditions: prs: 'REGEXP_EXTRACT(textPayload, "prs=\"([^\"]*)\"")' detail: 'REGEXP_EXTRACT(textPayload, "detail=\"(.*)\"")' alertStrategy: - # 5 minutes is the minimum. A run prints one line, so no result is lost. + # 5 minutes is the minimum. notificationRateLimit: period: 300s autoClose: 1800s From ab2b96c12babb6d009ff76e00fd5f929af171b0b Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad <omkargaikwad@google.com> Date: Wed, 30 Sep 2026 12:07:06 +0000 Subject: [PATCH 19/27] chore: rename weekly release alert policy condition display name --- .ci/release/alert_policy.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.ci/release/alert_policy.yaml b/.ci/release/alert_policy.yaml index 620d6dde..30636027 100644 --- a/.ci/release/alert_policy.yaml +++ b/.ci/release/alert_policy.yaml @@ -10,7 +10,7 @@ displayName: 'EvalBench weekly release' combiner: OR conditions: - - displayName: 'RELEASE_RESULT line in the weekly release build' + - displayName: 'Weekly release result' conditionMatchedLog: filter: >- resource.type="build" From 47bda426009755fb5affd0f0cc2707d50b1a7f26 Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad <omkargaikwad@google.com> Date: Wed, 30 Sep 2026 14:33:36 +0000 Subject: [PATCH 20/27] ci(release): cap the smoke gate and alert on runs with no result Add a 1200 s timeout to smoke-gate, so a hung eval cannot use the build time that deploy-gke needs for a rollout and a rollback. Add build_status_policy.yaml. It emails when a run ends as CANCELLED, TIMEOUT, or INTERNAL_ERROR, which print no RELEASE_RESULT line. It matches the build-end audit entry and skips FAILURE, because a failed step already prints a RELEASE_RESULT line. Each run sends one email. Put the detail right after the status in the result email. Name the tag and digest in the success detail, use notify-success as the stage, and print 'unavailable' when there is no compare link. --- .ci/release.cloudbuild.yaml | 10 +++--- .ci/release/alert_policy.yaml | 8 ++--- .ci/release/build_status_policy.yaml | 48 ++++++++++++++++++++++++++++ .ci/release/lib.sh | 5 +-- 4 files changed, 61 insertions(+), 10 deletions(-) create mode 100644 .ci/release/build_status_policy.yaml diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index 7d90539d..83fc2b5a 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -5,8 +5,9 @@ # to Corp Run. # # Each run prints one RELEASE_RESULT line (lib.sh), and alert_policy.yaml -# emails it. Cloud Build has no on-failure hook, so a failed step prints its -# own line. A timeout or a cancel prints no line. +# emails it. Cloud Build has no on-failure hook, so each failed step prints +# its own line. A timeout, a cancel, or an internal error prints no line, so +# build_status_policy.yaml emails those. # # Use $$ for shell variables and $$(...). Cloud Build expands a single $. @@ -126,6 +127,7 @@ steps: name: 'us-central1-docker.pkg.dev/$PROJECT_ID/evalbench/eval_server:$COMMIT_SHA' dir: '/evalbench' waitFor: ['resolve-release'] + timeout: '1200s' entrypoint: 'bash' args: - '-c' @@ -193,7 +195,7 @@ steps: fi # Scan from the passing attempt to the end. Any traceback or ERROR - # record fails the gate, also one from a handled exception. + # record fails the gate, even one from a handled exception. PROBLEMS=$$(tail -c +$$((LOG_OFFSET + 1)) "$$SERVER_LOG" | log_problems) if [[ -n "$$PROBLEMS" ]]; then echo "$$PROBLEMS" @@ -478,7 +480,7 @@ steps: - '-c' - | . /workspace/.ci/release/lib.sh - release_result SUCCESS release "GKE and Cloud Run serve $$(cat /workspace/image_digest.txt)." + release_result SUCCESS notify-success "GKE and Cloud Run serve $$(cat /workspace/release_tag.txt) ($$(cat /workspace/image_digest.txt))." substitutions: _EVAL_REGION: 'global' diff --git a/.ci/release/alert_policy.yaml b/.ci/release/alert_policy.yaml index 30636027..f05bde85 100644 --- a/.ci/release/alert_policy.yaml +++ b/.ci/release/alert_policy.yaml @@ -6,7 +6,7 @@ # TRIGGER_ID id of the trigger 'weekly-evalbench-release' (us-central1) # CHANNEL_NAME name of the notification channel 'EvalBench release' # To change the policy, fill in a copy of this file and run -# 'gcloud alpha monitoring policies update'. Do not create a second policy. +# 'gcloud alpha monitoring policies update'. Do not create a duplicate. displayName: 'EvalBench weekly release' combiner: OR conditions: @@ -37,13 +37,13 @@ documentation: content: | **Status:** ${log.extracted_label.status} + **Detail:** ${log.extracted_label.detail} + **Stage:** ${log.extracted_label.stage} **Tag:** ${log.extracted_label.tag} - **Detail:** ${log.extracted_label.detail} - - **PRs since the previous release:** ${log.extracted_label.prs} + **PRs since the last release:** ${log.extracted_label.prs} **Compare:** ${log.extracted_label.compare} diff --git a/.ci/release/build_status_policy.yaml b/.ci/release/build_status_policy.yaml new file mode 100644 index 00000000..09467f7b --- /dev/null +++ b/.ci/release/build_status_policy.yaml @@ -0,0 +1,48 @@ +# Log-based alert that emails when a weekly release run ends without a +# RELEASE_RESULT line: a timeout, a cancel, or an internal error. +# +# Cloud Build writes an audit entry when a build ends. Its status code is 1 +# for CANCELLED, 4 for TIMEOUT, 9 for FAILURE, and 13 for INTERNAL_ERROR. A +# failed step prints a RELEASE_RESULT line, so alert_policy.yaml emails a +# FAILURE and this policy skips code 9. Each run sends one email. +# +# Not covered: a build that fails before a step prints its line, for example +# when a step image does not pull, and a build that never starts. +# +# A log-based policy has only one condition, so this alert is a separate +# policy. TRIGGER_ID and CHANNEL_NAME are placeholders, as in +# alert_policy.yaml. To create the policy, fill in a copy of this file and +# run 'gcloud alpha monitoring policies create'. +displayName: 'EvalBench weekly release build status' +combiner: OR +conditions: + - displayName: 'Weekly release build ended without a result' + conditionMatchedLog: + filter: >- + logName:"cloudaudit.googleapis.com%2Factivity" + AND resource.type="build" + AND resource.labels.build_trigger_id="TRIGGER_ID" + AND protoPayload.methodName="google.devtools.cloudbuild.v1.CloudBuild.CreateBuild" + AND operation.last=true + AND protoPayload.status.code>0 + AND protoPayload.status.code!=9 + labelExtractors: + code: 'EXTRACT(protoPayload.status.code)' +alertStrategy: + # 5 minutes is the minimum. + notificationRateLimit: + period: 300s + autoClose: 1800s +documentation: + subject: 'EvalBench release build ended without a result (code ${log.extracted_label.code})' + mimeType: text/markdown + content: | + **Status code:** ${log.extracted_label.code} (1 = CANCELLED, 4 = TIMEOUT, 13 = INTERNAL_ERROR) + + The run ended before it printed a RELEASE_RESULT line, so no other email + comes for it. If deploy-gke or deploy-cloudrun did not finish, check the + image that GKE and Cloud Run serve. + + **Build log:** https://console.cloud.google.com/cloud-build/builds;region=us-central1/${resource.label.build_id}?project=${resource.label.project_id} +notificationChannels: + - CHANNEL_NAME diff --git a/.ci/release/lib.sh b/.ci/release/lib.sh index bfd5d146..d70cbd10 100644 --- a/.ci/release/lib.sh +++ b/.ci/release/lib.sh @@ -4,7 +4,8 @@ # POSIX sh, so the busybox shell in the gcrane image can source it. Needs # BUILD_ID, COMMIT_SHA, LOCATION and PROJECT_ID from options.env. -# alert_policy.yaml parses this line. Update it if the format changes. +# alert_policy.yaml matches this prefix and parses the line format. If you +# change either, update the policy. RELEASE_RESULT_PREFIX="RELEASE_RESULT" # Prints the result line of the run. @@ -23,7 +24,7 @@ release_result() { _prs=$(cat /workspace/release_prs.txt 2>/dev/null \ || echo "PR list not computed before this step ended the run.") _prs=$(printf '%s' "$_prs" | tr '"\n' "' ") - _compare=$(cat /workspace/release_compare.txt 2>/dev/null || echo "none") + _compare=$(cat /workspace/release_compare.txt 2>/dev/null || echo "unavailable") _log="https://console.cloud.google.com/cloud-build/builds;region=${LOCATION}/${BUILD_ID}?project=${PROJECT_ID}" _commit="https://github.com/GoogleCloudPlatform/evalbench/commit/${COMMIT_SHA}" echo "${RELEASE_RESULT_PREFIX} status=$1 stage=$2 tag=${_tag} commit=${_commit} log=${_log} compare=${_compare} prs=\"${_prs}\" detail=\"${_detail}\"" From 1539253f31d4eaf79c12846c42015299fc805f0f Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad <omkargaikwad@google.com> Date: Wed, 30 Sep 2026 14:40:04 +0000 Subject: [PATCH 21/27] ci(release): scan only the GKE pod startup log before rollback The Deployment has no readiness probe, so the new pod takes traffic as soon as it starts. A client request can then log an ERROR, for example a missing set_up_script, and roll back a healthy release. Stop the scan at the "Server started" line. If the line is missing, scan the full log. --- .ci/release.cloudbuild.yaml | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index 83fc2b5a..f9329901 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -365,12 +365,14 @@ steps: rollback "The new pod $$POD did not answer Ping." fi - echo "===== scan the new pod logs =====" + echo "===== scan the new pod startup log =====" + # The pod takes traffic as soon as it starts, so a client request can + # log an ERROR. Scan only the startup log, up to "Server started". PROBLEMS=$$(kubectl logs -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" \ - | log_problems) + | sed '/Server started/q' | log_problems) if [[ -n "$$PROBLEMS" ]]; then echo "$$PROBLEMS" - rollback "The new pod $$POD logged server errors or exceptions." + rollback "The new pod $$POD logged errors or exceptions at startup." fi # Record the release. An annotation does not restart the pod. From 88838515ffca962121aa89d6cce6f1b5e0daeac1 Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad <omkargaikwad@google.com> Date: Wed, 30 Sep 2026 15:05:18 +0000 Subject: [PATCH 22/27] ci(release): scan the startup log and remove redundant comments --- .ci/release.cloudbuild.yaml | 34 +++++++++++++++++++--------------- .ci/release/lib.sh | 8 ++------ .ci/release/wait_for_grpc.py | 1 - 3 files changed, 21 insertions(+), 22 deletions(-) diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index f9329901..5fc048e8 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -2,7 +2,7 @@ # # Gates the image that the Build trigger pushed for this commit, tags it, # deploys it to GKE with automatic rollback, moves :latest, and deploys it -# to Corp Run. +# to Cloud Run. # # Each run prints one RELEASE_RESULT line (lib.sh), and alert_policy.yaml # emails it. Cloud Build has no on-failure hook, so each failed step prints @@ -56,9 +56,9 @@ steps: echo "digest = $$DIGEST" echo "tag = $$TAG" - # Lists the PRs merged since the release that GKE serves now. lib.sh adds - # the list to the result line. Runs in parallel with the smoke gate. Every - # failure writes a reason and exits 0, so this step never fails the release. + # Lists the PRs merged since the release that GKE serves now, for the + # result line. Every failure writes a reason and exits 0, so this step + # never fails the release. - id: release-notes name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' waitFor: ['resolve-release'] @@ -158,6 +158,14 @@ steps: release_fail $$STAGE "The eval server did not answer Ping within 300 s." fi + # The final scan starts at the passing attempt, so check startup here. + PROBLEMS=$$(log_problems < "$$SERVER_LOG") + if [[ -n "$$PROBLEMS" ]]; then + echo "$$PROBLEMS" + dump_log + release_fail $$STAGE "The eval server logged errors or exceptions at startup." + fi + # Retry once if the client crashes or the verifier exits 2 (model). for ATTEMPT in 1 2; do echo "===== smoke eval, attempt $$ATTEMPT =====" @@ -185,8 +193,7 @@ steps: release_fail $$STAGE "The smoke eval failed on attempt $$ATTEMPT. The build log has the verifier output." done - # Client SQL, noop generator, EvalConfig resources, batch and - # streaming. Scores are fixed, so there is no retry. + # Scores are fixed, so there is no retry. echo "===== client SQL smoke =====" if ! EVALBENCH_INSECURE=true python .ci/release/client_sql_smoke.py \ --results-dir "$$RESULTS"; then @@ -194,8 +201,7 @@ steps: release_fail $$STAGE "The client SQL smoke failed. The build log lists the problems." fi - # Scan from the passing attempt to the end. Any traceback or ERROR - # record fails the gate, even one from a handled exception. + # Skip a failed model attempt. Handled exceptions still fail the gate. PROBLEMS=$$(tail -c +$$((LOG_OFFSET + 1)) "$$SERVER_LOG" | log_problems) if [[ -n "$$PROBLEMS" ]]; then echo "$$PROBLEMS" @@ -275,8 +281,8 @@ steps: echo "$$RUNNING" > /workspace/rollback_digest.txt echo "rollback target = $$RUNNING" - # Deploy, check, and rollback share one step, so a failed check can roll - # back. See the header. + # Deploy, check, and rollback share one step, because Cloud Build has no + # on-failure hook. - id: deploy-gke name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' waitFor: ['capture-rollback-target'] @@ -324,7 +330,6 @@ steps: --timeout="$$ROLLOUT_TIMEOUT" } - # Restore the previous digest and print the result line. rollback() { echo "===== rolling back to $$OLD_DIGEST =====" if pin "$$OLD_DIGEST"; then @@ -366,8 +371,8 @@ steps: fi echo "===== scan the new pod startup log =====" - # The pod takes traffic as soon as it starts, so a client request can - # log an ERROR. Scan only the startup log, up to "Server started". + # Client requests can log an ERROR after startup, so stop at + # "Server started". PROBLEMS=$$(kubectl logs -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" \ | sed '/Server started/q' | log_problems) if [[ -n "$$PROBLEMS" ]]; then @@ -375,7 +380,7 @@ steps: rollback "The new pod $$POD logged errors or exceptions at startup." fi - # Record the release. An annotation does not restart the pod. + # An annotation does not restart the pod. kubectl annotate "$$DEPLOYMENT" -n "$$NAMESPACE" --overwrite \ kubernetes.io/change-cause="release $$TAG ($$NEW_DIGEST)" || true @@ -438,7 +443,6 @@ steps: --format='value(status.latestReadyRevisionName)') echo "current revision = $$OLD_REVISION" - # Keep traffic on the current revision until the digest check passes. echo "===== deploy a revision with no traffic =====" if ! gcloud run deploy "$$SERVICE" "$${RUN[@]}" \ --image="$$CR_REPO@$$DIGEST" --no-traffic; then diff --git a/.ci/release/lib.sh b/.ci/release/lib.sh index d70cbd10..35387fef 100644 --- a/.ci/release/lib.sh +++ b/.ci/release/lib.sh @@ -15,9 +15,8 @@ RELEASE_RESULT_PREFIX="RELEASE_RESULT" # STAGE the id of the step that ended the run # DETAIL one sentence for the email # -# Reads the tag, PR list, and compare link from /workspace. If a step did not -# write them, it prints a fallback. Double quotes become single quotes. Keep -# detail last, because the alert policy reads it with a greedy regex. +# Reads the tag, PR list, and compare link from /workspace. Keep detail +# last, because the alert policy reads it with a greedy regex. release_result() { _tag=$(cat /workspace/release_tag.txt 2>/dev/null || echo "none") _detail=$(printf '%s' "$3" | tr '"\n' "' ") @@ -30,9 +29,6 @@ release_result() { echo "${RELEASE_RESULT_PREFIX} status=$1 stage=$2 tag=${_tag} commit=${_commit} log=${_log} compare=${_compare} prs=\"${_prs}\" detail=\"${_detail}\"" } -# Prints a FAILED result line and ends the step with an error. -# -# Usage: release_fail STAGE DETAIL release_fail() { release_result FAILED "$1" "$2" exit 1 diff --git a/.ci/release/wait_for_grpc.py b/.ci/release/wait_for_grpc.py index 4725185c..14c94241 100644 --- a/.ci/release/wait_for_grpc.py +++ b/.ci/release/wait_for_grpc.py @@ -21,7 +21,6 @@ from client.eval_client import EvalbenchClient # noqa: E402 -# Consecutive ALTS handshake failures before aborting. _ALTS_FAILURE_LIMIT = 5 From 107ad8f4b0e1f153123739f7861dd44d77a68143 Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad <omkargaikwad@google.com> Date: Wed, 30 Sep 2026 15:17:34 +0000 Subject: [PATCH 23/27] ci(release): wait for the Build trigger image with a step timeout --- .ci/release.cloudbuild.yaml | 26 ++++++++++++++++++-------- 1 file changed, 18 insertions(+), 8 deletions(-) diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index 5fc048e8..c3c30dd4 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -13,9 +13,10 @@ steps: - # The Build trigger must have pushed an image for this commit. + # Waits for the image that the Build trigger pushes after its tests pass. - id: resolve-release name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' + timeout: '3300s' entrypoint: 'bash' args: - '-c' @@ -40,13 +41,19 @@ steps: fi REPO="us-central1-docker.pkg.dev/$PROJECT_ID/evalbench/eval_server" + DEADLINE=$$(( $$(date +%s) + ${_IMAGE_WAIT_SECONDS} )) echo "Resolving $$REPO:$COMMIT_SHA" - DIGEST=$$(gcloud artifacts docker images describe \ - "$$REPO:$COMMIT_SHA" \ - --format='value(image_summary.digest)' 2>/dev/null) - if [[ -z "$$DIGEST" ]]; then - release_fail $$STAGE "No image at $$REPO:$COMMIT_SHA. The Build trigger has not published this commit." - fi + while true; do + DIGEST=$$(gcloud artifacts docker images describe \ + "$$REPO:$COMMIT_SHA" \ + --format='value(image_summary.digest)' 2>/dev/null) + [[ -n "$$DIGEST" ]] && break + if (( $$(date +%s) >= DEADLINE )); then + release_fail $$STAGE "No image at $$REPO:$COMMIT_SHA after ${_IMAGE_WAIT_SECONDS} s. The Build trigger failed or is still running for this commit." + fi + echo "No image yet. Checking again in 5 min." + sleep 300 + done TAG="v$$(date -u +%Y.%m.%d).$SHORT_SHA" @@ -490,8 +497,11 @@ steps: substitutions: _EVAL_REGION: 'global' + # The Build trigger took 15 to 26 min in 59 runs. + _IMAGE_WAIT_SECONDS: '2700' -timeout: '3600s' +# The image wait, one extra poll, and about 3000 s for the other steps. +timeout: '6600s' options: logging: CLOUD_LOGGING_ONLY From f70f779fb531daa64b14e711abc6833ec3835d03 Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad <omkargaikwad@google.com> Date: Thu, 1 Oct 2026 06:04:26 +0000 Subject: [PATCH 24/27] chore: optimize cloudbuild step timeouts and reduce wait durations for release pipeline efficiency --- .ci/release.cloudbuild.yaml | 26 ++++++++++++++++++-------- 1 file changed, 18 insertions(+), 8 deletions(-) diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index c3c30dd4..0f7b259e 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -16,7 +16,7 @@ steps: # Waits for the image that the Build trigger pushes after its tests pass. - id: resolve-release name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' - timeout: '3300s' + timeout: '2400s' entrypoint: 'bash' args: - '-c' @@ -111,9 +111,10 @@ steps: > /workspace/release_compare.txt # A treeless clone has every commit message and no file contents. - if ! git clone --quiet --filter=tree:0 --no-checkout \ + # A step timeout fails the release, so limit the clone instead. + if ! timeout 300 git clone --quiet --filter=tree:0 --no-checkout \ "$$GITHUB.git" /tmp/evalbench; then - unavailable "git clone of $$GITHUB failed." + unavailable "git clone of $$GITHUB failed or took more than 300 s." fi if ! python3 /workspace/.ci/release/release_notes.py \ --repo-dir /tmp/evalbench --old "$$OLD_SHA" --new "$COMMIT_SHA" \ @@ -226,6 +227,7 @@ steps: - id: tag-release name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' waitFor: ['smoke-gate'] + timeout: '120s' entrypoint: 'bash' args: - '-c' @@ -254,6 +256,7 @@ steps: - id: capture-rollback-target name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' waitFor: ['tag-release'] + timeout: '120s' entrypoint: 'bash' args: - '-c' @@ -293,6 +296,8 @@ steps: - id: deploy-gke name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' waitFor: ['capture-rollback-target'] + # 2 rollouts (600 s), confirm (60 s), and Ping (120 s). + timeout: '900s' entrypoint: 'bash' args: - '-c' @@ -304,7 +309,7 @@ steps: SELECTOR=app=evalbench-eval-server CONTAINER=evalbench-eval DEPLOYMENT=deployment/evalbench-eval-server-deploy - ROLLOUT_TIMEOUT=600s + ROLLOUT_TIMEOUT=300s REPO=$$(cat /workspace/image_repo.txt) NEW_DIGEST=$$(cat /workspace/image_digest.txt) @@ -372,7 +377,7 @@ steps: -o jsonpath='{.items[0].metadata.name}') echo "===== gRPC health check on $$POD =====" if ! kubectl exec -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" -- \ - python /evalbench/.ci/release/wait_for_grpc.py --timeout 300; then + python /evalbench/.ci/release/wait_for_grpc.py --timeout 120; then kubectl logs -n "$$NAMESPACE" "$$POD" -c "$$CONTAINER" --tail=150 rollback "The new pod $$POD did not answer Ping." fi @@ -398,6 +403,7 @@ steps: - id: promote-latest name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' waitFor: ['deploy-gke'] + timeout: '120s' entrypoint: 'bash' args: - '-c' @@ -415,6 +421,7 @@ steps: - id: copy-cloudrun-image name: 'gcr.io/go-containerregistry/gcrane:debug' waitFor: ['promote-latest'] + timeout: '300s' entrypoint: 'sh' args: - '-c' @@ -433,6 +440,7 @@ steps: - id: deploy-cloudrun name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' waitFor: ['copy-cloudrun-image'] + timeout: '600s' entrypoint: 'bash' args: - '-c' @@ -488,6 +496,7 @@ steps: - id: notify-success name: 'gcr.io/google.com/cloudsdktool/cloud-sdk' waitFor: ['deploy-cloudrun', 'release-notes'] + timeout: '60s' entrypoint: 'bash' args: - '-c' @@ -498,10 +507,11 @@ steps: substitutions: _EVAL_REGION: 'global' # The Build trigger took 15 to 26 min in 59 runs. - _IMAGE_WAIT_SECONDS: '2700' + _IMAGE_WAIT_SECONDS: '1800' -# The image wait, one extra poll, and about 3000 s for the other steps. -timeout: '6600s' +# Covers the step timeouts up to deploy-gke (4740 s), so a rollback always +# ends. +timeout: '4800s' options: logging: CLOUD_LOGGING_ONLY From 1f81961b15bce9639c9d9738b55eb14286ba0d46 Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad <omkargaikwad@google.com> Date: Thu, 1 Oct 2026 06:20:56 +0000 Subject: [PATCH 25/27] docs: update build status policy error notification message for clarity and actionable steps --- .ci/release/build_status_policy.yaml | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/.ci/release/build_status_policy.yaml b/.ci/release/build_status_policy.yaml index 09467f7b..08b59b95 100644 --- a/.ci/release/build_status_policy.yaml +++ b/.ci/release/build_status_policy.yaml @@ -39,9 +39,12 @@ documentation: content: | **Status code:** ${log.extracted_label.code} (1 = CANCELLED, 4 = TIMEOUT, 13 = INTERNAL_ERROR) - The run ended before it printed a RELEASE_RESULT line, so no other email - comes for it. If deploy-gke or deploy-cloudrun did not finish, check the - image that GKE and Cloud Run serve. + The release run stopped before it reported a result. + + 1. Open the build log to see where the run stopped. + 2. Check that all GKE pods and Cloud Run serve the same image digest. If + they do not, the deploy is incomplete. + 3. If necessary, start the release again. **Build log:** https://console.cloud.google.com/cloud-build/builds;region=us-central1/${resource.label.build_id}?project=${resource.label.project_id} notificationChannels: From 2dde56e8965db79bc669ee2ca26aa186f40da96e Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad <omkargaikwad@google.com> Date: Thu, 1 Oct 2026 06:50:55 +0000 Subject: [PATCH 26/27] ci: set the release build timeout to the sum of the step timeouts --- .ci/release.cloudbuild.yaml | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index 0f7b259e..a626c0fd 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -509,9 +509,8 @@ substitutions: # The Build trigger took 15 to 26 min in 59 runs. _IMAGE_WAIT_SECONDS: '1800' -# Covers the step timeouts up to deploy-gke (4740 s), so a rollback always -# ends. -timeout: '4800s' +# Sum of the step timeouts. +timeout: '5820s' options: logging: CLOUD_LOGGING_ONLY From 423e39f70c89ac59034cbe9a088d6119a67a2b4f Mon Sep 17 00:00:00 2001 From: Omkar Gaikwad <omkargaikwad@google.com> Date: Thu, 1 Oct 2026 07:01:28 +0000 Subject: [PATCH 27/27] refactor: update release build status policy to alert on all non-zero exit codes including code 9 --- .ci/release.cloudbuild.yaml | 4 ++-- .ci/release/build_status_policy.yaml | 20 +++++++++----------- 2 files changed, 11 insertions(+), 13 deletions(-) diff --git a/.ci/release.cloudbuild.yaml b/.ci/release.cloudbuild.yaml index a626c0fd..f60474dc 100644 --- a/.ci/release.cloudbuild.yaml +++ b/.ci/release.cloudbuild.yaml @@ -6,8 +6,8 @@ # # Each run prints one RELEASE_RESULT line (lib.sh), and alert_policy.yaml # emails it. Cloud Build has no on-failure hook, so each failed step prints -# its own line. A timeout, a cancel, or an internal error prints no line, so -# build_status_policy.yaml emails those. +# its own line. A step timeout or a cancel prints no line, so +# build_status_policy.yaml emails every run that fails. # # Use $$ for shell variables and $$(...). Cloud Build expands a single $. diff --git a/.ci/release/build_status_policy.yaml b/.ci/release/build_status_policy.yaml index 08b59b95..6c945813 100644 --- a/.ci/release/build_status_policy.yaml +++ b/.ci/release/build_status_policy.yaml @@ -1,13 +1,12 @@ -# Log-based alert that emails when a weekly release run ends without a -# RELEASE_RESULT line: a timeout, a cancel, or an internal error. +# Log-based alert that emails when a weekly release run fails. # # Cloud Build writes an audit entry when a build ends. Its status code is 1 # for CANCELLED, 4 for TIMEOUT, 9 for FAILURE, and 13 for INTERNAL_ERROR. A -# failed step prints a RELEASE_RESULT line, so alert_policy.yaml emails a -# FAILURE and this policy skips code 9. Each run sends one email. +# step timeout or a step image that does not pull gives code 9 and prints no +# RELEASE_RESULT line, so this policy also emails code 9. A failed step that +# prints its RELEASE_RESULT line sends a second email from alert_policy.yaml. # -# Not covered: a build that fails before a step prints its line, for example -# when a step image does not pull, and a build that never starts. +# Not covered: a build that never starts, for example an invalid config. # # A log-based policy has only one condition, so this alert is a separate # policy. TRIGGER_ID and CHANNEL_NAME are placeholders, as in @@ -16,7 +15,7 @@ displayName: 'EvalBench weekly release build status' combiner: OR conditions: - - displayName: 'Weekly release build ended without a result' + - displayName: 'Weekly release build failed' conditionMatchedLog: filter: >- logName:"cloudaudit.googleapis.com%2Factivity" @@ -25,7 +24,6 @@ conditions: AND protoPayload.methodName="google.devtools.cloudbuild.v1.CloudBuild.CreateBuild" AND operation.last=true AND protoPayload.status.code>0 - AND protoPayload.status.code!=9 labelExtractors: code: 'EXTRACT(protoPayload.status.code)' alertStrategy: @@ -34,12 +32,12 @@ alertStrategy: period: 300s autoClose: 1800s documentation: - subject: 'EvalBench release build ended without a result (code ${log.extracted_label.code})' + subject: 'EvalBench release build failed (code ${log.extracted_label.code})' mimeType: text/markdown content: | - **Status code:** ${log.extracted_label.code} (1 = CANCELLED, 4 = TIMEOUT, 13 = INTERNAL_ERROR) + **Status code:** ${log.extracted_label.code} (1 = CANCELLED, 4 = TIMEOUT, 9 = FAILURE, 13 = INTERNAL_ERROR) - The release run stopped before it reported a result. + The release run failed. 1. Open the build log to see where the run stopped. 2. Check that all GKE pods and Cloud Run serve the same image digest. If