Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
36 commits
Select commit Hold shift + click to select a range
258a235
ci: add a release smoke pipeline that runs a real eval against the de…
omkargaikwad23 Sep 17, 2026
08e4efb
Merge branch 'main' into feat/release-pipeline
omkargaikwad23 Sep 23, 2026
6c3a5a0
refactor: update NL2SQL verification to include dialect-specific scen…
omkargaikwad23 Sep 23, 2026
2da13d7
Merge branch 'main' into feat/release-pipeline
omkargaikwad23 Sep 24, 2026
119001b
refactor: remove Recreate deployment strategy in evalbench service yaml
omkargaikwad23 Sep 24, 2026
71ae37e
ci(release): capture the running digest before a release can deploy
omkargaikwad23 Sep 24, 2026
d1446a6
ci(release): deploy the gated digest with automatic rollback
omkargaikwad23 Sep 24, 2026
86132f3
ci(release): deploy the Mesop UI to Corp Run after the GKE gate
omkargaikwad23 Sep 24, 2026
635713b
ci(release): correct and tighten the pipeline comments
omkargaikwad23 Sep 24, 2026
6f05b65
ci(release): tighten roll_to and the convergence loop
omkargaikwad23 Sep 24, 2026
76fa1d1
ci(release): report a failed Cloud Run traffic pin
omkargaikwad23 Sep 24, 2026
677bcaf
ci(release): move tags in place instead of recreating them
omkargaikwad23 Sep 25, 2026
6e513d9
chore: update loop variable name and add progress logging to release …
omkargaikwad23 Sep 25, 2026
1da7338
chore: update image release tag format to use YYYY.MM.DD date convention
omkargaikwad23 Sep 25, 2026
07aa642
Merge branch 'main' into feat/release-pipeline
omkargaikwad23 Sep 28, 2026
60d9fca
ci(nl2sql): gate pull requests on a QueryData NL2SQL check (#611)
omkargaikwad23 Sep 29, 2026
05c72bf
Merge branch 'main' into feat/release-pipeline
omkargaikwad23 Sep 29, 2026
e0f238d
ci(release): gate, tag, and deploy the weekly release by digest
omkargaikwad23 Sep 29, 2026
09a29c6
Merge remote-tracking branch 'origin/main' into feat/release-pipeline
omkargaikwad23 Sep 29, 2026
46ba609
Merge branch 'main' into feat/release-pipeline
omkargaikwad23 Sep 30, 2026
78a611b
feat: add automated release notes generation and update alert policy …
omkargaikwad23 Sep 30, 2026
0129e11
ci(release): gate the client-generated SQL path in the smoke gate
omkargaikwad23 Sep 30, 2026
33cd126
ci(release): scan only real error records in server logs
omkargaikwad23 Sep 30, 2026
ed4d7bf
ci(release): use TRIGGER_NAME in the concurrency guard
omkargaikwad23 Sep 30, 2026
4312d3b
Merge branch 'main' into feat/release-pipeline
prernakakkar-google Sep 30, 2026
fb70681
Merge remote-tracking branch 'origin/main' into feat/release-pipeline
omkargaikwad23 Sep 30, 2026
ab2b96c
chore: rename weekly release alert policy condition display name
omkargaikwad23 Sep 30, 2026
47bda42
ci(release): cap the smoke gate and alert on runs with no result
omkargaikwad23 Sep 30, 2026
1539253
ci(release): scan only the GKE pod startup log before rollback
omkargaikwad23 Sep 30, 2026
8883851
ci(release): scan the startup log and remove redundant comments
omkargaikwad23 Sep 30, 2026
107ad8f
ci(release): wait for the Build trigger image with a step timeout
omkargaikwad23 Sep 30, 2026
a5bf69a
Merge branch 'main' into feat/release-pipeline
omkargaikwad23 Oct 1, 2026
f70f779
chore: optimize cloudbuild step timeouts and reduce wait durations fo…
omkargaikwad23 Oct 1, 2026
1f81961
docs: update build status policy error notification message for clari…
omkargaikwad23 Oct 1, 2026
2dde56e
ci: set the release build timeout to the sum of the step timeouts
omkargaikwad23 Oct 1, 2026
423e39f
refactor: update release build status policy to alert on all non-zero…
omkargaikwad23 Oct 1, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
523 changes: 523 additions & 0 deletions .ci/release.cloudbuild.yaml

Large diffs are not rendered by default.

54 changes: 54 additions & 0 deletions .ci/release/alert_policy.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,54 @@
# Log-based alert that emails the RELEASE_RESULT line (lib.sh) of each
# weekly release run.
#
# TRIGGER_ID and CHANNEL_NAME are placeholders, so this public file has no
# project resource IDs. The live policy already has the real values:
# TRIGGER_ID id of the trigger 'weekly-evalbench-release' (us-central1)
# CHANNEL_NAME name of the notification channel 'EvalBench release'
# To change the policy, fill in a copy of this file and run
# 'gcloud alpha monitoring policies update'. Do not create a duplicate.
displayName: 'EvalBench weekly release'
combiner: OR
conditions:
- displayName: 'Weekly release result'
conditionMatchedLog:
filter: >-
resource.type="build"
AND resource.labels.build_trigger_id="TRIGGER_ID"
AND textPayload:"RELEASE_RESULT status="
labelExtractors:
status: 'REGEXP_EXTRACT(textPayload, "status=(\\S+)")'
stage: 'REGEXP_EXTRACT(textPayload, "stage=(\\S+)")'
tag: 'REGEXP_EXTRACT(textPayload, "tag=(\\S+)")'
commit: 'REGEXP_EXTRACT(textPayload, "commit=(\\S+)")'
# 'log' is a reserved label name.
build_log: 'REGEXP_EXTRACT(textPayload, "log=(\\S+)")'
compare: 'REGEXP_EXTRACT(textPayload, "compare=(\\S+)")'
prs: 'REGEXP_EXTRACT(textPayload, "prs=\"([^\"]*)\"")'
detail: 'REGEXP_EXTRACT(textPayload, "detail=\"(.*)\"")'
alertStrategy:
# 5 minutes is the minimum.
notificationRateLimit:
period: 300s
autoClose: 1800s
documentation:
subject: 'EvalBench release ${log.extracted_label.status}: ${log.extracted_label.tag}'
mimeType: text/markdown
content: |
**Status:** ${log.extracted_label.status}

**Detail:** ${log.extracted_label.detail}

**Stage:** ${log.extracted_label.stage}

**Tag:** ${log.extracted_label.tag}

**PRs since the last release:** ${log.extracted_label.prs}

**Compare:** ${log.extracted_label.compare}

**Commit:** ${log.extracted_label.commit}

**Build log:** ${log.extracted_label.build_log}
notificationChannels:
- CHANNEL_NAME
49 changes: 49 additions & 0 deletions .ci/release/build_status_policy.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,49 @@
# Log-based alert that emails when a weekly release run fails.
#
# Cloud Build writes an audit entry when a build ends. Its status code is 1
# for CANCELLED, 4 for TIMEOUT, 9 for FAILURE, and 13 for INTERNAL_ERROR. A
# step timeout or a step image that does not pull gives code 9 and prints no
# RELEASE_RESULT line, so this policy also emails code 9. A failed step that
# prints its RELEASE_RESULT line sends a second email from alert_policy.yaml.
#
# Not covered: a build that never starts, for example an invalid config.
#
# A log-based policy has only one condition, so this alert is a separate
# policy. TRIGGER_ID and CHANNEL_NAME are placeholders, as in
# alert_policy.yaml. To create the policy, fill in a copy of this file and
# run 'gcloud alpha monitoring policies create'.
displayName: 'EvalBench weekly release build status'
combiner: OR
conditions:
- displayName: 'Weekly release build failed'
conditionMatchedLog:
filter: >-
logName:"cloudaudit.googleapis.com%2Factivity"
AND resource.type="build"
AND resource.labels.build_trigger_id="TRIGGER_ID"
AND protoPayload.methodName="google.devtools.cloudbuild.v1.CloudBuild.CreateBuild"
AND operation.last=true
AND protoPayload.status.code>0
labelExtractors:

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Step timeouts produce status.code=9 (FAILURE), not 4 (TIMEOUT): When an individual step hits its step timeout: (for example smoke-gate hitting 1200s), Cloud Build kills the step container immediately (so RELEASE_RESULT is never printed and alert_policy.yaml does not fire) and marks the overall build as FAILURE (protoPayload.status.code=9). Because this filter has protoPayload.status.code!=9, a step timeout won't trigger either alert policy. We should either wrap the smoke-gate body in an inner timeout 1140s ... || fail_release smoke_gate "smoke gate timed out" so RELEASE_RESULT is always logged, or drop protoPayload.status.code!=9 here.

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Valid, I removed code!=9, so every failed run now sends an email, including step timeouts.

code: 'EXTRACT(protoPayload.status.code)'
alertStrategy:
# 5 minutes is the minimum.
notificationRateLimit:
period: 300s
autoClose: 1800s
documentation:
subject: 'EvalBench release build failed (code ${log.extracted_label.code})'
mimeType: text/markdown
content: |
**Status code:** ${log.extracted_label.code} (1 = CANCELLED, 4 = TIMEOUT, 9 = FAILURE, 13 = INTERNAL_ERROR)

The release run failed.

1. Open the build log to see where the run stopped.
2. Check that all GKE pods and Cloud Run serve the same image digest. If
they do not, the deploy is incomplete.
3. If necessary, start the release again.

**Build log:** https://console.cloud.google.com/cloud-build/builds;region=us-central1/${resource.label.build_id}?project=${resource.label.project_id}
notificationChannels:
- CHANNEL_NAME
186 changes: 186 additions & 0 deletions .ci/release/client_sql_smoke.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,186 @@
#!/usr/bin/env python3
"""Runs the client-generated SQL path on the local eval server.

Most production clients generate SQL themselves and send it back. The server
uses the noop generator, reads its configs from the EvalConfig resources, and
only executes and scores the SQL. This script does the same with fixed SQL,
so every score is known before the run:

batch pass golden SQL on every item. Every scorer gives 100.
streaming pass wrong SQL on the last item. exact_match and set_match give
0 for it, which proves that the scorers can fail.

No model is called, so a failure is never model variance.

Exit codes: 0 every score matches, 1 otherwise.

Local test, with the server from 'python evalbench/eval_server.py --localhost':
EVALBENCH_INSECURE=true python .ci/release/client_sql_smoke.py
"""
import argparse
import asyncio
import os
import sys

import yaml

_HERE = os.path.dirname(os.path.abspath(__file__))
_REPO = os.path.dirname(os.path.dirname(_HERE))
# Add both module roots so eval_client and generated proto imports resolve.
for _path in (os.path.join(_REPO, "evalbench"),
os.path.join(_REPO, "evalbench", "evalproto")):
if _path not in sys.path:
sys.path.insert(0, _path)

from client.eval_client import EvalbenchClient # noqa: E402
from evalproto import eval_config_pb2, eval_connect_pb2 # noqa: E402
from evalproto import eval_request_pb2 # noqa: E402
from verify_release_smoke import ( # noqa: E402
HARD_ERROR_COLUMNS, as_prompt_id, as_score, is_empty, read_rows)

# The server replaces a config value that equals a resource address with the
# path of that resource in its session directory.
ROOT = "release_smoke"
DIALECT = "sqlite"
# Resource address -> repo file. None means the content is in RESOURCES_TEXT.
RESOURCES = {
f"{ROOT}/release_smoke.evalset.json":
".ci/release/release_smoke.evalset.json",
f"{ROOT}/sqlite.yaml": "datasets/bat/db_configs/sqlite.yaml",
f"{ROOT}/noop_model.yaml": None,
}
RESOURCES_TEXT = {f"{ROOT}/noop_model.yaml": "generator: noop\n"}
CONFIG = {
"dataset_config": f"{ROOT}/release_smoke.evalset.json",
"dataset_format": "evalbench-standard-format",
"database_configs": [f"{ROOT}/sqlite.yaml"],
"dialects": [DIALECT],
"dialect": DIALECT,
# The orchestrator sends read queries only. DML goldens also write
# CURRENT_TIMESTAMP, so their scores are not fixed.
"query_types": ["dql"],
"setup_directory": "datasets/bat/setup",
"model_config": f"{ROOT}/noop_model.yaml",
"prompt_generator": "NOOPGenerator",
"num_trials": 1,
"scorers": {"exact_match": None, "set_match": None,
"executable_sql": None, "returned_sql": None},
"reporting": {"csv": {"output_directory": "results"}},
}
WRONG_SQL = "SELECT -1 AS wrong_answer;"
# Expected scores for the item with WRONG_SQL. Golden SQL gives 100 on all.
WRONG_SCORES = {"exact_match": 0, "set_match": 0,
"executable_sql": 100, "returned_sql": 100}


def config_request():
request = eval_config_pb2.EvalConfigRequest(
yaml_config=yaml.safe_dump(CONFIG).encode("utf-8"))
for address, path in RESOURCES.items():
if path is None:
content = RESOURCES_TEXT[address].encode("utf-8")
else:
with open(os.path.join(_REPO, path), "rb") as f:
content = f.read()
request.resources.add(address=address, content=content)
return request


async def run_pass(streaming, wrong_last):
"""Returns (job_id, {item id: sent SQL}) for one Eval call."""
client = EvalbenchClient("local")
metadata = client.metadata
try:
await client.stub.Ping(eval_request_pb2.PingRequest(),
metadata=metadata)
await client.stub.Connect(
eval_connect_pb2.EvalConnectRequest(
client_id="release-smoke", streaming_eval=streaming),
metadata=metadata)
await client.stub.EvalConfig(config_request(), metadata=metadata)
items = [item async for item in client.stub.ListEvalInputs(
eval_request_pb2.EvalInputRequest(), metadata=metadata)]
if not items:
raise RuntimeError("ListEvalInputs returned no items")
sent = {}
for i, item in enumerate(items):
golden = item.golden_sql[DIALECT].sql_statements[0]
wrong = wrong_last and i == len(items) - 1
item.generated_sql = WRONG_SQL if wrong else golden
sent[str(item.id)] = item.generated_sql
response = await client.stub.Eval(iter(items), metadata=metadata)
return response.response, sent
finally:
await client.channel.close()


def check(name, results_dir, job_id, sent, wrong_id):
"""Returns the problems in the reports of one pass."""
job_dir = os.path.join(results_dir, job_id)
if not job_id or not os.path.isdir(job_dir):
return [f"no report directory {job_dir}"]
problems = []
# prompt_id is the dataset id. id is the eval id that scores.csv uses.
evals = {as_prompt_id(r.get("prompt_id")): r
for r in read_rows(job_dir, "evals.csv")}
eval_ids = {}
for item_id, sql in sorted(sent.items()):
row = evals.get(item_id)
if row is None:
problems.append(f"{item_id}: no eval row")
continue
eval_ids[item_id] = as_prompt_id(row.get("id"))
for column in HARD_ERROR_COLUMNS + ("sql_generator_error",
"generated_error"):
if not is_empty(row.get(column)):
problems.append(f"{item_id}: {column}: "
f"{row[column].strip()[:200]}")
# The noop generator must keep the client SQL.
if (row.get("generated_sql") or "").strip() != sql.strip():
problems.append(f"{item_id}: generated_sql is not the SQL that "
f"the client sent")

scores = {(r.get("comparator"), as_prompt_id(r.get("id"))): r
for r in read_rows(job_dir, "scores.csv")}
for scorer in sorted(CONFIG["scorers"]):
for item_id, eval_id in sorted(eval_ids.items()):
want = WRONG_SCORES[scorer] if item_id == wrong_id else 100
row = scores.get((scorer, eval_id))
got = as_score(row.get("score")) if row else None
if got != want:
problems.append(f"{scorer}: {item_id} scored {got}, "
f"expected {want}")
status = "[PASS]" if not problems else "[FAIL]"
print(f"{status} {name}: job {job_id}, {len(sent)} items, "
f"{len(problems)} problem(s)")
for problem in problems:
print(f" {problem}")
return problems


async def run(results_dir):
failed = False
for name, streaming, wrong_last in (("batch pass", False, False),
("streaming pass", True, True)):
try:
job_id, sent = await run_pass(streaming, wrong_last)
except Exception as e:
print(f"[FAIL] {name}: {type(e).__name__}: {e}")
failed = True
continue
wrong_id = list(sent)[-1] if wrong_last else None
failed |= bool(check(name, results_dir, job_id, sent, wrong_id))
return 1 if failed else 0


def main():
parser = argparse.ArgumentParser(description=__doc__)
# CsvReporter forces this directory on the gRPC path.
parser.add_argument("--results-dir", default="/tmp_session_files/results",
help="Directory that holds <job_id>/scores.csv.")
args = parser.parse_args()
return asyncio.run(run(args.results_dir))


if __name__ == "__main__":
sys.exit(main())
64 changes: 64 additions & 0 deletions .ci/release/lib.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,64 @@
# Helpers for the steps in release.cloudbuild.yaml:
# . /workspace/.ci/release/lib.sh
#
# POSIX sh, so the busybox shell in the gcrane image can source it. Needs
# BUILD_ID, COMMIT_SHA, LOCATION and PROJECT_ID from options.env.

# alert_policy.yaml matches this prefix and parses the line format. If you
# change either, update the policy.
RELEASE_RESULT_PREFIX="RELEASE_RESULT"

# Prints the result line of the run.
#
# Usage: release_result STATUS STAGE DETAIL
# STATUS SUCCESS, FAILED, ROLLED_BACK, or ROLLBACK_FAILED
# STAGE the id of the step that ended the run
# DETAIL one sentence for the email
#
# Reads the tag, PR list, and compare link from /workspace. Keep detail
# last, because the alert policy reads it with a greedy regex.
release_result() {
_tag=$(cat /workspace/release_tag.txt 2>/dev/null || echo "none")
_detail=$(printf '%s' "$3" | tr '"\n' "' ")
_prs=$(cat /workspace/release_prs.txt 2>/dev/null \
|| echo "PR list not computed before this step ended the run.")
_prs=$(printf '%s' "$_prs" | tr '"\n' "' ")
_compare=$(cat /workspace/release_compare.txt 2>/dev/null || echo "unavailable")
_log="https://console.cloud.google.com/cloud-build/builds;region=${LOCATION}/${BUILD_ID}?project=${PROJECT_ID}"
_commit="https://github.com/GoogleCloudPlatform/evalbench/commit/${COMMIT_SHA}"
echo "${RELEASE_RESULT_PREFIX} status=$1 stage=$2 tag=${_tag} commit=${_commit} log=${_log} compare=${_compare} prs=\"${_prs}\" detail=\"${_detail}\""
}

release_fail() {
release_result FAILED "$1" "$2"
exit 1
}

# Prints the ERROR and CRITICAL records and tracebacks in an eval_server log.
# Prints nothing if the log is clean. Matches only at the start of a line,
# because eval payloads can quote a traceback.
#
# Usage: log_problems < LOG
log_problems() {
_log=$(cat)
printf '%s\n' "$_log" \
| grep -nE '^[0-9-]{10} [0-9:,]+ \[[^]]*\] (ERROR|CRITICAL) ' | head -40
printf '%s\n' "$_log" \
| grep -n -A8 '^Traceback (most recent call last)' | head -80
}

# Points TAG at DIGEST. 'tags update' moves an existing tag without the
# tags.delete permission. 'docker tags add' creates a new tag.
#
# Usage: move_tag IMAGE TAG DIGEST
# IMAGE the image path, e.g. us-central1-docker.pkg.dev/PROJECT/REPO/PKG
move_tag() {
_host=$(echo "$1" | cut -d/ -f1)
_project=$(echo "$1" | cut -d/ -f2)
_repository=$(echo "$1" | cut -d/ -f3)
_package=$(echo "$1" | cut -d/ -f4-)
gcloud artifacts tags update "$2" --location="${_host%-docker.pkg.dev}" \
--project="$_project" --repository="$_repository" \
--package="$_package" --version="$3" 2>/dev/null \
|| gcloud artifacts docker tags add "$1@$3" "$1:$2"
}
Loading
Loading