From b5e4eaf57a3f2c5087b25751562c4b4ff7faf295 Mon Sep 17 00:00:00 2001 From: Grant Whitmer Date: Wed, 23 Sep 2026 11:27:53 -0400 Subject: [PATCH] telemetry: ci.job_cancelled from the janitor; interval_s on heartbeat The janitor now returns one JSON line per job it cancels (repo, workflow, job, reason, runs_on, waited_s) into a spool; the emitter ships them as ci.job_cancelled (declared with Telemetry Boss) and truncates the spool only after a 2xx. Run status recompute folded into the same statement. Co-Authored-By: Claude Opus 5.5 --- scripts/cancel_unrunnable.sh | 10 +++++++--- scripts/cancel_unrunnable.sql | 32 ++++++++++++++++++++++---------- scripts/telemetry_emit.py | 30 ++++++++++++++++++++++++++++++ 3 files changed, 59 insertions(+), 13 deletions(-) diff --git a/scripts/cancel_unrunnable.sh b/scripts/cancel_unrunnable.sh index 2a99966..dd08e2f 100755 --- a/scripts/cancel_unrunnable.sh +++ b/scripts/cancel_unrunnable.sh @@ -1,6 +1,10 @@ #!/usr/bin/env bash # Cancel jobs no runner can ever take (see cancel_unrunnable.sql). Run on Veron as root. set -euo pipefail -n=$(docker exec -i windy-git-db-1 sh -c 'psql -U "$POSTGRES_USER" -d gitea -At -v ON_ERROR_STOP=1' \ - < "$(dirname "$0")/cancel_unrunnable.sql" | grep -cE '^[0-9]+$' || true) -echo "[janitor] cancelled unrunnable jobs in ${n} run(s)" +SPOOL="${JANITOR_SPOOL:-/var/lib/windy-git/janitor-cancelled.jsonl}" +mkdir -p "$(dirname "$SPOOL")" +out=$(docker exec -i windy-git-db-1 sh -c 'psql -U "$POSTGRES_USER" -d gitea -At -v ON_ERROR_STOP=1' \ + < "$(dirname "$0")/cancel_unrunnable.sql") +printf '%s\n' "$out" | grep '^{' >> "$SPOOL" || true +n=$(printf '%s\n' "$out" | grep -c '^{' || true) +echo "[janitor] cancelled ${n} unrunnable job(s)" diff --git a/scripts/cancel_unrunnable.sql b/scripts/cancel_unrunnable.sql index 2bf25e7..5eb9930 100644 --- a/scripts/cancel_unrunnable.sql +++ b/scripts/cancel_unrunnable.sql @@ -15,17 +15,29 @@ WITH dead AS ( AND to_timestamp(j.created) < now() - interval '30 minutes' AND EXISTS (SELECT 1 FROM jsonb_array_elements_text(j.runs_on::jsonb) l WHERE l NOT IN ('veron-1', 'linux-x64', 'self-hosted', 'linux', 'x64')) - RETURNING j.run_id + RETURNING j.id, j.run_id, j.name, j.runs_on, j.created +), runs AS ( + UPDATE action_run r + SET status = CASE + WHEN EXISTS (SELECT 1 FROM action_run_job x WHERE x.run_id = r.id AND x.status = 2) THEN 2 + WHEN EXISTS (SELECT 1 FROM action_run_job x WHERE x.run_id = r.id AND x.status IN (5, 6, 7) + AND x.id NOT IN (SELECT id FROM dead)) THEN r.status + ELSE 3 END, + stopped = CASE WHEN r.stopped = 0 THEN extract(epoch from now())::bigint ELSE r.stopped END + WHERE r.id IN (SELECT DISTINCT run_id FROM dead) + RETURNING r.id ) -UPDATE action_run r - SET status = CASE - WHEN EXISTS (SELECT 1 FROM action_run_job x WHERE x.run_id = r.id AND x.status = 2) THEN 2 - WHEN EXISTS (SELECT 1 FROM action_run_job x WHERE x.run_id = r.id AND x.status IN (5, 6, 7)) THEN r.status - WHEN EXISTS (SELECT 1 FROM action_run_job x WHERE x.run_id = r.id AND x.status = 3) THEN 3 - ELSE 1 END, - stopped = CASE WHEN r.stopped = 0 THEN extract(epoch from now())::bigint ELSE r.stopped END - WHERE r.id IN (SELECT DISTINCT run_id FROM dead) -RETURNING r.id; +-- One JSON line per cancelled job: the telemetry emitter ships these as +-- ci.job_cancelled (declared with Telemetry Boss, 2026-09-23). +SELECT json_build_object( + 'repo', p.lower_name, + 'workflow', regexp_replace(r.workflow_id, '\.ya?ml$', ''), + 'job', d.name, + 'reason', 'unrunnable_label', + 'runs_on', (SELECT string_agg(l, ',') FROM jsonb_array_elements_text(d.runs_on::jsonb) l), + 'waited_s', (extract(epoch from now())::bigint - d.created))::text + FROM dead d JOIN action_run r ON r.id = d.run_id JOIN repository p ON p.id = r.repo_id + WHERE (SELECT count(*) FROM runs) >= 0; -- Jobs BLOCKED on `needs:` inside a run that has already finished (a needed job -- failed): Gitea leaves them status 7 forever. They were never going to run; diff --git a/scripts/telemetry_emit.py b/scripts/telemetry_emit.py index f8343ef..a2f4672 100644 --- a/scripts/telemetry_emit.py +++ b/scripts/telemetry_emit.py @@ -135,6 +135,7 @@ def main() -> int: (select coalesce(extract(epoch from now())::bigint - min(created), 0) from action_run_job where status in (5, 7)) as oldest_waiting_s""")[0] meta = {k: int(v) for k, v in h.items()} + meta["interval_s"] = int(now - since) # ecosystem-standard key for k in ("repos_synced", "repos_sync_failed", "statuses_posted", "bridge_errors"): v = os.environ.get(f"TELEMETRY_{k.upper()}") if v is not None and v.isdigit(): # absent = couldn't count; never invent 0 @@ -150,6 +151,33 @@ def main() -> int: } ) + # ci.job_cancelled: spooled by the janitor (cancel_unrunnable.sh), one JSON per job. + spool = os.environ.get("JANITOR_SPOOL", "/var/lib/windy-git/janitor-cancelled.jsonl") + spooled = 0 + try: + with open(spool) as f: + for line in f: + try: + m = json.loads(line) + except ValueError: + continue + events.append( + { + "ts": iso(now), + "platform": PLATFORM, + "service": SERVICE, + "event_type": "ci.job_cancelled", + "actor_type": "system", + "metadata": { + k: m[k] + for k in ("repo", "workflow", "job", "reason", "runs_on", "waited_s") + }, + } + ) + spooled += 1 + except OSError: + pass + if dry: print(json.dumps({"events": events}, indent=1)[:4000]) print(f"[telemetry] DRY RUN: {len(events)} events ({len(jobs)} ci.run)") @@ -183,6 +211,8 @@ def main() -> int: print(f"[telemetry] FAILED ingest: {e.reason}") return 1 + if spooled: + open(spool, "w").close() # only after every batch was accepted os.makedirs(os.path.dirname(STATE), exist_ok=True) new_last = max([j["id"] for j in jobs], default=last_job) with open(STATE + ".tmp", "w") as f: