telemetry: ci.job_cancelled from the janitor; interval_s on heartbeat
The janitor now returns one JSON line per job it cancels (repo, workflow, job, reason, runs_on, waited_s) into a spool; the emitter ships them as ci.job_cancelled (declared with Telemetry Boss) and truncates the spool only after a 2xx. Run status recompute folded into the same statement. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -1,6 +1,10 @@
|
||||
#!/usr/bin/env bash
|
||||
# Cancel jobs no runner can ever take (see cancel_unrunnable.sql). Run on Veron as root.
|
||||
set -euo pipefail
|
||||
n=$(docker exec -i windy-git-db-1 sh -c 'psql -U "$POSTGRES_USER" -d gitea -At -v ON_ERROR_STOP=1' \
|
||||
< "$(dirname "$0")/cancel_unrunnable.sql" | grep -cE '^[0-9]+$' || true)
|
||||
echo "[janitor] cancelled unrunnable jobs in ${n} run(s)"
|
||||
SPOOL="${JANITOR_SPOOL:-/var/lib/windy-git/janitor-cancelled.jsonl}"
|
||||
mkdir -p "$(dirname "$SPOOL")"
|
||||
out=$(docker exec -i windy-git-db-1 sh -c 'psql -U "$POSTGRES_USER" -d gitea -At -v ON_ERROR_STOP=1' \
|
||||
< "$(dirname "$0")/cancel_unrunnable.sql")
|
||||
printf '%s\n' "$out" | grep '^{' >> "$SPOOL" || true
|
||||
n=$(printf '%s\n' "$out" | grep -c '^{' || true)
|
||||
echo "[janitor] cancelled ${n} unrunnable job(s)"
|
||||
|
||||
@@ -15,17 +15,29 @@ WITH dead AS (
|
||||
AND to_timestamp(j.created) < now() - interval '30 minutes'
|
||||
AND EXISTS (SELECT 1 FROM jsonb_array_elements_text(j.runs_on::jsonb) l
|
||||
WHERE l NOT IN ('veron-1', 'linux-x64', 'self-hosted', 'linux', 'x64'))
|
||||
RETURNING j.run_id
|
||||
RETURNING j.id, j.run_id, j.name, j.runs_on, j.created
|
||||
), runs AS (
|
||||
UPDATE action_run r
|
||||
SET status = CASE
|
||||
WHEN EXISTS (SELECT 1 FROM action_run_job x WHERE x.run_id = r.id AND x.status = 2) THEN 2
|
||||
WHEN EXISTS (SELECT 1 FROM action_run_job x WHERE x.run_id = r.id AND x.status IN (5, 6, 7)
|
||||
AND x.id NOT IN (SELECT id FROM dead)) THEN r.status
|
||||
ELSE 3 END,
|
||||
stopped = CASE WHEN r.stopped = 0 THEN extract(epoch from now())::bigint ELSE r.stopped END
|
||||
WHERE r.id IN (SELECT DISTINCT run_id FROM dead)
|
||||
RETURNING r.id
|
||||
)
|
||||
UPDATE action_run r
|
||||
SET status = CASE
|
||||
WHEN EXISTS (SELECT 1 FROM action_run_job x WHERE x.run_id = r.id AND x.status = 2) THEN 2
|
||||
WHEN EXISTS (SELECT 1 FROM action_run_job x WHERE x.run_id = r.id AND x.status IN (5, 6, 7)) THEN r.status
|
||||
WHEN EXISTS (SELECT 1 FROM action_run_job x WHERE x.run_id = r.id AND x.status = 3) THEN 3
|
||||
ELSE 1 END,
|
||||
stopped = CASE WHEN r.stopped = 0 THEN extract(epoch from now())::bigint ELSE r.stopped END
|
||||
WHERE r.id IN (SELECT DISTINCT run_id FROM dead)
|
||||
RETURNING r.id;
|
||||
-- One JSON line per cancelled job: the telemetry emitter ships these as
|
||||
-- ci.job_cancelled (declared with Telemetry Boss, 2026-09-23).
|
||||
SELECT json_build_object(
|
||||
'repo', p.lower_name,
|
||||
'workflow', regexp_replace(r.workflow_id, '\.ya?ml$', ''),
|
||||
'job', d.name,
|
||||
'reason', 'unrunnable_label',
|
||||
'runs_on', (SELECT string_agg(l, ',') FROM jsonb_array_elements_text(d.runs_on::jsonb) l),
|
||||
'waited_s', (extract(epoch from now())::bigint - d.created))::text
|
||||
FROM dead d JOIN action_run r ON r.id = d.run_id JOIN repository p ON p.id = r.repo_id
|
||||
WHERE (SELECT count(*) FROM runs) >= 0;
|
||||
|
||||
-- Jobs BLOCKED on `needs:` inside a run that has already finished (a needed job
|
||||
-- failed): Gitea leaves them status 7 forever. They were never going to run;
|
||||
|
||||
@@ -135,6 +135,7 @@ def main() -> int:
|
||||
(select coalesce(extract(epoch from now())::bigint - min(created), 0)
|
||||
from action_run_job where status in (5, 7)) as oldest_waiting_s""")[0]
|
||||
meta = {k: int(v) for k, v in h.items()}
|
||||
meta["interval_s"] = int(now - since) # ecosystem-standard key
|
||||
for k in ("repos_synced", "repos_sync_failed", "statuses_posted", "bridge_errors"):
|
||||
v = os.environ.get(f"TELEMETRY_{k.upper()}")
|
||||
if v is not None and v.isdigit(): # absent = couldn't count; never invent 0
|
||||
@@ -150,6 +151,33 @@ def main() -> int:
|
||||
}
|
||||
)
|
||||
|
||||
# ci.job_cancelled: spooled by the janitor (cancel_unrunnable.sh), one JSON per job.
|
||||
spool = os.environ.get("JANITOR_SPOOL", "/var/lib/windy-git/janitor-cancelled.jsonl")
|
||||
spooled = 0
|
||||
try:
|
||||
with open(spool) as f:
|
||||
for line in f:
|
||||
try:
|
||||
m = json.loads(line)
|
||||
except ValueError:
|
||||
continue
|
||||
events.append(
|
||||
{
|
||||
"ts": iso(now),
|
||||
"platform": PLATFORM,
|
||||
"service": SERVICE,
|
||||
"event_type": "ci.job_cancelled",
|
||||
"actor_type": "system",
|
||||
"metadata": {
|
||||
k: m[k]
|
||||
for k in ("repo", "workflow", "job", "reason", "runs_on", "waited_s")
|
||||
},
|
||||
}
|
||||
)
|
||||
spooled += 1
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
if dry:
|
||||
print(json.dumps({"events": events}, indent=1)[:4000])
|
||||
print(f"[telemetry] DRY RUN: {len(events)} events ({len(jobs)} ci.run)")
|
||||
@@ -183,6 +211,8 @@ def main() -> int:
|
||||
print(f"[telemetry] FAILED ingest: {e.reason}")
|
||||
return 1
|
||||
|
||||
if spooled:
|
||||
open(spool, "w").close() # only after every batch was accepted
|
||||
os.makedirs(os.path.dirname(STATE), exist_ok=True)
|
||||
new_last = max([j["id"] for j in jobs], default=last_job)
|
||||
with open(STATE + ".tmp", "w") as f:
|
||||
|
||||
Reference in New Issue
Block a user