diff --git a/scripts/cancel_unrunnable.sh b/scripts/cancel_unrunnable.sh new file mode 100755 index 0000000..2a99966 --- /dev/null +++ b/scripts/cancel_unrunnable.sh @@ -0,0 +1,6 @@ +#!/usr/bin/env bash +# Cancel jobs no runner can ever take (see cancel_unrunnable.sql). Run on Veron as root. +set -euo pipefail +n=$(docker exec -i windy-git-db-1 sh -c 'psql -U "$POSTGRES_USER" -d gitea -At -v ON_ERROR_STOP=1' \ + < "$(dirname "$0")/cancel_unrunnable.sql" | grep -cE '^[0-9]+$' || true) +echo "[janitor] cancelled unrunnable jobs in ${n} run(s)" diff --git a/scripts/cancel_unrunnable.sql b/scripts/cancel_unrunnable.sql new file mode 100644 index 0000000..d0bccb9 --- /dev/null +++ b/scripts/cancel_unrunnable.sql @@ -0,0 +1,29 @@ +-- Cancel CI jobs that can never run (called by scripts/cancel_unrunnable.sh). +-- +-- A job whose runs-on names a label no Windy Git runner offers (ubuntu-latest, +-- macos-latest, windows-latest …) waits forever: Gitea evaluates a job's `if:` +-- only when a runner picks it, so even `if: false` / tag-only jobs sit in the +-- queue, invisible to /actions/tasks, and keep their run "waiting" for good. +-- After 30 minutes they are cancelled here; the run's status is then recomputed +-- (failure > still-active > cancelled > success), the same precedence Gitea uses. +-- Keep RUNNER_LABELS in step with deploy/runner/config.yaml. +BEGIN; +WITH dead AS ( + UPDATE action_run_job j + SET status = 3, stopped = extract(epoch from now())::bigint, updated = extract(epoch from now())::bigint + WHERE j.status IN (5, 7) + AND to_timestamp(j.created) < now() - interval '30 minutes' + AND EXISTS (SELECT 1 FROM jsonb_array_elements_text(j.runs_on::jsonb) l + WHERE l NOT IN ('veron-1', 'linux-x64', 'self-hosted', 'linux', 'x64')) + RETURNING j.run_id +) +UPDATE action_run r + SET status = CASE + WHEN EXISTS (SELECT 1 FROM action_run_job x WHERE x.run_id = r.id AND x.status = 2) THEN 2 + WHEN EXISTS (SELECT 1 FROM action_run_job x WHERE x.run_id = r.id AND x.status IN (5, 6, 7)) THEN r.status + WHEN EXISTS (SELECT 1 FROM action_run_job x WHERE x.run_id = r.id AND x.status = 3) THEN 3 + ELSE 1 END, + stopped = CASE WHEN r.stopped = 0 THEN extract(epoch from now())::bigint ELSE r.stopped END + WHERE r.id IN (SELECT DISTINCT run_id FROM dead) +RETURNING r.id; +COMMIT; diff --git a/scripts/sync_from_github.sh b/scripts/sync_from_github.sh index 01d54fd..8aeb886 100755 --- a/scripts/sync_from_github.sh +++ b/scripts/sync_from_github.sh @@ -80,6 +80,10 @@ for r in $REPOS; do fi done +# Jobs that name labels no runner has (ubuntu/macos/windows-latest) would wait +# forever and invisibly; cancel them after 30 min. Never fails the sync. +bash "$(dirname "$0")/cancel_unrunnable.sh" || log "janitor failed (non-fatal)" + # Private repos can't run GitHub Actions; mirror their open PRs here so CI # fires, and post the verdicts back to GitHub as commit statuses. if ! python3 "$(dirname "$0")/pr_status_bridge.py"; then