2 Commits

Author SHA1 Message Date
Kit OC5
fe3bce39ff bridge: post pending for queued jobs a runner can take
All checks were successful
check / gate (push) Successful in 28s
Gitea 1.24 lists only picked-up jobs, so a queued PR showed NOTHING on
GitHub and lanes asked whether their push was lost (Windy Mind #131,
Windy Cloud today). The bridge now reads waiting jobs from the gitea DB
and posts pending where nothing newer was picked up; a queued re-run
supersedes the stale failure it replaces.

Only status 5 jobs whose runs-on labels a live runner has: blocked jobs
often end skipped and label-unrunnable jobs are cancelled unpicked, and
neither ever reaches /actions/tasks, so their pending would never resolve.
Lookup is bounded (30 s) and non-fatal: the IO-stall lesson.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-09-23 14:20:32 -04:00
Kit OC5
6b0b60eb4a scripts: rerun_ci.sh — re-fire a PR's CI without the web button
Gitea 1.24 has no rerun API and the web button needs Grant's SSO identity.
Guarded branch rewind that the next sync undoes; restores the branch itself
on timeout. Used today for eternitas #166 and windy-mind #131 after the
Veron IO stall.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-09-23 14:17:10 -04:00
4 changed files with 187 additions and 1 deletions

View File

@@ -79,10 +79,11 @@ class Fake:
@pytest.fixture
def fake(monkeypatch):
def make(**kw):
def make(queued=(), **kw):
f = Fake(**kw)
monkeypatch.setattr(bridge, "gitea", f.gitea)
monkeypatch.setattr(bridge, "github", f.github)
monkeypatch.setattr(bridge, "queued_jobs", lambda repo, sha: list(queued))
return f
return make
@@ -272,3 +273,70 @@ def test_gitea_dir_wins_over_github_dir(fake):
)
def test_workflow_problem(text, problem):
assert bridge.workflow_problem(text) == problem
def _q(n, wf, job):
return {"run_number": n, "workflow_id": wf, "name": job}
def test_queued_job_shows_pending_instead_of_nothing(fake):
f = fake(queued=[_q(5, "ci.yml", "test")])
bridge.post_statuses("windy-chat", SHA)
assert [(p["context"], p["state"]) for p in f.posted] == [("windy-git/ci/test", "pending")]
assert f.posted[0]["target_url"].endswith("/actions/runs/5")
def test_queued_rerun_supersedes_the_stale_failure(fake):
f = fake(runs=[_run(1, "ci.yml", "test", "failure", n=4)], queued=[_q(7, "ci.yml", "test")])
bridge.post_statuses("windy-chat", SHA)
assert [(p["context"], p["state"]) for p in f.posted] == [("windy-git/ci/test", "pending")]
def test_older_queued_job_never_overrides_a_newer_verdict(fake):
f = fake(runs=[_run(1, "ci.yml", "test", "success", n=9)], queued=[_q(3, "ci.yml", "test")])
bridge.post_statuses("windy-chat", SHA)
assert [(p["context"], p["state"]) for p in f.posted] == [("windy-git/ci/test", "success")]
def test_queued_docker_and_non_blocking_jobs_stay_unposted(fake):
f = fake(queued=[_q(2, "ci.yml", "docker-build"), _q(2, "ci.yml", "build-desktop")])
bridge.post_statuses("windy-pro", SHA)
assert f.posted == []
def test_queued_lookup_refuses_unsafe_input():
assert bridge.queued_jobs("x'; drop table t;--", SHA) == []
assert bridge.queued_jobs("windy-chat", "not-a-sha") == []
def _db(monkeypatch, jobs, labels):
import json as _json
import subprocess as _sp
payload = _json.dumps({"jobs": jobs, "labels": [_json.dumps(x) for x in labels]})
monkeypatch.setattr(
bridge.subprocess, "run",
lambda *a, **k: _sp.CompletedProcess(a, 0, stdout=payload, stderr=""),
)
RUNNER = ["veron-1", "linux-x64", "self-hosted", "linux", "x64"]
def test_only_jobs_a_runner_can_take_are_pending(monkeypatch):
# macos-latest is cancelled unpicked by the janitor: pending would never resolve.
_db(monkeypatch, [
{"run_number": 3, "workflow_id": "ci.yml", "name": "test", "runs_on": '["self-hosted","linux","x64"]'},
{"run_number": 3, "workflow_id": "ci.yml", "name": "mac", "runs_on": '["macos-latest"]'},
], [RUNNER])
assert [j["name"] for j in bridge.queued_jobs("windy-chat", SHA)] == ["test"]
def test_lookup_failure_is_non_fatal(monkeypatch):
import subprocess as _sp
def boom(*a, **k):
raise _sp.TimeoutExpired("docker", 30)
monkeypatch.setattr(bridge.subprocess, "run", boom)
assert bridge.queued_jobs("windy-chat", SHA) == []

View File

@@ -152,3 +152,16 @@ disqualifying the moment a stranger depends on it. **The trigger is not a date
it is the first external push.** Move the control plane to a dedicated VPS (not
Kit 0), keep Veron 1 as the runner. It is an rsync, a Postgres dump and three
DNS record edits.
## Re-run a PR's CI (Gitea 1.24 has no rerun API)
```bash
ssh wg-veron
cd /srv/windygit/src && bash scripts/rerun_ci.sh <repo> <branch> <github-head-sha-prefix>
```
Moves the Windy Git branch back one commit; the next sync force-pushes the
GitHub head again and Gitea re-fires every workflow for that event on the same
commit. Guarded: refuses unless the branch is at the given sha, waits for a
sync that starts AFTER the rewind, restores the branch itself on timeout.
Don't use the web "Re-run" button: it needs a hub-SSO session as windyadmin,
which is Grant's identity.

View File

@@ -32,6 +32,7 @@ import base64
import json
import os
import re
import subprocess
import sys
import time
import urllib.error
@@ -222,6 +223,54 @@ def sync_prs(repo: str) -> list[str]:
return heads
SAFE_NAME = re.compile(r"^[A-Za-z0-9._-]+$")
SAFE_SHA = re.compile(r"^[0-9a-f]{40}$")
def queued_jobs(repo: str, sha: str) -> list[dict]:
"""Jobs at `sha` that are waiting for a runner and that a runner CAN take.
Gitea 1.24's API lists only PICKED-UP jobs (/actions/tasks), so a queued PR
showed nothing on GitHub and people asked whether the push was lost. The
truth is in the gitea DB. Bounded + non-fatal: during the 09-23 IO stall
`docker exec` hung for an hour and must never wedge the bridge again.
Only status 5 (waiting) with labels some live runner has. A `pending` we
post must end in a verdict we will also see, or it sits yellow forever:
blocked jobs (7) often end SKIPPED, and jobs for labels no runner has
(macos-/windows-/ubuntu-latest) are cancelled by the janitor unpicked;
neither ever appears in /actions/tasks.
"""
if not (SAFE_NAME.match(repo) and SAFE_NAME.match(WG_OWNER) and SAFE_SHA.match(sha)):
return []
query = (
"select json_build_object("
" 'jobs', (select coalesce(json_agg(t), '[]'::json) from ("
" select ar.index as run_number, ar.workflow_id, j.name, j.runs_on"
" from action_run_job j join action_run ar on ar.id = j.run_id"
" join repository r on r.id = j.repo_id join \"user\" o on o.id = r.owner_id"
f" where o.lower_name = '{WG_OWNER.lower()}' and r.lower_name = '{repo.lower()}'"
f" and ar.commit_sha = '{sha}' and j.status = 5) t),"
" 'labels', (select coalesce(json_agg(agent_labels), '[]'::json)"
" from action_runner where coalesce(deleted, 0) = 0));"
)
try:
out = subprocess.run(
["docker", "exec", "-i", "windy-git-db-1", "sh", "-c",
'psql -U "$POSTGRES_USER" -d gitea -At -v ON_ERROR_STOP=1'],
input=query, capture_output=True, text=True, check=True, timeout=30,
).stdout.strip()
got = json.loads(out or "{}")
runners = [set(json.loads(x or "[]")) for x in got.get("labels") or []]
return [
j for j in got.get("jobs") or []
if any(set(json.loads(j.get("runs_on") or "[]")) <= r for r in runners)
]
except (subprocess.SubprocessError, OSError, ValueError) as e:
print(f" {repo}: queued-job lookup skipped ({type(e).__name__})")
return []
def post_statuses(repo: str, sha: str) -> None:
# Gitea caps a page at 50 (MAX_RESPONSE_ITEMS) whatever `limit` says, and a
# daily scheduled workflow can push a quiet main's runs off page 1.
@@ -242,6 +291,17 @@ def post_statuses(repo: str, sha: str) -> None:
ctx = f"windy-git/{r['workflow_id'].removesuffix('.yml')}/{r['name']}"
if ctx not in latest or r["id"] > latest[ctx]["id"]:
latest[ctx] = r
# Queued jobs: `pending` where nothing newer has been picked up. A re-run
# queued behind an old failure must read pending, not the stale red.
for q in queued_jobs(repo, sha):
if NO_DAEMON_JOB.search(q["name"]):
continue
wf = q["workflow_id"].removesuffix(".yml")
if f"{wf}/{q['name']}" in NON_BLOCKING.get(repo, ()):
continue
ctx = f"windy-git/{wf}/{q['name']}"
if ctx not in latest or q["run_number"] > latest[ctx]["run_number"]:
latest[ctx] = {"id": 0, "status": "waiting", "run_number": q["run_number"]}
bad = invalid_workflows(repo, sha)
if not (latest or bad):
return

45
scripts/rerun_ci.sh Executable file
View File

@@ -0,0 +1,45 @@
#!/usr/bin/env bash
# Re-run a PR's (or branch's) CI on Windy Git. Runs ON Veron 1.
#
# bash scripts/rerun_ci.sh <repo> <branch> <sha-prefix>
#
# Gitea 1.24 has NO rerun API; the web button needs a hub-SSO session as
# windyadmin, which is Grant's identity, so we don't use it. Instead: move the
# Windy Git branch back one commit, let the next sync force-push the GitHub head
# again, and Gitea fires an ordinary push / pull_request_sync event on the SAME
# commit. Every workflow on that event re-runs, not only the failed one.
#
# Safety: refuses unless the branch is exactly at <sha-prefix> (GitHub's head),
# never rewinds while a sync is running (a run already past this repo would
# not push it back), waits for a sync that STARTS after the rewind, and if the
# branch is not verifiably back at <sha-prefix> by the deadline, restores it
# itself, so Windy Git is never left behind GitHub.
set -euo pipefail
repo="${1:?repo}"; branch="${2:?branch}"; want="${3:?sha prefix}"
G="sudo docker exec -u git windy-git-gitea-1 git -C /data/git/repositories/windyadmin/${repo}.git"
head=$($G rev-parse "refs/heads/${branch}")
[[ "$head" == "$want"* ]] || { echo "refusing: ${branch} is at ${head:0:7}, not ${want}"; exit 1; }
parent=$($G rev-parse "${head}^")
while systemctl is-active -q windygit-sync; do sleep 5; done
$G update-ref "refs/heads/${branch}" "$parent" "$head"
mark=$(awk '{print int($1*1000000)}' /proc/uptime)
echo "rewound ${repo}:${branch} ${head:0:7} -> ${parent:0:7}"
deadline=$(( $(date +%s) + 900 ))
until [ "$(systemctl show windygit-sync -p ExecMainStartTimestampMonotonic --value)" -gt "$mark" ] \
&& [ "$($G rev-parse "refs/heads/${branch}")" = "$head" ]; do
if [ "$(date +%s)" -ge "$deadline" ]; then
$G update-ref "refs/heads/${branch}" "$head" "$($G rev-parse "refs/heads/${branch}")" || true
echo "TIMEOUT: restored ${branch} to ${head:0:7} by hand; NO new run fired"; exit 1
fi
sleep 10
done
echo "restored by sync: ${branch} = ${head:0:7}"
sleep 5
~/bin/wg-q <<SQL
select ar.index, ar.workflow_id, ar.event, ar.status, to_char(to_timestamp(ar.created),'HH24:MI:SS')
from action_run ar join repository r on r.id = ar.repo_id
where r.name = '${repo}' and ar.commit_sha = '${head}' order by ar.id desc limit 6;
SQL