From 6b9608755441cc2e20c8dc824553f67d7fc1e3ad Mon Sep 17 00:00:00 2001 From: Grant Whitmer Date: Wed, 12 Aug 2026 11:18:51 -0400 Subject: [PATCH] =?UTF-8?q?G7:=20per-job=20networks=20=E2=80=94=20fixes=20?= =?UTF-8?q?service=20DNS=20and=20tightens=20isolation?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The migration step failed with 'could not translate host name postgres'. The Postgres service container was healthy; the job simply could not name it, because service DNS aliases only exist on a per-job network and I had pinned container.network to the flat 'bridge'. That choice was wrong in both directions: it broke service containers AND it was weaker isolation, since every concurrent job shared one bridge and could see its neighbours. A per-job network is stricter and correct — and still has no route to the forge, because these networks live inside the dind daemon, which has no forge attachment at all. Co-Authored-By: Claude Opus 5 --- api/tests/test_invariants.py | 10 ++++++++++ deploy/runner/config.yaml | 16 ++++++++++++---- 2 files changed, 22 insertions(+), 4 deletions(-) diff --git a/api/tests/test_invariants.py b/api/tests/test_invariants.py index 8aeb5cf..1877cd4 100644 --- a/api/tests/test_invariants.py +++ b/api/tests/test_invariants.py @@ -409,6 +409,16 @@ def test_i05_runner_never_mounts_the_host_docker_socket(): assert "/var/run/docker.sock" not in ln, "I-5: never mount the host docker socket" +def test_i05_jobs_get_a_network_per_job_not_a_shared_bridge(): + """A shared flat bridge lets concurrent jobs see each other, and breaks + service-container DNS (service aliases only exist on a per-job network). + Per-job is both stricter and correct.""" + cfg = (ROOT / "deploy" / "runner" / "config.yaml").read_text() + active = [ln for ln in cfg.splitlines() if ln.strip() and not ln.strip().startswith("#")] + net = [ln for ln in active if ln.strip().startswith("network:")] + assert net and net[0].strip() == 'network: ""', "jobs must get a per-job network" + + def test_i05_jobs_cannot_bind_mount_from_the_daemon_host(): cfg = (ROOT / "deploy" / "runner" / "config.yaml").read_text() assert "valid_volumes: []" in cfg diff --git a/deploy/runner/config.yaml b/deploy/runner/config.yaml index 78888d8..07e0230 100644 --- a/deploy/runner/config.yaml +++ b/deploy/runner/config.yaml @@ -26,10 +26,18 @@ cache: dir: /data/cache container: - # Job containers join the dind daemon's own bridge. NOT the forge network: - # untrusted code must never be able to reach the forge's Postgres or its - # environment (I-5). - network: bridge + # Empty = act creates a NETWORK PER JOB and removes it afterwards. + # + # This started as `bridge` for isolation, which was a mistake in both + # directions. It broke service containers — Postgres came up healthy but the + # job could not resolve the name `postgres`, because service DNS aliases only + # exist on a per-job network — and it was *weaker* isolation, since every + # concurrent job shared one flat bridge and could see its neighbours. + # + # A per-job network is both correct and stricter. Still no route to the forge: + # these networks live inside the dind daemon, which has no forge attachment + # at all. + network: "" privileged: false options: workdir_parent: /workspace