G1: stop probing the tunnel from inside a container
cloudflared binds 127.0.0.1:2000 on the HOST. This process runs in a container whose only route to the host is the bridge gateway (172.17.0.1), where nothing is listening — so the check was permanently red regardless of what the tunnel was actually doing. Binding the metrics endpoint wider would have fixed the probe and made a metrics bind failure capable of taking down ingress. That is a worse trade than losing one row on a dashboard. The check is not silently dropped: /health/full now carries a 'not_checked_here' map naming the tunnel and where its health actually lives (systemd windygit-tunnel). An observer should never have to wonder whether a missing check means healthy or means forgotten. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -50,12 +50,6 @@ class Settings(BaseSettings):
|
||||
# ---- account-server OIDC (human identity) -----------------------------
|
||||
account_server_base_url: str = "https://account.windyword.ai"
|
||||
|
||||
# cloudflared binds its metrics on the HOST, so from inside a container
|
||||
# `localhost` is the wrong box. A health check that is permanently red is as
|
||||
# useless as one that is permanently green -- it trains people to ignore the
|
||||
# dashboard, which is how a 37-day-dead fleet canary goes unnoticed.
|
||||
tunnel_metrics_url: str = "http://host.docker.internal:2000/metrics"
|
||||
|
||||
# ---- storage law (I-3, G4.4) ------------------------------------------
|
||||
# Git object databases MUST live on a POSIX filesystem. A test asserts this
|
||||
# path does not resolve to a network mount.
|
||||
|
||||
@@ -23,7 +23,6 @@ from api.app.providers.registry import (
|
||||
EternitasProvider,
|
||||
GiteaProvider,
|
||||
R2Provider,
|
||||
TunnelProvider,
|
||||
)
|
||||
from api.app.routes import health
|
||||
|
||||
@@ -86,7 +85,14 @@ async def lifespan(app: FastAPI):
|
||||
GiteaProvider(settings),
|
||||
R2Provider(settings),
|
||||
EternitasProvider(settings),
|
||||
TunnelProvider(settings),
|
||||
# NOTE: the tunnel is deliberately NOT probed from here. cloudflared
|
||||
# binds its metrics on the host's loopback, so a container can never
|
||||
# reach it -- the check would be permanently red no matter what the
|
||||
# tunnel is doing. A check that structurally cannot succeed is worse
|
||||
# than no check: it trains people to ignore the dashboard, which is
|
||||
# exactly how a fleet canary goes 37 days dead without anyone noticing.
|
||||
# Tunnel health is a host concern and lives where it can be observed:
|
||||
# systemd Restart=always, plus the runbook's `systemctl status`.
|
||||
]
|
||||
|
||||
yield
|
||||
|
||||
@@ -104,25 +104,9 @@ class DatabaseProvider(Provider):
|
||||
return ProbeResult(True, "postgres reachable", True)
|
||||
|
||||
|
||||
class TunnelProvider(Provider):
|
||||
"""cloudflared is the only ingress. No inbound port is ever opened (G1.2)."""
|
||||
|
||||
name = "tunnel"
|
||||
|
||||
def __init__(self, settings: Settings) -> None:
|
||||
self._s = settings
|
||||
|
||||
@property
|
||||
def configured(self) -> bool:
|
||||
# The tunnel is a host-level concern, not a credential we hold, so there
|
||||
# is nothing to "configure" here. The probe alone decides health, and in
|
||||
# dev it will honestly say cloudflared is not running (I-8).
|
||||
return True
|
||||
|
||||
async def probe(self) -> ProbeResult:
|
||||
async with httpx.AsyncClient(timeout=_TIMEOUT) as client:
|
||||
try:
|
||||
r = await client.get(self._s.tunnel_metrics_url)
|
||||
except httpx.RequestError as exc:
|
||||
return ProbeResult(False, f"cloudflared metrics unreachable: {exc}")
|
||||
return ProbeResult(r.status_code == 200, f"cloudflared metrics -> {r.status_code}", True)
|
||||
# TunnelProvider was removed deliberately. See the note in main.py: cloudflared
|
||||
# binds 127.0.0.1:2000 on the HOST, and this process runs in a container whose
|
||||
# only route to the host is the bridge gateway (172.17.0.1), where nothing is
|
||||
# listening. Binding the metrics endpoint wider would fix the probe and make a
|
||||
# metrics bind failure able to take down ingress -- a worse trade than losing
|
||||
# one row on a dashboard. The tunnel is supervised by systemd instead.
|
||||
|
||||
@@ -66,6 +66,14 @@ async def health_full(request: Request, response: Response) -> dict:
|
||||
"status": status,
|
||||
"commit_sha": info.commit_sha,
|
||||
"checks": checks,
|
||||
# Named, not hidden. An observer should never have to wonder whether a
|
||||
# missing check means healthy or means forgotten.
|
||||
"not_checked_here": {
|
||||
"tunnel": (
|
||||
"host-scoped: cloudflared binds host loopback and is supervised "
|
||||
"by systemd (windygit-tunnel). Verify with `systemctl status`."
|
||||
)
|
||||
},
|
||||
# Grandma-words, and the D-9 vocabulary law binds this string.
|
||||
"speak": (
|
||||
"Everything is working."
|
||||
|
||||
@@ -24,9 +24,6 @@ services:
|
||||
# nothing needs to be reachable from the LAN, let alone the internet (G1.6).
|
||||
ports: ["127.0.0.1:${API_PORT:-8600}:8600"]
|
||||
depends_on: {db: {condition: service_healthy}}
|
||||
# cloudflared runs on the host, not in this network. Without this the tunnel
|
||||
# probe is permanently red and stops meaning anything.
|
||||
extra_hosts: ["host.docker.internal:host-gateway"]
|
||||
restart: unless-stopped
|
||||
|
||||
gitea:
|
||||
|
||||
Reference in New Issue
Block a user