diff --git a/src/lifecycle/egress.ts b/src/lifecycle/egress.ts index ae1fd1b..386018b 100644 --- a/src/lifecycle/egress.ts +++ b/src/lifecycle/egress.ts @@ -76,6 +76,19 @@ export async function confirmEgress( ); } log(` ${machine} reaches the internet over its uplink (${UPSTREAM} answered ${code})`); + + // **Now that the path is proven, stop one dropped packet from costing a whole run.** + // + // The check above deliberately uses the uplink alone, because proving that path is the point + // of it. What follows is a different concern: a bed runs for hours after this, pulling images + // on four machines at once, and the uplink's resolver is a single server under exactly that + // load. Three runs have died at a lookup timeout well after this check passed — not because + // the path was broken, but because one query went unanswered. + // + // So the uplink stays first and keeps being used; public resolvers are appended behind it, and + // the retry is tightened so a silent server costs seconds rather than the step. A broken uplink + // still fails the check above, before any of this. + await resilientResolver(name); } } @@ -117,6 +130,23 @@ async function reaches(name: string, waitSeconds: number): Promise { + await incus([ + "exec", name, "--", "sh", "-c", + `via=$(ip -4 route show default | awk '{print $3}' | head -n1); ` + + `{ [ -n "$via" ] && printf 'nameserver %s\\n' "$via"; ` + + `printf 'nameserver 1.1.1.1\\nnameserver 8.8.8.8\\n'; ` + + `printf 'options timeout:2 attempts:3\\n'; } > /etc/resolv.conf; true`, + ], 30_000); +} + async function pointResolverAtTheUplink(name: string): Promise { await incus([ "exec", name, "--", "sh", "-c",