fix(tunnel): break the expired-leaf renewal deadlock
`POST /renew` is authenticated by mTLS with the very leaf it renews, so once that leaf lapsed the host could never renew it and the tunnel stayed down until an operator re-paired by hand. Production hit exactly this: the Mac slept through its 8h renewal window, the 24h leaf expired, and the agent then logged `client certificate has expired; renew before dialling` 6380 times over 8 days without recovering. Three layers independently refused an expired leaf, so all three had to move: - agent `buildTlsOptions` gains an opt-in `expiredGraceMs`. Absent/0 keeps the historical fail-closed behaviour, and the TUNNEL dial never passes it — only the renew transport does. Past the window it throws the new `CertExpiredBeyondGraceError`. - agent rotator routes an already-expired leaf to a separate recovery endpoint and treats "beyond grace" as TERMINAL: report once via `onExhausted`, stop retrying, and name the fix (re-pair) instead of spamming warnings forever. - control-plane `assertPresentedCertTrusted` grants a bounded grace on `notAfter` only. Chain validation, SPIFFE identity, `notBefore`, and the registry `active`/account checks are all unchanged, so revocation still bites. - new `deploy/nginx/recover-mtls.conf` (:8472). The strict enroll vhost cannot host this: under `ssl_verify_client optional` nginx answers a bare 400 as soon as a presented cert fails verification, so an expired leaf never reaches the location — and the directive is server-level, not per-location. Grace defaults to 30 days on both ends and is configurable (`RECOVER_URL`, `expiredRenewGraceMs`). The honest trade is recorded in the code: a stale stolen leaf stays reusable for the window, which widens an existing exposure (an unexpired stolen leaf already renews indefinitely) rather than opening a new one. Verified: agent 296/296, control-plane 286/286, tsc clean on both.
This commit is contained in:
@@ -14,6 +14,7 @@ import type { AgentIdentity } from '../keys/identity.js'
|
||||
import type { Keystore } from '../keys/keystore.js'
|
||||
import type { TimerLike } from '../transport/seams.js'
|
||||
import { createBackoff, type BackoffPolicy } from '../transport/backoff.js'
|
||||
import { CertExpiredBeyondGraceError } from '../transport/dial.js'
|
||||
import { buildCsr } from '../enroll/csr.js'
|
||||
import { certResponseToPem } from './pem.js'
|
||||
|
||||
@@ -26,6 +27,11 @@ export interface CertRotator {
|
||||
onRevoked(cb: () => void): void
|
||||
/** A renewal attempt failed (network/HTTP, NOT a 403 revoke). The rotator retries with backoff. */
|
||||
onError(cb: (err: unknown) => void): void
|
||||
/**
|
||||
* TERMINAL: the leaf expired past the renewal grace window, so no future attempt can succeed. The
|
||||
* rotator has stopped; recovery requires an operator re-pair.
|
||||
*/
|
||||
onExhausted(cb: (err: CertExpiredBeyondGraceError) => void): void
|
||||
}
|
||||
|
||||
export type RenewOutcome = 'rotated' | 'revoked'
|
||||
@@ -35,6 +41,36 @@ export function renewalUrlFor(cfg: AgentConfig): string {
|
||||
return cfg.enrollUrl.replace(/\/enroll$/, '/renew')
|
||||
}
|
||||
|
||||
/** The `enroll.<zone>` label the recovery vhost mirrors as `recover.<zone>`. */
|
||||
const ENROLL_LABEL = 'enroll.'
|
||||
const RECOVER_LABEL = 'recover.'
|
||||
|
||||
/**
|
||||
* Renewal route for an ALREADY-EXPIRED leaf, or null when this deployment has none.
|
||||
*
|
||||
* The normal `/renew` vhost runs `ssl_verify_client optional`, which makes nginx answer a bare
|
||||
* `400 Bad Request` as soon as a presented client cert fails verification — an expired leaf never
|
||||
* reaches the location, so it can never be forwarded to the control-plane. Recovery therefore lives
|
||||
* on a sibling `recover.` vhost running `optional_no_ca`, where nginx forwards the cert unverified
|
||||
* and the control-plane is the sole (and full) verifier: chain → SPIFFE → registry status → bounded
|
||||
* expiry grace. Explicit `cfg.recoverUrl` wins; otherwise it is derived by swapping the `enroll.`
|
||||
* label. A deployment whose enroll host has no `enroll.` label gets null (no recovery configured)
|
||||
* rather than a guessed hostname.
|
||||
*/
|
||||
export function recoveryRenewalUrlFor(cfg: AgentConfig): string | null {
|
||||
if (cfg.recoverUrl != null && cfg.recoverUrl.length > 0) return cfg.recoverUrl
|
||||
let url: URL
|
||||
try {
|
||||
url = new URL(cfg.enrollUrl)
|
||||
} catch {
|
||||
return null
|
||||
}
|
||||
if (!url.hostname.startsWith(ENROLL_LABEL)) return null
|
||||
url.hostname = RECOVER_LABEL + url.hostname.slice(ENROLL_LABEL.length)
|
||||
url.pathname = url.pathname.replace(/\/enroll$/, '/renew')
|
||||
return url.toString()
|
||||
}
|
||||
|
||||
/** Ms until (validTo − renewBeforeMs), clamped to ≥ 0. */
|
||||
export function computeRenewDelayMs(
|
||||
certPem: string,
|
||||
@@ -56,9 +92,10 @@ export async function renewCert(
|
||||
id: AgentIdentity,
|
||||
ks: Keystore,
|
||||
fetchImpl: typeof fetch,
|
||||
opts: { url?: string } = {},
|
||||
): Promise<RenewOutcome> {
|
||||
const csr = buildCsr(id, cfg.subdomain ?? 'web-terminal-agent')
|
||||
const res = await fetchImpl(renewalUrlFor(cfg), {
|
||||
const res = await fetchImpl(opts.url ?? renewalUrlFor(cfg), {
|
||||
method: 'POST',
|
||||
headers: { 'content-type': 'application/json' },
|
||||
body: JSON.stringify({ csr }),
|
||||
@@ -102,6 +139,7 @@ export function createCertRotator(
|
||||
let rotatedCb: (() => void) | null = null
|
||||
let revokedCb: (() => void) | null = null
|
||||
let errorCb: ((err: unknown) => void) | null = null
|
||||
let exhaustedCb: ((err: CertExpiredBeyondGraceError) => void) | null = null
|
||||
|
||||
function schedule(): void {
|
||||
const certs = ks.loadCert()
|
||||
@@ -110,8 +148,21 @@ export function createCertRotator(
|
||||
handle = timer.setTimeout(runRenewal, delay)
|
||||
}
|
||||
|
||||
/**
|
||||
* Pick the endpoint for THIS attempt. A leaf that has already lapsed cannot be forwarded by the
|
||||
* strict vhost, so it must go to the recovery vhost; a still-valid leaf always uses the normal one.
|
||||
* With no recovery endpoint configured we fall back to the normal URL and let it fail honestly.
|
||||
*/
|
||||
function endpointForNow(): string {
|
||||
const certs = ks.loadCert()
|
||||
if (certs === null) return renewalUrlFor(cfg)
|
||||
const expired = parseCert(certs.certPem).getTime() < now().getTime()
|
||||
if (!expired) return renewalUrlFor(cfg)
|
||||
return recoveryRenewalUrlFor(cfg) ?? renewalUrlFor(cfg)
|
||||
}
|
||||
|
||||
function runRenewal(): void {
|
||||
void renewCert(cfg, id, ks, doFetch)
|
||||
void renewCert(cfg, id, ks, doFetch, { url: endpointForNow() })
|
||||
.then((outcome) => {
|
||||
if (outcome === 'revoked') {
|
||||
revokedCb?.()
|
||||
@@ -122,6 +173,14 @@ export function createCertRotator(
|
||||
schedule()
|
||||
})
|
||||
.catch((err: unknown) => {
|
||||
// The grace window is spent: no amount of retrying can mint a new leaf, only an operator
|
||||
// re-pair can. Report it ONCE and stop — the old behaviour retried forever and buried the
|
||||
// real signal under thousands of identical warnings.
|
||||
if (err instanceof CertExpiredBeyondGraceError) {
|
||||
handle = null
|
||||
exhaustedCb?.(err)
|
||||
return
|
||||
}
|
||||
// Network/HTTP failure (never a 403 revoke): surface it (caller logs, no secret) and retry
|
||||
// with backoff. The cert is still valid until expiry, so the tunnel stays up meanwhile — a
|
||||
// failed renewal must NEVER tear the supervisor down.
|
||||
@@ -147,5 +206,8 @@ export function createCertRotator(
|
||||
onError(cb): void {
|
||||
errorCb = cb
|
||||
},
|
||||
onExhausted(cb): void {
|
||||
exhaustedCb = cb
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user