wulf-pulse/lib/services/route53-change-submit.ts
lorentz f4e151dedd feat(24-05): implement route53-change-submit (ChangeBatch, retry, bounded poll)
- buildChangeBatch: CREATE/UPSERT/DELETE, SetIdentifier omission, NS/SOA guard (D-01 defence in depth)
- isRetryableAwsError: classifies ThrottlingException/Throttling/PriorRequestNotComplete/ServiceUnavailable
- submitRecordChange: sends ChangeResourceRecordSetsCommand with bounded 2-retry backoff (750ms/1500ms)
- pollChangeStatus: bounded 15s/2s GetChange poll, no 30-minute SDK waiter
- 20/20 tests passing, tsc clean
2026-08-05 20:31:43 -04:00

204 lines
6.7 KiB
TypeScript

/**
* AWS Route 53 DNS sync — change-batch construction and bounded propagation
* poll.
*
* Lives under lib/** (not app/api/**) so vitest.config.ts's `include:
* ['lib/**\/*.test.ts']` glob can reach it — the AWS-command construction and
* polling loop are pure/testable, and route files are not.
*
* D-01 defence in depth: `buildChangeBatch` throws on NS/SOA independently of
* `validateRecordWrite` (lib/services/route53-record-validation.ts), which is
* the primary, earlier-running gate. This is a second backstop so no future
* caller can bypass it by calling this module directly.
*
* 24-RESEARCH.md Anti-Patterns: never use the SDK's built-in
* resource-record-sets-changed waiter inside a request handler — its default
* config is a 30-second interval with 60 attempts (up to 30 minutes).
* `pollChangeStatus` is a short, bounded, hand-rolled poll instead.
*/
import {
Route53Client,
ChangeResourceRecordSetsCommand,
GetChangeCommand,
type ChangeBatch,
type ChangeAction,
type ResourceRecordSet,
} from '@aws-sdk/client-route-53';
import { getRoute53Client } from './route53-factory';
const ZONE_DELEGATION_TYPES = new Set(['NS', 'SOA']);
const RETRYABLE_ERROR_NAMES = new Set([
'ThrottlingException',
'Throttling',
'PriorRequestNotComplete',
'ServiceUnavailable',
]);
export interface ChangeSubmitRecordSet {
name: string;
type: string;
ttl: number;
resourceRecords: Array<{ value: string }>;
setIdentifier?: string | null;
}
/**
* Build a `@aws-sdk/client-route-53` ChangeBatch for a single recordset
* change. Omits `SetIdentifier` from the emitted object when null/undefined
* rather than setting it to `undefined` — AWS's SDK serializer treats an
* explicit `undefined` property differently from an absent one in some
* marshalling paths, so we simply never set the key.
*
* Throws when `type.toUpperCase()` is `NS` or `SOA` — a second, independent
* D-01 enforcement point (T-24-01, defence in depth) so a caller cannot
* bypass `validateRecordWrite` by calling this module directly.
*/
export function buildChangeBatch(
action: 'CREATE' | 'UPSERT' | 'DELETE',
recordSet: ChangeSubmitRecordSet
): ChangeBatch {
const type = recordSet.type.toUpperCase();
if (ZONE_DELEGATION_TYPES.has(type)) {
throw new Error(
`Refusing to build a change batch for record type ${type} — NS and SOA are zone-delegation records (D-01)`
);
}
const resourceRecordSet: ResourceRecordSet = {
Name: recordSet.name,
Type: type as ResourceRecordSet['Type'],
TTL: recordSet.ttl,
ResourceRecords: recordSet.resourceRecords.map((r) => ({ Value: r.value })),
};
if (recordSet.setIdentifier !== null && recordSet.setIdentifier !== undefined) {
resourceRecordSet.SetIdentifier = recordSet.setIdentifier;
}
return {
Changes: [
{
Action: action as ChangeAction,
ResourceRecordSet: resourceRecordSet,
},
],
};
}
/**
* Classify an AWS SDK error as retryable. `PriorRequestNotComplete` is a
* per-zone serialization constraint (two writes to the same hosted zone in
* quick succession), not a hard failure — per 24-RESEARCH.md Pitfall 4.
*/
export function isRetryableAwsError(err: unknown): boolean {
if (!(err instanceof Error)) return false;
return RETRYABLE_ERROR_NAMES.has(err.name);
}
function sleep(ms: number): Promise<void> {
return new Promise((resolve) => setTimeout(resolve, ms));
}
export interface SubmitRecordChangeInput {
zoneId: string;
action: 'CREATE' | 'UPSERT' | 'DELETE';
recordSet: ChangeSubmitRecordSet;
client?: Route53Client;
}
export interface SubmitRecordChangeResult {
changeId: string | null;
awsResponse: unknown;
}
const RETRY_BACKOFFS_MS = [750, 1500];
/**
* Construct a ChangeResourceRecordSetsCommand and send it. On a retryable
* error (per `isRetryableAwsError`), retry up to 2 additional times with
* 750ms then 1500ms backoff; any other error rethrows immediately.
*
* Does not add a general backoff wrapper around every AWS call — the SDK's
* built-in retry strategy already handles transport-level retries
* (24-RESEARCH.md "Don't Hand-Roll"). This retry is specifically for the
* application-level PriorRequestNotComplete / throttling cases.
*/
export async function submitRecordChange(
input: SubmitRecordChangeInput
): Promise<SubmitRecordChangeResult> {
const client = input.client ?? getRoute53Client();
const command = new ChangeResourceRecordSetsCommand({
HostedZoneId: input.zoneId,
ChangeBatch: buildChangeBatch(input.action, input.recordSet),
});
let lastError: unknown;
for (let attempt = 0; attempt <= RETRY_BACKOFFS_MS.length; attempt++) {
try {
const response = await client.send(command);
return {
changeId: response.ChangeInfo?.Id ?? null,
awsResponse: response,
};
} catch (err) {
lastError = err;
if (!isRetryableAwsError(err) || attempt === RETRY_BACKOFFS_MS.length) {
throw err;
}
await sleep(RETRY_BACKOFFS_MS[attempt]);
}
}
// Unreachable — the loop above always returns or throws — but keeps
// TypeScript's control-flow analysis satisfied.
throw lastError;
}
export interface PollChangeStatusOptions {
client?: Route53Client;
timeoutMs?: number;
intervalMs?: number;
}
const DEFAULT_POLL_TIMEOUT_MS = 15000;
const DEFAULT_POLL_INTERVAL_MS = 2000;
/**
* Bounded poll of GetChangeCommand until ChangeInfo.Status === 'INSYNC' or
* the timeout budget elapses, in which case 'PENDING' is returned and no
* further calls are made. Per-attempt errors are swallowed and polling
* continues until the budget elapses — a transient GetChange failure is not
* a write failure; the write was already accepted by AWS.
*
* CRITICAL: do NOT use the SDK's built-in resource-record-sets-changed
* waiter here — its default config (30s interval, 60 attempts) can block an
* HTTP request handler for up to 30 minutes.
*/
export async function pollChangeStatus(
changeId: string,
opts?: PollChangeStatusOptions
): Promise<'INSYNC' | 'PENDING'> {
const client = opts?.client ?? getRoute53Client();
const timeoutMs = opts?.timeoutMs ?? DEFAULT_POLL_TIMEOUT_MS;
const intervalMs = opts?.intervalMs ?? DEFAULT_POLL_INTERVAL_MS;
const deadline = Date.now() + timeoutMs;
while (Date.now() < deadline) {
try {
const response = await client.send(new GetChangeCommand({ Id: changeId }));
if (response.ChangeInfo?.Status === 'INSYNC') {
return 'INSYNC';
}
} catch {
// Swallow — the write already succeeded; GetChange transiently
// failing is not a write failure. Keep polling until the budget
// elapses.
}
if (Date.now() >= deadline) break;
await sleep(intervalMs);
}
return 'PENDING';
}