Files
adr-sml/infra/cloudfront/configure.mjs
T
Pouya LajevardiandClaude Opus 5 bd282aa47d
Build and deploy / build-and-deploy (push) Failing after 4s
feat: production run — Q61 ramp, /404/, CloudFront router, cutover runbook
Five items of Pouya's production run, 2026-09-01.

Q61 — scroll-padding-top becomes a max() ramp on `10lh - 83px`, with the
plain calc() first as the fallback for engines without `lh`. Hidden focus
stops under minimumFontSize=32: 290 of 1,455 -> 0, control build still
290. Default settings byte-identical (0 differences over 352 page-widths x
17 fields). The 12 residual cells at minimumFontSize=16/20 are pre-existing
and unchanged-or-better; reported, not widened, per instruction.

Intake backend + CloudFront — docs/09-cutover-runbook.md is the
copy-paste sequence for admin execution: every command followed by its
verification and expected output, rollback per part, and Part 10 is Q60's
TTL test. infra/cloudfront/router.js is the trailing-slash function
(30-case suite; 8 fail against the pre-review version, incl. a
protocol-relative open redirect). infra/cloudfront/configure.mjs is
dry-run-by-default and idempotent. scripts/intake-env.mjs emits the six
Lambda env vars from src/data/site.ts.

Four launch blockers found by reading the running system:
  - handler.mjs wrote pk/sk; the live table's key is submissionId with no
    sort key, so every submission would have failed validation silently
  - the Lambda invoke permission is scoped to the old route path
  - 22 of 23 pages 403 without the router function
  - there was no 404 page; src/pages/404.astro adds it

Claims audit (D20 cutover pass) — five gloss over-reaches corrected on
/practice/energy/, /practice/insurance/ (x2), /practice/technology/ and
/med-arb/. Three findings left open for Pouya: Q62, the /med-arb/ gloss,
and Q60.

Q62 — one frozen-tripwire pattern added under the freeze's own breach
exception, with a probe and four negative fixtures. check:claims exits 1
until the false /legal/privacy/ sentence is corrected, so both deploy
paths are blocked by a mechanism rather than by memory.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Md3GndFqWPzK78xAoebsg5
2026-09-02 06:52:20 -04:00

330 lines
12 KiB
JavaScript

/**
* Applies the three distribution changes the site needs, as one reviewable
* transaction. `docs/09-cutover-runbook.md` Part 3 is what calls it.
*
* 1. FunctionAssociations on the default behaviour -> `router.js`, viewer
* request. Without it 22 of 23 pages return S3's AccessDenied XML.
* 2. CustomErrorResponses: 404 -> /404.html with response code 404.
* `docs/04` requires a genuine 404 status; `docs/06` calls a 200 here
* "the single most common misconfiguration in this stack".
* 3. A `/api/*` cache behaviour on a new origin pointing at the HTTP API, so
* the intake form's same-origin POST reaches the handler.
*
* ⚠️ DRY RUN BY DEFAULT. It prints what it would change and exits 0 without
* calling `update-distribution`. `--apply` is the only thing that writes, and it
* sends the `IfMatch` ETag it read, so a concurrent console edit fails the call
* rather than being overwritten.
*
* ⚠️ IDEMPOTENT ON PURPOSE. Every change is checked for before it is made, so a
* re-run after a partial failure completes the rest instead of adding a second
* `/api/*` behaviour. Re-running a runbook step is the normal case, not the
* exception.
*
* ⚠️ NO MANAGED POLICY ID IS WRITTEN IN THIS FILE. They are resolved by name
* from the account at run time — `CLAUDE.md`'s rule that a pin is verified
* against the registry rather than recalled applies to an AWS identifier just
* as much as to an npm version, and a wrong cache-policy id here would ship a
* cached POST endpoint.
*
* usage:
* node infra/cloudfront/configure.mjs --dist <id> --api-domain <host> [--function-arn <arn>]
* node infra/cloudfront/configure.mjs ... --apply
*/
import { execFileSync } from 'node:child_process';
const args = process.argv.slice(2);
const flag = (name) => {
const i = args.indexOf(`--${name}`);
return i === -1 ? undefined : args[i + 1];
};
const APPLY = args.includes('--apply');
const DIST = flag('dist');
const API_DOMAIN = flag('api-domain');
const FUNCTION_ARN = flag('function-arn');
if (!DIST || !API_DOMAIN) {
console.error(
'usage: node infra/cloudfront/configure.mjs --dist <distribution-id> ' +
'--api-domain <api-id>.execute-api.<region>.amazonaws.com ' +
'[--function-arn <router-function-arn>] [--apply]',
);
console.error('Values come from AGENTS.md §7.');
process.exit(2);
}
if (/^https?:/.test(API_DOMAIN) || API_DOMAIN.includes('/')) {
console.error(
`--api-domain must be a bare hostname, not a URL: got ${API_DOMAIN}`,
);
process.exit(2);
}
/* stderr is NEVER suppressed and the exit status is always read — the AWS CLI
reports an expired session, a missing permission and a typo'd id all on
stderr with a non-zero status, and swallowing that is how "it failed" becomes
"it found nothing" (CLAUDE.md, from AGENTS.md Q22). */
const aws = (argv) => {
const out = execFileSync('aws', argv, {
encoding: 'utf8',
stdio: ['ignore', 'pipe', 'inherit'],
maxBuffer: 64 * 1024 * 1024,
});
return out.trim() === '' ? null : JSON.parse(out);
};
const ORIGIN_ID = 'intake-api';
const PATH_PATTERN = '/api/*';
const ERROR_PAGE = '/404.html';
function managedId(kind, name) {
const listCmd = {
cache: ['list-cache-policies', 'CachePolicyList', 'CachePolicy'],
origreq: [
'list-origin-request-policies',
'OriginRequestPolicyList',
'OriginRequestPolicy',
],
}[kind];
const res = aws([
'cloudfront',
listCmd[0],
'--type',
'managed',
'--output',
'json',
]);
const items = res?.[listCmd[1]]?.Items ?? [];
const hit = items.find(
(i) => i[listCmd[2]][`${listCmd[2]}Config`].Name === name,
);
if (!hit) {
throw new Error(
`no managed ${kind} policy named ${name}${items.length} listed. ` +
'Do not substitute an id from memory.',
);
}
return hit[listCmd[2]].Id;
}
const cachingDisabled = managedId('cache', 'Managed-CachingDisabled');
/* AllViewerExceptHostHeader, and the exception is the whole reason: API Gateway
routes on the Host header, so forwarding the viewer's `adr.smlcompany.ca`
makes every request a 403 from the API. It forwards everything else, which is
what carries `Origin` and `Referer` — the handler's CSRF control reads both,
so a policy that dropped them would turn every real submission into a 403. */
const allViewerExceptHost = managedId(
'origreq',
'Managed-AllViewerExceptHostHeader',
);
console.log(`resolved Managed-CachingDisabled = ${cachingDisabled}`);
console.log(
`resolved Managed-AllViewerExceptHostHeader = ${allViewerExceptHost}`,
);
const current = aws([
'cloudfront',
'get-distribution-config',
'--id',
DIST,
'--output',
'json',
]);
const etag = current.ETag;
const cfg = current.DistributionConfig;
if (!etag || !cfg) throw new Error('could not read the distribution config');
const changes = [];
/* ---- 1. viewer-request function on the default behaviour ---------------- */
if (FUNCTION_ARN) {
const fa = cfg.DefaultCacheBehavior.FunctionAssociations ?? { Quantity: 0 };
const existing = (fa.Items ?? []).filter(
(i) => i.EventType === 'viewer-request',
);
if (existing.length === 1 && existing[0].FunctionARN === FUNCTION_ARN) {
console.log(
'· default behaviour already runs this function on viewer-request',
);
} else {
const items = (fa.Items ?? []).filter(
(i) => i.EventType !== 'viewer-request',
);
items.push({ EventType: 'viewer-request', FunctionARN: FUNCTION_ARN });
cfg.DefaultCacheBehavior.FunctionAssociations = {
Quantity: items.length,
Items: items,
};
changes.push(
`DefaultCacheBehavior.FunctionAssociations viewer-request -> ${FUNCTION_ARN}` +
(existing.length ? ` (replacing ${existing[0].FunctionARN})` : ''),
);
}
} else {
console.log('· no --function-arn given, leaving FunctionAssociations alone');
}
/* ---- 2. custom error response ------------------------------------------- */
/* ⚠️ ONLY 404 IS MAPPED, NOT 403, AND THAT IS DELIBERATE. Mapping 403 as well
would swallow two different real failures: a broken bucket policy or OAC
would render as "page not found" on every URL at once, and the intake
handler's Origin refusal (a 403 from the API origin) would come back as a 404
page. Custom error responses are distribution-wide — they cannot be scoped to
one behaviour — so the fix for missing keys is on the S3 side instead:
granting the OAC principal `s3:ListBucket` makes S3 answer 404 NoSuchKey
rather than 403 AccessDenied. Runbook Part 1 does that first, and its
verification step is what proves this mapping is reached. */
const cer = cfg.CustomErrorResponses ?? { Quantity: 0, Items: [] };
const cerItems = cer.Items ?? [];
const has404 = cerItems.some(
(i) =>
i.ErrorCode === 404 &&
i.ResponsePagePath === ERROR_PAGE &&
String(i.ResponseCode) === '404',
);
if (has404) {
console.log('· 404 -> /404.html (404) already configured');
} else {
/* Report a REPLACEMENT as a replacement. This branch filters out any existing
404 mapping, so on a distribution that maps 404 to a different page the
operator would otherwise be told a mapping was "added" while one was
silently changed — and Part 3 tells them to carry on when the change count is
lower than expected. The function-association branch above already names
what it replaces; this one did not. */
const replaced = cerItems.find((i) => i.ErrorCode === 404);
const items = cerItems.filter((i) => i.ErrorCode !== 404);
items.push({
ErrorCode: 404,
ResponsePagePath: ERROR_PAGE,
ResponseCode: '404',
/* Short, not zero. A 404 is cheap to re-fetch and this is the value that
decides how long a genuinely-missing URL keeps 404ing after the page it
should have been is deployed. */
ErrorCachingMinTTL: 10,
});
cfg.CustomErrorResponses = { Quantity: items.length, Items: items };
changes.push(
replaced
? `CustomErrorResponses 404 -> ${ERROR_PAGE} with status 404 (REPLACING ` +
`${replaced.ResponsePagePath} with status ${replaced.ResponseCode})`
: `CustomErrorResponses += 404 -> ${ERROR_PAGE} with status 404`,
);
}
/* ---- 3. the /api/* origin and behaviour --------------------------------- */
const origins = cfg.Origins.Items ?? [];
if (origins.some((o) => o.Id === ORIGIN_ID)) {
console.log(`· origin ${ORIGIN_ID} already present`);
} else {
origins.push({
Id: ORIGIN_ID,
DomainName: API_DOMAIN,
OriginPath: '',
CustomHeaders: { Quantity: 0 },
CustomOriginConfig: {
HTTPPort: 80,
HTTPSPort: 443,
/* https-only to the origin. The API is public over TLS and there is no
reason for a leg of this in plaintext. */
OriginProtocolPolicy: 'https-only',
OriginSslProtocols: { Quantity: 1, Items: ['TLSv1.2'] },
OriginReadTimeout: 30,
OriginKeepaliveTimeout: 5,
},
ConnectionAttempts: 3,
ConnectionTimeout: 10,
OriginShield: { Enabled: false },
});
cfg.Origins = { Quantity: origins.length, Items: origins };
changes.push(
`Origins += ${ORIGIN_ID} -> ${API_DOMAIN} (https-only, TLSv1.2)`,
);
}
const behaviours = cfg.CacheBehaviors?.Items ?? [];
if (behaviours.some((b) => b.PathPattern === PATH_PATTERN)) {
console.log(`· cache behaviour ${PATH_PATTERN} already present`);
} else {
behaviours.push({
PathPattern: PATH_PATTERN,
TargetOriginId: ORIGIN_ID,
ViewerProtocolPolicy: 'https-only',
/* POST is the one that matters; the rest are here because CloudFront only
offers the three fixed method sets and this is the set containing POST. */
AllowedMethods: {
Quantity: 7,
Items: ['GET', 'HEAD', 'POST', 'PUT', 'PATCH', 'OPTIONS', 'DELETE'],
CachedMethods: { Quantity: 2, Items: ['GET', 'HEAD'] },
},
CachePolicyId: cachingDisabled,
OriginRequestPolicyId: allViewerExceptHost,
Compress: false,
SmoothStreaming: false,
FieldLevelEncryptionId: '',
/* NO FUNCTION ASSOCIATION, AND THE OMISSION IS LOAD-BEARING. `router.js`
would 301 `/api/intake` to `/api/intake/`, and a 301 turns a POST into a
GET — the submission body would be dropped with a 200 at the end of it.
`infra/cloudfront/router.test.mjs` carries that case as documentation. */
FunctionAssociations: { Quantity: 0 },
LambdaFunctionAssociations: { Quantity: 0 },
TrustedKeyGroups: { Enabled: false, Quantity: 0 },
});
cfg.CacheBehaviors = { Quantity: behaviours.length, Items: behaviours };
changes.push(
`CacheBehaviors += ${PATH_PATTERN} -> ${ORIGIN_ID}, CachingDisabled, AllViewerExceptHostHeader, POST allowed`,
);
}
/* CloudFront matches cache behaviours in order and the FIRST match wins, so a
`/api/*` behaviour placed after a hypothetical `/*` one would never be
reached. There is no `/*` behaviour today — the default behaviour is the
catch-all and is not part of this list — but assert it rather than assume it. */
const catchAll = (cfg.CacheBehaviors?.Items ?? []).findIndex(
(b) => b.PathPattern === '*' || b.PathPattern === '/*',
);
const apiIndex = (cfg.CacheBehaviors?.Items ?? []).findIndex(
(b) => b.PathPattern === PATH_PATTERN,
);
if (catchAll !== -1 && catchAll < apiIndex) {
throw new Error(
`a catch-all behaviour at index ${catchAll} precedes ${PATH_PATTERN} at ${apiIndex} — ` +
'the API behaviour would never match. Reorder before applying.',
);
}
console.log('');
if (changes.length === 0) {
console.log(
'NOTHING TO CHANGE — the distribution already carries all three.',
);
process.exit(0);
}
console.log(
`${changes.length} change(s) to distribution ${DIST} (ETag ${etag}):`,
);
for (const c of changes) console.log(` + ${c}`);
console.log('');
if (!APPLY) {
console.log('DRY RUN — nothing was sent. Re-run with --apply to write it.');
process.exit(0);
}
const res = aws([
'cloudfront',
'update-distribution',
'--id',
DIST,
'--if-match',
etag,
'--distribution-config',
JSON.stringify(cfg),
'--output',
'json',
]);
console.log(
`APPLIED. Status=${res.Distribution.Status} ETag=${res.ETag}\n` +
'CloudFront takes a few minutes to deploy. Wait for Deployed, then run the ' +
"runbook's verification block:\n" +
` aws cloudfront wait distribution-deployed --id ${DIST}`,
);