/** * Applies the four distribution changes the site needs, as one reviewable * transaction. `docs/09-cutover-runbook.md` Part 3 is what calls it. * * 1. FunctionAssociations on the default behaviour -> `router.js`, viewer * request. Without it 22 of 23 pages return S3's AccessDenied XML. * 2. CustomErrorResponses: 404 -> /404.html with response code 404. * `docs/04` requires a genuine 404 status; `docs/06` calls a 200 here * "the single most common misconfiguration in this stack". * 3. A `/api/*` cache behaviour on a new origin pointing at the HTTP API, so * the intake form's same-origin POST reaches the handler. * 4. A `*.pdf` cache behaviour carrying a response-headers policy that adds * `X-Robots-Tag: noindex`, so the bio PDF is not indexed as a duplicate of * `/bio/`. `docs/06`'s checklist item carries the reasoning. * * ⚠️ DRY RUN BY DEFAULT. It prints what it would change and exits 0 without * calling `update-distribution`. `--apply` is the only thing that writes, and it * sends the `IfMatch` ETag it read, so a concurrent console edit fails the call * rather than being overwritten. * * ⚠️ IDEMPOTENT ON PURPOSE. Every change is checked for before it is made, so a * re-run after a partial failure completes the rest instead of adding a second * `/api/*` behaviour. Re-running a runbook step is the normal case, not the * exception. * * ⚠️ NO MANAGED POLICY ID IS WRITTEN IN THIS FILE. They are resolved by name * from the account at run time — `CLAUDE.md`'s rule that a pin is verified * against the registry rather than recalled applies to an AWS identifier just * as much as to an npm version, and a wrong cache-policy id here would ship a * cached POST endpoint. * * usage: * node infra/cloudfront/configure.mjs --dist --api-domain [--function-arn ] * node infra/cloudfront/configure.mjs ... --apply */ import { execFileSync } from 'node:child_process'; const args = process.argv.slice(2); const flag = (name) => { const i = args.indexOf(`--${name}`); return i === -1 ? undefined : args[i + 1]; }; const APPLY = args.includes('--apply'); const DIST = flag('dist'); const API_DOMAIN = flag('api-domain'); const FUNCTION_ARN = flag('function-arn'); if (!DIST || !API_DOMAIN) { console.error( 'usage: node infra/cloudfront/configure.mjs --dist ' + '--api-domain .execute-api..amazonaws.com ' + '[--function-arn ] [--apply]', ); console.error('Values come from AGENTS.md §7.'); process.exit(2); } if (/^https?:/.test(API_DOMAIN) || API_DOMAIN.includes('/')) { console.error( `--api-domain must be a bare hostname, not a URL: got ${API_DOMAIN}`, ); process.exit(2); } /* stderr is NEVER suppressed and the exit status is always read — the AWS CLI reports an expired session, a missing permission and a typo'd id all on stderr with a non-zero status, and swallowing that is how "it failed" becomes "it found nothing" (CLAUDE.md, from AGENTS.md Q22). */ const aws = (argv) => { const out = execFileSync('aws', argv, { encoding: 'utf8', stdio: ['ignore', 'pipe', 'inherit'], maxBuffer: 64 * 1024 * 1024, }); return out.trim() === '' ? null : JSON.parse(out); }; const ORIGIN_ID = 'intake-api'; const PATH_PATTERN = '/api/*'; const ERROR_PAGE = '/404.html'; function managedId(kind, name) { const listCmd = { cache: ['list-cache-policies', 'CachePolicyList', 'CachePolicy'], origreq: [ 'list-origin-request-policies', 'OriginRequestPolicyList', 'OriginRequestPolicy', ], }[kind]; const res = aws([ 'cloudfront', listCmd[0], '--type', 'managed', '--output', 'json', ]); const items = res?.[listCmd[1]]?.Items ?? []; const hit = items.find( (i) => i[listCmd[2]][`${listCmd[2]}Config`].Name === name, ); if (!hit) { throw new Error( `no managed ${kind} policy named ${name} — ${items.length} listed. ` + 'Do not substitute an id from memory.', ); } return hit[listCmd[2]].Id; } const cachingDisabled = managedId('cache', 'Managed-CachingDisabled'); /* AllViewerExceptHostHeader, and the exception is the whole reason: API Gateway routes on the Host header, so forwarding the viewer's `adr.smlcompany.ca` makes every request a 403 from the API. It forwards everything else, which is what carries `Origin` and `Referer` — the handler's CSRF control reads both, so a policy that dropped them would turn every real submission into a 403. */ const allViewerExceptHost = managedId( 'origreq', 'Managed-AllViewerExceptHostHeader', ); console.log(`resolved Managed-CachingDisabled = ${cachingDisabled}`); console.log( `resolved Managed-AllViewerExceptHostHeader = ${allViewerExceptHost}`, ); const current = aws([ 'cloudfront', 'get-distribution-config', '--id', DIST, '--output', 'json', ]); const etag = current.ETag; const cfg = current.DistributionConfig; if (!etag || !cfg) throw new Error('could not read the distribution config'); const changes = []; /* ---- 1. viewer-request function on the default behaviour ---------------- */ if (FUNCTION_ARN) { const fa = cfg.DefaultCacheBehavior.FunctionAssociations ?? { Quantity: 0 }; const existing = (fa.Items ?? []).filter( (i) => i.EventType === 'viewer-request', ); if (existing.length === 1 && existing[0].FunctionARN === FUNCTION_ARN) { console.log( '· default behaviour already runs this function on viewer-request', ); } else { const items = (fa.Items ?? []).filter( (i) => i.EventType !== 'viewer-request', ); items.push({ EventType: 'viewer-request', FunctionARN: FUNCTION_ARN }); cfg.DefaultCacheBehavior.FunctionAssociations = { Quantity: items.length, Items: items, }; changes.push( `DefaultCacheBehavior.FunctionAssociations viewer-request -> ${FUNCTION_ARN}` + (existing.length ? ` (replacing ${existing[0].FunctionARN})` : ''), ); } } else { console.log('· no --function-arn given, leaving FunctionAssociations alone'); } /* ---- 2. custom error response ------------------------------------------- */ /* ⚠️ ONLY 404 IS MAPPED, NOT 403, AND THAT IS DELIBERATE. Mapping 403 as well would swallow two different real failures: a broken bucket policy or OAC would render as "page not found" on every URL at once, and the intake handler's Origin refusal (a 403 from the API origin) would come back as a 404 page. Custom error responses are distribution-wide — they cannot be scoped to one behaviour — so the fix for missing keys is on the S3 side instead: granting the OAC principal `s3:ListBucket` makes S3 answer 404 NoSuchKey rather than 403 AccessDenied. Runbook Part 1 does that first, and its verification step is what proves this mapping is reached. */ const cer = cfg.CustomErrorResponses ?? { Quantity: 0, Items: [] }; const cerItems = cer.Items ?? []; const has404 = cerItems.some( (i) => i.ErrorCode === 404 && i.ResponsePagePath === ERROR_PAGE && String(i.ResponseCode) === '404', ); if (has404) { console.log('· 404 -> /404.html (404) already configured'); } else { /* Report a REPLACEMENT as a replacement. This branch filters out any existing 404 mapping, so on a distribution that maps 404 to a different page the operator would otherwise be told a mapping was "added" while one was silently changed — and Part 3 tells them to carry on when the change count is lower than expected. The function-association branch above already names what it replaces; this one did not. */ const replaced = cerItems.find((i) => i.ErrorCode === 404); const items = cerItems.filter((i) => i.ErrorCode !== 404); items.push({ ErrorCode: 404, ResponsePagePath: ERROR_PAGE, ResponseCode: '404', /* Short, not zero. A 404 is cheap to re-fetch and this is the value that decides how long a genuinely-missing URL keeps 404ing after the page it should have been is deployed. */ ErrorCachingMinTTL: 10, }); cfg.CustomErrorResponses = { Quantity: items.length, Items: items }; changes.push( replaced ? `CustomErrorResponses 404 -> ${ERROR_PAGE} with status 404 (REPLACING ` + `${replaced.ResponsePagePath} with status ${replaced.ResponseCode})` : `CustomErrorResponses += 404 -> ${ERROR_PAGE} with status 404`, ); } /* ---- 3. the /api/* origin and behaviour --------------------------------- */ const origins = cfg.Origins.Items ?? []; if (origins.some((o) => o.Id === ORIGIN_ID)) { console.log(`· origin ${ORIGIN_ID} already present`); } else { origins.push({ Id: ORIGIN_ID, DomainName: API_DOMAIN, OriginPath: '', CustomHeaders: { Quantity: 0 }, CustomOriginConfig: { HTTPPort: 80, HTTPSPort: 443, /* https-only to the origin. The API is public over TLS and there is no reason for a leg of this in plaintext. */ OriginProtocolPolicy: 'https-only', OriginSslProtocols: { Quantity: 1, Items: ['TLSv1.2'] }, OriginReadTimeout: 30, OriginKeepaliveTimeout: 5, }, ConnectionAttempts: 3, ConnectionTimeout: 10, OriginShield: { Enabled: false }, }); cfg.Origins = { Quantity: origins.length, Items: origins }; changes.push( `Origins += ${ORIGIN_ID} -> ${API_DOMAIN} (https-only, TLSv1.2)`, ); } const behaviours = cfg.CacheBehaviors?.Items ?? []; if (behaviours.some((b) => b.PathPattern === PATH_PATTERN)) { console.log(`· cache behaviour ${PATH_PATTERN} already present`); } else { behaviours.push({ PathPattern: PATH_PATTERN, TargetOriginId: ORIGIN_ID, ViewerProtocolPolicy: 'https-only', /* POST is the one that matters; the rest are here because CloudFront only offers the three fixed method sets and this is the set containing POST. */ AllowedMethods: { Quantity: 7, Items: ['GET', 'HEAD', 'POST', 'PUT', 'PATCH', 'OPTIONS', 'DELETE'], CachedMethods: { Quantity: 2, Items: ['GET', 'HEAD'] }, }, CachePolicyId: cachingDisabled, OriginRequestPolicyId: allViewerExceptHost, Compress: false, SmoothStreaming: false, FieldLevelEncryptionId: '', /* NO FUNCTION ASSOCIATION, AND THE OMISSION IS LOAD-BEARING. `router.js` would 301 `/api/intake` to `/api/intake/`, and a 301 turns a POST into a GET — the submission body would be dropped with a 200 at the end of it. `infra/cloudfront/router.test.mjs` carries that case as documentation. */ FunctionAssociations: { Quantity: 0 }, LambdaFunctionAssociations: { Quantity: 0 }, TrustedKeyGroups: { Enabled: false, Quantity: 0 }, }); cfg.CacheBehaviors = { Quantity: behaviours.length, Items: behaviours }; changes.push( `CacheBehaviors += ${PATH_PATTERN} -> ${ORIGIN_ID}, CachingDisabled, AllViewerExceptHostHeader, POST allowed`, ); } /* CloudFront matches cache behaviours in order and the FIRST match wins, so a `/api/*` behaviour placed after a hypothetical `/*` one would never be reached. There is no `/*` behaviour today — the default behaviour is the catch-all and is not part of this list — but assert it rather than assume it. */ const catchAll = (cfg.CacheBehaviors?.Items ?? []).findIndex( (b) => b.PathPattern === '*' || b.PathPattern === '/*', ); const apiIndex = (cfg.CacheBehaviors?.Items ?? []).findIndex( (b) => b.PathPattern === PATH_PATTERN, ); if (catchAll !== -1 && catchAll < apiIndex) { throw new Error( `a catch-all behaviour at index ${catchAll} precedes ${PATH_PATTERN} at ${apiIndex} — ` + 'the API behaviour would never match. Reorder before applying.', ); } /* ---- 4. X-Robots-Tag: noindex on the bio PDF ---------------------------- ⚠️ S3 OBJECT METADATA CANNOT DO THIS. `aws s3 sync --metadata` writes USER metadata, which S3 returns as `x-amz-meta-x-robots-tag` — a header no crawler reads. Only a literal `X-Robots-Tag` counts and the REST endpoint will not emit one, so the mechanism is a response-headers policy. `docs/06`'s checklist item carries why the PDF needs it at all; this comment carries only what the next implementer needs in order not to break it. ⚠️ THE ONE LIVE CONSTRAINT: A RESPONSE-HEADERS POLICY REPLACES, IT DOES NOT MERGE. Attaching a policy to `*.pdf` means the default behaviour's policy no longer applies there, so this one must carry everything that policy carries — hence the clone below, and hence the drift check that follows it. Measured 2026-09-03: all five security headers arrive on the live PDF today. */ const PDF_PATTERN = '*.pdf'; const PDF_POLICY_NAME = 'adr-sml-pdf-noindex'; const XRT = { Header: 'X-Robots-Tag', Value: 'noindex', Override: true }; const defaultRhpId = cfg.DefaultCacheBehavior.ResponseHeadersPolicyId; function getResponseHeadersPolicy(id) { return aws([ 'cloudfront', 'get-response-headers-policy', '--id', id, '--output', 'json', ]); } /* Only `custom` is listed: `adr-sml-pdf-noindex` is a name this script creates, so a managed hit is impossible and listing them would be a wasted call that reads as if one were possible. */ function findPdfPolicy() { const res = aws([ 'cloudfront', 'list-response-headers-policies', '--type', 'custom', '--output', 'json', ]); const items = res?.ResponseHeadersPolicyList?.Items ?? []; return ( items.find( (i) => i.ResponseHeadersPolicy.ResponseHeadersPolicyConfig.Name === PDF_POLICY_NAME, )?.ResponseHeadersPolicy ?? null ); } /* ⚠️ SKIP, DO NOT THROW. Sections 1–3 have already staged their mutations, and throwing here would make the script unusable for re-applying the router function or the 404 mapping — which is the re-run contract this file promises at the top, and `router.js` is what keeps 22 of 23 pages off S3's AccessDenied. A missing policy on the default behaviour is section 4's problem alone. */ /* ⚠️ A SKIP IS NOT A CHANGE AND MUST NOT ENTER `changes`. That array is printed under "N change(s)", `docs/09` Part 3 tells the operator to COUNT those lines, and the `NOTHING TO CHANGE` guard exits on its length — so a skip in there would both miscount and send an `update-distribution` carrying a config nothing mutated. Skips get their own list and their own heading. */ const skipped = []; if (!defaultRhpId) { skipped.push( `${PDF_PATTERN} / ${PDF_POLICY_NAME} — the default behaviour has no ResponseHeadersPolicyId, so there is nothing to clone the security headers from`, ); } else { const existingPdfPolicy = findPdfPolicy(); let pdfPolicyId = existingPdfPolicy?.Id ?? null; /* ⚠️ RECONCILE ON EVERY RUN, NEVER ONLY AT CREATION. The clone is a copy of a fact that lives somewhere else, so it goes stale the moment the default behaviour's policy changes — and it would go stale silently, as a uniform pass. `docs/05` already specifies a Content-Security-Policy (a field OF SecurityHeadersConfig) and a Permissions-Policy (which can only be a CUSTOM header) that the site does not ship yet; adding either to the default behaviour would reach the pages and not the PDF. This check fails loudly instead, naming the diff. */ const source = getResponseHeadersPolicy(defaultRhpId); const srcCfg = source?.ResponseHeadersPolicy?.ResponseHeadersPolicyConfig; /* ⚠️ SKIP, NOT THROW — same rule as the missing-id case above, and it was inconsistent for one round. A policy carrying only `CorsConfig` is legal; an ABSENT source is section 4's problem alone and must not stop sections 1-3 from re-applying `router.js`. The DRIFT throw below is different: that is a divergence, not an absence, and `docs/09` Part 3 argues for it. */ if (!srcCfg?.SecurityHeadersConfig) { skipped.push( `${PDF_PATTERN} / ${PDF_POLICY_NAME} — response-headers policy ${defaultRhpId} has no SecurityHeadersConfig to clone`, ); } else { const wanted = { SecurityHeadersConfig: srcCfg.SecurityHeadersConfig, ...(srcCfg.CorsConfig ? { CorsConfig: srcCfg.CorsConfig } : {}), ...(srcCfg.RemoveHeadersConfig ? { RemoveHeadersConfig: srcCfg.RemoveHeadersConfig } : {}), ...(srcCfg.ServerTimingHeadersConfig ? { ServerTimingHeadersConfig: srcCfg.ServerTimingHeadersConfig } : {}), CustomHeadersConfig: { Quantity: (srcCfg.CustomHeadersConfig?.Items ?? []).length + 1, Items: [...(srcCfg.CustomHeadersConfig?.Items ?? []), XRT], }, }; if (existingPdfPolicy) { const have = existingPdfPolicy.ResponseHeadersPolicyConfig; const norm = (o) => JSON.stringify(o ?? null); const drift = [ 'SecurityHeadersConfig', 'CorsConfig', 'RemoveHeadersConfig', 'ServerTimingHeadersConfig', ] .filter((k) => norm(have[k]) !== norm(wanted[k])) .concat( norm(have.CustomHeadersConfig?.Items) !== norm(wanted.CustomHeadersConfig.Items) ? ['CustomHeadersConfig'] : [], ); if (drift.length) { /* Print BOTH SIDES of every drifted key. Naming the field alone does not tell the operator which header moved, nor which direction to reconcile in — the same message fires whether the source gained a header or the PDF policy lost its X-Robots-Tag, and those need opposite repairs. */ const detail = drift .map( (k) => ` ${k}\n pdf policy : ${norm( k === 'CustomHeadersConfig' ? have.CustomHeadersConfig?.Items : have[k], )}\n default : ${norm( k === 'CustomHeadersConfig' ? wanted.CustomHeadersConfig.Items : wanted[k], )}`, ) .join('\n'); throw new Error( `${PDF_POLICY_NAME} has DRIFTED from the default behaviour's policy ` + `${defaultRhpId} on ${drift.length} field(s). The PDF is being served ` + `different headers from the pages — read which way before repairing:\n` + `${detail}\n` + `Reconcile with update-response-headers-policy (it needs the policy's ` + `own ETag), then re-run. This script will not silently paper over it.`, ); } console.log( `· response-headers policy ${PDF_POLICY_NAME} exists and matches the default behaviour`, ); } else if (!APPLY) { console.log(`· would CREATE response-headers policy ${PDF_POLICY_NAME}`); changes.push( `create response-headers policy ${PDF_POLICY_NAME} (SecurityHeadersConfig cloned from ${defaultRhpId} + X-Robots-Tag: noindex)`, ); } else { const created = aws([ 'cloudfront', 'create-response-headers-policy', '--response-headers-policy-config', JSON.stringify({ Name: PDF_POLICY_NAME, Comment: 'Cloned from the default behaviour, plus X-Robots-Tag: noindex for *.pdf. See infra/cloudfront/configure.mjs section 4.', ...wanted, }), '--output', 'json', ]); pdfPolicyId = created?.ResponseHeadersPolicy?.Id; if (!pdfPolicyId) { throw new Error('create-response-headers-policy returned no Id'); } console.log( `created response-headers policy ${PDF_POLICY_NAME} = ${pdfPolicyId}`, ); changes.push( `created response-headers policy ${PDF_POLICY_NAME} = ${pdfPolicyId}`, ); } const pdfBehaviours = cfg.CacheBehaviors?.Items ?? []; const foundPdf = pdfBehaviours.find((b) => b.PathPattern === PDF_PATTERN); if (foundPdf) { /* ⚠️ PRESENCE IS NOT CORRECTNESS. This checked only that a `*.pdf` behaviour existed, so one added by hand — while chasing the `aws s3 sync --metadata` route this file's header records as the original instruction — would report `already present`, push nothing, and print NOTHING TO CHANGE while the PDF served no `X-Robots-Tag` at all. Section 1 compares the FunctionARN before declaring a match; so does this now. */ const wrong = []; if (foundPdf.ResponseHeadersPolicyId !== pdfPolicyId) { wrong.push( `ResponseHeadersPolicyId is ${foundPdf.ResponseHeadersPolicyId ?? '(none)'}, expected ${pdfPolicyId ?? '(the policy this script manages)'}`, ); } if (foundPdf.TargetOriginId !== cfg.DefaultCacheBehavior.TargetOriginId) { wrong.push( `TargetOriginId is ${foundPdf.TargetOriginId}, expected ${cfg.DefaultCacheBehavior.TargetOriginId}`, ); } const hasViewerRequest = ( foundPdf.FunctionAssociations?.Items ?? [] ).some((i) => i.EventType === 'viewer-request'); if (!hasViewerRequest) { wrong.push( 'no viewer-request FunctionAssociation — router.js normalises `//` and `\\` on file paths, so `//pouya-lajevardi-bio.pdf` would 404 instead of 301', ); } if (wrong.length) { throw new Error( `a ${PDF_PATTERN} cache behaviour already exists but is NOT the one this ` + `script manages:\n - ${wrong.join('\n - ')}\n` + `Reconcile or remove it before re-running; this script will not adopt ` + `a behaviour it cannot account for.`, ); } console.log( `· cache behaviour ${PDF_PATTERN} already present and correctly configured`, ); } else { const d = cfg.DefaultCacheBehavior; const behaviour = { PathPattern: PDF_PATTERN, TargetOriginId: d.TargetOriginId, ViewerProtocolPolicy: d.ViewerProtocolPolicy, AllowedMethods: d.AllowedMethods, CachePolicyId: d.CachePolicyId, /* Placeholder only in a dry run — the real id exists by the time --apply reaches this line, because the branch above created it. */ ResponseHeadersPolicyId: pdfPolicyId ?? '', Compress: d.Compress, SmoothStreaming: false, FieldLevelEncryptionId: '', /* ⚠️ THE ROUTER FUNCTION IS ATTACHED, AND IT IS NOT A NO-OP ON FILE PATHS. `router.js` normalises `\` to `/` and collapses a leading `//` run BEFORE it tests for an extension, and 301s when normalisation changed anything — so `//pouya-lajevardi-bio.pdf` redirects to the canonical path today. Omitting the association here would silently drop that and hand S3 the doubled key instead. The `/api/*` reason for omitting it — a 301 turning a POST into a GET and losing the body — does not apply to a GET-only PDF. */ FunctionAssociations: d.FunctionAssociations ?? { Quantity: 0 }, LambdaFunctionAssociations: { Quantity: 0 }, TrustedKeyGroups: { Enabled: false, Quantity: 0 }, }; /* ⚠️ STAGE THE REPORT EVEN WHEN THE ID IS NOT KNOWN YET. The dry run's whole job is to show what would touch a distribution serving 23 pages; reporting only the harmless policy creation and staying silent about the behaviour would mean the first sight of it is `update-distribution` writing it. The `cfg` mutation stays gated on a real id; the REPORT does not. */ changes.push( `CacheBehaviors += ${PDF_PATTERN} -> ${d.TargetOriginId}, default cache policy, ${PDF_POLICY_NAME}` + (pdfPolicyId ? ` (${pdfPolicyId})` : ' (policy id created in the same --apply pass)'), ); if (!APPLY && !pdfPolicyId) { console.log( `· would ADD cache behaviour ${PDF_PATTERN}:\n` + JSON.stringify(behaviour, null, 2) .split('\n') .map((l) => ' ' + l) .join('\n'), ); } else { pdfBehaviours.push(behaviour); cfg.CacheBehaviors = { Quantity: pdfBehaviours.length, Items: pdfBehaviours, }; } } } } console.log(''); /* Skips print under their own heading and are NOT counted as changes — see the comment on `skipped`. A skip means section 4 did nothing and the PDF is probably not noindexed; that is louder than a silent omission and quieter than a false change. */ if (skipped.length) { console.log(`⚠ ${skipped.length} thing(s) SKIPPED, not changed:`); for (const k of skipped) console.log(` ! ${k}`); console.log(' Sections 1-3 are unaffected. Investigate before relying on'); console.log(` ${PDF_PATTERN} carrying X-Robots-Tag.`); console.log(''); } if (changes.length === 0) { console.log( skipped.length ? 'NOTHING TO CHANGE — but see the skips above; the distribution does NOT carry all four.' : 'NOTHING TO CHANGE — the distribution already carries all four.', ); process.exit(0); } console.log( `${changes.length} change(s) to distribution ${DIST} (ETag ${etag}):`, ); for (const c of changes) console.log(` + ${c}`); console.log(''); if (!APPLY) { console.log('DRY RUN — nothing was sent. Re-run with --apply to write it.'); process.exit(0); } const res = aws([ 'cloudfront', 'update-distribution', '--id', DIST, '--if-match', etag, '--distribution-config', JSON.stringify(cfg), '--output', 'json', ]); console.log( `APPLIED. Status=${res.Distribution.Status} ETag=${res.ETag}\n` + 'CloudFront takes a few minutes to deploy. Wait for Deployed, then run the ' + "runbook's verification block:\n" + ` aws cloudfront wait distribution-deployed --id ${DIST}`, );