diff --git a/site/astro.config.mjs b/site/astro.config.mjs
index 22ca0b05..fd27ac2b 100644
--- a/site/astro.config.mjs
+++ b/site/astro.config.mjs
@@ -70,6 +70,11 @@ export default defineConfig({
// take the slack so it sits at the bottom of a short page.
PageFrame: "./src/components/PageFrame.astro",
},
+ // Starlight derives the canonical URL from the on-disk file name, which
+ // under `build.format: "file"` carries a `.html` no link, sitemap entry
+ // or agent-facing URL on this site uses. See the middleware for why the
+ // tags are rewritten rather than re-declared.
+ routeMiddleware: "./src/starlight-route-data.ts",
lastUpdated: true,
pagination: true,
favicon: "/favicon.svg",
diff --git a/site/package.json b/site/package.json
index d8a621aa..29cc74be 100644
--- a/site/package.json
+++ b/site/package.json
@@ -9,7 +9,7 @@
"scripts": {
"dev": "astro dev",
"start": "astro dev",
- "build": "astro check && astro build && node scripts/check-tables.mjs && node scripts/check-social-card.mjs",
+ "build": "astro check && astro build && node scripts/check-tables.mjs && node scripts/check-social-card.mjs && node scripts/check-canonical.mjs",
"preview": "astro preview",
"astro": "astro"
},
diff --git a/site/scripts/check-canonical.mjs b/site/scripts/check-canonical.mjs
new file mode 100644
index 00000000..35c0960b
--- /dev/null
+++ b/site/scripts/check-canonical.mjs
@@ -0,0 +1,111 @@
+// Assert that every page's canonical URL is the address the rest of the site
+// actually publishes.
+//
+// Starlight derives the canonical from the file it wrote to disk, and under
+// `build.format: "file"` that carries a `.html` nothing else here uses: the
+// sitemap lists `/start/install`, links point at `/start/install`, `llms.txt`
+// points at `/start/install.md`, and Cloudflare Pages 301s the `.html` form to
+// the extensionless one. `src/starlight-route-data.ts` corrects the tags, and
+// this asserts the correction still holds — the failure it guards against is
+// silent, produces a build that looks perfect, and is only visible in a crawler
+// weeks later.
+//
+// Three properties, because each fails on its own:
+// - the canonical carries no `.html`, so it is not naming a redirect;
+// - `og:url` says the same thing, so the page does not claim two addresses;
+// - the canonical is in the sitemap, so the two halves of the same claim
+// about which URLs exist cannot drift apart.
+import { readdir, readFile } from "node:fs/promises";
+import { join } from "node:path";
+
+const dist = new URL("../dist/", import.meta.url).pathname;
+
+async function htmlFiles(dir) {
+ const found = [];
+ for (const entry of await readdir(dir, { withFileTypes: true })) {
+ const path = join(dir, entry.name);
+ if (entry.isDirectory()) found.push(...(await htmlFiles(path)));
+ else if (entry.name.endsWith(".html")) found.push(path);
+ }
+ return found;
+}
+
+function canonicalOf(html) {
+ return html.match(/]+rel="canonical"[^>]+href="([^"]+)"/i)?.[1] ?? null;
+}
+
+function ogUrlOf(html) {
+ return html.match(/]+property="og:url"[^>]+content="([^"]+)"/i)?.[1] ?? null;
+}
+
+// `https://fanout.run` and `https://fanout.run/` are the same address, and the
+// sitemap writes the first while the canonical writes the second. Comparing
+// them literally would fail the site's own landing page.
+function key(url) {
+ return url.replace(/\/$/, "");
+}
+
+let pages = [];
+try {
+ pages = await htmlFiles(dist);
+} catch (error) {
+ console.error(`check-canonical: cannot read ${dist} — run \`npm run build\` first`);
+ console.error(String(error));
+ process.exit(1);
+}
+
+if (pages.length === 0) {
+ console.error("check-canonical: the build produced no HTML, which cannot be right");
+ process.exit(1);
+}
+
+// The 404 is served for addresses that do not exist, so it is the one page the
+// sitemap must not list and the one whose canonical proves nothing.
+const NOT_FOUND = join(dist, "404.html");
+
+const failures = [];
+const canonicals = new Map();
+for (const page of pages) {
+ const name = page.replace(dist, "");
+ const html = await readFile(page, "utf8");
+ const canonical = canonicalOf(html);
+ const ogUrl = ogUrlOf(html);
+
+ if (canonical === null) {
+ failures.push(`${name}: no `);
+ continue;
+ }
+ if (canonical.endsWith(".html")) {
+ failures.push(`${name}: canonical is ${canonical}, which redirects to the extensionless path`);
+ }
+ if (ogUrl !== canonical) {
+ failures.push(`${name}: og:url is ${ogUrl ?? "absent"} but the canonical is ${canonical}`);
+ }
+ if (page !== NOT_FOUND) canonicals.set(key(canonical), name);
+}
+
+let sitemap;
+try {
+ sitemap = await readFile(join(dist, "sitemap-0.xml"), "utf8");
+} catch {
+ console.error("check-canonical: dist/sitemap-0.xml is missing, so nothing tells a crawler these pages exist");
+ process.exit(1);
+}
+const listed = new Set(
+ [...sitemap.matchAll(/([^<]+)<\/loc>/g)].map((match) => key(match[1])),
+);
+
+for (const [canonical, name] of canonicals) {
+ if (!listed.has(canonical)) {
+ failures.push(`${name}: canonical ${canonical} is not in the sitemap`);
+ }
+}
+
+if (failures.length > 0) {
+ console.error("check-canonical: pages whose canonical URL is not the one the site publishes:");
+ for (const failure of failures) console.error(` ${failure}`);
+ console.error("check-canonical: see src/starlight-route-data.ts, which rewrites these tags");
+ process.exit(1);
+}
+
+console.log(`check-canonical: ${pages.length} page(s), canonical matches og:url and the sitemap`);
diff --git a/site/src/starlight-route-data.ts b/site/src/starlight-route-data.ts
new file mode 100644
index 00000000..66f366b5
--- /dev/null
+++ b/site/src/starlight-route-data.ts
@@ -0,0 +1,52 @@
+import { defineRouteMiddleware } from "@astrojs/starlight/route-data";
+
+// Strips the `.html` Astro puts in the canonical URL.
+//
+// Starlight builds the canonical from `Astro.url.pathname`, and under
+// `build.format: "file"` that pathname is the file it wrote to disk, so the tag
+// read `https://fanout.run/start/install.html`. Nothing else on the site uses
+// that form: the sitemap lists `/start/install`, every internal link points at
+// `/start/install`, `llms.txt` points at `/start/install.md`, and Cloudflare
+// Pages — where this site is published — serves the extensionless path and 301s
+// the `.html` one to it. The single tag whose job is to state the page's real
+// address was naming a redirect, on every page.
+//
+// Starlight's own `formatCanonical` returns the href untouched when the format
+// is "file". That is right for a server that only serves `foo.html`, and wrong
+// for this one, which is why the correction lives here rather than in a
+// configuration flag.
+//
+// The tags are rewritten in place rather than appended to, because two
+// `` elements with different hrefs are worse than one
+// wrong href: a crawler that sees the pair discards both and picks a canonical
+// on its own.
+function withoutHtmlExtension(href: string): string {
+ const url = new URL(href);
+ if (url.pathname === "/index.html") {
+ url.pathname = "/";
+ } else if (url.pathname.endsWith(".html")) {
+ url.pathname = url.pathname.slice(0, -".html".length);
+ }
+ return url.href;
+}
+
+export const onRequest = defineRouteMiddleware((context) => {
+ for (const tag of context.locals.starlightRoute.head) {
+ // `og:url` is generated from the same string as the canonical, so a fix
+ // that touched only the `` would leave the two disagreeing about
+ // which URL the page is.
+ if (
+ tag.tag === "link" &&
+ tag.attrs?.rel === "canonical" &&
+ typeof tag.attrs.href === "string"
+ ) {
+ tag.attrs.href = withoutHtmlExtension(tag.attrs.href);
+ } else if (
+ tag.tag === "meta" &&
+ tag.attrs?.property === "og:url" &&
+ typeof tag.attrs.content === "string"
+ ) {
+ tag.attrs.content = withoutHtmlExtension(tag.attrs.content);
+ }
+ }
+});