learning-website-nextjs1-11 Exercise 3: Broken Links, Links Between Sites, and Cutting Over One Site at a Time ========================================================================================================================= PART A: BROKEN LINKS. Read every page and check each absolute internal link against the pages, the folders, the files that will be copied and the decided redirects. Save as packages/migration/src/links.ts: import type { Page } from "@lw/content"; import { unquote } from "@lw/content"; import { NoSiteError, siteForPath, siteUrl, type Environment, type SiteName } from "@lw/sites"; import { bareAddress } from "./redirects.ts"; import type { Resolver } from "./verify.ts"; const ABSOLUTE = /href=(["'])(\/[^"'#?]*)/g; const FILE = /\.(?:pdf|txt)$/i; /** The site that owns an absolute path such as /linux/x/, or null (a search page, a static file, the front page). */ export function ownerOf(path: string): SiteName | null { const bare = unquote(path).replace(/^\/+|\/+$/g, ""); if (!bare) return null; try { return siteForPath(bare); } catch (error) { if (error instanceof NoSiteError) return null; throw error; } } export interface BrokenLink { readonly page: string; readonly link: string; readonly why: "no such page" | "file not found" } /** * Read every page and check each absolute internal link against the pages, the folders, the files that will be copied and the decided * redirects. A link to ANOTHER site is fine here: it is rewritten when the page is shown (rewriteCrossSiteLinks), so it is not broken. */ export function checkLinks( pages: readonly Page[], fragmentFor: (page: Page) => string, resolver: Resolver, assets: ReadonlySet, ): { checked: number; broken: BrokenLink[] } { let checked = 0; const broken: BrokenLink[] = []; for (const page of pages) { for (const match of fragmentFor(page).matchAll(ABSOLUTE)) { const link = match[2] as string; if (link.startsWith("//")) continue; checked++; const owner = ownerOf(link); if (owner === null) continue; // not content: /search/, /static/..., /login const target = unquote(link); if (FILE.test(target)) { if (!assets.has(target.replace(/^\//, ""))) broken.push({ page: page.path, link, why: "file not found" }); continue; } const bare = bareAddress(target); if (!resolver.exists(owner, bare) && !resolver.redirectFor(owner, bare)) broken.push({ page: page.path, link, why: "no such page" }); } } return { checked, broken }; } /** * A page written for the old single site links to /linux/... with a plain absolute path. After the split /linux/ is on the systems site, so * that path on the languages site is a 404. Rather than edit thousands of files (and write a host name into them), a link into ANOTHER site's * folders gets that site's address in front WHEN THE PAGE IS BUILT. The files stay unchanged. */ export function rewriteCrossSiteLinks(html: string, site: SiteName, env: Environment): { html: string; rewritten: number } { let rewritten = 0; const out = html.replace(ABSOLUTE, (whole, quote: string, path: string) => { if (path.startsWith("//")) return whole; const owner = ownerOf(path); if (owner === null || owner === site) return whole; rewritten++; return `href=${quote}${siteUrl(owner, env, path)}`; }); return { html: out, rewritten }; } Save as check-links.mjs: // Check every internal link of every page, and count the links that will need an address on another site. // LW_CONTENT_ROOT= node check-links.mjs [redirects folder] import { existsSync, readFileSync } from "node:fs"; import { join } from "node:path"; import { assetPaths, loadPages, prepareFragment, readContentFile } from "./packages/content/src/index.ts"; import { Resolver, checkLinks, rewriteCrossSiteLinks } from "./packages/migration/src/index.ts"; import { SITE_NAMES } from "./packages/sites/src/index.ts"; const root = process.env.LW_CONTENT_ROOT; const dir = process.argv[2]; const redirects = dir ? SITE_NAMES.flatMap((s) => (existsSync(join(dir, `${s}.json`)) ? JSON.parse(readFileSync(join(dir, `${s}.json`), "utf8")) : [])) : []; const { pages } = loadPages(root); const assets = new Set(assetPaths(root)); const resolver = new Resolver(pages, assets, redirects); const fragmentFor = (page) => prepareFragment(readContentFile(root, page.path), page.path, page.site); const started = performance.now(); const { checked, broken } = checkLinks(pages, fragmentFor, resolver, assets); const ms = Math.round(performance.now() - started); const byWhy = {}; for (const b of broken) byWhy[b.why] = (byWhy[b.why] ?? 0) + 1; console.log(`${checked} absolute internal links checked in ${pages.length} pages (${ms} ms); ${broken.length} broken ${JSON.stringify(byWhy)}; ${redirects.length} redirects known`); const targets = {}; for (const b of broken) { const k = b.link.split("/").slice(1, 4).join("/"); targets[k] = (targets[k] ?? 0) + 1; } for (const [k, n] of Object.entries(targets).sort((a, b) => b[1] - a[1]).slice(0, 6)) console.log(` ${String(n).padStart(4)} /${k}/...`); let across = 0; for (const page of pages) across += rewriteCrossSiteLinks(fragmentFor(page), page.site, "dev").rewritten; console.log(`links that point into ANOTHER site (rewritten when a page is built): ${across}`); 11751 absolute internal links checked in 4422 pages (1178 ms); 483 broken {"no such page":305,"file not found":178}; 0 redirects known 141 /japan/hiragana/hiragana-tiles/... 141 /japan/katakana/katakana-tiles/... 72 /web-development/scripting-and-backend/node-js/... 54 /web-development/scripting-and-backend/javascript/... 48 /web-development/scripting-and-backend/express-js/... 23 /resources/japanese/kanji/... links that point into ANOTHER site (rewritten when a page is built): 0 This is the same result the Django project reached, category by category: 483 broken links, 305 to pages and 178 to files. The 282 links to /japan/hiragana/hiragana-tiles and /japan/katakana/katakana-tiles are to tile pages that exist on the old site and not yet on the new one; the 174 solution links under /web-development/scripting-and-backend are files that were moved or never made; the 23 under /resources/japanese need a look. The check is NOT set to fail the build: 483 known problems would stop every build until each is decided. PART B: LINKS BETWEEN SITES. A page written for the old single site that links to /linux/... breaks after the split, because /linux/ is on the systems site. The fix is made WHEN THE PAGE IS BUILT, not in the files: a link into another site's folders gets that site's address in front (the function is above, and the languages page now applies it: rewriteCrossSiteLinks(fragment, "languages", environment())). Only absolute links into another site's folders change; external links, // links, anchors, /search/ and the site's own links are left alone (a test lists them). links into ANOTHER site, in all 4,422 pages: 0 So on the real content nothing needs rewriting: every page links only inside its own site. The function is therefore tested only on made-up examples, which is said here because a function that has never met real data deserves to be treated with suspicion. PART C: ONE SITE AT A TIME. While the old site still answers on osztromok.com, a site moves by telling the OLD host's Apache to send that site's folders to its new subdomain. To go back, remove the line. Save as packages/migration/src/cutover.ts: import { DOMAIN, SIDEBAR_ROUTES, SITES, SPECIAL_PREFIXES, type SiteName } from "@lw/sites"; /** * Run side by side, move one site at a time. * * While the old site still answers on osztromok.com, a site is "cut over" by telling the OLD host's Apache to send that site's folders to * the new subdomain with a permanent redirect. Nothing else changes: the other sites' folders keep being served by the old site until their * own turn. To go back, remove that site's line. */ /** The top-level paths a site owns, including the sidebar subjects and special prefixes that route to it. */ export function foldersOf(site: SiteName): string[] { return [ ...SITES[site].folders, ...Object.entries(SIDEBAR_ROUTES).filter(([, owner]) => owner === site).map(([subject]) => `sidebar/${subject}`), ...Object.entries(SPECIAL_PREFIXES).filter(([, owner]) => owner === site).map(([prefix]) => prefix), ]; } const alternatives = (site: SiteName) => foldersOf(site).map((f) => f.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")).join("|"); /** One RedirectMatch line for a site: Apache's pattern is a regular expression, with $1 and $2 for the parts. */ export function ruleFor(site: SiteName, domain: string = DOMAIN): string { return `RedirectMatch 301 ^/(${alternatives(site)})(/.*)?$ https://${site}.${domain}/$1$2`; } /** The same pattern as a JavaScript regular expression, so it can be tested without Apache. */ export function patternFor(site: SiteName): RegExp { return new RegExp(`^/(${alternatives(site)})(/.*)?$`); } Save as cutover-check.mjs: // Test the Apache cut-over lines without Apache: run each site's pattern as a regular expression over EVERY address of the old site. // node cutover-check.mjs import { patternFor, ruleFor, scanOldSite } from "./packages/migration/src/index.ts"; import { SITE_NAMES } from "./packages/sites/src/index.ts"; const old = scanOldSite(process.argv[2]).filter((u) => u.kind === "page" || u.kind === "asset"); const patterns = Object.fromEntries(SITE_NAMES.map((site) => [site, patternFor(site)])); let owned = 0, wrong = 0, unowned = 0, unownedMatched = 0; for (const u of old) { const url = "/" + u.path + (u.kind === "page" ? "/" : ""); const hits = SITE_NAMES.filter((site) => patterns[site].test(url)); if (u.site === null) { unowned++; if (hits.length) { unownedMatched++; console.log("UNOWNED BUT MATCHED", url, hits); } } else { owned++; if (hits.length !== 1 || hits[0] !== u.site) { wrong++; console.log("WRONG", url, hits, "expected", u.site); } } } console.log(`addresses a site owns: ${owned}; matched by the wrong site or by two: ${wrong}`); console.log(`addresses no site owns: ${unowned}; matched by a rule anyway: ${unownedMatched}`); console.log(ruleFor("languages")); addresses a site owns: 8892; matched by the wrong site or by two: 0 addresses no site owns: 144; matched by a rule anyway: 0 RedirectMatch 301 ^/(france|germany|hungary|japan|culture|resources/japanese|resources/hungarian)(/.*)?$ https://languages.osztromok.com/$1$2 Without Apache, the same pattern was run as a regular expression over every address of the old site: 8,892 addresses that a site owns, none matched by the wrong site or by two, and none of the 144 that no site owns matched by any rule. (The same numbers as the Django project.) A planted mistake and what caught it: removing the "must be followed by a slash or the end" part of the pattern, so that /linuxfoo/ would match the systems rule, gave 0 wrong matches on the REAL addresses (no real address looks like that) and the unit test failed. So the real addresses prove the rules cover what exists, and only a test with an awkward address proves the rules do not cover too much. Both are needed. WHAT WAS NOT DONE: the Apache lines were not run (there is no Apache here): apachectl configtest, then curl -I on one address of that site expecting a 301 and the right Location, before relying on them. The other seven sites have no app yet, so only the languages site's redirects were tried over HTTP. The tile pages and the 25 unmatched pages were not built or decided. WHY THIS WORKS AS AN ANSWER --------------------------- Every number agrees with the Django project's (a second implementation), the one function the real data never exercises is named, and the cut-over rule is tested against both the real list of addresses and a deliberately awkward one.