learning-website-nextjs1-8 Exercise 1: PDFs and Solutions: Copy Them In, Serve Them Safely ===================================================================================================== Every course folder may hold a pdfs/ folder and a solutions/ folder, and a page links to them at the SAME relative path: /hungary/hungarian-basic-3/solutions/ex1.txt. Those addresses must keep working. Next serves everything in an app's public/ folder as it is, at the same path, so a site's build COPIES its own PDFs and solutions there. Save as packages/content/src/assets.ts: import { copyFileSync, existsSync, mkdirSync, readFileSync, readdirSync, rmSync, statSync, utimesSync, writeFileSync, } from "node:fs"; import { dirname, join } from "node:path"; import { NoSiteError, siteForPath, type SiteName } from "@lw/sites"; /** * PDFs and solution files. * * Every course folder may hold a pdfs/ folder and a solutions/ folder. A page links to them at the SAME relative path, so * /hungary/hungarian-basic-3/solutions/ex1.txt must keep working after the split. A site's build copies its own files into the * app's public/ folder at the same relative path, and Next serves public/ files as they are. */ export const ASSET_FOLDERS = ["pdfs", "solutions"] as const; const ASSET_EXTENSIONS = [".pdf", ".txt"]; /** * Where the list of files a build copied is kept, so that pruning only ever removes files this tool made. It sits NEXT TO the * target folder, never inside it: everything inside public/ is served to visitors, and a bookkeeping file is not for them. */ export function manifestPathFor(targetDir: string): string { return `${targetDir.replace(/[\\/]+$/, "")}.manifest.json`; } /** Relative paths (forward slashes) of every PDF and solution file under the content folder. */ export function* assetPaths(root: string, relDir = ""): Generator { const entries = readdirSync(join(root, relDir), { withFileTypes: true }).sort((a, b) => (a.name < b.name ? -1 : 1)); const inAssetFolder = (ASSET_FOLDERS as readonly string[]).includes(relDir.split("/").at(-1) ?? ""); for (const entry of entries) { const rel = relDir ? `${relDir}/${entry.name}` : entry.name; if (entry.isDirectory()) yield* assetPaths(root, rel); else if (inAssetFolder && ASSET_EXTENSIONS.some((ext) => entry.name.toLowerCase().endsWith(ext))) yield rel; } } export interface CollectResult { copied: number; unchanged: number; removed: number; bytesCopied: number; /** Files that belong to no site, with the reason: never copied. */ skipped: { path: string; reason: string }[]; } export interface CollectOptions { readonly dryRun?: boolean; /** Remove files that a PREVIOUS run copied and that no longer exist in the content. Files made by anything else are never touched. */ readonly prune?: boolean; } /** Copy one site's PDFs and solutions into targetDir. A file is copied only when it is missing or its size or time differs. */ export function collectAssets(contentRoot: string, targetDir: string, site: SiteName, options: CollectOptions = {}): CollectResult { const result: CollectResult = { copied: 0, unchanged: 0, removed: 0, bytesCopied: 0, skipped: [] }; const wanted: string[] = []; for (const rel of assetPaths(contentRoot)) { try { if (siteForPath(rel) !== site) continue; } catch (error) { if (!(error instanceof NoSiteError)) throw error; result.skipped.push({ path: rel, reason: "belongs to no site" }); continue; } wanted.push(rel); const source = join(contentRoot, rel); const target = join(targetDir, rel); const from = statSync(source); if (existsSync(target)) { const to = statSync(target); if (from.size === to.size && Math.floor(from.mtimeMs / 1000) <= Math.floor(to.mtimeMs / 1000)) { result.unchanged++; continue; } } result.copied++; result.bytesCopied += from.size; if (!options.dryRun) { mkdirSync(dirname(target), { recursive: true }); copyFileSync(source, target); utimesSync(target, from.atime, from.mtime); // keep the modified time, so the next run sees "same" } } const manifestPath = manifestPathFor(targetDir); if (options.prune && existsSync(manifestPath)) { const previous: string[] = (JSON.parse(readFileSync(manifestPath, "utf8")) as { files: string[] }).files; const keep = new Set(wanted); for (const rel of previous) { if (keep.has(rel)) continue; result.removed++; if (!options.dryRun) rmSync(join(targetDir, rel), { force: true }); } } if (!options.dryRun) { writeFileSync(manifestPath, JSON.stringify({ files: wanted }, null, 1)); } return result; } export interface AssetLink { readonly name: string; readonly href: string } export type AssetLists = Record<(typeof ASSET_FOLDERS)[number], AssetLink[]>; /** The PDFs and solution files that sit directly in /pdfs and /solutions. */ export function listAssets(targetDir: string, folder: string): AssetLists { const out: AssetLists = { pdfs: [], solutions: [] }; for (const kind of ASSET_FOLDERS) { const directory = join(targetDir, ...folder.split("/"), kind); if (!existsSync(directory)) continue; out[kind] = readdirSync(directory) .filter((name) => ASSET_EXTENSIONS.some((ext) => name.toLowerCase().endsWith(ext))) .sort((a, b) => (a < b ? -1 : 1)) .map((name) => ({ name, href: `/${folder}/${kind}/${name}` })); } return out; } Save as collect-assets.mjs: // Copy one site's PDFs and solution files into that app's public/ folder, at the same relative paths. // node collect-assets.mjs [--dry-run] [--prune] (LW_CONTENT_ROOT names the content folder) import { collectAssets } from "./packages/content/src/index.ts"; import { SITE_NAMES } from "./packages/sites/src/index.ts"; const [site, ...flags] = process.argv.slice(2); const root = process.env.LW_CONTENT_ROOT; if (!root) throw new Error("Set LW_CONTENT_ROOT to the content folder."); if (!SITE_NAMES.includes(site)) throw new Error(`usage: collect-assets.mjs <${SITE_NAMES.join("|")}> [--dry-run] [--prune]`); const started = performance.now(); const result = collectAssets(root, new URL(`./apps/${site}/public`, import.meta.url).pathname.replace(/^\/([A-Za-z]:)/, "$1"), site, { dryRun: flags.includes("--dry-run"), prune: flags.includes("--prune"), }); const seconds = ((performance.now() - started) / 1000).toFixed(1); console.log(`${site}: copied ${result.copied} (${(result.bytesCopied / 1048576).toFixed(0)} MB), unchanged ${result.unchanged}, removed ${result.removed}, skipped ${result.skipped.length} (${seconds}s)${flags.includes("--dry-run") ? " [dry run]" : ""}`); for (const s of result.skipped.slice(0, 5)) console.log(` skipped: ${s.path} (${s.reason})`); What it does, and why: - A site gets only its own files. The site is decided from the path (the same function as for pages), and a file that belongs to no site is reported and never copied. - A file is copied only when it is missing or its size or time differs, and the copy keeps the original's modified time, so a second run copies nothing. - --dry-run shows what would happen. --prune removes files a PREVIOUS run copied and that no longer exist in the content. - PDFs are build artefacts: they are produced by the Python builders (scripts/python) and are ignored by git (*.pdf). This build does not make them; it copies the ones that exist. On a fresh checkout there are none until the builders have run. Save as packages/content/src/assets.test.ts: import assert from "node:assert/strict"; import { existsSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, rmSync, statSync, utimesSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import { dirname, join } from "node:path"; import { test } from "node:test"; import { assetPaths, collectAssets, listAssets, manifestPathFor } from "./index.ts"; function tree(files: Record): string { const root = mkdtempSync(join(tmpdir(), "lw-assets-")); for (const [rel, data] of Object.entries(files)) { const full = join(root, rel); mkdirSync(dirname(full), { recursive: true }); writeFileSync(full, data); } return root; } const FILES = { "hungary/c/pdfs/Course.pdf": "pdf", "hungary/c/solutions/c_1_1_ex1.txt": "one", "hungary/c/solutions/c_1_1_ex2.TXT": "two", "hungary/c/solutions/notes.md": "no", "hungary/c/page_1_1.html": "

", "hungary/c/other/ex.txt": "not in a solutions folder", "linux/x/pdfs/Linux.pdf": "pdf", "nonsense/pdfs/A.pdf": "pdf", "france/f/pdfs/F.pdf": "pdf", }; test("only PDF and text files inside a pdfs or solutions folder are assets", () => { const root = tree(FILES); try { assert.deepEqual([...assetPaths(root)].sort(), [ "france/f/pdfs/F.pdf", "hungary/c/pdfs/Course.pdf", "hungary/c/solutions/c_1_1_ex1.txt", "hungary/c/solutions/c_1_1_ex2.TXT", "linux/x/pdfs/Linux.pdf", "nonsense/pdfs/A.pdf", ]); } finally { rmSync(root, { recursive: true }); } }); test("a site only gets its own files, and files of no site are reported, not copied", () => { const root = tree(FILES); const target = mkdtempSync(join(tmpdir(), "lw-public-")); try { const result = collectAssets(root, target, "languages"); assert.deepEqual([result.copied, result.unchanged, result.removed], [4, 0, 0]); assert.deepEqual(result.skipped.map((s) => s.path), ["nonsense/pdfs/A.pdf"]); assert.ok(existsSync(join(target, "hungary/c/pdfs/Course.pdf"))); assert.ok(!existsSync(join(target, "linux/x/pdfs/Linux.pdf"))); // another site's file assert.equal(readFileSync(join(target, "hungary/c/solutions/c_1_1_ex1.txt"), "utf8"), "one"); } finally { rmSync(root, { recursive: true }); rmSync(target, { recursive: true }); rmSync(manifestPathFor(target), { force: true }); } }); test("a second run copies nothing, and a changed file is copied again", () => { const root = tree(FILES); const target = mkdtempSync(join(tmpdir(), "lw-public-")); try { collectAssets(root, target, "languages"); const again = collectAssets(root, target, "languages"); assert.deepEqual([again.copied, again.unchanged], [0, 4]); const changed = join(root, "hungary/c/solutions/c_1_1_ex1.txt"); writeFileSync(changed, "longer text now"); utimesSync(changed, new Date(), new Date(Date.now() + 5000)); const third = collectAssets(root, target, "languages"); assert.deepEqual([third.copied, third.unchanged], [1, 3]); assert.equal(readFileSync(join(target, "hungary/c/solutions/c_1_1_ex1.txt"), "utf8"), "longer text now"); assert.equal(Math.floor(statSync(join(target, "hungary/c/pdfs/Course.pdf")).mtimeMs / 1000), Math.floor(statSync(join(root, "hungary/c/pdfs/Course.pdf")).mtimeMs / 1000)); // the modified time is kept } finally { rmSync(root, { recursive: true }); rmSync(target, { recursive: true }); rmSync(manifestPathFor(target), { force: true }); } }); test("a dry run changes nothing", () => { const root = tree(FILES); const target = mkdtempSync(join(tmpdir(), "lw-public-")); try { const result = collectAssets(root, target, "languages", { dryRun: true }); assert.equal(result.copied, 4); assert.ok(!existsSync(join(target, "hungary")) && !existsSync(manifestPathFor(target))); } finally { rmSync(root, { recursive: true }); rmSync(target, { recursive: true }); rmSync(manifestPathFor(target), { force: true }); } }); test("pruning removes only what an earlier run copied, never a file somebody else put there", () => { const root = tree(FILES); const target = mkdtempSync(join(tmpdir(), "lw-public-")); try { writeFileSync(join(target, "favicon.ico"), "mine"); mkdirSync(join(target, "hungary/c/solutions"), { recursive: true }); writeFileSync(join(target, "hungary/c/solutions/handmade.txt"), "mine too"); collectAssets(root, target, "languages"); rmSync(join(root, "hungary/c/solutions/c_1_1_ex2.TXT")); const result = collectAssets(root, target, "languages", { prune: true }); assert.equal(result.removed, 1); assert.ok(!existsSync(join(target, "hungary/c/solutions/c_1_1_ex2.TXT"))); assert.ok(existsSync(join(target, "favicon.ico")) && existsSync(join(target, "hungary/c/solutions/handmade.txt"))); } finally { rmSync(root, { recursive: true }); rmSync(target, { recursive: true }); rmSync(manifestPathFor(target), { force: true }); } }); test("listing a folder's downloads", () => { const root = tree(FILES); const target = mkdtempSync(join(tmpdir(), "lw-public-")); try { collectAssets(root, target, "languages"); const lists = listAssets(target, "hungary/c"); assert.deepEqual(lists.pdfs, [{ name: "Course.pdf", href: "/hungary/c/pdfs/Course.pdf" }]); assert.deepEqual(lists.solutions.map((a) => a.name), ["c_1_1_ex1.txt", "c_1_1_ex2.TXT"]); assert.deepEqual(listAssets(target, "hungary/missing"), { pdfs: [], solutions: [] }); } finally { rmSync(root, { recursive: true }); rmSync(target, { recursive: true }); rmSync(manifestPathFor(target), { force: true }); } }); test("the bookkeeping file is next to the folder, never inside it where visitors could fetch it", () => { const root = tree(FILES); const target = mkdtempSync(join(tmpdir(), "lw-public-")); try { collectAssets(root, target, "languages"); assert.ok(existsSync(manifestPathFor(target))); assert.ok(!manifestPathFor(target).startsWith(target + "/") && !manifestPathFor(target).startsWith(target + "\\")); assert.deepEqual(readdirSync(target).sort(), ["france", "hungary"]); // nothing but content in public } finally { rmSync(root, { recursive: true }); rmSync(target, { recursive: true }); rmSync(manifestPathFor(target), { force: true }); } }); npm test ℹ tests 63 ℹ pass 63 ℹ fail 0 (56 earlier and 7 new) Run on the real content: node collect-assets.mjs languages --dry-run languages: copied 236 (186 MiB), unchanged 0, removed 0, skipped 0 (0.4s) [dry run] node collect-assets.mjs languages copied 236 (186 MiB) (1.6 s) node collect-assets.mjs languages copied 0, unchanged 236 (0.3 s) node collect-assets.mjs languages --prune copied 0, unchanged 236, removed 0 (0.4 s) every site, dry run: languages 236 files, webdevelopment 1,016 (70 MiB), programming 3,658 (203 MiB), systems 1,419 (86 MiB), ai 317 (25 MiB), humanities 889 (36 MiB), lifeskills 412 (19 MiB), creative 484 (12 MiB) = 8,431 files, none skipped. (These are copies: about 640 MiB in all across the eight apps.) THREE THINGS THAT WENT WRONG, in the order I met them: 1. A bookkeeping file in public/. The first version kept its list of copied files (needed for --prune) INSIDE the target folder, and public/ is served to visitors. A request for it was answered, and with an error (500), which is its own bad sign. The list is now kept NEXT TO the folder (apps/languages/public.manifest.json), and a test checks that nothing but content is in public. Takeaway: whatever a folder serves, nothing in it should be for you only. 2. A wrong comment. I wrote in the configuration that "Next sends no validators for files in public/, so every visit would download the whole file again", and added a one-hour cache rule. I had filtered the header list too tightly to see them. Next DOES send ETag and Last-Modified (a second request with the ETag gives 304 Not Modified), and its default Cache-Control: public, max-age=0 means "ask again each time, cheaply". The rule and the comment were removed. Takeaway: look at ALL the headers before explaining them, and measure a conditional request before claiming caching is missing. 3. A test that assumed a language. The first version of the "only content is in public" test expected one folder and found two, because the test's own sample files included a French one, and French is a languages-site language. The expectation was corrected; the lesson is to derive what a test expects from the same rule the code uses. Serving, checked with curl on the running app (a solutions file from the web development app, since language courses have none): .txt 200 Content-Type: text/plain; charset=UTF-8 ETag, Last-Modified, Accept-Ranges: bytes .pdf 200 Content-Type: application/pdf Accept-Ranges: bytes; a range request (bytes 0-99) gives 206 Partial Content a missing file under a solutions folder 404 /solutions/../../../package.json and the %2e%2e form 404 and 404 a page next to its folders is still a page 200 One header is added, for the two folders (next.config.ts, in the languages and web development apps): async headers() { const download = [{ key: "X-Content-Type-Options", value: "nosniff" }]; return [{ source: "/:path*/pdfs/:file", headers: download }, { source: "/:path*/solutions/:file", headers: download }]; }, nosniff tells the browser never to guess a file's type. To see why it matters, a file named evil.txt containing "

hi

" was put in a solutions folder (and removed afterwards). It is served as text/plain with nosniff, and Chrome showed it as TEXT inside a
 block, with the tags escaped: nothing ran.

What was NOT done: Apache is not in front (in production Apache can serve these folders directly, Chapter 12), so none of this
was checked behind a proxy; the PDFs themselves were not rebuilt.

WHY THIS WORKS AS AN ANSWER
---------------------------
It keeps the old addresses, copies only what changed, serves with the right types, and records three mistakes with the way to
avoid each.