learning-website-nextjs1-3 Exercise 2: From a File to a Page, and a Whole Folder ================================================================================ Save as packages/content/src/parse.ts: import { siteForPath } from "@lw/sites"; import { parseBanner } from "./banner.ts"; import type { Banner, IsoDate, Kind, Page } from "./types.ts"; export class ContentParseError extends Error { constructor(message: string) { super(message); this.name = "ContentParseError"; } } // Chapter file names come in several generations (the same list as the Django project): // hungarian_basic_conversation_3_1.html course 3, chapter 1 // html_lesson_01_3.html chapter 1 of course 3: the numbers are the other way round // js1-4.html, psp-croll-1-2.html course 1, chapter 4 / 2: the older prompt-prefix names // setting_up_a_web_server_on_debian_03 chapter 3 only // linux_appendix_a.html no number at all const LESSON_NUMBERS = /_lesson_(\d+)_(\d+)\.html$/; const NEW_NUMBERS = /_(\d+)_(\d+)\.html$/; const OLD_NUMBERS = /[A-Za-z_-](\d+)-(\d+)\.html$/; const ONE_NUMBER = /_(\d+)\.html$/; /** [course number, chapter number] read from the file name; either may be null. */ export function numbersFromPath(path: string): [number | null, number | null] { const lesson = LESSON_NUMBERS.exec(path); if (lesson) return [Number(lesson[2]), Number(lesson[1])]; for (const pattern of [NEW_NUMBERS, OLD_NUMBERS]) { const match = pattern.exec(path); if (match) return [Number(match[1]), Number(match[2])]; } const one = ONE_NUMBER.exec(path); return one ? [null, Number(one[1])] : [null, null]; } const NAMED: Record = { amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", nbsp: " ", copy: "©", reg: "®", middot: "·", mdash: "—", ndash: "–", hellip: "…", lsquo: "‘", rsquo: "’", ldquo: "“", rdquo: "”", rarr: "→", larr: "←", times: "×", bull: "•", eacute: "é", egrave: "è", aacute: "á", oacute: "ó", uuml: "ü", ouml: "ö", auml: "ä", szlig: "ß", }; /** The parts of HTML character decoding that titles need: numeric references and the common named ones. */ export function decodeEntities(text: string): string { return text.replace(/&(#x[0-9a-f]+|#\d+|[a-z][a-z0-9]*);/gi, (whole, name: string) => { if (name.startsWith("#")) { const code = name[1] === "x" || name[1] === "X" ? parseInt(name.slice(2), 16) : parseInt(name.slice(1), 10); return Number.isFinite(code) && code > 0 && code <= 0x10ffff ? String.fromCodePoint(code) : whole; } return NAMED[name] ?? NAMED[name.toLowerCase()] ?? whole; }); } const HEADING = /]*>([\s\S]*?)<\/h[12]>/i; const TAGS = /<[^>]+>/g; /** "hungarian_lesson_vdd_to_be.html" becomes "Hungarian Lesson Vdd To Be". */ export function humanize(filename: string): string { const stem = filename.includes(".") ? filename.slice(0, filename.lastIndexOf(".")) : filename; return stem .split(/[-_]+/) .filter((word) => word !== "") .map((word) => word.charAt(0).toUpperCase() + word.slice(1).toLowerCase()) .join(" "); } /** The same order the live site uses: banner Chapter, , first h1 or h2, then the file name. */ export function bestTitle(raw: string, banner: Banner, filename: string): string { if (banner["chapter"]) return decodeEntities(banner["chapter"]); if (banner.title) return decodeEntities(banner.title); const heading = HEADING.exec(raw); if (heading) { const text = decodeEntities((heading[1] ?? "").replace(TAGS, "")).trim(); if (text) return text; } return humanize(filename); } const KINDS: Record<Banner["kind"], Kind> = { "course-chapter": "course_chapter", sidebar: "sidebar", "free-banner": "lesson", "full-page": "full_page", "no-banner": "other", }; /** A real calendar date in ISO form, or an error: "2026-13-45" is not a date. */ export function checkedDate(text: string | undefined, path: string): IsoDate | null { if (text === undefined) return null; const parsed = new Date(`${text}T00:00:00Z`); if (Number.isNaN(parsed.getTime()) || parsed.toISOString().slice(0, 10) !== text) { throw new ContentParseError(`${path}: "${text}" is not a real date`); } return text; } /** At most `limit` characters, counting what a person sees as a character (not UTF-16 halves). */ function clip(text: string, limit: number): string { const characters = Array.from(text); return characters.length <= limit ? text : characters.slice(0, limit).join(""); } /** Turn one content file into a Page. Throws NoSiteError for a path no site owns, ContentParseError for a bad date. */ export function parsePage(raw: string, path: string): Page { const banner = parseBanner(raw.slice(0, 6000)); const [courseNo, chapterNo] = banner.kind === "course-chapter" ? numbersFromPath(path) : [null, null]; const filename = path.slice(path.lastIndexOf("/") + 1); return { path, site: siteForPath(path), kind: KINDS[banner.kind], title: clip(bestTitle(raw, banner, filename), 300), summary: clip(banner["topic"] ?? "", 300), courseName: banner["course"] ?? null, courseNo, chapterNo, created: checkedDate(banner.created, path), updated: checkedDate(banner.updated, path), }; } Save as packages/content/src/scan.ts: import { readFileSync, readdirSync, statSync } from "node:fs"; import { join } from "node:path"; import { NoSiteError } from "@lw/sites"; import { ContentParseError, parsePage } from "./parse.ts"; import type { ContentError, Course, Page } from "./types.ts"; const MAX_BYTES = 2_000_000; // the largest real page is 0.77 MB, so 2 MB is generous const SKIP_DIRS = new Set(["pdfs", "solutions"]); const EXCLUDED_PREFIXES = ["japan/japanese-language/reference-materials/kanji"]; // the archive copy, not routed /** Relative paths (forward slashes) of every page file in the content folder, in a stable order. */ export function* contentPaths(root: string, relDir = ""): Generator<string> { if (EXCLUDED_PREFIXES.some((prefix) => relDir === prefix || relDir.startsWith(prefix + "/"))) return; const entries = readdirSync(join(root, relDir), { withFileTypes: true }).sort((a, b) => (a.name < b.name ? -1 : 1)); for (const entry of entries) { const rel = relDir ? `${relDir}/${entry.name}` : entry.name; if (entry.isDirectory()) { if (!SKIP_DIRS.has(entry.name)) yield* contentPaths(root, rel); } else if (entry.name.endsWith(".html") && !entry.name.endsWith("_print.html") && !entry.name.startsWith("_")) { yield rel; } } } /** Read one file as strict UTF-8: a file that is not UTF-8 is an error, not a guess. */ export function readContentFile(root: string, rel: string): string { if (rel.startsWith("/") || rel.includes("\\") || rel.split("/").includes("..")) { throw new ContentParseError(`${rel}: not a safe relative path`); } const full = join(root, rel); if (statSync(full).size > MAX_BYTES) throw new ContentParseError(`${rel}: larger than ${MAX_BYTES} bytes`); try { return new TextDecoder("utf-8", { fatal: true }).decode(readFileSync(full)); } catch { throw new ContentParseError(`${rel}: not valid UTF-8`); } } export interface LoadResult { readonly pages: readonly Page[]; readonly errors: readonly ContentError[]; } /** Parse every page of a content folder. A bad file is reported in `errors`; it never stops the others. */ export function loadPages(root: string): LoadResult { const pages: Page[] = []; const errors: ContentError[] = []; for (const rel of contentPaths(root)) { try { pages.push(parsePage(readContentFile(root, rel), rel)); } catch (error) { if (error instanceof NoSiteError || error instanceof ContentParseError) { errors.push({ path: rel, message: `${error.name}: ${error.message}` }); } else { throw error; // a real bug must not be filed away as "a bad file" } } } return { pages, errors }; } /** Group course chapters by folder. A chapter's own banner names its course; the first one seen is used. */ export function groupCourses(pages: readonly Page[]): Course[] { const byFolder = new Map<string, { page: Page; chapters: Page[] }>(); for (const page of pages) { if (page.kind !== "course_chapter") continue; const folder = page.path.slice(0, page.path.lastIndexOf("/")); const entry = byFolder.get(folder); if (entry) entry.chapters.push(page); else byFolder.set(folder, { page, chapters: [page] }); } return [...byFolder.entries()] .sort(([a], [b]) => (a < b ? -1 : 1)) .map(([folder, { page, chapters }]) => ({ folder, site: page.site, name: page.courseName ?? folder, courseNo: page.courseNo, // chapter 2 before chapter 10; a chapter without a number goes last, then by path chapters: [...chapters].sort((a, b) => (a.chapterNo ?? 1e9) - (b.chapterNo ?? 1e9) || (a.path < b.path ? -1 : 1)), })); } Save as packages/content/src/index.ts: export { parseBanner } from "./banner.ts"; export { ContentParseError, bestTitle, checkedDate, decodeEntities, humanize, numbersFromPath, parsePage, } from "./parse.ts"; export { contentPaths, groupCourses, loadPages, readContentFile } from "./scan.ts"; export type { LoadResult } from "./scan.ts"; export type { Banner, ContentError, Course, IsoDate, Kind, Page } from "./types.ts"; Points worth knowing: - Titles follow the order the live site uses: banner Chapter, then <title>, then the first h1 or h2, then the file name made readable ("hungarian_lesson_vdd_to_be.html" becomes "Hungarian Lesson Vdd To Be"). - Course and chapter numbers come from the FILE NAME, which has come in five generations (the comments in parse.ts list them); html_lesson_01_3 means chapter 1 of course 3, the other way round from the rest. - A date that is not a real date ("2026-13-01", "2025-02-29") is an error, not a null. - Character references in titles (&, é, é) are decoded once only: "&lt;" becomes "<", not "<". Only the common named references are decoded (a small table); an unknown one is left as written. - A file that is not valid UTF-8, is over 2 MB, or has a path that climbs out of the folder is an error. - loadPages never stops for one bad file: it returns {pages, errors}. But it re-throws anything that is NOT a content problem (a missing folder is ENOENT, not "a bad file"), so a real bug cannot hide in the error list. - Pages ending in _print.html, files starting with _, the pdfs and solutions folders and the archive copy of the kanji pages are not pages. Save as packages/content/src/content.test.ts: import assert from "node:assert/strict"; import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { test } from "node:test"; import { NoSiteError } from "@lw/sites"; import { ContentParseError, bestTitle, checkedDate, contentPaths, decodeEntities, groupCourses, humanize, loadPages, numbersFromPath, parseBanner, parsePage, readContentFile, } from "./index.ts"; const CHAPTER = `<!-- ============================================================ Course: German Basic Conversation 3 Chapter: At the Station & Beyond File: german_basic_conversation_3_1.html Topic: Buying tickets and asking about trains Date Created: 2026-10-01 Date Updated: 2026-10-05 ============================================================ --> <style>.x{}</style><div>body</div>`; const SIDEBAR = `<!-- ============================================================ Category: Sidebar Subcategory: Linux Chapter: Vim Cheat Sheet File: cheat_sheet_vim.html Date Created: 2026-08-07 Date Updated: 2026-08-07 ============================================================ -->`; test("a course banner is read, with its decorations and dates", () => { const banner = parseBanner(CHAPTER); assert.equal(banner.kind, "course-chapter"); assert.equal(banner["course"], "German Basic Conversation 3"); assert.equal(banner["chapter"], "At the Station & Beyond"); assert.equal(banner["topic"], "Buying tickets and asking about trains"); assert.deepEqual([banner.created, banner.updated], ["2026-10-01", "2026-10-05"]); }); test("the other banner kinds", () => { assert.equal(parseBanner(SIDEBAR).kind, "sidebar"); assert.equal(parseBanner("<!-- HUNGARIAN LESSON: Food -->\n<div>").kind, "free-banner"); assert.equal(parseBanner("<!-- HUNGARIAN LESSON: Food -->")["headline"], "HUNGARIAN LESSON: Food"); assert.deepEqual(parseBanner("<div>no banner at all</div>"), { kind: "no-banner" }); const page = parseBanner(" <!DOCTYPE html><html><head><title> Kanji: Water "); assert.deepEqual([page.kind, page.title], ["full-page", "Kanji: Water"]); }); test("older banners have no dates, and the first value of a repeated key wins", () => { const banner = parseBanner(""); assert.equal(banner.created, undefined); assert.equal(banner["course"], "A"); }); test("a banner that is not at the very top is not a banner", () => { assert.equal(parseBanner("

x

").kind, "no-banner"); }); test("file names give course and chapter numbers in every generation", () => { assert.deepEqual(numbersFromPath("hungary/b/hungarian_basic_conversation_3_1.html"), [3, 1]); assert.deepEqual(numbersFromPath("x/html_lesson_01_3.html"), [3, 1]); // the numbers are the other way round assert.deepEqual(numbersFromPath("x/js1-4.html"), [1, 4]); assert.deepEqual(numbersFromPath("x/psp-croll-1-2.html"), [1, 2]); assert.deepEqual(numbersFromPath("x/setting_up_a_web_server_on_debian_03.html"), [null, 3]); assert.deepEqual(numbersFromPath("x/linux_appendix_a.html"), [null, null]); }); test("character references are decoded once, not twice", () => { assert.equal(decodeEntities("A & B é A é …"), "A & B é A é …"); assert.equal(decodeEntities("&lt;"), "<"); // not "<" assert.equal(decodeEntities("&nosuchthing; �"), "&nosuchthing; �"); }); test("a title comes from the banner, then , then a heading, then the file name", () => { const banner = parseBanner(CHAPTER); assert.equal(bestTitle(CHAPTER, banner, "x.html"), "At the Station & Beyond"); assert.equal(bestTitle("", { kind: "full-page", title: "T & U" }, "x.html"), "T & U"); assert.equal(bestTitle("<h2 class='a'>The <b>Heading</b></h2>", { kind: "no-banner" }, "x.html"), "The Heading"); assert.equal(bestTitle("<h1> </h1><p>x</p>", { kind: "no-banner" }, "my-file_name.html"), "My File Name"); }); test("humanize capitalises the first letter and lower-cases the rest", () => { assert.equal(humanize("hungarian_lesson_vdd_to_be.html"), "Hungarian Lesson Vdd To Be"); assert.equal(humanize("HTML-basics"), "Html Basics"); }); test("only real calendar dates are accepted", () => { assert.equal(checkedDate("2024-02-29", "p"), "2024-02-29"); assert.equal(checkedDate(undefined, "p"), null); for (const bad of ["2026-13-01", "2026-02-30", "2025-02-29", "2026-00-10"]) { assert.throws(() => checkedDate(bad, "p"), ContentParseError, bad); } }); test("a whole page is described", () => { const page = parsePage(CHAPTER, "germany/german-basic-3/german_basic_conversation_3_1.html"); assert.deepEqual(page, { path: "germany/german-basic-3/german_basic_conversation_3_1.html", site: "languages", kind: "course_chapter", title: "At the Station & Beyond", summary: "Buying tickets and asking about trains", courseName: "German Basic Conversation 3", courseNo: 3, chapterNo: 1, created: "2026-10-01", updated: "2026-10-05", }); const sidebar = parsePage(SIDEBAR, "sidebar/linux/cheat_sheet_vim.html"); assert.deepEqual([sidebar.site, sidebar.kind, sidebar.chapterNo, sidebar.courseName], ["systems", "sidebar", null, null]); }); test("a path no site owns is an error, never a guess", () => { assert.throws(() => parsePage(CHAPTER, "nonsense/x.html"), NoSiteError); }); test("a long title is cut at 300 characters, counting emoji as one", () => { const raw = `<!-- Course: C\n Chapter: ${"\u{1F600}".repeat(400)} -->`; const page = parsePage(raw, "france/x/a_1_1.html"); assert.equal(Array.from(page.title).length, 300); }); function tree(files: Record<string, string | Uint8Array>): string { const root = mkdtempSync(join(tmpdir(), "lw-content-")); for (const [rel, data] of Object.entries(files)) { const full = join(root, rel); mkdirSync(join(full, ".."), { recursive: true }); writeFileSync(full, data); } return root; } test("scanning skips what is not a page", () => { const root = tree({ "hungary/c/a_1_1.html": CHAPTER, "hungary/c/a_1_1_print.html": CHAPTER, "hungary/c/_hidden.html": CHAPTER, "hungary/c/pdfs/x.html": CHAPTER, "hungary/c/solutions/y.html": CHAPTER, "hungary/c/notes.txt": "x", "japan/japanese-language/reference-materials/kanji/kanji_x.html": CHAPTER, }); try { assert.deepEqual([...contentPaths(root)], ["hungary/c/a_1_1.html"]); } finally { rmSync(root, { recursive: true }); } }); test("a bad file is reported and the others still load", () => { const root = tree({ "france/c/ok_1_1.html": CHAPTER, "france/c/bad_date_1_2.html": "<!-- Course: A\n Chapter: B\n Date Created: 2026-13-01 Date Updated: 2026-13-02 -->", "france/c/not_utf8_1_3.html": new Uint8Array([0x3c, 0x70, 0x3e, 0xff, 0xfe]), "nonsense/x.html": CHAPTER, }); try { const { pages, errors } = loadPages(root); assert.deepEqual(pages.map((p) => p.path), ["france/c/ok_1_1.html"]); assert.deepEqual(errors.map((e) => e.path).sort(), ["france/c/bad_date_1_2.html", "france/c/not_utf8_1_3.html", "nonsense/x.html"]); assert.match(errors.find((e) => e.path.includes("utf8"))!.message, /not valid UTF-8/); assert.match(errors.find((e) => e.path.startsWith("nonsense"))!.message, /^NoSiteError/); } finally { rmSync(root, { recursive: true }); } }); test("a path that climbs out of the folder is refused", () => { const root = tree({ "france/a.html": CHAPTER }); try { for (const bad of ["../x.html", "/etc/passwd", "france\\a.html", "france/../../x.html"]) { assert.throws(() => readContentFile(root, bad), ContentParseError, bad); } } finally { rmSync(root, { recursive: true }); } }); test("a programming error is not hidden as a bad file", () => { assert.throws(() => loadPages(join(tmpdir(), "lw-this-folder-does-not-exist")), /ENOENT/); }); test("courses are grouped by folder with chapters in number order", () => { const make = (n: number) => CHAPTER.replace("Chapter: At", `Chapter: Ch${n} At`); const root = tree({ "germany/german-basic-3/german_basic_conversation_3_10.html": make(10), "germany/german-basic-3/german_basic_conversation_3_2.html": make(2), "germany/german-basic-3/german_basic_conversation_3_1.html": make(1), "germany/german-basic-3/german_basic_appendix.html": CHAPTER, "france/x/other.html": "<p>no banner</p>", }); try { const courses = groupCourses(loadPages(root).pages); assert.equal(courses.length, 1); assert.equal(courses[0]?.folder, "germany/german-basic-3"); assert.equal(courses[0]?.name, "German Basic Conversation 3"); assert.deepEqual(courses[0]?.chapters.map((c) => c.chapterNo), [1, 2, 10, null]); } finally { rmSync(root, { recursive: true }); } }); Run (the root test script now covers every package): npm test ℹ tests 26 ℹ pass 26 ℹ fail 0 (9 tests from @lw/sites and 17 new ones; typecheck of the package passes.) WHY THIS WORKS AS AN ANSWER --------------------------- The awkward cases (five file-name generations, impossible dates, a double-encoded entity, bad UTF-8, a path that climbs out, a file that is not a page) are each a test, and the failure policy is explicit: report the bad file, stop for a real bug.