learning-website-nextjs1-9 Exercise 3: The Search Page, Sitemaps, Canonical Addresses and Open Graph ======================================================================================================= PART A: THE PAGE. The index is made when the site is built, as a file: /search-index.json (a route handler marked force-static). The search page is a client component that reads ?q=, fetches the index once, searches it and shows 20 results at a time. A plain form in the header goes to it (it works without any script). Save as apps/languages/lib/search.ts: import { buildSearchIndex, htmlToText, type SearchIndex } from "@lw/search"; import { content } from "./site"; let cached: SearchIndex | undefined; /** The search index of this site: every page's title, summary and text, made once per build. */ export function siteSearchIndex(): SearchIndex { cached ??= buildSearchIndex(content.all().map((page) => { const text = htmlToText(content.fragmentFor(page)); return { path: page.path, title: page.title, summary: page.summary || Array.from(text).slice(0, 160).join(""), text }; })); return cached; } Save as apps/languages/app/search-index.json/route.ts: import { siteSearchIndex } from "../../lib/search"; // made once, when the site is built: the browser fetches it as a plain file the first time anyone searches export const dynamic = "force-static"; export function GET() { return Response.json(siteSearchIndex()); } Save as apps/languages/app/search/page.tsx: import type { Metadata } from "next"; import { Suspense } from "react"; import { SearchResults } from "@lw/ui"; export const metadata: Metadata = { title: "Search | Philip's Learning Notes", robots: { index: false }, // a list of results is not a page worth finding in a search engine }; export default function SearchPage() { return ( <>

Search

{/* useSearchParams needs a Suspense boundary so the rest of the page can be made when the site is built */} Loading…

}>
); } Save as packages/ui/src/SearchResults.tsx: "use client"; import Link from "next/link"; import { useSearchParams } from "next/navigation"; import { useEffect, useState } from "react"; import { prepare, search, type PreparedIndex } from "@lw/search/client"; import styles from "./Search.module.css"; const PER_PAGE = 20; /** The index is fetched once, the first time anyone searches, and kept for the rest of the visit. */ let loading: Promise | undefined; function loadIndex(): Promise { loading ??= fetch("/search-index.json").then((response) => { if (!response.ok) throw new Error(`the search index could not be loaded (${response.status})`); return response.json(); }).then((index) => prepare(index)); loading.catch(() => { loading = undefined; }); // after a failure, the next search tries again return loading; } /** The address of a page from its file path: hungary/x/y.html is /hungary/x/y. */ const href = (path: string) => "/" + path.replace(/\.html$/, ""); export function SearchResults() { const params = useSearchParams(); const q = (params.get("q") ?? "").slice(0, 100); const page = Math.max(1, Number.parseInt(params.get("page") ?? "1", 10) || 1); const [index, setIndex] = useState(null); const [failed, setFailed] = useState(null); useEffect(() => { if (!q) return; loadIndex().then(setIndex, (error: Error) => setFailed(error.message)); }, [q]); const found = index && q ? search(index, q, { limit: PER_PAGE, offset: (page - 1) * PER_PAGE }) : null; const pages = found ? Math.max(1, Math.ceil(found.total / PER_PAGE)) : 1; const link = (n: number) => `/search?${new URLSearchParams({ q, page: String(n) })}`; return ( <>
{!q ?

Type a word or part of a word. Accents and capital letters do not matter, and Japanese is found by its characters.

: null} {failed ?

{failed}. Try again in a moment.

: null} {q && !found && !failed ?

Searching…

: null} {found ? ( <>

{found.total} result{found.total === 1 ? "" : "s"} for {q}

    {found.results.map((result) => (
  1. {result.title}

    {result.summary}

    {href(result.path)}

  2. ))}
{found.total === 0 ?

Nothing found. Try fewer or shorter words.

: null} {pages > 1 ? ( ) : null} ) : null} ); } Save as packages/ui/src/SearchBox.tsx: import styles from "./Search.module.css"; /** A search box for the site's header. It is a plain form: it works without any script, and goes to /search?q=... */ export function SearchBox({ label }: { label: string }) { return (
); } (The header form appears when SiteLayout is given search; only the languages app has the page, so only it passes it.) The page is marked noindex: a list of results is not something to find in a search engine. useSearchParams needs a Suspense boundary so the rest of the page can still be made at build time. PART B: ONE ADDRESS FOR EVERY PAGE. The sitemap, the canonical link and the Open Graph address all come from ONE function, so they cannot disagree: Save as apps/languages/lib/seo.ts: import type { Page } from "@lw/content"; import { pageHref } from "@lw/content"; import { htmlToText } from "@lw/search"; import { environment, siteUrl } from "@lw/sites"; import { content } from "./site"; /** The address of this site, without a trailing slash: https://languages.osztromok.com in production, http://languages.localhost:3001 in development. */ export const ORIGIN = siteUrl("languages", environment()).replace(/\/$/, ""); /** The ONE address of a page. The sitemap, the canonical link and the Open Graph address all come from here, so they cannot disagree. */ export function canonicalUrl(page: Page): string { return new URL(pageHref(page), ORIGIN).href; // non-ASCII letters and spaces are percent-encoded } /** The page's description: the banner's Topic line, or the first 160 characters of its text (the same rule as the Django project). */ export function descriptionFor(page: Page): string { if (page.summary) return page.summary; return Array.from(htmlToText(content.fragmentFor(page))).slice(0, 160).join(""); } Save as apps/languages/app/sitemap.ts: import type { MetadataRoute } from "next"; import { content } from "../lib/site"; import { canonicalUrl } from "../lib/seo"; /** /sitemap.xml: every page of this site, with the same address its canonical link uses. A date only where the page's banner has one. */ export default function sitemap(): MetadataRoute.Sitemap { return content.all() .sort((a, b) => (a.path < b.path ? -1 : 1)) .map((page) => (page.updated ? { url: canonicalUrl(page), lastModified: page.updated } : { url: canonicalUrl(page) })); } Save as apps/languages/app/robots.ts: import type { MetadataRoute } from "next"; import { ORIGIN } from "../lib/seo"; /** * /robots.txt. Crawling stays blocked until launch: set LW_ALLOW_CRAWLING=1 when BUILDING the site for the real domain. * (It is decided at build time here; the Django project decides at request time.) */ export default function robots(): MetadataRoute.Robots { if (process.env["LW_ALLOW_CRAWLING"] === "1") { return { rules: { userAgent: "*", allow: "/" }, sitemap: `${ORIGIN}/sitemap.xml` }; } return { rules: { userAgent: "*", disallow: "/" } }; } and in the page's generateMetadata (apps/languages/app/[...path]/page.tsx): title, description, alternates.canonical, and openGraph (type article, title, description, url, siteName, locale, published and modified times from the banner), plus a twitter card. The root layout sets metadataBase to the site's own address, so every relative address in the metadata is made absolute from it. DECISION: the canonical form has NO trailing slash (Next's way), where the Django project used one. Old addresses with a slash are redirected (308): /hungary/hungarian-basic-3/hungarian_basic_conversation_3_1/ 308 -> .../hungarian_basic_conversation_3_1 /hungary/hungarian-basic-3/ 308 -> /hungary/hungarian-basic-3 /search/ -> /search /sitemap.xml/ -> /sitemap.xml so no link anyone made is lost, and each page has exactly one address. CHECKED on the built site (check-seo.mjs reads every built page): Save as check-seo.mjs: // Check the built languages site's search-engine metadata: every page's canonical address is in the sitemap, descriptions match the // Django project's, Open Graph tags are complete, and robots.txt says what it should. // node check-seo.mjs (LW_CONTENT_ROOT = the content folder; the site must have been built) import { readFileSync, existsSync } from "node:fs"; import { join } from "node:path"; import { decodeEntities, loadPages, urlSegments } from "./packages/content/src/index.ts"; const root = process.env.LW_CONTENT_ROOT; const dump = JSON.parse(readFileSync(process.argv[2], "utf8")); const summary = new Map(dump.pages.map((p) => [p.path, p.summary])); const built = join(import.meta.dirname, "apps/languages/.next/server/app"); const { pages } = loadPages(root, { site: "languages" }); const sitemap = readFileSync(join(built, "sitemap.xml.body"), "utf8"); const locs = [...sitemap.matchAll(/([^<]*)<\/loc>/g)].map((m) => decodeEntities(m[1])); const dated = [...sitemap.matchAll(//g)].length; console.log(`sitemap: ${locs.length} addresses (${new Set(locs).size} different), ${dated} with a date; pages on the site: ${pages.length}`); console.log("sitemap first address:", locs[0]); const encoded = locs.find((l) => /%[0-9A-F]{2}/.test(l)); console.log("an address with non-ASCII letters:", encoded); const meta = (html, re) => { const m = re.exec(html); return m ? decodeEntities(m[1]) : null; }; const tally = { canonicalInSitemap: 0, canonicalMissing: 0, descriptionSame: 0, descriptionDifferent: 0, ogComplete: 0, ogIncomplete: 0, htmlMissing: 0 }; const problems = []; const locSet = new Set(locs); for (const page of pages) { const file = join(built, ...urlSegments(page)) + ".html"; if (!existsSync(file)) { tally.htmlMissing++; continue; } const html = readFileSync(file, "utf8"); const canonical = meta(html, / !pages.some((p) => l === new URL(`/${urlSegments(p).join("/")}`, locs[0]).href)).length); const robots = readFileSync(join(built, "robots.txt.body"), "utf8"); console.log("robots.txt:", JSON.stringify(robots)); const search = readFileSync(join(built, "search.html"), "utf8"); console.log("search page robots meta:", meta(search, / import { spawn } from "node:child_process"; import { mkdtempSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; const base = process.argv[2] ?? "http://localhost:3001"; const chrome = spawn("C:/Program Files/Google/Chrome/Application/chrome.exe", [ "--headless=new", "--disable-gpu", "--remote-debugging-port=9338", `--user-data-dir=${mkdtempSync(join(tmpdir(), "lw-chrome-"))}`, "about:blank", ], { stdio: "ignore" }); const sleep = (ms) => new Promise((r) => setTimeout(r, ms)); let targets; for (let i = 0; i < 40; i++) { try { targets = await (await fetch("http://127.0.0.1:9338/json")).json(); if (targets.length) break; } catch { /* not up yet */ } await sleep(250); } const ws = new WebSocket(targets.find((t) => t.type === "page").webSocketDebuggerUrl); await new Promise((r) => ws.addEventListener("open", r)); let id = 0; const pending = new Map(); ws.addEventListener("message", (e) => { const m = JSON.parse(e.data); if (m.id && pending.has(m.id)) { pending.get(m.id)(m); pending.delete(m.id); } }); const send = (method, params = {}) => new Promise((r) => { const n = ++id; pending.set(n, r); ws.send(JSON.stringify({ id: n, method, params })); }); const evaluate = async (expression) => (await send("Runtime.evaluate", { expression, returnByValue: true })).result?.result?.value; const rows = []; const record = (ok, what, detail) => rows.push([ok ? "PASS" : "FAIL", what, detail]); /** Open a search and wait until the results (or the "nothing found" note) are on the page; returns how long that took. */ async function searchFor(q, page = 1) { await send("Page.enable"); const started = Date.now(); await send("Page.navigate", { url: `${base}/search?${new URLSearchParams(page === 1 ? { q } : { q, page: String(page) })}` }); for (let i = 0; i < 100; i++) { await sleep(50); const done = await evaluate('!!document.querySelector("ol li") || /Nothing found|Type a word|could not be loaded/.test(document.body.innerText)'); if (done) break; } return Date.now() - started; } const count = () => evaluate('document.querySelectorAll("ol li").length'); const summaryLine = () => evaluate('(document.body.innerText.match(/(\\d+) results? for/) || [])[1] || null'); for (const [q, expected] of [["szeretnek", 18], ["Szeretnék", 18], ["brotchen", 1], ["polite cond", 10], ["ありがとう", 25], ["ありが", 25], ["水", 5], ["zzzqqq", 0]]) { const ms = await searchFor(q); const shown = await count(); const total = await summaryLine(); record(Number(total ?? 0) === expected && shown === Math.min(20, expected), `search "${q}": ${expected} results (as in Django)`, `${shown} shown, count says ${total ?? "0"}, ${ms} ms from opening the page`); } await searchFor('"; DROP TABLE x; --'); record((await count()) === 0 && (await evaluate('document.body.innerText.includes("Nothing found")')), "an injection-like query is just a search with no results", await evaluate("document.querySelector('main').innerText.slice(0, 60)")); await searchFor(''); record((await evaluate("document.querySelectorAll('main img').length")) === 0 && !(await evaluate("window.__xss")), "HTML typed into the search is shown as text, not run", await evaluate("document.querySelector('input[name=q]').value")); // the index: how big is it on the wire, and is it fetched only once per visit? await searchFor("hotel"); const resource = await evaluate(`JSON.stringify(performance.getEntriesByType("resource").filter(r => r.name.endsWith("search-index.json")).map(r => ({ transferKB: Math.round(r.transferSize / 1024), decodedKB: Math.round(r.decodedBodySize / 1024), ms: Math.round(r.duration) })))`); record(JSON.parse(resource).length === 1, "the index is fetched once", resource); // pages of results await searchFor("the"); const first = await evaluate('JSON.stringify([...document.querySelectorAll("ol li a")].map(a => a.getAttribute("href")))'); record((await count()) === 20 && (await summaryLine()) === "348", 'page 1 of "the": 20 of 348', `${await count()} shown`); await evaluate('[...document.querySelectorAll("nav a")].find(a => a.textContent.includes("Next")).click()'); await sleep(800); const second = await evaluate('JSON.stringify([...document.querySelectorAll("ol li a")].map(a => a.getAttribute("href")))'); record(JSON.parse(second).length === 20 && JSON.parse(second).every((h) => !JSON.parse(first).includes(h)), "Next shows the next 20, different ones", `url ${await evaluate("location.search")}`); await searchFor("the", 18); record((await count()) === 8, "the last page (18 of 18) has the remaining 8", `${await count()} shown`); await searchFor("the", 999); record((await count()) === 0, "a page number past the end shows no results, no crash", `${await count()} shown`); // the box in the header await send("Page.navigate", { url: `${base}/hungary/hungarian-basic-3` }); await sleep(1500); await evaluate('(() => { const i = document.getElementById("site-q"); i.value = "kabát"; i.form.requestSubmit(); })()'); await sleep(2500); record((await evaluate("location.pathname + location.search")).startsWith("/search?q="), "the header search box goes to /search?q=...", await evaluate("location.pathname + location.search")); record((await count()) > 0, "and accented text typed there finds pages", `${await count()} shown for kabát`); // a result is a normal link: a click is a client-side navigation await evaluate("window.__marker = 1; true"); await evaluate('document.querySelector("ol li a").click()'); await sleep(1500); record((await evaluate("window.__marker === 1")) && !(await evaluate("location.pathname")).startsWith("/search"), "clicking a result moves to the page without a reload", await evaluate("location.pathname")); for (const [status, what, detail] of rows) console.log(status, what.padEnd(76), String(detail).slice(0, 80)); console.log(`\n${rows.filter((r) => r[0] === "PASS").length} passed, ${rows.filter((r) => r[0] === "FAIL").length} failed`); ws.close(); chrome.kill(); process.exit(0); 18 passed, 0 failed "szeretnek" 18, "Szeretnék" 18, "brotchen" 1, "polite cond" 10, "ありがとう" 25 (20 shown), "ありが" 25, "水" 5, "zzzqqq" 0: the same counts as the Django search. an injection-like query: a search with no results; HTML typed into the box (an ): shown as text, nothing ran; the index is fetched once per visit; "the" gives 348 results in pages of 20 (20, 20, ..., the last page 8; a page past the end shows none); the header box with accented text ("kabát") finds pages; clicking a result is a client-side navigation (no reload). Time from opening the search page to seeing results: 64 to 136 ms after the first (the first, which loads the index, 439 ms), on this machine. FOUND ON THE WAY, and what I did: - The index is 995 KB on the wire here. "next start" compresses pages (a 107 KB page went out as 20 KB) but NOT this file, and a generated file has no ETag, so a browser would download it again on every visit. Two changes: Cache-Control: public, max-age=3600 for /search-index.json (an index up to an hour old after a rebuild is acceptable), and compression is left to Apache (gzip would send 320 KB). The Apache part is Chapter 12 and NOT tested. - A build warning (Turbopack: "dynamic filesystem access causes tracing of the whole project") from the Chapter 8 code that lists a folder's downloads. That code only runs while the site is built, so it is marked to be ignored; the warning is gone. WHAT WAS NOT DONE: the results show the page's summary, not the matching words highlighted (the index holds no page text, to stay small; the Django search showed highlighted snippets); there is no og:image (no page has an image, so there is nothing to show: one picture per site would be a decision for you); other sites' search pages (only the languages app has routes yet); other browsers. WHY THIS WORKS AS AN ANSWER --------------------------- Everything a search engine or a visitor sees is checked on the real, built site (561 pages, 250 dates, 18 browser checks), the one fact that must agree (the page's address) has one source, and the things given up (snippets, og:image, compression) are named.