learning-website-django1-8 Exercise 1: Mirror the PDFs and Solutions ================================================================ Pages link to PDFs and solution files at the same relative path as in the content folder (the importer rewrites every .txt link to //solutions/), so those addresses must keep working after the split. Mirror every pdfs/ and solutions/ file into one folder PER SITE, so a site can only ever serve its own files. Django 6.1.2. Settings (config/settings/base.py): ASSET_ROOT = Path(os.environ.get("LW_ASSET_ROOT", BASE_DIR / "assets")) SERVE_ASSETS_WITH_DJANGO = False # production: Apache serves them; dev.py sets True Save as apps/content/assets.py: """PDFs and solution files. Every course folder may hold a pdfs/ folder and a solutions/ folder. A page links to them at the SAME relative path, so /hungary/hungarian-basic-3/solutions/ex1.txt must keep working after the split. They are mirrored into one folder per site, ASSET_ROOT//, so a site can only ever serve its own files, and Apache can serve them directly (Django's own file serving is for development only).""" import os import shutil from dataclasses import dataclass, field from config.sites_config import NoSiteError, site_for_path ASSET_FOLDERS = ("pdfs", "solutions") ASSET_EXTENSIONS = (".pdf", ".txt") @dataclass class CollectResult: copied: int = 0 unchanged: int = 0 removed: int = 0 skipped: list = field(default_factory=list) # (path, reason): never copied bytes_copied: int = 0 def iter_asset_paths(root): """Relative paths (forward slashes) of every PDF and solution file in the content folder.""" for dirpath, dirnames, filenames in os.walk(root): rel_dir = os.path.relpath(dirpath, root).replace(os.sep, "/") parts = rel_dir.split("/") if parts[-1] not in ASSET_FOLDERS: continue for name in sorted(filenames): if name.lower().endswith(ASSET_EXTENSIONS): yield f"{rel_dir}/{name}" def collect_assets(content_root, asset_root, *, only_site=None, dry_run=False, prune=False): """Copy a file only when it is missing or different (size or modified time). A second run does nothing.""" result = CollectResult() wanted = set() for rel in iter_asset_paths(content_root): try: site = site_for_path(rel) except NoSiteError: result.skipped.append((rel, "belongs to no site")) continue if only_site and site != only_site: continue wanted.add(f"{site}/{rel}") source = os.path.join(content_root, *rel.split("/")) target = os.path.join(asset_root, site, *rel.split("/")) if os.path.exists(target): a, b = os.stat(source), os.stat(target) if a.st_size == b.st_size and int(a.st_mtime) <= int(b.st_mtime): result.unchanged += 1 continue result.copied += 1 result.bytes_copied += os.path.getsize(source) if not dry_run: os.makedirs(os.path.dirname(target), exist_ok=True) shutil.copy2(source, target) # copy2 keeps the modified time, so the next run sees "same" if prune and os.path.isdir(asset_root): for site_dir in os.listdir(asset_root): if only_site and site_dir != only_site: continue for dirpath, _, filenames in os.walk(os.path.join(asset_root, site_dir)): for name in filenames: full = os.path.join(dirpath, name) key = os.path.relpath(full, asset_root).replace(os.sep, "/") if key not in wanted: result.removed += 1 if not dry_run: os.remove(full) return result def list_assets(asset_root, site, folder): """The PDFs and solution files that sit directly in /pdfs and /solutions on this site.""" found = {} for kind in ASSET_FOLDERS: directory = os.path.join(asset_root, site, *folder.split("/"), kind) names = sorted(n for n in os.listdir(directory) if n.lower().endswith(ASSET_EXTENSIONS)) if os.path.isdir(directory) else [] found[kind] = [{"name": n, "href": f"/{folder}/{kind}/{n}"} for n in names] return found Save as apps/content/management/commands/collect_assets.py: import time from django.conf import settings from django.core.management.base import BaseCommand from apps.content.assets import collect_assets from config.sites_config import SITES class Command(BaseCommand): help = "Mirror every pdfs/ and solutions/ file into ASSET_ROOT//, at the same relative path." def add_arguments(self, parser): parser.add_argument("--root", default=None, help="content folder (default: settings.CONTENT_ROOT)") parser.add_argument("--dry-run", action="store_true", help="report what would be copied; copy nothing") parser.add_argument("--prune", action="store_true", help="delete mirrored files whose source has gone") parser.add_argument("--site", choices=sorted(SITES), help="only this site's files") def handle(self, *args, **opts): started = time.perf_counter() result = collect_assets(opts["root"] or settings.CONTENT_ROOT, settings.ASSET_ROOT, only_site=opts["site"], dry_run=opts["dry_run"], prune=opts["prune"]) mode = "DRY RUN (nothing copied): " if opts["dry_run"] else "" self.stdout.write(f"{mode}copied {result.copied} ({result.bytes_copied / 1e6:.0f} MB), " f"unchanged {result.unchanged}, removed {result.removed}, " f"skipped {len(result.skipped)} ({time.perf_counter() - started:.1f}s)") for path, reason in result.skipped[:5]: self.stdout.write(f" skipped {path}: {reason}") Save as tests/test_assets.py (the first part, CollectTests; the second part is Exercise 2): import os import tempfile import time from pathlib import Path from django.core.management import call_command from django.test import Client, TestCase, override_settings from apps.content import importer from apps.content.assets import collect_assets, iter_asset_paths, list_assets from apps.content.models import Page from tests.test_importer import ContentTree CHAPTER = "hungary/hungarian-basic-3/hungarian_basic_conversation_3_1.html" PDF = "hungary/hungarian-basic-3/pdfs/Hungarian_Basic_Conversation_3_Course.pdf" SOLUTION = "hungary/hungarian-basic-3/solutions/ex1.txt" class AssetTree(TestCase): """A content folder with a chapter, a PDF and a solution, and a separate asset folder.""" def setUp(self): self.tree = ContentTree(self) self.tree.write(CHAPTER, '\nsolution') self.tree.write(PDF, b"%PDF-1.4 fake pdf", binary=True) self.tree.write(SOLUTION, "print('hello')\n") self.assets = tempfile.TemporaryDirectory() self.addCleanup(self.assets.cleanup) self.asset_root = Path(self.assets.name) def collect(self, **kw): return collect_assets(self.tree.root, self.asset_root, **kw) class CollectTests(AssetTree): def test_only_files_in_pdfs_and_solutions_folders_are_found(self): self.tree.write("hungary/hungarian-basic-3/notes.txt", "not an asset") self.tree.write("hungary/hungarian-basic-3/pdfs/readme.md", "not an asset") self.assertEqual(sorted(iter_asset_paths(self.tree.root)), sorted([PDF, SOLUTION])) def test_files_are_mirrored_under_their_site_at_the_same_relative_path(self): result = self.collect() self.assertEqual((result.copied, result.unchanged), (2, 0)) self.assertTrue((self.asset_root / "languages" / PDF).is_file()) self.assertEqual((self.asset_root / "languages" / SOLUTION).read_text(), "print('hello')\n") def test_a_second_run_copies_nothing(self): self.collect() again = self.collect() self.assertEqual((again.copied, again.unchanged), (0, 2)) def test_a_changed_file_is_copied_again_and_only_that_one(self): self.collect() time.sleep(1.1) self.tree.write(SOLUTION, "print('changed, and longer')\n") again = self.collect() self.assertEqual((again.copied, again.unchanged), (1, 1)) def test_a_dry_run_copies_nothing(self): result = self.collect(dry_run=True) self.assertEqual(result.copied, 2) self.assertFalse(any(self.asset_root.iterdir())) def test_a_file_in_no_site_is_reported_and_skipped(self): self.tree.write("nonsense/x/solutions/a.txt", "orphan") result = self.collect() self.assertEqual(result.copied, 2) self.assertEqual(result.skipped, [("nonsense/x/solutions/a.txt", "belongs to no site")]) def test_one_site_only(self): self.tree.write("linux/vim/solutions/v.txt", "vim") result = self.collect(only_site="systems") self.assertEqual(result.copied, 1) self.assertTrue((self.asset_root / "systems" / "linux/vim/solutions/v.txt").is_file()) def test_prune_removes_mirrored_files_whose_source_has_gone(self): self.collect() os.remove(self.tree.root / SOLUTION) self.assertEqual(self.collect().removed, 0) # reported only with --prune self.assertEqual(self.collect(prune=True).removed, 1) self.assertFalse((self.asset_root / "languages" / SOLUTION).exists()) def test_the_sidebar_follows_its_subject(self): self.tree.write("sidebar/linux/solutions/s.txt", "x") self.collect() self.assertTrue((self.asset_root / "systems" / "sidebar/linux/solutions/s.txt").is_file()) def test_list_assets_gives_addresses_in_the_same_form_as_the_pages_links(self): self.collect() found = list_assets(self.asset_root, "languages", "hungary/hungarian-basic-3") self.assertEqual(found["pdfs"], [{"name": "Hungarian_Basic_Conversation_3_Course.pdf", "href": "/" + PDF}]) self.assertEqual(found["solutions"][0]["href"], "/" + SOLUTION) def test_the_command_reports_what_it_did(self): from io import StringIO out = StringIO() with override_settings(ASSET_ROOT=self.asset_root): call_command("collect_assets", root=str(self.tree.root), stdout=out) self.assertIn("copied 2", out.getvalue()) class ServeAssetTests(AssetTree): def setUp(self): super().setUp() self.collect() self.override = override_settings(ASSET_ROOT=self.asset_root) self.override.enable() self.addCleanup(self.override.disable) self.http = Client(HTTP_HOST="languages.localhost") def get(self, path, client=None, **headers): """A request whose response is closed afterwards: Windows will not delete a file that is still open.""" response = (client or self.http).get(path, **headers) self.addCleanup(response.close) return response def test_a_pdf_is_served_with_its_type(self): response = self.get("/" + PDF) self.assertEqual(response.status_code, 200) self.assertEqual(response["Content-Type"], "application/pdf") self.assertEqual(b"".join(response.streaming_content), b"%PDF-1.4 fake pdf") def test_a_solution_is_plain_text_and_the_browser_may_not_guess_otherwise(self): response = self.get("/" + SOLUTION) self.assertEqual(response.status_code, 200) self.assertTrue(response["Content-Type"].startswith("text/plain")) self.assertEqual(response["X-Content-Type-Options"], "nosniff") def test_a_repeat_request_with_the_modified_date_gets_304(self): first = self.get("/" + SOLUTION) again = self.get("/" + SOLUTION, HTTP_IF_MODIFIED_SINCE=first["Last-Modified"]) self.assertEqual(again.status_code, 304) def test_another_sites_assets_are_not_served_here(self): self.assertEqual(self.get("/" + PDF, client=Client(HTTP_HOST="systems.localhost")).status_code, 404) def test_a_missing_file_and_a_directory_are_404(self): self.assertEqual(self.get("/hungary/hungarian-basic-3/solutions/nope.txt").status_code, 404) self.assertEqual(self.get("/hungary/hungarian-basic-3/solutions/").status_code, 404) def test_a_path_that_climbs_out_is_refused(self): for bad in ("/hungary/hungarian-basic-3/solutions/..%2f..%2f..%2fcollect.txt", "/hungary/hungarian-basic-3/solutions/%2e%2e/%2e%2e/x.txt"): self.assertIn(self.get(bad).status_code, (400, 404), bad) def test_only_pdf_and_txt_files_in_those_folders_are_routed(self): (self.asset_root / "languages" / "hungary/hungarian-basic-3/solutions/run.sh").write_text("echo hi") self.assertEqual(self.get("/hungary/hungarian-basic-3/solutions/run.sh").status_code, 404) def test_the_link_written_by_the_importer_reaches_the_file(self): importer.import_tree(self.tree.root) link = Page.objects.get().fragment.split('href="')[1].split('"')[0] self.assertEqual(link, "/" + SOLUTION) self.assertEqual(self.get(link).status_code, 200) def test_the_course_page_lists_the_downloads(self): importer.import_tree(self.tree.root) html = self.get("/hungary/hungarian-basic-3/").content.decode() self.assertIn('href="/' + PDF + '" download', html) self.assertIn("Exercise solutions (1 files)", html) def test_in_production_django_does_not_serve_assets_itself(self): # the routes are built when the project starts, and Django caches the resolver: rebuild, then clear the cache from django.urls import clear_url_caches from config.urlconfs import build try: with override_settings(SERVE_ASSETS_WITH_DJANGO=False): build("languages") clear_url_caches() self.assertEqual(self.get("/" + PDF).status_code, 404) finally: with override_settings(SERVE_ASSETS_WITH_DJANGO=True): build("languages") clear_url_caches() self.assertEqual(self.get("/" + PDF).status_code, 200) Run on the real content (a dry run for every site, then two sites for real, then a repeat): python manage.py collect_assets --dry-run python manage.py collect_assets --site languages python manage.py collect_assets --site systems python manage.py collect_assets --site languages Output (checked by running it; times are for a OneDrive-synced folder): DRY RUN (nothing copied): copied 8394 (666 MB), unchanged 0, removed 0, skipped 0 (8.1s) copied 236 (195 MB), unchanged 0, removed 0, skipped 0 (2.0s) copied 1419 (90 MB), unchanged 0, removed 0, skipped 0 (3.4s) copied 0 (0 MB), unchanged 236, removed 0, skipped 0 (0.3s) What it shows ------------- - The real content has 8,394 PDFs and solution files, 666 MB, and every one belongs to a site: nothing was skipped. - Languages has 236 files (195 MB), systems 1,419 (90 MB). The repeat run copies nothing and takes 0.3 seconds, because a file is copied only when it is missing or its size or modified time differs. - copy2 keeps the modified time, which is what lets the next run see "same". - A file in a folder that belongs to no site is reported and skipped, never copied somewhere by guess. --prune removes mirrored files whose source has gone, and is a separate flag for the same reason as in the importer: deleting should never be a default. - PDFs are build outputs and are not in git, so on a server the content folder (or the mirrored folder) must be put there some other way: rsync the folder, or rebuild the PDFs with their scripts. WHY THIS WORKS AS AN ANSWER --------------------------- One folder per site makes the site boundary a property of the file system, not of code that has to remember to check. The same relative path keeps every old link working, and the second run costing nothing means the step can run on every deploy.