learning-website-django1-4 Exercise 1: Turn a File Into a Stored Fragment ========================================================================== Add the code that turns a raw content file into the HTML the site shows: unwrap a complete document, drop the banner comment, rewrite every stale solution link to the course's own solutions folder, and refuse any path that would leave the content folder. These are the same steps the live site performs (Learning Website: Framework & Architecture 6), written as small Python functions that need no database. Save as apps/content/fragments.py: """Turn a raw content file into the fragment the site shows. These are the same steps as the live site's transform (Learning Website: Framework & Architecture 6): a complete document is unwrapped, a fragment loses its banner comment, and every stale link to a solution file is rewritten to the course's own solutions folder.""" import os import re from urllib.parse import unquote BODY_OPEN = re.compile(r"]*>", re.I) BODY_CLOSE = re.compile(r"", re.I) HEAD = re.compile(r"]*>(.*?)", re.I | re.S) STYLE = re.compile(r"]*>.*?", re.I | re.S) LEADING_COMMENT = re.compile(r"^\s*\s*", re.S) TXT_HREF = re.compile(r'href="([^"]*?)([^"/]+\.txt)"', re.I) class MalformedDocument(ValueError): """A tag with no closing .""" class UnsafePath(ValueError): """A path that would leave the content folder.""" def safe_join(root, relative_path): """root + relative_path, refusing anything that resolves outside root (.., symlinks, absolute paths).""" root_real = os.path.realpath(root) full = os.path.realpath(os.path.join(root_real, *relative_path.split("/"))) if full != root_real and not full.startswith(root_real + os.sep): raise UnsafePath(relative_path) return full def detect_shape(raw): return "wrapped" if BODY_OPEN.search(raw) else "fragment" def extract_fragment(raw): """A complete document gives its body (plus the styles from its head); a fragment loses its banner.""" if detect_shape(raw) == "fragment": return LEADING_COMMENT.sub("", raw, count=1) opened, closed = BODY_OPEN.search(raw), BODY_CLOSE.search(raw) if not closed or closed.start() < opened.start(): raise MalformedDocument(" without a matching ") inner = raw[opened.end():closed.start()] head = HEAD.search(raw) styles = "\n".join(STYLE.findall(head.group(1))) + "\n" if head and STYLE.search(head.group(1)) else "" return styles + inner def rewrite_solution_links(fragment, course_url_path): """Every .txt link is a solution file: point it at /solutions/, whatever it said before.""" prefix = course_url_path if course_url_path.startswith("/") else "/" + course_url_path return TXT_HREF.sub(lambda m: f'href="{prefix}/solutions/{unquote(m.group(2))}"', fragment) def prepare_fragment(raw, path): course_url_path = "/" + path.rsplit("/", 1)[0] if "/" in path else "" return rewrite_solution_links(extract_fragment(raw), course_url_path) Save as tests/test_fragments.py: import os import tempfile from django.test import SimpleTestCase from apps.content.fragments import (MalformedDocument, UnsafePath, detect_shape, extract_fragment, prepare_fragment, rewrite_solution_links, safe_join) class ExtractTests(SimpleTestCase): def test_a_fragment_loses_only_its_leading_banner(self): raw = "\n\n
body
" self.assertEqual(detect_shape(raw), "fragment") self.assertEqual(extract_fragment(raw), "\n
body
") def test_a_complete_document_gives_its_body_and_the_head_styles(self): raw = ("t" "

kanji

") self.assertEqual(detect_shape(raw), "wrapped") out = extract_fragment(raw) self.assertIn("", out) self.assertIn("

kanji

", out) self.assertNotIn("", out) self.assertNotIn("<html", out) def test_a_body_without_a_close_is_an_error(self): with self.assertRaises(MalformedDocument): extract_fragment("<html><body><p>never closed</p></html>") class SolutionLinkTests(SimpleTestCase): COURSE = "/linux/system-administration/debian-development-machine-setup" def fix(self, href): out = rewrite_solution_links(f'<a href="{href}">x</a>', self.COURSE) return out.split('"')[1] def test_every_stale_form_points_at_the_courses_solutions_folder(self): target = f"{self.COURSE}/solutions/devsetup1-1_exercise1.txt" for href in ("solutions/devsetup1-1_exercise1.txt", "devsetup1-1_exercise1.txt", "/lessons/solutions/exercises/devsetup1-1_exercise1.txt", "../../../lessons/solutions/exercises/devsetup1-1_exercise1.txt", "../text%20files/devsetup1-1_exercise1.txt"): self.assertEqual(self.fix(href), target, href) def test_an_encoded_file_name_is_decoded(self): self.assertTrue(self.fix("solutions/my%20file.txt").endswith("/solutions/my file.txt")) def test_other_links_are_untouched(self): raw = '<a href="/japan/kanji-tiles/">k</a> <a href="https://example.org/a.pdf">p</a>' self.assertEqual(rewrite_solution_links(raw, self.COURSE), raw) def test_prepare_fragment_uses_the_pages_own_folder(self): raw = '<!-- b -->\n<a href="x_exercise1.txt">s</a>' out = prepare_fragment(raw, "web-servers/apache-in-depth/apache_in_depth_1_1.html") self.assertIn('href="/web-servers/apache-in-depth/solutions/x_exercise1.txt"', out) self.assertNotIn("<!--", out) class SafeJoinTests(SimpleTestCase): def test_a_path_inside_the_root_is_allowed(self): with tempfile.TemporaryDirectory() as root: self.assertTrue(safe_join(root, "a/b.html").startswith(os.path.realpath(root))) def test_a_path_that_climbs_out_is_refused(self): with tempfile.TemporaryDirectory() as root: for bad in ("../outside.html", "a/../../outside.html"): with self.assertRaises(UnsafePath, msg=bad): safe_join(root, bad) Run them (Django 6.1.2); the full run is below, in Exercise 2. python manage.py test tests.test_fragments Design points ------------- - extract_fragment keeps only the leading banner out. A comment further down the page is content and stays (1,357 stored pages still contain one). - A complete document (the kanji pages) gives its body plus the style blocks from its head, so its styling survives. A body with no closing tag is an error, not something to repair silently. - rewrite_solution_links matches ANY .txt link and rebuilds it from the file name alone, so it copes with every stale form the old site used. The test lists five of them. - safe_join resolves the real path, including symbolic links, and refuses anything that ends up outside the root. It is the guard against a file or link that points somewhere it should not. PART B: Addresses with non-ASCII characters ------------------------------------------- 282 of the real page paths contain non-ASCII characters (the hiragana and katakana reference pages are named after the character, such as hiragana_あ.html). Check that the routing from Chapter 2 and the page's address agree, whether the browser sends the character percent-encoded or raw. Save as tests/test_unicode_routes.py: from django.test import Client, SimpleTestCase class NonAsciiAddressTests(SimpleTestCase): """Some page names are Japanese characters (hiragana_あ.html). The address is the same path.""" def get(self, path): return Client().get(path, HTTP_HOST="languages.localhost") def test_a_percent_encoded_address_reaches_the_page(self): response = self.get("/japan/japanese-language/reference-materials/hiragana/hiragana_%E3%81%82/") self.assertEqual(response.status_code, 200) self.assertIn("hiragana_あ/", response.content.decode()) def test_the_raw_character_in_an_address_works_too(self): response = self.get("/japan/japanese-language/reference-materials/hiragana/hiragana_あ/") self.assertEqual(response.status_code, 200) def test_a_page_stored_with_such_a_path_has_the_matching_address(self): from urllib.parse import quote from apps.content.models import Page page = Page(path="japan/japanese-language/reference-materials/hiragana/hiragana_あ.html") self.assertEqual(quote(page.url_path), "/japan/japanese-language/reference-materials/hiragana/hiragana_%E3%81%82/") All three pass (79 tests in the project at this point). Django decodes the percent-encoded address before the URL pattern is matched, so the same page path works for both forms. The address a page links to should be the percent-encoded form (urllib.parse.quote), which is what a browser sends. WHY THIS WORKS AS AN ANSWER --------------------------- Each step is a pure function with its own tests, so a change to one rule (say a new stale link form) is a one-line change with a one-line test, and none of it needs a database to check.