learning-website-django1-4 Exercise 1: Turn a File Into a Stored Fragment ========================================================================== Add the code that turns a raw content file into the HTML the site shows: unwrap a complete document, drop the banner comment, rewrite every stale solution link to the course's own solutions folder, and refuse any path that would leave the content folder. These are the same steps the live site performs (Learning Website: Framework & Architecture 6), written as small Python functions that need no database. Save as apps/content/fragments.py: """Turn a raw content file into the fragment the site shows. These are the same steps as the live site's transform (Learning Website: Framework & Architecture 6): a complete document is unwrapped, a fragment loses its banner comment, and every stale link to a solution file is rewritten to the course's own solutions folder.""" import os import re from urllib.parse import unquote BODY_OPEN = re.compile(r"
]*>", re.I) BODY_CLOSE = re.compile(r"", re.I) HEAD = re.compile(r"]*>(.*?)", re.I | re.S) STYLE = re.compile(r"", re.I | re.S) LEADING_COMMENT = re.compile(r"^\s*\s*", re.S) TXT_HREF = re.compile(r'href="([^"]*?)([^"/]+\.txt)"', re.I) class MalformedDocument(ValueError): """A tag with no closing .""" class UnsafePath(ValueError): """A path that would leave the content folder.""" def safe_join(root, relative_path): """root + relative_path, refusing anything that resolves outside root (.., symlinks, absolute paths).""" root_real = os.path.realpath(root) full = os.path.realpath(os.path.join(root_real, *relative_path.split("/"))) if full != root_real and not full.startswith(root_real + os.sep): raise UnsafePath(relative_path) return full def detect_shape(raw): return "wrapped" if BODY_OPEN.search(raw) else "fragment" def extract_fragment(raw): """A complete document gives its body (plus the styles from its head); a fragment loses its banner.""" if detect_shape(raw) == "fragment": return LEADING_COMMENT.sub("", raw, count=1) opened, closed = BODY_OPEN.search(raw), BODY_CLOSE.search(raw) if not closed or closed.start() < opened.start(): raise MalformedDocument(" without a matching ") inner = raw[opened.end():closed.start()] head = HEAD.search(raw) styles = "\n".join(STYLE.findall(head.group(1))) + "\n" if head and STYLE.search(head.group(1)) else "" return styles + inner def rewrite_solution_links(fragment, course_url_path): """Every .txt link is a solution file: point it atkanji
") self.assertEqual(detect_shape(raw), "wrapped") out = extract_fragment(raw) self.assertIn("", out) self.assertIn("kanji
", out) self.assertNotIn("never closed
") class SolutionLinkTests(SimpleTestCase): COURSE = "/linux/system-administration/debian-development-machine-setup" def fix(self, href): out = rewrite_solution_links(f'x', self.COURSE) return out.split('"')[1] def test_every_stale_form_points_at_the_courses_solutions_folder(self): target = f"{self.COURSE}/solutions/devsetup1-1_exercise1.txt" for href in ("solutions/devsetup1-1_exercise1.txt", "devsetup1-1_exercise1.txt", "/lessons/solutions/exercises/devsetup1-1_exercise1.txt", "../../../lessons/solutions/exercises/devsetup1-1_exercise1.txt", "../text%20files/devsetup1-1_exercise1.txt"): self.assertEqual(self.fix(href), target, href) def test_an_encoded_file_name_is_decoded(self): self.assertTrue(self.fix("solutions/my%20file.txt").endswith("/solutions/my file.txt")) def test_other_links_are_untouched(self): raw = 'k p' self.assertEqual(rewrite_solution_links(raw, self.COURSE), raw) def test_prepare_fragment_uses_the_pages_own_folder(self): raw = '\ns' out = prepare_fragment(raw, "web-servers/apache-in-depth/apache_in_depth_1_1.html") self.assertIn('href="/web-servers/apache-in-depth/solutions/x_exercise1.txt"', out) self.assertNotIn("