learning-website-django1-11 Exercise 3: Links Between Sites, Broken Links and the Cut-Over ========================================================================================= Part A: links between sites. A page written for the old single site links to /linux/... with a plain absolute path. After the split /linux/ is on the systems site, so that path on the languages site is a 404. Rather than edit thousands of files (and bake a host name into them), the link is rewritten WHEN THE PAGE IS SHOWN. The stored page is unchanged, and the address comes from SITE_URL_TEMPLATE, so development and production each get their own. Save as apps/content/crosslinks.py: """Links between sites. A page written for the old single site links to /linux/... with a plain absolute path. After the split /linux/ is on the systems site, so the same path on the languages site is a 404. Rather than edit 4,000 files (and bake a host name into them), the link is rewritten WHEN THE PAGE IS SHOWN: a link to another site's folder gets that site's address in front. Files stay unchanged and independent of where the site is hosted; the address comes from SITE_URL_TEMPLATE, which is different in development and production.""" import re from urllib.parse import unquote from django.conf import settings from config.sites_config import NoSiteError, site_for_path HREF = re.compile(r"""(href=(["']))(/[^"'#?]*)""") def owner_of(path): """The site that owns an absolute path such as /linux/x/, or None (a search page, a static file, the front page).""" parts = unquote(path).strip("/") if not parts: return None try: return site_for_path(parts) except NoSiteError: return None def rewrite_cross_site_links(html, site): """Give every link that points into ANOTHER site's folders that site's address. Returns the new html.""" if 'href="/' not in html and "href='/" not in html: return html def replace(match): path = match.group(3) if path.startswith("//"): return match.group(0) owner = owner_of(path) if owner is None or owner == site: return match.group(0) return match.group(1) + settings.SITE_URL_TEMPLATE.format(site=owner) + path return HREF.sub(replace, html) Page.html (apps/content/models.py) now returns mark_safe(rewrite_cross_site_links(self.fragment, self.site)). On the real content this found NOTHING to rewrite: of 11,691 internal links, every one points into its own site (and 24 are not content at all, such as /about/ and /wp-content/ in sample code). That is a good result for the site map (the folder split is clean) and it means the rewriter is only exercised by tests today; it is there for the day a page links across. Part B: broken links. Read every stored page and check each absolute internal link. Save as apps/migration/links.py: """Find internal links that lead nowhere, by reading every stored page.""" import re from collections import Counter from pathlib import Path from urllib.parse import unquote from django.conf import settings from apps.content.crosslinks import owner_of from apps.content.models import Page from .models import Redirect from .redirects import bare_path ABSOLUTE = re.compile(r"""href=(["'])(/[^"'#?]*)""") FILE_EXTENSIONS = (".pdf", ".txt") def known_addresses(): """(set of page addresses, set of folder addresses), both as "hungary/x/y" with no slashes at the ends.""" pages, folders = set(), set() for path in Page.objects.values_list("path", flat=True): bare = path[:-5] pages.add(bare) parts = bare.split("/") for i in range(1, len(parts)): folders.add("/".join(parts[:i])) return pages, folders def broken_links(asset_root=None, limit_per_page=None): """Returns (checked, broken) where broken is a list of (page path, link, why). A link is fine when it names a page, a folder that has pages, a mirrored PDF or solution file, a search or account address, or an old address that has a redirect. Links to another SITE are fine too: they are rewritten when the page is shown (apps.content.crosslinks), so they are not counted as broken here.""" pages, folders = known_addresses() redirects = set(Redirect.objects.values_list("old_path", flat=True)) root = Path(asset_root or settings.ASSET_ROOT) checked = 0 broken = [] for path, fragment in Page.objects.values_list("path", "fragment").iterator(chunk_size=200): for match in ABSOLUTE.finditer(fragment): link = match.group(2) if link.startswith("//"): continue checked += 1 target = unquote(link) owner = owner_of(target) if owner is None: continue # /search/, /static/..., /accounts/...: not content if target.lower().endswith(FILE_EXTENSIONS): if not (root / owner / target.strip("/")).is_file(): broken.append((path, link, "file not found")) continue bare = bare_path(target) if bare in pages or bare in folders or bare in redirects: continue broken.append((path, link, "no such page")) return checked, broken def summarise(broken): return Counter(why for _, _, why in broken) Save as apps/migration/management/commands/check_links.py: from django.core.management.base import BaseCommand, CommandError from apps.migration.links import broken_links, summarise class Command(BaseCommand): help = "List internal links in the stored pages that lead nowhere." def add_arguments(self, parser): parser.add_argument("--list", metavar="FILE", help="write every broken link to this file") def handle(self, list=None, **options): checked, broken = broken_links() self.stdout.write(f"{checked} absolute internal links checked; {len(broken)} broken " + str(dict(summarise(broken)))) if list: with open(list, "w", encoding="utf-8") as handle: for path, link, why in broken: handle.write(f"{path}\t{link}\t{why}\n") if broken: raise CommandError(f"{len(broken)} broken links") Run on the real data (3 seconds): 11715 absolute internal links checked; 483 broken {'no such page': 305, 'file not found': 178} What they are: - 282 links to /japan/hiragana/hiragana-tiles and /japan/katakana/katakana-tiles (141 each), from the hiragana and katakana character pages: those tile pages are dedicated pages on the old site that the new site does not have yet (like japan/kanji-tiles) - 174 links to .txt solutions in /web-development/scripting-and-backend/... that do not exist: the same solution-link problem already noted for several courses in website_checks.md - 23 links under /resources/japanese, 3 in office-and-productivity-software/excel, 1 in security Part C: cutting over one site at a time. While the old site still answers on osztromok.com, a site moves by telling the OLD host to send that site's folders to the new subdomain: Save as apps/migration/cutover.py: """Run side by side, move one site at a time. While the old site still answers on osztromok.com, a site is "cut over" by telling the old host's Apache to send that site's folders to the new subdomain with a permanent redirect. Nothing else changes: the other sites' folders keep being served by the old site until their own turn.""" import re from config.sites_config import SIDEBAR_ROUTES, SITES, SPECIAL_PREFIXES def folders_of(site): """The top-level paths a site owns, including the sidebar subjects and special prefixes that route to it.""" names = list(SITES[site]["folders"]) names += [f"sidebar/{subject}" for subject, owner in SIDEBAR_ROUTES.items() if owner == site] names += [prefix for prefix, owner in SPECIAL_PREFIXES.items() if owner == site] return names def rule_for(site, domain="osztromok.com"): """One RedirectMatch line for a site. Apache's pattern is a regular expression with $1 for the rest of the path.""" alternatives = "|".join(re.escape(f).replace("\\-", "-") for f in folders_of(site)) return f"RedirectMatch 301 ^/({alternatives})(/.*)?$ https://{site}.{domain}/$1$2" def pattern_for(site): """The same pattern as a Python regular expression, so it can be tested without Apache.""" alternatives = "|".join(re.escape(f).replace("\\-", "-") for f in folders_of(site)) return re.compile(rf"^/({alternatives})(/.*)?$") def snippet(sites, domain="osztromok.com"): return "\n".join(rule_for(site, domain) for site in sites) Save as apps/migration/management/commands/apache_cutover.py: from django.core.management.base import BaseCommand, CommandError from apps.migration.cutover import snippet from config.sites_config import SITES class Command(BaseCommand): help = "Print the Apache lines that send a site's old addresses to its new subdomain." def add_arguments(self, parser): parser.add_argument("sites", nargs="+", help="site names, for example: languages") def handle(self, sites, **options): unknown = [s for s in sites if s not in SITES] if unknown: raise CommandError(f"unknown site: {', '.join(unknown)}") self.stdout.write(snippet(sites)) python manage.py apache_cutover languages RedirectMatch 301 ^/(france|germany|hungary|japan|culture|resources/japanese|resources/hungarian)(/.*)?$ https://languages.osztromok.com/$1$2 Checked without Apache, by running the same pattern as a Python regular expression over every old address: of 8,892 old page and file addresses that a site owns, 0 were matched by the wrong site or by two sites, and none of the 144 addresses that no site owns was matched. A test checks that no folder is claimed by two sites and that /linuxfoo/ does not match the systems rule. What was NOT verified: these lines were not run in Apache (the development machine has none). Before using them: apachectl configtest, then curl -I against one address of the site, expecting 301 and the right Location, and keep the old files in place until you have checked. Tests for the whole chapter are in tests/test_migration.py: Save as tests/test_migration.py: import re import tempfile from pathlib import Path from django.core.management import call_command from django.core.management.base import CommandError from django.db import IntegrityError from django.test import Client, SimpleTestCase, TestCase, override_settings from apps.content.crosslinks import owner_of, rewrite_cross_site_links from apps.content.models import Course, Page from apps.migration import cutover, links, redirects from apps.migration.models import Redirect from apps.migration.oldsite import scan_old_site from apps.migration.verify import verify from config.sites_config import SITES def make(path, site="languages", fragment="
x
"): return Page.objects.create(path=path, site=site, kind="lesson", title=path, fragment=fragment) def old_tree(testcase, files): """A temporary 'old site': {relative path: text}.""" folder = tempfile.TemporaryDirectory() testcase.addCleanup(folder.cleanup) for rel, text in files.items(): target = Path(folder.name) / rel target.parent.mkdir(parents=True, exist_ok=True) target.write_text(text, encoding="utf-8") return folder.name class OldSiteScanTests(TestCase): def test_every_kind_of_old_file_is_classified(self): root = old_tree(self, { "index.html": "", "hungary/x/index.html": "", "hungary/x/pdfs/a.pdf": "", "hungary/x/solutions/s.txt": "", "linux/y/index.html": "", "anime/vault.php": "", "_astro/app.css": "", "css/site.css": "", "sidebar/ai/index.html": "", "resources/japanese/kanji/kanji_水.html": "", }) found = {u.path: u for u in scan_old_site(root)} self.assertEqual(found["hungary/x"].kind, "page") self.assertEqual(found["hungary/x"].site, "languages") self.assertEqual(found["linux/y"].site, "systems") self.assertEqual(found["hungary/x/pdfs/a.pdf"].kind, "asset") self.assertEqual(found["hungary/x/solutions/s.txt"].site, "languages") self.assertEqual(found["anime/vault.php"].kind, "dynamic") self.assertIsNone(found["anime/vault.php"].site) self.assertEqual(found["css/site.css"].kind, "static") self.assertEqual(found["resources/japanese/kanji/kanji_水.html"].kind, "static") # a bare .html is not a page address self.assertNotIn("_astro/app.css", found) # the old build's own files are skipped self.assertIsNone(found[""].site) # the old front page belongs to no site class VerifyTests(TestCase): def setUp(self): make("hungary/x/a.html") make("hungary/new/b.html") Redirect.objects.create(site="languages", old_path="hungary/old/b", new_site="languages", new_path="hungary/new/b") Redirect.objects.create(site="languages", old_path="hungary/old/gone", new_site="languages", new_path="hungary/nowhere") self.root = old_tree(self, {"hungary/x/index.html": "", "hungary/old/b/index.html": "", "hungary/old/gone/index.html": "", "hungary/lost/index.html": "", "hungary/x/pdfs/a.pdf": "", "hungary/x/pdfs/missing.pdf": "", "anime/index.php": ""}) self.assets = Path(old_tree(self, {"languages/hungary/x/pdfs/a.pdf": "pdf"})) def test_same_redirected_and_missing_are_told_apart(self): counts, missing = verify(scan_old_site(self.root), asset_root=self.assets) self.assertEqual(dict(counts["page"]), {"same": 1, "redirected": 1, "missing": 2}) paths = {path for _, _, path in missing} self.assertIn("hungary/lost", paths) # no redirect at all self.assertIn("hungary/old/gone", paths) # a redirect to a page that does not exist is NOT success self.assertEqual(dict(counts["asset"]), {"same": 1, "missing": 1}) def test_the_command_fails_while_anything_is_missing_and_lists_it(self): listing = Path(tempfile.mkdtemp()) / "missing.tsv" with self.assertRaises(CommandError): call_command("verify_old_urls", self.root, list=str(listing)) self.assertIn("hungary/lost", listing.read_text(encoding="utf-8")) class ProposalTests(TestCase): def setUp(self): make("hungary/hungarian-lessons/lesson_one.html") make("a/b/shared_name.html") make("c/d/shared_name.html") make("hungary/hungarian-lessons/page_two.html") def test_one_file_name_match_is_a_proposal(self): proposals, ambiguous, nothing = redirects.propose([("languages", "hungary/hungarian-language/lesson_one")]) self.assertEqual(proposals, [("languages", "hungary/hungarian-language/lesson_one", "languages", "hungary/hungarian-lessons/lesson_one", "same file name")]) def test_case_and_dashes_do_not_matter(self): proposals, _, _ = redirects.propose([("languages", "old/Lesson-One")]) self.assertEqual(len(proposals), 1) def test_several_matches_are_left_for_a_person(self): proposals, ambiguous, _ = redirects.propose([("languages", "old/shared_name")]) self.assertEqual(proposals, []) self.assertEqual(sorted(ambiguous[0][2]), ["a/b/shared_name", "c/d/shared_name"]) def test_no_match_is_reported_not_guessed(self): _, _, nothing = redirects.propose([("languages", "old/unknown_page")]) self.assertEqual(nothing, [("languages", "old/unknown_page")]) def test_a_print_version_points_at_the_page_itself(self): proposals, _, _ = redirects.propose([("languages", "old/page_two_print")]) self.assertEqual(proposals[0][3:], ("hungary/hungarian-lessons/page_two", "was the print version of this page")) def test_a_real_page_name_ending_in_print_is_matched_as_itself(self): make("x/blueprint_print.html") proposals, _, _ = redirects.propose([("languages", "old/blueprint_print")]) self.assertEqual(proposals[0][3:], ("x/blueprint_print", "same file name")) def test_only_rows_marked_yes_are_loaded(self): proposals, _, _ = redirects.propose([("languages", "old/lesson_one"), ("languages", "old/page_two")]) path = Path(tempfile.mkdtemp()) / "r.csv" redirects.write_csv(path, proposals) self.assertEqual(redirects.load_csv(path), (0, 0, 2)) # nothing reviewed yet: nothing loaded text = path.read_text(encoding="utf-8").replace("same file name,", "same file name,yes", 1) path.write_text(text, encoding="utf-8") self.assertEqual(redirects.load_csv(path), (1, 0, 1)) self.assertEqual(redirects.load_csv(path), (0, 1, 1)) # loading again updates, never duplicates self.assertEqual(Redirect.objects.count(), 1) class RedirectViewTests(TestCase): def setUp(self): make("hungary/new/b.html") make("linux/new/c.html", site="systems") Redirect.objects.create(site="languages", old_path="hungary/old/b", new_site="languages", new_path="hungary/new/b") Redirect.objects.create(site="languages", old_path="football/old", new_site="systems", new_path="linux/new/c") self.client = Client(HTTP_HOST="languages.localhost") def test_an_old_address_gets_a_permanent_redirect_on_the_same_site(self): response = self.client.get("/hungary/old/b/") self.assertEqual((response.status_code, response["Location"]), (301, "/hungary/new/b/")) def test_a_page_that_moved_to_another_site_gets_that_sites_address(self): # the old folder must be one the languages site routes, so use a folder it owns Redirect.objects.create(site="languages", old_path="japan/old", new_site="systems", new_path="linux/new/c") response = self.client.get("/japan/old/") self.assertEqual(response.status_code, 301) self.assertEqual(response["Location"], "http://systems.localhost:8000/linux/new/c/") def test_a_real_page_is_never_replaced_by_a_redirect(self): Redirect.objects.create(site="languages", old_path="hungary/new/b", new_site="languages", new_path="hungary/elsewhere") self.assertEqual(self.client.get("/hungary/new/b/").status_code, 200) def test_an_address_with_no_redirect_is_still_a_404(self): self.assertEqual(self.client.get("/hungary/old/nothing/").status_code, 404) def test_a_redirect_belongs_to_one_site(self): self.assertEqual(Client(HTTP_HOST="systems.localhost").get("/hungary/old/b/").status_code, 404) def test_one_redirect_per_old_address(self): with self.assertRaises(IntegrityError): Redirect.objects.create(site="languages", old_path="hungary/old/b", new_site="languages", new_path="x") def test_old_address_forms_are_normalised(self): for form in ("/hungary/old/b/index.html", "/hungary/old/b.html"): self.assertEqual(redirects.bare_path(form), "hungary/old/b") class CutoverTests(SimpleTestCase): def test_every_site_gets_a_rule_that_names_its_own_host(self): for site in SITES: rule = cutover.rule_for(site) self.assertTrue(rule.startswith("RedirectMatch 301 ^/(")) self.assertTrue(rule.endswith(f" https://{site}.osztromok.com/$1$2")) def test_a_rule_matches_its_own_folders_and_nothing_longer_by_accident(self): pattern = cutover.pattern_for("systems") for url in ("/linux", "/linux/", "/linux/x/y/", "/sidebar/linux/cheat_sheet_vim/", "/windows/a.pdf"): self.assertTrue(pattern.match(url), url) for url in ("/linuxfoo/", "/sidebar/", "/sidebar/football/x/", "/hungary/x/", "/"): self.assertFalse(pattern.match(url), url) def test_no_address_is_claimed_by_two_sites(self): patterns = {site: cutover.pattern_for(site) for site in SITES} for site, info in SITES.items(): for folder in info["folders"]: owners = [s for s, p in patterns.items() if p.match(f"/{folder}/x/")] self.assertEqual(owners, [site], folder) def test_sidebar_subjects_follow_their_site(self): self.assertEqual([s for s in SITES if cutover.pattern_for(s).match("/sidebar/football/x/")], ["humanities"]) self.assertEqual([s for s in SITES if cutover.pattern_for(s).match("/resources/japanese/kanji/x.html")], ["languages"]) def test_the_command_prints_one_line_per_site_and_rejects_a_typo(self): from io import StringIO out = StringIO() call_command("apache_cutover", "languages", "systems", stdout=out) self.assertEqual(len(out.getvalue().strip().splitlines()), 2) with self.assertRaises(CommandError): call_command("apache_cutover", "languagez") @override_settings(SITE_URL_TEMPLATE="https://{site}.example.org") class CrossLinkTests(SimpleTestCase): def test_a_link_into_another_site_gets_that_sites_address(self): html = 'x' self.assertEqual(rewrite_cross_site_links(html, "languages"), 'x') def test_a_link_inside_the_site_is_left_alone(self): html = 'x y' self.assertEqual(rewrite_cross_site_links(html, "languages"), html) def test_other_kinds_of_link_are_left_alone(self): html = ('ab' 'cdef' 'g') self.assertEqual(rewrite_cross_site_links(html, "languages"), html) def test_query_and_fragment_survive(self): out = rewrite_cross_site_links('x', "languages") self.assertEqual(out, 'x') def test_single_quotes_work_too(self): self.assertIn("https://systems.example.org/linux/", rewrite_cross_site_links("x", "ai")) def test_sidebar_pages_follow_their_subject(self): self.assertEqual(owner_of("/sidebar/football/x/"), "humanities") self.assertEqual(owner_of("/sidebar/ai/x/"), "ai") self.assertIsNone(owner_of("/sidebar/")) self.assertEqual(owner_of("/resources/japanese/kanji/kanji_%E6%B0%B4.html"), "languages") def test_a_page_without_links_is_returned_untouched(self): html = "no links, and no href at all
" self.assertIs(rewrite_cross_site_links(html, "languages"), html) class PageHtmlTests(TestCase): def test_the_stored_fragment_is_not_changed_only_what_is_shown(self): page = make("hungary/x/a.html", fragment='x') self.assertIn("systems.localhost", page.html) self.assertEqual(Page.objects.get(pk=page.pk).fragment, 'x') def test_the_shown_page_carries_the_rewritten_link(self): make("hungary/x/a.html", fragment='x') html = Client(HTTP_HOST="languages.localhost").get("/hungary/x/a/").content.decode() self.assertIn('href="http://systems.localhost:8000/linux/a/"', html) class BrokenLinkTests(TestCase): def setUp(self): self.assets = Path(old_tree(self, {"languages/hungary/x/solutions/s.txt": "s"})) Course.objects.create(site="languages", folder="hungary/x", name="X") make("hungary/x/a.html") Redirect.objects.create(site="languages", old_path="hungary/old/b", new_site="languages", new_path="hungary/x/a") def broken(self, fragment): Page.objects.filter(path="hungary/x/host.html").delete() make("hungary/x/host.html", fragment=fragment) checked, broken = links.broken_links(asset_root=self.assets) return checked, [(link, why) for _, link, why in broken] def test_good_links_are_not_reported(self): good = ('pfoldermoved' 'filese') checked, broken = self.broken(good) self.assertEqual(broken, []) self.assertEqual(checked, 5) # the https link is not an internal link def test_dead_links_are_reported_with_a_reason(self): _, broken = self.broken('ab') self.assertEqual(broken, [("/hungary/x/missing/", "no such page"), ("/hungary/x/solutions/none.txt", "file not found")]) def test_old_style_html_links_are_understood(self): _, broken = self.broken('ab') self.assertEqual(broken, []) def test_a_link_to_another_site_is_not_broken_because_it_is_rewritten_when_shown(self): make("linux/y/z.html", site="systems") _, broken = self.broken('a') self.assertEqual(broken, []) Test run (whole project): Found 280 test(s). System check identified no issues (0 silenced). Old site: 2 asset, 1 dynamic, 4 page No new site owns 0 page/asset addresses (decide: keep on the old host, or drop). pages owned by a site: 4 2 missing, 1 redirected, 1 same assets owned by a site: 2 2 missing Missing addresses written to C:\Users\emuba\AppData\Local\Temp\tmph45a5ylj\missing.tsv Creating test database for alias 'default'... ........................................................................................................................................................................................................................................................................................ ---------------------------------------------------------------------- Ran 280 tests in 74.792s OK Destroying test database for alias 'default'... WHY THIS WORKS AS AN ANSWER --------------------------- Each part reports the real numbers and its own limits, and the cut-over rule is tested against the actual list of old addresses instead of a handful of examples.