"""Post-migration checks. Run after `hugo --gc --minify`. 1. Inventory - every published WordPress page/post has a content file. 2. Media - every attachment is either placed in the site or explicitly accounted for as an unused media-library leftover. 3. Links - no internal href/src in public/ points at a missing file, and no absolute wp-content URL survived the import. Exits non-zero if any check fails. """ from __future__ import annotations import re import sys from pathlib import Path from urllib.parse import unquote, urlsplit from wp2hugo import PAGE_MAP, SKIP, SLUG_FIXES from wxr import REPO_ROOT, clean_media_name, load_items, slugify PUBLIC = REPO_ROOT / "public" CONTENT = REPO_ROOT / "content" failures: list[str] = [] notes: list[str] = [] def check_inventory(items) -> None: expected = 0 for item in items.values(): if item.post_id in SKIP or item.status != "publish": continue if item.post_type == "page": mapping = PAGE_MAP.get(item.post_id) if not mapping: failures.append(f"page {item.post_id} ({item.slug}) is unmapped") continue path = CONTENT / mapping[0] elif item.post_type == "post": path = CONTENT / "aktuelles" / SLUG_FIXES.get( item.slug, slugify(item.slug)) / "index.md" else: continue expected += 1 if not path.exists(): failures.append(f"missing content file for {item.post_type} " f"{item.post_id} ({item.slug}): {path}") notes.append(f"inventory: {expected} published pages/posts accounted for") for post_id, reason in SKIP.items(): notes.append(f" skipped {post_id}: {reason}") def check_media(items) -> None: placed = {p.name for p in CONTENT.rglob("*") if p.is_file() and p.suffix.lower() != ".md"} placed |= {p.name for p in (REPO_ROOT / "assets" / "img").rglob("*") if p.is_file()} placed |= {p.name for p in (REPO_ROOT / "static").rglob("*") if p.is_file()} # assets/img/logo.png and banner.jpg are renamed copies. placed |= {"Emblem_transparent.png", "cropped-Banner_small_web.jpg"} unused = [] for item in items.values(): if item.post_type != "attachment" or not item.attachment_url: continue original = item.attachment_url.rsplit("/", 1)[-1] if original in placed or clean_media_name(original) in placed: continue unused.append(item.attachment_url) total = sum(1 for i in items.values() if i.post_type == "attachment") notes.append(f"media: {total - len(unused)}/{total} attachments placed") if unused: notes.append(f" {len(unused)} media-library leftovers not shown on the " f"live site (see MIGRATION.md)") HREF = re.compile(r'(?:href|src)=["\']?([^"\'\s>]+)', re.I) def check_links() -> None: if not PUBLIC.exists(): failures.append("public/ does not exist - run `hugo` first") return checked = broken = 0 for html in PUBLIC.rglob("*.html"): text = html.read_text(encoding="utf-8", errors="replace") for stale in re.findall(r'https?://motorradclub-giebelwald\.de/wp-content/\S*', text): failures.append(f"{html.relative_to(PUBLIC)}: stale WordPress URL {stale}") for raw in HREF.findall(text): url = urlsplit(raw) if url.scheme or url.netloc or not url.path: continue # external, mailto:, protocol-relative if raw.startswith("#"): continue checked += 1 path = unquote(url.path) target = (PUBLIC / path.lstrip("/") if path.startswith("/") else html.parent / path) if target.is_dir(): target = target / "index.html" if not target.exists(): broken += 1 failures.append(f"{html.relative_to(PUBLIC)}: broken link -> {raw}") notes.append(f"links: {checked} internal references checked, {broken} broken") def main() -> int: items = load_items() check_inventory(items) check_media(items) check_links() for note in notes: print(note) if failures: print(f"\n{len(failures)} FAILURE(S):", file=sys.stderr) for f in failures: print(f" - {f}", file=sys.stderr) return 1 print("\nall checks passed") return 0 if __name__ == "__main__": raise SystemExit(main())