Files
mcg-website/tools/html2md.py
T
luxick bef5182545 Migrate WordPress site to Hugo
Replaces the WordPress 7.0.2 / HitMag site with a static Hugo build,
deployed by Gitea Actions over rsync/SSH.

Content: 26 files (9 pages + 17 posts) as Markdown page bundles, plus
127 of 165 media attachments. The remaining 38 are media-library
leftovers that appear nowhere on the live site; they are listed in
MIGRATION.md.

The WXR export contains neither media binaries nor widgets, so both were
recovered from the live host before it is retired:
  - tools/fetch_media.py downloads all uploads, capping them at 2000px
  - tools/build_data.py scrapes the "Unsere Mitglieder" and
    "In Erinnerung" widgets into data/*.yaml

TablePress' six Termine tables became data/termine.yaml, rendered with
the newest year expanded and earlier years collapsed. Old permalinks are
not preserved (clean slugs, no redirects, as agreed).

Custom layouts, no third-party theme. Plain CSS, since the pinned Hugo
0.164.0 is the non-extended build and cannot compile Sass.

Verified: clean build with zero warnings, 718 internal links checked and
none broken, no surviving wp-content URLs.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-07-28 19:20:11 +02:00

172 lines
6.5 KiB
Python
Raw Blame History

This file contains invisible Unicode characters
This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Minimal HTML -> Markdown converter for the imported WordPress content.
Deliberately narrow: it only handles the tags that actually occur in this export
(paragraphs, headings, lists, emphasis, links, images, breaks, rules). Anything
unrecognised is passed through as raw HTML, which Goldmark renders because
``markup.goldmark.renderer.unsafe`` is on.
"""
from __future__ import annotations
import re
from html.parser import HTMLParser
INLINE = {"strong", "b", "em", "i", "a", "br", "span", "code", "u", "s"}
BLOCK = {"p", "h1", "h2", "h3", "h4", "h5", "h6", "ul", "ol", "li", "hr",
"blockquote", "figure", "figcaption", "div"}
ESCAPE = re.compile(r"([\\`*_\[\]])")
class Converter(HTMLParser):
def __init__(self, resolve_image=None):
super().__init__(convert_charrefs=True)
self.out: list[tuple[str, str]] = []
self.resolve_image = resolve_image or (lambda src, cls: src)
self._stack: list[str] = []
self._list: list[str] = [] # 'ul' / 'ol' nesting
self._li_index: list[int] = []
self._link: str | None = None
self._buf: list[str] = []
# -- helpers ---------------------------------------------------------
def _emit(self, text: str) -> None:
self._buf.append(text)
def _flush_block(self, prefix: str = "", kind: str = "block") -> None:
text = "".join(self._buf).strip()
self._buf = []
if text:
self.out.append((kind, prefix + text))
def _open_emphasis(self, marker: str) -> None:
# Emitted lazily: WordPress often writes "<strong> 57258 ...</strong>",
# and "** 57258**" is not valid emphasis in CommonMark. The marker is
# placed after any leading whitespace instead (see handle_data).
self._pending_emphasis = marker
def _close_emphasis(self, marker: str) -> None:
if getattr(self, "_pending_emphasis", None) == marker:
self._pending_emphasis = None # empty <strong></strong>
return
# Likewise move trailing whitespace out: "**Samstag **" -> "**Samstag** ".
trailing = ""
while self._buf and self._buf[-1] and self._buf[-1][-1] in " \t\n":
trailing = self._buf[-1][-1] + trailing
self._buf[-1] = self._buf[-1][:-1]
if not self._buf[-1]:
self._buf.pop()
if not self._buf:
return
self._emit(marker)
if trailing:
self._emit(trailing)
# -- tags ------------------------------------------------------------
def handle_starttag(self, tag, attrs):
a = dict(attrs)
if tag == "p":
self._flush_block()
elif tag in ("h1", "h2", "h3", "h4", "h5", "h6"):
self._flush_block()
self._pending_prefix = "#" * int(tag[1]) + " "
elif tag in ("ul", "ol"):
self._flush_block()
self._list.append(tag)
self._li_index.append(0)
elif tag == "li":
self._flush_block()
if self._list:
self._li_index[-1] += 1
indent = " " * (len(self._list) - 1)
marker = ("- " if self._list[-1] == "ul"
else f"{self._li_index[-1]}. ")
self._pending_prefix = indent + marker
elif tag == "hr":
self._flush_block()
self.out.append(("block", "---"))
elif tag == "br":
self._emit(" \n")
elif tag in ("strong", "b"):
self._open_emphasis("**")
elif tag in ("em", "i"):
self._open_emphasis("*")
elif tag == "a":
self._link = a.get("href", "")
self._emit("[")
elif tag == "img":
src = self.resolve_image(a.get("src", ""), a.get("class", ""))
alt = (a.get("alt") or "").replace("]", "")
self._emit(f"![{alt}]({src})")
self._stack.append(tag)
def handle_endtag(self, tag):
while self._stack and self._stack.pop() != tag:
pass
if tag == "p":
self._flush_block()
elif tag in ("h1", "h2", "h3", "h4", "h5", "h6", "li"):
prefix = getattr(self, "_pending_prefix", "")
self._pending_prefix = ""
self._flush_block(prefix, kind="li" if tag == "li" else "block")
elif tag in ("ul", "ol"):
self._flush_block()
if self._list:
self._list.pop()
self._li_index.pop()
elif tag in ("strong", "b"):
self._close_emphasis("**")
elif tag in ("em", "i"):
self._close_emphasis("*")
elif tag == "a":
self._emit(f"]({self._link or ''})")
self._link = None
def handle_data(self, data):
if not data:
return
# Collapse WordPress' aggressive whitespace, but keep explicit breaks.
text = data.replace(" ", " ")
text = re.sub(r"[ \t]*\n[ \t]*", "\n", text)
pending = getattr(self, "_pending_emphasis", None)
if pending:
if not text.strip():
return # whitespace only; keep waiting for real content
lead = text[:len(text) - len(text.lstrip())]
if lead:
self._emit(lead)
text = text.lstrip()
self._emit(pending)
self._pending_emphasis = None
elif not self._buf and not text.strip():
return
self._emit(ESCAPE.sub(r"\\\1", text) if self._link is None
else text.replace("[", r"\[").replace("]", r"\]"))
def result(self) -> str:
self._flush_block(getattr(self, "_pending_prefix", ""))
blocks = [(kind, text.strip()) for kind, text in self.out if text.strip()]
out = ""
for index, (kind, text) in enumerate(blocks):
if index == 0:
out = text
continue
# Keep consecutive list items tight; a blank line between them makes
# Goldmark render a "loose" list, wrapping every item in <p>.
sep = "\n" if kind == "li" and blocks[index - 1][0] == "li" else "\n\n"
out += sep + text
return out
def html_to_markdown(html_text: str, resolve_image=None) -> str:
conv = Converter(resolve_image=resolve_image)
conv.feed(html_text)
conv.close()
text = conv.result()
# An escaped em-dash entity or stray double-escape looks worse than the source.
text = text.replace("\\_\\_", "__")
return re.sub(r"\n{3,}", "\n\n", text).strip()