Files
blog/build.py
Pascal Maas a1fb581d21
All checks were successful
Build blog / build (push) Successful in 0s
Add blog template, build script and CI workflow
2026-10-03 14:50:40 +02:00

111 lines
3.7 KiB
Python

#!/usr/bin/env python3
"""Build the blog: every topics/**/*.md becomes a section in one index.html.
Each post may start with a small header:
---
title: My first post
date: 2026-10-03
---
Without it, the title comes from the first "# Heading" (or the file name)
and the date from a YYYY-MM-DD file-name prefix, if any. A post in a
subfolder (topics/python/foo.md) is tagged with that folder's name.
Usage: python3 build.py [source_dir] [output_dir]
"""
import html
import re
import shutil
import sys
from datetime import date
from pathlib import Path
import markdown
SRC = Path(sys.argv[1] if len(sys.argv) > 1 else ".").resolve()
OUT = Path(sys.argv[2] if len(sys.argv) > 2 else "public").resolve()
TOPICS = SRC / "topics"
def slugify(text):
return re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-") or "post"
def parse(path):
text = path.read_text(encoding="utf-8")
meta = {}
m = re.match(r"^---\s*\n(.*?)\n---\s*\n", text, re.S)
if m:
for line in m.group(1).splitlines():
key, sep, value = line.partition(":")
if sep:
meta[key.strip().lower()] = value.strip().strip("\"'")
text = text[m.end():]
title = meta.get("title")
if not title:
h1 = re.match(r"^\s*#\s+(.+)\n", text)
if h1:
title = h1.group(1).strip()
text = text[h1.end():] # don't render the title twice
else:
title = re.sub(r"^\d{4}-\d{2}-\d{2}-", "", path.stem).replace("-", " ").capitalize()
raw_date = meta.get("date") or (re.match(r"^\d{4}-\d{2}-\d{2}", path.stem) or [None])[0]
try:
when = date.fromisoformat(raw_date) if raw_date else None
except ValueError:
sys.exit(f"{path.relative_to(SRC)}: date '{raw_date}' is not YYYY-MM-DD")
rel = path.relative_to(TOPICS)
return {
"title": title,
"date": when,
"topic": rel.parts[0] if len(rel.parts) > 1 else meta.get("topic"),
"slug": slugify(meta.get("slug") or re.sub(r"^\d{4}-\d{2}-\d{2}-", "", path.stem)),
"body": markdown.markdown(text, extensions=["fenced_code", "tables", "footnotes"]),
}
def main():
posts = [parse(p) for p in sorted(TOPICS.rglob("*.md"))]
# Newest first; undated posts go last.
posts.sort(key=lambda p: (p["date"] is not None, p["date"] or date.min), reverse=True)
seen = {}
for p in posts: # keep anchors unique
n = seen.get(p["slug"], 0)
seen[p["slug"]] = n + 1
if n:
p["slug"] += f"-{n + 1}"
toc, sections = [], []
for p in posts:
title = html.escape(p["title"])
when = f'<time datetime="{p["date"].isoformat()}">{p["date"]:%d %B %Y}</time>' if p["date"] else ""
tag = f'<span class="tag">{html.escape(p["topic"])}</span>' if p["topic"] else ""
toc.append(f'<li><a href="#{p["slug"]}">{title}</a>{when}</li>')
sections.append(
f'<article id="{p["slug"]}">\n'
f'<header><h2><a href="#{p["slug"]}">{title}</a></h2><p class="meta">{when}{tag}</p></header>\n'
f'{p["body"]}\n</article>'
)
page = (SRC / "template.html").read_text(encoding="utf-8")
page = page.replace("{{toc}}", "\n".join(toc) or "<li>No posts yet.</li>")
page = page.replace("{{posts}}", "\n\n".join(sections))
page = page.replace("{{year}}", str(date.today().year))
if OUT.exists():
shutil.rmtree(OUT)
OUT.mkdir(parents=True)
(OUT / "index.html").write_text(page, encoding="utf-8")
if (SRC / "static").is_dir(): # images etc.: topics can link to static/foo.png
shutil.copytree(SRC / "static", OUT / "static")
print(f"Built {len(posts)} post(s) into {OUT}")
if __name__ == "__main__":
main()