244 lines
7.9 KiB
Python
244 lines
7.9 KiB
Python
#!/usr/bin/env python3
|
|
"""Build the blog: every topics/**/*.md becomes its own page at <slug>/,
|
|
index.html lists them all with a short teaser, and feed.xml is an RSS feed
|
|
with every post in full.
|
|
|
|
Each post may start with a small header:
|
|
|
|
---
|
|
title: My first post
|
|
date: 2026-10-03
|
|
description: One or two sentences for the teaser and search results.
|
|
---
|
|
|
|
Without it, the title comes from the first "# Heading" (or the file name),
|
|
the date from a YYYY-MM-DD file-name prefix, if any, and the description
|
|
from the first paragraph. A post in a subfolder (topics/python/foo.md) is
|
|
tagged with that folder's name.
|
|
|
|
Usage: python3 build.py [source_dir] [output_dir]
|
|
"""
|
|
import html
|
|
import re
|
|
import shutil
|
|
import sys
|
|
from datetime import date, datetime, time, timezone
|
|
from email.utils import format_datetime
|
|
from pathlib import Path
|
|
|
|
import markdown
|
|
|
|
SRC = Path(sys.argv[1] if len(sys.argv) > 1 else ".").resolve()
|
|
OUT = Path(sys.argv[2] if len(sys.argv) > 2 else "public").resolve()
|
|
TOPICS = SRC / "topics"
|
|
SITE_URL = "https://pascal-maas.nl"
|
|
SITE_NAME = "Pascal Maas"
|
|
SITE_DESCRIPTION = "A blog about bioinformatics in metabolomics"
|
|
|
|
|
|
def slugify(text):
|
|
return re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-") or "post"
|
|
|
|
|
|
def parse(path):
|
|
text = path.read_text(encoding="utf-8")
|
|
meta = {}
|
|
m = re.match(r"^---\s*\n(.*?)\n---\s*\n", text, re.S)
|
|
if m:
|
|
for line in m.group(1).splitlines():
|
|
key, sep, value = line.partition(":")
|
|
if sep:
|
|
meta[key.strip().lower()] = value.strip().strip("\"'")
|
|
text = text[m.end():]
|
|
|
|
title = meta.get("title")
|
|
if not title:
|
|
h1 = re.match(r"^\s*#\s+(.+)\n", text)
|
|
if h1:
|
|
title = h1.group(1).strip()
|
|
text = text[h1.end():] # don't render the title twice
|
|
else:
|
|
title = re.sub(r"^\d{4}-\d{2}-\d{2}-", "", path.stem).replace("-", " ").capitalize()
|
|
|
|
raw_date = meta.get("date") or (re.match(r"^\d{4}-\d{2}-\d{2}", path.stem) or [None])[0]
|
|
try:
|
|
when = date.fromisoformat(raw_date) if raw_date else None
|
|
except ValueError:
|
|
sys.exit(f"{path.relative_to(SRC)}: date '{raw_date}' is not YYYY-MM-DD")
|
|
|
|
rel = path.relative_to(TOPICS)
|
|
body = markdown.markdown(
|
|
text,
|
|
extensions=[
|
|
"fenced_code", "tables", "footnotes", "codehilite", "abbr",
|
|
],
|
|
extension_configs={"codehilite": {"guess_lang": False}},
|
|
)
|
|
return {
|
|
"title": title,
|
|
"date": when,
|
|
"topic": rel.parts[0] if len(rel.parts) > 1 else meta.get("topic"),
|
|
"slug": slugify(meta.get("slug") or re.sub(r"^\d{4}-\d{2}-\d{2}-", "", path.stem)),
|
|
"summary": meta.get("description") or first_paragraph(body),
|
|
"body": body,
|
|
}
|
|
|
|
|
|
def first_paragraph(body):
|
|
"""Plain text of the first paragraph, without footnote markers."""
|
|
match = re.search(r"<p>(.*?)</p>", body, re.S)
|
|
if not match:
|
|
return ""
|
|
text = re.sub(r"<sup.*?</sup>", "", match.group(1), flags=re.S)
|
|
text = html.unescape(re.sub(r"<[^>]+>", "", text))
|
|
return " ".join(text.split())
|
|
|
|
|
|
def load_posts():
|
|
"""All posts, newest first (undated last), with unique slugs."""
|
|
posts = [parse(p) for p in sorted(TOPICS.rglob("*.md"))]
|
|
posts.sort(
|
|
key=lambda p: (p["date"] is not None, p["date"] or date.min),
|
|
reverse=True,
|
|
)
|
|
seen = {}
|
|
for p in posts: # keep page folders unique
|
|
n = seen.get(p["slug"], 0)
|
|
seen[p["slug"]] = n + 1
|
|
if n:
|
|
p["slug"] += f"-{n + 1}"
|
|
return posts
|
|
|
|
|
|
def time_tag(post):
|
|
if not post["date"]:
|
|
return ""
|
|
return (
|
|
f'<time datetime="{post["date"].isoformat()}">'
|
|
f'{post["date"]:%d %B %Y}</time>'
|
|
)
|
|
|
|
|
|
def meta_line(post):
|
|
tag = ""
|
|
if post["topic"]:
|
|
tag = f'<span class="tag">{html.escape(post["topic"])}</span>'
|
|
return f'<p class="meta">{time_tag(post)}{tag}</p>'
|
|
|
|
|
|
def render_toc(posts, current, root):
|
|
"""The side list; the post being shown (if any) is marked current."""
|
|
items = []
|
|
for post in posts:
|
|
css = ' class="current"' if post is current else ""
|
|
items.append(
|
|
f'<li><a href="{root}{post["slug"]}/"{css}>'
|
|
f'{html.escape(post["title"])}</a>{time_tag(post)}</li>'
|
|
)
|
|
return "\n".join(items) or "<li>No posts yet.</li>"
|
|
|
|
|
|
def index_content(posts):
|
|
teasers = []
|
|
for post in posts:
|
|
link = f'./{post["slug"]}/'
|
|
teasers.append(
|
|
f'<article>\n<header><h2><a href="{link}">'
|
|
f'{html.escape(post["title"])}</a></h2>{meta_line(post)}'
|
|
f'</header>\n<p>{html.escape(post["summary"])}</p>\n'
|
|
f'<p><a href="{link}">Read more</a></p>\n</article>'
|
|
)
|
|
return "\n\n".join(teasers)
|
|
|
|
|
|
def rebase_static(body, prefix):
|
|
"""Put prefix in front of the static/ paths in a post body."""
|
|
return re.sub(r'(src|href)="static/', rf'\1="{prefix}static/', body)
|
|
|
|
|
|
def post_content(post):
|
|
# Post pages sit one folder deeper, so static/ paths need ../ in front.
|
|
body = rebase_static(post["body"], "../")
|
|
return (
|
|
f'<article>\n<header><h2>{html.escape(post["title"])}</h2>'
|
|
f'{meta_line(post)}</header>\n{body}\n</article>'
|
|
)
|
|
|
|
|
|
def write_page(folder, template, posts, current, title, description, content):
|
|
"""Fill the template for one page and write it as folder/index.html."""
|
|
root = "./" if current is None else "../"
|
|
fields = {
|
|
"title": html.escape(title),
|
|
"description": html.escape(description),
|
|
"root": root,
|
|
"toc": render_toc(posts, current, root),
|
|
"content": content,
|
|
"year": str(date.today().year),
|
|
}
|
|
for key, value in fields.items():
|
|
template = template.replace("{{" + key + "}}", value)
|
|
folder.mkdir(parents=True, exist_ok=True)
|
|
(folder / "index.html").write_text(template, encoding="utf-8")
|
|
|
|
|
|
def feed_item(post):
|
|
link = f'{SITE_URL}/{post["slug"]}/'
|
|
# Feed readers show posts away from the site, so paths must be full.
|
|
body = rebase_static(post["body"], f"{SITE_URL}/")
|
|
lines = [
|
|
"<item>",
|
|
f'<title>{html.escape(post["title"])}</title>',
|
|
f"<link>{link}</link>",
|
|
f'<guid isPermaLink="true">{link}</guid>',
|
|
]
|
|
if post["date"]:
|
|
moment = datetime.combine(post["date"], time(), timezone.utc)
|
|
lines.append(f"<pubDate>{format_datetime(moment)}</pubDate>")
|
|
lines.append(f"<description>{html.escape(body)}</description>")
|
|
lines.append("</item>")
|
|
return "\n".join(lines)
|
|
|
|
|
|
def write_feed(posts):
|
|
"""feed.xml: an RSS feed with every post in full, newest first."""
|
|
items = "\n".join(feed_item(post) for post in posts)
|
|
feed = (
|
|
'<?xml version="1.0" encoding="utf-8"?>\n'
|
|
'<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">\n'
|
|
"<channel>\n"
|
|
f"<title>{html.escape(SITE_NAME)}</title>\n"
|
|
f"<link>{SITE_URL}/</link>\n"
|
|
f"<description>{html.escape(SITE_DESCRIPTION)}</description>\n"
|
|
"<language>en</language>\n"
|
|
f'<atom:link href="{SITE_URL}/feed.xml" rel="self" '
|
|
'type="application/rss+xml"/>\n'
|
|
f"{items}\n</channel>\n</rss>\n"
|
|
)
|
|
(OUT / "feed.xml").write_text(feed, encoding="utf-8")
|
|
|
|
|
|
def main():
|
|
posts = load_posts()
|
|
template = (SRC / "template.html").read_text(encoding="utf-8")
|
|
if OUT.exists():
|
|
shutil.rmtree(OUT)
|
|
write_page(
|
|
OUT, template, posts, None, f"{SITE_NAME} · Blog", SITE_DESCRIPTION,
|
|
index_content(posts),
|
|
)
|
|
for post in posts:
|
|
write_page(
|
|
OUT / post["slug"], template, posts, post,
|
|
f'{post["title"]} · {SITE_NAME}', post["summary"],
|
|
post_content(post),
|
|
)
|
|
write_feed(posts)
|
|
if (SRC / "static").is_dir(): # images etc.: topics can link to static/foo.png
|
|
shutil.copytree(SRC / "static", OUT / "static")
|
|
print(f"Built {len(posts)} post(s) into {OUT}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|