From b6bb7dd7892c66e6d8609827b8751dc5b81f005b Mon Sep 17 00:00:00 2001 From: Pascal Maas Date: Sun, 4 Oct 2026 17:58:43 +0200 Subject: [PATCH] added RSS and pages --- build.py | 193 +++++++++++++++++++++++++++++++------- template.html | 50 +++------- topics/starting-a-blog.md | 1 + 3 files changed, 176 insertions(+), 68 deletions(-) diff --git a/build.py b/build.py index 8e83cd8..7960e32 100644 --- a/build.py +++ b/build.py @@ -1,16 +1,20 @@ #!/usr/bin/env python3 -"""Build the blog: every topics/**/*.md becomes a section in one index.html. +"""Build the blog: every topics/**/*.md becomes its own page at /, +index.html lists them all with a short teaser, and feed.xml is an RSS feed +with every post in full. Each post may start with a small header: --- title: My first post date: 2026-10-03 + description: One or two sentences for the teaser and search results. --- -Without it, the title comes from the first "# Heading" (or the file name) -and the date from a YYYY-MM-DD file-name prefix, if any. A post in a -subfolder (topics/python/foo.md) is tagged with that folder's name. +Without it, the title comes from the first "# Heading" (or the file name), +the date from a YYYY-MM-DD file-name prefix, if any, and the description +from the first paragraph. A post in a subfolder (topics/python/foo.md) is +tagged with that folder's name. Usage: python3 build.py [source_dir] [output_dir] """ @@ -18,7 +22,8 @@ import html import re import shutil import sys -from datetime import date +from datetime import date, datetime, time, timezone +from email.utils import format_datetime from pathlib import Path import markdown @@ -26,6 +31,9 @@ import markdown SRC = Path(sys.argv[1] if len(sys.argv) > 1 else ".").resolve() OUT = Path(sys.argv[2] if len(sys.argv) > 2 else "public").resolve() TOPICS = SRC / "topics" +SITE_URL = "https://pascal-maas.nl" +SITE_NAME = "Pascal Maas" +SITE_DESCRIPTION = "A blog about bioinformatics in metabolomics" def slugify(text): @@ -59,54 +67,173 @@ def parse(path): sys.exit(f"{path.relative_to(SRC)}: date '{raw_date}' is not YYYY-MM-DD") rel = path.relative_to(TOPICS) + body = markdown.markdown( + text, + extensions=[ + "fenced_code", "tables", "footnotes", "codehilite", "abbr", + ], + extension_configs={"codehilite": {"guess_lang": False}}, + ) return { "title": title, "date": when, "topic": rel.parts[0] if len(rel.parts) > 1 else meta.get("topic"), "slug": slugify(meta.get("slug") or re.sub(r"^\d{4}-\d{2}-\d{2}-", "", path.stem)), - "body": markdown.markdown( - text, - extensions=[ - "fenced_code", "tables", "footnotes", "codehilite", "abbr", - ], - extension_configs={"codehilite": {"guess_lang": False}}, - ), + "summary": meta.get("description") or first_paragraph(body), + "body": body, } -def main(): - posts = [parse(p) for p in sorted(TOPICS.rglob("*.md"))] - # Newest first; undated posts go last. - posts.sort(key=lambda p: (p["date"] is not None, p["date"] or date.min), reverse=True) +def first_paragraph(body): + """Plain text of the first paragraph, without footnote markers.""" + match = re.search(r"

(.*?)

", body, re.S) + if not match: + return "" + text = re.sub(r"", "", match.group(1), flags=re.S) + text = html.unescape(re.sub(r"<[^>]+>", "", text)) + return " ".join(text.split()) + +def load_posts(): + """All posts, newest first (undated last), with unique slugs.""" + posts = [parse(p) for p in sorted(TOPICS.rglob("*.md"))] + posts.sort( + key=lambda p: (p["date"] is not None, p["date"] or date.min), + reverse=True, + ) seen = {} - for p in posts: # keep anchors unique + for p in posts: # keep page folders unique n = seen.get(p["slug"], 0) seen[p["slug"]] = n + 1 if n: p["slug"] += f"-{n + 1}" + return posts - toc, sections = [], [] - for p in posts: - title = html.escape(p["title"]) - when = f'' if p["date"] else "" - tag = f'{html.escape(p["topic"])}' if p["topic"] else "" - toc.append(f'
  • {title}{when}
  • ') - sections.append( - f'
    \n' - f'

    {title}

    {when}{tag}

    \n' - f'{p["body"]}\n
    ' + +def time_tag(post): + if not post["date"]: + return "" + return ( + f'' + ) + + +def meta_line(post): + tag = "" + if post["topic"]: + tag = f'{html.escape(post["topic"])}' + return f'

    {time_tag(post)}{tag}

    ' + + +def render_toc(posts, current, root): + """The side list; the post being shown (if any) is marked current.""" + items = [] + for post in posts: + css = ' class="current"' if post is current else "" + items.append( + f'
  • ' + f'{html.escape(post["title"])}{time_tag(post)}
  • ' ) + return "\n".join(items) or "
  • No posts yet.
  • " - page = (SRC / "template.html").read_text(encoding="utf-8") - page = page.replace("{{toc}}", "\n".join(toc) or "
  • No posts yet.
  • ") - page = page.replace("{{posts}}", "\n\n".join(sections)) - page = page.replace("{{year}}", str(date.today().year)) +def index_content(posts): + teasers = [] + for post in posts: + link = f'./{post["slug"]}/' + teasers.append( + f'' + ) + return "\n\n".join(teasers) + + +def rebase_static(body, prefix): + """Put prefix in front of the static/ paths in a post body.""" + return re.sub(r'(src|href)="static/', rf'\1="{prefix}static/', body) + + +def post_content(post): + # Post pages sit one folder deeper, so static/ paths need ../ in front. + body = rebase_static(post["body"], "../") + return ( + f'
    \n

    {html.escape(post["title"])}

    ' + f'{meta_line(post)}
    \n{body}\n
    ' + ) + + +def write_page(folder, template, posts, current, title, description, content): + """Fill the template for one page and write it as folder/index.html.""" + root = "./" if current is None else "../" + fields = { + "title": html.escape(title), + "description": html.escape(description), + "root": root, + "toc": render_toc(posts, current, root), + "content": content, + "year": str(date.today().year), + } + for key, value in fields.items(): + template = template.replace("{{" + key + "}}", value) + folder.mkdir(parents=True, exist_ok=True) + (folder / "index.html").write_text(template, encoding="utf-8") + + +def feed_item(post): + link = f'{SITE_URL}/{post["slug"]}/' + # Feed readers show posts away from the site, so paths must be full. + body = rebase_static(post["body"], f"{SITE_URL}/") + lines = [ + "", + f'{html.escape(post["title"])}', + f"{link}", + f'{link}', + ] + if post["date"]: + moment = datetime.combine(post["date"], time(), timezone.utc) + lines.append(f"{format_datetime(moment)}") + lines.append(f"{html.escape(body)}") + lines.append("") + return "\n".join(lines) + + +def write_feed(posts): + """feed.xml: an RSS feed with every post in full, newest first.""" + items = "\n".join(feed_item(post) for post in posts) + feed = ( + '\n' + '\n' + "\n" + f"{html.escape(SITE_NAME)}\n" + f"{SITE_URL}/\n" + f"{html.escape(SITE_DESCRIPTION)}\n" + "en\n" + f'\n' + f"{items}\n\n\n" + ) + (OUT / "feed.xml").write_text(feed, encoding="utf-8") + + +def main(): + posts = load_posts() + template = (SRC / "template.html").read_text(encoding="utf-8") if OUT.exists(): shutil.rmtree(OUT) - OUT.mkdir(parents=True) - (OUT / "index.html").write_text(page, encoding="utf-8") + write_page( + OUT, template, posts, None, f"{SITE_NAME} · Blog", SITE_DESCRIPTION, + index_content(posts), + ) + for post in posts: + write_page( + OUT / post["slug"], template, posts, post, + f'{post["title"]} · {SITE_NAME}', post["summary"], + post_content(post), + ) + write_feed(posts) if (SRC / "static").is_dir(): # images etc.: topics can link to static/foo.png shutil.copytree(SRC / "static", OUT / "static") print(f"Built {len(posts)} post(s) into {OUT}") diff --git a/template.html b/template.html index 54850e0..dd44b78 100644 --- a/template.html +++ b/template.html @@ -3,7 +3,12 @@ -Pascal Maas · Blog +{{title}} + + + + diff --git a/topics/starting-a-blog.md b/topics/starting-a-blog.md index 09d465c..cd754ac 100644 --- a/topics/starting-a-blog.md +++ b/topics/starting-a-blog.md @@ -1,5 +1,6 @@ --- title: Starting a blog +description: Hi there! This is my first post in hopefully many. I'm Pascal and work as a bioinformatician in the field of metabolomics. I studied bioinformatics, getting my master degree in 2020 at the University of Amsterdam and since then I work at the Metabolomics Analytical Centre, which is part of the University of Leiden. Here I develop software to support the data analysis of large volumes of samples. This however, remains a challenge for reasons that hopefully become clear through posts on this blog. Ofcourse, I will also attempt to provide solutions for practical challenges I've encounted. date: 2026-10-04 ---