Files
novoyuuparosk-auto-wiki/pipelines/songs/render.py
T
mikkeli df7195511f render: replace generated_at timestamp with commit SHA in banner
Timestamps made the idempotency check always false — every run would
rewrite every page. Commit SHA is stable: same source commit produces
identical wikitext, so unchanged pages are correctly skipped.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-09 18:01:46 +09:00

184 lines
5.6 KiB
Python

#!/usr/bin/env python3
"""Songs pipeline renderer — converts source .md files to MediaWiki pages."""
import argparse
import os
import subprocess
import sys
from datetime import datetime
from pathlib import Path
import frontmatter
import mwclient
def validate(fm: dict, path: Path) -> None:
errors = []
for field in ("title", "album"):
val = fm.get(field)
if not val or not str(val).strip():
errors.append(f"missing or empty '{field}'")
wiki = fm.get("wiki")
if not isinstance(wiki, dict) or "publish" not in wiki:
errors.append("missing 'wiki.publish'")
elif not isinstance(wiki["publish"], bool):
errors.append("'wiki.publish' must be a boolean (true/false), not a string")
if "release_date" in fm and fm["release_date"]:
try:
datetime.strptime(str(fm["release_date"]), "%Y-%m-%d")
except ValueError:
errors.append("'release_date' must be YYYY-MM-DD")
if errors:
raise ValueError(f"{path}: " + "; ".join(errors))
def strip_first_h1(text: str) -> str:
"""Remove the first '# Heading' line and any immediately following blank line."""
lines = text.split("\n")
for i, line in enumerate(lines):
if line.strip().startswith("# "):
del lines[i]
if i < len(lines) and lines[i].strip() == "":
del lines[i]
break
return "\n".join(lines)
def markdown_to_wikitext(body: str) -> str:
result = subprocess.run(
["pandoc", "-f", "markdown", "-t", "mediawiki"],
input=body,
capture_output=True,
text=True,
)
if result.returncode != 0:
raise RuntimeError(f"pandoc failed: {result.stderr.strip()}")
return result.stdout
def build_wikitext(fm: dict, body_wikitext: str, source_url: str, source_ref: str) -> str:
banner = f"{{{{Auto-generated|source={source_url}|commit={source_ref}}}}}"
category = f"[[Category:{fm['album']}]]"
return f"{banner}\n\n{body_wikitext}\n{category}\n"
def connect_wiki() -> mwclient.Site:
from urllib.parse import urlparse
api_url = os.environ["WIKI_API_URL"]
parsed = urlparse(api_url)
host = parsed.netloc
# Strip api.php to get the wiki root path (e.g. "/" or "/w/")
path = parsed.path[: parsed.path.rfind("/") + 1] or "/"
site = mwclient.Site(host, path=path, scheme=parsed.scheme)
site.login(os.environ["WIKI_BOT_USER"], os.environ["WIKI_BOT_PASSWORD"])
return site
def publish_one(rel: Path, post, site: mwclient.Site, source_ref: str, gitea_repo_url: str) -> str:
"""Returns 'published', 'noop', or 'skipped'."""
fm = post.metadata
if not fm["wiki"]["publish"]:
return "skipped"
body_wikitext = markdown_to_wikitext(strip_first_h1(post.content))
source_url = f"{gitea_repo_url}/src/commit/{source_ref}/{rel}"
page_content = build_wikitext(fm, body_wikitext, source_url, source_ref)
page = site.pages[fm["title"]]
if page.text() == page_content:
return "noop"
ref_short = source_ref[:8] if source_ref else "unknown"
page.save(page_content, summary=f"Auto-published from {ref_short} (songs pipeline)")
return "published"
def load_and_validate(files: list[Path], source_dir: Path) -> dict[Path, object]:
posts = {}
titles: dict[str, Path] = {}
for path in files:
if not path.exists():
print(f" warn: {path} not found, skipping", file=sys.stderr)
continue
post = frontmatter.load(str(path))
validate(post.metadata, path)
if post.metadata["wiki"]["publish"]:
title = str(post.metadata["title"])
if title in titles:
raise ValueError(
f"Duplicate wiki title '{title}': {path} and {titles[title]}"
)
titles[title] = path
posts[path] = post
return posts
def main() -> None:
parser = argparse.ArgumentParser(description="Publish song .md files to MediaWiki.")
parser.add_argument("--source-dir", required=True, type=Path)
parser.add_argument("--files", nargs="*", default=[], help="Relative paths within source-dir")
parser.add_argument("--all", action="store_true", help="Publish all publishable files")
args = parser.parse_args()
source_dir = args.source_dir.resolve()
source_ref = os.environ.get("SOURCE_REF", "")
gitea_repo_url = os.environ.get("GITEA_REPO_URL", "").rstrip("/")
if args.all:
candidates = list(source_dir.rglob("*.md"))
elif args.files:
candidates = [source_dir / f for f in args.files]
else:
print("Nothing to do: pass --files or --all.")
return
files = [
p for p in candidates
if p.exists()
and p.suffix == ".md"
and "wip" not in p.relative_to(source_dir).parts
and p.name != "README.md"
]
if not files:
print("No publishable candidates after filtering.")
return
print(f"Validating {len(files)} file(s)...")
posts = load_and_validate(files, source_dir)
print("Validation passed.")
site = connect_wiki()
counts = {"published": 0, "noop": 0, "skipped": 0}
for abs_path, post in posts.items():
rel = abs_path.relative_to(source_dir)
outcome = publish_one(rel, post, site, source_ref, gitea_repo_url)
counts[outcome] += 1
print(f" [{outcome}] {rel}")
print(
f"\nDone: {counts['published']} published, "
f"{counts['noop']} unchanged, "
f"{counts['skipped']} skipped (wiki.publish=false)."
)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"ERROR: {e}", file=sys.stderr)
sys.exit(1)