#!/usr/bin/env python3 """Songs pipeline — validates, renders, and publishes song .md files to MediaWiki.""" import argparse import hashlib import io import os import subprocess import sys from datetime import datetime from pathlib import Path import frontmatter import mwclient def validate(fm: dict, path: Path) -> None: errors = [] for field in ("title", "album"): val = fm.get(field) if not val or not str(val).strip(): errors.append(f"missing or empty '{field}'") if not isinstance(fm.get("wiki", {}).get("publish"), bool): errors.append("'wiki.publish' must be a boolean (true/false), not a string") if "release_date" in fm and fm["release_date"]: try: datetime.strptime(str(fm["release_date"]), "%Y-%m-%d") except ValueError: errors.append("'release_date' must be YYYY-MM-DD") if errors: raise ValueError(f"{path}: " + "; ".join(errors)) def strip_first_h1(text: str) -> str: """Remove the first '# Heading' line and any immediately following blank line.""" lines = text.split("\n") for i, line in enumerate(lines): if line.strip().startswith("# "): del lines[i] if i < len(lines) and lines[i].strip() == "": del lines[i] break return "\n".join(lines) def markdown_to_wikitext(body: str) -> str: result = subprocess.run( ["pandoc", "-f", "markdown", "-t", "mediawiki"], input=body, capture_output=True, text=True, ) if result.returncode != 0: raise RuntimeError(f"pandoc failed: {result.stderr.strip()}") return result.stdout def build_wikitext(fm: dict, body_wikitext: str, source_url: str, source_ref: str, lrc_filename: str | None = None) -> str: banner = f"{{{{Auto-generated|source={source_url}|commit={source_ref}}}}}" category = f"[[Category:{fm['album']}]]" lrc_line = f"[[Media:{lrc_filename}|Synced lyrics (.lrc)]]\n" if lrc_filename else "" return f"{banner}\n\n{body_wikitext}{lrc_line}{category}\n" def connect_wiki() -> mwclient.Site: from urllib.parse import urlparse api_url = os.environ["WIKI_API_URL"] parsed = urlparse(api_url) host = parsed.netloc # Strip api.php to get the wiki root path (e.g. "/" or "/w/") path = parsed.path[: parsed.path.rfind("/") + 1] or "/" site = mwclient.Site(host, path=path, scheme=parsed.scheme) site.login(os.environ["WIKI_BOT_USER"], os.environ["WIKI_BOT_PASSWORD"]) return site def upload_lrc(lrc_path: Path, title: str, site: mwclient.Site) -> bool: """Upload LRC file to wiki if content has changed. Returns True if uploaded.""" if not lrc_path.exists(): print(f" warn: LRC file not found: {lrc_path}", file=sys.stderr) return False content = lrc_path.read_bytes() try: content.decode("utf-8") except UnicodeDecodeError: print(f" warn: {lrc_path} is not valid UTF-8 text, skipping LRC upload", file=sys.stderr) return False wiki_filename = f"{title}.lrc" local_sha1 = hashlib.sha1(content).hexdigest() if site.images[wiki_filename].imageinfo.get("sha1") == local_sha1: return False site.upload( file=io.BytesIO(content), filename=wiki_filename, description=f"Synced lyrics for [[{title}]] (auto-published)", ignore=True, ) return True AUTO_BANNER_PREFIX = "{{Auto-generated" def publish_one(abs_path: Path, source_dir: Path, post, site: mwclient.Site, source_ref: str, gitea_repo_url: str) -> str: """Returns 'published', 'noop', or 'skipped-manual'.""" rel = abs_path.relative_to(source_dir) fm = post.metadata title = str(fm["title"]) body_wikitext = markdown_to_wikitext(strip_first_h1(post.content)) source_url = f"{gitea_repo_url}/src/commit/{source_ref}/{rel}" lrc_filename = None lrc_rel = fm.get("lrc") if lrc_rel: lrc_uploaded = upload_lrc(abs_path.parent / str(lrc_rel), title, site) lrc_filename = f"{title}.lrc" if lrc_uploaded: print(f" [lrc-uploaded] {lrc_filename}") page_content = build_wikitext(fm, body_wikitext, source_url, source_ref, lrc_filename) page = site.pages[title] existing = page.text() if existing and not existing.startswith(AUTO_BANNER_PREFIX): return "skipped-manual" if existing == page_content: return "noop" ref_short = source_ref[:8] if source_ref else "unknown" page.save(page_content, summary=f"Auto-published from {ref_short} (songs pipeline)") return "published" def load_and_validate(files: list[Path], source_dir: Path) -> dict[Path, object]: posts = {} titles: dict[str, Path] = {} for path in files: if not path.exists(): print(f" warn: {path} not found, skipping", file=sys.stderr) continue post = frontmatter.load(str(path)) wiki = post.metadata.get("wiki") if not isinstance(wiki, dict) or not wiki.get("publish"): continue validate(post.metadata, path) title = str(post.metadata["title"]) if title in titles: raise ValueError( f"Duplicate wiki title '{title}': {path} and {titles[title]}" ) titles[title] = path posts[path] = post return posts def main() -> None: parser = argparse.ArgumentParser(description="Publish song .md files to MediaWiki.") parser.add_argument("--source-dir", required=True, type=Path) parser.add_argument("--files", nargs="*", default=[], help="Relative paths within source-dir") parser.add_argument("--all", action="store_true", help="Publish all publishable files") args = parser.parse_args() source_dir = args.source_dir.resolve() source_ref = os.environ.get("SOURCE_REF", "") gitea_repo_url = os.environ.get("GITEA_REPO_URL", "").rstrip("/") if args.all: candidates = list(source_dir.rglob("*.md")) elif args.files: candidates = [source_dir / f for f in args.files] else: print("Nothing to do: pass --files or --all.") return files = [ p for p in candidates if p.exists() and p.suffix == ".md" and "wip" not in p.relative_to(source_dir).parts and p.name != "README.md" ] if not files: print("No publishable candidates after filtering.") return print(f"Validating {len(files)} file(s)...") posts = load_and_validate(files, source_dir) print("Validation passed.") site = connect_wiki() counts = {"published": 0, "noop": 0, "skipped-manual": 0} for abs_path, post in posts.items(): rel = abs_path.relative_to(source_dir) outcome = publish_one(abs_path, source_dir, post, site, source_ref, gitea_repo_url) counts[outcome] += 1 if outcome == "skipped-manual": print(f" [skipped-manual] {rel} ← page exists without auto-gen banner; delete or add banner to hand over to bot", file=sys.stderr) else: print(f" [{outcome}] {rel}") print( f"\nDone: {counts['published']} published, {counts['noop']} unchanged, " f"{counts['skipped-manual']} skipped (existing manual pages)." ) if __name__ == "__main__": try: main() except Exception as e: print(f"ERROR: {e}", file=sys.stderr) sys.exit(1)