Files
novoyuuparosk-auto-wiki/pipelines/songs/publish.py
T
mikkeli 580ef2109e feat(songs): LRC file upload and footer link
For songs with an `lrc` field in frontmatter:
- Verifies the file exists and is valid UTF-8 (warns and skips otherwise)
- Uploads to wiki as `File:<title>.lrc` via MediaWiki file API
- Idempotent: skips upload if SHA1 matches existing wiki file
- Appends `[[Media:<title>.lrc|Synced lyrics (.lrc)]]` before the
  category tag at the bottom of the wiki page

Songs without `lrc` in frontmatter are unaffected.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-09 23:11:00 +09:00

233 lines
7.3 KiB
Python

#!/usr/bin/env python3
"""Songs pipeline — validates, renders, and publishes song .md files to MediaWiki."""
import argparse
import hashlib
import io
import os
import subprocess
import sys
from datetime import datetime
from pathlib import Path
import frontmatter
import mwclient
def validate(fm: dict, path: Path) -> None:
errors = []
for field in ("title", "album"):
val = fm.get(field)
if not val or not str(val).strip():
errors.append(f"missing or empty '{field}'")
if not isinstance(fm.get("wiki", {}).get("publish"), bool):
errors.append("'wiki.publish' must be a boolean (true/false), not a string")
if "release_date" in fm and fm["release_date"]:
try:
datetime.strptime(str(fm["release_date"]), "%Y-%m-%d")
except ValueError:
errors.append("'release_date' must be YYYY-MM-DD")
if errors:
raise ValueError(f"{path}: " + "; ".join(errors))
def strip_first_h1(text: str) -> str:
"""Remove the first '# Heading' line and any immediately following blank line."""
lines = text.split("\n")
for i, line in enumerate(lines):
if line.strip().startswith("# "):
del lines[i]
if i < len(lines) and lines[i].strip() == "":
del lines[i]
break
return "\n".join(lines)
def markdown_to_wikitext(body: str) -> str:
result = subprocess.run(
["pandoc", "-f", "markdown", "-t", "mediawiki"],
input=body,
capture_output=True,
text=True,
)
if result.returncode != 0:
raise RuntimeError(f"pandoc failed: {result.stderr.strip()}")
return result.stdout
def build_wikitext(fm: dict, body_wikitext: str, source_url: str, source_ref: str, lrc_filename: str | None = None) -> str:
banner = f"{{{{Auto-generated|source={source_url}|commit={source_ref}}}}}"
category = f"[[Category:{fm['album']}]]"
lrc_line = f"[[Media:{lrc_filename}|Synced lyrics (.lrc)]]\n" if lrc_filename else ""
return f"{banner}\n\n{body_wikitext}{lrc_line}{category}\n"
def connect_wiki() -> mwclient.Site:
from urllib.parse import urlparse
api_url = os.environ["WIKI_API_URL"]
parsed = urlparse(api_url)
host = parsed.netloc
# Strip api.php to get the wiki root path (e.g. "/" or "/w/")
path = parsed.path[: parsed.path.rfind("/") + 1] or "/"
site = mwclient.Site(host, path=path, scheme=parsed.scheme)
site.login(os.environ["WIKI_BOT_USER"], os.environ["WIKI_BOT_PASSWORD"])
return site
def upload_lrc(lrc_path: Path, title: str, site: mwclient.Site) -> bool:
"""Upload LRC file to wiki if content has changed. Returns True if uploaded."""
if not lrc_path.exists():
print(f" warn: LRC file not found: {lrc_path}", file=sys.stderr)
return False
content = lrc_path.read_bytes()
try:
content.decode("utf-8")
except UnicodeDecodeError:
print(f" warn: {lrc_path} is not valid UTF-8 text, skipping LRC upload", file=sys.stderr)
return False
wiki_filename = f"{title}.lrc"
local_sha1 = hashlib.sha1(content).hexdigest()
if site.images[wiki_filename].imageinfo.get("sha1") == local_sha1:
return False
site.upload(
file=io.BytesIO(content),
filename=wiki_filename,
description=f"Synced lyrics for [[{title}]] (auto-published)",
ignore=True,
)
return True
AUTO_BANNER_PREFIX = "{{Auto-generated"
def publish_one(abs_path: Path, source_dir: Path, post, site: mwclient.Site, source_ref: str, gitea_repo_url: str) -> str:
"""Returns 'published', 'noop', or 'skipped-manual'."""
rel = abs_path.relative_to(source_dir)
fm = post.metadata
title = str(fm["title"])
body_wikitext = markdown_to_wikitext(strip_first_h1(post.content))
source_url = f"{gitea_repo_url}/src/commit/{source_ref}/{rel}"
lrc_filename = None
lrc_rel = fm.get("lrc")
if lrc_rel:
lrc_uploaded = upload_lrc(abs_path.parent / str(lrc_rel), title, site)
lrc_filename = f"{title}.lrc"
if lrc_uploaded:
print(f" [lrc-uploaded] {lrc_filename}")
page_content = build_wikitext(fm, body_wikitext, source_url, source_ref, lrc_filename)
page = site.pages[title]
existing = page.text()
if existing and not existing.startswith(AUTO_BANNER_PREFIX):
return "skipped-manual"
if existing == page_content:
return "noop"
ref_short = source_ref[:8] if source_ref else "unknown"
page.save(page_content, summary=f"Auto-published from {ref_short} (songs pipeline)")
return "published"
def load_and_validate(files: list[Path], source_dir: Path) -> dict[Path, object]:
posts = {}
titles: dict[str, Path] = {}
for path in files:
if not path.exists():
print(f" warn: {path} not found, skipping", file=sys.stderr)
continue
post = frontmatter.load(str(path))
wiki = post.metadata.get("wiki")
if not isinstance(wiki, dict) or not wiki.get("publish"):
continue
validate(post.metadata, path)
title = str(post.metadata["title"])
if title in titles:
raise ValueError(
f"Duplicate wiki title '{title}': {path} and {titles[title]}"
)
titles[title] = path
posts[path] = post
return posts
def main() -> None:
parser = argparse.ArgumentParser(description="Publish song .md files to MediaWiki.")
parser.add_argument("--source-dir", required=True, type=Path)
parser.add_argument("--files", nargs="*", default=[], help="Relative paths within source-dir")
parser.add_argument("--all", action="store_true", help="Publish all publishable files")
args = parser.parse_args()
source_dir = args.source_dir.resolve()
source_ref = os.environ.get("SOURCE_REF", "")
gitea_repo_url = os.environ.get("GITEA_REPO_URL", "").rstrip("/")
if args.all:
candidates = list(source_dir.rglob("*.md"))
elif args.files:
candidates = [source_dir / f for f in args.files]
else:
print("Nothing to do: pass --files or --all.")
return
files = [
p for p in candidates
if p.exists()
and p.suffix == ".md"
and "wip" not in p.relative_to(source_dir).parts
and p.name != "README.md"
]
if not files:
print("No publishable candidates after filtering.")
return
print(f"Validating {len(files)} file(s)...")
posts = load_and_validate(files, source_dir)
print("Validation passed.")
site = connect_wiki()
counts = {"published": 0, "noop": 0, "skipped-manual": 0}
for abs_path, post in posts.items():
rel = abs_path.relative_to(source_dir)
outcome = publish_one(abs_path, source_dir, post, site, source_ref, gitea_repo_url)
counts[outcome] += 1
if outcome == "skipped-manual":
print(f" [skipped-manual] {rel} ← page exists without auto-gen banner; delete or add banner to hand over to bot", file=sys.stderr)
else:
print(f" [{outcome}] {rel}")
print(
f"\nDone: {counts['published']} published, {counts['noop']} unchanged, "
f"{counts['skipped-manual']} skipped (existing manual pages)."
)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"ERROR: {e}", file=sys.stderr)
sys.exit(1)