Files
novoyuuparosk-auto-wiki/pipelines/songs/render.py
T
mikkeli 65aa617935 fix(songs): skip pages with existing non-auto-generated content
If a wiki page exists but was written manually (no Auto-generated
banner), the bot now skips it instead of attempting an overwrite
that triggers MediaWiki CAPTCHA. Clear stderr warning tells user
to delete the page or add the banner to hand ownership to the bot.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-09 18:23:14 +09:00

190 lines
5.9 KiB
Python

#!/usr/bin/env python3
"""Songs pipeline renderer — converts source .md files to MediaWiki pages."""
import argparse
import os
import subprocess
import sys
from datetime import datetime
from pathlib import Path
import frontmatter
import mwclient
def validate(fm: dict, path: Path) -> None:
errors = []
for field in ("title", "album"):
val = fm.get(field)
if not val or not str(val).strip():
errors.append(f"missing or empty '{field}'")
if not isinstance(fm.get("wiki", {}).get("publish"), bool):
errors.append("'wiki.publish' must be a boolean (true/false), not a string")
if "release_date" in fm and fm["release_date"]:
try:
datetime.strptime(str(fm["release_date"]), "%Y-%m-%d")
except ValueError:
errors.append("'release_date' must be YYYY-MM-DD")
if errors:
raise ValueError(f"{path}: " + "; ".join(errors))
def strip_first_h1(text: str) -> str:
"""Remove the first '# Heading' line and any immediately following blank line."""
lines = text.split("\n")
for i, line in enumerate(lines):
if line.strip().startswith("# "):
del lines[i]
if i < len(lines) and lines[i].strip() == "":
del lines[i]
break
return "\n".join(lines)
def markdown_to_wikitext(body: str) -> str:
result = subprocess.run(
["pandoc", "-f", "markdown", "-t", "mediawiki"],
input=body,
capture_output=True,
text=True,
)
if result.returncode != 0:
raise RuntimeError(f"pandoc failed: {result.stderr.strip()}")
return result.stdout
def build_wikitext(fm: dict, body_wikitext: str, source_url: str, source_ref: str) -> str:
banner = f"{{{{Auto-generated|source={source_url}|commit={source_ref}}}}}"
category = f"[[Category:{fm['album']}]]"
return f"{banner}\n\n{body_wikitext}\n{category}\n"
def connect_wiki() -> mwclient.Site:
from urllib.parse import urlparse
api_url = os.environ["WIKI_API_URL"]
parsed = urlparse(api_url)
host = parsed.netloc
# Strip api.php to get the wiki root path (e.g. "/" or "/w/")
path = parsed.path[: parsed.path.rfind("/") + 1] or "/"
site = mwclient.Site(host, path=path, scheme=parsed.scheme)
site.login(os.environ["WIKI_BOT_USER"], os.environ["WIKI_BOT_PASSWORD"])
return site
AUTO_BANNER_PREFIX = "{{Auto-generated"
def publish_one(rel: Path, post, site: mwclient.Site, source_ref: str, gitea_repo_url: str) -> str:
"""Returns 'published', 'noop', or 'skipped-manual'."""
fm = post.metadata
body_wikitext = markdown_to_wikitext(strip_first_h1(post.content))
source_url = f"{gitea_repo_url}/src/commit/{source_ref}/{rel}"
page_content = build_wikitext(fm, body_wikitext, source_url, source_ref)
page = site.pages[fm["title"]]
existing = page.text()
if existing and not existing.startswith(AUTO_BANNER_PREFIX):
return "skipped-manual"
if existing == page_content:
return "noop"
ref_short = source_ref[:8] if source_ref else "unknown"
page.save(page_content, summary=f"Auto-published from {ref_short} (songs pipeline)")
return "published"
def load_and_validate(files: list[Path], source_dir: Path) -> dict[Path, object]:
posts = {}
titles: dict[str, Path] = {}
for path in files:
if not path.exists():
print(f" warn: {path} not found, skipping", file=sys.stderr)
continue
post = frontmatter.load(str(path))
wiki = post.metadata.get("wiki")
if not isinstance(wiki, dict) or not wiki.get("publish"):
continue
validate(post.metadata, path)
title = str(post.metadata["title"])
if title in titles:
raise ValueError(
f"Duplicate wiki title '{title}': {path} and {titles[title]}"
)
titles[title] = path
posts[path] = post
return posts
def main() -> None:
parser = argparse.ArgumentParser(description="Publish song .md files to MediaWiki.")
parser.add_argument("--source-dir", required=True, type=Path)
parser.add_argument("--files", nargs="*", default=[], help="Relative paths within source-dir")
parser.add_argument("--all", action="store_true", help="Publish all publishable files")
args = parser.parse_args()
source_dir = args.source_dir.resolve()
source_ref = os.environ.get("SOURCE_REF", "")
gitea_repo_url = os.environ.get("GITEA_REPO_URL", "").rstrip("/")
if args.all:
candidates = list(source_dir.rglob("*.md"))
elif args.files:
candidates = [source_dir / f for f in args.files]
else:
print("Nothing to do: pass --files or --all.")
return
files = [
p for p in candidates
if p.exists()
and p.suffix == ".md"
and "wip" not in p.relative_to(source_dir).parts
and p.name != "README.md"
]
if not files:
print("No publishable candidates after filtering.")
return
print(f"Validating {len(files)} file(s)...")
posts = load_and_validate(files, source_dir)
print("Validation passed.")
site = connect_wiki()
counts = {"published": 0, "noop": 0, "skipped-manual": 0}
for abs_path, post in posts.items():
rel = abs_path.relative_to(source_dir)
outcome = publish_one(rel, post, site, source_ref, gitea_repo_url)
counts[outcome] += 1
if outcome == "skipped-manual":
print(f" [skipped-manual] {rel} ← page exists without auto-gen banner; delete or add banner to hand over to bot", file=sys.stderr)
else:
print(f" [{outcome}] {rel}")
print(
f"\nDone: {counts['published']} published, {counts['noop']} unchanged, "
f"{counts['skipped-manual']} skipped (existing manual pages)."
)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"ERROR: {e}", file=sys.stderr)
sys.exit(1)