"""Shared utilities for novoyuuparosk-auto-wiki pipelines.""" import html import os import re import subprocess from urllib.parse import urlparse import mwclient AUTO_BANNER_PREFIX = "{{Auto-generated" # A ```columns fenced block is passed through Pandoc verbatim as #
; columns within it are separated by a line of ===. _COLUMNS_BLOCK_RE = re.compile(r'
(.*?)
', re.DOTALL) _COLUMN_SEP_RE = re.compile(r"^\s*===\s*$", re.MULTILINE) # Minimal Markdown emphasis → wikitext, for use inside a verbatim columns block # (Pandoc doesn't process Markdown there). Asterisk style only, single line. # Order matters: bold-italic (***) before bold (**) before italic (*). _BOLD_ITALIC_RE = re.compile(r"\*\*\*(.+?)\*\*\*") _BOLD_RE = re.compile(r"\*\*(.+?)\*\*") _ITALIC_RE = re.compile(r"\*(.+?)\*") def _md_emphasis_to_wikitext(text: str) -> str: text = _BOLD_ITALIC_RE.sub(r"'''''\1'''''", text) text = _BOLD_RE.sub(r"'''\1'''", text) text = _ITALIC_RE.sub(r"''\1''", text) return text def strip_first_h1(text: str) -> str: """Remove the first '# Heading' line and any immediately following blank line.""" lines = text.split("\n") for i, line in enumerate(lines): if line.strip().startswith("# "): del lines[i] if i < len(lines) and lines[i].strip() == "": del lines[i] break return "\n".join(lines) def expand_columns(wikitext: str) -> str: """Expand ```columns fenced blocks into a flex row of columns. Authors write a fenced code block tagged ``columns``; Pandoc passes its body through verbatim as ``
`` (line breaks and blank lines preserved, inline markup entity-escaped). Columns within the block are separated by a line containing only ``===``. Each column is wrapped in so its line breaks survive MediaWiki parsing. Content is HTML-unescaped (so inline markup such as renders rather than appearing as literal text) and Markdown emphasis (``*italic*``, ``**bold**``, ``***bold-italic***``) is converted to wikitext, since Pandoc does not process Markdown inside the verbatim block. """ def render(match: re.Match) -> str: body = _md_emphasis_to_wikitext(html.unescape(match.group(1))) columns = _COLUMN_SEP_RE.split(body) poems = "".join("\n" + col.strip("\n") + "\n" for col in columns) return '
' + poems + "
" return _COLUMNS_BLOCK_RE.sub(render, wikitext) def markdown_to_wikitext(body: str) -> str: result = subprocess.run( ["pandoc", "-f", "markdown", "-t", "mediawiki"], input=body, capture_output=True, text=True, ) if result.returncode != 0: raise RuntimeError(f"pandoc failed: {result.stderr.strip()}") return expand_columns(result.stdout) def connect_wiki() -> mwclient.Site: api_url = os.environ["WIKI_API_URL"] parsed = urlparse(api_url) host = parsed.netloc path = parsed.path[: parsed.path.rfind("/") + 1] or "/" site = mwclient.Site(host, path=path, scheme=parsed.scheme) site.login(os.environ["WIKI_BOT_USER"], os.environ["WIKI_BOT_PASSWORD"]) return site