Files
2026-linux-sumka/pdf_builder.py
2026-09-23 03:26:09 +03:00

299 lines
11 KiB
Python

"""Local Markdown to PDF rendering."""
from __future__ import annotations
import html
import os
import re
import shutil
import subprocess
import tempfile
from pathlib import Path
_IMAGE_RE = re.compile(r"^!\[([^]]*)\]\((?:<([^>]+)>|([^\s)]+))\)$")
_HEADING_RE = re.compile(r"^(#{1,6})\s+(.+)$")
_ORDERED_RE = re.compile(r"^\d+[.)]\s+(.+)$")
_UNORDERED_RE = re.compile(r"^[-*+]\s+(.+)$")
_MATH_RE = re.compile(r"(\$\$.+?\$\$|(?<!\\)\$(?!\$).+?(?<!\\)\$)", re.DOTALL)
_BOLD_RE = re.compile(r"\*\*(.+?)\*\*", re.DOTALL)
_LINK_RE = re.compile(r"\[([^]]+)]\(([^)]+)\)")
class PdfBuilder:
"""Render the Markdown subset produced by MarkdownBuilder as a styled PDF."""
def __init__(self, browser: str | None = None) -> None:
self._browser = browser
@staticmethod
def _render_math(source: str, *, display: bool) -> str:
try:
from latex2mathml.converter import convert
except ImportError as error:
raise RuntimeError(
"PDF formulas require latex2mathml: pip install latex2mathml"
) from error
try:
mathml = convert(source.strip())
except Exception as error:
raise ValueError(f"Invalid LaTeX formula: {source}") from error
css_class = "math display" if display else "math inline"
return f'<span class="{css_class}">{mathml}</span>'
@classmethod
def _render_plain_inline(cls, source: str) -> str:
parts: list[str] = []
cursor = 0
for match in _BOLD_RE.finditer(source):
parts.append(html.escape(source[cursor:match.start()]))
parts.append(f"<strong>{html.escape(match.group(1))}</strong>")
cursor = match.end()
parts.append(html.escape(source[cursor:]))
rendered = "".join(parts)
def link(match: re.Match[str]) -> str:
label = html.escape(match.group(1))
target = html.escape(match.group(2), quote=True)
return f'<a href="{target}">{label}</a>'
return _LINK_RE.sub(link, rendered)
@classmethod
def _render_inline(cls, source: str) -> str:
result: list[str] = []
cursor = 0
for match in _MATH_RE.finditer(source):
result.append(cls._render_plain_inline(source[cursor:match.start()]))
token = match.group(0)
display = token.startswith("$$")
formula = token[2:-2] if display else token[1:-1]
result.append(cls._render_math(formula, display=display))
cursor = match.end()
result.append(cls._render_plain_inline(source[cursor:]))
return "".join(result)
@staticmethod
def _image_uri(markdown_path: Path, image_path: str) -> str:
path = Path(image_path)
if not path.is_absolute():
path = markdown_path.parent / path
path = path.resolve()
if not path.is_file():
raise FileNotFoundError(f"Markdown image not found: {path}")
return path.as_uri()
@classmethod
def _render_blocks(cls, markdown: str, markdown_path: Path) -> str:
lines = markdown.splitlines()
blocks: list[str] = []
index = 0
first_heading = True
while index < len(lines):
line = lines[index].strip()
if not line:
index += 1
continue
heading = _HEADING_RE.match(line)
if heading:
level = min(len(heading.group(1)), 4)
title_class = ' class="document-title"' if first_heading else ""
blocks.append(
f"<h{level}{title_class}>"
f"{cls._render_inline(heading.group(2))}</h{level}>"
)
first_heading = False
index += 1
continue
image = _IMAGE_RE.match(line)
if image:
alt = image.group(1).strip()
source = image.group(2) or image.group(3)
uri = cls._image_uri(markdown_path, source)
caption = (
f"<figcaption>{cls._render_inline(alt)}</figcaption>"
if alt else ""
)
blocks.append(
"<figure>"
f'<img src="{html.escape(uri, quote=True)}" alt="{html.escape(alt, quote=True)}">'
f"{caption}</figure>"
)
index += 1
continue
if line.startswith(">"):
quote: list[str] = []
while index < len(lines) and lines[index].lstrip().startswith(">"):
quote.append(lines[index].lstrip()[1:].strip())
index += 1
blocks.append(
f"<blockquote>{cls._render_inline(' '.join(quote))}</blockquote>"
)
continue
ordered = _ORDERED_RE.match(line)
unordered = _UNORDERED_RE.match(line)
if ordered or unordered:
pattern = _ORDERED_RE if ordered else _UNORDERED_RE
tag = "ol" if ordered else "ul"
items: list[str] = []
while index < len(lines):
item = pattern.match(lines[index].strip())
if not item:
break
items.append(f"<li>{cls._render_inline(item.group(1))}</li>")
index += 1
blocks.append(f"<{tag}>{''.join(items)}</{tag}>")
continue
if line.startswith("$$") and line.endswith("$$") and len(line) > 4:
blocks.append(cls._render_math(line[2:-2], display=True))
index += 1
continue
paragraph = [line]
index += 1
while index < len(lines) and lines[index].strip():
candidate = lines[index].strip()
if (
_HEADING_RE.match(candidate)
or _IMAGE_RE.match(candidate)
or candidate.startswith(">")
or _ORDERED_RE.match(candidate)
or _UNORDERED_RE.match(candidate)
):
break
paragraph.append(candidate)
index += 1
blocks.append(f"<p>{cls._render_inline(' '.join(paragraph))}</p>")
return "\n".join(blocks)
@staticmethod
def _document(body: str, title: str) -> str:
return f"""<!doctype html>
<html lang="ru">
<head>
<meta charset="utf-8">
<title>{html.escape(title)}</title>
<style>
@page {{
size: A4;
margin: 19mm 18mm 20mm;
@bottom-center {{
content: "SUMKA · " counter(page);
color: #718096;
font: 8.5pt "Ubuntu", "DejaVu Sans", sans-serif;
letter-spacing: .08em;
}}
}}
* {{ box-sizing: border-box; }}
html {{ color: #172033; font-family: "Ubuntu", "DejaVu Sans", sans-serif; }}
body {{ margin: 0; font-size: 10.7pt; line-height: 1.62; }}
h1, h2, h3, h4 {{ color: #102a43; line-height: 1.22; break-after: avoid; }}
h1 {{ font-size: 28pt; }}
h2 {{ font-size: 21pt; margin: 1.2em 0 .55em; }}
h3 {{ font-size: 15.5pt; margin: 1.45em 0 .5em; padding-bottom: .22em;
border-bottom: 1px solid #d9e2ec; }}
h4 {{ font-size: 12.5pt; margin: 1.25em 0 .4em; color: #1f5f8b; }}
.document-title {{ margin: 0 0 1.15em; padding: .75em .9em .8em;
border-left: 5px solid #2f80ed;
background: linear-gradient(120deg, #eef6ff, #f8fbff 70%);
border-radius: 0 10px 10px 0; }}
p {{ margin: 0 0 .85em; text-align: justify; hyphens: auto; orphans: 3; widows: 3; }}
strong {{ color: #0f3557; font-weight: 700; }}
ul, ol {{ margin: .3em 0 1em; padding-left: 1.55em; }}
li {{ margin: .24em 0; padding-left: .2em; }}
li::marker {{ color: #2f80ed; font-weight: 700; }}
blockquote {{ margin: 1.1em 0; padding: .8em 1em .8em 1.15em;
border-left: 4px solid #f2b84b; background: #fff8e8; color: #483b22;
border-radius: 0 8px 8px 0; break-inside: avoid; }}
figure {{ margin: 1.25em auto 1.45em; text-align: center; break-inside: avoid; }}
figure img {{ display: block; max-width: 100%; max-height: 188mm; margin: auto;
object-fit: contain; border: 1px solid #d9e2ec; border-radius: 7px;
box-shadow: 0 5px 18px rgba(23, 32, 51, .12); }}
figcaption {{ margin: .55em auto 0; max-width: 88%; color: #637083;
font-size: 8.7pt; line-height: 1.4; }}
.math.inline {{ display: inline-block; white-space: nowrap; vertical-align: -.12em; }}
.math.display {{ display: block; margin: 1.1em auto; padding: .75em 1em;
overflow: hidden; text-align: center; background: #f6f9fc;
border: 1px solid #e3eaf2; border-radius: 7px; break-inside: avoid; }}
math {{ font-family: "STIX Two Math", "STIX Math", "Cambria Math", serif; }}
a {{ color: #1769aa; text-decoration: none; }}
</style>
</head>
<body>{body}</body>
</html>
"""
def _find_browser(self) -> str:
if self._browser:
browser = shutil.which(self._browser) or self._browser
if Path(browser).is_file():
return browser
raise RuntimeError(f"PDF browser not found: {self._browser}")
for candidate in (
"chromium",
"chromium-browser",
"google-chrome",
"google-chrome-stable",
):
browser = shutil.which(candidate)
if browser:
return browser
raise RuntimeError("PDF generation requires Chromium or Google Chrome")
def build(self, markdown_path: str | Path, output_path: str | Path) -> None:
source = Path(markdown_path).resolve()
destination = Path(output_path).resolve()
if not source.is_file():
raise FileNotFoundError(f"Markdown file not found: {source}")
markdown = source.read_text(encoding="utf-8")
body = self._render_blocks(markdown, source)
title_match = re.search(r"^#{1,6}\s+(.+)$", markdown, re.MULTILINE)
title = title_match.group(1).replace("**", "") if title_match else "Конспект"
document = self._document(body, title)
browser = self._find_browser()
destination.parent.mkdir(parents=True, exist_ok=True)
with tempfile.TemporaryDirectory(prefix=".sumka-pdf-", dir=source.parent) as tmp:
workdir = Path(tmp)
html_path = workdir / "document.html"
html_path.write_text(document, encoding="utf-8")
command = [
browser,
"--headless",
"--disable-gpu",
"--disable-dev-shm-usage",
"--allow-file-access-from-files",
"--no-pdf-header-footer",
f"--user-data-dir={workdir / 'profile'}",
f"--print-to-pdf={destination}",
html_path.as_uri(),
]
if os.geteuid() == 0:
command.insert(1, "--no-sandbox")
try:
result = subprocess.run(
command,
capture_output=True,
text=True,
timeout=120,
check=False,
)
except subprocess.TimeoutExpired as error:
raise RuntimeError("Chromium timed out while generating PDF") from error
if result.returncode != 0 or not destination.is_file():
details = (result.stderr or result.stdout).strip()
raise RuntimeError(
f"Chromium failed to generate PDF: {details or result.returncode}"
)