diff --git a/.gitignore b/.gitignore index 29e1a43..3799ed7 100644 --- a/.gitignore +++ b/.gitignore @@ -5,6 +5,7 @@ runtime/ images/ debug/ output.md +output.pdf *.json *.mkv @@ -14,4 +15,4 @@ output.md *.wav *.ogg *.mp3 -*.m4a \ No newline at end of file +*.m4a diff --git a/README.md b/README.md index 14e1de8..175307d 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,6 @@ # 2026-linux-sumka -Сервис для построения полноценного файла коспекта (главным образом Markdown) из +Сервис для построения полноценного конспекта в Markdown и PDF из записи в одном из видов: 1. **Только звук.** Создаст файл лекции, основываясь только на звуке. 2. **Видео со звуком.** Создаст файл лекции на основе звука, и дополнит @@ -31,6 +31,8 @@ python3 -m venv .venv # Установить зависимости pip install -r requirements.txt +# Также должны быть установлены ffmpeg и Chromium/Google Chrome + # Запуск вот так (см. далее) python main.py ``` @@ -451,3 +453,11 @@ Telegram ботом, которому отправили видео). - В итоге создаётся файл `output.md`, который может включать в себя ссылки на изображения из директории `images/` (относительно директории промежуточных данных) +10. **Собрать PDF из Markdown** + - Сборка выполняется локально, без ИИ, и создаёт файл `output.pdf`. + - LaTeX-формулы преобразуются в MathML, а изображения встраиваются из + относительных путей Markdown. Пакет `latex2mathml` входит в + `requirements.txt`; для печати нужен установленный Chromium или Google + Chrome. + - PDF оформляется для A4: с типографикой, выделенными важными блоками, + подписями к изображениям и нумерацией страниц. diff --git a/main.py b/main.py index 9d5fbd2..9e5b639 100644 --- a/main.py +++ b/main.py @@ -19,6 +19,7 @@ from reference_resolver import ReferenceResolver from structure_builder import Structure, StructureBuilder from structure_refiner import StructureRefiner from markdown_builder import MarkdownBuilder +from pdf_builder import PdfBuilder from windowizer import Windowizer from agent import Agent @@ -36,6 +37,7 @@ class Step(Enum): STRUCTURE_BUILDER = "structure_builder" STRUCTURE_REFINER = "structure_refiner" MARKDOWN_BUILDER = "markdown_builder" + PDF_BUILDER = "pdf_builder" # # Utility @@ -345,7 +347,7 @@ def on_structure_refiner(current_step: Step, input_data: dict | None) -> tuple[S def on_markdown_builder(current_step: Step, input_data: dict | None) -> tuple[Step | None, str | None]: if os.path.isfile(WORKFLOW_DATA[current_step][0]): logging.info("Skipping Markdown builder") - return (None, None) + return (Step.PDF_BUILDER, None) if input_data is None: logging.error("Can't build Markdown without input_data") return (None, None) @@ -353,7 +355,19 @@ def on_markdown_builder(current_step: Step, input_data: dict | None) -> tuple[St timeline = Timeline(**json.load(f)) logging.info("Building Markdown...") builder = MarkdownBuilder(timeline) - return (None, builder.build(Structure(**input_data))) + return (Step.PDF_BUILDER, builder.build(Structure(**input_data))) + +def on_pdf_builder(current_step: Step, input_data: dict | None) -> tuple[None, None]: + if os.path.isfile(WORKFLOW_DATA[current_step][0]): + logging.info("Skipping PDF builder") + return (None, None) + if not os.path.isfile("output.md"): + logging.error("Can't build PDF without output.md") + return (None, None) + logging.info("Building PDF...") + PdfBuilder().build("output.md", WORKFLOW_DATA[current_step][0]) + logging.info("PDF is saved to %s", WORKFLOW_DATA[current_step][0]) + return (None, None) # # Main @@ -367,7 +381,8 @@ WORKFLOW_DATA: dict[Step, tuple[str, Callable[[Step, dict | None], tuple[Step | Step.REFERENCE_RESOLVER: ("events.json", on_reference_resolver), Step.STRUCTURE_BUILDER: ("structure.json", on_structure_builder), Step.STRUCTURE_REFINER: ("structure_refined.json", on_structure_refiner), - Step.MARKDOWN_BUILDER: ("output.md", on_markdown_builder) + Step.MARKDOWN_BUILDER: ("output.md", on_markdown_builder), + Step.PDF_BUILDER: ("output.pdf", on_pdf_builder), } """Information about workflow. diff --git a/pdf_builder.py b/pdf_builder.py new file mode 100644 index 0000000..5850989 --- /dev/null +++ b/pdf_builder.py @@ -0,0 +1,295 @@ +"""Local Markdown to PDF rendering.""" + +from __future__ import annotations + +import html +import re +import shutil +import subprocess +import tempfile +from pathlib import Path + + +_IMAGE_RE = re.compile(r"^!\[([^]]*)\]\((?:<([^>]+)>|([^\s)]+))\)$") +_HEADING_RE = re.compile(r"^(#{1,6})\s+(.+)$") +_ORDERED_RE = re.compile(r"^\d+[.)]\s+(.+)$") +_UNORDERED_RE = re.compile(r"^[-*+]\s+(.+)$") +_MATH_RE = re.compile(r"(\$\$.+?\$\$|(? None: + self._browser = browser + + @staticmethod + def _render_math(source: str, *, display: bool) -> str: + try: + from latex2mathml.converter import convert + except ImportError as error: + raise RuntimeError( + "PDF formulas require latex2mathml: pip install latex2mathml" + ) from error + + try: + mathml = convert(source.strip()) + except Exception as error: + raise ValueError(f"Invalid LaTeX formula: {source}") from error + css_class = "math display" if display else "math inline" + return f'{mathml}' + + @classmethod + def _render_plain_inline(cls, source: str) -> str: + parts: list[str] = [] + cursor = 0 + for match in _BOLD_RE.finditer(source): + parts.append(html.escape(source[cursor:match.start()])) + parts.append(f"{html.escape(match.group(1))}") + cursor = match.end() + parts.append(html.escape(source[cursor:])) + rendered = "".join(parts) + + def link(match: re.Match[str]) -> str: + label = html.escape(match.group(1)) + target = html.escape(match.group(2), quote=True) + return f'{label}' + + return _LINK_RE.sub(link, rendered) + + @classmethod + def _render_inline(cls, source: str) -> str: + result: list[str] = [] + cursor = 0 + for match in _MATH_RE.finditer(source): + result.append(cls._render_plain_inline(source[cursor:match.start()])) + token = match.group(0) + display = token.startswith("$$") + formula = token[2:-2] if display else token[1:-1] + result.append(cls._render_math(formula, display=display)) + cursor = match.end() + result.append(cls._render_plain_inline(source[cursor:])) + return "".join(result) + + @staticmethod + def _image_uri(markdown_path: Path, image_path: str) -> str: + path = Path(image_path) + if not path.is_absolute(): + path = markdown_path.parent / path + path = path.resolve() + if not path.is_file(): + raise FileNotFoundError(f"Markdown image not found: {path}") + return path.as_uri() + + @classmethod + def _render_blocks(cls, markdown: str, markdown_path: Path) -> str: + lines = markdown.splitlines() + blocks: list[str] = [] + index = 0 + first_heading = True + + while index < len(lines): + line = lines[index].strip() + if not line: + index += 1 + continue + + heading = _HEADING_RE.match(line) + if heading: + level = min(len(heading.group(1)), 4) + title_class = ' class="document-title"' if first_heading else "" + blocks.append( + f"" + f"{cls._render_inline(heading.group(2))}" + ) + first_heading = False + index += 1 + continue + + image = _IMAGE_RE.match(line) + if image: + alt = image.group(1).strip() + source = image.group(2) or image.group(3) + uri = cls._image_uri(markdown_path, source) + caption = ( + f"
{cls._render_inline(alt)}
" + if alt else "" + ) + blocks.append( + "
" + f'{html.escape(alt, quote=True)}' + f"{caption}
" + ) + index += 1 + continue + + if line.startswith(">"): + quote: list[str] = [] + while index < len(lines) and lines[index].lstrip().startswith(">"): + quote.append(lines[index].lstrip()[1:].strip()) + index += 1 + blocks.append( + f"
{cls._render_inline(' '.join(quote))}
" + ) + continue + + ordered = _ORDERED_RE.match(line) + unordered = _UNORDERED_RE.match(line) + if ordered or unordered: + pattern = _ORDERED_RE if ordered else _UNORDERED_RE + tag = "ol" if ordered else "ul" + items: list[str] = [] + while index < len(lines): + item = pattern.match(lines[index].strip()) + if not item: + break + items.append(f"
  • {cls._render_inline(item.group(1))}
  • ") + index += 1 + blocks.append(f"<{tag}>{''.join(items)}") + continue + + if line.startswith("$$") and line.endswith("$$") and len(line) > 4: + blocks.append(cls._render_math(line[2:-2], display=True)) + index += 1 + continue + + paragraph = [line] + index += 1 + while index < len(lines) and lines[index].strip(): + candidate = lines[index].strip() + if ( + _HEADING_RE.match(candidate) + or _IMAGE_RE.match(candidate) + or candidate.startswith(">") + or _ORDERED_RE.match(candidate) + or _UNORDERED_RE.match(candidate) + ): + break + paragraph.append(candidate) + index += 1 + blocks.append(f"

    {cls._render_inline(' '.join(paragraph))}

    ") + + return "\n".join(blocks) + + @staticmethod + def _document(body: str, title: str) -> str: + return f""" + + + +{html.escape(title)} + + +{body} + +""" + + def _find_browser(self) -> str: + if self._browser: + browser = shutil.which(self._browser) or self._browser + if Path(browser).is_file(): + return browser + raise RuntimeError(f"PDF browser not found: {self._browser}") + + for candidate in ( + "chromium", + "chromium-browser", + "google-chrome", + "google-chrome-stable", + ): + browser = shutil.which(candidate) + if browser: + return browser + raise RuntimeError("PDF generation requires Chromium or Google Chrome") + + def build(self, markdown_path: str | Path, output_path: str | Path) -> None: + source = Path(markdown_path).resolve() + destination = Path(output_path).resolve() + if not source.is_file(): + raise FileNotFoundError(f"Markdown file not found: {source}") + + markdown = source.read_text(encoding="utf-8") + body = self._render_blocks(markdown, source) + title_match = re.search(r"^#{1,6}\s+(.+)$", markdown, re.MULTILINE) + title = title_match.group(1).replace("**", "") if title_match else "Конспект" + document = self._document(body, title) + browser = self._find_browser() + destination.parent.mkdir(parents=True, exist_ok=True) + + with tempfile.TemporaryDirectory(prefix=".sumka-pdf-", dir=source.parent) as tmp: + workdir = Path(tmp) + html_path = workdir / "document.html" + html_path.write_text(document, encoding="utf-8") + command = [ + browser, + "--headless", + "--disable-gpu", + "--disable-dev-shm-usage", + "--allow-file-access-from-files", + "--no-pdf-header-footer", + f"--user-data-dir={workdir / 'profile'}", + f"--print-to-pdf={destination}", + html_path.as_uri(), + ] + try: + result = subprocess.run( + command, + capture_output=True, + text=True, + timeout=120, + check=False, + ) + except subprocess.TimeoutExpired as error: + raise RuntimeError("Chromium timed out while generating PDF") from error + + if result.returncode != 0 or not destination.is_file(): + details = (result.stderr or result.stdout).strip() + raise RuntimeError( + f"Chromium failed to generate PDF: {details or result.returncode}" + ) diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..7fa5704 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,6 @@ +httpx2 +latex2mathml +openai +openai-whisper +pydantic>=2 +torch