Added PDF output (codex)
This commit is contained in:
1
.gitignore
vendored
1
.gitignore
vendored
@@ -5,6 +5,7 @@ runtime/
|
|||||||
images/
|
images/
|
||||||
debug/
|
debug/
|
||||||
output.md
|
output.md
|
||||||
|
output.pdf
|
||||||
*.json
|
*.json
|
||||||
|
|
||||||
*.mkv
|
*.mkv
|
||||||
|
|||||||
12
README.md
12
README.md
@@ -1,6 +1,6 @@
|
|||||||
# 2026-linux-sumka
|
# 2026-linux-sumka
|
||||||
|
|
||||||
Сервис для построения полноценного файла коспекта (главным образом Markdown) из
|
Сервис для построения полноценного конспекта в Markdown и PDF из
|
||||||
записи в одном из видов:
|
записи в одном из видов:
|
||||||
1. **Только звук.** Создаст файл лекции, основываясь только на звуке.
|
1. **Только звук.** Создаст файл лекции, основываясь только на звуке.
|
||||||
2. **Видео со звуком.** Создаст файл лекции на основе звука, и дополнит
|
2. **Видео со звуком.** Создаст файл лекции на основе звука, и дополнит
|
||||||
@@ -31,6 +31,8 @@ python3 -m venv .venv
|
|||||||
# Установить зависимости
|
# Установить зависимости
|
||||||
pip install -r requirements.txt
|
pip install -r requirements.txt
|
||||||
|
|
||||||
|
# Также должны быть установлены ffmpeg и Chromium/Google Chrome
|
||||||
|
|
||||||
# Запуск вот так (см. далее)
|
# Запуск вот так (см. далее)
|
||||||
python main.py
|
python main.py
|
||||||
```
|
```
|
||||||
@@ -451,3 +453,11 @@ Telegram ботом, которому отправили видео).
|
|||||||
- В итоге создаётся файл `output.md`, который может включать в себя ссылки
|
- В итоге создаётся файл `output.md`, который может включать в себя ссылки
|
||||||
на изображения из директории `images/` (относительно директории
|
на изображения из директории `images/` (относительно директории
|
||||||
промежуточных данных)
|
промежуточных данных)
|
||||||
|
10. **Собрать PDF из Markdown**
|
||||||
|
- Сборка выполняется локально, без ИИ, и создаёт файл `output.pdf`.
|
||||||
|
- LaTeX-формулы преобразуются в MathML, а изображения встраиваются из
|
||||||
|
относительных путей Markdown. Пакет `latex2mathml` входит в
|
||||||
|
`requirements.txt`; для печати нужен установленный Chromium или Google
|
||||||
|
Chrome.
|
||||||
|
- PDF оформляется для A4: с типографикой, выделенными важными блоками,
|
||||||
|
подписями к изображениям и нумерацией страниц.
|
||||||
|
|||||||
21
main.py
21
main.py
@@ -19,6 +19,7 @@ from reference_resolver import ReferenceResolver
|
|||||||
from structure_builder import Structure, StructureBuilder
|
from structure_builder import Structure, StructureBuilder
|
||||||
from structure_refiner import StructureRefiner
|
from structure_refiner import StructureRefiner
|
||||||
from markdown_builder import MarkdownBuilder
|
from markdown_builder import MarkdownBuilder
|
||||||
|
from pdf_builder import PdfBuilder
|
||||||
from windowizer import Windowizer
|
from windowizer import Windowizer
|
||||||
|
|
||||||
from agent import Agent
|
from agent import Agent
|
||||||
@@ -36,6 +37,7 @@ class Step(Enum):
|
|||||||
STRUCTURE_BUILDER = "structure_builder"
|
STRUCTURE_BUILDER = "structure_builder"
|
||||||
STRUCTURE_REFINER = "structure_refiner"
|
STRUCTURE_REFINER = "structure_refiner"
|
||||||
MARKDOWN_BUILDER = "markdown_builder"
|
MARKDOWN_BUILDER = "markdown_builder"
|
||||||
|
PDF_BUILDER = "pdf_builder"
|
||||||
|
|
||||||
#
|
#
|
||||||
# Utility
|
# Utility
|
||||||
@@ -345,7 +347,7 @@ def on_structure_refiner(current_step: Step, input_data: dict | None) -> tuple[S
|
|||||||
def on_markdown_builder(current_step: Step, input_data: dict | None) -> tuple[Step | None, str | None]:
|
def on_markdown_builder(current_step: Step, input_data: dict | None) -> tuple[Step | None, str | None]:
|
||||||
if os.path.isfile(WORKFLOW_DATA[current_step][0]):
|
if os.path.isfile(WORKFLOW_DATA[current_step][0]):
|
||||||
logging.info("Skipping Markdown builder")
|
logging.info("Skipping Markdown builder")
|
||||||
return (None, None)
|
return (Step.PDF_BUILDER, None)
|
||||||
if input_data is None:
|
if input_data is None:
|
||||||
logging.error("Can't build Markdown without input_data")
|
logging.error("Can't build Markdown without input_data")
|
||||||
return (None, None)
|
return (None, None)
|
||||||
@@ -353,7 +355,19 @@ def on_markdown_builder(current_step: Step, input_data: dict | None) -> tuple[St
|
|||||||
timeline = Timeline(**json.load(f))
|
timeline = Timeline(**json.load(f))
|
||||||
logging.info("Building Markdown...")
|
logging.info("Building Markdown...")
|
||||||
builder = MarkdownBuilder(timeline)
|
builder = MarkdownBuilder(timeline)
|
||||||
return (None, builder.build(Structure(**input_data)))
|
return (Step.PDF_BUILDER, builder.build(Structure(**input_data)))
|
||||||
|
|
||||||
|
def on_pdf_builder(current_step: Step, input_data: dict | None) -> tuple[None, None]:
|
||||||
|
if os.path.isfile(WORKFLOW_DATA[current_step][0]):
|
||||||
|
logging.info("Skipping PDF builder")
|
||||||
|
return (None, None)
|
||||||
|
if not os.path.isfile("output.md"):
|
||||||
|
logging.error("Can't build PDF without output.md")
|
||||||
|
return (None, None)
|
||||||
|
logging.info("Building PDF...")
|
||||||
|
PdfBuilder().build("output.md", WORKFLOW_DATA[current_step][0])
|
||||||
|
logging.info("PDF is saved to %s", WORKFLOW_DATA[current_step][0])
|
||||||
|
return (None, None)
|
||||||
|
|
||||||
#
|
#
|
||||||
# Main
|
# Main
|
||||||
@@ -367,7 +381,8 @@ WORKFLOW_DATA: dict[Step, tuple[str, Callable[[Step, dict | None], tuple[Step |
|
|||||||
Step.REFERENCE_RESOLVER: ("events.json", on_reference_resolver),
|
Step.REFERENCE_RESOLVER: ("events.json", on_reference_resolver),
|
||||||
Step.STRUCTURE_BUILDER: ("structure.json", on_structure_builder),
|
Step.STRUCTURE_BUILDER: ("structure.json", on_structure_builder),
|
||||||
Step.STRUCTURE_REFINER: ("structure_refined.json", on_structure_refiner),
|
Step.STRUCTURE_REFINER: ("structure_refined.json", on_structure_refiner),
|
||||||
Step.MARKDOWN_BUILDER: ("output.md", on_markdown_builder)
|
Step.MARKDOWN_BUILDER: ("output.md", on_markdown_builder),
|
||||||
|
Step.PDF_BUILDER: ("output.pdf", on_pdf_builder),
|
||||||
}
|
}
|
||||||
"""Information about workflow.
|
"""Information about workflow.
|
||||||
|
|
||||||
|
|||||||
295
pdf_builder.py
Normal file
295
pdf_builder.py
Normal file
@@ -0,0 +1,295 @@
|
|||||||
|
"""Local Markdown to PDF rendering."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import html
|
||||||
|
import re
|
||||||
|
import shutil
|
||||||
|
import subprocess
|
||||||
|
import tempfile
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
_IMAGE_RE = re.compile(r"^!\[([^]]*)\]\((?:<([^>]+)>|([^\s)]+))\)$")
|
||||||
|
_HEADING_RE = re.compile(r"^(#{1,6})\s+(.+)$")
|
||||||
|
_ORDERED_RE = re.compile(r"^\d+[.)]\s+(.+)$")
|
||||||
|
_UNORDERED_RE = re.compile(r"^[-*+]\s+(.+)$")
|
||||||
|
_MATH_RE = re.compile(r"(\$\$.+?\$\$|(?<!\\)\$(?!\$).+?(?<!\\)\$)", re.DOTALL)
|
||||||
|
_BOLD_RE = re.compile(r"\*\*(.+?)\*\*", re.DOTALL)
|
||||||
|
_LINK_RE = re.compile(r"\[([^]]+)]\(([^)]+)\)")
|
||||||
|
|
||||||
|
|
||||||
|
class PdfBuilder:
|
||||||
|
"""Render the Markdown subset produced by MarkdownBuilder as a styled PDF."""
|
||||||
|
|
||||||
|
def __init__(self, browser: str | None = None) -> None:
|
||||||
|
self._browser = browser
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _render_math(source: str, *, display: bool) -> str:
|
||||||
|
try:
|
||||||
|
from latex2mathml.converter import convert
|
||||||
|
except ImportError as error:
|
||||||
|
raise RuntimeError(
|
||||||
|
"PDF formulas require latex2mathml: pip install latex2mathml"
|
||||||
|
) from error
|
||||||
|
|
||||||
|
try:
|
||||||
|
mathml = convert(source.strip())
|
||||||
|
except Exception as error:
|
||||||
|
raise ValueError(f"Invalid LaTeX formula: {source}") from error
|
||||||
|
css_class = "math display" if display else "math inline"
|
||||||
|
return f'<span class="{css_class}">{mathml}</span>'
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def _render_plain_inline(cls, source: str) -> str:
|
||||||
|
parts: list[str] = []
|
||||||
|
cursor = 0
|
||||||
|
for match in _BOLD_RE.finditer(source):
|
||||||
|
parts.append(html.escape(source[cursor:match.start()]))
|
||||||
|
parts.append(f"<strong>{html.escape(match.group(1))}</strong>")
|
||||||
|
cursor = match.end()
|
||||||
|
parts.append(html.escape(source[cursor:]))
|
||||||
|
rendered = "".join(parts)
|
||||||
|
|
||||||
|
def link(match: re.Match[str]) -> str:
|
||||||
|
label = html.escape(match.group(1))
|
||||||
|
target = html.escape(match.group(2), quote=True)
|
||||||
|
return f'<a href="{target}">{label}</a>'
|
||||||
|
|
||||||
|
return _LINK_RE.sub(link, rendered)
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def _render_inline(cls, source: str) -> str:
|
||||||
|
result: list[str] = []
|
||||||
|
cursor = 0
|
||||||
|
for match in _MATH_RE.finditer(source):
|
||||||
|
result.append(cls._render_plain_inline(source[cursor:match.start()]))
|
||||||
|
token = match.group(0)
|
||||||
|
display = token.startswith("$$")
|
||||||
|
formula = token[2:-2] if display else token[1:-1]
|
||||||
|
result.append(cls._render_math(formula, display=display))
|
||||||
|
cursor = match.end()
|
||||||
|
result.append(cls._render_plain_inline(source[cursor:]))
|
||||||
|
return "".join(result)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _image_uri(markdown_path: Path, image_path: str) -> str:
|
||||||
|
path = Path(image_path)
|
||||||
|
if not path.is_absolute():
|
||||||
|
path = markdown_path.parent / path
|
||||||
|
path = path.resolve()
|
||||||
|
if not path.is_file():
|
||||||
|
raise FileNotFoundError(f"Markdown image not found: {path}")
|
||||||
|
return path.as_uri()
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def _render_blocks(cls, markdown: str, markdown_path: Path) -> str:
|
||||||
|
lines = markdown.splitlines()
|
||||||
|
blocks: list[str] = []
|
||||||
|
index = 0
|
||||||
|
first_heading = True
|
||||||
|
|
||||||
|
while index < len(lines):
|
||||||
|
line = lines[index].strip()
|
||||||
|
if not line:
|
||||||
|
index += 1
|
||||||
|
continue
|
||||||
|
|
||||||
|
heading = _HEADING_RE.match(line)
|
||||||
|
if heading:
|
||||||
|
level = min(len(heading.group(1)), 4)
|
||||||
|
title_class = ' class="document-title"' if first_heading else ""
|
||||||
|
blocks.append(
|
||||||
|
f"<h{level}{title_class}>"
|
||||||
|
f"{cls._render_inline(heading.group(2))}</h{level}>"
|
||||||
|
)
|
||||||
|
first_heading = False
|
||||||
|
index += 1
|
||||||
|
continue
|
||||||
|
|
||||||
|
image = _IMAGE_RE.match(line)
|
||||||
|
if image:
|
||||||
|
alt = image.group(1).strip()
|
||||||
|
source = image.group(2) or image.group(3)
|
||||||
|
uri = cls._image_uri(markdown_path, source)
|
||||||
|
caption = (
|
||||||
|
f"<figcaption>{cls._render_inline(alt)}</figcaption>"
|
||||||
|
if alt else ""
|
||||||
|
)
|
||||||
|
blocks.append(
|
||||||
|
"<figure>"
|
||||||
|
f'<img src="{html.escape(uri, quote=True)}" alt="{html.escape(alt, quote=True)}">'
|
||||||
|
f"{caption}</figure>"
|
||||||
|
)
|
||||||
|
index += 1
|
||||||
|
continue
|
||||||
|
|
||||||
|
if line.startswith(">"):
|
||||||
|
quote: list[str] = []
|
||||||
|
while index < len(lines) and lines[index].lstrip().startswith(">"):
|
||||||
|
quote.append(lines[index].lstrip()[1:].strip())
|
||||||
|
index += 1
|
||||||
|
blocks.append(
|
||||||
|
f"<blockquote>{cls._render_inline(' '.join(quote))}</blockquote>"
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
|
||||||
|
ordered = _ORDERED_RE.match(line)
|
||||||
|
unordered = _UNORDERED_RE.match(line)
|
||||||
|
if ordered or unordered:
|
||||||
|
pattern = _ORDERED_RE if ordered else _UNORDERED_RE
|
||||||
|
tag = "ol" if ordered else "ul"
|
||||||
|
items: list[str] = []
|
||||||
|
while index < len(lines):
|
||||||
|
item = pattern.match(lines[index].strip())
|
||||||
|
if not item:
|
||||||
|
break
|
||||||
|
items.append(f"<li>{cls._render_inline(item.group(1))}</li>")
|
||||||
|
index += 1
|
||||||
|
blocks.append(f"<{tag}>{''.join(items)}</{tag}>")
|
||||||
|
continue
|
||||||
|
|
||||||
|
if line.startswith("$$") and line.endswith("$$") and len(line) > 4:
|
||||||
|
blocks.append(cls._render_math(line[2:-2], display=True))
|
||||||
|
index += 1
|
||||||
|
continue
|
||||||
|
|
||||||
|
paragraph = [line]
|
||||||
|
index += 1
|
||||||
|
while index < len(lines) and lines[index].strip():
|
||||||
|
candidate = lines[index].strip()
|
||||||
|
if (
|
||||||
|
_HEADING_RE.match(candidate)
|
||||||
|
or _IMAGE_RE.match(candidate)
|
||||||
|
or candidate.startswith(">")
|
||||||
|
or _ORDERED_RE.match(candidate)
|
||||||
|
or _UNORDERED_RE.match(candidate)
|
||||||
|
):
|
||||||
|
break
|
||||||
|
paragraph.append(candidate)
|
||||||
|
index += 1
|
||||||
|
blocks.append(f"<p>{cls._render_inline(' '.join(paragraph))}</p>")
|
||||||
|
|
||||||
|
return "\n".join(blocks)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _document(body: str, title: str) -> str:
|
||||||
|
return f"""<!doctype html>
|
||||||
|
<html lang="ru">
|
||||||
|
<head>
|
||||||
|
<meta charset="utf-8">
|
||||||
|
<title>{html.escape(title)}</title>
|
||||||
|
<style>
|
||||||
|
@page {{
|
||||||
|
size: A4;
|
||||||
|
margin: 19mm 18mm 20mm;
|
||||||
|
@bottom-center {{
|
||||||
|
content: "SUMKA · " counter(page);
|
||||||
|
color: #718096;
|
||||||
|
font: 8.5pt "Ubuntu", "DejaVu Sans", sans-serif;
|
||||||
|
letter-spacing: .08em;
|
||||||
|
}}
|
||||||
|
}}
|
||||||
|
* {{ box-sizing: border-box; }}
|
||||||
|
html {{ color: #172033; font-family: "Ubuntu", "DejaVu Sans", sans-serif; }}
|
||||||
|
body {{ margin: 0; font-size: 10.7pt; line-height: 1.62; }}
|
||||||
|
h1, h2, h3, h4 {{ color: #102a43; line-height: 1.22; break-after: avoid; }}
|
||||||
|
h1 {{ font-size: 28pt; }}
|
||||||
|
h2 {{ font-size: 21pt; margin: 1.2em 0 .55em; }}
|
||||||
|
h3 {{ font-size: 15.5pt; margin: 1.45em 0 .5em; padding-bottom: .22em;
|
||||||
|
border-bottom: 1px solid #d9e2ec; }}
|
||||||
|
h4 {{ font-size: 12.5pt; margin: 1.25em 0 .4em; color: #1f5f8b; }}
|
||||||
|
.document-title {{ margin: 0 0 1.15em; padding: .75em .9em .8em;
|
||||||
|
border-left: 5px solid #2f80ed;
|
||||||
|
background: linear-gradient(120deg, #eef6ff, #f8fbff 70%);
|
||||||
|
border-radius: 0 10px 10px 0; }}
|
||||||
|
p {{ margin: 0 0 .85em; text-align: justify; hyphens: auto; orphans: 3; widows: 3; }}
|
||||||
|
strong {{ color: #0f3557; font-weight: 700; }}
|
||||||
|
ul, ol {{ margin: .3em 0 1em; padding-left: 1.55em; }}
|
||||||
|
li {{ margin: .24em 0; padding-left: .2em; }}
|
||||||
|
li::marker {{ color: #2f80ed; font-weight: 700; }}
|
||||||
|
blockquote {{ margin: 1.1em 0; padding: .8em 1em .8em 1.15em;
|
||||||
|
border-left: 4px solid #f2b84b; background: #fff8e8; color: #483b22;
|
||||||
|
border-radius: 0 8px 8px 0; break-inside: avoid; }}
|
||||||
|
figure {{ margin: 1.25em auto 1.45em; text-align: center; break-inside: avoid; }}
|
||||||
|
figure img {{ display: block; max-width: 100%; max-height: 188mm; margin: auto;
|
||||||
|
object-fit: contain; border: 1px solid #d9e2ec; border-radius: 7px;
|
||||||
|
box-shadow: 0 5px 18px rgba(23, 32, 51, .12); }}
|
||||||
|
figcaption {{ margin: .55em auto 0; max-width: 88%; color: #637083;
|
||||||
|
font-size: 8.7pt; line-height: 1.4; }}
|
||||||
|
.math.inline {{ display: inline-block; white-space: nowrap; vertical-align: -.12em; }}
|
||||||
|
.math.display {{ display: block; margin: 1.1em auto; padding: .75em 1em;
|
||||||
|
overflow: hidden; text-align: center; background: #f6f9fc;
|
||||||
|
border: 1px solid #e3eaf2; border-radius: 7px; break-inside: avoid; }}
|
||||||
|
math {{ font-family: "STIX Two Math", "STIX Math", "Cambria Math", serif; }}
|
||||||
|
a {{ color: #1769aa; text-decoration: none; }}
|
||||||
|
</style>
|
||||||
|
</head>
|
||||||
|
<body>{body}</body>
|
||||||
|
</html>
|
||||||
|
"""
|
||||||
|
|
||||||
|
def _find_browser(self) -> str:
|
||||||
|
if self._browser:
|
||||||
|
browser = shutil.which(self._browser) or self._browser
|
||||||
|
if Path(browser).is_file():
|
||||||
|
return browser
|
||||||
|
raise RuntimeError(f"PDF browser not found: {self._browser}")
|
||||||
|
|
||||||
|
for candidate in (
|
||||||
|
"chromium",
|
||||||
|
"chromium-browser",
|
||||||
|
"google-chrome",
|
||||||
|
"google-chrome-stable",
|
||||||
|
):
|
||||||
|
browser = shutil.which(candidate)
|
||||||
|
if browser:
|
||||||
|
return browser
|
||||||
|
raise RuntimeError("PDF generation requires Chromium or Google Chrome")
|
||||||
|
|
||||||
|
def build(self, markdown_path: str | Path, output_path: str | Path) -> None:
|
||||||
|
source = Path(markdown_path).resolve()
|
||||||
|
destination = Path(output_path).resolve()
|
||||||
|
if not source.is_file():
|
||||||
|
raise FileNotFoundError(f"Markdown file not found: {source}")
|
||||||
|
|
||||||
|
markdown = source.read_text(encoding="utf-8")
|
||||||
|
body = self._render_blocks(markdown, source)
|
||||||
|
title_match = re.search(r"^#{1,6}\s+(.+)$", markdown, re.MULTILINE)
|
||||||
|
title = title_match.group(1).replace("**", "") if title_match else "Конспект"
|
||||||
|
document = self._document(body, title)
|
||||||
|
browser = self._find_browser()
|
||||||
|
destination.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
with tempfile.TemporaryDirectory(prefix=".sumka-pdf-", dir=source.parent) as tmp:
|
||||||
|
workdir = Path(tmp)
|
||||||
|
html_path = workdir / "document.html"
|
||||||
|
html_path.write_text(document, encoding="utf-8")
|
||||||
|
command = [
|
||||||
|
browser,
|
||||||
|
"--headless",
|
||||||
|
"--disable-gpu",
|
||||||
|
"--disable-dev-shm-usage",
|
||||||
|
"--allow-file-access-from-files",
|
||||||
|
"--no-pdf-header-footer",
|
||||||
|
f"--user-data-dir={workdir / 'profile'}",
|
||||||
|
f"--print-to-pdf={destination}",
|
||||||
|
html_path.as_uri(),
|
||||||
|
]
|
||||||
|
try:
|
||||||
|
result = subprocess.run(
|
||||||
|
command,
|
||||||
|
capture_output=True,
|
||||||
|
text=True,
|
||||||
|
timeout=120,
|
||||||
|
check=False,
|
||||||
|
)
|
||||||
|
except subprocess.TimeoutExpired as error:
|
||||||
|
raise RuntimeError("Chromium timed out while generating PDF") from error
|
||||||
|
|
||||||
|
if result.returncode != 0 or not destination.is_file():
|
||||||
|
details = (result.stderr or result.stdout).strip()
|
||||||
|
raise RuntimeError(
|
||||||
|
f"Chromium failed to generate PDF: {details or result.returncode}"
|
||||||
|
)
|
||||||
6
requirements.txt
Normal file
6
requirements.txt
Normal file
@@ -0,0 +1,6 @@
|
|||||||
|
httpx2
|
||||||
|
latex2mathml
|
||||||
|
openai
|
||||||
|
openai-whisper
|
||||||
|
pydantic>=2
|
||||||
|
torch
|
||||||
Reference in New Issue
Block a user