Files
kkpab-site/frontend/scripts/convert_documents_to_pdf.py

126 lines
4.2 KiB
Python

from __future__ import annotations
import re
import zipfile
from pathlib import Path
from typing import Iterable
from xml.etree import ElementTree as ET
from docx import Document as DocxDocument
from reportlab.lib.pagesizes import A4
from reportlab.lib.styles import ParagraphStyle, getSampleStyleSheet
from reportlab.lib.units import mm
from reportlab.pdfbase import pdfmetrics
from reportlab.pdfbase.ttfonts import TTFont
from reportlab.platypus import Paragraph, SimpleDocTemplate, Spacer
ROOT = Path(__file__).resolve().parents[1]
DOCS_DIR = ROOT / "public" / "documents"
FONT_PATH = Path("C:/Windows/Fonts/arial.ttf")
FONT_NAME = "ArialLocal"
def register_font() -> None:
if FONT_NAME not in pdfmetrics.getRegisteredFontNames():
pdfmetrics.registerFont(TTFont(FONT_NAME, str(FONT_PATH)))
def normalize_lines(lines: Iterable[str]) -> list[str]:
cleaned: list[str] = []
for line in lines:
value = re.sub(r"\s+", " ", line.replace("\xa0", " ")).strip()
if value:
cleaned.append(value)
return cleaned
def read_docx(path: Path) -> list[str]:
doc = DocxDocument(path)
return normalize_lines(paragraph.text for paragraph in doc.paragraphs)
def read_rtf(path: Path) -> list[str]:
raw = path.read_text(encoding="utf-8", errors="ignore")
raw = re.sub(r"\\par[d]?", "\n", raw)
raw = re.sub(r"\\'[0-9a-fA-F]{2}", "", raw)
raw = re.sub(r"\\[a-zA-Z]+-?\d* ?", "", raw)
raw = raw.replace("{", "").replace("}", "")
return normalize_lines(raw.splitlines())
def read_odt(path: Path) -> list[str]:
with zipfile.ZipFile(path) as archive:
content = archive.read("content.xml")
root = ET.fromstring(content)
ns = {
"text": "urn:oasis:names:tc:opendocument:xmlns:text:1.0",
}
paragraphs = [node.itertext() for node in root.findall(".//text:p", ns)]
return normalize_lines("".join(text) for text in paragraphs)
def write_pdf(lines: list[str], output_path: Path, title: str) -> None:
register_font()
styles = getSampleStyleSheet()
body_style = ParagraphStyle(
"BodyRussian",
parent=styles["BodyText"],
fontName=FONT_NAME,
fontSize=11,
leading=15,
spaceAfter=7,
)
title_style = ParagraphStyle(
"TitleRussian",
parent=styles["Heading1"],
fontName=FONT_NAME,
fontSize=16,
leading=20,
spaceAfter=14,
)
story = [Paragraph(title, title_style), Spacer(1, 4 * mm)]
for line in lines:
story.append(Paragraph(line.replace("&", "&amp;").replace("<", "&lt;").replace(">", "&gt;"), body_style))
doc = SimpleDocTemplate(
str(output_path),
pagesize=A4,
leftMargin=18 * mm,
rightMargin=18 * mm,
topMargin=18 * mm,
bottomMargin=18 * mm,
title=title,
)
doc.build(story)
def convert(source_name: str, output_name: str, title: str) -> None:
source_path = DOCS_DIR / source_name
output_path = DOCS_DIR / output_name
extension = source_path.suffix.lower()
if extension == ".docx":
lines = read_docx(source_path)
elif extension == ".rtf":
lines = read_rtf(source_path)
elif extension == ".odt":
lines = read_odt(source_path)
else:
raise ValueError(f"Unsupported extension: {extension}")
write_pdf(lines, output_path, title)
print(f"Converted {source_name} -> {output_name}")
if __name__ == "__main__":
convert("price_site.odt", "price_site.pdf", "Перечень платных медицинских услуг")
convert("perechen.docx", "perechen.pdf", "Лекарственное обеспечение")
convert("o_pravilah_gospitalizacii.docx", "o_pravilah_gospitalizacii.pdf", "О правилах и сроках госпитализации")
convert("o_diagnostike.docx", "o_diagnostike.pdf", "О правилах подготовки к диагностическим исследованиям")
convert("ob_obyome.docx", "ob_obyome.pdf", "О порядке, объеме и условиях оказания медицинской помощи")
convert("postanovlenie.rtf", "postanovlenie.pdf", "О государственной поддержке развития медицинской промышленности")