from __future__ import annotations import re import zipfile from pathlib import Path from typing import Iterable from xml.etree import ElementTree as ET from docx import Document as DocxDocument from reportlab.lib.pagesizes import A4 from reportlab.lib.styles import ParagraphStyle, getSampleStyleSheet from reportlab.lib.units import mm from reportlab.pdfbase import pdfmetrics from reportlab.pdfbase.ttfonts import TTFont from reportlab.platypus import Paragraph, SimpleDocTemplate, Spacer ROOT = Path(__file__).resolve().parents[1] DOCS_DIR = ROOT / "public" / "documents" FONT_PATH = Path("C:/Windows/Fonts/arial.ttf") FONT_NAME = "ArialLocal" def register_font() -> None: if FONT_NAME not in pdfmetrics.getRegisteredFontNames(): pdfmetrics.registerFont(TTFont(FONT_NAME, str(FONT_PATH))) def normalize_lines(lines: Iterable[str]) -> list[str]: cleaned: list[str] = [] for line in lines: value = re.sub(r"\s+", " ", line.replace("\xa0", " ")).strip() if value: cleaned.append(value) return cleaned def read_docx(path: Path) -> list[str]: doc = DocxDocument(path) return normalize_lines(paragraph.text for paragraph in doc.paragraphs) def read_rtf(path: Path) -> list[str]: raw = path.read_text(encoding="utf-8", errors="ignore") raw = re.sub(r"\\par[d]?", "\n", raw) raw = re.sub(r"\\'[0-9a-fA-F]{2}", "", raw) raw = re.sub(r"\\[a-zA-Z]+-?\d* ?", "", raw) raw = raw.replace("{", "").replace("}", "") return normalize_lines(raw.splitlines()) def read_odt(path: Path) -> list[str]: with zipfile.ZipFile(path) as archive: content = archive.read("content.xml") root = ET.fromstring(content) ns = { "text": "urn:oasis:names:tc:opendocument:xmlns:text:1.0", } paragraphs = [node.itertext() for node in root.findall(".//text:p", ns)] return normalize_lines("".join(text) for text in paragraphs) def write_pdf(lines: list[str], output_path: Path, title: str) -> None: register_font() styles = getSampleStyleSheet() body_style = ParagraphStyle( "BodyRussian", parent=styles["BodyText"], fontName=FONT_NAME, fontSize=11, leading=15, spaceAfter=7, ) title_style = ParagraphStyle( "TitleRussian", parent=styles["Heading1"], fontName=FONT_NAME, fontSize=16, leading=20, spaceAfter=14, ) story = [Paragraph(title, title_style), Spacer(1, 4 * mm)] for line in lines: story.append(Paragraph(line.replace("&", "&").replace("<", "<").replace(">", ">"), body_style)) doc = SimpleDocTemplate( str(output_path), pagesize=A4, leftMargin=18 * mm, rightMargin=18 * mm, topMargin=18 * mm, bottomMargin=18 * mm, title=title, ) doc.build(story) def convert(source_name: str, output_name: str, title: str) -> None: source_path = DOCS_DIR / source_name output_path = DOCS_DIR / output_name extension = source_path.suffix.lower() if extension == ".docx": lines = read_docx(source_path) elif extension == ".rtf": lines = read_rtf(source_path) elif extension == ".odt": lines = read_odt(source_path) else: raise ValueError(f"Unsupported extension: {extension}") write_pdf(lines, output_path, title) print(f"Converted {source_name} -> {output_name}") if __name__ == "__main__": convert("price_site.odt", "price_site.pdf", "Перечень платных медицинских услуг") convert("perechen.docx", "perechen.pdf", "Лекарственное обеспечение") convert("o_pravilah_gospitalizacii.docx", "o_pravilah_gospitalizacii.pdf", "О правилах и сроках госпитализации") convert("o_diagnostike.docx", "o_diagnostike.pdf", "О правилах подготовки к диагностическим исследованиям") convert("ob_obyome.docx", "ob_obyome.pdf", "О порядке, объеме и условиях оказания медицинской помощи") convert("postanovlenie.rtf", "postanovlenie.pdf", "О государственной поддержке развития медицинской промышленности")