Files
visual-vue-v1/scripts/docx2pdf.py
T
2026-07-26 14:41:16 +08:00

144 lines
4.9 KiB
Python

#!/usr/bin/env python3
"""Convert docx to PDF for China Copyright Center submission.
Handles Chinese fonts, page breaks, headers and footers.
"""
import sys
import os
import re
from docx import Document
from docx.enum.text import WD_ALIGN_PARAGRAPH
from fpdf import FPDF
class PDF(FPDF):
def __init__(self, title="", header_text=""):
super().__init__(unit="mm", format="A4")
self.title = title
self.header_text = header_text
self.set_auto_page_break(auto=True, margin=18)
self.add_fonts()
self.page_no = 0
def add_fonts(self):
# Try common macOS Chinese fonts
candidates = [
"/System/Library/AssetsV2/com_apple_MobileAsset_Font8/86ba2c91f017a3749571a82f2c6d890ac7ffb2fb.asset/AssetData/PingFang.ttc",
"/System/Library/Fonts/PingFang.ttc",
"/System/Library/PrivateFrameworks/FontServices.framework/Versions/A/Resources/Reserved/PingFangUI.ttc",
"/System/Library/Fonts/STHeiti Medium.ttc",
"/System/Library/Fonts/Hiragino Sans GB.ttc",
"/Library/Fonts/Arial Unicode.ttf",
]
font_path = None
for c in candidates:
if os.path.exists(c):
font_path = c
break
if not font_path:
raise RuntimeError("No Chinese font found on this system")
# Use the same Chinese-capable font for code to avoid missing CJK glyphs
code_font_path = font_path
self.add_font("zh", "", font_path, uni=True)
self.add_font("zh_bold", "", font_path, uni=True)
self.add_font("code", "", code_font_path, uni=True)
def header(self):
if self.header_text:
self.set_font("zh", "", 8)
self.set_text_color(100, 100, 100)
self.cell(0, 8, self.header_text, border="B", align="C", new_x="LMARGIN", new_y="NEXT")
self.ln(2)
def footer(self):
self.set_y(-15)
self.set_font("zh", "", 8)
self.set_text_color(100, 100, 100)
self.cell(0, 10, f"第 {self.page_no} 页", align="C")
def get_page_breaks(doc):
"""Return indices of paragraphs that contain page breaks."""
breaks = []
for i, para in enumerate(doc.paragraphs):
for run in para.runs:
if "lastRenderedPageBreak" in run._element.xml or run._element.xml.find("<w:br") != -1 and "page" in run._element.xml:
breaks.append(i)
return breaks
def is_heading(para):
return para.style.name.startswith("Heading")
def convert(input_path, output_path, header_text=""):
doc = Document(input_path)
pdf = PDF(title=os.path.basename(input_path), header_text=header_text)
pdf.add_page()
margin_x = 20
max_y = 277
line_height_map = {
"Heading 1": 10,
"Heading 2": 8,
"Heading 3": 7,
"Normal": 5,
}
for para in doc.paragraphs:
text = para.text
if not text:
pdf.ln(3)
continue
# Page break detection via explicit PageBreak runs is not reliable in python-docx;
# rely on vertical overflow via auto_page_break.
style_name = para.style.name if para.style else "Normal"
if "Heading 1" in style_name:
pdf.set_font("zh_bold", "", 18)
pdf.set_text_color(0, 0, 0)
pdf.cell(0, 10, text, new_x="LMARGIN", new_y="NEXT")
pdf.ln(2)
elif "Heading 2" in style_name:
pdf.set_font("zh_bold", "", 14)
pdf.set_text_color(0, 0, 0)
pdf.cell(0, 8, text, new_x="LMARGIN", new_y="NEXT")
pdf.ln(1)
elif "Heading 3" in style_name:
pdf.set_font("zh_bold", "", 12)
pdf.set_text_color(0, 0, 0)
pdf.cell(0, 7, text, new_x="LMARGIN", new_y="NEXT")
else:
# Normal paragraph
pdf.set_font("code" if looks_like_code(text) else "zh", "", 10)
pdf.set_text_color(0, 0, 0)
# Multi-cell wrapping
pdf.multi_cell(0, 4.5, text)
pdf.ln(1)
pdf.output(output_path)
print(f"Converted {input_path} -> {output_path}")
def looks_like_code(text):
# Heuristic: if line has lots of code symbols
code_chars = set("{}[]<>();=+-*/|&!?.,:\"'")
if not text:
return False
ratio = sum(1 for c in text if c in code_chars) / len(text)
return ratio > 0.05 or text.lstrip().startswith("//")
if __name__ == "__main__":
root = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
docs_dir = os.path.join(root, "docs")
convert(
os.path.join(docs_dir, "源程序文档.docx"),
os.path.join(docs_dir, "源程序文档.pdf"),
header_text="可视化大屏导航站 V1.0"
)
convert(
os.path.join(docs_dir, "软件说明书.docx"),
os.path.join(docs_dir, "软件说明书.pdf"),
header_text="可视化大屏导航站 — 软件说明书"
)