144 lines
4.9 KiB
Python
144 lines
4.9 KiB
Python
#!/usr/bin/env python3
|
|||
|
|
"""Convert docx to PDF for China Copyright Center submission.
|
||
|
|
Handles Chinese fonts, page breaks, headers and footers.
|
||
|
|
"""
|
||
|
|
import sys
|
||
|
|
import os
|
||
|
|
import re
|
||
|
|
from docx import Document
|
||
|
|
from docx.enum.text import WD_ALIGN_PARAGRAPH
|
||
|
|
from fpdf import FPDF
|
||
|
|
|
||
|
|
|
||
|
|
class PDF(FPDF):
|
||
|
|
def __init__(self, title="", header_text=""):
|
||
|
|
super().__init__(unit="mm", format="A4")
|
||
|
|
self.title = title
|
||
|
|
self.header_text = header_text
|
||
|
|
self.set_auto_page_break(auto=True, margin=18)
|
||
|
|
self.add_fonts()
|
||
|
|
self.page_no = 0
|
||
|
|
|
||
|
|
def add_fonts(self):
|
||
|
|
# Try common macOS Chinese fonts
|
||
|
|
candidates = [
|
||
|
|
"/System/Library/AssetsV2/com_apple_MobileAsset_Font8/86ba2c91f017a3749571a82f2c6d890ac7ffb2fb.asset/AssetData/PingFang.ttc",
|
||
|
|
"/System/Library/Fonts/PingFang.ttc",
|
||
|
|
"/System/Library/PrivateFrameworks/FontServices.framework/Versions/A/Resources/Reserved/PingFangUI.ttc",
|
||
|
|
"/System/Library/Fonts/STHeiti Medium.ttc",
|
||
|
|
"/System/Library/Fonts/Hiragino Sans GB.ttc",
|
||
|
|
"/Library/Fonts/Arial Unicode.ttf",
|
||
|
|
]
|
||
|
|
font_path = None
|
||
|
|
for c in candidates:
|
||
|
|
if os.path.exists(c):
|
||
|
|
font_path = c
|
||
|
|
break
|
||
|
|
if not font_path:
|
||
|
|
raise RuntimeError("No Chinese font found on this system")
|
||
|
|
# Use the same Chinese-capable font for code to avoid missing CJK glyphs
|
||
|
|
code_font_path = font_path
|
||
|
|
self.add_font("zh", "", font_path, uni=True)
|
||
|
|
self.add_font("zh_bold", "", font_path, uni=True)
|
||
|
|
self.add_font("code", "", code_font_path, uni=True)
|
||
|
|
|
||
|
|
def header(self):
|
||
|
|
if self.header_text:
|
||
|
|
self.set_font("zh", "", 8)
|
||
|
|
self.set_text_color(100, 100, 100)
|
||
|
|
self.cell(0, 8, self.header_text, border="B", align="C", new_x="LMARGIN", new_y="NEXT")
|
||
|
|
self.ln(2)
|
||
|
|
|
||
|
|
def footer(self):
|
||
|
|
self.set_y(-15)
|
||
|
|
self.set_font("zh", "", 8)
|
||
|
|
self.set_text_color(100, 100, 100)
|
||
|
|
self.cell(0, 10, f"第 {self.page_no} 页", align="C")
|
||
|
|
|
||
|
|
|
||
|
|
def get_page_breaks(doc):
|
||
|
|
"""Return indices of paragraphs that contain page breaks."""
|
||
|
|
breaks = []
|
||
|
|
for i, para in enumerate(doc.paragraphs):
|
||
|
|
for run in para.runs:
|
||
|
|
if "lastRenderedPageBreak" in run._element.xml or run._element.xml.find("<w:br") != -1 and "page" in run._element.xml:
|
||
|
|
breaks.append(i)
|
||
|
|
return breaks
|
||
|
|
|
||
|
|
|
||
|
|
def is_heading(para):
|
||
|
|
return para.style.name.startswith("Heading")
|
||
|
|
|
||
|
|
|
||
|
|
def convert(input_path, output_path, header_text=""):
|
||
|
|
doc = Document(input_path)
|
||
|
|
pdf = PDF(title=os.path.basename(input_path), header_text=header_text)
|
||
|
|
pdf.add_page()
|
||
|
|
|
||
|
|
margin_x = 20
|
||
|
|
max_y = 277
|
||
|
|
line_height_map = {
|
||
|
|
"Heading 1": 10,
|
||
|
|
"Heading 2": 8,
|
||
|
|
"Heading 3": 7,
|
||
|
|
"Normal": 5,
|
||
|
|
}
|
||
|
|
|
||
|
|
for para in doc.paragraphs:
|
||
|
|
text = para.text
|
||
|
|
if not text:
|
||
|
|
pdf.ln(3)
|
||
|
|
continue
|
||
|
|
|
||
|
|
# Page break detection via explicit PageBreak runs is not reliable in python-docx;
|
||
|
|
# rely on vertical overflow via auto_page_break.
|
||
|
|
style_name = para.style.name if para.style else "Normal"
|
||
|
|
if "Heading 1" in style_name:
|
||
|
|
pdf.set_font("zh_bold", "", 18)
|
||
|
|
pdf.set_text_color(0, 0, 0)
|
||
|
|
pdf.cell(0, 10, text, new_x="LMARGIN", new_y="NEXT")
|
||
|
|
pdf.ln(2)
|
||
|
|
elif "Heading 2" in style_name:
|
||
|
|
pdf.set_font("zh_bold", "", 14)
|
||
|
|
pdf.set_text_color(0, 0, 0)
|
||
|
|
pdf.cell(0, 8, text, new_x="LMARGIN", new_y="NEXT")
|
||
|
|
pdf.ln(1)
|
||
|
|
elif "Heading 3" in style_name:
|
||
|
|
pdf.set_font("zh_bold", "", 12)
|
||
|
|
pdf.set_text_color(0, 0, 0)
|
||
|
|
pdf.cell(0, 7, text, new_x="LMARGIN", new_y="NEXT")
|
||
|
|
else:
|
||
|
|
# Normal paragraph
|
||
|
|
pdf.set_font("code" if looks_like_code(text) else "zh", "", 10)
|
||
|
|
pdf.set_text_color(0, 0, 0)
|
||
|
|
# Multi-cell wrapping
|
||
|
|
pdf.multi_cell(0, 4.5, text)
|
||
|
|
pdf.ln(1)
|
||
|
|
|
||
|
|
pdf.output(output_path)
|
||
|
|
print(f"Converted {input_path} -> {output_path}")
|
||
|
|
|
||
|
|
|
||
|
|
def looks_like_code(text):
|
||
|
|
# Heuristic: if line has lots of code symbols
|
||
|
|
code_chars = set("{}[]<>();=+-*/|&!?.,:\"'")
|
||
|
|
if not text:
|
||
|
|
return False
|
||
|
|
ratio = sum(1 for c in text if c in code_chars) / len(text)
|
||
|
|
return ratio > 0.05 or text.lstrip().startswith("//")
|
||
|
|
|
||
|
|
|
||
|
|
if __name__ == "__main__":
|
||
|
|
root = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||
|
|
docs_dir = os.path.join(root, "docs")
|
||
|
|
convert(
|
||
|
|
os.path.join(docs_dir, "源程序文档.docx"),
|
||
|
|
os.path.join(docs_dir, "源程序文档.pdf"),
|
||
|
|
header_text="可视化大屏导航站 V1.0"
|
||
|
|
)
|
||
|
|
convert(
|
||
|
|
os.path.join(docs_dir, "软件说明书.docx"),
|
||
|
|
os.path.join(docs_dir, "软件说明书.pdf"),
|
||
|
|
header_text="可视化大屏导航站 — 软件说明书"
|
||
|
|
)
|