feat(convert): apply reference-doc typography in docx post-processing

Extend _fix_docx_postprocess to copy Normal style paragraph properties
from the reference doc, convert BodyText to Normal, add heading spacing
before/after, and format reference entries with hanging-style spacing.
This commit is contained in:
Zhengshou Lai
2026-06-09 14:38:30 +08:00
parent 5d0ed21e07
commit fc4dab1629
+118 -5
View File
@@ -113,7 +113,7 @@ def convert_cmd(inputs, output, template, quality, density, reference_doc, bookm
if ref_doc: if ref_doc:
cmd.extend(["--reference-doc", ref_doc]) cmd.extend(["--reference-doc", ref_doc])
run_command(cmd, check=True) run_command(cmd, check=True)
_fix_docx_postprocess(out_path) _fix_docx_postprocess(out_path, reference_doc_path=ref_doc)
click.secho(f"Word document generated: {output}", fg="green") click.secho(f"Word document generated: {output}", fg="green")
return return
@@ -207,10 +207,15 @@ def _images_to_pdf(images: list[Path], output: Path) -> None:
writer.write(f) writer.write(f)
def _fix_docx_postprocess(docx_path: Path) -> None: def _fix_docx_postprocess(docx_path: Path, reference_doc_path: str | None = None) -> None:
"""Post-process pandoc-generated docx for Chinese academic typography. """Post-process pandoc-generated docx for Chinese academic typography.
Applies the following fixes: Applies the following fixes:
- Apply paragraph-level typography from reference-doc Normal style
(first-line indent, spacing, line spacing, justification)
- Convert BodyText paragraphs to Normal and apply same typography
- Add heading spacing before/after Heading 1-4
- Add 6pt spacing after reference entries (detected after "参考文献")
- Three-line table borders (1.5pt top/bottom, 0.75pt header separator) - Three-line table borders (1.5pt top/bottom, 0.75pt header separator)
- Remove table styles that override custom borders - Remove table styles that override custom borders
- Remove first-line indent inside table cells - Remove first-line indent inside table cells
@@ -226,15 +231,54 @@ def _fix_docx_postprocess(docx_path: Path) -> None:
import zipfile import zipfile
import os import os
import copy
from tempfile import TemporaryDirectory from tempfile import TemporaryDirectory
W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
wtag = lambda t: "{" + W + "}" + t wtag = lambda t: "{" + W + "}" + t
def _paragraph_text(p):
parts = []
for r in p.findall(wtag("r")):
t = r.find(wtag("t"))
if t is not None and t.text:
parts.append(t.text)
return "".join(parts)
def _copy_ppr_children(src_ppr, dst_ppr):
for child in src_ppr:
tag_name = child.tag.split("}")[-1]
# Remove existing same-tag elements
for existing in list(dst_ppr.findall(wtag(tag_name))):
dst_ppr.remove(existing)
dst_ppr.append(copy.deepcopy(child))
with TemporaryDirectory() as workdir: with TemporaryDirectory() as workdir:
with zipfile.ZipFile(docx_path, "r") as z: with zipfile.ZipFile(docx_path, "r") as z:
z.extractall(workdir) z.extractall(workdir)
# --- Load reference-doc Normal style paragraph properties ---
normal_ppr = None
if reference_doc_path and Path(reference_doc_path).exists():
try:
with zipfile.ZipFile(reference_doc_path, "r") as z:
styles_content = z.read("word/styles.xml")
ref_styles_root = ET.fromstring(styles_content)
for style in ref_styles_root.findall(wtag("style")):
if style.get(wtag("styleId")) == "Normal":
normal_ppr = style.find(wtag("pPr"))
break
except Exception:
normal_ppr = None
# Heading spacing in twips (1pt = 20 twips)
heading_spacing = {
"Heading1": {"before": "360", "after": "120"},
"Heading2": {"before": "280", "after": "120"},
"Heading3": {"before": "200", "after": "80"},
"Heading4": {"before": "160", "after": "80"},
}
doc_xml = os.path.join(workdir, "word", "document.xml") doc_xml = os.path.join(workdir, "word", "document.xml")
tree = ET.parse(doc_xml) tree = ET.parse(doc_xml)
root = tree.getroot() root = tree.getroot()
@@ -336,7 +380,76 @@ def _fix_docx_postprocess(docx_path: Path) -> None:
if spacing is not None: if spacing is not None:
spacing.set(wtag("before"), "0") spacing.set(wtag("before"), "0")
# --- 4. Add spacing after tables --- # --- 4. Apply paragraph-level typography from reference doc ---
in_references = False
for p in body.findall(wtag("p")):
# Skip paragraphs inside table cells (handled separately)
parent = p.getparent()
if parent is not None and parent.tag == wtag("tc"):
continue
pPr = p.find(wtag("pPr"))
if pPr is None:
pPr = ET.SubElement(p, wtag("pPr"))
p.insert(0, pPr)
pStyle = pPr.find(wtag("pStyle"))
style_id = pStyle.get(wtag("val")) if pStyle is not None else "Normal"
# Detect references section by heading text
if style_id.startswith("Heading") and "参考文献" in _paragraph_text(p):
in_references = True
# Apply heading spacing
hs = heading_spacing.get(style_id)
if hs is not None:
spacing = pPr.find(wtag("spacing"))
if spacing is None:
spacing = ET.SubElement(pPr, wtag("spacing"))
spacing.set(wtag("before"), hs["before"])
spacing.set(wtag("after"), hs["after"])
continue
# Reference entry formatting
if in_references:
text = _paragraph_text(p).strip()
if not text:
# Empty paragraph in refs: remove indent but keep normal spacing
for ind in list(pPr.findall(wtag("ind"))):
pPr.remove(ind)
ET.SubElement(pPr, wtag("ind")).set(wtag("firstLine"), "0")
continue
# Convert to Normal and apply normal pPr
if pStyle is None:
pStyle = ET.SubElement(pPr, wtag("pStyle"))
pStyle.set(wtag("val"), "Normal")
if normal_ppr is not None:
_copy_ppr_children(normal_ppr, pPr)
spacing = pPr.find(wtag("spacing"))
if spacing is None:
spacing = ET.SubElement(pPr, wtag("spacing"))
spacing.set(wtag("after"), "120") # 6pt after each reference
continue
# BodyText / FirstParagraph -> Normal
if style_id in ("BodyText", "FirstParagraph"):
if pStyle is None:
pStyle = ET.SubElement(pPr, wtag("pStyle"))
pStyle.set(wtag("val"), "Normal")
style_id = "Normal"
# Apply Normal paragraph properties from reference doc
if style_id in ("Normal", "FirstParagraph") and normal_ppr is not None:
_copy_ppr_children(normal_ppr, pPr)
# Apply heading spacing
if style_id in heading_spacing:
spacing = pPr.find(wtag("spacing"))
if spacing is None:
spacing = ET.SubElement(pPr, wtag("spacing"))
spacing.set(wtag("before"), heading_spacing[style_id]["before"])
spacing.set(wtag("after"), heading_spacing[style_id]["after"])
# --- 5. Add spacing after tables ---
for tbl in body.findall(wtag("tbl")): for tbl in body.findall(wtag("tbl")):
tbl_idx = list(body).index(tbl) tbl_idx = list(body).index(tbl)
for i in range(tbl_idx + 1, len(list(body))): for i in range(tbl_idx + 1, len(list(body))):
@@ -351,7 +464,7 @@ def _fix_docx_postprocess(docx_path: Path) -> None:
spacing.set(wtag("before"), "120") spacing.set(wtag("before"), "120")
break break
# --- 5. Inject suppressAutoHyphens into all paragraphs --- # --- 6. Inject suppressAutoHyphens into all paragraphs ---
for p in body.findall(wtag("p")): for p in body.findall(wtag("p")):
pPr = p.find(wtag("pPr")) pPr = p.find(wtag("pPr"))
if pPr is None: if pPr is None:
@@ -362,7 +475,7 @@ def _fix_docx_postprocess(docx_path: Path) -> None:
tree.write(doc_xml, xml_declaration=True, encoding="UTF-8", pretty_print=False) tree.write(doc_xml, xml_declaration=True, encoding="UTF-8", pretty_print=False)
# --- 6. Inject HorizontalRule style --- # --- 7. Inject HorizontalRule style ---
styles_xml = os.path.join(workdir, "word", "styles.xml") styles_xml = os.path.join(workdir, "word", "styles.xml")
styles_tree = ET.parse(styles_xml) styles_tree = ET.parse(styles_xml)
styles_root = styles_tree.getroot() styles_root = styles_tree.getroot()