1153 lines
45 KiB
Python
1153 lines
45 KiB
Python
"""Deterministic Markdown -> DOCX export with an embedded title font.
|
||
|
||
The implementation intentionally uses Python's standard library only. It
|
||
builds the small set of OOXML parts required by Word instead of depending on a
|
||
desktop Office installation or a document-conversion service. The bundled
|
||
方正小标宋简体 font is validated and embedded according to ECMA-376 so the
|
||
recipient's computer does not need the font installed.
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import base64
|
||
import binascii
|
||
import hashlib
|
||
import io
|
||
import os
|
||
import re
|
||
import struct
|
||
import uuid
|
||
import zipfile
|
||
from dataclasses import dataclass
|
||
from datetime import UTC, datetime
|
||
from pathlib import Path
|
||
from xml.sax.saxutils import escape
|
||
|
||
DOCX_MEDIA_TYPE = "application/vnd.openxmlformats-officedocument.wordprocessingml.document"
|
||
|
||
TITLE_FONT_FAMILY = "方正小标宋简体"
|
||
BODY_FONT_FAMILY = "仿宋"
|
||
HEADING_1_FONT_FAMILY = "黑体"
|
||
HEADING_2_FONT_FAMILY = "楷体"
|
||
TABLE_FONT_FAMILY = "宋体"
|
||
|
||
TITLE_SIZE_HALF_POINTS = 44 # 二号 / 22 pt
|
||
BODY_SIZE_HALF_POINTS = 32 # 三号 / 16 pt
|
||
TABLE_SIZE_HALF_POINTS = 28 # 四号 / 14 pt
|
||
EXACT_LINE_TWIPS = 579 # 28.95 pt * 20
|
||
|
||
A4_WIDTH_TWIPS = 11_906
|
||
A4_HEIGHT_TWIPS = 16_838
|
||
MARGIN_TOP_TWIPS = 2_098 # 3.7 cm
|
||
MARGIN_BOTTOM_TWIPS = 1_984 # 3.5 cm
|
||
MARGIN_LEFT_TWIPS = 1_587 # 2.8 cm
|
||
MARGIN_RIGHT_TWIPS = 1_474 # 2.6 cm
|
||
CONTENT_WIDTH_TWIPS = A4_WIDTH_TWIPS - MARGIN_LEFT_TWIPS - MARGIN_RIGHT_TWIPS
|
||
|
||
_XML_DECL = '<?xml version="1.0" encoding="UTF-8" standalone="yes"?>'
|
||
_W_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
|
||
_R_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships"
|
||
_PACKAGE_REL_NS = "http://schemas.openxmlformats.org/package/2006/relationships"
|
||
|
||
_FONT_DIR = Path(__file__).resolve().parent / "assets" / "fonts"
|
||
_BUNDLED_FONT_PATH = _FONT_DIR / "方正小标宋简体.ttf"
|
||
_BUNDLED_FONT_SHA256 = "C52D577DD3AA719EF5FB9FF2043B7F7CA36FAE468EF34A47628A3A555D006A49"
|
||
|
||
|
||
class WordExportError(RuntimeError):
|
||
"""Base error returned to the Word export API."""
|
||
|
||
|
||
class WordExportFontError(WordExportError):
|
||
"""The required server-side font is missing, invalid, or not embeddable."""
|
||
|
||
|
||
@dataclass(slots=True)
|
||
class FontAsset:
|
||
data: bytes
|
||
path: Path
|
||
names: frozenset[str]
|
||
fs_type: int
|
||
sha256: str
|
||
|
||
|
||
@dataclass(slots=True)
|
||
class ParagraphBlock:
|
||
kind: str
|
||
text: str
|
||
level: int = 0
|
||
|
||
|
||
@dataclass(slots=True)
|
||
class TableBlock:
|
||
header: list[str]
|
||
rows: list[list[str]]
|
||
|
||
|
||
@dataclass(slots=True)
|
||
class ImageBlock:
|
||
alt: str
|
||
data: bytes
|
||
media_name: str
|
||
width_px: int
|
||
height_px: int
|
||
|
||
|
||
Block = ParagraphBlock | TableBlock | ImageBlock
|
||
|
||
_EMU_PER_TWIP = 635
|
||
_PNG_SIGNATURE = b"\x89PNG\r\n\x1a\n"
|
||
_MAX_EMBEDDED_IMAGES = 40
|
||
_MAX_EMBEDDED_IMAGE_BYTES = 8 * 1024 * 1024
|
||
_EMBEDDED_IMAGE_ID = re.compile(r"^embedded:([A-Za-z0-9_-]{1,80})$")
|
||
_DATA_PNG_URI = re.compile(r"^data:image/png;base64,([A-Za-z0-9+/=\s]+)$", re.IGNORECASE)
|
||
|
||
|
||
def _safe_xml_text(value: str) -> str:
|
||
"""Remove XML-illegal controls and invisible Word-breaking selectors."""
|
||
value = re.sub(r"[\u0000-\u0008\u000b\u000c\u000e-\u001f]", "", value)
|
||
value = re.sub(r"[\u200b-\u200d\ufeff\ufe00-\ufe0f]", "", value)
|
||
value = value.replace("\u00ad", "").replace("\ufffc", "")
|
||
value = re.sub(r"</?(?:details|summary)\b[^>]*>", "", value, flags=re.I)
|
||
return value
|
||
|
||
|
||
def _xml_attr(value: str) -> str:
|
||
return escape(_safe_xml_text(value), {'"': """, "'": "'"})
|
||
|
||
|
||
def _decode_name_record(platform_id: int, raw: bytes) -> str | None:
|
||
try:
|
||
if platform_id in {0, 3}:
|
||
return raw.decode("utf-16-be").strip("\x00")
|
||
if platform_id == 1:
|
||
return raw.decode("mac_roman").strip("\x00")
|
||
except (UnicodeDecodeError, LookupError):
|
||
return None
|
||
return None
|
||
|
||
|
||
def _read_sfnt_metadata(data: bytes) -> tuple[frozenset[str], int]:
|
||
if len(data) < 12:
|
||
raise WordExportFontError("Word 标题字体文件不是有效的 OpenType/TrueType 字体。")
|
||
try:
|
||
table_count = struct.unpack_from(">H", data, 4)[0]
|
||
tables: dict[str, tuple[int, int]] = {}
|
||
for index in range(table_count):
|
||
position = 12 + index * 16
|
||
tag, _checksum, offset, length = struct.unpack_from(">4sIII", data, position)
|
||
tag_text = tag.decode("latin-1")
|
||
if offset + length > len(data):
|
||
raise WordExportFontError("Word 标题字体的表目录已损坏。")
|
||
tables[tag_text] = (offset, length)
|
||
|
||
if "OS/2" not in tables or "name" not in tables:
|
||
raise WordExportFontError("Word 标题字体缺少 OS/2 或 name 元数据表。")
|
||
|
||
os2_offset, os2_length = tables["OS/2"]
|
||
if os2_length < 10:
|
||
raise WordExportFontError("Word 标题字体的 OS/2 元数据无效。")
|
||
fs_type = struct.unpack_from(">H", data, os2_offset + 8)[0]
|
||
|
||
name_offset, name_length = tables["name"]
|
||
if name_length < 6:
|
||
raise WordExportFontError("Word 标题字体的 name 元数据无效。")
|
||
name_count, string_offset = struct.unpack_from(">HH", data, name_offset + 2)
|
||
names: set[str] = set()
|
||
for index in range(name_count):
|
||
record_offset = name_offset + 6 + index * 12
|
||
if record_offset + 12 > name_offset + name_length:
|
||
break
|
||
platform_id, _encoding, _language, name_id, length, offset = struct.unpack_from(
|
||
">HHHHHH", data, record_offset
|
||
)
|
||
if name_id not in {1, 4, 6}:
|
||
continue
|
||
value_start = name_offset + string_offset + offset
|
||
value_end = value_start + length
|
||
if value_end > name_offset + name_length:
|
||
continue
|
||
decoded = _decode_name_record(platform_id, data[value_start:value_end])
|
||
if decoded:
|
||
names.add(decoded)
|
||
return frozenset(names), fs_type
|
||
except struct.error as exc:
|
||
raise WordExportFontError("Word 标题字体文件结构不完整。") from exc
|
||
|
||
|
||
def resolve_title_font_path() -> Path:
|
||
configured = os.getenv("DEERFLOW_WORD_TITLE_FONT_PATH", "").strip()
|
||
return Path(configured).expanduser() if configured else _BUNDLED_FONT_PATH
|
||
|
||
|
||
def load_title_font(path: Path | None = None) -> FontAsset:
|
||
font_path = (path or resolve_title_font_path()).resolve()
|
||
if not font_path.is_file():
|
||
raise WordExportFontError(
|
||
f"服务器缺少“{TITLE_FONT_FAMILY}”字体资源:{font_path}。"
|
||
"请将已授权字体放入后端字体目录后再导出。"
|
||
)
|
||
data = font_path.read_bytes()
|
||
if not data or len(data) > 25 * 1024 * 1024:
|
||
raise WordExportFontError("Word 标题字体文件为空或超过 25 MB 限制。")
|
||
|
||
names, fs_type = _read_sfnt_metadata(data)
|
||
accepted_name = TITLE_FONT_FAMILY in names or any("FZXiaoBiaoSong" in name for name in names)
|
||
if not accepted_name:
|
||
readable_names = "、".join(sorted(names)[:6]) or "无法读取"
|
||
raise WordExportFontError(
|
||
f"服务器字体不是“{TITLE_FONT_FAMILY}”(检测到:{readable_names}),已阻止导出。"
|
||
)
|
||
if fs_type & 0x0002:
|
||
raise WordExportFontError(f"“{TITLE_FONT_FAMILY}”授权标志禁止文档嵌入,已阻止导出。")
|
||
if fs_type & 0x0200:
|
||
raise WordExportFontError(f"“{TITLE_FONT_FAMILY}”仅允许位图嵌入,不能用于本次 Word 导出。")
|
||
|
||
digest = hashlib.sha256(data).hexdigest().upper()
|
||
if font_path == _BUNDLED_FONT_PATH.resolve() and digest != _BUNDLED_FONT_SHA256:
|
||
raise WordExportFontError("代码仓库中的 Word 标题字体校验和不匹配,已阻止导出。")
|
||
return FontAsset(data=data, path=font_path, names=names, fs_type=fs_type, sha256=digest)
|
||
|
||
|
||
def _obfuscate_font(data: bytes, font_key: str) -> bytes:
|
||
"""Apply the ECMA-376 ODTTF XOR obfuscation to the first 32 bytes."""
|
||
key_hex = font_key.replace("-", "")
|
||
if len(key_hex) != 32:
|
||
raise WordExportFontError("生成 Word 嵌入字体密钥失败。")
|
||
key = bytes.fromhex(key_hex)[::-1]
|
||
output = bytearray(data)
|
||
for index in range(min(32, len(output))):
|
||
output[index] ^= key[index % len(key)]
|
||
return bytes(output)
|
||
|
||
|
||
def _strip_inline_markdown(text: str) -> str:
|
||
text = _safe_xml_text(text)
|
||
text = re.sub(r"^\s*#+\s*", "", text)
|
||
text = re.sub(r"!\[([^\]]*)\]\([^)]*\)", r"\1", text)
|
||
text = re.sub(r"\[([^\]]+)\]\([^)]*\)", r"\1", text)
|
||
text = re.sub(r"(?:\*\*|__)(.*?)(?:\*\*|__)", r"\1", text)
|
||
return re.sub(r"[`*_~]", "", text).strip()
|
||
|
||
|
||
def _strip_manual_heading_number(text: str) -> str:
|
||
return re.sub(
|
||
r"^\s*(?:"
|
||
r"[一二三四五六七八九十百千]+[、..]"
|
||
r"|[((]\s*[一二三四五六七八九十百千]+\s*[))]"
|
||
r"|[((]\s*\d+(?:[..]\d+)*\s*[))]"
|
||
r"|\d+(?:[..]\d+)+(?:[..、])?"
|
||
r"|\d+[..、))]"
|
||
r")\s*",
|
||
"",
|
||
text,
|
||
).strip()
|
||
|
||
|
||
def _split_table_row(line: str) -> list[str]:
|
||
value = re.sub(r"^\s*\|", "", line)
|
||
value = re.sub(r"\|\s*$", "", value)
|
||
return [part.replace(r"\|", "|").strip() for part in re.split(r"(?<!\\)\|", value)]
|
||
|
||
|
||
def _is_table_separator(line: str) -> bool:
|
||
cells = _split_table_row(line)
|
||
return bool(cells) and all(re.fullmatch(r":?-{3,}:?", cell.replace(" ", "")) for cell in cells)
|
||
|
||
|
||
def _looks_like_table_row(line: str) -> bool:
|
||
return bool(re.match(r"^\s*\|.+", line))
|
||
|
||
|
||
def _is_table_caption(text: str) -> bool:
|
||
return bool(re.match(r"^\s*表\s*(?:\d+|[一二三四五六七八九十百]+|[::])", _strip_inline_markdown(text)))
|
||
|
||
|
||
def _extract_title(markdown: str, explicit_title: str | None) -> tuple[str | None, list[str]]:
|
||
lines = markdown.replace("\r\n", "\n").replace("\r", "\n").split("\n")
|
||
in_fence = False
|
||
for index, line in enumerate(lines):
|
||
if line.strip().startswith("```"):
|
||
in_fence = not in_fence
|
||
continue
|
||
if not in_fence:
|
||
match = re.match(r"^#\s+(.+)$", line)
|
||
if match:
|
||
del lines[index]
|
||
return _strip_inline_markdown(match.group(1)) or (explicit_title or "").strip() or None, lines
|
||
return (explicit_title or "").strip() or None, lines
|
||
|
||
|
||
def _png_size(data: bytes) -> tuple[int, int]:
|
||
if len(data) < 24 or data[:8] != _PNG_SIGNATURE:
|
||
raise WordExportError("嵌入的流程图不是有效的 PNG 图片。")
|
||
width, height = struct.unpack(">II", data[16:24])
|
||
if width < 1 or height < 1 or width > 20_000 or height > 20_000:
|
||
raise WordExportError("嵌入图片尺寸无效。")
|
||
return width, height
|
||
|
||
|
||
def _decode_png_base64(payload: str) -> bytes:
|
||
try:
|
||
data = base64.b64decode(re.sub(r"\s+", "", payload), validate=False)
|
||
except (ValueError, binascii.Error) as exc:
|
||
raise WordExportError("嵌入图片编码无效。") from exc
|
||
if not data or len(data) > _MAX_EMBEDDED_IMAGE_BYTES:
|
||
raise WordExportError("嵌入图片为空或超过 8 MB 限制。")
|
||
_png_size(data)
|
||
return data
|
||
|
||
|
||
def _resolve_markdown_image(
|
||
alt: str,
|
||
target: str,
|
||
embedded_images: dict[str, str],
|
||
image_index: int,
|
||
) -> Block:
|
||
payload: str | None = None
|
||
embedded = _EMBEDDED_IMAGE_ID.match(target.strip())
|
||
if embedded:
|
||
payload = embedded_images.get(embedded.group(1))
|
||
else:
|
||
data_uri = _DATA_PNG_URI.match(target.strip())
|
||
if data_uri:
|
||
payload = data_uri.group(1)
|
||
if not payload:
|
||
return ParagraphBlock("image", f"图片:{alt}|{target}")
|
||
try:
|
||
data = _decode_png_base64(payload)
|
||
width, height = _png_size(data)
|
||
except WordExportError:
|
||
return ParagraphBlock("image", f"图片:{alt}|{target}")
|
||
return ImageBlock(
|
||
alt=alt,
|
||
data=data,
|
||
media_name=f"image{image_index}.png",
|
||
width_px=width,
|
||
height_px=height,
|
||
)
|
||
|
||
|
||
def _parse_markdown(
|
||
markdown: str,
|
||
explicit_title: str | None,
|
||
embedded_images: dict[str, str] | None = None,
|
||
) -> tuple[str | None, list[Block]]:
|
||
title, lines = _extract_title(markdown, explicit_title)
|
||
blocks: list[Block] = []
|
||
paragraph_buffer: list[str] = []
|
||
in_fence = False
|
||
images = embedded_images or {}
|
||
image_count = 0
|
||
|
||
def flush_paragraph() -> None:
|
||
if paragraph_buffer:
|
||
blocks.append(ParagraphBlock("body", " ".join(paragraph_buffer)))
|
||
paragraph_buffer.clear()
|
||
|
||
index = 0
|
||
while index < len(lines):
|
||
raw = lines[index].replace("\t", " ")
|
||
if raw.strip().startswith("```"):
|
||
flush_paragraph()
|
||
in_fence = not in_fence
|
||
index += 1
|
||
continue
|
||
if in_fence:
|
||
blocks.append(ParagraphBlock("code", raw or " "))
|
||
index += 1
|
||
continue
|
||
if not raw.strip():
|
||
flush_paragraph()
|
||
index += 1
|
||
continue
|
||
|
||
heading = re.match(r"^(#{1,6})\s+(.+)$", raw)
|
||
if heading:
|
||
flush_paragraph()
|
||
markdown_level = max(2, len(heading.group(1)))
|
||
section_level = max(0, min(markdown_level - 2, 4))
|
||
blocks.append(ParagraphBlock("heading", _strip_manual_heading_number(heading.group(2)), section_level))
|
||
index += 1
|
||
continue
|
||
|
||
if _looks_like_table_row(raw) and index + 1 < len(lines) and _is_table_separator(lines[index + 1]):
|
||
caption: str | None = None
|
||
if paragraph_buffer and _is_table_caption(paragraph_buffer[-1]):
|
||
caption = paragraph_buffer.pop()
|
||
flush_paragraph()
|
||
if caption:
|
||
blocks.append(ParagraphBlock("table_caption", caption))
|
||
header = _split_table_row(raw)
|
||
index += 2
|
||
rows: list[list[str]] = []
|
||
while index < len(lines) and _looks_like_table_row(lines[index]):
|
||
rows.append(_split_table_row(lines[index]))
|
||
index += 1
|
||
blocks.append(TableBlock(header, rows))
|
||
continue
|
||
|
||
image = re.match(r'^\s*!\[([^\]]*)\]\(([^)]+)(?:\s+["\'][^"\']*["\'])?\)\s*$', raw)
|
||
if image:
|
||
flush_paragraph()
|
||
alt = image.group(1).strip() or "未命名图片"
|
||
target = image.group(2).strip()
|
||
if image_count >= _MAX_EMBEDDED_IMAGES and (
|
||
_EMBEDDED_IMAGE_ID.match(target) or _DATA_PNG_URI.match(target)
|
||
):
|
||
blocks.append(ParagraphBlock("image", f"图片:{alt}|{target}"))
|
||
else:
|
||
resolved = _resolve_markdown_image(alt, target, images, image_count + 1)
|
||
if isinstance(resolved, ImageBlock):
|
||
image_count += 1
|
||
blocks.append(resolved)
|
||
index += 1
|
||
continue
|
||
|
||
quote = re.match(r"^>\s?(.*)$", raw)
|
||
if quote:
|
||
flush_paragraph()
|
||
blocks.append(ParagraphBlock("quote", quote.group(1)))
|
||
index += 1
|
||
continue
|
||
|
||
ordered = re.match(r"^(\s*)(\d+)[.)、]\s+(.+)$", raw)
|
||
if ordered:
|
||
flush_paragraph()
|
||
blocks.append(ParagraphBlock("ordered_list", ordered.group(3), min(len(ordered.group(1)) // 2, 8)))
|
||
index += 1
|
||
continue
|
||
|
||
unordered = re.match(r"^(\s*)[-*+]\s+(.+)$", raw)
|
||
if unordered:
|
||
flush_paragraph()
|
||
blocks.append(ParagraphBlock("unordered_list", unordered.group(2), min(len(unordered.group(1)) // 2, 8)))
|
||
index += 1
|
||
continue
|
||
|
||
if re.fullmatch(r"\s*(?:---|\*\*\*|___)\s*", raw):
|
||
flush_paragraph()
|
||
index += 1
|
||
continue
|
||
|
||
paragraph_buffer.append(raw.strip())
|
||
index += 1
|
||
|
||
flush_paragraph()
|
||
return title, blocks
|
||
|
||
|
||
class _DocumentRelationships:
|
||
def __init__(self) -> None:
|
||
self._hyperlinks: list[tuple[str, str]] = []
|
||
self._images: list[tuple[str, str]] = []
|
||
self._drawing_id = 0
|
||
|
||
def add_hyperlink(self, target: str) -> str:
|
||
relation_id = f"rId{100 + len(self._hyperlinks)}"
|
||
self._hyperlinks.append((relation_id, target))
|
||
return relation_id
|
||
|
||
def add_image(self, target: str) -> tuple[str, int]:
|
||
self._drawing_id += 1
|
||
relation_id = f"rId{200 + len(self._images)}"
|
||
self._images.append((relation_id, target))
|
||
return relation_id, self._drawing_id
|
||
|
||
def xml(self) -> str:
|
||
fixed = [
|
||
("rId1", "http://schemas.openxmlformats.org/officeDocument/2006/relationships/styles", "styles.xml"),
|
||
("rId2", "http://schemas.openxmlformats.org/officeDocument/2006/relationships/numbering", "numbering.xml"),
|
||
("rId3", "http://schemas.openxmlformats.org/officeDocument/2006/relationships/settings", "settings.xml"),
|
||
("rId4", "http://schemas.openxmlformats.org/officeDocument/2006/relationships/fontTable", "fontTable.xml"),
|
||
("rId5", "http://schemas.openxmlformats.org/officeDocument/2006/relationships/footer", "footer1.xml"),
|
||
("rId6", "http://schemas.openxmlformats.org/officeDocument/2006/relationships/footer", "footer2.xml"),
|
||
]
|
||
relationships = "".join(
|
||
f'<Relationship Id="{relation_id}" Type="{rel_type}" Target="{target}"/>'
|
||
for relation_id, rel_type, target in fixed
|
||
)
|
||
relationships += "".join(
|
||
f'<Relationship Id="{relation_id}" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/hyperlink" Target="{_xml_attr(target)}" TargetMode="External"/>'
|
||
for relation_id, target in self._hyperlinks
|
||
)
|
||
relationships += "".join(
|
||
f'<Relationship Id="{relation_id}" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/image" Target="{_xml_attr(target)}"/>'
|
||
for relation_id, target in self._images
|
||
)
|
||
return f'{_XML_DECL}<Relationships xmlns="{_PACKAGE_REL_NS}">{relationships}</Relationships>'
|
||
|
||
|
||
_INLINE_PATTERN = re.compile(
|
||
r"\[([^\]]+)\]\(([^)\s]+)(?:\s+[\"\'][^\"\']*[\"\'])?\)|(`+)([^`]+)\3|\*\*([^*]+)\*\*|__([^_]+)__|\*([^*]+)\*|_([^_]+)_"
|
||
)
|
||
|
||
|
||
def _run_xml(
|
||
text: str,
|
||
*,
|
||
font: str = BODY_FONT_FAMILY,
|
||
size: int = BODY_SIZE_HALF_POINTS,
|
||
bold: bool = False,
|
||
italic: bool = False,
|
||
underline: bool = False,
|
||
) -> str:
|
||
text = _safe_xml_text(text)
|
||
preserve = text.startswith(" ") or text.endswith(" ") or " " in text
|
||
properties = (
|
||
f'<w:rFonts w:ascii="{_xml_attr(font)}" w:hAnsi="{_xml_attr(font)}" '
|
||
f'w:eastAsia="{_xml_attr(font)}" w:cs="{_xml_attr(font)}"/>'
|
||
'<w:color w:val="000000"/>'
|
||
f'<w:sz w:val="{size}"/><w:szCs w:val="{size}"/>'
|
||
)
|
||
if bold:
|
||
properties += "<w:b/><w:bCs/>"
|
||
if italic:
|
||
properties += "<w:i/><w:iCs/>"
|
||
if underline:
|
||
properties += '<w:u w:val="single"/>'
|
||
space = ' xml:space="preserve"' if preserve else ""
|
||
return f"<w:r><w:rPr>{properties}</w:rPr><w:t{space}>{escape(text)}</w:t></w:r>"
|
||
|
||
|
||
def _inline_xml(
|
||
text: str,
|
||
relationships: _DocumentRelationships,
|
||
*,
|
||
font: str = BODY_FONT_FAMILY,
|
||
size: int = BODY_SIZE_HALF_POINTS,
|
||
) -> str:
|
||
cleaned = _safe_xml_text(text)
|
||
output: list[str] = []
|
||
cursor = 0
|
||
for match in _INLINE_PATTERN.finditer(cleaned):
|
||
if match.start() > cursor:
|
||
output.append(_run_xml(cleaned[cursor : match.start()], font=font, size=size))
|
||
if match.group(1) is not None and match.group(2) is not None:
|
||
relation_id = relationships.add_hyperlink(match.group(2))
|
||
output.append(
|
||
f'<w:hyperlink r:id="{relation_id}" w:history="1">'
|
||
f'{_run_xml(match.group(1), font=font, size=size, underline=True)}</w:hyperlink>'
|
||
)
|
||
elif match.group(4) is not None:
|
||
output.append(_run_xml(match.group(4), font=font, size=size))
|
||
elif match.group(5) is not None:
|
||
output.append(_run_xml(match.group(5), font=font, size=size, bold=True))
|
||
elif match.group(6) is not None:
|
||
output.append(_run_xml(match.group(6), font=font, size=size, bold=True))
|
||
elif match.group(7) is not None:
|
||
output.append(_run_xml(match.group(7), font=font, size=size, italic=True))
|
||
elif match.group(8) is not None:
|
||
output.append(_run_xml(match.group(8), font=font, size=size, italic=True))
|
||
cursor = match.end()
|
||
if cursor < len(cleaned):
|
||
output.append(_run_xml(cleaned[cursor:], font=font, size=size))
|
||
return "".join(output) or _run_xml(" ", font=font, size=size)
|
||
|
||
|
||
def _paragraph_xml(
|
||
style: str,
|
||
children: str,
|
||
*,
|
||
numbering_level: int | None = None,
|
||
numbering_id: int = 1,
|
||
keep_next: bool = False,
|
||
alignment: str | None = None,
|
||
) -> str:
|
||
properties = f'<w:pStyle w:val="{style}"/>'
|
||
if numbering_level is not None:
|
||
properties += (
|
||
f'<w:numPr><w:ilvl w:val="{numbering_level}"/><w:numId w:val="{numbering_id}"/></w:numPr>'
|
||
)
|
||
if keep_next:
|
||
properties += "<w:keepNext/>"
|
||
if alignment:
|
||
properties += f'<w:jc w:val="{alignment}"/>'
|
||
return f"<w:p><w:pPr>{properties}</w:pPr>{children}</w:p>"
|
||
|
||
|
||
def _image_extent_emu(width_px: int, height_px: int) -> tuple[int, int]:
|
||
max_width_emu = CONTENT_WIDTH_TWIPS * _EMU_PER_TWIP
|
||
width_emu = max(1, int(width_px * 9525))
|
||
height_emu = max(1, int(height_px * 9525))
|
||
if width_emu > max_width_emu:
|
||
scale = max_width_emu / width_emu
|
||
width_emu = max_width_emu
|
||
height_emu = max(1, int(height_emu * scale))
|
||
return width_emu, height_emu
|
||
|
||
|
||
def _embedded_image_xml(block: ImageBlock, relationships: _DocumentRelationships) -> str:
|
||
relation_id, drawing_id = relationships.add_image(f"media/{block.media_name}")
|
||
cx, cy = _image_extent_emu(block.width_px, block.height_px)
|
||
name = _xml_attr(block.alt or block.media_name)
|
||
drawing = (
|
||
'<w:r><w:drawing><wp:inline distT="0" distB="0" distL="0" distR="0">'
|
||
f'<wp:extent cx="{cx}" cy="{cy}"/>'
|
||
f'<wp:docPr id="{drawing_id}" name="{name}"/>'
|
||
'<a:graphic xmlns:a="http://schemas.openxmlformats.org/drawingml/2006/main">'
|
||
'<a:graphicData uri="http://schemas.openxmlformats.org/drawingml/2006/picture">'
|
||
'<pic:pic xmlns:pic="http://schemas.openxmlformats.org/drawingml/2006/picture">'
|
||
f'<pic:nvPicPr><pic:cNvPr id="0" name="{name}"/><pic:cNvPicPr/></pic:nvPicPr>'
|
||
f'<pic:blipFill><a:blip r:embed="{relation_id}"/><a:stretch><a:fillRect/></a:stretch></pic:blipFill>'
|
||
"<pic:spPr>"
|
||
f'<a:xfrm><a:off x="0" y="0"/><a:ext cx="{cx}" cy="{cy}"/></a:xfrm>'
|
||
'<a:prstGeom prst="rect"><a:avLst/></a:prstGeom>'
|
||
"</pic:spPr></pic:pic></a:graphicData></a:graphic></wp:inline></w:drawing></w:r>"
|
||
)
|
||
return (
|
||
"<w:p><w:pPr>"
|
||
'<w:pStyle w:val="ImageCaption"/>'
|
||
'<w:jc w:val="center"/>'
|
||
'<w:spacing w:before="160" w:after="160" w:line="240" w:lineRule="auto"/>'
|
||
'<w:ind w:left="0" w:hanging="0" w:firstLineChars="0"/>'
|
||
f"</w:pPr>{drawing}</w:p>"
|
||
)
|
||
|
||
|
||
def _image_paragraph_xml(text: str, relationships: _DocumentRelationships) -> str:
|
||
label, _, target = text.partition("|")
|
||
children = _run_xml(label)
|
||
if target:
|
||
relation_id = relationships.add_hyperlink(target)
|
||
children += (
|
||
f'<w:hyperlink r:id="{relation_id}" w:history="1">'
|
||
f'{_run_xml(f"({target})", underline=True)}</w:hyperlink>'
|
||
)
|
||
return _paragraph_xml("ImageCaption", children, alignment="center")
|
||
|
||
|
||
def _cell_xml(text: str, width: int, relationships: _DocumentRelationships, *, header: bool) -> str:
|
||
style = "TableHeader" if header else "TableBody"
|
||
paragraph = _paragraph_xml(
|
||
style,
|
||
_inline_xml(text, relationships, font=TABLE_FONT_FAMILY, size=TABLE_SIZE_HALF_POINTS),
|
||
)
|
||
return (
|
||
"<w:tc>"
|
||
f'<w:tcPr><w:tcW w:w="{width}" w:type="dxa"/><w:vAlign w:val="center"/>'
|
||
'<w:tcMar><w:top w:w="80" w:type="dxa"/><w:left w:w="100" w:type="dxa"/>'
|
||
'<w:bottom w:w="80" w:type="dxa"/><w:right w:w="100" w:type="dxa"/></w:tcMar></w:tcPr>'
|
||
f"{paragraph}</w:tc>"
|
||
)
|
||
|
||
|
||
def _table_xml(block: TableBlock, relationships: _DocumentRelationships) -> str:
|
||
column_count = max(1, len(block.header))
|
||
base_width = CONTENT_WIDTH_TWIPS // column_count
|
||
widths = [base_width] * column_count
|
||
widths[-1] = CONTENT_WIDTH_TWIPS - base_width * (column_count - 1)
|
||
grid = "".join(f'<w:gridCol w:w="{width}"/>' for width in widths)
|
||
border = (
|
||
'<w:top w:val="single" w:sz="4" w:color="000000"/>'
|
||
'<w:left w:val="single" w:sz="4" w:color="000000"/>'
|
||
'<w:bottom w:val="single" w:sz="4" w:color="000000"/>'
|
||
'<w:right w:val="single" w:sz="4" w:color="000000"/>'
|
||
'<w:insideH w:val="single" w:sz="4" w:color="000000"/>'
|
||
'<w:insideV w:val="single" w:sz="4" w:color="000000"/>'
|
||
)
|
||
properties = (
|
||
f'<w:tblPr><w:tblW w:w="{CONTENT_WIDTH_TWIPS}" w:type="dxa"/>'
|
||
'<w:jc w:val="center"/><w:tblLayout w:type="fixed"/>'
|
||
f"<w:tblBorders>{border}</w:tblBorders></w:tblPr>"
|
||
)
|
||
|
||
header = list(block.header[:column_count]) + [""] * max(0, column_count - len(block.header))
|
||
header_cells = "".join(
|
||
_cell_xml(header[index], widths[index], relationships, header=True) for index in range(column_count)
|
||
)
|
||
rows = [f"<w:tr><w:trPr><w:tblHeader/><w:cantSplit/></w:trPr>{header_cells}</w:tr>"]
|
||
for source_row in block.rows:
|
||
normalized = list(source_row[:column_count]) + [""] * max(0, column_count - len(source_row))
|
||
cells = "".join(
|
||
_cell_xml(normalized[index], widths[index], relationships, header=False)
|
||
for index in range(column_count)
|
||
)
|
||
rows.append(f"<w:tr><w:trPr><w:cantSplit/></w:trPr>{cells}</w:tr>")
|
||
return f"<w:tbl>{properties}<w:tblGrid>{grid}</w:tblGrid>{''.join(rows)}</w:tbl>"
|
||
|
||
|
||
def _blocks_xml(blocks: list[Block], relationships: _DocumentRelationships) -> str:
|
||
output: list[str] = []
|
||
for block in blocks:
|
||
if isinstance(block, ImageBlock):
|
||
output.append(_embedded_image_xml(block, relationships))
|
||
continue
|
||
if isinstance(block, TableBlock):
|
||
output.append(_table_xml(block, relationships))
|
||
continue
|
||
if block.kind == "heading":
|
||
style = f"Heading{block.level + 1}"
|
||
font = (
|
||
HEADING_1_FONT_FAMILY
|
||
if block.level == 0
|
||
else HEADING_2_FONT_FAMILY
|
||
if block.level == 1
|
||
else BODY_FONT_FAMILY
|
||
)
|
||
output.append(
|
||
_paragraph_xml(
|
||
style,
|
||
_inline_xml(block.text, relationships, font=font),
|
||
numbering_level=block.level,
|
||
numbering_id=1,
|
||
keep_next=True,
|
||
)
|
||
)
|
||
elif block.kind == "table_caption":
|
||
output.append(
|
||
_paragraph_xml(
|
||
"TableCaption",
|
||
_inline_xml(block.text, relationships, font=HEADING_1_FONT_FAMILY, size=TABLE_SIZE_HALF_POINTS),
|
||
keep_next=True,
|
||
)
|
||
)
|
||
elif block.kind == "ordered_list":
|
||
output.append(
|
||
_paragraph_xml(
|
||
"ListText",
|
||
_inline_xml(block.text, relationships),
|
||
numbering_level=block.level,
|
||
numbering_id=2,
|
||
)
|
||
)
|
||
elif block.kind == "unordered_list":
|
||
output.append(
|
||
_paragraph_xml(
|
||
"ListText",
|
||
_inline_xml(block.text, relationships),
|
||
numbering_level=block.level,
|
||
numbering_id=3,
|
||
)
|
||
)
|
||
elif block.kind == "code":
|
||
output.append(_paragraph_xml("CodeText", _run_xml(block.text)))
|
||
elif block.kind == "quote":
|
||
output.append(_paragraph_xml("QuoteText", _inline_xml(block.text, relationships)))
|
||
elif block.kind == "image":
|
||
output.append(_image_paragraph_xml(block.text, relationships))
|
||
else:
|
||
output.append(_paragraph_xml("BodyText", _inline_xml(block.text, relationships)))
|
||
return "".join(output)
|
||
|
||
|
||
def _font_run_properties(font: str, size: int, *, bold: bool = False) -> str:
|
||
value = (
|
||
f'<w:rFonts w:ascii="{_xml_attr(font)}" w:hAnsi="{_xml_attr(font)}" '
|
||
f'w:eastAsia="{_xml_attr(font)}" w:cs="{_xml_attr(font)}"/>'
|
||
'<w:color w:val="000000"/>'
|
||
f'<w:sz w:val="{size}"/><w:szCs w:val="{size}"/>'
|
||
)
|
||
return value + ("<w:b/><w:bCs/>" if bold else '<w:b w:val="0"/><w:bCs w:val="0"/>')
|
||
|
||
|
||
def _style_xml(
|
||
style_id: str,
|
||
name: str,
|
||
*,
|
||
font: str,
|
||
size: int,
|
||
based_on: str | None = "Normal",
|
||
next_style: str = "BodyText",
|
||
alignment: str = "both",
|
||
first_line_chars: int = 0,
|
||
left: int = 0,
|
||
hanging: int = 0,
|
||
keep_next: bool = False,
|
||
keep_lines: bool = False,
|
||
outline_level: int | None = None,
|
||
bold: bool = False,
|
||
) -> str:
|
||
paragraph = (
|
||
f'<w:jc w:val="{alignment}"/><w:spacing w:before="0" w:after="0" '
|
||
f'w:line="{EXACT_LINE_TWIPS}" w:lineRule="exact"/>'
|
||
f'<w:ind w:left="{left}" w:hanging="{hanging}" w:firstLineChars="{first_line_chars}"/>'
|
||
)
|
||
if keep_next:
|
||
paragraph += "<w:keepNext/>"
|
||
if keep_lines:
|
||
paragraph += "<w:keepLines/>"
|
||
if outline_level is not None:
|
||
paragraph += f'<w:outlineLvl w:val="{outline_level}"/>'
|
||
based_on_xml = f'<w:basedOn w:val="{based_on}"/>' if based_on else ""
|
||
return (
|
||
f'<w:style w:type="paragraph" w:styleId="{style_id}"><w:name w:val="{_xml_attr(name)}"/>'
|
||
f'{based_on_xml}<w:next w:val="{next_style}"/><w:qFormat/>'
|
||
f"<w:pPr>{paragraph}</w:pPr><w:rPr>{_font_run_properties(font, size, bold=bold)}</w:rPr></w:style>"
|
||
)
|
||
|
||
|
||
def _styles_xml() -> str:
|
||
defaults = (
|
||
"<w:docDefaults><w:rPrDefault><w:rPr>"
|
||
f"{_font_run_properties(BODY_FONT_FAMILY, BODY_SIZE_HALF_POINTS)}"
|
||
'<w:lang w:val="zh-CN" w:eastAsia="zh-CN"/></w:rPr></w:rPrDefault>'
|
||
'<w:pPrDefault><w:pPr><w:jc w:val="both"/>'
|
||
f'<w:spacing w:before="0" w:after="0" w:line="{EXACT_LINE_TWIPS}" w:lineRule="exact"/>'
|
||
'<w:ind w:firstLineChars="200"/></w:pPr></w:pPrDefault></w:docDefaults>'
|
||
)
|
||
styles = [
|
||
_style_xml(
|
||
"Normal",
|
||
"Normal",
|
||
font=BODY_FONT_FAMILY,
|
||
size=BODY_SIZE_HALF_POINTS,
|
||
based_on=None,
|
||
next_style="BodyText",
|
||
first_line_chars=200,
|
||
),
|
||
_style_xml(
|
||
"DocumentTitle",
|
||
"Document Title",
|
||
font=TITLE_FONT_FAMILY,
|
||
size=TITLE_SIZE_HALF_POINTS,
|
||
alignment="center",
|
||
keep_next=True,
|
||
),
|
||
_style_xml(
|
||
"BodyText",
|
||
"Body Text",
|
||
font=BODY_FONT_FAMILY,
|
||
size=BODY_SIZE_HALF_POINTS,
|
||
first_line_chars=200,
|
||
),
|
||
_style_xml(
|
||
"Heading1",
|
||
"heading 1",
|
||
font=HEADING_1_FONT_FAMILY,
|
||
size=BODY_SIZE_HALF_POINTS,
|
||
alignment="left",
|
||
first_line_chars=200,
|
||
keep_next=True,
|
||
keep_lines=True,
|
||
outline_level=0,
|
||
),
|
||
_style_xml(
|
||
"Heading2",
|
||
"heading 2",
|
||
font=HEADING_2_FONT_FAMILY,
|
||
size=BODY_SIZE_HALF_POINTS,
|
||
alignment="left",
|
||
first_line_chars=200,
|
||
keep_next=True,
|
||
keep_lines=True,
|
||
outline_level=1,
|
||
),
|
||
]
|
||
for level in range(3, 6):
|
||
styles.append(
|
||
_style_xml(
|
||
f"Heading{level}",
|
||
f"heading {level}",
|
||
font=BODY_FONT_FAMILY,
|
||
size=BODY_SIZE_HALF_POINTS,
|
||
alignment="left",
|
||
first_line_chars=200 if level == 3 else 0,
|
||
keep_next=True,
|
||
keep_lines=True,
|
||
outline_level=level - 1,
|
||
)
|
||
)
|
||
styles.extend(
|
||
[
|
||
_style_xml(
|
||
"ListText",
|
||
"List Text",
|
||
font=BODY_FONT_FAMILY,
|
||
size=BODY_SIZE_HALF_POINTS,
|
||
alignment="both",
|
||
),
|
||
_style_xml(
|
||
"QuoteText",
|
||
"Quote Text",
|
||
font=BODY_FONT_FAMILY,
|
||
size=BODY_SIZE_HALF_POINTS,
|
||
alignment="both",
|
||
first_line_chars=200,
|
||
left=640,
|
||
),
|
||
_style_xml(
|
||
"CodeText",
|
||
"Code Text",
|
||
font=BODY_FONT_FAMILY,
|
||
size=BODY_SIZE_HALF_POINTS,
|
||
alignment="left",
|
||
),
|
||
_style_xml(
|
||
"TableCaption",
|
||
"Table Caption",
|
||
font=HEADING_1_FONT_FAMILY,
|
||
size=TABLE_SIZE_HALF_POINTS,
|
||
alignment="center",
|
||
keep_next=True,
|
||
),
|
||
_style_xml(
|
||
"TableHeader",
|
||
"Table Header",
|
||
font=TABLE_FONT_FAMILY,
|
||
size=TABLE_SIZE_HALF_POINTS,
|
||
alignment="center",
|
||
keep_lines=True,
|
||
),
|
||
_style_xml(
|
||
"TableBody",
|
||
"Table Body",
|
||
font=TABLE_FONT_FAMILY,
|
||
size=TABLE_SIZE_HALF_POINTS,
|
||
alignment="both",
|
||
keep_lines=True,
|
||
),
|
||
_style_xml(
|
||
"ImageCaption",
|
||
"Image Caption",
|
||
font=BODY_FONT_FAMILY,
|
||
size=BODY_SIZE_HALF_POINTS,
|
||
alignment="center",
|
||
keep_lines=True,
|
||
),
|
||
]
|
||
)
|
||
return f'{_XML_DECL}<w:styles xmlns:w="{_W_NS}">{defaults}{"".join(styles)}</w:styles>'
|
||
|
||
|
||
def _heading_level_xml(level: int, format_value: str, text: str, font: str) -> str:
|
||
restart = f'<w:lvlRestart w:val="{level}"/>' if level > 0 else ""
|
||
first_line_chars = ' w:firstLineChars="200"' if level <= 2 else ""
|
||
return (
|
||
f'<w:lvl w:ilvl="{level}"><w:start w:val="1"/>{restart}<w:numFmt w:val="{format_value}"/>'
|
||
f'<w:lvlText w:val="{_xml_attr(text)}"/><w:lvlJc w:val="left"/>'
|
||
f'<w:pPr><w:ind w:left="0" w:hanging="0"{first_line_chars}/></w:pPr>'
|
||
f'<w:rPr>{_font_run_properties(font, BODY_SIZE_HALF_POINTS)}</w:rPr></w:lvl>'
|
||
)
|
||
|
||
|
||
def _list_abstract_xml(abstract_id: int, *, ordered: bool) -> str:
|
||
levels: list[str] = []
|
||
bullets = ["●", "○", "▪"]
|
||
for level in range(9):
|
||
format_value = "decimal" if ordered else "bullet"
|
||
text = f"%{level + 1}." if ordered else bullets[level % len(bullets)]
|
||
left = 720 * (level + 1)
|
||
levels.append(
|
||
f'<w:lvl w:ilvl="{level}"><w:start w:val="1"/><w:numFmt w:val="{format_value}"/>'
|
||
f'<w:lvlText w:val="{_xml_attr(text)}"/><w:lvlJc w:val="left"/>'
|
||
f'<w:pPr><w:tabs><w:tab w:val="num" w:pos="{left}"/></w:tabs>'
|
||
f'<w:ind w:left="{left}" w:hanging="360"/></w:pPr>'
|
||
f'<w:rPr>{_font_run_properties(BODY_FONT_FAMILY, BODY_SIZE_HALF_POINTS)}</w:rPr></w:lvl>'
|
||
)
|
||
return f'<w:abstractNum w:abstractNumId="{abstract_id}"><w:multiLevelType w:val="multilevel"/>{"".join(levels)}</w:abstractNum>'
|
||
|
||
|
||
def _numbering_xml() -> str:
|
||
heading_levels = "".join(
|
||
[
|
||
_heading_level_xml(0, "chineseCounting", "%1、", HEADING_1_FONT_FAMILY),
|
||
_heading_level_xml(1, "chineseCounting", "(%2)", HEADING_2_FONT_FAMILY),
|
||
_heading_level_xml(2, "decimal", "%3.", BODY_FONT_FAMILY),
|
||
_heading_level_xml(3, "decimal", "(%4)", BODY_FONT_FAMILY),
|
||
_heading_level_xml(4, "decimal", "%5)", BODY_FONT_FAMILY),
|
||
]
|
||
)
|
||
return (
|
||
f'{_XML_DECL}<w:numbering xmlns:w="{_W_NS}">'
|
||
f'<w:abstractNum w:abstractNumId="0"><w:multiLevelType w:val="multilevel">'
|
||
f"</w:multiLevelType>{heading_levels}</w:abstractNum>"
|
||
f'{_list_abstract_xml(1, ordered=True)}{_list_abstract_xml(2, ordered=False)}'
|
||
'<w:num w:numId="1"><w:abstractNumId w:val="0"/></w:num>'
|
||
'<w:num w:numId="2"><w:abstractNumId w:val="1"/></w:num>'
|
||
'<w:num w:numId="3"><w:abstractNumId w:val="2"/></w:num>'
|
||
"</w:numbering>"
|
||
)
|
||
|
||
|
||
def _footer_xml(alignment: str) -> str:
|
||
run_properties = _font_run_properties(TABLE_FONT_FAMILY, TABLE_SIZE_HALF_POINTS)
|
||
return (
|
||
f'{_XML_DECL}<w:ftr xmlns:w="{_W_NS}" xmlns:r="{_R_NS}">'
|
||
f'<w:p><w:pPr><w:jc w:val="{alignment}"/><w:spacing w:before="0" w:after="0"/></w:pPr>'
|
||
f'<w:r><w:rPr>{run_properties}</w:rPr><w:fldChar w:fldCharType="begin"/></w:r>'
|
||
f'<w:r><w:rPr>{run_properties}</w:rPr><w:instrText xml:space="preserve"> PAGE </w:instrText></w:r>'
|
||
f'<w:r><w:rPr>{run_properties}</w:rPr><w:fldChar w:fldCharType="separate"/></w:r>'
|
||
f'<w:r><w:rPr>{run_properties}</w:rPr><w:t>1</w:t></w:r>'
|
||
f'<w:r><w:rPr>{run_properties}</w:rPr><w:fldChar w:fldCharType="end"/></w:r>'
|
||
"</w:p></w:ftr>"
|
||
)
|
||
|
||
|
||
def _font_table_xml(font_key: str) -> str:
|
||
def font(name: str, *, embedded: bool = False) -> str:
|
||
embed = f'<w:embedRegular r:id="rId1" w:fontKey="{{{font_key}}}"/>' if embedded else ""
|
||
return (
|
||
f'<w:font w:name="{_xml_attr(name)}"><w:family w:val="roman"/>'
|
||
f'<w:charset w:val="86"/>{embed}</w:font>'
|
||
)
|
||
|
||
fonts = "".join(
|
||
[
|
||
font(TITLE_FONT_FAMILY, embedded=True),
|
||
font(BODY_FONT_FAMILY),
|
||
font(HEADING_1_FONT_FAMILY),
|
||
font(HEADING_2_FONT_FAMILY),
|
||
font(TABLE_FONT_FAMILY),
|
||
]
|
||
)
|
||
return f'{_XML_DECL}<w:fonts xmlns:w="{_W_NS}" xmlns:r="{_R_NS}">{fonts}</w:fonts>'
|
||
|
||
|
||
def _font_table_relationships_xml() -> str:
|
||
return (
|
||
f'{_XML_DECL}<Relationships xmlns="{_PACKAGE_REL_NS}">'
|
||
'<Relationship Id="rId1" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/font" Target="fonts/font1.odttf"/>'
|
||
"</Relationships>"
|
||
)
|
||
|
||
|
||
def _settings_xml() -> str:
|
||
return (
|
||
f'{_XML_DECL}<w:settings xmlns:w="{_W_NS}">'
|
||
'<w:zoom w:percent="100"/><w:evenAndOddHeaders/><w:updateFields w:val="true"/>'
|
||
'<w:compat><w:compatSetting w:name="compatibilityMode" '
|
||
'w:uri="http://schemas.microsoft.com/office/word" w:val="15"/></w:compat>'
|
||
"</w:settings>"
|
||
)
|
||
|
||
|
||
def _document_xml(title: str | None, blocks: list[Block], relationships: _DocumentRelationships) -> str:
|
||
body: list[str] = []
|
||
if title:
|
||
body.append(
|
||
_paragraph_xml(
|
||
"DocumentTitle",
|
||
_run_xml(title, font=TITLE_FONT_FAMILY, size=TITLE_SIZE_HALF_POINTS),
|
||
keep_next=True,
|
||
alignment="center",
|
||
)
|
||
)
|
||
body.append(_blocks_xml(blocks, relationships))
|
||
section = (
|
||
'<w:sectPr><w:footerReference w:type="default" r:id="rId5"/>'
|
||
'<w:footerReference w:type="even" r:id="rId6"/>'
|
||
f'<w:pgSz w:w="{A4_WIDTH_TWIPS}" w:h="{A4_HEIGHT_TWIPS}" w:orient="portrait"/>'
|
||
f'<w:pgMar w:top="{MARGIN_TOP_TWIPS}" w:right="{MARGIN_RIGHT_TWIPS}" '
|
||
f'w:bottom="{MARGIN_BOTTOM_TWIPS}" w:left="{MARGIN_LEFT_TWIPS}" '
|
||
'w:header="720" w:footer="992" w:gutter="0"/>'
|
||
'<w:pgNumType w:start="1" w:fmt="decimal"/><w:cols w:space="720"/></w:sectPr>'
|
||
)
|
||
return (
|
||
f'{_XML_DECL}<w:document xmlns:w="{_W_NS}" xmlns:r="{_R_NS}" '
|
||
'xmlns:wp="http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing">'
|
||
f"<w:body>{''.join(body)}{section}</w:body></w:document>"
|
||
)
|
||
|
||
|
||
def _content_types_xml() -> str:
|
||
return (
|
||
f'{_XML_DECL}<Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types">'
|
||
'<Default Extension="rels" ContentType="application/vnd.openxmlformats-package.relationships+xml"/>'
|
||
'<Default Extension="xml" ContentType="application/xml"/>'
|
||
'<Default Extension="odttf" ContentType="application/vnd.openxmlformats-officedocument.obfuscatedFont"/>'
|
||
'<Default Extension="png" ContentType="image/png"/>'
|
||
'<Override PartName="/word/document.xml" ContentType="application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml"/>'
|
||
'<Override PartName="/word/styles.xml" ContentType="application/vnd.openxmlformats-officedocument.wordprocessingml.styles+xml"/>'
|
||
'<Override PartName="/word/numbering.xml" ContentType="application/vnd.openxmlformats-officedocument.wordprocessingml.numbering+xml"/>'
|
||
'<Override PartName="/word/settings.xml" ContentType="application/vnd.openxmlformats-officedocument.wordprocessingml.settings+xml"/>'
|
||
'<Override PartName="/word/fontTable.xml" ContentType="application/vnd.openxmlformats-officedocument.wordprocessingml.fontTable+xml"/>'
|
||
'<Override PartName="/word/footer1.xml" ContentType="application/vnd.openxmlformats-officedocument.wordprocessingml.footer+xml"/>'
|
||
'<Override PartName="/word/footer2.xml" ContentType="application/vnd.openxmlformats-officedocument.wordprocessingml.footer+xml"/>'
|
||
'<Override PartName="/docProps/core.xml" ContentType="application/vnd.openxmlformats-package.core-properties+xml"/>'
|
||
'<Override PartName="/docProps/app.xml" ContentType="application/vnd.openxmlformats-officedocument.extended-properties+xml"/>'
|
||
"</Types>"
|
||
)
|
||
|
||
|
||
def _package_relationships_xml() -> str:
|
||
return (
|
||
f'{_XML_DECL}<Relationships xmlns="{_PACKAGE_REL_NS}">'
|
||
'<Relationship Id="rId1" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" Target="word/document.xml"/>'
|
||
'<Relationship Id="rId2" Type="http://schemas.openxmlformats.org/package/2006/relationships/metadata/core-properties" Target="docProps/core.xml"/>'
|
||
'<Relationship Id="rId3" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/extended-properties" Target="docProps/app.xml"/>'
|
||
"</Relationships>"
|
||
)
|
||
|
||
|
||
def _core_properties_xml(title: str | None) -> str:
|
||
now = datetime.now(UTC).replace(microsecond=0).isoformat().replace("+00:00", "Z")
|
||
safe_title = escape(title or "Word 文档")
|
||
return (
|
||
f'{_XML_DECL}<cp:coreProperties xmlns:cp="http://schemas.openxmlformats.org/package/2006/metadata/core-properties" '
|
||
'xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:dcterms="http://purl.org/dc/terms/" '
|
||
'xmlns:dcmitype="http://purl.org/dc/dcmitype/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">'
|
||
f"<dc:title>{safe_title}</dc:title><dc:creator>DeerFlow</dc:creator>"
|
||
f'<dcterms:created xsi:type="dcterms:W3CDTF">{now}</dcterms:created>'
|
||
f'<dcterms:modified xsi:type="dcterms:W3CDTF">{now}</dcterms:modified></cp:coreProperties>'
|
||
)
|
||
|
||
|
||
def _app_properties_xml() -> str:
|
||
return (
|
||
f'{_XML_DECL}<Properties xmlns="http://schemas.openxmlformats.org/officeDocument/2006/extended-properties" '
|
||
'xmlns:vt="http://schemas.openxmlformats.org/officeDocument/2006/docPropsVTypes">'
|
||
"<Application>DeerFlow</Application><AppVersion>1.0</AppVersion></Properties>"
|
||
)
|
||
|
||
|
||
def build_markdown_docx(
|
||
markdown: str,
|
||
title: str | None = None,
|
||
*,
|
||
font_path: Path | None = None,
|
||
embedded_images: dict[str, str] | None = None,
|
||
) -> bytes:
|
||
"""Build a formal DOCX and embed the validated server-side title font."""
|
||
if not markdown or not markdown.strip():
|
||
raise WordExportError("文档内容为空。")
|
||
if len(markdown) > 2_000_000:
|
||
raise WordExportError("文档内容超过 200 万字符,无法导出。")
|
||
if embedded_images and len(embedded_images) > _MAX_EMBEDDED_IMAGES:
|
||
raise WordExportError("嵌入图片数量超过限制。")
|
||
|
||
font = load_title_font(font_path)
|
||
resolved_title, blocks = _parse_markdown(markdown, title, embedded_images)
|
||
relationships = _DocumentRelationships()
|
||
document_xml = _document_xml(resolved_title, blocks, relationships)
|
||
font_key = str(uuid.uuid4()).upper()
|
||
media_files = {
|
||
f"word/media/{block.media_name}": block.data for block in blocks if isinstance(block, ImageBlock)
|
||
}
|
||
|
||
output = io.BytesIO()
|
||
with zipfile.ZipFile(output, mode="w", compression=zipfile.ZIP_DEFLATED, compresslevel=6) as archive:
|
||
parts: dict[str, str | bytes] = {
|
||
"[Content_Types].xml": _content_types_xml(),
|
||
"_rels/.rels": _package_relationships_xml(),
|
||
"docProps/core.xml": _core_properties_xml(resolved_title),
|
||
"docProps/app.xml": _app_properties_xml(),
|
||
"word/document.xml": document_xml,
|
||
"word/_rels/document.xml.rels": relationships.xml(),
|
||
"word/styles.xml": _styles_xml(),
|
||
"word/numbering.xml": _numbering_xml(),
|
||
"word/settings.xml": _settings_xml(),
|
||
"word/fontTable.xml": _font_table_xml(font_key),
|
||
"word/_rels/fontTable.xml.rels": _font_table_relationships_xml(),
|
||
# Word applies the default footer to odd pages and the even footer
|
||
# to even pages. The requested placement is odd-left/even-right.
|
||
"word/footer1.xml": _footer_xml("left"),
|
||
"word/footer2.xml": _footer_xml("right"),
|
||
"word/fonts/font1.odttf": _obfuscate_font(font.data, font_key),
|
||
**media_files,
|
||
}
|
||
for name, value in parts.items():
|
||
archive.writestr(name, value.encode("utf-8") if isinstance(value, str) else value)
|
||
return output.getvalue()
|
||
|
||
|
||
__all__ = [
|
||
"DOCX_MEDIA_TYPE",
|
||
"TITLE_FONT_FAMILY",
|
||
"WordExportError",
|
||
"WordExportFontError",
|
||
"build_markdown_docx",
|
||
"load_title_font",
|
||
"resolve_title_font_path",
|
||
]
|