deerflow-code/offline-backend-20260512/backend/app/gateway/word_export.py
2026-09-07 18:24:55 +08:00

1153 lines
45 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""Deterministic Markdown -> DOCX export with an embedded title font.
The implementation intentionally uses Python's standard library only. It
builds the small set of OOXML parts required by Word instead of depending on a
desktop Office installation or a document-conversion service. The bundled
方正小标宋简体 font is validated and embedded according to ECMA-376 so the
recipient's computer does not need the font installed.
"""
from __future__ import annotations
import base64
import binascii
import hashlib
import io
import os
import re
import struct
import uuid
import zipfile
from dataclasses import dataclass
from datetime import UTC, datetime
from pathlib import Path
from xml.sax.saxutils import escape
DOCX_MEDIA_TYPE = "application/vnd.openxmlformats-officedocument.wordprocessingml.document"
TITLE_FONT_FAMILY = "方正小标宋简体"
BODY_FONT_FAMILY = "仿宋"
HEADING_1_FONT_FAMILY = "黑体"
HEADING_2_FONT_FAMILY = "楷体"
TABLE_FONT_FAMILY = "宋体"
TITLE_SIZE_HALF_POINTS = 44 # 二号 / 22 pt
BODY_SIZE_HALF_POINTS = 32 # 三号 / 16 pt
TABLE_SIZE_HALF_POINTS = 28 # 四号 / 14 pt
EXACT_LINE_TWIPS = 579 # 28.95 pt * 20
A4_WIDTH_TWIPS = 11_906
A4_HEIGHT_TWIPS = 16_838
MARGIN_TOP_TWIPS = 2_098 # 3.7 cm
MARGIN_BOTTOM_TWIPS = 1_984 # 3.5 cm
MARGIN_LEFT_TWIPS = 1_587 # 2.8 cm
MARGIN_RIGHT_TWIPS = 1_474 # 2.6 cm
CONTENT_WIDTH_TWIPS = A4_WIDTH_TWIPS - MARGIN_LEFT_TWIPS - MARGIN_RIGHT_TWIPS
_XML_DECL = '<?xml version="1.0" encoding="UTF-8" standalone="yes"?>'
_W_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
_R_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships"
_PACKAGE_REL_NS = "http://schemas.openxmlformats.org/package/2006/relationships"
_FONT_DIR = Path(__file__).resolve().parent / "assets" / "fonts"
_BUNDLED_FONT_PATH = _FONT_DIR / "方正小标宋简体.ttf"
_BUNDLED_FONT_SHA256 = "C52D577DD3AA719EF5FB9FF2043B7F7CA36FAE468EF34A47628A3A555D006A49"
class WordExportError(RuntimeError):
"""Base error returned to the Word export API."""
class WordExportFontError(WordExportError):
"""The required server-side font is missing, invalid, or not embeddable."""
@dataclass(slots=True)
class FontAsset:
data: bytes
path: Path
names: frozenset[str]
fs_type: int
sha256: str
@dataclass(slots=True)
class ParagraphBlock:
kind: str
text: str
level: int = 0
@dataclass(slots=True)
class TableBlock:
header: list[str]
rows: list[list[str]]
@dataclass(slots=True)
class ImageBlock:
alt: str
data: bytes
media_name: str
width_px: int
height_px: int
Block = ParagraphBlock | TableBlock | ImageBlock
_EMU_PER_TWIP = 635
_PNG_SIGNATURE = b"\x89PNG\r\n\x1a\n"
_MAX_EMBEDDED_IMAGES = 40
_MAX_EMBEDDED_IMAGE_BYTES = 8 * 1024 * 1024
_EMBEDDED_IMAGE_ID = re.compile(r"^embedded:([A-Za-z0-9_-]{1,80})$")
_DATA_PNG_URI = re.compile(r"^data:image/png;base64,([A-Za-z0-9+/=\s]+)$", re.IGNORECASE)
def _safe_xml_text(value: str) -> str:
"""Remove XML-illegal controls and invisible Word-breaking selectors."""
value = re.sub(r"[\u0000-\u0008\u000b\u000c\u000e-\u001f]", "", value)
value = re.sub(r"[\u200b-\u200d\ufeff\ufe00-\ufe0f]", "", value)
value = value.replace("\u00ad", "").replace("\ufffc", "")
value = re.sub(r"</?(?:details|summary)\b[^>]*>", "", value, flags=re.I)
return value
def _xml_attr(value: str) -> str:
return escape(_safe_xml_text(value), {'"': "&quot;", "'": "&apos;"})
def _decode_name_record(platform_id: int, raw: bytes) -> str | None:
try:
if platform_id in {0, 3}:
return raw.decode("utf-16-be").strip("\x00")
if platform_id == 1:
return raw.decode("mac_roman").strip("\x00")
except (UnicodeDecodeError, LookupError):
return None
return None
def _read_sfnt_metadata(data: bytes) -> tuple[frozenset[str], int]:
if len(data) < 12:
raise WordExportFontError("Word 标题字体文件不是有效的 OpenType/TrueType 字体。")
try:
table_count = struct.unpack_from(">H", data, 4)[0]
tables: dict[str, tuple[int, int]] = {}
for index in range(table_count):
position = 12 + index * 16
tag, _checksum, offset, length = struct.unpack_from(">4sIII", data, position)
tag_text = tag.decode("latin-1")
if offset + length > len(data):
raise WordExportFontError("Word 标题字体的表目录已损坏。")
tables[tag_text] = (offset, length)
if "OS/2" not in tables or "name" not in tables:
raise WordExportFontError("Word 标题字体缺少 OS/2 或 name 元数据表。")
os2_offset, os2_length = tables["OS/2"]
if os2_length < 10:
raise WordExportFontError("Word 标题字体的 OS/2 元数据无效。")
fs_type = struct.unpack_from(">H", data, os2_offset + 8)[0]
name_offset, name_length = tables["name"]
if name_length < 6:
raise WordExportFontError("Word 标题字体的 name 元数据无效。")
name_count, string_offset = struct.unpack_from(">HH", data, name_offset + 2)
names: set[str] = set()
for index in range(name_count):
record_offset = name_offset + 6 + index * 12
if record_offset + 12 > name_offset + name_length:
break
platform_id, _encoding, _language, name_id, length, offset = struct.unpack_from(
">HHHHHH", data, record_offset
)
if name_id not in {1, 4, 6}:
continue
value_start = name_offset + string_offset + offset
value_end = value_start + length
if value_end > name_offset + name_length:
continue
decoded = _decode_name_record(platform_id, data[value_start:value_end])
if decoded:
names.add(decoded)
return frozenset(names), fs_type
except struct.error as exc:
raise WordExportFontError("Word 标题字体文件结构不完整。") from exc
def resolve_title_font_path() -> Path:
configured = os.getenv("DEERFLOW_WORD_TITLE_FONT_PATH", "").strip()
return Path(configured).expanduser() if configured else _BUNDLED_FONT_PATH
def load_title_font(path: Path | None = None) -> FontAsset:
font_path = (path or resolve_title_font_path()).resolve()
if not font_path.is_file():
raise WordExportFontError(
f"服务器缺少“{TITLE_FONT_FAMILY}”字体资源:{font_path}。"
"请将已授权字体放入后端字体目录后再导出。"
)
data = font_path.read_bytes()
if not data or len(data) > 25 * 1024 * 1024:
raise WordExportFontError("Word 标题字体文件为空或超过 25 MB 限制。")
names, fs_type = _read_sfnt_metadata(data)
accepted_name = TITLE_FONT_FAMILY in names or any("FZXiaoBiaoSong" in name for name in names)
if not accepted_name:
readable_names = "、".join(sorted(names)[:6]) or "无法读取"
raise WordExportFontError(
f"服务器字体不是“{TITLE_FONT_FAMILY}”(检测到:{readable_names}),已阻止导出。"
)
if fs_type & 0x0002:
raise WordExportFontError(f"“{TITLE_FONT_FAMILY}”授权标志禁止文档嵌入,已阻止导出。")
if fs_type & 0x0200:
raise WordExportFontError(f"“{TITLE_FONT_FAMILY}”仅允许位图嵌入,不能用于本次 Word 导出。")
digest = hashlib.sha256(data).hexdigest().upper()
if font_path == _BUNDLED_FONT_PATH.resolve() and digest != _BUNDLED_FONT_SHA256:
raise WordExportFontError("代码仓库中的 Word 标题字体校验和不匹配,已阻止导出。")
return FontAsset(data=data, path=font_path, names=names, fs_type=fs_type, sha256=digest)
def _obfuscate_font(data: bytes, font_key: str) -> bytes:
"""Apply the ECMA-376 ODTTF XOR obfuscation to the first 32 bytes."""
key_hex = font_key.replace("-", "")
if len(key_hex) != 32:
raise WordExportFontError("生成 Word 嵌入字体密钥失败。")
key = bytes.fromhex(key_hex)[::-1]
output = bytearray(data)
for index in range(min(32, len(output))):
output[index] ^= key[index % len(key)]
return bytes(output)
def _strip_inline_markdown(text: str) -> str:
text = _safe_xml_text(text)
text = re.sub(r"^\s*#+\s*", "", text)
text = re.sub(r"!\[([^\]]*)\]\([^)]*\)", r"\1", text)
text = re.sub(r"\[([^\]]+)\]\([^)]*\)", r"\1", text)
text = re.sub(r"(?:\*\*|__)(.*?)(?:\*\*|__)", r"\1", text)
return re.sub(r"[`*_~]", "", text).strip()
def _strip_manual_heading_number(text: str) -> str:
return re.sub(
r"^\s*(?:"
r"[一二三四五六七八九十百千]+[、..]"
r"|[((]\s*[一二三四五六七八九十百千]+\s*[))]"
r"|[((]\s*\d+(?:[..]\d+)*\s*[))]"
r"|\d+(?:[..]\d+)+(?:[..、])?"
r"|\d+[..、))]"
r")\s*",
"",
text,
).strip()
def _split_table_row(line: str) -> list[str]:
value = re.sub(r"^\s*\|", "", line)
value = re.sub(r"\|\s*$", "", value)
return [part.replace(r"\|", "|").strip() for part in re.split(r"(?<!\\)\|", value)]
def _is_table_separator(line: str) -> bool:
cells = _split_table_row(line)
return bool(cells) and all(re.fullmatch(r":?-{3,}:?", cell.replace(" ", "")) for cell in cells)
def _looks_like_table_row(line: str) -> bool:
return bool(re.match(r"^\s*\|.+", line))
def _is_table_caption(text: str) -> bool:
return bool(re.match(r"^\s*表\s*(?:\d+|[一二三四五六七八九十百]+|[::])", _strip_inline_markdown(text)))
def _extract_title(markdown: str, explicit_title: str | None) -> tuple[str | None, list[str]]:
lines = markdown.replace("\r\n", "\n").replace("\r", "\n").split("\n")
in_fence = False
for index, line in enumerate(lines):
if line.strip().startswith("```"):
in_fence = not in_fence
continue
if not in_fence:
match = re.match(r"^#\s+(.+)$", line)
if match:
del lines[index]
return _strip_inline_markdown(match.group(1)) or (explicit_title or "").strip() or None, lines
return (explicit_title or "").strip() or None, lines
def _png_size(data: bytes) -> tuple[int, int]:
if len(data) < 24 or data[:8] != _PNG_SIGNATURE:
raise WordExportError("嵌入的流程图不是有效的 PNG 图片。")
width, height = struct.unpack(">II", data[16:24])
if width < 1 or height < 1 or width > 20_000 or height > 20_000:
raise WordExportError("嵌入图片尺寸无效。")
return width, height
def _decode_png_base64(payload: str) -> bytes:
try:
data = base64.b64decode(re.sub(r"\s+", "", payload), validate=False)
except (ValueError, binascii.Error) as exc:
raise WordExportError("嵌入图片编码无效。") from exc
if not data or len(data) > _MAX_EMBEDDED_IMAGE_BYTES:
raise WordExportError("嵌入图片为空或超过 8 MB 限制。")
_png_size(data)
return data
def _resolve_markdown_image(
alt: str,
target: str,
embedded_images: dict[str, str],
image_index: int,
) -> Block:
payload: str | None = None
embedded = _EMBEDDED_IMAGE_ID.match(target.strip())
if embedded:
payload = embedded_images.get(embedded.group(1))
else:
data_uri = _DATA_PNG_URI.match(target.strip())
if data_uri:
payload = data_uri.group(1)
if not payload:
return ParagraphBlock("image", f"图片:{alt}|{target}")
try:
data = _decode_png_base64(payload)
width, height = _png_size(data)
except WordExportError:
return ParagraphBlock("image", f"图片:{alt}|{target}")
return ImageBlock(
alt=alt,
data=data,
media_name=f"image{image_index}.png",
width_px=width,
height_px=height,
)
def _parse_markdown(
markdown: str,
explicit_title: str | None,
embedded_images: dict[str, str] | None = None,
) -> tuple[str | None, list[Block]]:
title, lines = _extract_title(markdown, explicit_title)
blocks: list[Block] = []
paragraph_buffer: list[str] = []
in_fence = False
images = embedded_images or {}
image_count = 0
def flush_paragraph() -> None:
if paragraph_buffer:
blocks.append(ParagraphBlock("body", " ".join(paragraph_buffer)))
paragraph_buffer.clear()
index = 0
while index < len(lines):
raw = lines[index].replace("\t", " ")
if raw.strip().startswith("```"):
flush_paragraph()
in_fence = not in_fence
index += 1
continue
if in_fence:
blocks.append(ParagraphBlock("code", raw or " "))
index += 1
continue
if not raw.strip():
flush_paragraph()
index += 1
continue
heading = re.match(r"^(#{1,6})\s+(.+)$", raw)
if heading:
flush_paragraph()
markdown_level = max(2, len(heading.group(1)))
section_level = max(0, min(markdown_level - 2, 4))
blocks.append(ParagraphBlock("heading", _strip_manual_heading_number(heading.group(2)), section_level))
index += 1
continue
if _looks_like_table_row(raw) and index + 1 < len(lines) and _is_table_separator(lines[index + 1]):
caption: str | None = None
if paragraph_buffer and _is_table_caption(paragraph_buffer[-1]):
caption = paragraph_buffer.pop()
flush_paragraph()
if caption:
blocks.append(ParagraphBlock("table_caption", caption))
header = _split_table_row(raw)
index += 2
rows: list[list[str]] = []
while index < len(lines) and _looks_like_table_row(lines[index]):
rows.append(_split_table_row(lines[index]))
index += 1
blocks.append(TableBlock(header, rows))
continue
image = re.match(r'^\s*!\[([^\]]*)\]\(([^)]+)(?:\s+["\'][^"\']*["\'])?\)\s*$', raw)
if image:
flush_paragraph()
alt = image.group(1).strip() or "未命名图片"
target = image.group(2).strip()
if image_count >= _MAX_EMBEDDED_IMAGES and (
_EMBEDDED_IMAGE_ID.match(target) or _DATA_PNG_URI.match(target)
):
blocks.append(ParagraphBlock("image", f"图片:{alt}|{target}"))
else:
resolved = _resolve_markdown_image(alt, target, images, image_count + 1)
if isinstance(resolved, ImageBlock):
image_count += 1
blocks.append(resolved)
index += 1
continue
quote = re.match(r"^>\s?(.*)$", raw)
if quote:
flush_paragraph()
blocks.append(ParagraphBlock("quote", quote.group(1)))
index += 1
continue
ordered = re.match(r"^(\s*)(\d+)[.)、]\s+(.+)$", raw)
if ordered:
flush_paragraph()
blocks.append(ParagraphBlock("ordered_list", ordered.group(3), min(len(ordered.group(1)) // 2, 8)))
index += 1
continue
unordered = re.match(r"^(\s*)[-*+]\s+(.+)$", raw)
if unordered:
flush_paragraph()
blocks.append(ParagraphBlock("unordered_list", unordered.group(2), min(len(unordered.group(1)) // 2, 8)))
index += 1
continue
if re.fullmatch(r"\s*(?:---|\*\*\*|___)\s*", raw):
flush_paragraph()
index += 1
continue
paragraph_buffer.append(raw.strip())
index += 1
flush_paragraph()
return title, blocks
class _DocumentRelationships:
def __init__(self) -> None:
self._hyperlinks: list[tuple[str, str]] = []
self._images: list[tuple[str, str]] = []
self._drawing_id = 0
def add_hyperlink(self, target: str) -> str:
relation_id = f"rId{100 + len(self._hyperlinks)}"
self._hyperlinks.append((relation_id, target))
return relation_id
def add_image(self, target: str) -> tuple[str, int]:
self._drawing_id += 1
relation_id = f"rId{200 + len(self._images)}"
self._images.append((relation_id, target))
return relation_id, self._drawing_id
def xml(self) -> str:
fixed = [
("rId1", "http://schemas.openxmlformats.org/officeDocument/2006/relationships/styles", "styles.xml"),
("rId2", "http://schemas.openxmlformats.org/officeDocument/2006/relationships/numbering", "numbering.xml"),
("rId3", "http://schemas.openxmlformats.org/officeDocument/2006/relationships/settings", "settings.xml"),
("rId4", "http://schemas.openxmlformats.org/officeDocument/2006/relationships/fontTable", "fontTable.xml"),
("rId5", "http://schemas.openxmlformats.org/officeDocument/2006/relationships/footer", "footer1.xml"),
("rId6", "http://schemas.openxmlformats.org/officeDocument/2006/relationships/footer", "footer2.xml"),
]
relationships = "".join(
f'<Relationship Id="{relation_id}" Type="{rel_type}" Target="{target}"/>'
for relation_id, rel_type, target in fixed
)
relationships += "".join(
f'<Relationship Id="{relation_id}" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/hyperlink" Target="{_xml_attr(target)}" TargetMode="External"/>'
for relation_id, target in self._hyperlinks
)
relationships += "".join(
f'<Relationship Id="{relation_id}" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/image" Target="{_xml_attr(target)}"/>'
for relation_id, target in self._images
)
return f'{_XML_DECL}<Relationships xmlns="{_PACKAGE_REL_NS}">{relationships}</Relationships>'
_INLINE_PATTERN = re.compile(
r"\[([^\]]+)\]\(([^)\s]+)(?:\s+[\"\'][^\"\']*[\"\'])?\)|(`+)([^`]+)\3|\*\*([^*]+)\*\*|__([^_]+)__|\*([^*]+)\*|_([^_]+)_"
)
def _run_xml(
text: str,
*,
font: str = BODY_FONT_FAMILY,
size: int = BODY_SIZE_HALF_POINTS,
bold: bool = False,
italic: bool = False,
underline: bool = False,
) -> str:
text = _safe_xml_text(text)
preserve = text.startswith(" ") or text.endswith(" ") or " " in text
properties = (
f'<w:rFonts w:ascii="{_xml_attr(font)}" w:hAnsi="{_xml_attr(font)}" '
f'w:eastAsia="{_xml_attr(font)}" w:cs="{_xml_attr(font)}"/>'
'<w:color w:val="000000"/>'
f'<w:sz w:val="{size}"/><w:szCs w:val="{size}"/>'
)
if bold:
properties += "<w:b/><w:bCs/>"
if italic:
properties += "<w:i/><w:iCs/>"
if underline:
properties += '<w:u w:val="single"/>'
space = ' xml:space="preserve"' if preserve else ""
return f"<w:r><w:rPr>{properties}</w:rPr><w:t{space}>{escape(text)}</w:t></w:r>"
def _inline_xml(
text: str,
relationships: _DocumentRelationships,
*,
font: str = BODY_FONT_FAMILY,
size: int = BODY_SIZE_HALF_POINTS,
) -> str:
cleaned = _safe_xml_text(text)
output: list[str] = []
cursor = 0
for match in _INLINE_PATTERN.finditer(cleaned):
if match.start() > cursor:
output.append(_run_xml(cleaned[cursor : match.start()], font=font, size=size))
if match.group(1) is not None and match.group(2) is not None:
relation_id = relationships.add_hyperlink(match.group(2))
output.append(
f'<w:hyperlink r:id="{relation_id}" w:history="1">'
f'{_run_xml(match.group(1), font=font, size=size, underline=True)}</w:hyperlink>'
)
elif match.group(4) is not None:
output.append(_run_xml(match.group(4), font=font, size=size))
elif match.group(5) is not None:
output.append(_run_xml(match.group(5), font=font, size=size, bold=True))
elif match.group(6) is not None:
output.append(_run_xml(match.group(6), font=font, size=size, bold=True))
elif match.group(7) is not None:
output.append(_run_xml(match.group(7), font=font, size=size, italic=True))
elif match.group(8) is not None:
output.append(_run_xml(match.group(8), font=font, size=size, italic=True))
cursor = match.end()
if cursor < len(cleaned):
output.append(_run_xml(cleaned[cursor:], font=font, size=size))
return "".join(output) or _run_xml(" ", font=font, size=size)
def _paragraph_xml(
style: str,
children: str,
*,
numbering_level: int | None = None,
numbering_id: int = 1,
keep_next: bool = False,
alignment: str | None = None,
) -> str:
properties = f'<w:pStyle w:val="{style}"/>'
if numbering_level is not None:
properties += (
f'<w:numPr><w:ilvl w:val="{numbering_level}"/><w:numId w:val="{numbering_id}"/></w:numPr>'
)
if keep_next:
properties += "<w:keepNext/>"
if alignment:
properties += f'<w:jc w:val="{alignment}"/>'
return f"<w:p><w:pPr>{properties}</w:pPr>{children}</w:p>"
def _image_extent_emu(width_px: int, height_px: int) -> tuple[int, int]:
max_width_emu = CONTENT_WIDTH_TWIPS * _EMU_PER_TWIP
width_emu = max(1, int(width_px * 9525))
height_emu = max(1, int(height_px * 9525))
if width_emu > max_width_emu:
scale = max_width_emu / width_emu
width_emu = max_width_emu
height_emu = max(1, int(height_emu * scale))
return width_emu, height_emu
def _embedded_image_xml(block: ImageBlock, relationships: _DocumentRelationships) -> str:
relation_id, drawing_id = relationships.add_image(f"media/{block.media_name}")
cx, cy = _image_extent_emu(block.width_px, block.height_px)
name = _xml_attr(block.alt or block.media_name)
drawing = (
'<w:r><w:drawing><wp:inline distT="0" distB="0" distL="0" distR="0">'
f'<wp:extent cx="{cx}" cy="{cy}"/>'
f'<wp:docPr id="{drawing_id}" name="{name}"/>'
'<a:graphic xmlns:a="http://schemas.openxmlformats.org/drawingml/2006/main">'
'<a:graphicData uri="http://schemas.openxmlformats.org/drawingml/2006/picture">'
'<pic:pic xmlns:pic="http://schemas.openxmlformats.org/drawingml/2006/picture">'
f'<pic:nvPicPr><pic:cNvPr id="0" name="{name}"/><pic:cNvPicPr/></pic:nvPicPr>'
f'<pic:blipFill><a:blip r:embed="{relation_id}"/><a:stretch><a:fillRect/></a:stretch></pic:blipFill>'
"<pic:spPr>"
f'<a:xfrm><a:off x="0" y="0"/><a:ext cx="{cx}" cy="{cy}"/></a:xfrm>'
'<a:prstGeom prst="rect"><a:avLst/></a:prstGeom>'
"</pic:spPr></pic:pic></a:graphicData></a:graphic></wp:inline></w:drawing></w:r>"
)
return (
"<w:p><w:pPr>"
'<w:pStyle w:val="ImageCaption"/>'
'<w:jc w:val="center"/>'
'<w:spacing w:before="160" w:after="160" w:line="240" w:lineRule="auto"/>'
'<w:ind w:left="0" w:hanging="0" w:firstLineChars="0"/>'
f"</w:pPr>{drawing}</w:p>"
)
def _image_paragraph_xml(text: str, relationships: _DocumentRelationships) -> str:
label, _, target = text.partition("|")
children = _run_xml(label)
if target:
relation_id = relationships.add_hyperlink(target)
children += (
f'<w:hyperlink r:id="{relation_id}" w:history="1">'
f'{_run_xml(f"({target})", underline=True)}</w:hyperlink>'
)
return _paragraph_xml("ImageCaption", children, alignment="center")
def _cell_xml(text: str, width: int, relationships: _DocumentRelationships, *, header: bool) -> str:
style = "TableHeader" if header else "TableBody"
paragraph = _paragraph_xml(
style,
_inline_xml(text, relationships, font=TABLE_FONT_FAMILY, size=TABLE_SIZE_HALF_POINTS),
)
return (
"<w:tc>"
f'<w:tcPr><w:tcW w:w="{width}" w:type="dxa"/><w:vAlign w:val="center"/>'
'<w:tcMar><w:top w:w="80" w:type="dxa"/><w:left w:w="100" w:type="dxa"/>'
'<w:bottom w:w="80" w:type="dxa"/><w:right w:w="100" w:type="dxa"/></w:tcMar></w:tcPr>'
f"{paragraph}</w:tc>"
)
def _table_xml(block: TableBlock, relationships: _DocumentRelationships) -> str:
column_count = max(1, len(block.header))
base_width = CONTENT_WIDTH_TWIPS // column_count
widths = [base_width] * column_count
widths[-1] = CONTENT_WIDTH_TWIPS - base_width * (column_count - 1)
grid = "".join(f'<w:gridCol w:w="{width}"/>' for width in widths)
border = (
'<w:top w:val="single" w:sz="4" w:color="000000"/>'
'<w:left w:val="single" w:sz="4" w:color="000000"/>'
'<w:bottom w:val="single" w:sz="4" w:color="000000"/>'
'<w:right w:val="single" w:sz="4" w:color="000000"/>'
'<w:insideH w:val="single" w:sz="4" w:color="000000"/>'
'<w:insideV w:val="single" w:sz="4" w:color="000000"/>'
)
properties = (
f'<w:tblPr><w:tblW w:w="{CONTENT_WIDTH_TWIPS}" w:type="dxa"/>'
'<w:jc w:val="center"/><w:tblLayout w:type="fixed"/>'
f"<w:tblBorders>{border}</w:tblBorders></w:tblPr>"
)
header = list(block.header[:column_count]) + [""] * max(0, column_count - len(block.header))
header_cells = "".join(
_cell_xml(header[index], widths[index], relationships, header=True) for index in range(column_count)
)
rows = [f"<w:tr><w:trPr><w:tblHeader/><w:cantSplit/></w:trPr>{header_cells}</w:tr>"]
for source_row in block.rows:
normalized = list(source_row[:column_count]) + [""] * max(0, column_count - len(source_row))
cells = "".join(
_cell_xml(normalized[index], widths[index], relationships, header=False)
for index in range(column_count)
)
rows.append(f"<w:tr><w:trPr><w:cantSplit/></w:trPr>{cells}</w:tr>")
return f"<w:tbl>{properties}<w:tblGrid>{grid}</w:tblGrid>{''.join(rows)}</w:tbl>"
def _blocks_xml(blocks: list[Block], relationships: _DocumentRelationships) -> str:
output: list[str] = []
for block in blocks:
if isinstance(block, ImageBlock):
output.append(_embedded_image_xml(block, relationships))
continue
if isinstance(block, TableBlock):
output.append(_table_xml(block, relationships))
continue
if block.kind == "heading":
style = f"Heading{block.level + 1}"
font = (
HEADING_1_FONT_FAMILY
if block.level == 0
else HEADING_2_FONT_FAMILY
if block.level == 1
else BODY_FONT_FAMILY
)
output.append(
_paragraph_xml(
style,
_inline_xml(block.text, relationships, font=font),
numbering_level=block.level,
numbering_id=1,
keep_next=True,
)
)
elif block.kind == "table_caption":
output.append(
_paragraph_xml(
"TableCaption",
_inline_xml(block.text, relationships, font=HEADING_1_FONT_FAMILY, size=TABLE_SIZE_HALF_POINTS),
keep_next=True,
)
)
elif block.kind == "ordered_list":
output.append(
_paragraph_xml(
"ListText",
_inline_xml(block.text, relationships),
numbering_level=block.level,
numbering_id=2,
)
)
elif block.kind == "unordered_list":
output.append(
_paragraph_xml(
"ListText",
_inline_xml(block.text, relationships),
numbering_level=block.level,
numbering_id=3,
)
)
elif block.kind == "code":
output.append(_paragraph_xml("CodeText", _run_xml(block.text)))
elif block.kind == "quote":
output.append(_paragraph_xml("QuoteText", _inline_xml(block.text, relationships)))
elif block.kind == "image":
output.append(_image_paragraph_xml(block.text, relationships))
else:
output.append(_paragraph_xml("BodyText", _inline_xml(block.text, relationships)))
return "".join(output)
def _font_run_properties(font: str, size: int, *, bold: bool = False) -> str:
value = (
f'<w:rFonts w:ascii="{_xml_attr(font)}" w:hAnsi="{_xml_attr(font)}" '
f'w:eastAsia="{_xml_attr(font)}" w:cs="{_xml_attr(font)}"/>'
'<w:color w:val="000000"/>'
f'<w:sz w:val="{size}"/><w:szCs w:val="{size}"/>'
)
return value + ("<w:b/><w:bCs/>" if bold else '<w:b w:val="0"/><w:bCs w:val="0"/>')
def _style_xml(
style_id: str,
name: str,
*,
font: str,
size: int,
based_on: str | None = "Normal",
next_style: str = "BodyText",
alignment: str = "both",
first_line_chars: int = 0,
left: int = 0,
hanging: int = 0,
keep_next: bool = False,
keep_lines: bool = False,
outline_level: int | None = None,
bold: bool = False,
) -> str:
paragraph = (
f'<w:jc w:val="{alignment}"/><w:spacing w:before="0" w:after="0" '
f'w:line="{EXACT_LINE_TWIPS}" w:lineRule="exact"/>'
f'<w:ind w:left="{left}" w:hanging="{hanging}" w:firstLineChars="{first_line_chars}"/>'
)
if keep_next:
paragraph += "<w:keepNext/>"
if keep_lines:
paragraph += "<w:keepLines/>"
if outline_level is not None:
paragraph += f'<w:outlineLvl w:val="{outline_level}"/>'
based_on_xml = f'<w:basedOn w:val="{based_on}"/>' if based_on else ""
return (
f'<w:style w:type="paragraph" w:styleId="{style_id}"><w:name w:val="{_xml_attr(name)}"/>'
f'{based_on_xml}<w:next w:val="{next_style}"/><w:qFormat/>'
f"<w:pPr>{paragraph}</w:pPr><w:rPr>{_font_run_properties(font, size, bold=bold)}</w:rPr></w:style>"
)
def _styles_xml() -> str:
defaults = (
"<w:docDefaults><w:rPrDefault><w:rPr>"
f"{_font_run_properties(BODY_FONT_FAMILY, BODY_SIZE_HALF_POINTS)}"
'<w:lang w:val="zh-CN" w:eastAsia="zh-CN"/></w:rPr></w:rPrDefault>'
'<w:pPrDefault><w:pPr><w:jc w:val="both"/>'
f'<w:spacing w:before="0" w:after="0" w:line="{EXACT_LINE_TWIPS}" w:lineRule="exact"/>'
'<w:ind w:firstLineChars="200"/></w:pPr></w:pPrDefault></w:docDefaults>'
)
styles = [
_style_xml(
"Normal",
"Normal",
font=BODY_FONT_FAMILY,
size=BODY_SIZE_HALF_POINTS,
based_on=None,
next_style="BodyText",
first_line_chars=200,
),
_style_xml(
"DocumentTitle",
"Document Title",
font=TITLE_FONT_FAMILY,
size=TITLE_SIZE_HALF_POINTS,
alignment="center",
keep_next=True,
),
_style_xml(
"BodyText",
"Body Text",
font=BODY_FONT_FAMILY,
size=BODY_SIZE_HALF_POINTS,
first_line_chars=200,
),
_style_xml(
"Heading1",
"heading 1",
font=HEADING_1_FONT_FAMILY,
size=BODY_SIZE_HALF_POINTS,
alignment="left",
first_line_chars=200,
keep_next=True,
keep_lines=True,
outline_level=0,
),
_style_xml(
"Heading2",
"heading 2",
font=HEADING_2_FONT_FAMILY,
size=BODY_SIZE_HALF_POINTS,
alignment="left",
first_line_chars=200,
keep_next=True,
keep_lines=True,
outline_level=1,
),
]
for level in range(3, 6):
styles.append(
_style_xml(
f"Heading{level}",
f"heading {level}",
font=BODY_FONT_FAMILY,
size=BODY_SIZE_HALF_POINTS,
alignment="left",
first_line_chars=200 if level == 3 else 0,
keep_next=True,
keep_lines=True,
outline_level=level - 1,
)
)
styles.extend(
[
_style_xml(
"ListText",
"List Text",
font=BODY_FONT_FAMILY,
size=BODY_SIZE_HALF_POINTS,
alignment="both",
),
_style_xml(
"QuoteText",
"Quote Text",
font=BODY_FONT_FAMILY,
size=BODY_SIZE_HALF_POINTS,
alignment="both",
first_line_chars=200,
left=640,
),
_style_xml(
"CodeText",
"Code Text",
font=BODY_FONT_FAMILY,
size=BODY_SIZE_HALF_POINTS,
alignment="left",
),
_style_xml(
"TableCaption",
"Table Caption",
font=HEADING_1_FONT_FAMILY,
size=TABLE_SIZE_HALF_POINTS,
alignment="center",
keep_next=True,
),
_style_xml(
"TableHeader",
"Table Header",
font=TABLE_FONT_FAMILY,
size=TABLE_SIZE_HALF_POINTS,
alignment="center",
keep_lines=True,
),
_style_xml(
"TableBody",
"Table Body",
font=TABLE_FONT_FAMILY,
size=TABLE_SIZE_HALF_POINTS,
alignment="both",
keep_lines=True,
),
_style_xml(
"ImageCaption",
"Image Caption",
font=BODY_FONT_FAMILY,
size=BODY_SIZE_HALF_POINTS,
alignment="center",
keep_lines=True,
),
]
)
return f'{_XML_DECL}<w:styles xmlns:w="{_W_NS}">{defaults}{"".join(styles)}</w:styles>'
def _heading_level_xml(level: int, format_value: str, text: str, font: str) -> str:
restart = f'<w:lvlRestart w:val="{level}"/>' if level > 0 else ""
first_line_chars = ' w:firstLineChars="200"' if level <= 2 else ""
return (
f'<w:lvl w:ilvl="{level}"><w:start w:val="1"/>{restart}<w:numFmt w:val="{format_value}"/>'
f'<w:lvlText w:val="{_xml_attr(text)}"/><w:lvlJc w:val="left"/>'
f'<w:pPr><w:ind w:left="0" w:hanging="0"{first_line_chars}/></w:pPr>'
f'<w:rPr>{_font_run_properties(font, BODY_SIZE_HALF_POINTS)}</w:rPr></w:lvl>'
)
def _list_abstract_xml(abstract_id: int, *, ordered: bool) -> str:
levels: list[str] = []
bullets = ["●", "○", "▪"]
for level in range(9):
format_value = "decimal" if ordered else "bullet"
text = f"%{level + 1}." if ordered else bullets[level % len(bullets)]
left = 720 * (level + 1)
levels.append(
f'<w:lvl w:ilvl="{level}"><w:start w:val="1"/><w:numFmt w:val="{format_value}"/>'
f'<w:lvlText w:val="{_xml_attr(text)}"/><w:lvlJc w:val="left"/>'
f'<w:pPr><w:tabs><w:tab w:val="num" w:pos="{left}"/></w:tabs>'
f'<w:ind w:left="{left}" w:hanging="360"/></w:pPr>'
f'<w:rPr>{_font_run_properties(BODY_FONT_FAMILY, BODY_SIZE_HALF_POINTS)}</w:rPr></w:lvl>'
)
return f'<w:abstractNum w:abstractNumId="{abstract_id}"><w:multiLevelType w:val="multilevel"/>{"".join(levels)}</w:abstractNum>'
def _numbering_xml() -> str:
heading_levels = "".join(
[
_heading_level_xml(0, "chineseCounting", "%1、", HEADING_1_FONT_FAMILY),
_heading_level_xml(1, "chineseCounting", "(%2)", HEADING_2_FONT_FAMILY),
_heading_level_xml(2, "decimal", "%3.", BODY_FONT_FAMILY),
_heading_level_xml(3, "decimal", "(%4)", BODY_FONT_FAMILY),
_heading_level_xml(4, "decimal", "%5)", BODY_FONT_FAMILY),
]
)
return (
f'{_XML_DECL}<w:numbering xmlns:w="{_W_NS}">'
f'<w:abstractNum w:abstractNumId="0"><w:multiLevelType w:val="multilevel">'
f"</w:multiLevelType>{heading_levels}</w:abstractNum>"
f'{_list_abstract_xml(1, ordered=True)}{_list_abstract_xml(2, ordered=False)}'
'<w:num w:numId="1"><w:abstractNumId w:val="0"/></w:num>'
'<w:num w:numId="2"><w:abstractNumId w:val="1"/></w:num>'
'<w:num w:numId="3"><w:abstractNumId w:val="2"/></w:num>'
"</w:numbering>"
)
def _footer_xml(alignment: str) -> str:
run_properties = _font_run_properties(TABLE_FONT_FAMILY, TABLE_SIZE_HALF_POINTS)
return (
f'{_XML_DECL}<w:ftr xmlns:w="{_W_NS}" xmlns:r="{_R_NS}">'
f'<w:p><w:pPr><w:jc w:val="{alignment}"/><w:spacing w:before="0" w:after="0"/></w:pPr>'
f'<w:r><w:rPr>{run_properties}</w:rPr><w:fldChar w:fldCharType="begin"/></w:r>'
f'<w:r><w:rPr>{run_properties}</w:rPr><w:instrText xml:space="preserve"> PAGE </w:instrText></w:r>'
f'<w:r><w:rPr>{run_properties}</w:rPr><w:fldChar w:fldCharType="separate"/></w:r>'
f'<w:r><w:rPr>{run_properties}</w:rPr><w:t>1</w:t></w:r>'
f'<w:r><w:rPr>{run_properties}</w:rPr><w:fldChar w:fldCharType="end"/></w:r>'
"</w:p></w:ftr>"
)
def _font_table_xml(font_key: str) -> str:
def font(name: str, *, embedded: bool = False) -> str:
embed = f'<w:embedRegular r:id="rId1" w:fontKey="{{{font_key}}}"/>' if embedded else ""
return (
f'<w:font w:name="{_xml_attr(name)}"><w:family w:val="roman"/>'
f'<w:charset w:val="86"/>{embed}</w:font>'
)
fonts = "".join(
[
font(TITLE_FONT_FAMILY, embedded=True),
font(BODY_FONT_FAMILY),
font(HEADING_1_FONT_FAMILY),
font(HEADING_2_FONT_FAMILY),
font(TABLE_FONT_FAMILY),
]
)
return f'{_XML_DECL}<w:fonts xmlns:w="{_W_NS}" xmlns:r="{_R_NS}">{fonts}</w:fonts>'
def _font_table_relationships_xml() -> str:
return (
f'{_XML_DECL}<Relationships xmlns="{_PACKAGE_REL_NS}">'
'<Relationship Id="rId1" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/font" Target="fonts/font1.odttf"/>'
"</Relationships>"
)
def _settings_xml() -> str:
return (
f'{_XML_DECL}<w:settings xmlns:w="{_W_NS}">'
'<w:zoom w:percent="100"/><w:evenAndOddHeaders/><w:updateFields w:val="true"/>'
'<w:compat><w:compatSetting w:name="compatibilityMode" '
'w:uri="http://schemas.microsoft.com/office/word" w:val="15"/></w:compat>'
"</w:settings>"
)
def _document_xml(title: str | None, blocks: list[Block], relationships: _DocumentRelationships) -> str:
body: list[str] = []
if title:
body.append(
_paragraph_xml(
"DocumentTitle",
_run_xml(title, font=TITLE_FONT_FAMILY, size=TITLE_SIZE_HALF_POINTS),
keep_next=True,
alignment="center",
)
)
body.append(_blocks_xml(blocks, relationships))
section = (
'<w:sectPr><w:footerReference w:type="default" r:id="rId5"/>'
'<w:footerReference w:type="even" r:id="rId6"/>'
f'<w:pgSz w:w="{A4_WIDTH_TWIPS}" w:h="{A4_HEIGHT_TWIPS}" w:orient="portrait"/>'
f'<w:pgMar w:top="{MARGIN_TOP_TWIPS}" w:right="{MARGIN_RIGHT_TWIPS}" '
f'w:bottom="{MARGIN_BOTTOM_TWIPS}" w:left="{MARGIN_LEFT_TWIPS}" '
'w:header="720" w:footer="992" w:gutter="0"/>'
'<w:pgNumType w:start="1" w:fmt="decimal"/><w:cols w:space="720"/></w:sectPr>'
)
return (
f'{_XML_DECL}<w:document xmlns:w="{_W_NS}" xmlns:r="{_R_NS}" '
'xmlns:wp="http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing">'
f"<w:body>{''.join(body)}{section}</w:body></w:document>"
)
def _content_types_xml() -> str:
return (
f'{_XML_DECL}<Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types">'
'<Default Extension="rels" ContentType="application/vnd.openxmlformats-package.relationships+xml"/>'
'<Default Extension="xml" ContentType="application/xml"/>'
'<Default Extension="odttf" ContentType="application/vnd.openxmlformats-officedocument.obfuscatedFont"/>'
'<Default Extension="png" ContentType="image/png"/>'
'<Override PartName="/word/document.xml" ContentType="application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml"/>'
'<Override PartName="/word/styles.xml" ContentType="application/vnd.openxmlformats-officedocument.wordprocessingml.styles+xml"/>'
'<Override PartName="/word/numbering.xml" ContentType="application/vnd.openxmlformats-officedocument.wordprocessingml.numbering+xml"/>'
'<Override PartName="/word/settings.xml" ContentType="application/vnd.openxmlformats-officedocument.wordprocessingml.settings+xml"/>'
'<Override PartName="/word/fontTable.xml" ContentType="application/vnd.openxmlformats-officedocument.wordprocessingml.fontTable+xml"/>'
'<Override PartName="/word/footer1.xml" ContentType="application/vnd.openxmlformats-officedocument.wordprocessingml.footer+xml"/>'
'<Override PartName="/word/footer2.xml" ContentType="application/vnd.openxmlformats-officedocument.wordprocessingml.footer+xml"/>'
'<Override PartName="/docProps/core.xml" ContentType="application/vnd.openxmlformats-package.core-properties+xml"/>'
'<Override PartName="/docProps/app.xml" ContentType="application/vnd.openxmlformats-officedocument.extended-properties+xml"/>'
"</Types>"
)
def _package_relationships_xml() -> str:
return (
f'{_XML_DECL}<Relationships xmlns="{_PACKAGE_REL_NS}">'
'<Relationship Id="rId1" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" Target="word/document.xml"/>'
'<Relationship Id="rId2" Type="http://schemas.openxmlformats.org/package/2006/relationships/metadata/core-properties" Target="docProps/core.xml"/>'
'<Relationship Id="rId3" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/extended-properties" Target="docProps/app.xml"/>'
"</Relationships>"
)
def _core_properties_xml(title: str | None) -> str:
now = datetime.now(UTC).replace(microsecond=0).isoformat().replace("+00:00", "Z")
safe_title = escape(title or "Word 文档")
return (
f'{_XML_DECL}<cp:coreProperties xmlns:cp="http://schemas.openxmlformats.org/package/2006/metadata/core-properties" '
'xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:dcterms="http://purl.org/dc/terms/" '
'xmlns:dcmitype="http://purl.org/dc/dcmitype/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">'
f"<dc:title>{safe_title}</dc:title><dc:creator>DeerFlow</dc:creator>"
f'<dcterms:created xsi:type="dcterms:W3CDTF">{now}</dcterms:created>'
f'<dcterms:modified xsi:type="dcterms:W3CDTF">{now}</dcterms:modified></cp:coreProperties>'
)
def _app_properties_xml() -> str:
return (
f'{_XML_DECL}<Properties xmlns="http://schemas.openxmlformats.org/officeDocument/2006/extended-properties" '
'xmlns:vt="http://schemas.openxmlformats.org/officeDocument/2006/docPropsVTypes">'
"<Application>DeerFlow</Application><AppVersion>1.0</AppVersion></Properties>"
)
def build_markdown_docx(
markdown: str,
title: str | None = None,
*,
font_path: Path | None = None,
embedded_images: dict[str, str] | None = None,
) -> bytes:
"""Build a formal DOCX and embed the validated server-side title font."""
if not markdown or not markdown.strip():
raise WordExportError("文档内容为空。")
if len(markdown) > 2_000_000:
raise WordExportError("文档内容超过 200 万字符,无法导出。")
if embedded_images and len(embedded_images) > _MAX_EMBEDDED_IMAGES:
raise WordExportError("嵌入图片数量超过限制。")
font = load_title_font(font_path)
resolved_title, blocks = _parse_markdown(markdown, title, embedded_images)
relationships = _DocumentRelationships()
document_xml = _document_xml(resolved_title, blocks, relationships)
font_key = str(uuid.uuid4()).upper()
media_files = {
f"word/media/{block.media_name}": block.data for block in blocks if isinstance(block, ImageBlock)
}
output = io.BytesIO()
with zipfile.ZipFile(output, mode="w", compression=zipfile.ZIP_DEFLATED, compresslevel=6) as archive:
parts: dict[str, str | bytes] = {
"[Content_Types].xml": _content_types_xml(),
"_rels/.rels": _package_relationships_xml(),
"docProps/core.xml": _core_properties_xml(resolved_title),
"docProps/app.xml": _app_properties_xml(),
"word/document.xml": document_xml,
"word/_rels/document.xml.rels": relationships.xml(),
"word/styles.xml": _styles_xml(),
"word/numbering.xml": _numbering_xml(),
"word/settings.xml": _settings_xml(),
"word/fontTable.xml": _font_table_xml(font_key),
"word/_rels/fontTable.xml.rels": _font_table_relationships_xml(),
# Word applies the default footer to odd pages and the even footer
# to even pages. The requested placement is odd-left/even-right.
"word/footer1.xml": _footer_xml("left"),
"word/footer2.xml": _footer_xml("right"),
"word/fonts/font1.odttf": _obfuscate_font(font.data, font_key),
**media_files,
}
for name, value in parts.items():
archive.writestr(name, value.encode("utf-8") if isinstance(value, str) else value)
return output.getvalue()
__all__ = [
"DOCX_MEDIA_TYPE",
"TITLE_FONT_FAMILY",
"WordExportError",
"WordExportFontError",
"build_markdown_docx",
"load_title_font",
"resolve_title_font_path",
]