deerflow-code/offline-backend-20260512/backend/tests/test_word_export.py
2026-09-07 18:24:55 +08:00

178 lines
6.9 KiB
Python

from __future__ import annotations
import base64
import hashlib
import importlib.util
import io
import re
import sys
import zipfile
from pathlib import Path
from xml.etree import ElementTree
import pytest
BACKEND_ROOT = Path(__file__).resolve().parents[1]
MODULE_PATH = BACKEND_ROOT / "app" / "gateway" / "word_export.py"
def _load_word_export_module():
"""Load the leaf module without executing app.gateway.__init__."""
spec = importlib.util.spec_from_file_location("word_export_test_module", MODULE_PATH)
assert spec is not None and spec.loader is not None
module = importlib.util.module_from_spec(spec)
sys.modules[spec.name] = module
spec.loader.exec_module(module)
return module
word_export = _load_word_export_module()
def test_bundled_font_is_exact_and_allows_editable_embedding() -> None:
font = word_export.load_title_font()
assert font.sha256 == "C52D577DD3AA719EF5FB9FF2043B7F7CA36FAE468EF34A47628A3A555D006A49"
assert font.fs_type == 0x0008
assert any("FZXiaoBiaoSong" in name for name in font.names)
def test_formal_docx_contains_required_layout_numbering_and_embedded_font(tmp_path: Path) -> None:
markdown = """# 样例标题
## 一、一级标题
这是正文,用于验证首行缩进、两端对齐和固定行距。
### 2.1 二级标题
#### 2.1.1 三级标题
---
表1 测试表格
| 项目 | 数值 |
|---|---:|
| 页面 | A4 纵向 |
"""
docx_path = tmp_path / "sample.docx"
docx_path.write_bytes(word_export.build_markdown_docx(markdown))
with zipfile.ZipFile(docx_path) as archive:
names = set(archive.namelist())
assert "word/fonts/font1.odttf" in names
assert "word/_rels/fontTable.xml.rels" in names
for name in names:
if name.endswith(".xml") or name.endswith(".rels"):
ElementTree.fromstring(archive.read(name))
document = archive.read("word/document.xml").decode("utf-8")
styles = archive.read("word/styles.xml").decode("utf-8")
numbering = archive.read("word/numbering.xml").decode("utf-8")
settings = archive.read("word/settings.xml").decode("utf-8")
font_table = archive.read("word/fontTable.xml").decode("utf-8")
default_footer = archive.read("word/footer1.xml").decode("utf-8")
even_footer = archive.read("word/footer2.xml").decode("utf-8")
assert '<w:pgSz w:w="11906" w:h="16838" w:orient="portrait"/>' in document
assert 'w:top="2098"' in document
assert 'w:bottom="1984"' in document
assert 'w:left="1587"' in document
assert 'w:right="1474"' in document
assert '<w:pgNumType w:start="1" w:fmt="decimal"/>' in document
assert '<w:pStyle w:val="DocumentTitle"/>' in document
assert "方正小标宋简体" in document
assert "一、一级标题" not in document
assert "2.1 二级标题" not in document
assert "2.1.1 三级标题" not in document
assert "────────" not in document
assert "二级标题" in document
assert "三级标题" in document
assert 'w:line="579" w:lineRule="exact"' in styles
assert '<w:ind w:left="0" w:hanging="0" w:firstLineChars="200"/>' in styles
assert "方正小标宋简体" in styles
assert 'w:val="44"' in styles
assert "仿宋" in styles and 'w:val="32"' in styles
assert "宋体" in styles and 'w:val="28"' in styles
assert styles.count('w:firstLineChars="200"') >= 6
assert '<w:numFmt w:val="chineseCounting"/>' in numbering
assert '<w:lvlText w:val="%1、"/>' in numbering
assert '<w:lvlText w:val="(%2)"/>' in numbering
assert '<w:lvlText w:val="%3."/>' in numbering
assert numbering.count('w:firstLineChars="200"') == 3
assert '<w:evenAndOddHeaders/>' in settings
assert '<w:updateFields w:val="true"/>' in settings
assert "方正小标宋简体" in font_table
key_match = re.search(r'w:fontKey="\{([0-9A-F-]{36})\}"', font_table)
assert key_match is not None
font_key = key_match.group(1)
restored = word_export._obfuscate_font(archive.read("word/fonts/font1.odttf"), font_key)
assert hashlib.sha256(restored).hexdigest().upper() == word_export._BUNDLED_FONT_SHA256
# Word's default footer applies to odd pages; the separate even footer
# applies to even pages. The requested alignment is odd-left/even-right.
assert '<w:jc w:val="left"/>' in default_footer
assert '<w:jc w:val="right"/>' in even_footer
assert " PAGE " in default_footer and " PAGE " in even_footer
_TINY_PNG = base64.b64decode(
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
)
def test_embedded_png_is_written_into_docx_media() -> None:
markdown = "正文前\n\n![流程图](embedded:mermaid-1)\n\n正文后"
payload = word_export.build_markdown_docx(
markdown,
embedded_images={"mermaid-1": base64.b64encode(_TINY_PNG).decode("ascii")},
)
with zipfile.ZipFile(io.BytesIO(payload)) as archive:
assert archive.read("word/media/image1.png") == _TINY_PNG
document = archive.read("word/document.xml").decode("utf-8")
rels = archive.read("word/_rels/document.xml.rels").decode("utf-8")
types = archive.read("[Content_Types].xml").decode("utf-8")
assert "<w:drawing>" in document
assert 'w:lineRule="auto"' in document
assert "flowchart TB" not in document
assert 'Target="media/image1.png"' in rels
assert 'Extension="png"' in types
def test_mermaid_fence_without_image_stays_code() -> None:
markdown = "```mermaid\nflowchart TB\nA-->B\n```"
payload = word_export.build_markdown_docx(markdown)
with zipfile.ZipFile(io.BytesIO(payload)) as archive:
document = archive.read("word/document.xml").decode("utf-8")
assert "flowchart TB" in document
assert "<w:drawing>" not in document
assert "word/media/image1.png" not in archive.namelist()
def test_export_rejects_wrong_font(tmp_path: Path) -> None:
wrong_font = tmp_path / "wrong.ttf"
wrong_font.write_bytes(b"not-a-font")
with pytest.raises(word_export.WordExportFontError):
word_export.build_markdown_docx("正文", font_path=wrong_font)
@pytest.mark.parametrize(
("source", "expected"),
[
("一、一级标题", "一级标题"),
("(一) 二级标题", "二级标题"),
("(一)二级标题", "二级标题"),
("1. 三级标题", "三级标题"),
("2.1 二级标题", "二级标题"),
("2.1.1 三级标题", "三级标题"),
("2.1.1. 三级标题", "三级标题"),
("正文标题", "正文标题"),
],
)
def test_manual_heading_number_is_removed_as_one_complete_prefix(source: str, expected: str) -> None:
assert word_export._strip_manual_heading_number(source) == expected