from __future__ import annotations
import base64
import hashlib
import importlib.util
import io
import re
import sys
import zipfile
from pathlib import Path
from xml.etree import ElementTree
import pytest
BACKEND_ROOT = Path(__file__).resolve().parents[1]
MODULE_PATH = BACKEND_ROOT / "app" / "gateway" / "word_export.py"
def _load_word_export_module():
"""Load the leaf module without executing app.gateway.__init__."""
spec = importlib.util.spec_from_file_location("word_export_test_module", MODULE_PATH)
assert spec is not None and spec.loader is not None
module = importlib.util.module_from_spec(spec)
sys.modules[spec.name] = module
spec.loader.exec_module(module)
return module
word_export = _load_word_export_module()
def test_bundled_font_is_exact_and_allows_editable_embedding() -> None:
font = word_export.load_title_font()
assert font.sha256 == "C52D577DD3AA719EF5FB9FF2043B7F7CA36FAE468EF34A47628A3A555D006A49"
assert font.fs_type == 0x0008
assert any("FZXiaoBiaoSong" in name for name in font.names)
def test_formal_docx_contains_required_layout_numbering_and_embedded_font(tmp_path: Path) -> None:
markdown = """# 样例标题
## 一、一级标题
这是正文,用于验证首行缩进、两端对齐和固定行距。
### 2.1 二级标题
#### 2.1.1 三级标题
---
表1 测试表格
| 项目 | 数值 |
|---|---:|
| 页面 | A4 纵向 |
"""
docx_path = tmp_path / "sample.docx"
docx_path.write_bytes(word_export.build_markdown_docx(markdown))
with zipfile.ZipFile(docx_path) as archive:
names = set(archive.namelist())
assert "word/fonts/font1.odttf" in names
assert "word/_rels/fontTable.xml.rels" in names
for name in names:
if name.endswith(".xml") or name.endswith(".rels"):
ElementTree.fromstring(archive.read(name))
document = archive.read("word/document.xml").decode("utf-8")
styles = archive.read("word/styles.xml").decode("utf-8")
numbering = archive.read("word/numbering.xml").decode("utf-8")
settings = archive.read("word/settings.xml").decode("utf-8")
font_table = archive.read("word/fontTable.xml").decode("utf-8")
default_footer = archive.read("word/footer1.xml").decode("utf-8")
even_footer = archive.read("word/footer2.xml").decode("utf-8")
assert '' in document
assert 'w:top="2098"' in document
assert 'w:bottom="1984"' in document
assert 'w:left="1587"' in document
assert 'w:right="1474"' in document
assert '' in document
assert '' in document
assert "方正小标宋简体" in document
assert "一、一级标题" not in document
assert "2.1 二级标题" not in document
assert "2.1.1 三级标题" not in document
assert "────────" not in document
assert "二级标题" in document
assert "三级标题" in document
assert 'w:line="579" w:lineRule="exact"' in styles
assert '' in styles
assert "方正小标宋简体" in styles
assert 'w:val="44"' in styles
assert "仿宋" in styles and 'w:val="32"' in styles
assert "宋体" in styles and 'w:val="28"' in styles
assert styles.count('w:firstLineChars="200"') >= 6
assert '' in numbering
assert '' in numbering
assert '' in numbering
assert '' in numbering
assert numbering.count('w:firstLineChars="200"') == 3
assert '' in settings
assert '' in settings
assert "方正小标宋简体" in font_table
key_match = re.search(r'w:fontKey="\{([0-9A-F-]{36})\}"', font_table)
assert key_match is not None
font_key = key_match.group(1)
restored = word_export._obfuscate_font(archive.read("word/fonts/font1.odttf"), font_key)
assert hashlib.sha256(restored).hexdigest().upper() == word_export._BUNDLED_FONT_SHA256
# Word's default footer applies to odd pages; the separate even footer
# applies to even pages. The requested alignment is odd-left/even-right.
assert '' in default_footer
assert '' in even_footer
assert " PAGE " in default_footer and " PAGE " in even_footer
_TINY_PNG = base64.b64decode(
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
)
def test_embedded_png_is_written_into_docx_media() -> None:
markdown = "正文前\n\n\n\n正文后"
payload = word_export.build_markdown_docx(
markdown,
embedded_images={"mermaid-1": base64.b64encode(_TINY_PNG).decode("ascii")},
)
with zipfile.ZipFile(io.BytesIO(payload)) as archive:
assert archive.read("word/media/image1.png") == _TINY_PNG
document = archive.read("word/document.xml").decode("utf-8")
rels = archive.read("word/_rels/document.xml.rels").decode("utf-8")
types = archive.read("[Content_Types].xml").decode("utf-8")
assert "" in document
assert 'w:lineRule="auto"' in document
assert "flowchart TB" not in document
assert 'Target="media/image1.png"' in rels
assert 'Extension="png"' in types
def test_mermaid_fence_without_image_stays_code() -> None:
markdown = "```mermaid\nflowchart TB\nA-->B\n```"
payload = word_export.build_markdown_docx(markdown)
with zipfile.ZipFile(io.BytesIO(payload)) as archive:
document = archive.read("word/document.xml").decode("utf-8")
assert "flowchart TB" in document
assert "" not in document
assert "word/media/image1.png" not in archive.namelist()
def test_export_rejects_wrong_font(tmp_path: Path) -> None:
wrong_font = tmp_path / "wrong.ttf"
wrong_font.write_bytes(b"not-a-font")
with pytest.raises(word_export.WordExportFontError):
word_export.build_markdown_docx("正文", font_path=wrong_font)
@pytest.mark.parametrize(
("source", "expected"),
[
("一、一级标题", "一级标题"),
("(一) 二级标题", "二级标题"),
("(一)二级标题", "二级标题"),
("1. 三级标题", "三级标题"),
("2.1 二级标题", "二级标题"),
("2.1.1 三级标题", "三级标题"),
("2.1.1. 三级标题", "三级标题"),
("正文标题", "正文标题"),
],
)
def test_manual_heading_number_is_removed_as_one_complete_prefix(source: str, expected: str) -> None:
assert word_export._strip_manual_heading_number(source) == expected