797 lines
29 KiB
Python
797 lines
29 KiB
Python
"""HTML 模板「JSON 填充」纯引擎 + 模板注册表。
|
||
|
||
定时任务的「模板数据转换器」内置 agent 把用户传入的任意 JSON 智能映射成某个
|
||
HTML 模板需要的 ``DATA`` 结构后,本模块负责把数据**填充**进模板并产出一份完整、
|
||
可预览的 HTML 文档。
|
||
|
||
设计要点
|
||
--------
|
||
* **模板自身即 schema 来源**:每个模板的 HTML 里都内嵌一行 ``const DATA={...};``
|
||
(台湾滨海防卫展示系统 no.1 还内嵌 ``const MANIFEST={...};``)。我们直接解析这行
|
||
拿到「参考数据」,既当**目标格式的范例**(喂给大模型),又当**结构校验**的基准
|
||
(顶层键 + 类型)。注册表因此不需要手写一份庞大的 schema 字面量,改模板即改契约。
|
||
* **行级安全替换**:模板把每个常量压在**单独一行**(``const DATA=...;`` 以分号结尾、
|
||
行内无换行),所以用 ``^const VAR=.*;$`` (MULTILINE) 精确命中该行整体替换,绝不
|
||
误伤模板里别处出现的同名子串。替换值用函数回调写入,避开 ``re.sub`` 把 JSON 里的
|
||
反斜杠当反向引用的坑。
|
||
* **harness 层、零 app.* 依赖**:可被调度器(harness)与 Gateway 路由(app)共同 import。
|
||
|
||
公开入口:``list_templates`` / ``get_template`` / ``reference_data`` /
|
||
``schema_skeleton_json`` / ``validate_data`` / ``field_coverage_gaps`` /
|
||
``shape_mismatches`` / ``fill_template``。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import json
|
||
import re
|
||
from dataclasses import dataclass
|
||
from datetime import date
|
||
from functools import lru_cache
|
||
from pathlib import Path
|
||
from typing import Any
|
||
|
||
_FENCE_RE = re.compile(r"```(?:json)?\s*(.*?)```", re.DOTALL | re.IGNORECASE)
|
||
|
||
|
||
def _scan_balanced(text: str, open_ch: str, close_ch: str) -> Any | None:
|
||
"""从 text 里扫第一个**平衡**的 ``open_ch…close_ch`` 块并 ``json.loads``。
|
||
|
||
字符串感知(跳过引号内的括号/转义),失败返回 None。供从「含 JSON 的文本」里
|
||
把 JSON 段抠出来用。
|
||
"""
|
||
start = text.find(open_ch)
|
||
if start == -1:
|
||
return None
|
||
depth = 0
|
||
in_string = False
|
||
escape = False
|
||
for i in range(start, len(text)):
|
||
ch = text[i]
|
||
if in_string:
|
||
if escape:
|
||
escape = False
|
||
elif ch == "\\":
|
||
escape = True
|
||
elif ch == '"':
|
||
in_string = False
|
||
continue
|
||
if ch == '"':
|
||
in_string = True
|
||
elif ch == open_ch:
|
||
depth += 1
|
||
elif ch == close_ch:
|
||
depth -= 1
|
||
if depth == 0:
|
||
blob = text[start : i + 1]
|
||
try:
|
||
return json.loads(blob)
|
||
except (ValueError, TypeError):
|
||
return None
|
||
return None
|
||
|
||
|
||
def extract_embedded_json(text: str | None) -> Any | None:
|
||
"""尽力从一段文本里抽出内嵌的 JSON(对象或数组)。
|
||
|
||
覆盖三种用户输入:纯 JSON(直接解析)、含 JSON 的文本(抠出第一个平衡的
|
||
``{…}`` 或 ``[…]``)、纯文本(返回 ``None``)。也识别 ```json fence。
|
||
"""
|
||
if not text or not isinstance(text, str):
|
||
return None
|
||
s = text.strip()
|
||
fence = _FENCE_RE.search(s)
|
||
if fence:
|
||
inner = fence.group(1).strip()
|
||
try:
|
||
return json.loads(inner)
|
||
except (ValueError, TypeError):
|
||
s = inner # fence 内不是完整 JSON,继续往下扫
|
||
try:
|
||
return json.loads(s)
|
||
except (ValueError, TypeError):
|
||
pass
|
||
# 取最靠前出现的那种括号块(对象优先于数组,谁先出现谁优先)。
|
||
brace = _scan_balanced(s, "{", "}")
|
||
bracket = _scan_balanced(s, "[", "]")
|
||
if brace is not None and bracket is not None:
|
||
return brace if s.find("{") <= s.find("[") else bracket
|
||
return brace if brace is not None else bracket
|
||
|
||
# 模板资源根目录(随 git 发布,与本模块同级 ``templates/`` 子目录)。
|
||
_TEMPLATES_ROOT = Path(__file__).parent / "templates"
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class TemplateSpec:
|
||
"""一个 HTML 模板的注册信息。
|
||
|
||
``file`` 相对 :data:`_TEMPLATES_ROOT`;``data_var``/``manifest_var`` 是模板里
|
||
待替换的 JS 常量名(``manifest_var`` 为空表示该模板没有 MANIFEST 行)。
|
||
|
||
``convertible=False`` 表示该模板暂只「直接展示已有数据」——执行时原样输出模板
|
||
HTML,不做 JSON 转换/校验/填充(无 ``const DATA=`` 注入行的模板用它,后续接入
|
||
转换后改回 True)。
|
||
"""
|
||
|
||
id: str
|
||
name: str
|
||
description: str
|
||
file: str
|
||
data_var: str = "DATA"
|
||
manifest_var: str = ""
|
||
convertible: bool = True
|
||
|
||
|
||
# 已注册模板。新增 no.2/no.3 时在此追加一条即可——前端下拉、调度器填充全自动生效。
|
||
TEMPLATES: tuple[TemplateSpec, ...] = (
|
||
TemplateSpec(
|
||
id="no1",
|
||
name="no.1",
|
||
description="",
|
||
file="no1/taiwan_littoral_ui_redwhite_v2.html",
|
||
data_var="DATA",
|
||
manifest_var="MANIFEST",
|
||
),
|
||
TemplateSpec(
|
||
id="no2",
|
||
name="no.2",
|
||
description="",
|
||
file="no2/taiwan_air_force_v3_2.html",
|
||
# 暂只展示已有数据,不做 JSON 转换/填充(后续接入转换后再开)。
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no3",
|
||
name="no.3",
|
||
description="",
|
||
file="no3/no3.html",
|
||
# 无 const DATA= 注入点,暂只展示已有数据(后续接入转换后再开)。
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no4",
|
||
name="no.4",
|
||
description="",
|
||
file="no4/no4.html",
|
||
# 数据存于 <script id="app-data" type="application/json"> 元素(非 const DATA= 行),
|
||
# 当前引擎只支持 const VAR= 行注入,故暂注册为只展示;接转换需扩展元素注入。
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no5",
|
||
name="no.5",
|
||
description="",
|
||
file="no5/no5.html",
|
||
# 数据用 `let DATA = {…}`(let + 空格,非 const DATA= 行),当前引擎不匹配,
|
||
# 故暂注册为只展示;接转换需扩展行注入正则。
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no6",
|
||
name="no.6",
|
||
description="",
|
||
file="no6/no6.html",
|
||
# 数据用 `const DATA = {…}`(= 前有空格,当前引擎正则要求 const DATA=),不匹配,
|
||
# 故暂注册为只展示;接转换需放宽行注入正则。
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no7",
|
||
name="no.7",
|
||
description="",
|
||
file="no7/no7.html",
|
||
# 数据存于 <script id="PAGE_DATA" type="application/json"> 元素(同 no.4 类元素注入),
|
||
# 当前引擎只支持 const VAR= 行注入,故暂注册为只展示;接转换需扩展元素注入。
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
# 显示 no.8;id 用 no8b(no8 已被显示名 no.12 的模板占用)。
|
||
id="no8b",
|
||
name="no.8",
|
||
description="",
|
||
file="no8b/no8b.html",
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
# 显示 no.9;id 用 no9b(no9 已被显示名 no.13 的模板占用)。
|
||
id="no9b",
|
||
name="no.9",
|
||
description="",
|
||
file="no9b/no9b.html",
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no10",
|
||
name="no.10",
|
||
description="",
|
||
file="no10/no10.html",
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no11",
|
||
name="no.11",
|
||
description="",
|
||
file="no11/no11.html",
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no8",
|
||
name="no.12",
|
||
description="",
|
||
file="no8/no8.html",
|
||
# 数据存于 <script id="DATA" type="application/json"> 元素(同 no.4/no.7 元素注入),
|
||
# 当前引擎只支持 const VAR= 行注入,故暂注册为只展示;接转换需扩展元素注入。
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no9",
|
||
name="no.13",
|
||
description="",
|
||
file="no9/no9.html",
|
||
# 数据用 `const DATA = {…}`(= 前有空格,同 no.6),当前引擎正则要求 const DATA=,不匹配,
|
||
# 故暂注册为只展示;接转换需放宽行注入正则。
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no14",
|
||
name="no.14",
|
||
description="",
|
||
file="no14/no14.html",
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no15",
|
||
name="no.15",
|
||
description="",
|
||
file="no15/no15.html",
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no16",
|
||
name="no.16",
|
||
description="",
|
||
file="no16/no16.html",
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no17",
|
||
name="no.17",
|
||
description="",
|
||
file="no17/no17.html",
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no18",
|
||
name="no.18",
|
||
description="",
|
||
file="no18/no18.html",
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no19",
|
||
name="no.19",
|
||
description="",
|
||
file="no19/no.19.html",
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no20",
|
||
name="no.20",
|
||
description="",
|
||
file="no20/no.20.html",
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no21",
|
||
name="no.21",
|
||
description="",
|
||
file="no21/no.21.html",
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no22",
|
||
name="no.22",
|
||
description="",
|
||
file="no22/no.22.html",
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no23",
|
||
name="no.23",
|
||
description="",
|
||
file="no23/no.23.html",
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no24",
|
||
name="no.24",
|
||
description="",
|
||
file="no24/no.24.html",
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no25",
|
||
name="no.25",
|
||
description="",
|
||
file="no25/no.25.html",
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no26",
|
||
name="no.26",
|
||
description="",
|
||
file="no26/no.26.html",
|
||
convertible=False,
|
||
),
|
||
TemplateSpec(
|
||
id="no27",
|
||
name="no.27",
|
||
description="",
|
||
file="no27/no.27.html",
|
||
convertible=False,
|
||
),
|
||
)
|
||
|
||
_TEMPLATES_BY_ID: dict[str, TemplateSpec] = {t.id: t for t in TEMPLATES}
|
||
|
||
|
||
class TemplateError(Exception):
|
||
"""模板缺失 / 解析失败 / 填充失败。"""
|
||
|
||
|
||
def list_templates() -> list[dict[str, Any]]:
|
||
"""前端下拉用的轻量清单(只含 id/name/description/convertible,不含庞大数据)。"""
|
||
return [
|
||
{"id": t.id, "name": t.name, "description": t.description, "convertible": t.convertible}
|
||
for t in TEMPLATES
|
||
]
|
||
|
||
|
||
def get_template(template_id: str) -> TemplateSpec | None:
|
||
return _TEMPLATES_BY_ID.get(template_id)
|
||
|
||
|
||
def is_convertible(template_id: str) -> bool:
|
||
spec = _TEMPLATES_BY_ID.get(template_id)
|
||
return bool(spec and spec.convertible)
|
||
|
||
|
||
def _require(template_id: str) -> TemplateSpec:
|
||
spec = _TEMPLATES_BY_ID.get(template_id)
|
||
if spec is None:
|
||
raise TemplateError(f"未知模板 id:{template_id!r}")
|
||
return spec
|
||
|
||
|
||
def _template_path(spec: TemplateSpec) -> Path:
|
||
return _TEMPLATES_ROOT / spec.file
|
||
|
||
|
||
def read_template_text(template_id: str) -> str:
|
||
spec = _require(template_id)
|
||
path = _template_path(spec)
|
||
if not path.is_file():
|
||
raise TemplateError(f"模板文件缺失:{path}")
|
||
return path.read_text(encoding="utf-8")
|
||
|
||
|
||
def _const_line_re(var: str) -> re.Pattern[str]:
|
||
# 单行 ``const VAR={...};`` —— 默认点号不跨行,故 ``.*`` 恰好吃到该物理行的结尾分号。
|
||
return re.compile(r"^const " + re.escape(var) + r"=.*;[ \t]*$", re.MULTILINE)
|
||
|
||
|
||
def _extract_const_object(text: str, var: str) -> Any:
|
||
"""解析模板里 ``const VAR={...};`` 那一行的 JSON 值。"""
|
||
match = _const_line_re(var).search(text)
|
||
if match is None:
|
||
raise TemplateError(f"模板中未找到 const {var}=… 行")
|
||
line = match.group(0).strip()
|
||
body = line[len("const ") + len(var) + 1 :] # 去掉 ``const VAR=``
|
||
body = body.rstrip().rstrip(";").rstrip()
|
||
try:
|
||
return json.loads(body)
|
||
except json.JSONDecodeError as exc: # pragma: no cover - 模板损坏才会触发
|
||
raise TemplateError(f"解析 const {var} 失败:{exc}") from exc
|
||
|
||
|
||
@lru_cache(maxsize=16)
|
||
def reference_data(template_id: str) -> dict[str, Any]:
|
||
"""模板内嵌的参考 ``DATA`` 对象(目标格式范例 + 校验基准),按模板缓存。"""
|
||
spec = _require(template_id)
|
||
text = read_template_text(template_id)
|
||
obj = _extract_const_object(text, spec.data_var)
|
||
if not isinstance(obj, dict):
|
||
raise TemplateError(f"模板 {template_id} 的 {spec.data_var} 不是对象")
|
||
return obj
|
||
|
||
|
||
def _skeleton(value: Any, *, list_sample: int = 1, depth: int = 0, max_depth: int = 6) -> Any:
|
||
"""把一份具体数据压成「空占位骨架」:保留全部字段名与类型,值一律抹空。
|
||
|
||
模板只是样式外壳,**示例值(模板自带的模拟内容)绝不能进入转换 prompt**——
|
||
否则大模型会照抄示例内容、或被其领域(如台海防卫)限制住,不肯把用户主题的
|
||
内容映射进来。叶子按类型给空占位:字符串→\"\"、数字→0、布尔→false、
|
||
其它→null;数组元素同构,留 1 个抹空样例即可暴露该元素的完整字段集。
|
||
"""
|
||
if depth >= max_depth:
|
||
return None
|
||
if isinstance(value, dict):
|
||
return {k: _skeleton(v, list_sample=list_sample, depth=depth + 1, max_depth=max_depth) for k, v in value.items()}
|
||
if isinstance(value, list):
|
||
if not value:
|
||
return []
|
||
return [_skeleton(v, list_sample=list_sample, depth=depth + 1, max_depth=max_depth) for v in value[:list_sample]]
|
||
if isinstance(value, bool):
|
||
return False
|
||
if isinstance(value, (int, float)):
|
||
return 0
|
||
if isinstance(value, str):
|
||
return ""
|
||
return None
|
||
|
||
|
||
@lru_cache(maxsize=16)
|
||
def schema_skeleton_json(template_id: str) -> str:
|
||
"""目标格式的空占位骨架(JSON 字符串),注入到大模型的转换 prompt 里。"""
|
||
skeleton = _skeleton(reference_data(template_id))
|
||
return json.dumps(skeleton, ensure_ascii=False, indent=1)
|
||
|
||
|
||
def top_level_keys(template_id: str) -> list[str]:
|
||
return list(reference_data(template_id).keys())
|
||
|
||
|
||
def validate_data(template_id: str, obj: Any) -> list[str]:
|
||
"""硬校验:转换结果必须是对象,且顶层键齐全、容器类型与模板一致。
|
||
|
||
返回错误列表(空 = 通过)。只校验顶层结构——字段级缺失通过
|
||
:func:`field_coverage_gaps` 作为软反馈回喂大模型,而不在这里硬卡死。
|
||
"""
|
||
errors: list[str] = []
|
||
if not isinstance(obj, dict):
|
||
return ["转换结果不是 JSON 对象(应为 {…})"]
|
||
ref = reference_data(template_id)
|
||
for key, ref_val in ref.items():
|
||
if key not in obj:
|
||
errors.append(f"缺少顶层字段「{key}」")
|
||
continue
|
||
got = obj[key]
|
||
if isinstance(ref_val, list) and not isinstance(got, list):
|
||
errors.append(f"字段「{key}」应为数组")
|
||
elif isinstance(ref_val, dict) and not isinstance(got, dict):
|
||
errors.append(f"字段「{key}」应为对象")
|
||
return errors
|
||
|
||
|
||
def _element_fields(value: Any) -> set[str]:
|
||
"""取一个数组里所有 dict 元素字段名的并集。"""
|
||
fields: set[str] = set()
|
||
if isinstance(value, list):
|
||
for item in value:
|
||
if isinstance(item, dict):
|
||
fields.update(item.keys())
|
||
return fields
|
||
|
||
|
||
def field_coverage_gaps(template_id: str, obj: Any, *, max_items: int = 40) -> list[str]:
|
||
"""软反馈:对照参考数据,列出转换结果里缺失的字段(数组元素字段 / 对象子键)。
|
||
|
||
用于把「字段没填全」回喂大模型让它补齐,不作为硬性失败。
|
||
"""
|
||
gaps: list[str] = []
|
||
if not isinstance(obj, dict):
|
||
return gaps
|
||
ref = reference_data(template_id)
|
||
for key, ref_val in ref.items():
|
||
if key not in obj:
|
||
continue
|
||
got = obj[key]
|
||
if isinstance(ref_val, dict) and isinstance(got, dict):
|
||
for sub in ref_val:
|
||
if sub not in got:
|
||
gaps.append(f"{key}.{sub} 缺失")
|
||
elif isinstance(ref_val, list) and isinstance(got, list):
|
||
want = _element_fields(ref_val)
|
||
have = _element_fields(got)
|
||
for f in sorted(want - have):
|
||
gaps.append(f"{key}[].{f} 缺失")
|
||
if len(gaps) >= max_items:
|
||
return gaps[:max_items] + ["…(更多字段省略)"]
|
||
return gaps
|
||
|
||
|
||
def _has_content(value: Any) -> bool:
|
||
"""该值是否含「有效内容」(用于判断子页面 / 字段是否抽到东西)。
|
||
|
||
空 / None / 空串 / 全空的列表或字典 视为无内容;数字(含 0)、布尔、非空字符串
|
||
视为有内容。
|
||
"""
|
||
if value is None:
|
||
return False
|
||
if isinstance(value, str):
|
||
return bool(value.strip())
|
||
if isinstance(value, (list, tuple)):
|
||
return any(_has_content(v) for v in value)
|
||
if isinstance(value, dict):
|
||
return any(_has_content(v) for v in value.values())
|
||
return True # 数字 / 布尔 / 其它标量
|
||
|
||
|
||
# 元素字段中至少要有这个比例能在模板参考数据的元素字段里找到,否则视为「自造字段」。
|
||
_LIST_FIELD_OVERLAP_MIN = 0.5
|
||
# 元素字段至少要覆盖模板参考元素字段的这个比例:低于它意味着大部分展示字段缺失
|
||
# (卡片大面积空白/undefined),不如整支回退模板原数据。
|
||
_LIST_FIELD_RECALL_MIN = 0.4
|
||
|
||
|
||
def _list_misaligned(got: Any, ref_val: Any) -> bool:
|
||
"""非空列表与模板参考列表是否「无法有效渲染」。
|
||
|
||
顶层校验挡不住这种情况:转换结果键齐全、类型也对,但列表元素的**字段**不对,
|
||
填进模板后 JS 读不到字段,页面满屏 ``undefined``。判定(满足其一即错位):
|
||
|
||
* 模板元素是对象而转换给了非对象(字符串/数字读不出属性);
|
||
* 元素字段里能对上模板的比例过低(:data:`_LIST_FIELD_OVERLAP_MIN`,
|
||
自造字段名占主导);
|
||
* 元素字段对模板参考字段的**覆盖率**过低(:data:`_LIST_FIELD_RECALL_MIN`,
|
||
只填了零星几个字段,展示字段大面积缺失)。
|
||
|
||
空列表不算错位(交由「无内容回退」处理)。
|
||
"""
|
||
if not isinstance(got, list) or not got:
|
||
return False
|
||
ref_fields = _element_fields(ref_val)
|
||
if not ref_fields:
|
||
return False # 模板元素本身没有对象字段,无从比对
|
||
ref_dicts = [item for item in ref_val if isinstance(item, dict)]
|
||
got_dicts = [item for item in got if isinstance(item, dict)]
|
||
if ref_dicts and not got_dicts:
|
||
return True # 模板元素是对象,转换给了字符串/数字 → 必然读不到字段
|
||
if not got_dicts:
|
||
return False
|
||
got_fields: set[str] = set()
|
||
for item in got_dicts:
|
||
got_fields.update(item.keys())
|
||
if not got_fields:
|
||
return False
|
||
overlap = got_fields & ref_fields
|
||
if len(overlap) / len(got_fields) < _LIST_FIELD_OVERLAP_MIN:
|
||
return True # 自造字段名占主导
|
||
return len(overlap) / len(ref_fields) < _LIST_FIELD_RECALL_MIN # 覆盖率过低
|
||
|
||
|
||
def _mismatch_reason(got: Any, ref_val: Any) -> str:
|
||
"""错位原因的人话描述(回喂重试用),与 :func:`_list_misaligned` 同判定。"""
|
||
ref_fields = _element_fields(ref_val)
|
||
got_fields: set[str] = set()
|
||
for item in got or []:
|
||
if isinstance(item, dict):
|
||
got_fields.update(item.keys())
|
||
overlap = got_fields & ref_fields
|
||
if got_fields and len(overlap) / len(got_fields) < _LIST_FIELD_OVERLAP_MIN:
|
||
return "的元素字段名与模板不一致(疑似自造字段名,必须原样沿用目标结构的字段名)"
|
||
return "的元素字段覆盖过低(样例元素的字段大部分缺失;缺数据的字段按类型留空补齐,不要省略字段)"
|
||
|
||
|
||
def shape_mismatches(template_id: str, obj: Any, *, max_items: int = 20) -> list[str]:
|
||
"""硬反馈:列出非空列表分支里「无法有效渲染」的项(供回喂重试)。
|
||
|
||
与 :func:`field_coverage_gaps`(软反馈,字段没填全)不同,本函数抓的是
|
||
**填了也渲染不出来**的情况:列表有内容但元素字段自造(对不上模板)或覆盖
|
||
过低(展示字段大面积缺失) → 渲染必然大面积 undefined,必须回喂重试。
|
||
"""
|
||
out: list[str] = []
|
||
if not isinstance(obj, dict):
|
||
return out
|
||
ref = reference_data(template_id)
|
||
for key, ref_val in ref.items():
|
||
got = obj.get(key)
|
||
if isinstance(ref_val, list) and _list_misaligned(got, ref_val):
|
||
out.append(f"「{key}」{_mismatch_reason(got, ref_val)}")
|
||
if len(out) >= max_items:
|
||
break
|
||
return out
|
||
|
||
|
||
def _ref_element_types(ref_val: list) -> dict[str, Any]:
|
||
"""参考列表各字段的类型样本(取第一个携带该字段的元素)。"""
|
||
types: dict[str, Any] = {}
|
||
for item in ref_val:
|
||
if isinstance(item, dict):
|
||
for key, value in item.items():
|
||
types.setdefault(key, value)
|
||
return types
|
||
|
||
|
||
def _blank_like(sample: Any) -> Any:
|
||
"""按参考类型给出「空值」:字符串→空串、列表→空数组、对象→空对象、其它→null。"""
|
||
if isinstance(sample, str):
|
||
return ""
|
||
if isinstance(sample, list):
|
||
return []
|
||
if isinstance(sample, dict):
|
||
return {}
|
||
return None
|
||
|
||
|
||
def _fill_missing_fields(got: list, ref_val: list) -> list:
|
||
"""给保留的转换列表逐元素补齐模板元素字段(缺失的按参考类型给空值)。
|
||
|
||
模板 JS 常裸读 ``${item.字段}``(不经过空值兜底函数),字段缺失会直接把
|
||
``undefined`` 打到页面上;补齐成空值后渲染为空白,不再出现字面 undefined。
|
||
只补缺、不改已有值,多出来的自造键原样保留(JS 不读,无害)。
|
||
"""
|
||
types = _ref_element_types(ref_val)
|
||
if not types:
|
||
return got
|
||
out: list[Any] = []
|
||
for item in got:
|
||
if isinstance(item, dict):
|
||
filled = dict(item)
|
||
for key, sample in types.items():
|
||
if key not in filled:
|
||
filled[key] = _blank_like(sample)
|
||
out.append(filled)
|
||
else:
|
||
out.append(item)
|
||
return out
|
||
|
||
|
||
def _merge_fill(got: Any, ref: Any) -> Any:
|
||
"""用 ``got`` 覆盖,但 ``got`` 里「没有有效内容」的部分回退 ``ref``。
|
||
|
||
字典逐键递归(空分支回退);列表有内容就整体保留 ``got``(不逐元素合并,避免与
|
||
原模板元素错配)——**例外**:列表元素无法有效渲染(:func:`_list_misaligned`,
|
||
自造字段名占主导/覆盖率过低/元素类型不符)时整支回退 ``ref``,否则填进模板
|
||
JS 读不到字段,页面满屏 undefined。保留的列表会逐元素补齐缺失字段的空值
|
||
(:func:`_fill_missing_fields`),防止局部缺字段裸读出 undefined。
|
||
"""
|
||
if not _has_content(got):
|
||
return ref
|
||
if isinstance(ref, dict) and isinstance(got, dict):
|
||
out = dict(got) # 保留 got 里 ref 没有的额外键
|
||
for key, ref_val in ref.items():
|
||
out[key] = _merge_fill(got.get(key), ref_val)
|
||
return out
|
||
if isinstance(ref, list) and isinstance(got, list):
|
||
if _list_misaligned(got, ref):
|
||
return ref
|
||
return _fill_missing_fields(got, ref)
|
||
return got
|
||
|
||
|
||
def merge_with_reference(template_id: str, data: Any) -> dict[str, Any]:
|
||
"""用转换结果覆盖,但「没抽到有效内容」的子页面 / 字段回退模板原有数据。
|
||
|
||
顶层键即子页面:某子页面(及其内部空字段)无有效内容时,用模板自带的原始数据补上,
|
||
保证页面与子页面不空白。整体为空时等价于返回模板默认数据。
|
||
"""
|
||
ref = reference_data(template_id)
|
||
merged = _merge_fill(data, ref)
|
||
return merged if isinstance(merged, dict) else dict(ref)
|
||
|
||
|
||
def _dump_const(var: str, obj: Any) -> str:
|
||
body = json.dumps(obj, ensure_ascii=False)
|
||
return f"const {var}={body};"
|
||
|
||
|
||
def _escape_html(text: str) -> str:
|
||
return (
|
||
text.replace("&", "&")
|
||
.replace("<", "<")
|
||
.replace(">", ">")
|
||
.replace('"', """)
|
||
)
|
||
|
||
|
||
# 转换结果里约定的两个额外顶层键:页面大标题/副标题(填充时替换页头,不进 DATA)。
|
||
PAGE_TITLE_KEY = "页面标题"
|
||
PAGE_SUBTITLE_KEY = "页面副标题"
|
||
|
||
|
||
def pop_page_titles(obj: dict[str, Any]) -> tuple[str | None, str | None]:
|
||
"""取出并移除转换结果里的「页面标题/页面副标题」额外键。
|
||
|
||
模型按转换 prompt 的约定额外输出这两个顶层字符串键;它们不是模板 DATA 的
|
||
字段(校验本就忽略额外键),填充前摘出来用于替换页头,避免漏进 ``const DATA``。
|
||
"""
|
||
def _clean(value: Any) -> str | None:
|
||
if isinstance(value, str) and value.strip():
|
||
return value.strip()
|
||
return None
|
||
|
||
title = _clean(obj.pop(PAGE_TITLE_KEY, None))
|
||
subtitle = _clean(obj.pop(PAGE_SUBTITLE_KEY, None))
|
||
return title, subtitle
|
||
|
||
|
||
def fill_template(
|
||
template_id: str,
|
||
data_obj: Any,
|
||
*,
|
||
manifest_overrides: dict[str, Any] | None = None,
|
||
update_date: date | None = None,
|
||
title: str | None = None,
|
||
subtitle: str | None = None,
|
||
) -> str:
|
||
"""把 ``data_obj`` 填进模板的 ``const DATA=…`` 行,产出完整 HTML 文档。
|
||
|
||
可选地更新 MANIFEST 的「系统.更新时间」(``update_date``,默认不动) 以及任意
|
||
``manifest_overrides``(深合并到解析后的 MANIFEST,如改系统名称)。
|
||
|
||
``title``/``subtitle`` 来自转换结果里的「页面标题/页面副标题」:模板页头
|
||
(<title>、第一个 <h1>、其后紧跟的副标题 <p>)是写死的静态 HTML,不随 DATA
|
||
变化——给出新标题时替换这三处,并把 MANIFEST 的 系统.名称 一并同步;
|
||
不给则保持模板原文。
|
||
"""
|
||
spec = _require(template_id)
|
||
text = read_template_text(template_id)
|
||
|
||
if title:
|
||
overrides = dict(manifest_overrides or {})
|
||
system = overrides.get("系统")
|
||
system = dict(system) if isinstance(system, dict) else {}
|
||
system["名称"] = title
|
||
overrides["系统"] = system
|
||
manifest_overrides = overrides
|
||
escaped = _escape_html(title)
|
||
text = re.sub(r"<title>.*?</title>", lambda _m: f"<title>{escaped}</title>", text, count=1, flags=re.DOTALL)
|
||
text = re.sub(r"<h1>.*?</h1>", lambda _m: f"<h1>{escaped}</h1>", text, count=1, flags=re.DOTALL)
|
||
if subtitle:
|
||
escaped_sub = _escape_html(subtitle)
|
||
text = re.sub(
|
||
r"(<h1>.*?</h1>\s*<p>).*?(</p>)",
|
||
lambda m: f"{m.group(1)}{escaped_sub}{m.group(2)}",
|
||
text,
|
||
count=1,
|
||
flags=re.DOTALL,
|
||
)
|
||
|
||
data_re = _const_line_re(spec.data_var)
|
||
if data_re.search(text) is None:
|
||
raise TemplateError(f"模板 {template_id} 中未找到 const {spec.data_var}= 行")
|
||
new_data_line = _dump_const(spec.data_var, data_obj)
|
||
text = data_re.sub(lambda _m: new_data_line, text, count=1)
|
||
|
||
if spec.manifest_var and (manifest_overrides or update_date is not None):
|
||
manifest_re = _const_line_re(spec.manifest_var)
|
||
match = manifest_re.search(text)
|
||
if match is not None:
|
||
try:
|
||
manifest = _extract_const_object(text, spec.manifest_var)
|
||
except TemplateError:
|
||
manifest = None
|
||
if isinstance(manifest, dict):
|
||
if update_date is not None:
|
||
system = manifest.get("系统")
|
||
if isinstance(system, dict):
|
||
system["更新时间"] = update_date.isoformat()
|
||
if manifest_overrides:
|
||
_deep_merge(manifest, manifest_overrides)
|
||
new_manifest_line = _dump_const(spec.manifest_var, manifest)
|
||
text = manifest_re.sub(lambda _m: new_manifest_line, text, count=1)
|
||
return text
|
||
|
||
|
||
def _deep_merge(base: dict[str, Any], overrides: dict[str, Any]) -> None:
|
||
for k, v in overrides.items():
|
||
if isinstance(v, dict) and isinstance(base.get(k), dict):
|
||
_deep_merge(base[k], v)
|
||
else:
|
||
base[k] = v
|
||
|
||
|
||
# ── require_html 模式选「内置模板」作参考 ─────────────────────────────────────────
|
||
# 「参考收藏页面」下拉里模板选项的值存成 ``tpl:<id>``,与用户收藏(普通 id)区分。
|
||
# 选了模板 → 调度器用**真实模板**出页(保留全部子页面),不抽风格摘要(那会丢子页面)。
|
||
REFERENCE_PREFIX = "tpl:"
|
||
|
||
|
||
def reference_template_id(reference: str | None) -> str | None:
|
||
"""``tpl:no1`` → ``no1``;非模板参考(普通收藏 id / 空)返回 None。"""
|
||
if isinstance(reference, str) and reference.startswith(REFERENCE_PREFIX):
|
||
return reference[len(REFERENCE_PREFIX) :] or None
|
||
return None
|