454 lines
18 KiB
Python
454 lines
18 KiB
Python
"""HTML page generation helpers for ``html_page`` scheduled tasks.
|
||
|
||
Pure, dependency-light utilities the scheduler uses to turn a GLM5 response
|
||
into a safe, previewable HTML document:
|
||
|
||
- :func:`extract_json_object` — fault-tolerant JSON extraction from a model
|
||
reply that may be wrapped in prose or ```json fences.
|
||
- :func:`sanitize_html` — strip script-execution vectors (``<script>``,
|
||
``<iframe>``, ``on*`` handlers, ``javascript:`` URLs, …) using BeautifulSoup.
|
||
- :func:`validate_html` — cheap structural checks (doctype/html/head/body,
|
||
non-empty body, no residual scripts) that gate the optional repair pass.
|
||
- ``build_*_prompt`` — the GLM5 prompt templates (generation / repair / style
|
||
extraction).
|
||
- :func:`screenshot_html` — best-effort Playwright screenshot, a no-op when
|
||
Playwright is not installed (the offline image ships without a browser).
|
||
|
||
Kept free of any ``app.*`` import so it stays inside the harness boundary.
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import json
|
||
import logging
|
||
import re
|
||
from typing import Any
|
||
|
||
logger = logging.getLogger(__name__)
|
||
|
||
# Tags removed wholesale — they either execute code or load remote active
|
||
# content that the iframe sandbox alone shouldn't be relied on to contain.
|
||
_DANGEROUS_TAGS = frozenset(
|
||
{"script", "iframe", "object", "embed", "applet", "frame", "frameset", "base"}
|
||
)
|
||
|
||
# URL schemes that can execute script when navigated/loaded.
|
||
_DANGEROUS_URL_RE = re.compile(r"^\s*(javascript|vbscript|data:text/html)", re.IGNORECASE)
|
||
|
||
# Attributes carrying a URL we must scheme-check.
|
||
_URL_ATTRS = frozenset({"href", "src", "action", "formaction", "xlink:href", "background", "poster"})
|
||
|
||
_LOCAL_DOC_LINK_RE = re.compile(
|
||
r"^(?![a-z][a-z0-9+.-]*:)(?!//)(?P<path>[^#?]*\.(?:html?|md|markdown))(?P<query>\?[^#]*)?(?P<fragment>#.+)?$",
|
||
re.IGNORECASE,
|
||
)
|
||
|
||
|
||
def parse_frontmatter(text: str) -> tuple[dict[str, Any], str]:
|
||
"""Split a ``---`` YAML frontmatter block from the body of a SKILL.md.
|
||
|
||
Returns ``(frontmatter_dict, body)``. When there is no frontmatter the dict
|
||
is empty and the body is the whole text.
|
||
"""
|
||
if not text or not text.lstrip().startswith("---"):
|
||
return {}, text or ""
|
||
lines = text.splitlines()
|
||
# Skip leading blank lines before the opening fence.
|
||
start = 0
|
||
while start < len(lines) and not lines[start].strip():
|
||
start += 1
|
||
end = None
|
||
for i in range(start + 1, len(lines)):
|
||
if lines[i].strip() == "---":
|
||
end = i
|
||
break
|
||
if end is None:
|
||
return {}, text
|
||
fm_block = "\n".join(lines[start + 1 : end])
|
||
body = "\n".join(lines[end + 1 :])
|
||
try:
|
||
import yaml
|
||
|
||
data = yaml.safe_load(fm_block) or {}
|
||
if not isinstance(data, dict):
|
||
data = {}
|
||
except Exception:
|
||
data = {}
|
||
return data, body
|
||
|
||
|
||
def extract_json_object(text: str | None) -> dict[str, Any] | None:
|
||
"""Best-effort parse of a JSON object from a model reply.
|
||
|
||
Handles three common shapes: a bare JSON object, a ```json fenced block,
|
||
and a JSON object embedded in surrounding prose. Returns ``None`` when no
|
||
balanced object can be recovered.
|
||
"""
|
||
if not text or not isinstance(text, str):
|
||
return None
|
||
candidate = text.strip()
|
||
|
||
# Strip a single ```json ... ``` (or ``` ... ```) fence if present.
|
||
fence = re.search(r"```(?:json)?\s*(.*?)```", candidate, flags=re.DOTALL | re.IGNORECASE)
|
||
if fence:
|
||
candidate = fence.group(1).strip()
|
||
|
||
# Fast path: the whole thing is already valid JSON.
|
||
try:
|
||
parsed = json.loads(candidate)
|
||
return parsed if isinstance(parsed, dict) else None
|
||
except (ValueError, TypeError):
|
||
pass
|
||
|
||
# Fallback: scan for the first balanced ``{ ... }`` and parse that.
|
||
start = candidate.find("{")
|
||
if start == -1:
|
||
return None
|
||
depth = 0
|
||
in_string = False
|
||
escape = False
|
||
for index in range(start, len(candidate)):
|
||
char = candidate[index]
|
||
if in_string:
|
||
if escape:
|
||
escape = False
|
||
elif char == "\\":
|
||
escape = True
|
||
elif char == '"':
|
||
in_string = False
|
||
continue
|
||
if char == '"':
|
||
in_string = True
|
||
elif char == "{":
|
||
depth += 1
|
||
elif char == "}":
|
||
depth -= 1
|
||
if depth == 0:
|
||
blob = candidate[start : index + 1]
|
||
try:
|
||
parsed = json.loads(blob)
|
||
except (ValueError, TypeError):
|
||
return None
|
||
return parsed if isinstance(parsed, dict) else None
|
||
return None
|
||
|
||
|
||
def recover_html_document(text: str | None) -> str | None:
|
||
"""Slice a complete HTML document out of a model reply.
|
||
|
||
Used as a fallback when the JSON wrapper around the ``html`` field can't be
|
||
parsed — the model commonly emits *literal* newlines and *unescaped* quotes
|
||
(``"zh-CN"``) inside the value, which is invalid JSON, so
|
||
:func:`extract_json_object` returns ``None``. Without this, the caller would
|
||
archive the whole ``{"title": ..., "html": "<!doctype html>…"}`` blob as the
|
||
page. We locate the ``<!doctype html>`` / ``<html …>`` start and the final
|
||
``</html>`` and return just that span, dropping the surrounding JSON
|
||
scaffolding. Also accepts a reply that is already bare HTML. Returns
|
||
``None`` when no HTML document can be found.
|
||
"""
|
||
if not text or not isinstance(text, str):
|
||
return None
|
||
lowered = text.lower()
|
||
start = lowered.find("<!doctype html")
|
||
if start == -1:
|
||
start = lowered.find("<html")
|
||
if start == -1:
|
||
return None
|
||
end = lowered.rfind("</html>")
|
||
candidate = text[start:] if end == -1 else text[start : end + len("</html>")]
|
||
candidate = candidate.strip()
|
||
# If the html had been a properly-escaped JSON string value (the JSON failed
|
||
# to parse for some *other* reason), unescape the common sequences so they
|
||
# don't survive into the rendered page. Only triggers when escapes exist, so
|
||
# genuinely-literal HTML (the usual broken-JSON case) is left untouched.
|
||
if "\\n" in candidate or '\\"' in candidate or "\\/" in candidate or "\\t" in candidate:
|
||
candidate = (
|
||
candidate.replace('\\"', '"')
|
||
.replace("\\n", "\n")
|
||
.replace("\\r", "\r")
|
||
.replace("\\t", "\t")
|
||
.replace("\\/", "/")
|
||
)
|
||
return candidate or None
|
||
|
||
|
||
def sanitize_html(html: str) -> tuple[str, list[str]]:
|
||
"""Remove script-execution vectors from ``html``.
|
||
|
||
Returns ``(cleaned_html, removed)`` where ``removed`` is a human-readable
|
||
report of what was stripped (used for logging / repair feedback). The
|
||
cleaner is defensive but not a full HTML validator — pair it with an
|
||
``iframe sandbox`` (no ``allow-scripts``) on the preview side.
|
||
"""
|
||
if not html or not isinstance(html, str):
|
||
return "", []
|
||
try:
|
||
from bs4 import BeautifulSoup
|
||
except Exception: # pragma: no cover - bs4 is a declared dependency
|
||
logger.warning("BeautifulSoup unavailable; returning HTML unsanitized")
|
||
return html, []
|
||
|
||
removed: list[str] = []
|
||
soup = BeautifulSoup(html, "html.parser")
|
||
|
||
# 1) Drop dangerous tags entirely.
|
||
for tag in soup.find_all(True):
|
||
name = (tag.name or "").lower()
|
||
if name in _DANGEROUS_TAGS:
|
||
removed.append(f"<{name}>")
|
||
tag.decompose()
|
||
continue
|
||
# External stylesheet / preload links pull remote resources.
|
||
if name == "link":
|
||
href = str(tag.get("href") or "")
|
||
rel = " ".join(tag.get("rel") or []) if isinstance(tag.get("rel"), list) else str(tag.get("rel") or "")
|
||
if re.match(r"^\s*(https?:)?//", href) and "stylesheet" in rel.lower():
|
||
removed.append("<link stylesheet>")
|
||
tag.decompose()
|
||
continue
|
||
# meta refresh can redirect to arbitrary URLs.
|
||
if name == "meta" and str(tag.get("http-equiv") or "").lower() == "refresh":
|
||
removed.append("<meta refresh>")
|
||
tag.decompose()
|
||
continue
|
||
|
||
# 2) Strip event handlers and dangerous URL schemes from surviving tags.
|
||
for tag in soup.find_all(True):
|
||
for attr in list(tag.attrs.keys()):
|
||
lowered = attr.lower()
|
||
if lowered.startswith("on"):
|
||
removed.append(f"{lowered}=")
|
||
del tag.attrs[attr]
|
||
continue
|
||
if lowered in _URL_ATTRS:
|
||
value = tag.attrs[attr]
|
||
value_str = value if isinstance(value, str) else " ".join(value)
|
||
if _DANGEROUS_URL_RE.match(value_str or ""):
|
||
removed.append(f"{lowered}={value_str[:24]}")
|
||
del tag.attrs[attr]
|
||
|
||
return str(soup), removed
|
||
|
||
|
||
def constrain_same_page_anchor_links(html: str) -> tuple[str, list[str]]:
|
||
"""Keep generated HTML document jumps inside the current HTML file."""
|
||
if not html or not isinstance(html, str):
|
||
return "", []
|
||
try:
|
||
from bs4 import BeautifulSoup
|
||
except Exception: # pragma: no cover - bs4 is a declared dependency
|
||
logger.warning("BeautifulSoup unavailable; returning HTML links unconstrained")
|
||
return html, []
|
||
|
||
changed: list[str] = []
|
||
soup = BeautifulSoup(html, "html.parser")
|
||
for tag in soup.find_all("a"):
|
||
href = tag.get("href")
|
||
if not isinstance(href, str):
|
||
continue
|
||
value = href.strip()
|
||
if not value or value.startswith("#"):
|
||
continue
|
||
match = _LOCAL_DOC_LINK_RE.match(value)
|
||
if not match:
|
||
continue
|
||
fragment = match.group("fragment")
|
||
if fragment:
|
||
tag["href"] = fragment
|
||
changed.append(f"{value}->{fragment}")
|
||
else:
|
||
del tag.attrs["href"]
|
||
changed.append(f"{value}->removed")
|
||
return str(soup), changed
|
||
|
||
|
||
def validate_html(html: str) -> list[str]:
|
||
"""Return a list of structural problems with ``html`` (empty == valid).
|
||
|
||
A page only counts as a complete document when it has the full skeleton
|
||
(doctype/html/head/body), a non-empty ``<title>``, visible body content
|
||
(no white-screen), no residual scripts, and no external stylesheet links.
|
||
"""
|
||
errors: list[str] = []
|
||
if not html or not isinstance(html, str):
|
||
return ["empty html"]
|
||
lowered = html.lower()
|
||
if "<!doctype html" not in lowered:
|
||
errors.append("缺少 <!doctype html> 声明")
|
||
if "<html" not in lowered:
|
||
errors.append("缺少 <html> 根元素")
|
||
if "<head" not in lowered:
|
||
errors.append("缺少 <head> 元素")
|
||
if "<body" not in lowered:
|
||
errors.append("缺少 <body> 元素")
|
||
if "<script" in lowered:
|
||
errors.append("仍包含 <script> 标签(不允许脚本)")
|
||
|
||
try:
|
||
from bs4 import BeautifulSoup
|
||
|
||
soup = BeautifulSoup(html, "html.parser")
|
||
# Title must exist and be non-empty.
|
||
title_el = soup.find("title")
|
||
if title_el is None or not title_el.get_text(strip=True):
|
||
errors.append("缺少非空 <title> 标题")
|
||
# External stylesheet links pull remote resources — not allowed offline.
|
||
for link in soup.find_all("link"):
|
||
href = str(link.get("href") or "")
|
||
rel = " ".join(link.get("rel") or []) if isinstance(link.get("rel"), list) else str(link.get("rel") or "")
|
||
if re.match(r"^\s*(https?:)?//", href) and "stylesheet" in rel.lower():
|
||
errors.append("包含外链样式表(不允许外部 CDN/资源)")
|
||
break
|
||
# Body must carry some visible content — guards against white-screen.
|
||
body = soup.body or soup
|
||
text = body.get_text(strip=True) if body else ""
|
||
has_media = bool(body.find(["img", "svg", "canvas", "table", "ul", "ol"])) if body else False
|
||
if len(text) < 5 and not has_media:
|
||
errors.append("页面正文为空,可能白屏")
|
||
except Exception: # pragma: no cover
|
||
pass
|
||
return errors
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Prompt templates
|
||
# ---------------------------------------------------------------------------
|
||
|
||
_GENERATION_TEMPLATE = """你是一个内网页面生成助手。请根据输入数据生成一个完整的 HTML 页面。
|
||
|
||
用户页面要求:
|
||
{page_prompt}
|
||
{constraint_block}本次数据:
|
||
{data}
|
||
|
||
参考页面风格:
|
||
{reference_style}
|
||
|
||
参考页面布局:
|
||
{reference_layout}
|
||
{skill_instructions}
|
||
要求:
|
||
1. 只输出 JSON,不要输出 Markdown,不要输出多余解释。
|
||
2. JSON 字段必须包含 title、html、style_summary、layout_summary。
|
||
3. html 字段的值必须是合法 JSON 字符串:其中所有换行必须写成 \\n、所有双引号必须转义为 \\",不要在字符串里直接换行或使用未转义的引号。
|
||
4. html 必须是完整 HTML 文档,包含 <!doctype html>、html、head、body。
|
||
5. 所有 CSS 写在 <style> 标签内,不要引用外部 CDN 或外链样式表。
|
||
6. 不要包含 <script>、外链脚本、iframe、表单提交、on* 事件属性。
|
||
7. 参考页面只用于风格和布局,页面内容必须来自“本次数据”,不要复用参考页旧数据。
|
||
8. 页面使用中文,排版适合在浏览器中直接预览。
|
||
9. HTML 内的目录/锚点跳转必须只跳转到当前这个 HTML 文件内部:只使用 href="#section-id",不要使用 href="index.html#section-id"、".md#..."、其它文件路径或任何跨文件锚点;目标区块必须有匹配的 id。
|
||
"""
|
||
|
||
_REPAIR_TEMPLATE = """你是一个 HTML 修复助手。下面的 HTML 存在问题,请修复后重新输出完整页面。
|
||
|
||
原始页面要求:
|
||
{requirements}
|
||
|
||
发现的问题:
|
||
{errors}
|
||
|
||
待修复 HTML:
|
||
{html}
|
||
|
||
要求:
|
||
1. 只输出 JSON,字段包含 fixed_html、changes(修复说明数组)。
|
||
2. fixed_html 必须是完整 HTML 文档,包含 <!doctype html>、html、head、body。
|
||
3. 移除所有 <script>、iframe、外链脚本、on* 事件属性、表单提交。
|
||
4. 保留原有内容与数据,只修复结构与安全问题。
|
||
5. HTML 内的目录/锚点跳转必须只跳转到当前这个 HTML 文件内部:只使用 href="#section-id",不要链接到 .html/.md 文件作为锚点。
|
||
"""
|
||
|
||
_STYLE_EXTRACT_TEMPLATE = """你是一个页面风格分析助手。请分析下面的 HTML 页面,提取可复用的风格与布局特征。
|
||
|
||
HTML 页面:
|
||
{html}
|
||
|
||
要求:
|
||
1. 只输出 JSON,字段包含 style_summary、layout_summary、content_structure_summary、reuse_prompt。
|
||
2. style_summary:颜色、字体、间距、圆角、阴影、视觉密度等视觉风格。
|
||
3. layout_summary:页面分区、卡片、表格、图表、导航等布局结构。
|
||
4. content_structure_summary:信息组织顺序与层级。
|
||
5. reuse_prompt:一段可直接拼接到下次生成 prompt 的中文风格说明,不要包含本页的具体数据。
|
||
"""
|
||
|
||
|
||
def _clip(value: Any, limit: int) -> str:
|
||
text = value if isinstance(value, str) else json.dumps(value, ensure_ascii=False, default=str)
|
||
text = text or ""
|
||
return text if len(text) <= limit else text[:limit] + "\n…(内容过长已截断)"
|
||
|
||
|
||
def build_generation_prompt(
|
||
*,
|
||
page_prompt: str,
|
||
negative_prompt: str | None,
|
||
data: Any,
|
||
reference_style: str | None,
|
||
reference_layout: str | None,
|
||
skill_instructions: str | None = None,
|
||
data_limit: int = 16000,
|
||
skill_limit: int = 30000,
|
||
) -> str:
|
||
skill_block = ""
|
||
if skill_instructions and skill_instructions.strip():
|
||
skill_block = "\n可用页面生成技能规范(请严格遵循):\n" + _clip(skill_instructions.strip(), skill_limit) + "\n"
|
||
# The constraint block is only emitted when the user actually supplied one,
|
||
# and is framed as a top-priority hard rule that overrides the page
|
||
# requirements / reference page / data on conflict — otherwise the model
|
||
# tends to treat "限制条件" as a soft hint and ignore it.
|
||
constraint_block = ""
|
||
neg = (negative_prompt or "").strip()
|
||
if neg:
|
||
constraint_block = (
|
||
"\n限制条件(最高优先级的硬性约束,必须逐条严格遵守;"
|
||
"若与上面的页面要求、参考页面或本次数据发生冲突,一律以这里的限制条件为准):\n"
|
||
+ neg
|
||
+ "\n\n"
|
||
)
|
||
return _GENERATION_TEMPLATE.format(
|
||
page_prompt=(page_prompt or "(未提供,使用通用看板风格)").strip(),
|
||
constraint_block=constraint_block,
|
||
data=_clip(data, data_limit),
|
||
reference_style=(reference_style or "无").strip(),
|
||
reference_layout=(reference_layout or "无").strip(),
|
||
skill_instructions=skill_block,
|
||
)
|
||
|
||
|
||
def build_repair_prompt(*, html: str, errors: list[str], requirements: str | None) -> str:
|
||
return _REPAIR_TEMPLATE.format(
|
||
requirements=(requirements or "无").strip(),
|
||
errors="\n".join(f"- {e}" for e in errors) or "- 未知问题",
|
||
html=_clip(html, 32000),
|
||
)
|
||
|
||
|
||
def build_style_extract_prompt(*, html: str) -> str:
|
||
return _STYLE_EXTRACT_TEMPLATE.format(html=_clip(html, 32000))
|
||
|
||
|
||
async def screenshot_html(html: str, output_path) -> str | None:
|
||
"""Best-effort PNG screenshot of ``html`` for white-screen / console QA.
|
||
|
||
Returns the screenshot path on success, or ``None`` when Playwright (or a
|
||
usable browser) is unavailable — the offline image ships without one, so
|
||
this must degrade silently rather than fail the run.
|
||
"""
|
||
try:
|
||
from playwright.async_api import async_playwright
|
||
except Exception:
|
||
return None
|
||
try:
|
||
async with async_playwright() as pw:
|
||
browser = await pw.chromium.launch()
|
||
try:
|
||
page = await browser.new_page()
|
||
await page.set_content(html, wait_until="networkidle")
|
||
await page.screenshot(path=str(output_path), full_page=True)
|
||
finally:
|
||
await browser.close()
|
||
return str(output_path)
|
||
except Exception:
|
||
logger.debug("Playwright screenshot failed; skipping QA screenshot", exc_info=True)
|
||
return None
|