deerflow-code/offline-backend-20260512/backend/packages/harness/deerflow/runtime/scheduler/html_page.py
2026-09-07 18:24:55 +08:00

454 lines
18 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""HTML page generation helpers for ``html_page`` scheduled tasks.
Pure, dependency-light utilities the scheduler uses to turn a GLM5 response
into a safe, previewable HTML document:
- :func:`extract_json_object` — fault-tolerant JSON extraction from a model
reply that may be wrapped in prose or ```json fences.
- :func:`sanitize_html` — strip script-execution vectors (``<script>``,
``<iframe>``, ``on*`` handlers, ``javascript:`` URLs, …) using BeautifulSoup.
- :func:`validate_html` — cheap structural checks (doctype/html/head/body,
non-empty body, no residual scripts) that gate the optional repair pass.
- ``build_*_prompt`` — the GLM5 prompt templates (generation / repair / style
extraction).
- :func:`screenshot_html` — best-effort Playwright screenshot, a no-op when
Playwright is not installed (the offline image ships without a browser).
Kept free of any ``app.*`` import so it stays inside the harness boundary.
"""
from __future__ import annotations
import json
import logging
import re
from typing import Any
logger = logging.getLogger(__name__)
# Tags removed wholesale — they either execute code or load remote active
# content that the iframe sandbox alone shouldn't be relied on to contain.
_DANGEROUS_TAGS = frozenset(
{"script", "iframe", "object", "embed", "applet", "frame", "frameset", "base"}
)
# URL schemes that can execute script when navigated/loaded.
_DANGEROUS_URL_RE = re.compile(r"^\s*(javascript|vbscript|data:text/html)", re.IGNORECASE)
# Attributes carrying a URL we must scheme-check.
_URL_ATTRS = frozenset({"href", "src", "action", "formaction", "xlink:href", "background", "poster"})
_LOCAL_DOC_LINK_RE = re.compile(
r"^(?![a-z][a-z0-9+.-]*:)(?!//)(?P<path>[^#?]*\.(?:html?|md|markdown))(?P<query>\?[^#]*)?(?P<fragment>#.+)?$",
re.IGNORECASE,
)
def parse_frontmatter(text: str) -> tuple[dict[str, Any], str]:
"""Split a ``---`` YAML frontmatter block from the body of a SKILL.md.
Returns ``(frontmatter_dict, body)``. When there is no frontmatter the dict
is empty and the body is the whole text.
"""
if not text or not text.lstrip().startswith("---"):
return {}, text or ""
lines = text.splitlines()
# Skip leading blank lines before the opening fence.
start = 0
while start < len(lines) and not lines[start].strip():
start += 1
end = None
for i in range(start + 1, len(lines)):
if lines[i].strip() == "---":
end = i
break
if end is None:
return {}, text
fm_block = "\n".join(lines[start + 1 : end])
body = "\n".join(lines[end + 1 :])
try:
import yaml
data = yaml.safe_load(fm_block) or {}
if not isinstance(data, dict):
data = {}
except Exception:
data = {}
return data, body
def extract_json_object(text: str | None) -> dict[str, Any] | None:
"""Best-effort parse of a JSON object from a model reply.
Handles three common shapes: a bare JSON object, a ```json fenced block,
and a JSON object embedded in surrounding prose. Returns ``None`` when no
balanced object can be recovered.
"""
if not text or not isinstance(text, str):
return None
candidate = text.strip()
# Strip a single ```json ... ``` (or ``` ... ```) fence if present.
fence = re.search(r"```(?:json)?\s*(.*?)```", candidate, flags=re.DOTALL | re.IGNORECASE)
if fence:
candidate = fence.group(1).strip()
# Fast path: the whole thing is already valid JSON.
try:
parsed = json.loads(candidate)
return parsed if isinstance(parsed, dict) else None
except (ValueError, TypeError):
pass
# Fallback: scan for the first balanced ``{ ... }`` and parse that.
start = candidate.find("{")
if start == -1:
return None
depth = 0
in_string = False
escape = False
for index in range(start, len(candidate)):
char = candidate[index]
if in_string:
if escape:
escape = False
elif char == "\\":
escape = True
elif char == '"':
in_string = False
continue
if char == '"':
in_string = True
elif char == "{":
depth += 1
elif char == "}":
depth -= 1
if depth == 0:
blob = candidate[start : index + 1]
try:
parsed = json.loads(blob)
except (ValueError, TypeError):
return None
return parsed if isinstance(parsed, dict) else None
return None
def recover_html_document(text: str | None) -> str | None:
"""Slice a complete HTML document out of a model reply.
Used as a fallback when the JSON wrapper around the ``html`` field can't be
parsed — the model commonly emits *literal* newlines and *unescaped* quotes
(``"zh-CN"``) inside the value, which is invalid JSON, so
:func:`extract_json_object` returns ``None``. Without this, the caller would
archive the whole ``{"title": ..., "html": "<!doctype html>…"}`` blob as the
page. We locate the ``<!doctype html>`` / ``<html …>`` start and the final
``</html>`` and return just that span, dropping the surrounding JSON
scaffolding. Also accepts a reply that is already bare HTML. Returns
``None`` when no HTML document can be found.
"""
if not text or not isinstance(text, str):
return None
lowered = text.lower()
start = lowered.find("<!doctype html")
if start == -1:
start = lowered.find("<html")
if start == -1:
return None
end = lowered.rfind("</html>")
candidate = text[start:] if end == -1 else text[start : end + len("</html>")]
candidate = candidate.strip()
# If the html had been a properly-escaped JSON string value (the JSON failed
# to parse for some *other* reason), unescape the common sequences so they
# don't survive into the rendered page. Only triggers when escapes exist, so
# genuinely-literal HTML (the usual broken-JSON case) is left untouched.
if "\\n" in candidate or '\\"' in candidate or "\\/" in candidate or "\\t" in candidate:
candidate = (
candidate.replace('\\"', '"')
.replace("\\n", "\n")
.replace("\\r", "\r")
.replace("\\t", "\t")
.replace("\\/", "/")
)
return candidate or None
def sanitize_html(html: str) -> tuple[str, list[str]]:
"""Remove script-execution vectors from ``html``.
Returns ``(cleaned_html, removed)`` where ``removed`` is a human-readable
report of what was stripped (used for logging / repair feedback). The
cleaner is defensive but not a full HTML validator — pair it with an
``iframe sandbox`` (no ``allow-scripts``) on the preview side.
"""
if not html or not isinstance(html, str):
return "", []
try:
from bs4 import BeautifulSoup
except Exception: # pragma: no cover - bs4 is a declared dependency
logger.warning("BeautifulSoup unavailable; returning HTML unsanitized")
return html, []
removed: list[str] = []
soup = BeautifulSoup(html, "html.parser")
# 1) Drop dangerous tags entirely.
for tag in soup.find_all(True):
name = (tag.name or "").lower()
if name in _DANGEROUS_TAGS:
removed.append(f"<{name}>")
tag.decompose()
continue
# External stylesheet / preload links pull remote resources.
if name == "link":
href = str(tag.get("href") or "")
rel = " ".join(tag.get("rel") or []) if isinstance(tag.get("rel"), list) else str(tag.get("rel") or "")
if re.match(r"^\s*(https?:)?//", href) and "stylesheet" in rel.lower():
removed.append("<link stylesheet>")
tag.decompose()
continue
# meta refresh can redirect to arbitrary URLs.
if name == "meta" and str(tag.get("http-equiv") or "").lower() == "refresh":
removed.append("<meta refresh>")
tag.decompose()
continue
# 2) Strip event handlers and dangerous URL schemes from surviving tags.
for tag in soup.find_all(True):
for attr in list(tag.attrs.keys()):
lowered = attr.lower()
if lowered.startswith("on"):
removed.append(f"{lowered}=")
del tag.attrs[attr]
continue
if lowered in _URL_ATTRS:
value = tag.attrs[attr]
value_str = value if isinstance(value, str) else " ".join(value)
if _DANGEROUS_URL_RE.match(value_str or ""):
removed.append(f"{lowered}={value_str[:24]}")
del tag.attrs[attr]
return str(soup), removed
def constrain_same_page_anchor_links(html: str) -> tuple[str, list[str]]:
"""Keep generated HTML document jumps inside the current HTML file."""
if not html or not isinstance(html, str):
return "", []
try:
from bs4 import BeautifulSoup
except Exception: # pragma: no cover - bs4 is a declared dependency
logger.warning("BeautifulSoup unavailable; returning HTML links unconstrained")
return html, []
changed: list[str] = []
soup = BeautifulSoup(html, "html.parser")
for tag in soup.find_all("a"):
href = tag.get("href")
if not isinstance(href, str):
continue
value = href.strip()
if not value or value.startswith("#"):
continue
match = _LOCAL_DOC_LINK_RE.match(value)
if not match:
continue
fragment = match.group("fragment")
if fragment:
tag["href"] = fragment
changed.append(f"{value}->{fragment}")
else:
del tag.attrs["href"]
changed.append(f"{value}->removed")
return str(soup), changed
def validate_html(html: str) -> list[str]:
"""Return a list of structural problems with ``html`` (empty == valid).
A page only counts as a complete document when it has the full skeleton
(doctype/html/head/body), a non-empty ``<title>``, visible body content
(no white-screen), no residual scripts, and no external stylesheet links.
"""
errors: list[str] = []
if not html or not isinstance(html, str):
return ["empty html"]
lowered = html.lower()
if "<!doctype html" not in lowered:
errors.append("缺少 <!doctype html> 声明")
if "<html" not in lowered:
errors.append("缺少 <html> 根元素")
if "<head" not in lowered:
errors.append("缺少 <head> 元素")
if "<body" not in lowered:
errors.append("缺少 <body> 元素")
if "<script" in lowered:
errors.append("仍包含 <script> 标签(不允许脚本)")
try:
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, "html.parser")
# Title must exist and be non-empty.
title_el = soup.find("title")
if title_el is None or not title_el.get_text(strip=True):
errors.append("缺少非空 <title> 标题")
# External stylesheet links pull remote resources — not allowed offline.
for link in soup.find_all("link"):
href = str(link.get("href") or "")
rel = " ".join(link.get("rel") or []) if isinstance(link.get("rel"), list) else str(link.get("rel") or "")
if re.match(r"^\s*(https?:)?//", href) and "stylesheet" in rel.lower():
errors.append("包含外链样式表(不允许外部 CDN/资源)")
break
# Body must carry some visible content — guards against white-screen.
body = soup.body or soup
text = body.get_text(strip=True) if body else ""
has_media = bool(body.find(["img", "svg", "canvas", "table", "ul", "ol"])) if body else False
if len(text) < 5 and not has_media:
errors.append("页面正文为空,可能白屏")
except Exception: # pragma: no cover
pass
return errors
# ---------------------------------------------------------------------------
# Prompt templates
# ---------------------------------------------------------------------------
_GENERATION_TEMPLATE = """你是一个内网页面生成助手。请根据输入数据生成一个完整的 HTML 页面。
用户页面要求:
{page_prompt}
{constraint_block}本次数据:
{data}
参考页面风格:
{reference_style}
参考页面布局:
{reference_layout}
{skill_instructions}
要求:
1. 只输出 JSON,不要输出 Markdown,不要输出多余解释。
2. JSON 字段必须包含 title、html、style_summary、layout_summary。
3. html 字段的值必须是合法 JSON 字符串:其中所有换行必须写成 \\n、所有双引号必须转义为 \\",不要在字符串里直接换行或使用未转义的引号。
4. html 必须是完整 HTML 文档,包含 <!doctype html>、html、head、body。
5. 所有 CSS 写在 <style> 标签内,不要引用外部 CDN 或外链样式表。
6. 不要包含 <script>、外链脚本、iframe、表单提交、on* 事件属性。
7. 参考页面只用于风格和布局,页面内容必须来自“本次数据”,不要复用参考页旧数据。
8. 页面使用中文,排版适合在浏览器中直接预览。
9. HTML 内的目录/锚点跳转必须只跳转到当前这个 HTML 文件内部:只使用 href="#section-id",不要使用 href="index.html#section-id"、".md#..."、其它文件路径或任何跨文件锚点;目标区块必须有匹配的 id。
"""
_REPAIR_TEMPLATE = """你是一个 HTML 修复助手。下面的 HTML 存在问题,请修复后重新输出完整页面。
原始页面要求:
{requirements}
发现的问题:
{errors}
待修复 HTML:
{html}
要求:
1. 只输出 JSON,字段包含 fixed_html、changes(修复说明数组)。
2. fixed_html 必须是完整 HTML 文档,包含 <!doctype html>、html、head、body。
3. 移除所有 <script>、iframe、外链脚本、on* 事件属性、表单提交。
4. 保留原有内容与数据,只修复结构与安全问题。
5. HTML 内的目录/锚点跳转必须只跳转到当前这个 HTML 文件内部:只使用 href="#section-id",不要链接到 .html/.md 文件作为锚点。
"""
_STYLE_EXTRACT_TEMPLATE = """你是一个页面风格分析助手。请分析下面的 HTML 页面,提取可复用的风格与布局特征。
HTML 页面:
{html}
要求:
1. 只输出 JSON,字段包含 style_summary、layout_summary、content_structure_summary、reuse_prompt。
2. style_summary:颜色、字体、间距、圆角、阴影、视觉密度等视觉风格。
3. layout_summary:页面分区、卡片、表格、图表、导航等布局结构。
4. content_structure_summary:信息组织顺序与层级。
5. reuse_prompt:一段可直接拼接到下次生成 prompt 的中文风格说明,不要包含本页的具体数据。
"""
def _clip(value: Any, limit: int) -> str:
text = value if isinstance(value, str) else json.dumps(value, ensure_ascii=False, default=str)
text = text or ""
return text if len(text) <= limit else text[:limit] + "\n…(内容过长已截断)"
def build_generation_prompt(
*,
page_prompt: str,
negative_prompt: str | None,
data: Any,
reference_style: str | None,
reference_layout: str | None,
skill_instructions: str | None = None,
data_limit: int = 16000,
skill_limit: int = 30000,
) -> str:
skill_block = ""
if skill_instructions and skill_instructions.strip():
skill_block = "\n可用页面生成技能规范(请严格遵循):\n" + _clip(skill_instructions.strip(), skill_limit) + "\n"
# The constraint block is only emitted when the user actually supplied one,
# and is framed as a top-priority hard rule that overrides the page
# requirements / reference page / data on conflict — otherwise the model
# tends to treat "限制条件" as a soft hint and ignore it.
constraint_block = ""
neg = (negative_prompt or "").strip()
if neg:
constraint_block = (
"\n限制条件(最高优先级的硬性约束,必须逐条严格遵守;"
"若与上面的页面要求、参考页面或本次数据发生冲突,一律以这里的限制条件为准):\n"
+ neg
+ "\n\n"
)
return _GENERATION_TEMPLATE.format(
page_prompt=(page_prompt or "(未提供,使用通用看板风格)").strip(),
constraint_block=constraint_block,
data=_clip(data, data_limit),
reference_style=(reference_style or "无").strip(),
reference_layout=(reference_layout or "无").strip(),
skill_instructions=skill_block,
)
def build_repair_prompt(*, html: str, errors: list[str], requirements: str | None) -> str:
return _REPAIR_TEMPLATE.format(
requirements=(requirements or "无").strip(),
errors="\n".join(f"- {e}" for e in errors) or "- 未知问题",
html=_clip(html, 32000),
)
def build_style_extract_prompt(*, html: str) -> str:
return _STYLE_EXTRACT_TEMPLATE.format(html=_clip(html, 32000))
async def screenshot_html(html: str, output_path) -> str | None:
"""Best-effort PNG screenshot of ``html`` for white-screen / console QA.
Returns the screenshot path on success, or ``None`` when Playwright (or a
usable browser) is unavailable — the offline image ships without one, so
this must degrade silently rather than fail the run.
"""
try:
from playwright.async_api import async_playwright
except Exception:
return None
try:
async with async_playwright() as pw:
browser = await pw.chromium.launch()
try:
page = await browser.new_page()
await page.set_content(html, wait_until="networkidle")
await page.screenshot(path=str(output_path), full_page=True)
finally:
await browser.close()
return str(output_path)
except Exception:
logger.debug("Playwright screenshot failed; skipping QA screenshot", exc_info=True)
return None