deerflow-code/offline-backend-20260512/backend/packages/harness/deerflow/knowledge/chunking.py
2026-09-07 18:24:55 +08:00

61 lines
2.0 KiB
Python

"""Markdown chunking for embedding/index (phase 2).
Splits a note body into overlapping chunks along Markdown structure (headings →
paragraphs) so each embedded chunk is semantically coherent and bounded in
size. Frontmatter is assumed already stripped by the caller.
"""
from __future__ import annotations
import re
_HEADING_RE = re.compile(r"^#{1,6}\s+", re.MULTILINE)
def _split_paragraphs(text: str) -> list[str]:
return [p.strip() for p in re.split(r"\n\s*\n", text) if p.strip()]
def chunk_markdown(content: str, *, max_chars: int = 800, overlap: int = 120) -> list[str]:
"""Return a list of text chunks for ``content``.
Sections are cut at Markdown headings; long sections are further split into
~``max_chars`` windows with ``overlap`` characters of carry-over.
"""
text = (content or "").strip()
if not text:
return []
# Split into heading-delimited sections, keeping the heading with its body.
sections: list[str] = []
last = 0
for m in _HEADING_RE.finditer(text):
if m.start() > last:
sections.append(text[last : m.start()].strip())
last = m.start()
sections.append(text[last:].strip())
sections = [s for s in sections if s]
chunks: list[str] = []
for section in sections:
if len(section) <= max_chars:
chunks.append(section)
continue
# Pack paragraphs into windows, then hard-split anything still too long.
buf = ""
for para in _split_paragraphs(section):
if len(buf) + len(para) + 2 <= max_chars:
buf = f"{buf}\n\n{para}".strip()
else:
if buf:
chunks.append(buf)
if len(para) <= max_chars:
buf = para
else:
for i in range(0, len(para), max_chars - overlap):
chunks.append(para[i : i + max_chars])
buf = ""
if buf:
chunks.append(buf)
return [c for c in chunks if c.strip()]