61 lines
2.0 KiB
Python
61 lines
2.0 KiB
Python
"""Markdown chunking for embedding/index (phase 2).
|
|
|
|
Splits a note body into overlapping chunks along Markdown structure (headings →
|
|
paragraphs) so each embedded chunk is semantically coherent and bounded in
|
|
size. Frontmatter is assumed already stripped by the caller.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
|
|
_HEADING_RE = re.compile(r"^#{1,6}\s+", re.MULTILINE)
|
|
|
|
|
|
def _split_paragraphs(text: str) -> list[str]:
|
|
return [p.strip() for p in re.split(r"\n\s*\n", text) if p.strip()]
|
|
|
|
|
|
def chunk_markdown(content: str, *, max_chars: int = 800, overlap: int = 120) -> list[str]:
|
|
"""Return a list of text chunks for ``content``.
|
|
|
|
Sections are cut at Markdown headings; long sections are further split into
|
|
~``max_chars`` windows with ``overlap`` characters of carry-over.
|
|
"""
|
|
text = (content or "").strip()
|
|
if not text:
|
|
return []
|
|
|
|
# Split into heading-delimited sections, keeping the heading with its body.
|
|
sections: list[str] = []
|
|
last = 0
|
|
for m in _HEADING_RE.finditer(text):
|
|
if m.start() > last:
|
|
sections.append(text[last : m.start()].strip())
|
|
last = m.start()
|
|
sections.append(text[last:].strip())
|
|
sections = [s for s in sections if s]
|
|
|
|
chunks: list[str] = []
|
|
for section in sections:
|
|
if len(section) <= max_chars:
|
|
chunks.append(section)
|
|
continue
|
|
# Pack paragraphs into windows, then hard-split anything still too long.
|
|
buf = ""
|
|
for para in _split_paragraphs(section):
|
|
if len(buf) + len(para) + 2 <= max_chars:
|
|
buf = f"{buf}\n\n{para}".strip()
|
|
else:
|
|
if buf:
|
|
chunks.append(buf)
|
|
if len(para) <= max_chars:
|
|
buf = para
|
|
else:
|
|
for i in range(0, len(para), max_chars - overlap):
|
|
chunks.append(para[i : i + max_chars])
|
|
buf = ""
|
|
if buf:
|
|
chunks.append(buf)
|
|
return [c for c in chunks if c.strip()]
|