"""Markdown chunking for embedding/index (phase 2). Splits a note body into overlapping chunks along Markdown structure (headings → paragraphs) so each embedded chunk is semantically coherent and bounded in size. Frontmatter is assumed already stripped by the caller. """ from __future__ import annotations import re _HEADING_RE = re.compile(r"^#{1,6}\s+", re.MULTILINE) def _split_paragraphs(text: str) -> list[str]: return [p.strip() for p in re.split(r"\n\s*\n", text) if p.strip()] def chunk_markdown(content: str, *, max_chars: int = 800, overlap: int = 120) -> list[str]: """Return a list of text chunks for ``content``. Sections are cut at Markdown headings; long sections are further split into ~``max_chars`` windows with ``overlap`` characters of carry-over. """ text = (content or "").strip() if not text: return [] # Split into heading-delimited sections, keeping the heading with its body. sections: list[str] = [] last = 0 for m in _HEADING_RE.finditer(text): if m.start() > last: sections.append(text[last : m.start()].strip()) last = m.start() sections.append(text[last:].strip()) sections = [s for s in sections if s] chunks: list[str] = [] for section in sections: if len(section) <= max_chars: chunks.append(section) continue # Pack paragraphs into windows, then hard-split anything still too long. buf = "" for para in _split_paragraphs(section): if len(buf) + len(para) + 2 <= max_chars: buf = f"{buf}\n\n{para}".strip() else: if buf: chunks.append(buf) if len(para) <= max_chars: buf = para else: for i in range(0, len(para), max_chars - overlap): chunks.append(para[i : i + max_chars]) buf = "" if buf: chunks.append(buf) return [c for c in chunks if c.strip()]