package portfolio import ( "regexp" "strings" ) // Chunk is one indexable unit of a project document. type Chunk struct { ID string ProjectID string SourceFile string Index int Section string // frontmatter | H1 title | H2 section name Content string } // ChunkerConfig controls the heading-based chunker. type ChunkerConfig struct { MaxSectionChars int // sections longer than this are sub-split (default 1000) MergeUnderChars int // sections shorter than this are merged with the next (default 50) OverlapChars int // overlap when sub-splitting (default 80) } func DefaultChunkerConfig() ChunkerConfig { return ChunkerConfig{ MaxSectionChars: 1000, MergeUnderChars: 50, OverlapChars: 80, } } // SplitMarkdownSections produces chunks using heading boundaries: // 1. Frontmatter (between leading --- ... ---) → "frontmatter" chunk // 2. Each H1 (project title) → one chunk // 3. Each H2 section → one chunk (with its sub-content under H3/H4 etc.) // 4. Sections > MaxSectionChars are sub-split by size with overlap // 5. Sections < MergeUnderChars are merged with the next section func SplitMarkdownSections(markdown string, cfg ChunkerConfig) []section { if cfg.MaxSectionChars <= 0 { cfg = DefaultChunkerConfig() } if cfg.MergeUnderChars < 0 { cfg.MergeUnderChars = 0 } body := markdown frontmatter := "" if fm, rest, ok := extractFrontmatter(markdown); ok { frontmatter = fm body = rest } var sections []section if frontmatter != "" { sections = append(sections, section{Heading: "frontmatter", Body: frontmatter}) } for _, s := range splitByHeadings(body) { sections = append(sections, s) } sections = dropEmpty(sections) sections = subSplit(sections, cfg.MaxSectionChars, cfg.OverlapChars) return sections } type section struct { Heading string Body string } var ( frontmatterRe = regexp.MustCompile(`^---\n([\s\S]*?)\n---\n?`) h1Re = regexp.MustCompile(`(?m)^# [^#].*$`) ) func extractFrontmatter(s string) (string, string, bool) { m := frontmatterRe.FindStringSubmatchIndex(s) if m == nil { return "", s, false } return s[m[2]:m[3]], s[m[1]:], true } // splitByHeadings splits the body into sections keyed by H1 or H2. // H3+ content stays attached to its parent H2 (no extra split), which keeps // the natural unit coherent: the bot answers about "Description" or "Tech // stack", not about individual bullet points. func splitByHeadings(body string) []section { lines := strings.Split(body, "\n") var sections []section var current section inSection := false for _, line := range lines { if isH1(line) || isH2(line) { if inSection { sections = append(sections, current) } current = section{Heading: strings.TrimSpace(strings.TrimLeft(strings.TrimSpace(line), "# "))} inSection = true continue } if inSection { current.Body += line + "\n" } } if inSection { sections = append(sections, current) } // Strip trailing whitespace per section for i := range sections { sections[i].Body = strings.TrimRight(sections[i].Body, "\n ") } return sections } func isH1(l string) bool { return strings.HasPrefix(l, "# ") && !strings.HasPrefix(l, "## ") } func isH2(l string) bool { return strings.HasPrefix(l, "## ") } // dropEmpty removes sections whose body is empty or just whitespace. // Keeps the heading-level chunking honest: a "Description" with no body // is not a useful retrieval target. func dropEmpty(sections []section) []section { out := sections[:0] for _, s := range sections { if strings.TrimSpace(s.Body) == "" { continue } out = append(out, s) } return out } // subSplit breaks down any section whose body exceeds max into overlapping // slices, preserving the heading as a prefix on each piece so context isn't // lost mid-section. func subSplit(sections []section, max, overlap int) []section { if max <= 0 { return sections } if overlap < 0 || overlap >= max { overlap = max / 10 } var out []section for _, s := range sections { if len(s.Body) <= max { out = append(out, s) continue } for i := 0; i < len(s.Body); { end := i + max if end > len(s.Body) { end = len(s.Body) } piece := s.Body[i:end] if i > 0 { piece = "... " + piece } if end < len(s.Body) { piece = piece + " ..." } out = append(out, section{Heading: s.Heading, Body: piece}) if end == len(s.Body) { break } i += max - overlap } } return out }