A doc with no headings and no blank-line paragraphs (JSON-lines, e.g. wiki/telos/decisions/human-intent-column.md) survived both chunk passes whole and was sent to nomic-embed over its context window → 'input length exceeds the context length' (400, the steady embed errors=1). Add a final hard-split pass (line then UTF-8 rune boundaries) so no chunk exceeds maxBytes. TDD. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
203 lines
5.6 KiB
Go
203 lines
5.6 KiB
Go
package vectorstore
|
|
|
|
import (
|
|
"fmt"
|
|
"strings"
|
|
"unicode/utf8"
|
|
)
|
|
|
|
// NumberedChunk pairs a chunk's body with the storage path it will use
|
|
// in brain_embeddings. Path format: "<parent>#NNNN" where NNNN is the
|
|
// 1-based chunk index zero-padded to 4 digits.
|
|
type NumberedChunk struct {
|
|
Path string
|
|
Content string
|
|
}
|
|
|
|
// ParentPath returns the file path with any "#NNNN" chunk suffix removed.
|
|
// Inputs without a "#" are returned unchanged. Used by search to dedupe
|
|
// chunk-level hits back to a single document per result.
|
|
func ParentPath(p string) string {
|
|
if i := strings.Index(p, "#"); i >= 0 {
|
|
return p[:i]
|
|
}
|
|
return p
|
|
}
|
|
|
|
// NumberChunks assigns "<parent>#NNNN" storage paths to a slice of chunk
|
|
// bodies, indexed from 0001. Empty chunks are dropped.
|
|
func NumberChunks(parent string, chunks []string) []NumberedChunk {
|
|
out := make([]NumberedChunk, 0, len(chunks))
|
|
idx := 1
|
|
for _, c := range chunks {
|
|
if strings.TrimSpace(c) == "" {
|
|
continue
|
|
}
|
|
out = append(out, NumberedChunk{
|
|
Path: fmt.Sprintf("%s#%04d", parent, idx),
|
|
Content: c,
|
|
})
|
|
idx++
|
|
}
|
|
return out
|
|
}
|
|
|
|
// ChunkMarkdown splits a markdown document into embedding-sized pieces.
|
|
// Strategy:
|
|
// 1. Split at H1/H2 headings (top-of-line "#" or "##"). The intro before
|
|
// the first heading is its own chunk.
|
|
// 2. Any section larger than maxBytes is further split at paragraph
|
|
// boundaries (blank lines), packing paragraphs greedily under the
|
|
// byte budget.
|
|
//
|
|
// The function aims for "fits comfortably under nomic-embed-text's 2048-
|
|
// token context" — at ~4 chars/token for English markdown, maxBytes ≈ 4000
|
|
// is a safe call-site default.
|
|
func ChunkMarkdown(content string, maxBytes int) []string {
|
|
if maxBytes <= 0 {
|
|
maxBytes = 4000
|
|
}
|
|
sections := splitAtHeadings(content)
|
|
|
|
out := make([]string, 0, len(sections))
|
|
for _, s := range sections {
|
|
if len(s) <= maxBytes {
|
|
out = append(out, s)
|
|
continue
|
|
}
|
|
out = append(out, splitAtParagraphs(s, maxBytes)...)
|
|
}
|
|
|
|
// Final guarantee: no chunk exceeds maxBytes. A single heading-less,
|
|
// paragraph-less block (JSON-lines, minified content) survives the two
|
|
// passes above whole — splitAtParagraphs emits an over-budget paragraph
|
|
// rather than truncating prose. Hard-split any such chunk at line/rune
|
|
// boundaries so the embedder never rejects an over-context chunk.
|
|
final := make([]string, 0, len(out))
|
|
for _, c := range out {
|
|
if len(c) <= maxBytes {
|
|
final = append(final, c)
|
|
continue
|
|
}
|
|
final = append(final, hardSplit(c, maxBytes)...)
|
|
}
|
|
return final
|
|
}
|
|
|
|
// hardSplit slices s into pieces no larger than maxBytes, breaking at line
|
|
// boundaries where possible and otherwise mid-line at a UTF-8 rune boundary.
|
|
// Last resort for content that has neither headings nor blank-line paragraphs.
|
|
func hardSplit(s string, maxBytes int) []string {
|
|
var out []string
|
|
var cur strings.Builder
|
|
flush := func() {
|
|
if cur.Len() > 0 {
|
|
out = append(out, cur.String())
|
|
cur.Reset()
|
|
}
|
|
}
|
|
for _, line := range strings.SplitAfter(s, "\n") {
|
|
if line == "" {
|
|
continue
|
|
}
|
|
if len(line) > maxBytes {
|
|
flush()
|
|
out = append(out, runeSplit(line, maxBytes)...)
|
|
continue
|
|
}
|
|
if cur.Len() > 0 && cur.Len()+len(line) > maxBytes {
|
|
flush()
|
|
}
|
|
cur.WriteString(line)
|
|
}
|
|
flush()
|
|
return out
|
|
}
|
|
|
|
// runeSplit slices s into <=maxBytes pieces without splitting a UTF-8 rune.
|
|
func runeSplit(s string, maxBytes int) []string {
|
|
var out []string
|
|
for len(s) > maxBytes {
|
|
cut := maxBytes
|
|
for cut > 0 && !utf8.RuneStart(s[cut]) {
|
|
cut--
|
|
}
|
|
if cut == 0 { // single rune wider than the budget; emit it whole
|
|
cut = maxBytes
|
|
}
|
|
out = append(out, s[:cut])
|
|
s = s[cut:]
|
|
}
|
|
if len(s) > 0 {
|
|
out = append(out, s)
|
|
}
|
|
return out
|
|
}
|
|
|
|
// splitAtHeadings cuts content into sections that each start with an
|
|
// "# " or "## " line (intro before any heading is the leading section).
|
|
func splitAtHeadings(content string) []string {
|
|
lines := strings.Split(content, "\n")
|
|
var sections []string
|
|
var cur strings.Builder
|
|
flush := func() {
|
|
if cur.Len() == 0 {
|
|
return
|
|
}
|
|
// Trim all trailing whitespace then re-add a single newline so a
|
|
// single-paragraph file round-trips to its original content rather
|
|
// than accumulating extra newlines from the empty-line split.
|
|
s := strings.TrimRight(cur.String(), "\n")
|
|
sections = append(sections, s+"\n")
|
|
cur.Reset()
|
|
}
|
|
for _, ln := range lines {
|
|
trimmed := strings.TrimLeft(ln, " ")
|
|
isH := strings.HasPrefix(trimmed, "# ") || strings.HasPrefix(trimmed, "## ")
|
|
if isH && cur.Len() > 0 {
|
|
flush()
|
|
}
|
|
cur.WriteString(ln)
|
|
cur.WriteByte('\n')
|
|
}
|
|
flush()
|
|
// Drop empty / whitespace-only trailing section (common when content
|
|
// itself ends with a "\n" — Split leaves a final empty element).
|
|
if n := len(sections); n > 0 && strings.TrimSpace(sections[n-1]) == "" {
|
|
sections = sections[:n-1]
|
|
}
|
|
return sections
|
|
}
|
|
|
|
// splitAtParagraphs packs paragraphs (blank-line separated blocks) into
|
|
// sub-chunks of at most maxBytes. A single paragraph that itself exceeds
|
|
// maxBytes is emitted as one over-budget chunk rather than being split
|
|
// mid-sentence — better to over-spend a little than truncate prose.
|
|
func splitAtParagraphs(section string, maxBytes int) []string {
|
|
paras := strings.Split(section, "\n\n")
|
|
var out []string
|
|
var cur strings.Builder
|
|
for _, p := range paras {
|
|
if p == "" {
|
|
continue
|
|
}
|
|
// +2 for the "\n\n" rejoin if cur isn't empty
|
|
need := len(p)
|
|
if cur.Len() > 0 {
|
|
need += 2
|
|
}
|
|
if cur.Len() > 0 && cur.Len()+need > maxBytes {
|
|
out = append(out, cur.String())
|
|
cur.Reset()
|
|
}
|
|
if cur.Len() > 0 {
|
|
cur.WriteString("\n\n")
|
|
}
|
|
cur.WriteString(p)
|
|
}
|
|
if cur.Len() > 0 {
|
|
out = append(out, cur.String())
|
|
}
|
|
return out
|
|
}
|