fix(vectorstore): hard-split oversized heading-less chunks
A doc with no headings and no blank-line paragraphs (JSON-lines, e.g. wiki/telos/decisions/human-intent-column.md) survived both chunk passes whole and was sent to nomic-embed over its context window → 'input length exceeds the context length' (400, the steady embed errors=1). Add a final hard-split pass (line then UTF-8 rune boundaries) so no chunk exceeds maxBytes. TDD. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -30,6 +30,23 @@ func TestChunkMarkdown_SplitsAtHeadings(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestChunkMarkdown_HardSplitsHeadinglessOversizedBlock(t *testing.T) {
|
||||
// A document with no headings and no blank-line paragraph breaks (e.g.
|
||||
// JSON-lines like wiki/telos/decisions/human-intent-column.md). The old
|
||||
// chunker emitted it as one over-budget chunk → nomic-embed returned
|
||||
// "input length exceeds the context length" (400). Every chunk must now
|
||||
// fit the budget, with no content lost.
|
||||
maxBytes := 200
|
||||
src := strings.Repeat("x", 1000) // one 1000-byte blob, no headings, no \n\n
|
||||
out := vectorstore.ChunkMarkdown(src, maxBytes)
|
||||
|
||||
require.Greater(t, len(out), 1, "oversized blob must be split")
|
||||
for i, c := range out {
|
||||
assert.LessOrEqual(t, len(c), maxBytes, "chunk %d over budget: %d bytes", i, len(c))
|
||||
}
|
||||
assert.Equal(t, 1000, strings.Count(strings.Join(out, ""), "x"), "no content lost")
|
||||
}
|
||||
|
||||
func TestChunkMarkdown_FurtherSplitsOversizedSection(t *testing.T) {
|
||||
// One H2 section with 4 paragraphs of ~80 chars each, limit 100.
|
||||
src := "## big\n\n" +
|
||||
|
||||
Reference in New Issue
Block a user