This repository has been archived on 2026-04-03. You can view files and clone it. You cannot open issues or pull requests or push a commit.
Files
core/server/knowledge/chunker_test.go
2026-02-26 15:59:26 +00:00

228 lines
6.4 KiB
Go

package knowledge
import (
"strings"
"testing"
"unicode/utf8"
)
func TestSplitText_EmptyInput(t *testing.T) {
chunks := SplitText("", DefaultChunkConfig())
if chunks != nil {
t.Errorf("expected nil for empty input, got %d chunks", len(chunks))
}
}
func TestSplitText_WhitespaceOnly(t *testing.T) {
chunks := SplitText(" \n\n ", DefaultChunkConfig())
if chunks != nil {
t.Errorf("expected nil for whitespace input, got %d chunks", len(chunks))
}
}
func TestSplitText_SmallText(t *testing.T) {
text := "Hello, world."
chunks := SplitText(text, DefaultChunkConfig())
if len(chunks) != 1 {
t.Fatalf("expected 1 chunk, got %d", len(chunks))
}
if chunks[0].Content != text {
t.Errorf("content mismatch: %q", chunks[0].Content)
}
if chunks[0].Index != 0 {
t.Errorf("expected index 0, got %d", chunks[0].Index)
}
if chunks[0].TokenCount <= 0 {
t.Error("expected positive token count")
}
}
func TestSplitText_ParagraphSplit(t *testing.T) {
// Two paragraphs, each under chunk size, but together over.
para := strings.Repeat("word ", 60) // ~300 chars each
text := para + "\n\n" + para
cfg := ChunkConfig{ChunkSize: 400, ChunkOverlap: 0, Separators: []string{"\n\n", "\n", ". ", " "}}
chunks := SplitText(text, cfg)
if len(chunks) < 2 {
t.Fatalf("expected at least 2 chunks from paragraph split, got %d", len(chunks))
}
for i, c := range chunks {
if utf8.RuneCountInString(c.Content) == 0 {
t.Errorf("chunk %d is empty", i)
}
}
}
func TestSplitText_OverlapPresent(t *testing.T) {
// Build text that will split into multiple chunks.
sentences := make([]string, 20)
for i := range sentences {
sentences[i] = "This is sentence number " + strings.Repeat("x", 40) + ". "
}
text := strings.Join(sentences, "")
cfg := ChunkConfig{ChunkSize: 200, ChunkOverlap: 50, Separators: []string{". ", " "}}
chunks := SplitText(text, cfg)
if len(chunks) < 3 {
t.Fatalf("expected at least 3 chunks, got %d", len(chunks))
}
// Verify chunks have sequential indexes.
for i, c := range chunks {
if c.Index != i {
t.Errorf("chunk %d has index %d", i, c.Index)
}
}
// With overlap > 0 and multiple chunks, later chunks should share
// some content with the previous chunk's tail.
// Just verify the overlap logic ran (chunks 1+ are longer or contain
// content from the previous chunk's ending).
if len(chunks) >= 2 {
// Chunk 1 should contain some overlap from chunk 0's tail.
// We can't check exact content easily, but the chunk should be
// non-trivially sized.
if len(chunks[1].Content) < 50 {
t.Errorf("chunk 1 seems too short for overlap: %d chars", len(chunks[1].Content))
}
}
}
func TestSplitText_ZeroOverlap(t *testing.T) {
text := strings.Repeat("A", 500) + "\n\n" + strings.Repeat("B", 500)
cfg := ChunkConfig{ChunkSize: 400, ChunkOverlap: 0, Separators: []string{"\n\n"}}
chunks := SplitText(text, cfg)
if len(chunks) < 2 {
t.Fatalf("expected at least 2 chunks, got %d", len(chunks))
}
// Without overlap, chunks should not share content.
for i, c := range chunks {
if c.Index != i {
t.Errorf("chunk %d index = %d", i, c.Index)
}
}
}
func TestSplitText_HardSplitFallback(t *testing.T) {
// No separators at all in the text — forces hard character split.
text := strings.Repeat("x", 500)
cfg := ChunkConfig{ChunkSize: 200, ChunkOverlap: 0, Separators: []string{"\n\n", "\n", ". ", " "}}
chunks := SplitText(text, cfg)
if len(chunks) < 2 {
t.Fatalf("expected multiple chunks from hard split, got %d", len(chunks))
}
// All chunks should be at most chunkSize.
for i, c := range chunks {
if utf8.RuneCountInString(c.Content) > cfg.ChunkSize {
t.Errorf("chunk %d exceeds chunk size: %d > %d",
i, utf8.RuneCountInString(c.Content), cfg.ChunkSize)
}
}
}
func TestSplitText_TokenEstimate(t *testing.T) {
text := strings.Repeat("word ", 100) // 500 chars
chunks := SplitText(text, ChunkConfig{ChunkSize: 2000, ChunkOverlap: 0})
if len(chunks) != 1 {
t.Fatalf("expected 1 chunk, got %d", len(chunks))
}
// ~500 chars / 4 = ~125 tokens
if chunks[0].TokenCount < 100 || chunks[0].TokenCount > 150 {
t.Errorf("expected ~125 tokens, got %d", chunks[0].TokenCount)
}
}
func TestSplitText_InvalidConfig(t *testing.T) {
text := "Hello world. This is a test."
// ChunkSize 0 should use default.
chunks := SplitText(text, ChunkConfig{ChunkSize: 0})
if len(chunks) != 1 {
t.Errorf("expected 1 chunk with zero chunk size (uses default), got %d", len(chunks))
}
// Overlap >= ChunkSize should be clamped.
chunks = SplitText(strings.Repeat("word ", 200), ChunkConfig{
ChunkSize: 100,
ChunkOverlap: 100,
})
if len(chunks) == 0 {
t.Error("expected chunks with overlap == chunk_size (should clamp)")
}
}
func TestSplitText_SentenceSplit(t *testing.T) {
text := "First sentence. Second sentence. Third sentence. Fourth sentence. Fifth sentence."
cfg := ChunkConfig{ChunkSize: 40, ChunkOverlap: 0, Separators: []string{"\n\n", "\n", ". ", " "}}
chunks := SplitText(text, cfg)
if len(chunks) < 2 {
t.Fatalf("expected multiple sentence-split chunks, got %d", len(chunks))
}
// Each chunk should be within bounds.
for i, c := range chunks {
if utf8.RuneCountInString(c.Content) == 0 {
t.Errorf("chunk %d is empty", i)
}
}
}
func TestSplitText_MetadataInitialized(t *testing.T) {
chunks := SplitText("Some text here.", DefaultChunkConfig())
if len(chunks) != 1 {
t.Fatal("expected 1 chunk")
}
if chunks[0].Metadata == nil {
t.Error("metadata should be initialized, got nil")
}
}
func TestEstimateTokens(t *testing.T) {
cases := []struct {
input string
minTok int
maxTok int
}{
{"", 0, 0},
{"hi", 1, 1},
{"hello world", 2, 4},
{strings.Repeat("a", 100), 20, 30},
}
for _, tc := range cases {
got := estimateTokens(tc.input)
if got < tc.minTok || got > tc.maxTok {
t.Errorf("estimateTokens(%q): got %d, want [%d, %d]", tc.input, got, tc.minTok, tc.maxTok)
}
}
}
func TestOverlapSuffix(t *testing.T) {
s := "the quick brown fox jumps over the lazy dog"
// Should return last ~20 chars, breaking at word boundary.
suffix := overlapSuffix(s, 20)
if len(suffix) == 0 {
t.Error("expected non-empty suffix")
}
if len(suffix) > 25 { // some slack for word boundary adjustment
t.Errorf("suffix too long: %d chars: %q", len(suffix), suffix)
}
// Short string: return whole thing.
short := "hello"
if got := overlapSuffix(short, 100); got != short {
t.Errorf("expected %q, got %q", short, got)
}
}