agent build
This commit is contained in:
101
go-api/internal/knowledge/chunk_test.go
Normal file
101
go-api/internal/knowledge/chunk_test.go
Normal file
@@ -0,0 +1,101 @@
|
||||
package knowledge
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
"unicode/utf8"
|
||||
)
|
||||
|
||||
func TestHeadingsStartNewChunks(t *testing.T) {
|
||||
// A heading is the author's own statement that a new topic begins. Carrying
|
||||
// text across one puts two topics in a single unit and labels it with the
|
||||
// wrong section — which then cites wrongly.
|
||||
body := "# Attendance\n\n" +
|
||||
strings.Repeat("Lateness is measured against the scheduled start. ", 4) + "\n\n" +
|
||||
"# Breaks\n\n" +
|
||||
strings.Repeat("A shift over six hours carries a thirty minute break. ", 4)
|
||||
|
||||
chunks := Split("Staff Handbook", body)
|
||||
if len(chunks) < 2 {
|
||||
t.Fatalf("%d chunks; a two-section document should not be one chunk", len(chunks))
|
||||
}
|
||||
for _, c := range chunks {
|
||||
if strings.Contains(c.Text, "Lateness") && strings.Contains(c.Text, "thirty minute") {
|
||||
t.Error("text was carried across a heading boundary")
|
||||
}
|
||||
if !strings.HasPrefix(c.Heading, "Staff Handbook") {
|
||||
t.Errorf("chunk heading %q does not start from the document title", c.Heading)
|
||||
}
|
||||
}
|
||||
if !strings.Contains(chunks[0].Heading, "Attendance") {
|
||||
t.Errorf("first chunk heading is %q, want it to name its section", chunks[0].Heading)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnOverlongParagraphIsSplitWithOverlap(t *testing.T) {
|
||||
// The boundary problem: the sentence that answers a question often straddles
|
||||
// a break, and without overlap neither side retrieves well.
|
||||
long := strings.Repeat("The venue manager approves every shift swap in advance. ", 120)
|
||||
chunks := Split("Handbook", long)
|
||||
|
||||
if len(chunks) < 2 {
|
||||
t.Fatalf("a %d-rune paragraph produced %d chunks", utf8.RuneCountInString(long), len(chunks))
|
||||
}
|
||||
for _, c := range chunks {
|
||||
if n := utf8.RuneCountInString(c.Text); n > MaxChunkRunes {
|
||||
t.Errorf("a chunk is %d runes, over the %d cap", n, MaxChunkRunes)
|
||||
}
|
||||
}
|
||||
// Consecutive chunks should share a tail/head.
|
||||
tail := chunks[0].Text
|
||||
if len(tail) > 60 {
|
||||
tail = tail[len(tail)-60:]
|
||||
}
|
||||
if !strings.Contains(chunks[1].Text, strings.TrimSpace(tail[:30])) {
|
||||
t.Error("consecutive chunks do not overlap; a sentence spanning the break would be lost")
|
||||
}
|
||||
}
|
||||
|
||||
func TestAFragmentIsFoldedIntoItsNeighbour(t *testing.T) {
|
||||
// A stray line indexed on its own will match on a keyword and then say
|
||||
// nothing, which is worse than not matching at all.
|
||||
body := strings.Repeat("Shift swaps need approval from the venue manager. ", 6) + "\n\nSee above."
|
||||
chunks := Split("Handbook", body)
|
||||
|
||||
for _, c := range chunks {
|
||||
if strings.TrimSpace(c.Text) == "See above." {
|
||||
t.Error("a two-word fragment was indexed as its own chunk")
|
||||
}
|
||||
}
|
||||
if !strings.Contains(chunks[len(chunks)-1].Text, "See above.") {
|
||||
t.Error("the fragment was dropped rather than folded in")
|
||||
}
|
||||
}
|
||||
|
||||
func TestOrdinalsAreContiguousFromZero(t *testing.T) {
|
||||
// The schema has UNIQUE (document_id, ordinal) and citations say "chunk 3
|
||||
// of this document". A gap or a repeat breaks both.
|
||||
chunks := Split("Handbook", strings.Repeat("Some policy text here. ", 400))
|
||||
for i, c := range chunks {
|
||||
if c.Ordinal != i {
|
||||
t.Fatalf("chunk %d has ordinal %d", i, c.Ordinal)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestProseThatStartsWithAHashIsNotAHeading(t *testing.T) {
|
||||
// `# 1 applies to agency staff` inside a paragraph is prose. Treating it as
|
||||
// a heading would split mid-thought and mislabel everything after it.
|
||||
body := "# Attendance\n\n#3 on the rota is the closing shift and it is not covered by this section."
|
||||
_, headings := parse("Handbook", body)
|
||||
|
||||
hashPrefixed := 0
|
||||
for _, h := range headings {
|
||||
if h != "" {
|
||||
hashPrefixed++
|
||||
}
|
||||
}
|
||||
if hashPrefixed != 1 {
|
||||
t.Errorf("%d headings detected, want 1 — prose beginning with a hash was misread", hashPrefixed)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user