package knowledge import ( "strings" "testing" "unicode/utf8" ) func TestHeadingsStartNewChunks(t *testing.T) { // A heading is the author's own statement that a new topic begins. Carrying // text across one puts two topics in a single unit and labels it with the // wrong section — which then cites wrongly. body := "# Attendance\n\n" + strings.Repeat("Lateness is measured against the scheduled start. ", 4) + "\n\n" + "# Breaks\n\n" + strings.Repeat("A shift over six hours carries a thirty minute break. ", 4) chunks := Split("Staff Handbook", body) if len(chunks) < 2 { t.Fatalf("%d chunks; a two-section document should not be one chunk", len(chunks)) } for _, c := range chunks { if strings.Contains(c.Text, "Lateness") && strings.Contains(c.Text, "thirty minute") { t.Error("text was carried across a heading boundary") } if !strings.HasPrefix(c.Heading, "Staff Handbook") { t.Errorf("chunk heading %q does not start from the document title", c.Heading) } } if !strings.Contains(chunks[0].Heading, "Attendance") { t.Errorf("first chunk heading is %q, want it to name its section", chunks[0].Heading) } } func TestAnOverlongParagraphIsSplitWithOverlap(t *testing.T) { // The boundary problem: the sentence that answers a question often straddles // a break, and without overlap neither side retrieves well. long := strings.Repeat("The venue manager approves every shift swap in advance. ", 120) chunks := Split("Handbook", long) if len(chunks) < 2 { t.Fatalf("a %d-rune paragraph produced %d chunks", utf8.RuneCountInString(long), len(chunks)) } for _, c := range chunks { if n := utf8.RuneCountInString(c.Text); n > MaxChunkRunes { t.Errorf("a chunk is %d runes, over the %d cap", n, MaxChunkRunes) } } // Consecutive chunks should share a tail/head. tail := chunks[0].Text if len(tail) > 60 { tail = tail[len(tail)-60:] } if !strings.Contains(chunks[1].Text, strings.TrimSpace(tail[:30])) { t.Error("consecutive chunks do not overlap; a sentence spanning the break would be lost") } } func TestAFragmentIsFoldedIntoItsNeighbour(t *testing.T) { // A stray line indexed on its own will match on a keyword and then say // nothing, which is worse than not matching at all. body := strings.Repeat("Shift swaps need approval from the venue manager. ", 6) + "\n\nSee above." chunks := Split("Handbook", body) for _, c := range chunks { if strings.TrimSpace(c.Text) == "See above." { t.Error("a two-word fragment was indexed as its own chunk") } } if !strings.Contains(chunks[len(chunks)-1].Text, "See above.") { t.Error("the fragment was dropped rather than folded in") } } func TestOrdinalsAreContiguousFromZero(t *testing.T) { // The schema has UNIQUE (document_id, ordinal) and citations say "chunk 3 // of this document". A gap or a repeat breaks both. chunks := Split("Handbook", strings.Repeat("Some policy text here. ", 400)) for i, c := range chunks { if c.Ordinal != i { t.Fatalf("chunk %d has ordinal %d", i, c.Ordinal) } } } func TestProseThatStartsWithAHashIsNotAHeading(t *testing.T) { // `# 1 applies to agency staff` inside a paragraph is prose. Treating it as // a heading would split mid-thought and mislabel everything after it. body := "# Attendance\n\n#3 on the rota is the closing shift and it is not covered by this section." _, headings := parse("Handbook", body) hashPrefixed := 0 for _, h := range headings { if h != "" { hashPrefixed++ } } if hashPrefixed != 1 { t.Errorf("%d headings detected, want 1 — prose beginning with a hash was misread", hashPrefixed) } }