Files
krow_backend/go-api/internal/knowledge/chunk.go
2026-08-28 12:21:44 +05:30

267 lines
8.2 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
package knowledge
import (
"strings"
"unicode/utf8"
)
// Chunking: turning a document into the units retrieval ranks.
//
// The size is a retrieval decision, not a storage one. Too large and a chunk
// matches on a paragraph the reader does not want, then spends the model's
// context on the rest of the page; too small and the sentence that answers the
// question arrives without the sentence that gives it meaning — "this does not
// apply to agency staff" is worse than useless detached from what "this" is.
//
// Paragraph-first, because a document's own paragraph breaks are the author's
// judgement about what belongs together, and they are better than any window
// this code could pick. Windows are the fallback for text with no structure.
const (
// TargetChunkRunes is what a chunk aims for. Roughly 250 words, which sits
// inside every current embedding model's window with room to spare and is
// about the size of a section a person would quote.
TargetChunkRunes = 1400
// MaxChunkRunes is the hard cap. A paragraph longer than this is split.
MaxChunkRunes = 2200
// OverlapRunes is how much of the previous chunk a split one repeats.
//
// Overlap exists for the boundary problem: the answer to a question often
// straddles a break, and without overlap neither side retrieves well. The
// cost is duplicated text in the index and occasionally two near-identical
// results, which the fusion step deduplicates by document and ordinal.
OverlapRunes = 180
// MinChunkRunes is the floor. A fragment shorter than this — a heading on
// its own, a stray line — is folded into its neighbour rather than indexed,
// because it will match on a keyword and then say nothing.
MinChunkRunes = 80
)
// Chunk is one indexable unit.
type Chunk struct {
Ordinal int
Text string
// Heading is the trail of headings above this chunk — "Handbook ›
// Attendance › Lateness". Weighted above the body in the tsvector, and it
// is what makes a citation read like a location rather than a row id.
Heading string
TokenEstimate int
}
// Split turns a document into chunks.
//
// `title` seeds the heading trail, so every chunk carries at least the document
// it came from. Markdown ATX headings (`#`, `##`) update the trail as they are
// passed; anything else is body text.
func Split(title, body string) []Chunk {
paragraphs, headings := parse(title, body)
var (
chunks []Chunk
current strings.Builder
heading string
)
flush := func() {
text := strings.TrimSpace(current.String())
current.Reset()
if text == "" {
return
}
// Too short to stand alone: fold it into the previous chunk rather than
// index a fragment that matches and then says nothing.
if utf8.RuneCountInString(text) < MinChunkRunes && len(chunks) > 0 {
last := &chunks[len(chunks)-1]
last.Text += "\n\n" + text
last.TokenEstimate = estimateTokens(last.Text)
return
}
chunks = append(chunks, Chunk{
Ordinal: len(chunks), Text: text, Heading: heading,
TokenEstimate: estimateTokens(text),
})
}
for i, p := range paragraphs {
if h := headings[i]; h != "" {
// A new section starts a new chunk. Carrying text across a heading
// would put two topics in one unit and give it the wrong label.
flush()
heading = h
continue
}
// A paragraph over the cap is split on its own, with overlap.
if utf8.RuneCountInString(p) > MaxChunkRunes {
flush()
for _, piece := range window(p) {
chunks = append(chunks, Chunk{
Ordinal: len(chunks), Text: piece, Heading: heading,
TokenEstimate: estimateTokens(piece),
})
}
continue
}
if current.Len() > 0 && utf8.RuneCountInString(current.String())+utf8.RuneCountInString(p) > TargetChunkRunes {
flush()
}
if current.Len() > 0 {
current.WriteString("\n\n")
}
current.WriteString(p)
}
flush()
return chunks
}
// parse splits a body into paragraphs, tracking the heading trail.
//
// Returns paragraphs and, in step, the heading each one introduces — empty for
// ordinary text. Two parallel slices rather than a struct because the caller
// walks them together exactly once.
func parse(title, body string) (paragraphs []string, headings []string) {
trail := []string{}
if t := strings.TrimSpace(title); t != "" {
trail = append(trail, t)
}
for _, block := range strings.Split(strings.ReplaceAll(body, "\r\n", "\n"), "\n\n") {
block = strings.TrimSpace(block)
if block == "" {
continue
}
if level, text, ok := atxHeading(block); ok {
// Trim the trail to this heading's depth, then push. The document
// title is always element 0, so a level-1 heading sits at index 1.
depth := level
if depth > len(trail) {
depth = len(trail)
}
trail = append(trail[:depth], text)
paragraphs = append(paragraphs, block)
headings = append(headings, strings.Join(trail, " › "))
continue
}
paragraphs = append(paragraphs, block)
headings = append(headings, "")
}
return paragraphs, headings
}
// atxHeading recognises a markdown heading line.
//
// Stricter than "starts with a hash", and it has to be. `#3 on the rota is the
// closing shift` is prose, and treating it as a heading splits a paragraph
// mid-thought and mislabels every chunk after it — a mislabelled chunk then
// cites wrongly, which is the failure that survives longest because the text is
// right and only the attribution is wrong.
//
// Three conditions, all from CommonMark's ATX rule plus one of our own:
//
// - One to six hashes, followed by WHITESPACE. This is the condition that
// `#3` fails, and it is the one CommonMark actually specifies.
// - A single line. `# Something` followed by prose in the same block is prose
// that begins with a hash.
// - Short. A "heading" the length of a paragraph is a paragraph — the cap is
// ours, not the spec's, and it exists because a heading becomes a citation
// label and a 400-character label is unusable.
func atxHeading(block string) (level int, text string, ok bool) {
if strings.Contains(block, "\n") {
return 0, "", false
}
trimmed := strings.TrimLeft(block, "#")
level = len(block) - len(trimmed)
if level == 0 || level > 6 {
return 0, "", false
}
// CommonMark: the hashes must be followed by a space or the end of line.
if trimmed != "" && !strings.HasPrefix(trimmed, " ") && !strings.HasPrefix(trimmed, "\t") {
return 0, "", false
}
text = strings.TrimSpace(trimmed)
if text == "" {
return 0, "", false
}
if utf8.RuneCountInString(text) > MaxHeadingRunes {
return 0, "", false
}
return level, text, true
}
// MaxHeadingRunes is how long a heading may be before it is read as a
// paragraph. A heading becomes a citation label, and a label the length of a
// paragraph is not a label.
const MaxHeadingRunes = 120
// window splits an over-long paragraph into overlapping pieces.
//
// Break points prefer a sentence end near the target, then a space, then the
// raw offset. Cutting mid-word produces a token nothing matches and a citation
// that reads as though it were corrupted.
func window(p string) []string {
runes := []rune(p)
var out []string
for start := 0; start < len(runes); {
end := start + TargetChunkRunes
if end >= len(runes) {
out = append(out, strings.TrimSpace(string(runes[start:])))
break
}
end = breakNear(runes, start, end)
out = append(out, strings.TrimSpace(string(runes[start:end])))
next := end - OverlapRunes
if next <= start {
// Defensive: a pathological break point must not stall the loop.
next = end
}
start = next
}
return out
}
// breakNear finds a readable break at or before `end`.
func breakNear(runes []rune, start, end int) int {
const look = 220
floor := end - look
if floor <= start {
floor = start + 1
}
for i := end; i > floor; i-- {
switch runes[i-1] {
case '.', '!', '?', '\n':
return i
}
}
for i := end; i > floor; i-- {
if runes[i-1] == ' ' {
return i
}
}
return end
}
// estimateTokens is a rough token count.
//
// Four characters per token, the usual English approximation. Deliberately an
// estimate: it is used to budget how much context a retrieval may spend, and
// paying a tokeniser to be exact about a number that is then compared to a soft
// budget would be precision nobody spends.
func estimateTokens(s string) int {
n := utf8.RuneCountInString(s) / 4
if n < 1 {
return 1
}
return n
}