agent build
This commit is contained in:
266
go-api/internal/knowledge/chunk.go
Normal file
266
go-api/internal/knowledge/chunk.go
Normal file
@@ -0,0 +1,266 @@
|
||||
package knowledge
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"unicode/utf8"
|
||||
)
|
||||
|
||||
// Chunking: turning a document into the units retrieval ranks.
|
||||
//
|
||||
// The size is a retrieval decision, not a storage one. Too large and a chunk
|
||||
// matches on a paragraph the reader does not want, then spends the model's
|
||||
// context on the rest of the page; too small and the sentence that answers the
|
||||
// question arrives without the sentence that gives it meaning — "this does not
|
||||
// apply to agency staff" is worse than useless detached from what "this" is.
|
||||
//
|
||||
// Paragraph-first, because a document's own paragraph breaks are the author's
|
||||
// judgement about what belongs together, and they are better than any window
|
||||
// this code could pick. Windows are the fallback for text with no structure.
|
||||
|
||||
const (
|
||||
// TargetChunkRunes is what a chunk aims for. Roughly 250 words, which sits
|
||||
// inside every current embedding model's window with room to spare and is
|
||||
// about the size of a section a person would quote.
|
||||
TargetChunkRunes = 1400
|
||||
|
||||
// MaxChunkRunes is the hard cap. A paragraph longer than this is split.
|
||||
MaxChunkRunes = 2200
|
||||
|
||||
// OverlapRunes is how much of the previous chunk a split one repeats.
|
||||
//
|
||||
// Overlap exists for the boundary problem: the answer to a question often
|
||||
// straddles a break, and without overlap neither side retrieves well. The
|
||||
// cost is duplicated text in the index and occasionally two near-identical
|
||||
// results, which the fusion step deduplicates by document and ordinal.
|
||||
OverlapRunes = 180
|
||||
|
||||
// MinChunkRunes is the floor. A fragment shorter than this — a heading on
|
||||
// its own, a stray line — is folded into its neighbour rather than indexed,
|
||||
// because it will match on a keyword and then say nothing.
|
||||
MinChunkRunes = 80
|
||||
)
|
||||
|
||||
// Chunk is one indexable unit.
|
||||
type Chunk struct {
|
||||
Ordinal int
|
||||
Text string
|
||||
|
||||
// Heading is the trail of headings above this chunk — "Handbook ›
|
||||
// Attendance › Lateness". Weighted above the body in the tsvector, and it
|
||||
// is what makes a citation read like a location rather than a row id.
|
||||
Heading string
|
||||
|
||||
TokenEstimate int
|
||||
}
|
||||
|
||||
// Split turns a document into chunks.
|
||||
//
|
||||
// `title` seeds the heading trail, so every chunk carries at least the document
|
||||
// it came from. Markdown ATX headings (`#`, `##`) update the trail as they are
|
||||
// passed; anything else is body text.
|
||||
func Split(title, body string) []Chunk {
|
||||
paragraphs, headings := parse(title, body)
|
||||
|
||||
var (
|
||||
chunks []Chunk
|
||||
current strings.Builder
|
||||
heading string
|
||||
)
|
||||
|
||||
flush := func() {
|
||||
text := strings.TrimSpace(current.String())
|
||||
current.Reset()
|
||||
if text == "" {
|
||||
return
|
||||
}
|
||||
// Too short to stand alone: fold it into the previous chunk rather than
|
||||
// index a fragment that matches and then says nothing.
|
||||
if utf8.RuneCountInString(text) < MinChunkRunes && len(chunks) > 0 {
|
||||
last := &chunks[len(chunks)-1]
|
||||
last.Text += "\n\n" + text
|
||||
last.TokenEstimate = estimateTokens(last.Text)
|
||||
return
|
||||
}
|
||||
chunks = append(chunks, Chunk{
|
||||
Ordinal: len(chunks), Text: text, Heading: heading,
|
||||
TokenEstimate: estimateTokens(text),
|
||||
})
|
||||
}
|
||||
|
||||
for i, p := range paragraphs {
|
||||
if h := headings[i]; h != "" {
|
||||
// A new section starts a new chunk. Carrying text across a heading
|
||||
// would put two topics in one unit and give it the wrong label.
|
||||
flush()
|
||||
heading = h
|
||||
continue
|
||||
}
|
||||
|
||||
// A paragraph over the cap is split on its own, with overlap.
|
||||
if utf8.RuneCountInString(p) > MaxChunkRunes {
|
||||
flush()
|
||||
for _, piece := range window(p) {
|
||||
chunks = append(chunks, Chunk{
|
||||
Ordinal: len(chunks), Text: piece, Heading: heading,
|
||||
TokenEstimate: estimateTokens(piece),
|
||||
})
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
if current.Len() > 0 && utf8.RuneCountInString(current.String())+utf8.RuneCountInString(p) > TargetChunkRunes {
|
||||
flush()
|
||||
}
|
||||
if current.Len() > 0 {
|
||||
current.WriteString("\n\n")
|
||||
}
|
||||
current.WriteString(p)
|
||||
}
|
||||
flush()
|
||||
|
||||
return chunks
|
||||
}
|
||||
|
||||
// parse splits a body into paragraphs, tracking the heading trail.
|
||||
//
|
||||
// Returns paragraphs and, in step, the heading each one introduces — empty for
|
||||
// ordinary text. Two parallel slices rather than a struct because the caller
|
||||
// walks them together exactly once.
|
||||
func parse(title, body string) (paragraphs []string, headings []string) {
|
||||
trail := []string{}
|
||||
if t := strings.TrimSpace(title); t != "" {
|
||||
trail = append(trail, t)
|
||||
}
|
||||
|
||||
for _, block := range strings.Split(strings.ReplaceAll(body, "\r\n", "\n"), "\n\n") {
|
||||
block = strings.TrimSpace(block)
|
||||
if block == "" {
|
||||
continue
|
||||
}
|
||||
|
||||
if level, text, ok := atxHeading(block); ok {
|
||||
// Trim the trail to this heading's depth, then push. The document
|
||||
// title is always element 0, so a level-1 heading sits at index 1.
|
||||
depth := level
|
||||
if depth > len(trail) {
|
||||
depth = len(trail)
|
||||
}
|
||||
trail = append(trail[:depth], text)
|
||||
paragraphs = append(paragraphs, block)
|
||||
headings = append(headings, strings.Join(trail, " › "))
|
||||
continue
|
||||
}
|
||||
|
||||
paragraphs = append(paragraphs, block)
|
||||
headings = append(headings, "")
|
||||
}
|
||||
return paragraphs, headings
|
||||
}
|
||||
|
||||
// atxHeading recognises a markdown heading line.
|
||||
//
|
||||
// Stricter than "starts with a hash", and it has to be. `#3 on the rota is the
|
||||
// closing shift` is prose, and treating it as a heading splits a paragraph
|
||||
// mid-thought and mislabels every chunk after it — a mislabelled chunk then
|
||||
// cites wrongly, which is the failure that survives longest because the text is
|
||||
// right and only the attribution is wrong.
|
||||
//
|
||||
// Three conditions, all from CommonMark's ATX rule plus one of our own:
|
||||
//
|
||||
// - One to six hashes, followed by WHITESPACE. This is the condition that
|
||||
// `#3` fails, and it is the one CommonMark actually specifies.
|
||||
// - A single line. `# Something` followed by prose in the same block is prose
|
||||
// that begins with a hash.
|
||||
// - Short. A "heading" the length of a paragraph is a paragraph — the cap is
|
||||
// ours, not the spec's, and it exists because a heading becomes a citation
|
||||
// label and a 400-character label is unusable.
|
||||
func atxHeading(block string) (level int, text string, ok bool) {
|
||||
if strings.Contains(block, "\n") {
|
||||
return 0, "", false
|
||||
}
|
||||
trimmed := strings.TrimLeft(block, "#")
|
||||
level = len(block) - len(trimmed)
|
||||
if level == 0 || level > 6 {
|
||||
return 0, "", false
|
||||
}
|
||||
// CommonMark: the hashes must be followed by a space or the end of line.
|
||||
if trimmed != "" && !strings.HasPrefix(trimmed, " ") && !strings.HasPrefix(trimmed, "\t") {
|
||||
return 0, "", false
|
||||
}
|
||||
text = strings.TrimSpace(trimmed)
|
||||
if text == "" {
|
||||
return 0, "", false
|
||||
}
|
||||
if utf8.RuneCountInString(text) > MaxHeadingRunes {
|
||||
return 0, "", false
|
||||
}
|
||||
return level, text, true
|
||||
}
|
||||
|
||||
// MaxHeadingRunes is how long a heading may be before it is read as a
|
||||
// paragraph. A heading becomes a citation label, and a label the length of a
|
||||
// paragraph is not a label.
|
||||
const MaxHeadingRunes = 120
|
||||
|
||||
// window splits an over-long paragraph into overlapping pieces.
|
||||
//
|
||||
// Break points prefer a sentence end near the target, then a space, then the
|
||||
// raw offset. Cutting mid-word produces a token nothing matches and a citation
|
||||
// that reads as though it were corrupted.
|
||||
func window(p string) []string {
|
||||
runes := []rune(p)
|
||||
var out []string
|
||||
|
||||
for start := 0; start < len(runes); {
|
||||
end := start + TargetChunkRunes
|
||||
if end >= len(runes) {
|
||||
out = append(out, strings.TrimSpace(string(runes[start:])))
|
||||
break
|
||||
}
|
||||
end = breakNear(runes, start, end)
|
||||
out = append(out, strings.TrimSpace(string(runes[start:end])))
|
||||
|
||||
next := end - OverlapRunes
|
||||
if next <= start {
|
||||
// Defensive: a pathological break point must not stall the loop.
|
||||
next = end
|
||||
}
|
||||
start = next
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// breakNear finds a readable break at or before `end`.
|
||||
func breakNear(runes []rune, start, end int) int {
|
||||
const look = 220
|
||||
floor := end - look
|
||||
if floor <= start {
|
||||
floor = start + 1
|
||||
}
|
||||
for i := end; i > floor; i-- {
|
||||
switch runes[i-1] {
|
||||
case '.', '!', '?', '\n':
|
||||
return i
|
||||
}
|
||||
}
|
||||
for i := end; i > floor; i-- {
|
||||
if runes[i-1] == ' ' {
|
||||
return i
|
||||
}
|
||||
}
|
||||
return end
|
||||
}
|
||||
|
||||
// estimateTokens is a rough token count.
|
||||
//
|
||||
// Four characters per token, the usual English approximation. Deliberately an
|
||||
// estimate: it is used to budget how much context a retrieval may spend, and
|
||||
// paying a tokeniser to be exact about a number that is then compared to a soft
|
||||
// budget would be precision nobody spends.
|
||||
func estimateTokens(s string) int {
|
||||
n := utf8.RuneCountInString(s) / 4
|
||||
if n < 1 {
|
||||
return 1
|
||||
}
|
||||
return n
|
||||
}
|
||||
Reference in New Issue
Block a user