agent build

This commit is contained in:
2026-08-28 12:21:44 +05:30
parent b6f8655909
commit f7df96c973
138 changed files with 24164 additions and 207 deletions

View File

@@ -0,0 +1,266 @@
package knowledge
import (
"strings"
"unicode/utf8"
)
// Chunking: turning a document into the units retrieval ranks.
//
// The size is a retrieval decision, not a storage one. Too large and a chunk
// matches on a paragraph the reader does not want, then spends the model's
// context on the rest of the page; too small and the sentence that answers the
// question arrives without the sentence that gives it meaning — "this does not
// apply to agency staff" is worse than useless detached from what "this" is.
//
// Paragraph-first, because a document's own paragraph breaks are the author's
// judgement about what belongs together, and they are better than any window
// this code could pick. Windows are the fallback for text with no structure.
const (
// TargetChunkRunes is what a chunk aims for. Roughly 250 words, which sits
// inside every current embedding model's window with room to spare and is
// about the size of a section a person would quote.
TargetChunkRunes = 1400
// MaxChunkRunes is the hard cap. A paragraph longer than this is split.
MaxChunkRunes = 2200
// OverlapRunes is how much of the previous chunk a split one repeats.
//
// Overlap exists for the boundary problem: the answer to a question often
// straddles a break, and without overlap neither side retrieves well. The
// cost is duplicated text in the index and occasionally two near-identical
// results, which the fusion step deduplicates by document and ordinal.
OverlapRunes = 180
// MinChunkRunes is the floor. A fragment shorter than this — a heading on
// its own, a stray line — is folded into its neighbour rather than indexed,
// because it will match on a keyword and then say nothing.
MinChunkRunes = 80
)
// Chunk is one indexable unit.
type Chunk struct {
Ordinal int
Text string
// Heading is the trail of headings above this chunk — "Handbook ›
// Attendance › Lateness". Weighted above the body in the tsvector, and it
// is what makes a citation read like a location rather than a row id.
Heading string
TokenEstimate int
}
// Split turns a document into chunks.
//
// `title` seeds the heading trail, so every chunk carries at least the document
// it came from. Markdown ATX headings (`#`, `##`) update the trail as they are
// passed; anything else is body text.
func Split(title, body string) []Chunk {
paragraphs, headings := parse(title, body)
var (
chunks []Chunk
current strings.Builder
heading string
)
flush := func() {
text := strings.TrimSpace(current.String())
current.Reset()
if text == "" {
return
}
// Too short to stand alone: fold it into the previous chunk rather than
// index a fragment that matches and then says nothing.
if utf8.RuneCountInString(text) < MinChunkRunes && len(chunks) > 0 {
last := &chunks[len(chunks)-1]
last.Text += "\n\n" + text
last.TokenEstimate = estimateTokens(last.Text)
return
}
chunks = append(chunks, Chunk{
Ordinal: len(chunks), Text: text, Heading: heading,
TokenEstimate: estimateTokens(text),
})
}
for i, p := range paragraphs {
if h := headings[i]; h != "" {
// A new section starts a new chunk. Carrying text across a heading
// would put two topics in one unit and give it the wrong label.
flush()
heading = h
continue
}
// A paragraph over the cap is split on its own, with overlap.
if utf8.RuneCountInString(p) > MaxChunkRunes {
flush()
for _, piece := range window(p) {
chunks = append(chunks, Chunk{
Ordinal: len(chunks), Text: piece, Heading: heading,
TokenEstimate: estimateTokens(piece),
})
}
continue
}
if current.Len() > 0 && utf8.RuneCountInString(current.String())+utf8.RuneCountInString(p) > TargetChunkRunes {
flush()
}
if current.Len() > 0 {
current.WriteString("\n\n")
}
current.WriteString(p)
}
flush()
return chunks
}
// parse splits a body into paragraphs, tracking the heading trail.
//
// Returns paragraphs and, in step, the heading each one introduces — empty for
// ordinary text. Two parallel slices rather than a struct because the caller
// walks them together exactly once.
func parse(title, body string) (paragraphs []string, headings []string) {
trail := []string{}
if t := strings.TrimSpace(title); t != "" {
trail = append(trail, t)
}
for _, block := range strings.Split(strings.ReplaceAll(body, "\r\n", "\n"), "\n\n") {
block = strings.TrimSpace(block)
if block == "" {
continue
}
if level, text, ok := atxHeading(block); ok {
// Trim the trail to this heading's depth, then push. The document
// title is always element 0, so a level-1 heading sits at index 1.
depth := level
if depth > len(trail) {
depth = len(trail)
}
trail = append(trail[:depth], text)
paragraphs = append(paragraphs, block)
headings = append(headings, strings.Join(trail, " › "))
continue
}
paragraphs = append(paragraphs, block)
headings = append(headings, "")
}
return paragraphs, headings
}
// atxHeading recognises a markdown heading line.
//
// Stricter than "starts with a hash", and it has to be. `#3 on the rota is the
// closing shift` is prose, and treating it as a heading splits a paragraph
// mid-thought and mislabels every chunk after it — a mislabelled chunk then
// cites wrongly, which is the failure that survives longest because the text is
// right and only the attribution is wrong.
//
// Three conditions, all from CommonMark's ATX rule plus one of our own:
//
// - One to six hashes, followed by WHITESPACE. This is the condition that
// `#3` fails, and it is the one CommonMark actually specifies.
// - A single line. `# Something` followed by prose in the same block is prose
// that begins with a hash.
// - Short. A "heading" the length of a paragraph is a paragraph — the cap is
// ours, not the spec's, and it exists because a heading becomes a citation
// label and a 400-character label is unusable.
func atxHeading(block string) (level int, text string, ok bool) {
if strings.Contains(block, "\n") {
return 0, "", false
}
trimmed := strings.TrimLeft(block, "#")
level = len(block) - len(trimmed)
if level == 0 || level > 6 {
return 0, "", false
}
// CommonMark: the hashes must be followed by a space or the end of line.
if trimmed != "" && !strings.HasPrefix(trimmed, " ") && !strings.HasPrefix(trimmed, "\t") {
return 0, "", false
}
text = strings.TrimSpace(trimmed)
if text == "" {
return 0, "", false
}
if utf8.RuneCountInString(text) > MaxHeadingRunes {
return 0, "", false
}
return level, text, true
}
// MaxHeadingRunes is how long a heading may be before it is read as a
// paragraph. A heading becomes a citation label, and a label the length of a
// paragraph is not a label.
const MaxHeadingRunes = 120
// window splits an over-long paragraph into overlapping pieces.
//
// Break points prefer a sentence end near the target, then a space, then the
// raw offset. Cutting mid-word produces a token nothing matches and a citation
// that reads as though it were corrupted.
func window(p string) []string {
runes := []rune(p)
var out []string
for start := 0; start < len(runes); {
end := start + TargetChunkRunes
if end >= len(runes) {
out = append(out, strings.TrimSpace(string(runes[start:])))
break
}
end = breakNear(runes, start, end)
out = append(out, strings.TrimSpace(string(runes[start:end])))
next := end - OverlapRunes
if next <= start {
// Defensive: a pathological break point must not stall the loop.
next = end
}
start = next
}
return out
}
// breakNear finds a readable break at or before `end`.
func breakNear(runes []rune, start, end int) int {
const look = 220
floor := end - look
if floor <= start {
floor = start + 1
}
for i := end; i > floor; i-- {
switch runes[i-1] {
case '.', '!', '?', '\n':
return i
}
}
for i := end; i > floor; i-- {
if runes[i-1] == ' ' {
return i
}
}
return end
}
// estimateTokens is a rough token count.
//
// Four characters per token, the usual English approximation. Deliberately an
// estimate: it is used to budget how much context a retrieval may spend, and
// paying a tokeniser to be exact about a number that is then compared to a soft
// budget would be precision nobody spends.
func estimateTokens(s string) int {
n := utf8.RuneCountInString(s) / 4
if n < 1 {
return 1
}
return n
}