Files
krow_backend/go-api/internal/definition/frontmatter.go
2026-08-24 13:06:29 +05:30

333 lines
8.7 KiB
Go

package definition
import (
"strings"
)
// The document layer: what is frontmatter, what is body, and what a `## Heading`
// section contains.
//
// A port of the four exported readers in src/lib/skills/registry.js —
// normalizeDefinition, hasFrontmatter, parseFrontmatter and the section
// readers. Agent and skill definitions are read by the same code on the
// frontend, deliberately, so that the two formats cannot drift; the same is
// true here.
//
// The regular expressions the JavaScript uses are hand-rolled rather than
// translated, because two of them rely on lookahead and lazy matching that RE2
// does not have. Each is written out below with the JavaScript it reproduces.
// Normalize is the frontend's `normalizeDefinition`: a definition's text as the
// parser needs to see it.
//
// Files arrive from editors, from Windows, from copy-paste and from downloads,
// and four of the things they arrive with used to take the whole frontmatter
// block down — a UTF-8 byte-order mark before the opening fence, blank lines
// above it, CRLF endings, and trailing spaces after `---`. In each case the
// fence did not match and the definition registered as untitled with no pages.
//
// This is NOT a lenient parser. The subset inside the fences is exactly as
// strict as it was. This is only about recognising that a fence is a fence.
//
// String(raw ?? '')
// .replace(/^\uFEFF/, '')
// .replace(/\r\n?/g, '\n')
// .replace(/^\s*\n+/, '')
//
// The order is load bearing: the BOM goes first so it cannot be counted as the
// leading whitespace, and CR normalisation goes before the blank-line strip so
// that a CRLF blank line is one.
func Normalize(raw string) string {
text := strings.TrimPrefix(raw, "\uFEFF")
// `\r\n?` → `\n`: a CRLF pair and a lone CR both become one newline.
if strings.IndexByte(text, '\r') >= 0 {
var b strings.Builder
b.Grow(len(text))
for i := 0; i < len(text); i++ {
if text[i] != '\r' {
b.WriteByte(text[i])
continue
}
b.WriteByte('\n')
if i+1 < len(text) && text[i+1] == '\n' {
i++
}
}
text = b.String()
}
// `^\s*\n+` → ``. Greedy `\s*` then at least one newline: the effect is to
// drop the leading whitespace run up to and including its LAST newline, and
// to drop nothing at all when that run contains no newline. A definition
// indented by one space is therefore still unfenced, which is what the
// editor decides too.
end, last := 0, -1
for i, r := range text {
if !jsIsSpace(r) {
break
}
if r == '\n' {
last = i
}
end = i + len(string(r))
}
_ = end
if last >= 0 {
text = text[last+1:]
}
return text
}
// fence locates the frontmatter block in already-normalized text.
//
// /^---[ \t]*\n([\s\S]*?)\n---[ \t]*(?=\n|$)/
//
// Returns the YAML source, the offset just past the closing fence, and whether
// there was one. Lazy: the FIRST closing fence wins, which is why a definition
// carrying a second `---` document keeps only the first and reads the rest as
// body.
func fence(text string) (yaml string, end int, ok bool) {
if !strings.HasPrefix(text, "---") {
return "", 0, false
}
i := 3
for i < len(text) && (text[i] == ' ' || text[i] == '\t') {
i++
}
if i >= len(text) || text[i] != '\n' {
return "", 0, false
}
start := i + 1
for at := start - 1; at >= 0 && at < len(text); {
nl := strings.IndexByte(text[at+1:], '\n')
if nl < 0 {
return "", 0, false
}
at = at + 1 + nl // index of the newline that must precede the fence
rest := text[at+1:]
if !strings.HasPrefix(rest, "---") {
continue
}
j := 3
for j < len(rest) && (rest[j] == ' ' || rest[j] == '\t') {
j++
}
// `(?=\n|$)` — end of the document, or the end of this line. `$` has no
// multiline flag on the frontend either, so it means end of document.
if j < len(rest) && rest[j] != '\n' {
continue
}
return text[start:at], at + 1 + j, true
}
return "", 0, false
}
// HasFrontmatter reports whether this text opens with a frontmatter block at
// all. It does not say whether that block parses.
func HasFrontmatter(raw string) bool {
_, _, ok := fence(Normalize(raw))
return ok
}
// Document is a definition split into its two halves.
type Document struct {
// Data is the frontmatter as plain data. Always a mapping: a frontmatter
// block that parses to a sequence is discarded, exactly as the frontend
// discards it, because every reader downstream indexes it by key.
Data map[string]any
// Body is everything after the closing fence, trimmed. A document with no
// frontmatter is all body.
Body string
// Fenced records whether a frontmatter block was found, which Data alone
// cannot express — an empty fence pair and a missing one both give an
// empty mapping.
Fenced bool
}
// ParseFrontmatter splits a definition and reads its frontmatter.
//
// Returns a *Error when the YAML subset refuses a line. A document with no
// recognisable fence is NOT an error: it is a document with no frontmatter,
// and what happens to it is the validator's decision — the same division the
// frontend makes.
func ParseFrontmatter(raw string) (Document, error) {
text := Normalize(raw)
yaml, end, ok := fence(text)
if !ok {
return Document{Data: map[string]any{}, Body: text, Fenced: false}, nil
}
value, err := ParseYAML(yaml)
if err != nil {
return Document{}, err
}
data, _ := value.(map[string]any)
if data == nil {
data = map[string]any{}
}
return Document{Data: data, Body: jsTrim(text[end:]), Fenced: true}, nil
}
// sectionSource is the text under a `## Heading`, up to the next one.
//
// new RegExp(`##\\s+${escaped}\\s*\\n([\\s\\S]*?)(?=\\n##\\s|$)`, 'i')
//
// Case-insensitive, and deliberately not anchored to the start of a line —
// that is what the frontend does. The heading is compared literally rather
// than compiled into a pattern, which is the same protection the frontend gets
// by escaping it: a heading containing regular-expression punctuation must
// match the words it is built from.
//
// Returns ok=false for a section that is not there, which is a different thing
// from a section that is there and empty.
func sectionSource(body, heading string) (string, bool) {
lower := strings.ToLower(body)
want := strings.ToLower(heading)
for at := 0; ; {
h := strings.Index(lower[at:], "##")
if h < 0 {
return "", false
}
h += at
at = h + 2
// `##` then `\s+` then the heading.
i := h + 2
gap := i
for i < len(body) {
r, size := decodeRune(body[i:])
if !jsIsSpace(r) {
break
}
i += size
}
if i == gap {
continue // `\s+` needs at least one
}
if !strings.HasPrefix(lower[i:], want) {
continue
}
i += len(want)
// `\s*\n`: a whitespace run that ends in a newline.
j, nl := i, -1
for j < len(body) {
r, size := decodeRune(body[j:])
if !jsIsSpace(r) {
break
}
if r == '\n' {
nl = j
break
}
j += size
}
if nl < 0 {
continue
}
start := nl + 1
// `(?=\n##\s|$)`, lazily: the first following line that opens a new
// `##` heading. `###` does not, because the character after `##` must
// be whitespace.
for k := start; ; {
n := strings.Index(body[k:], "\n##")
if n < 0 {
return body[start:], true
}
n += k
after := n + 3
if after < len(body) {
r, _ := decodeRune(body[after:])
if jsIsSpace(r) {
return body[start:n], true
}
}
k = n + 1
}
}
}
// decodeRune is utf8.DecodeRuneInString, kept local so the section reader has
// one obvious way to step through the body.
func decodeRune(s string) (rune, int) {
for i, r := range s {
_ = i
return r, len(string(r))
}
return 0, 0
}
// SectionText is the prose under a `## Heading`, with its bullets and blank
// lines flattened to one line.
//
// Used for a workforce level's description, which is a sentence rather than a
// list.
func SectionText(body, heading string) string {
source, ok := sectionSource(body, heading)
if !ok {
return ""
}
parts := []string{}
for _, l := range strings.Split(source, "\n") {
l = jsTrim(stripListMarker(l))
if l == "" {
continue
}
parts = append(parts, l)
}
return jsTrim(strings.Join(parts, " "))
}
// stripListMarker removes a leading `-`, `*` or `1.` / `1)` bullet.
//
// /^\s*(?:[-*]|\d+[.)])\s+/
func stripListMarker(l string) string {
i := 0
for i < len(l) {
r, size := decodeRune(l[i:])
if !jsIsSpace(r) {
break
}
i += size
}
marker := i
switch {
case i < len(l) && (l[i] == '-' || l[i] == '*'):
i++
default:
digits := i
for i < len(l) && l[i] >= '0' && l[i] <= '9' {
i++
}
if i == digits || i >= len(l) || (l[i] != '.' && l[i] != ')') {
return l
}
i++
}
// `\s+` after the marker is required; without it there is no list item.
space := i
for i < len(l) {
r, size := decodeRune(l[i:])
if !jsIsSpace(r) {
break
}
i += size
}
if i == space {
return l
}
_ = marker
return l[i:]
}