package definition import ( "strings" ) // The document layer: what is frontmatter, what is body, and what a `## Heading` // section contains. // // A port of the four exported readers in src/lib/skills/registry.js — // normalizeDefinition, hasFrontmatter, parseFrontmatter and the section // readers. Agent and skill definitions are read by the same code on the // frontend, deliberately, so that the two formats cannot drift; the same is // true here. // // The regular expressions the JavaScript uses are hand-rolled rather than // translated, because two of them rely on lookahead and lazy matching that RE2 // does not have. Each is written out below with the JavaScript it reproduces. // Normalize is the frontend's `normalizeDefinition`: a definition's text as the // parser needs to see it. // // Files arrive from editors, from Windows, from copy-paste and from downloads, // and four of the things they arrive with used to take the whole frontmatter // block down — a UTF-8 byte-order mark before the opening fence, blank lines // above it, CRLF endings, and trailing spaces after `---`. In each case the // fence did not match and the definition registered as untitled with no pages. // // This is NOT a lenient parser. The subset inside the fences is exactly as // strict as it was. This is only about recognising that a fence is a fence. // // String(raw ?? '') // .replace(/^\uFEFF/, '') // .replace(/\r\n?/g, '\n') // .replace(/^\s*\n+/, '') // // The order is load bearing: the BOM goes first so it cannot be counted as the // leading whitespace, and CR normalisation goes before the blank-line strip so // that a CRLF blank line is one. func Normalize(raw string) string { text := strings.TrimPrefix(raw, "\uFEFF") // `\r\n?` → `\n`: a CRLF pair and a lone CR both become one newline. if strings.IndexByte(text, '\r') >= 0 { var b strings.Builder b.Grow(len(text)) for i := 0; i < len(text); i++ { if text[i] != '\r' { b.WriteByte(text[i]) continue } b.WriteByte('\n') if i+1 < len(text) && text[i+1] == '\n' { i++ } } text = b.String() } // `^\s*\n+` → ``. Greedy `\s*` then at least one newline: the effect is to // drop the leading whitespace run up to and including its LAST newline, and // to drop nothing at all when that run contains no newline. A definition // indented by one space is therefore still unfenced, which is what the // editor decides too. end, last := 0, -1 for i, r := range text { if !jsIsSpace(r) { break } if r == '\n' { last = i } end = i + len(string(r)) } _ = end if last >= 0 { text = text[last+1:] } return text } // fence locates the frontmatter block in already-normalized text. // // /^---[ \t]*\n([\s\S]*?)\n---[ \t]*(?=\n|$)/ // // Returns the YAML source, the offset just past the closing fence, and whether // there was one. Lazy: the FIRST closing fence wins, which is why a definition // carrying a second `---` document keeps only the first and reads the rest as // body. func fence(text string) (yaml string, end int, ok bool) { if !strings.HasPrefix(text, "---") { return "", 0, false } i := 3 for i < len(text) && (text[i] == ' ' || text[i] == '\t') { i++ } if i >= len(text) || text[i] != '\n' { return "", 0, false } start := i + 1 for at := start - 1; at >= 0 && at < len(text); { nl := strings.IndexByte(text[at+1:], '\n') if nl < 0 { return "", 0, false } at = at + 1 + nl // index of the newline that must precede the fence rest := text[at+1:] if !strings.HasPrefix(rest, "---") { continue } j := 3 for j < len(rest) && (rest[j] == ' ' || rest[j] == '\t') { j++ } // `(?=\n|$)` — end of the document, or the end of this line. `$` has no // multiline flag on the frontend either, so it means end of document. if j < len(rest) && rest[j] != '\n' { continue } return text[start:at], at + 1 + j, true } return "", 0, false } // HasFrontmatter reports whether this text opens with a frontmatter block at // all. It does not say whether that block parses. func HasFrontmatter(raw string) bool { _, _, ok := fence(Normalize(raw)) return ok } // Document is a definition split into its two halves. type Document struct { // Data is the frontmatter as plain data. Always a mapping: a frontmatter // block that parses to a sequence is discarded, exactly as the frontend // discards it, because every reader downstream indexes it by key. Data map[string]any // Body is everything after the closing fence, trimmed. A document with no // frontmatter is all body. Body string // Fenced records whether a frontmatter block was found, which Data alone // cannot express — an empty fence pair and a missing one both give an // empty mapping. Fenced bool } // ParseFrontmatter splits a definition and reads its frontmatter. // // Returns a *Error when the YAML subset refuses a line. A document with no // recognisable fence is NOT an error: it is a document with no frontmatter, // and what happens to it is the validator's decision — the same division the // frontend makes. func ParseFrontmatter(raw string) (Document, error) { text := Normalize(raw) yaml, end, ok := fence(text) if !ok { return Document{Data: map[string]any{}, Body: text, Fenced: false}, nil } value, err := ParseYAML(yaml) if err != nil { return Document{}, err } data, _ := value.(map[string]any) if data == nil { data = map[string]any{} } return Document{Data: data, Body: jsTrim(text[end:]), Fenced: true}, nil } // sectionSource is the text under a `## Heading`, up to the next one. // // new RegExp(`##\\s+${escaped}\\s*\\n([\\s\\S]*?)(?=\\n##\\s|$)`, 'i') // // Case-insensitive, and deliberately not anchored to the start of a line — // that is what the frontend does. The heading is compared literally rather // than compiled into a pattern, which is the same protection the frontend gets // by escaping it: a heading containing regular-expression punctuation must // match the words it is built from. // // Returns ok=false for a section that is not there, which is a different thing // from a section that is there and empty. func sectionSource(body, heading string) (string, bool) { lower := strings.ToLower(body) want := strings.ToLower(heading) for at := 0; ; { h := strings.Index(lower[at:], "##") if h < 0 { return "", false } h += at at = h + 2 // `##` then `\s+` then the heading. i := h + 2 gap := i for i < len(body) { r, size := decodeRune(body[i:]) if !jsIsSpace(r) { break } i += size } if i == gap { continue // `\s+` needs at least one } if !strings.HasPrefix(lower[i:], want) { continue } i += len(want) // `\s*\n`: a whitespace run that ends in a newline. j, nl := i, -1 for j < len(body) { r, size := decodeRune(body[j:]) if !jsIsSpace(r) { break } if r == '\n' { nl = j break } j += size } if nl < 0 { continue } start := nl + 1 // `(?=\n##\s|$)`, lazily: the first following line that opens a new // `##` heading. `###` does not, because the character after `##` must // be whitespace. for k := start; ; { n := strings.Index(body[k:], "\n##") if n < 0 { return body[start:], true } n += k after := n + 3 if after < len(body) { r, _ := decodeRune(body[after:]) if jsIsSpace(r) { return body[start:n], true } } k = n + 1 } } } // decodeRune is utf8.DecodeRuneInString, kept local so the section reader has // one obvious way to step through the body. func decodeRune(s string) (rune, int) { for i, r := range s { _ = i return r, len(string(r)) } return 0, 0 } // SectionText is the prose under a `## Heading`, with its bullets and blank // lines flattened to one line. // // Used for a workforce level's description, which is a sentence rather than a // list. func SectionText(body, heading string) string { source, ok := sectionSource(body, heading) if !ok { return "" } parts := []string{} for _, l := range strings.Split(source, "\n") { l = jsTrim(stripListMarker(l)) if l == "" { continue } parts = append(parts, l) } return jsTrim(strings.Join(parts, " ")) } // stripListMarker removes a leading `-`, `*` or `1.` / `1)` bullet. // // /^\s*(?:[-*]|\d+[.)])\s+/ func stripListMarker(l string) string { i := 0 for i < len(l) { r, size := decodeRune(l[i:]) if !jsIsSpace(r) { break } i += size } marker := i switch { case i < len(l) && (l[i] == '-' || l[i] == '*'): i++ default: digits := i for i < len(l) && l[i] >= '0' && l[i] <= '9' { i++ } if i == digits || i >= len(l) || (l[i] != '.' && l[i] != ')') { return l } i++ } // `\s+` after the marker is required; without it there is no list item. space := i for i < len(l) { r, size := decodeRune(l[i:]) if !jsIsSpace(r) { break } i += size } if i == space { return l } _ = marker return l[i:] }