333 lines
8.7 KiB
Go
333 lines
8.7 KiB
Go
package definition
|
|
|
|
import (
|
|
"strings"
|
|
)
|
|
|
|
// The document layer: what is frontmatter, what is body, and what a `## Heading`
|
|
// section contains.
|
|
//
|
|
// A port of the four exported readers in src/lib/skills/registry.js —
|
|
// normalizeDefinition, hasFrontmatter, parseFrontmatter and the section
|
|
// readers. Agent and skill definitions are read by the same code on the
|
|
// frontend, deliberately, so that the two formats cannot drift; the same is
|
|
// true here.
|
|
//
|
|
// The regular expressions the JavaScript uses are hand-rolled rather than
|
|
// translated, because two of them rely on lookahead and lazy matching that RE2
|
|
// does not have. Each is written out below with the JavaScript it reproduces.
|
|
|
|
// Normalize is the frontend's `normalizeDefinition`: a definition's text as the
|
|
// parser needs to see it.
|
|
//
|
|
// Files arrive from editors, from Windows, from copy-paste and from downloads,
|
|
// and four of the things they arrive with used to take the whole frontmatter
|
|
// block down — a UTF-8 byte-order mark before the opening fence, blank lines
|
|
// above it, CRLF endings, and trailing spaces after `---`. In each case the
|
|
// fence did not match and the definition registered as untitled with no pages.
|
|
//
|
|
// This is NOT a lenient parser. The subset inside the fences is exactly as
|
|
// strict as it was. This is only about recognising that a fence is a fence.
|
|
//
|
|
// String(raw ?? '')
|
|
// .replace(/^\uFEFF/, '')
|
|
// .replace(/\r\n?/g, '\n')
|
|
// .replace(/^\s*\n+/, '')
|
|
//
|
|
// The order is load bearing: the BOM goes first so it cannot be counted as the
|
|
// leading whitespace, and CR normalisation goes before the blank-line strip so
|
|
// that a CRLF blank line is one.
|
|
func Normalize(raw string) string {
|
|
text := strings.TrimPrefix(raw, "\uFEFF")
|
|
|
|
// `\r\n?` → `\n`: a CRLF pair and a lone CR both become one newline.
|
|
if strings.IndexByte(text, '\r') >= 0 {
|
|
var b strings.Builder
|
|
b.Grow(len(text))
|
|
for i := 0; i < len(text); i++ {
|
|
if text[i] != '\r' {
|
|
b.WriteByte(text[i])
|
|
continue
|
|
}
|
|
b.WriteByte('\n')
|
|
if i+1 < len(text) && text[i+1] == '\n' {
|
|
i++
|
|
}
|
|
}
|
|
text = b.String()
|
|
}
|
|
|
|
// `^\s*\n+` → ``. Greedy `\s*` then at least one newline: the effect is to
|
|
// drop the leading whitespace run up to and including its LAST newline, and
|
|
// to drop nothing at all when that run contains no newline. A definition
|
|
// indented by one space is therefore still unfenced, which is what the
|
|
// editor decides too.
|
|
end, last := 0, -1
|
|
for i, r := range text {
|
|
if !jsIsSpace(r) {
|
|
break
|
|
}
|
|
if r == '\n' {
|
|
last = i
|
|
}
|
|
end = i + len(string(r))
|
|
}
|
|
_ = end
|
|
if last >= 0 {
|
|
text = text[last+1:]
|
|
}
|
|
|
|
return text
|
|
}
|
|
|
|
// fence locates the frontmatter block in already-normalized text.
|
|
//
|
|
// /^---[ \t]*\n([\s\S]*?)\n---[ \t]*(?=\n|$)/
|
|
//
|
|
// Returns the YAML source, the offset just past the closing fence, and whether
|
|
// there was one. Lazy: the FIRST closing fence wins, which is why a definition
|
|
// carrying a second `---` document keeps only the first and reads the rest as
|
|
// body.
|
|
func fence(text string) (yaml string, end int, ok bool) {
|
|
if !strings.HasPrefix(text, "---") {
|
|
return "", 0, false
|
|
}
|
|
i := 3
|
|
for i < len(text) && (text[i] == ' ' || text[i] == '\t') {
|
|
i++
|
|
}
|
|
if i >= len(text) || text[i] != '\n' {
|
|
return "", 0, false
|
|
}
|
|
start := i + 1
|
|
|
|
for at := start - 1; at >= 0 && at < len(text); {
|
|
nl := strings.IndexByte(text[at+1:], '\n')
|
|
if nl < 0 {
|
|
return "", 0, false
|
|
}
|
|
at = at + 1 + nl // index of the newline that must precede the fence
|
|
|
|
rest := text[at+1:]
|
|
if !strings.HasPrefix(rest, "---") {
|
|
continue
|
|
}
|
|
j := 3
|
|
for j < len(rest) && (rest[j] == ' ' || rest[j] == '\t') {
|
|
j++
|
|
}
|
|
// `(?=\n|$)` — end of the document, or the end of this line. `$` has no
|
|
// multiline flag on the frontend either, so it means end of document.
|
|
if j < len(rest) && rest[j] != '\n' {
|
|
continue
|
|
}
|
|
return text[start:at], at + 1 + j, true
|
|
}
|
|
return "", 0, false
|
|
}
|
|
|
|
// HasFrontmatter reports whether this text opens with a frontmatter block at
|
|
// all. It does not say whether that block parses.
|
|
func HasFrontmatter(raw string) bool {
|
|
_, _, ok := fence(Normalize(raw))
|
|
return ok
|
|
}
|
|
|
|
// Document is a definition split into its two halves.
|
|
type Document struct {
|
|
// Data is the frontmatter as plain data. Always a mapping: a frontmatter
|
|
// block that parses to a sequence is discarded, exactly as the frontend
|
|
// discards it, because every reader downstream indexes it by key.
|
|
Data map[string]any
|
|
|
|
// Body is everything after the closing fence, trimmed. A document with no
|
|
// frontmatter is all body.
|
|
Body string
|
|
|
|
// Fenced records whether a frontmatter block was found, which Data alone
|
|
// cannot express — an empty fence pair and a missing one both give an
|
|
// empty mapping.
|
|
Fenced bool
|
|
}
|
|
|
|
// ParseFrontmatter splits a definition and reads its frontmatter.
|
|
//
|
|
// Returns a *Error when the YAML subset refuses a line. A document with no
|
|
// recognisable fence is NOT an error: it is a document with no frontmatter,
|
|
// and what happens to it is the validator's decision — the same division the
|
|
// frontend makes.
|
|
func ParseFrontmatter(raw string) (Document, error) {
|
|
text := Normalize(raw)
|
|
|
|
yaml, end, ok := fence(text)
|
|
if !ok {
|
|
return Document{Data: map[string]any{}, Body: text, Fenced: false}, nil
|
|
}
|
|
|
|
value, err := ParseYAML(yaml)
|
|
if err != nil {
|
|
return Document{}, err
|
|
}
|
|
|
|
data, _ := value.(map[string]any)
|
|
if data == nil {
|
|
data = map[string]any{}
|
|
}
|
|
|
|
return Document{Data: data, Body: jsTrim(text[end:]), Fenced: true}, nil
|
|
}
|
|
|
|
// sectionSource is the text under a `## Heading`, up to the next one.
|
|
//
|
|
// new RegExp(`##\\s+${escaped}\\s*\\n([\\s\\S]*?)(?=\\n##\\s|$)`, 'i')
|
|
//
|
|
// Case-insensitive, and deliberately not anchored to the start of a line —
|
|
// that is what the frontend does. The heading is compared literally rather
|
|
// than compiled into a pattern, which is the same protection the frontend gets
|
|
// by escaping it: a heading containing regular-expression punctuation must
|
|
// match the words it is built from.
|
|
//
|
|
// Returns ok=false for a section that is not there, which is a different thing
|
|
// from a section that is there and empty.
|
|
func sectionSource(body, heading string) (string, bool) {
|
|
lower := strings.ToLower(body)
|
|
want := strings.ToLower(heading)
|
|
|
|
for at := 0; ; {
|
|
h := strings.Index(lower[at:], "##")
|
|
if h < 0 {
|
|
return "", false
|
|
}
|
|
h += at
|
|
at = h + 2
|
|
|
|
// `##` then `\s+` then the heading.
|
|
i := h + 2
|
|
gap := i
|
|
for i < len(body) {
|
|
r, size := decodeRune(body[i:])
|
|
if !jsIsSpace(r) {
|
|
break
|
|
}
|
|
i += size
|
|
}
|
|
if i == gap {
|
|
continue // `\s+` needs at least one
|
|
}
|
|
if !strings.HasPrefix(lower[i:], want) {
|
|
continue
|
|
}
|
|
i += len(want)
|
|
|
|
// `\s*\n`: a whitespace run that ends in a newline.
|
|
j, nl := i, -1
|
|
for j < len(body) {
|
|
r, size := decodeRune(body[j:])
|
|
if !jsIsSpace(r) {
|
|
break
|
|
}
|
|
if r == '\n' {
|
|
nl = j
|
|
break
|
|
}
|
|
j += size
|
|
}
|
|
if nl < 0 {
|
|
continue
|
|
}
|
|
|
|
start := nl + 1
|
|
// `(?=\n##\s|$)`, lazily: the first following line that opens a new
|
|
// `##` heading. `###` does not, because the character after `##` must
|
|
// be whitespace.
|
|
for k := start; ; {
|
|
n := strings.Index(body[k:], "\n##")
|
|
if n < 0 {
|
|
return body[start:], true
|
|
}
|
|
n += k
|
|
after := n + 3
|
|
if after < len(body) {
|
|
r, _ := decodeRune(body[after:])
|
|
if jsIsSpace(r) {
|
|
return body[start:n], true
|
|
}
|
|
}
|
|
k = n + 1
|
|
}
|
|
}
|
|
}
|
|
|
|
// decodeRune is utf8.DecodeRuneInString, kept local so the section reader has
|
|
// one obvious way to step through the body.
|
|
func decodeRune(s string) (rune, int) {
|
|
for i, r := range s {
|
|
_ = i
|
|
return r, len(string(r))
|
|
}
|
|
return 0, 0
|
|
}
|
|
|
|
// SectionText is the prose under a `## Heading`, with its bullets and blank
|
|
// lines flattened to one line.
|
|
//
|
|
// Used for a workforce level's description, which is a sentence rather than a
|
|
// list.
|
|
func SectionText(body, heading string) string {
|
|
source, ok := sectionSource(body, heading)
|
|
if !ok {
|
|
return ""
|
|
}
|
|
parts := []string{}
|
|
for _, l := range strings.Split(source, "\n") {
|
|
l = jsTrim(stripListMarker(l))
|
|
if l == "" {
|
|
continue
|
|
}
|
|
parts = append(parts, l)
|
|
}
|
|
return jsTrim(strings.Join(parts, " "))
|
|
}
|
|
|
|
// stripListMarker removes a leading `-`, `*` or `1.` / `1)` bullet.
|
|
//
|
|
// /^\s*(?:[-*]|\d+[.)])\s+/
|
|
func stripListMarker(l string) string {
|
|
i := 0
|
|
for i < len(l) {
|
|
r, size := decodeRune(l[i:])
|
|
if !jsIsSpace(r) {
|
|
break
|
|
}
|
|
i += size
|
|
}
|
|
marker := i
|
|
switch {
|
|
case i < len(l) && (l[i] == '-' || l[i] == '*'):
|
|
i++
|
|
default:
|
|
digits := i
|
|
for i < len(l) && l[i] >= '0' && l[i] <= '9' {
|
|
i++
|
|
}
|
|
if i == digits || i >= len(l) || (l[i] != '.' && l[i] != ')') {
|
|
return l
|
|
}
|
|
i++
|
|
}
|
|
// `\s+` after the marker is required; without it there is no list item.
|
|
space := i
|
|
for i < len(l) {
|
|
r, size := decodeRune(l[i:])
|
|
if !jsIsSpace(r) {
|
|
break
|
|
}
|
|
i += size
|
|
}
|
|
if i == space {
|
|
return l
|
|
}
|
|
_ = marker
|
|
return l[i:]
|
|
}
|