first commit
This commit is contained in:
332
go-api/internal/definition/frontmatter.go
Normal file
332
go-api/internal/definition/frontmatter.go
Normal file
@@ -0,0 +1,332 @@
|
||||
package definition
|
||||
|
||||
import (
|
||||
"strings"
|
||||
)
|
||||
|
||||
// The document layer: what is frontmatter, what is body, and what a `## Heading`
|
||||
// section contains.
|
||||
//
|
||||
// A port of the four exported readers in src/lib/skills/registry.js —
|
||||
// normalizeDefinition, hasFrontmatter, parseFrontmatter and the section
|
||||
// readers. Agent and skill definitions are read by the same code on the
|
||||
// frontend, deliberately, so that the two formats cannot drift; the same is
|
||||
// true here.
|
||||
//
|
||||
// The regular expressions the JavaScript uses are hand-rolled rather than
|
||||
// translated, because two of them rely on lookahead and lazy matching that RE2
|
||||
// does not have. Each is written out below with the JavaScript it reproduces.
|
||||
|
||||
// Normalize is the frontend's `normalizeDefinition`: a definition's text as the
|
||||
// parser needs to see it.
|
||||
//
|
||||
// Files arrive from editors, from Windows, from copy-paste and from downloads,
|
||||
// and four of the things they arrive with used to take the whole frontmatter
|
||||
// block down — a UTF-8 byte-order mark before the opening fence, blank lines
|
||||
// above it, CRLF endings, and trailing spaces after `---`. In each case the
|
||||
// fence did not match and the definition registered as untitled with no pages.
|
||||
//
|
||||
// This is NOT a lenient parser. The subset inside the fences is exactly as
|
||||
// strict as it was. This is only about recognising that a fence is a fence.
|
||||
//
|
||||
// String(raw ?? '')
|
||||
// .replace(/^\uFEFF/, '')
|
||||
// .replace(/\r\n?/g, '\n')
|
||||
// .replace(/^\s*\n+/, '')
|
||||
//
|
||||
// The order is load bearing: the BOM goes first so it cannot be counted as the
|
||||
// leading whitespace, and CR normalisation goes before the blank-line strip so
|
||||
// that a CRLF blank line is one.
|
||||
func Normalize(raw string) string {
|
||||
text := strings.TrimPrefix(raw, "\uFEFF")
|
||||
|
||||
// `\r\n?` → `\n`: a CRLF pair and a lone CR both become one newline.
|
||||
if strings.IndexByte(text, '\r') >= 0 {
|
||||
var b strings.Builder
|
||||
b.Grow(len(text))
|
||||
for i := 0; i < len(text); i++ {
|
||||
if text[i] != '\r' {
|
||||
b.WriteByte(text[i])
|
||||
continue
|
||||
}
|
||||
b.WriteByte('\n')
|
||||
if i+1 < len(text) && text[i+1] == '\n' {
|
||||
i++
|
||||
}
|
||||
}
|
||||
text = b.String()
|
||||
}
|
||||
|
||||
// `^\s*\n+` → ``. Greedy `\s*` then at least one newline: the effect is to
|
||||
// drop the leading whitespace run up to and including its LAST newline, and
|
||||
// to drop nothing at all when that run contains no newline. A definition
|
||||
// indented by one space is therefore still unfenced, which is what the
|
||||
// editor decides too.
|
||||
end, last := 0, -1
|
||||
for i, r := range text {
|
||||
if !jsIsSpace(r) {
|
||||
break
|
||||
}
|
||||
if r == '\n' {
|
||||
last = i
|
||||
}
|
||||
end = i + len(string(r))
|
||||
}
|
||||
_ = end
|
||||
if last >= 0 {
|
||||
text = text[last+1:]
|
||||
}
|
||||
|
||||
return text
|
||||
}
|
||||
|
||||
// fence locates the frontmatter block in already-normalized text.
|
||||
//
|
||||
// /^---[ \t]*\n([\s\S]*?)\n---[ \t]*(?=\n|$)/
|
||||
//
|
||||
// Returns the YAML source, the offset just past the closing fence, and whether
|
||||
// there was one. Lazy: the FIRST closing fence wins, which is why a definition
|
||||
// carrying a second `---` document keeps only the first and reads the rest as
|
||||
// body.
|
||||
func fence(text string) (yaml string, end int, ok bool) {
|
||||
if !strings.HasPrefix(text, "---") {
|
||||
return "", 0, false
|
||||
}
|
||||
i := 3
|
||||
for i < len(text) && (text[i] == ' ' || text[i] == '\t') {
|
||||
i++
|
||||
}
|
||||
if i >= len(text) || text[i] != '\n' {
|
||||
return "", 0, false
|
||||
}
|
||||
start := i + 1
|
||||
|
||||
for at := start - 1; at >= 0 && at < len(text); {
|
||||
nl := strings.IndexByte(text[at+1:], '\n')
|
||||
if nl < 0 {
|
||||
return "", 0, false
|
||||
}
|
||||
at = at + 1 + nl // index of the newline that must precede the fence
|
||||
|
||||
rest := text[at+1:]
|
||||
if !strings.HasPrefix(rest, "---") {
|
||||
continue
|
||||
}
|
||||
j := 3
|
||||
for j < len(rest) && (rest[j] == ' ' || rest[j] == '\t') {
|
||||
j++
|
||||
}
|
||||
// `(?=\n|$)` — end of the document, or the end of this line. `$` has no
|
||||
// multiline flag on the frontend either, so it means end of document.
|
||||
if j < len(rest) && rest[j] != '\n' {
|
||||
continue
|
||||
}
|
||||
return text[start:at], at + 1 + j, true
|
||||
}
|
||||
return "", 0, false
|
||||
}
|
||||
|
||||
// HasFrontmatter reports whether this text opens with a frontmatter block at
|
||||
// all. It does not say whether that block parses.
|
||||
func HasFrontmatter(raw string) bool {
|
||||
_, _, ok := fence(Normalize(raw))
|
||||
return ok
|
||||
}
|
||||
|
||||
// Document is a definition split into its two halves.
|
||||
type Document struct {
|
||||
// Data is the frontmatter as plain data. Always a mapping: a frontmatter
|
||||
// block that parses to a sequence is discarded, exactly as the frontend
|
||||
// discards it, because every reader downstream indexes it by key.
|
||||
Data map[string]any
|
||||
|
||||
// Body is everything after the closing fence, trimmed. A document with no
|
||||
// frontmatter is all body.
|
||||
Body string
|
||||
|
||||
// Fenced records whether a frontmatter block was found, which Data alone
|
||||
// cannot express — an empty fence pair and a missing one both give an
|
||||
// empty mapping.
|
||||
Fenced bool
|
||||
}
|
||||
|
||||
// ParseFrontmatter splits a definition and reads its frontmatter.
|
||||
//
|
||||
// Returns a *Error when the YAML subset refuses a line. A document with no
|
||||
// recognisable fence is NOT an error: it is a document with no frontmatter,
|
||||
// and what happens to it is the validator's decision — the same division the
|
||||
// frontend makes.
|
||||
func ParseFrontmatter(raw string) (Document, error) {
|
||||
text := Normalize(raw)
|
||||
|
||||
yaml, end, ok := fence(text)
|
||||
if !ok {
|
||||
return Document{Data: map[string]any{}, Body: text, Fenced: false}, nil
|
||||
}
|
||||
|
||||
value, err := ParseYAML(yaml)
|
||||
if err != nil {
|
||||
return Document{}, err
|
||||
}
|
||||
|
||||
data, _ := value.(map[string]any)
|
||||
if data == nil {
|
||||
data = map[string]any{}
|
||||
}
|
||||
|
||||
return Document{Data: data, Body: jsTrim(text[end:]), Fenced: true}, nil
|
||||
}
|
||||
|
||||
// sectionSource is the text under a `## Heading`, up to the next one.
|
||||
//
|
||||
// new RegExp(`##\\s+${escaped}\\s*\\n([\\s\\S]*?)(?=\\n##\\s|$)`, 'i')
|
||||
//
|
||||
// Case-insensitive, and deliberately not anchored to the start of a line —
|
||||
// that is what the frontend does. The heading is compared literally rather
|
||||
// than compiled into a pattern, which is the same protection the frontend gets
|
||||
// by escaping it: a heading containing regular-expression punctuation must
|
||||
// match the words it is built from.
|
||||
//
|
||||
// Returns ok=false for a section that is not there, which is a different thing
|
||||
// from a section that is there and empty.
|
||||
func sectionSource(body, heading string) (string, bool) {
|
||||
lower := strings.ToLower(body)
|
||||
want := strings.ToLower(heading)
|
||||
|
||||
for at := 0; ; {
|
||||
h := strings.Index(lower[at:], "##")
|
||||
if h < 0 {
|
||||
return "", false
|
||||
}
|
||||
h += at
|
||||
at = h + 2
|
||||
|
||||
// `##` then `\s+` then the heading.
|
||||
i := h + 2
|
||||
gap := i
|
||||
for i < len(body) {
|
||||
r, size := decodeRune(body[i:])
|
||||
if !jsIsSpace(r) {
|
||||
break
|
||||
}
|
||||
i += size
|
||||
}
|
||||
if i == gap {
|
||||
continue // `\s+` needs at least one
|
||||
}
|
||||
if !strings.HasPrefix(lower[i:], want) {
|
||||
continue
|
||||
}
|
||||
i += len(want)
|
||||
|
||||
// `\s*\n`: a whitespace run that ends in a newline.
|
||||
j, nl := i, -1
|
||||
for j < len(body) {
|
||||
r, size := decodeRune(body[j:])
|
||||
if !jsIsSpace(r) {
|
||||
break
|
||||
}
|
||||
if r == '\n' {
|
||||
nl = j
|
||||
break
|
||||
}
|
||||
j += size
|
||||
}
|
||||
if nl < 0 {
|
||||
continue
|
||||
}
|
||||
|
||||
start := nl + 1
|
||||
// `(?=\n##\s|$)`, lazily: the first following line that opens a new
|
||||
// `##` heading. `###` does not, because the character after `##` must
|
||||
// be whitespace.
|
||||
for k := start; ; {
|
||||
n := strings.Index(body[k:], "\n##")
|
||||
if n < 0 {
|
||||
return body[start:], true
|
||||
}
|
||||
n += k
|
||||
after := n + 3
|
||||
if after < len(body) {
|
||||
r, _ := decodeRune(body[after:])
|
||||
if jsIsSpace(r) {
|
||||
return body[start:n], true
|
||||
}
|
||||
}
|
||||
k = n + 1
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// decodeRune is utf8.DecodeRuneInString, kept local so the section reader has
|
||||
// one obvious way to step through the body.
|
||||
func decodeRune(s string) (rune, int) {
|
||||
for i, r := range s {
|
||||
_ = i
|
||||
return r, len(string(r))
|
||||
}
|
||||
return 0, 0
|
||||
}
|
||||
|
||||
// SectionText is the prose under a `## Heading`, with its bullets and blank
|
||||
// lines flattened to one line.
|
||||
//
|
||||
// Used for a workforce level's description, which is a sentence rather than a
|
||||
// list.
|
||||
func SectionText(body, heading string) string {
|
||||
source, ok := sectionSource(body, heading)
|
||||
if !ok {
|
||||
return ""
|
||||
}
|
||||
parts := []string{}
|
||||
for _, l := range strings.Split(source, "\n") {
|
||||
l = jsTrim(stripListMarker(l))
|
||||
if l == "" {
|
||||
continue
|
||||
}
|
||||
parts = append(parts, l)
|
||||
}
|
||||
return jsTrim(strings.Join(parts, " "))
|
||||
}
|
||||
|
||||
// stripListMarker removes a leading `-`, `*` or `1.` / `1)` bullet.
|
||||
//
|
||||
// /^\s*(?:[-*]|\d+[.)])\s+/
|
||||
func stripListMarker(l string) string {
|
||||
i := 0
|
||||
for i < len(l) {
|
||||
r, size := decodeRune(l[i:])
|
||||
if !jsIsSpace(r) {
|
||||
break
|
||||
}
|
||||
i += size
|
||||
}
|
||||
marker := i
|
||||
switch {
|
||||
case i < len(l) && (l[i] == '-' || l[i] == '*'):
|
||||
i++
|
||||
default:
|
||||
digits := i
|
||||
for i < len(l) && l[i] >= '0' && l[i] <= '9' {
|
||||
i++
|
||||
}
|
||||
if i == digits || i >= len(l) || (l[i] != '.' && l[i] != ')') {
|
||||
return l
|
||||
}
|
||||
i++
|
||||
}
|
||||
// `\s+` after the marker is required; without it there is no list item.
|
||||
space := i
|
||||
for i < len(l) {
|
||||
r, size := decodeRune(l[i:])
|
||||
if !jsIsSpace(r) {
|
||||
break
|
||||
}
|
||||
i += size
|
||||
}
|
||||
if i == space {
|
||||
return l
|
||||
}
|
||||
_ = marker
|
||||
return l[i:]
|
||||
}
|
||||
Reference in New Issue
Block a user