first commit
This commit is contained in:
543
go-api/internal/definition/agent.go
Normal file
543
go-api/internal/definition/agent.go
Normal file
@@ -0,0 +1,543 @@
|
||||
package definition
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// One Markdown definition → one agent.
|
||||
//
|
||||
// A port of parseAgent / validateAgentSource in src/lib/agents/registry.js and
|
||||
// normalizeAgent in src/lib/agents/agentConfig.js.
|
||||
//
|
||||
// The contract normalizeAgent holds, and this holds with it:
|
||||
//
|
||||
// - Everything is optional. A definition declaring only an id and a name
|
||||
// normalizes to a working agent with documented defaults.
|
||||
// - Nothing unknown survives. Statuses, reasoning modes, pages, icons,
|
||||
// knowledge kinds and permission roles are checked against the closed
|
||||
// tables in vocabulary.go; an unrecognised value is a named error rather
|
||||
// than a dropped key.
|
||||
// - What validates is kept. One bad entry costs its author that entry and a
|
||||
// message, never the rest of the file.
|
||||
//
|
||||
// One rule is deliberately absent, and it is absent on the frontend for the
|
||||
// same reason: an agent with NO SKILLS is not refused. Five of this product's
|
||||
// pages have no Owliver skills and answer from their own page responder, so
|
||||
// refusing a skill-less agent would mean inventing placeholder skills to make
|
||||
// those pages configurable.
|
||||
|
||||
// Starter is one conversation starter.
|
||||
type Starter struct {
|
||||
Label string `json:"label"`
|
||||
Prompt string `json:"prompt"`
|
||||
}
|
||||
|
||||
// Knowledge is one thing an agent has been told, as distinct from something it
|
||||
// can do. Modelled as a document with an id and a body because that is the
|
||||
// shape a retrieval layer reads.
|
||||
type Knowledge struct {
|
||||
ID string `json:"id"`
|
||||
Label string `json:"label"`
|
||||
Kind string `json:"kind"`
|
||||
Body string `json:"body"`
|
||||
URL string `json:"url"`
|
||||
}
|
||||
|
||||
// Person is one named grant on an agent.
|
||||
type Person struct {
|
||||
User string `json:"user"`
|
||||
Role string `json:"role"`
|
||||
}
|
||||
|
||||
// Permissions is who owns an agent, who may reach it, and what they may do.
|
||||
//
|
||||
// Parsed and NOT enforced. Migration 000005 deliberately has no
|
||||
// definition_permissions table: the block stays inside the Markdown until its
|
||||
// semantics are defined.
|
||||
type Permissions struct {
|
||||
Owner string `json:"owner"`
|
||||
Access string `json:"access"`
|
||||
People []Person `json:"people"`
|
||||
}
|
||||
|
||||
// Agent is a definition as the backend reads it.
|
||||
type Agent struct {
|
||||
ID string `json:"id"`
|
||||
Name string `json:"name"`
|
||||
Description string `json:"description"`
|
||||
Status string `json:"status"`
|
||||
Version int `json:"version"`
|
||||
|
||||
// Pages as CANONICAL surface ids.
|
||||
//
|
||||
// Unlike Skill.Pages, which keeps what the author wrote. The two are
|
||||
// genuinely different on the frontend — normalizeAgent maps every page
|
||||
// through canonicalPage and parseSkill does not — so agent_definitions.pages
|
||||
// and skill_definitions.pages hold different vocabularies for the same
|
||||
// concept. Reproduced rather than reconciled: making them agree here would
|
||||
// make each one disagree with its own editor.
|
||||
Pages []string `json:"pages"`
|
||||
|
||||
Icon string `json:"icon"`
|
||||
Reasoning string `json:"reasoning"`
|
||||
Trigger string `json:"trigger"`
|
||||
WebSearch bool `json:"webSearch"`
|
||||
|
||||
Skills []string `json:"skills"`
|
||||
Subagents []string `json:"subagents"`
|
||||
Starters []Starter `json:"starters"`
|
||||
Knowledge []Knowledge `json:"knowledge"`
|
||||
Permissions Permissions `json:"permissions"`
|
||||
|
||||
// Instructions is the body's `## Instructions` section. Prose belongs under
|
||||
// a heading where it can be written and read as prose, not in a
|
||||
// frontmatter string.
|
||||
Instructions string `json:"instructions"`
|
||||
|
||||
// Errors is what this definition lost on the way in, in the order
|
||||
// normalizeAgent produces them. Carried on the record rather than thrown,
|
||||
// so one bad entry costs its author that entry and a message.
|
||||
Errors []string `json:"errors"`
|
||||
|
||||
Body string `json:"-"`
|
||||
}
|
||||
|
||||
// asList is agentConfig.js's own coercion: an array stays an array, nothing
|
||||
// becomes nothing, and anything else becomes a list of one.
|
||||
//
|
||||
// This is why `pages: candidates` is accepted for an AGENT and refused for a
|
||||
// SKILL — parseSkill requires a real sequence and normalizeAgent coerces.
|
||||
func asList(v any) []any {
|
||||
switch x := v.(type) {
|
||||
case []any:
|
||||
return x
|
||||
case nil:
|
||||
return []any{}
|
||||
case string:
|
||||
if x == "" {
|
||||
return []any{}
|
||||
}
|
||||
}
|
||||
return []any{v}
|
||||
}
|
||||
|
||||
// uniqueStrings keeps order and drops repeats; a blank entry is an error rather
|
||||
// than a silent gap, because a blank id is an address that points nowhere.
|
||||
func uniqueStrings(raw any, where, label string, errs *[]string) []string {
|
||||
seen := map[string]bool{}
|
||||
out := []string{}
|
||||
for i, entry := range asList(raw) {
|
||||
value := jsTrimmed(entry)
|
||||
if value == "" {
|
||||
*errs = append(*errs, fmt.Sprintf("%s[%d]: %s cannot be blank.", where, i, label))
|
||||
continue
|
||||
}
|
||||
if seen[value] {
|
||||
continue
|
||||
}
|
||||
seen[value] = true
|
||||
out = append(out, value)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// normalizePages resolves every declared page to a canonical surface key.
|
||||
//
|
||||
// Through CanonicalPage, so a definition may write an alias — `university` for
|
||||
// `krow-forge` — exactly as a skill may. An unknown page is an error rather
|
||||
// than a silently dropped entry, because a page nobody recognises is an agent
|
||||
// that will never appear anywhere and give no reason why.
|
||||
func normalizePages(raw any, errs *[]string) []string {
|
||||
seen := map[string]bool{}
|
||||
pages := []string{}
|
||||
for i, entry := range asList(raw) {
|
||||
written := jsTrimmed(entry)
|
||||
if written == "" {
|
||||
*errs = append(*errs, fmt.Sprintf("pages[%d]: a page cannot be blank.", i))
|
||||
continue
|
||||
}
|
||||
canonical := CanonicalPage(written)
|
||||
if canonical == "" {
|
||||
*errs = append(*errs, fmt.Sprintf(
|
||||
"pages[%d]: `%s` is not a page this product has.", i, written))
|
||||
continue
|
||||
}
|
||||
if seen[canonical] {
|
||||
continue
|
||||
}
|
||||
seen[canonical] = true
|
||||
pages = append(pages, canonical)
|
||||
}
|
||||
return pages
|
||||
}
|
||||
|
||||
// normalizeStarter reads one starter, in either the plain-string or the mapping
|
||||
// form. A starter with no prompt of its own asks what it says.
|
||||
func normalizeStarter(raw any, index int, errs *[]string) *Starter {
|
||||
where := fmt.Sprintf("starters[%d]", index)
|
||||
|
||||
switch v := raw.(type) {
|
||||
case string, float64:
|
||||
label := jsTrimmed(v)
|
||||
if label == "" {
|
||||
*errs = append(*errs, where+": a starter needs text.")
|
||||
return nil
|
||||
}
|
||||
return &Starter{Label: label, Prompt: label}
|
||||
case map[string]any:
|
||||
// `raw.label ?? raw.prompt` — nullish, so an absent or null label
|
||||
// falls through to the prompt and a starter written as a bare prompt
|
||||
// still has something to show.
|
||||
source := v["label"]
|
||||
if source == nil {
|
||||
source = v["prompt"]
|
||||
}
|
||||
label := jsTrimmed(source)
|
||||
if label == "" {
|
||||
*errs = append(*errs, where+": a starter needs a `label`.")
|
||||
return nil
|
||||
}
|
||||
prompt := jsTrimmed(v["prompt"])
|
||||
if prompt == "" {
|
||||
prompt = label
|
||||
}
|
||||
return &Starter{Label: label, Prompt: prompt}
|
||||
}
|
||||
|
||||
*errs = append(*errs, where+": a starter must be a line of text, or a mapping of options.")
|
||||
return nil
|
||||
}
|
||||
|
||||
// normalizeKnowledge reads one knowledge entry.
|
||||
func normalizeKnowledge(raw any, index int, errs *[]string) *Knowledge {
|
||||
where := fmt.Sprintf("knowledge[%d]", index)
|
||||
|
||||
switch v := raw.(type) {
|
||||
case string, float64:
|
||||
body := jsTrimmed(v)
|
||||
if body == "" {
|
||||
*errs = append(*errs, where+": a knowledge entry needs text.")
|
||||
return nil
|
||||
}
|
||||
id := slugify(runeSlice(body, 40))
|
||||
if id == "" {
|
||||
id = fmt.Sprintf("k%d", index+1)
|
||||
}
|
||||
return &Knowledge{
|
||||
ID: id, Label: runeSlice(body, 60), Kind: DefaultKnowledgeKind, Body: body,
|
||||
}
|
||||
case map[string]any:
|
||||
label := jsTrimmed(v["label"])
|
||||
body := jsTrimmed(v["body"])
|
||||
url := jsTrimmed(v["url"])
|
||||
|
||||
if label == "" && body == "" {
|
||||
*errs = append(*errs, where+": a knowledge entry needs a `label` or a `body`.")
|
||||
return nil
|
||||
}
|
||||
|
||||
kind := jsTrimmed(v["kind"])
|
||||
if kind == "" {
|
||||
kind = DefaultKnowledgeKind
|
||||
}
|
||||
if !contains(KnowledgeKinds, kind) {
|
||||
*errs = append(*errs, fmt.Sprintf(
|
||||
"%s: `%s` is not a knowledge kind. Use one of %s.",
|
||||
where, kind, strings.Join(KnowledgeKinds, ", ")))
|
||||
return nil
|
||||
}
|
||||
if kind == "link" && url == "" {
|
||||
*errs = append(*errs, where+": a `link` needs a `url`.")
|
||||
return nil
|
||||
}
|
||||
|
||||
id := jsTrimmed(v["id"])
|
||||
if id == "" {
|
||||
id = slugify(label)
|
||||
}
|
||||
if id == "" {
|
||||
id = fmt.Sprintf("k%d", index+1)
|
||||
}
|
||||
if label == "" {
|
||||
label = runeSlice(body, 60)
|
||||
}
|
||||
return &Knowledge{ID: id, Label: label, Kind: kind, Body: body, URL: url}
|
||||
}
|
||||
|
||||
*errs = append(*errs, where+": a knowledge entry must be a line of text, or a mapping of options.")
|
||||
return nil
|
||||
}
|
||||
|
||||
// runeSlice is JavaScript's String.prototype.slice(0, n), which counts UTF-16
|
||||
// units. Counting runes instead differs only for astral characters, and cutting
|
||||
// a surrogate pair in half — which the frontend can do — would produce a label
|
||||
// no comparison could match. Runes are used deliberately; the conformance suite
|
||||
// carries no case that distinguishes them.
|
||||
func runeSlice(s string, n int) string {
|
||||
r := []rune(s)
|
||||
if len(r) <= n {
|
||||
return s
|
||||
}
|
||||
return string(r[:n])
|
||||
}
|
||||
|
||||
// normalizePermissions reads the `permissions:` block.
|
||||
func normalizePermissions(raw any, errs *[]string) Permissions {
|
||||
none := Permissions{Access: DefaultAgentAccess, People: []Person{}}
|
||||
if raw == nil {
|
||||
return none
|
||||
}
|
||||
|
||||
mapping, ok := raw.(map[string]any)
|
||||
if !ok {
|
||||
*errs = append(*errs, "permissions: must be a mapping of `owner`, `access` and `people`.")
|
||||
return none
|
||||
}
|
||||
|
||||
access := jsTrimmed(mapping["access"])
|
||||
if access == "" {
|
||||
access = DefaultAgentAccess
|
||||
}
|
||||
if !contains(AgentAccess, access) {
|
||||
*errs = append(*errs, fmt.Sprintf(
|
||||
"permissions.access: `%s` is not an access mode. Use one of %s.",
|
||||
access, strings.Join(AgentAccess, ", ")))
|
||||
}
|
||||
|
||||
people := []Person{}
|
||||
for i, entry := range asList(mapping["people"]) {
|
||||
where := fmt.Sprintf("permissions.people[%d]", i)
|
||||
person, ok := entry.(map[string]any)
|
||||
if !ok {
|
||||
*errs = append(*errs, where+": must be a mapping of `user` and `role`.")
|
||||
continue
|
||||
}
|
||||
user := jsTrimmed(person["user"])
|
||||
if user == "" {
|
||||
*errs = append(*errs, where+": needs a `user`.")
|
||||
continue
|
||||
}
|
||||
role := jsTrimmed(person["role"])
|
||||
if role == "" {
|
||||
role = DefaultPermission
|
||||
}
|
||||
if !contains(PermissionRole, role) {
|
||||
*errs = append(*errs, fmt.Sprintf(
|
||||
"%s: `%s` is not a role. Use one of %s.",
|
||||
where, role, strings.Join(PermissionRole, ", ")))
|
||||
continue
|
||||
}
|
||||
people = append(people, Person{User: user, Role: role})
|
||||
}
|
||||
|
||||
result := Permissions{Owner: jsTrimmed(mapping["owner"]), Access: access, People: people}
|
||||
if !contains(AgentAccess, access) {
|
||||
result.Access = DefaultAgentAccess
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// ParseAgent reads an agent definition.
|
||||
//
|
||||
// The order in which errors accumulate is part of the contract: validateAgent
|
||||
// reports the FIRST one, so a definition with two problems must name the same
|
||||
// one the editor names. That order is status, reasoning, icon, version,
|
||||
// subagents, starters, knowledge, pages, skills, permissions — which is
|
||||
// evaluation order in normalizeAgent, counting the object literal it returns.
|
||||
func ParseAgent(raw string, opts Options) (*Agent, error) {
|
||||
doc, err := ParseFrontmatter(raw)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
data := doc.Data
|
||||
errs := []string{}
|
||||
|
||||
id := jsTrim(jsString(data["id"]))
|
||||
if !jsTruthy(data["id"]) {
|
||||
id = slugify(data["name"])
|
||||
if id == "" {
|
||||
id = fileStem(opts.path())
|
||||
}
|
||||
}
|
||||
|
||||
status := jsTrimmed(data["status"])
|
||||
if status == "" {
|
||||
status = DefaultAgentStatus
|
||||
}
|
||||
if !contains(AgentStatuses, status) {
|
||||
errs = append(errs, fmt.Sprintf("status: `%s` is not a status. Use one of %s.",
|
||||
status, strings.Join(AgentStatuses, ", ")))
|
||||
}
|
||||
|
||||
reasoning := jsTrimmed(data["reasoning"])
|
||||
if reasoning == "" {
|
||||
reasoning = DefaultReasoning
|
||||
}
|
||||
if !contains(ReasoningModes, reasoning) {
|
||||
errs = append(errs, fmt.Sprintf("reasoning: `%s` is not a reasoning mode. Use one of %s.",
|
||||
reasoning, strings.Join(ReasoningModes, ", ")))
|
||||
}
|
||||
|
||||
icon := jsTrimmed(data["icon"])
|
||||
if icon == "" {
|
||||
icon = DefaultAgentIcon
|
||||
}
|
||||
if !contains(AgentIcons, icon) {
|
||||
errs = append(errs, fmt.Sprintf("icon: `%s` is not an icon this product has.", icon))
|
||||
}
|
||||
|
||||
// A version is an integer that only ever goes up. Anything else is an
|
||||
// authoring slip, and reading it as 1 is kinder than refusing the file —
|
||||
// but it is still reported, because a definition that thinks it is v3 and
|
||||
// registers as v1 will publish over something.
|
||||
version := 1
|
||||
if v, present := data["version"]; present && v != nil && v != "" {
|
||||
parsed := jsNumber(v)
|
||||
if math.IsNaN(parsed) || parsed != math.Trunc(parsed) || math.IsInf(parsed, 0) || parsed < 1 {
|
||||
errs = append(errs, fmt.Sprintf(
|
||||
"version: `%s` is not a whole number of 1 or more.", jsString(v)))
|
||||
} else if parsed > maxExactInteger {
|
||||
// Beyond 2^53-1 a float64 no longer names one integer, so there is
|
||||
// no value to carry. Saturating keeps the conversion below defined,
|
||||
// and ValidateAgent refuses everything above MaxVersion anyway, so
|
||||
// a saturated version can never reach a column.
|
||||
version = maxExactInteger
|
||||
} else {
|
||||
version = int(parsed)
|
||||
}
|
||||
}
|
||||
|
||||
subagents := uniqueStrings(data["subagents"], "subagents", "a subagent id", &errs)
|
||||
kept := subagents[:0]
|
||||
for _, s := range subagents {
|
||||
if id != "" && s == id {
|
||||
errs = append(errs, "subagents: an agent cannot be its own subagent.")
|
||||
continue
|
||||
}
|
||||
kept = append(kept, s)
|
||||
}
|
||||
subagents = kept
|
||||
|
||||
starters := []Starter{}
|
||||
for i, entry := range asList(data["starters"]) {
|
||||
if s := normalizeStarter(entry, i, &errs); s != nil {
|
||||
starters = append(starters, *s)
|
||||
}
|
||||
}
|
||||
|
||||
knowledge := []Knowledge{}
|
||||
for i, entry := range asList(data["knowledge"]) {
|
||||
if k := normalizeKnowledge(entry, i, &errs); k != nil {
|
||||
knowledge = append(knowledge, *k)
|
||||
}
|
||||
}
|
||||
|
||||
// From here the order follows the object literal normalizeAgent returns.
|
||||
pages := normalizePages(data["pages"], &errs)
|
||||
skills := uniqueStrings(data["skills"], "skills", "a skill id", &errs)
|
||||
permissions := normalizePermissions(data["permissions"], &errs)
|
||||
|
||||
instructions, _ := sectionSource(doc.Body, "Instructions")
|
||||
|
||||
agent := &Agent{
|
||||
ID: id,
|
||||
Name: "Untitled agent",
|
||||
Status: status,
|
||||
Version: version,
|
||||
Pages: pages,
|
||||
Icon: icon,
|
||||
Reasoning: reasoning,
|
||||
Trigger: jsTrimmed(data["trigger"]),
|
||||
WebSearch: data["webSearch"] == true || data["web_search"] == true,
|
||||
Skills: skills,
|
||||
Subagents: subagents,
|
||||
Starters: starters,
|
||||
Knowledge: knowledge,
|
||||
Permissions: permissions,
|
||||
Instructions: jsTrim(instructions),
|
||||
Errors: errs,
|
||||
Body: doc.Body,
|
||||
}
|
||||
|
||||
if jsTruthy(data["name"]) {
|
||||
agent.Name = jsString(data["name"])
|
||||
}
|
||||
if jsTruthy(data["description"]) {
|
||||
agent.Description = jsString(data["description"])
|
||||
}
|
||||
if !contains(AgentStatuses, status) {
|
||||
agent.Status = DefaultAgentStatus
|
||||
}
|
||||
if !contains(ReasoningModes, reasoning) {
|
||||
agent.Reasoning = DefaultReasoning
|
||||
}
|
||||
if !contains(AgentIcons, icon) {
|
||||
agent.Icon = DefaultAgentIcon
|
||||
}
|
||||
|
||||
return agent, nil
|
||||
}
|
||||
|
||||
// ValidateAgent decides whether an agent definition may be stored.
|
||||
//
|
||||
// Returns nil when it may. The order is the order an author would fix things
|
||||
// in, which is why it reads the same way ValidateSkill does. Note that the
|
||||
// `id` message differs from the skill one by two words — that difference is
|
||||
// the frontend's, and it is reproduced rather than tidied.
|
||||
func ValidateAgent(raw string) error {
|
||||
if jsTrim(raw) == "" {
|
||||
return &Rejection{Message: "Paste or upload a Markdown definition."}
|
||||
}
|
||||
|
||||
if n := len([]rune(raw)); n > MaxMarkdownLength {
|
||||
return &Rejection{
|
||||
BackendOnly: true,
|
||||
Message: fmt.Sprintf(
|
||||
"That definition is %d characters. The limit is %d.", n, MaxMarkdownLength),
|
||||
}
|
||||
}
|
||||
|
||||
agent, err := ParseAgent(raw, Options{})
|
||||
if err != nil {
|
||||
return &Rejection{Message: jsTrim("That definition could not be parsed. " + err.Error())}
|
||||
}
|
||||
|
||||
if agent.ID == "" {
|
||||
return &Rejection{Message: "The frontmatter needs an `id`."}
|
||||
}
|
||||
if !isDefinitionID(agent.ID) {
|
||||
return &Rejection{Message: "The `id` must be lower-case letters, numbers and dashes."}
|
||||
}
|
||||
// `!raw.includes('name:') || agent.name === 'Untitled agent'` — the literal
|
||||
// substring test is the frontend's, and it is why a definition whose name
|
||||
// resolves to the fallback is refused even when some other key happens to
|
||||
// spell `name:`.
|
||||
if !strings.Contains(raw, "name:") || agent.Name == "Untitled agent" {
|
||||
return &Rejection{Message: "The frontmatter needs a `name`."}
|
||||
}
|
||||
if len(agent.Pages) == 0 {
|
||||
return &Rejection{Message: "An agent needs at least one `pages:` entry, or it can never be offered anywhere."}
|
||||
}
|
||||
if len(agent.Errors) > 0 {
|
||||
return &Rejection{Message: agent.Errors[0]}
|
||||
}
|
||||
|
||||
// The backend's own bound, the companion to the size rule above:
|
||||
// agent_definitions.version is a PostgreSQL `integer`, and the frontend
|
||||
// accepts any whole number of 1 or more. It is checked here rather than in
|
||||
// ParseAgent so the normalized record stays identical to the frontend's for
|
||||
// every definition the frontend accepts, and last among the rules so a
|
||||
// definition the frontend also refuses is refused with the frontend's own
|
||||
// message.
|
||||
if agent.Version > MaxVersion {
|
||||
return &Rejection{
|
||||
BackendOnly: true,
|
||||
Message: fmt.Sprintf(
|
||||
"version: `%d` is larger than %d.", agent.Version, MaxVersion),
|
||||
}
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
928
go-api/internal/definition/conformance_test.go
Normal file
928
go-api/internal/definition/conformance_test.go
Normal file
@@ -0,0 +1,928 @@
|
||||
package definition_test
|
||||
|
||||
import (
|
||||
"encoding/base64"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"reflect"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/krow/krow-backend/go-api/internal/definition"
|
||||
)
|
||||
|
||||
// Phase 4D — JS/Go parser conformance.
|
||||
//
|
||||
// The fixture these tests read (testdata/oracle.json) is not written by hand.
|
||||
// It is captured by scripts/oracle.mjs, which loads the REAL frontend module
|
||||
// graph through Vite — import.meta.glob, the `@/` alias and raw Markdown
|
||||
// loading all behave exactly as they do in the app — and records what the
|
||||
// JavaScript parser did with every shipped definition and every adversarial
|
||||
// case. So the assertion below is not "Go agrees with a description of the
|
||||
// frontend"; it is "Go agrees with the frontend", replayed.
|
||||
//
|
||||
// Regenerate after any change to src/lib/skills or src/lib/agents:
|
||||
//
|
||||
// node scripts/oracle.mjs go-api/internal/definition/testdata/oracle.json
|
||||
//
|
||||
// A frontend change that alters parsing therefore fails these tests, which is
|
||||
// the point: the contract cannot drift silently in either direction.
|
||||
|
||||
type oracle struct {
|
||||
Vocabulary struct {
|
||||
Pages []struct {
|
||||
ID string `json:"id"`
|
||||
Aliases []string `json:"aliases"`
|
||||
} `json:"pages"`
|
||||
AgentStatuses []string `json:"agentStatuses"`
|
||||
Reasoning []string `json:"reasoning"`
|
||||
Icons []string `json:"icons"`
|
||||
KnowledgeKinds []string `json:"knowledgeKinds"`
|
||||
Access []string `json:"access"`
|
||||
Roles []string `json:"roles"`
|
||||
} `json:"vocabulary"`
|
||||
Corpus []observation `json:"corpus"`
|
||||
Cases []observation `json:"cases"`
|
||||
}
|
||||
|
||||
// observation is one definition as the JavaScript saw it, end to end.
|
||||
type observation struct {
|
||||
// Exactly one of these identifies the row.
|
||||
Path string `json:"path"`
|
||||
Name string `json:"name"`
|
||||
Type string `json:"type"`
|
||||
|
||||
Kind string `json:"kind"` // agent | skill
|
||||
RawBase64 string `json:"rawBase64"`
|
||||
|
||||
HasFrontmatter bool `json:"hasFrontmatter"`
|
||||
|
||||
Frontmatter struct {
|
||||
OK bool `json:"ok"`
|
||||
Data map[string]any `json:"data"`
|
||||
Body string `json:"body"`
|
||||
Error string `json:"error"`
|
||||
} `json:"frontmatter"`
|
||||
|
||||
Parse struct {
|
||||
OK bool `json:"ok"`
|
||||
Error string `json:"error"`
|
||||
} `json:"parse"`
|
||||
|
||||
Normalized map[string]any `json:"normalized"`
|
||||
|
||||
Accepted bool `json:"accepted"`
|
||||
Rejection *string `json:"rejection"`
|
||||
}
|
||||
|
||||
func (o observation) id() string {
|
||||
if o.Path != "" {
|
||||
return o.Path
|
||||
}
|
||||
return o.Name
|
||||
}
|
||||
|
||||
func (o observation) raw(t *testing.T) string {
|
||||
t.Helper()
|
||||
b, err := base64.StdEncoding.DecodeString(o.RawBase64)
|
||||
if err != nil {
|
||||
t.Fatalf("%s: undecodable fixture: %v", o.id(), err)
|
||||
}
|
||||
return string(b)
|
||||
}
|
||||
|
||||
func load(t *testing.T) *oracle {
|
||||
t.Helper()
|
||||
b, err := os.ReadFile("testdata/oracle.json")
|
||||
if err != nil {
|
||||
t.Fatalf("read fixture: %v", err)
|
||||
}
|
||||
var o oracle
|
||||
if err := json.Unmarshal(b, &o); err != nil {
|
||||
t.Fatalf("parse fixture: %v", err)
|
||||
}
|
||||
if len(o.Corpus) == 0 || len(o.Cases) == 0 {
|
||||
t.Fatal("fixture is empty; regenerate with scripts/oracle.mjs")
|
||||
}
|
||||
return &o
|
||||
}
|
||||
|
||||
func all(o *oracle) []observation { return append(append([]observation{}, o.Corpus...), o.Cases...) }
|
||||
|
||||
/* ── 1. The corpus is the corpus ──────────────────────────────────────────── */
|
||||
|
||||
// The shipped definition count, asserted rather than assumed. A definition
|
||||
// added to or removed from the product without regenerating the fixture leaves
|
||||
// these tests passing against a corpus that no longer exists, which is the one
|
||||
// way this suite could quietly stop meaning anything.
|
||||
func TestCorpusShape(t *testing.T) {
|
||||
o := load(t)
|
||||
|
||||
counts := map[string]int{}
|
||||
for _, c := range o.Corpus {
|
||||
counts[c.Type]++
|
||||
}
|
||||
|
||||
for _, want := range []struct {
|
||||
kind string
|
||||
n int
|
||||
}{{"agent", 9}, {"skill", 23}, {"example", 5}} {
|
||||
if counts[want.kind] != want.n {
|
||||
t.Errorf("%s definitions: got %d, want %d", want.kind, counts[want.kind], want.n)
|
||||
}
|
||||
}
|
||||
if len(o.Corpus) != 37 {
|
||||
t.Errorf("shipped definitions: got %d, want 37", len(o.Corpus))
|
||||
}
|
||||
}
|
||||
|
||||
/* ── 2. The vocabulary has not drifted ────────────────────────────────────── */
|
||||
|
||||
// Every closed table in vocabulary.go, checked against the table the frontend
|
||||
// actually exports. A page added to surfaces.js fails here rather than becoming
|
||||
// a definition the editor accepts and the API rejects.
|
||||
func TestVocabularyMatchesFrontend(t *testing.T) {
|
||||
o := load(t)
|
||||
|
||||
wantPages := make([]string, len(o.Vocabulary.Pages))
|
||||
for i, p := range o.Vocabulary.Pages {
|
||||
wantPages[i] = p.ID
|
||||
}
|
||||
if !reflect.DeepEqual(definition.SupportedPages, wantPages) {
|
||||
t.Errorf("supported pages differ\n go %v\n js %v", definition.SupportedPages, wantPages)
|
||||
}
|
||||
|
||||
// Aliases resolve, and resolve to the same canonical id.
|
||||
for _, p := range o.Vocabulary.Pages {
|
||||
for _, alias := range append([]string{p.ID}, p.Aliases...) {
|
||||
if got := definition.CanonicalPage(alias); got != p.ID {
|
||||
t.Errorf("CanonicalPage(%q) = %q, want %q", alias, got, p.ID)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for _, table := range []struct {
|
||||
name string
|
||||
got []string
|
||||
wanted []string
|
||||
}{
|
||||
{"agent statuses", definition.AgentStatuses, o.Vocabulary.AgentStatuses},
|
||||
{"reasoning modes", definition.ReasoningModes, o.Vocabulary.Reasoning},
|
||||
{"icons", definition.AgentIcons, o.Vocabulary.Icons},
|
||||
{"knowledge kinds", definition.KnowledgeKinds, o.Vocabulary.KnowledgeKinds},
|
||||
{"access modes", definition.AgentAccess, o.Vocabulary.Access},
|
||||
{"permission roles", definition.PermissionRole, o.Vocabulary.Roles},
|
||||
} {
|
||||
if !reflect.DeepEqual(table.got, table.wanted) {
|
||||
t.Errorf("%s differ\n go %v\n js %v", table.name, table.got, table.wanted)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* ── 3. The frontmatter tree ──────────────────────────────────────────────── */
|
||||
|
||||
// The deepest parity check available: the YAML subset must produce the same
|
||||
// data structure the JavaScript produced, for every definition and every
|
||||
// adversarial case. Not a projection of it — the whole tree.
|
||||
func TestFrontmatterTreeParity(t *testing.T) {
|
||||
for _, c := range all(load(t)) {
|
||||
t.Run(c.id(), func(t *testing.T) {
|
||||
raw := c.raw(t)
|
||||
doc, err := definition.ParseFrontmatter(raw)
|
||||
|
||||
if !c.Frontmatter.OK {
|
||||
if err == nil {
|
||||
t.Fatalf("JS refused this frontmatter (%s); Go accepted it", c.Frontmatter.Error)
|
||||
}
|
||||
if err.Error() != c.Frontmatter.Error {
|
||||
t.Errorf("error text differs\n go %q\n js %q", err.Error(), c.Frontmatter.Error)
|
||||
}
|
||||
return
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatalf("JS read this frontmatter; Go refused it: %v", err)
|
||||
}
|
||||
|
||||
if got, want := normalizeTree(doc.Data), normalizeTree(c.Frontmatter.Data); !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("frontmatter differs\n go %s\n js %s", show(got), show(want))
|
||||
}
|
||||
if doc.Body != c.Frontmatter.Body {
|
||||
t.Errorf("body differs\n go %q\n js %q", doc.Body, c.Frontmatter.Body)
|
||||
}
|
||||
if got := definition.HasFrontmatter(raw); got != c.HasFrontmatter {
|
||||
t.Errorf("HasFrontmatter = %v, JS said %v", got, c.HasFrontmatter)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// normalizeTree puts a parsed tree into the shape `encoding/json` would have
|
||||
// produced, so the Go value and the value round-tripped through the fixture's
|
||||
// JSON are comparable. Numbers become float64 on both sides, which is what
|
||||
// JavaScript had in the first place.
|
||||
func normalizeTree(v any) any {
|
||||
b, err := json.Marshal(v)
|
||||
if err != nil {
|
||||
return fmt.Sprintf("unmarshalable: %v", err)
|
||||
}
|
||||
var out any
|
||||
if err := json.Unmarshal(b, &out); err != nil {
|
||||
return fmt.Sprintf("unmarshalable: %v", err)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func show(v any) string {
|
||||
b, _ := json.Marshal(v)
|
||||
return string(b)
|
||||
}
|
||||
|
||||
/* ── 4. Accept / reject parity ────────────────────────────────────────────── */
|
||||
|
||||
// knownDivergence is the complete list of definitions where the two parsers
|
||||
// disagree, each with the reason. It is a CLOSED list: anything not on it that
|
||||
// disagrees fails, and anything on it that stops disagreeing fails too, so the
|
||||
// list cannot quietly grow and cannot quietly go stale.
|
||||
//
|
||||
// Every entry is a case where Go is stricter, except the first — and the first
|
||||
// is the one asymmetry this package documents as deferred.
|
||||
var knownDivergence = map[string]string{
|
||||
"skill-examples/board-invalid-context.md": "" +
|
||||
"JS rejects on `ui:` placement/source semantics, which this package defers " +
|
||||
"to the frontend. Go accepts and reports Deferred: [ui].",
|
||||
|
||||
"oversized-markdown": "" +
|
||||
"JS accepts; the database refuses it (markdown_size CHECK). Go refuses it " +
|
||||
"first, so an author gets a message instead of a constraint violation.",
|
||||
"oversized-agent": "" +
|
||||
"JS accepts; the database refuses it (markdown_size CHECK). Go refuses it " +
|
||||
"first, so an author gets a message instead of a constraint violation.",
|
||||
|
||||
"version-above-int32-agent": "" +
|
||||
"JS accepts any whole number of 1 or more; agent_definitions.version is a " +
|
||||
"PostgreSQL `integer`, so the database refuses this one. Go refuses it " +
|
||||
"first, for the same reason as the size bound.",
|
||||
}
|
||||
|
||||
func TestAcceptanceParity(t *testing.T) {
|
||||
seen := map[string]bool{}
|
||||
|
||||
for _, c := range all(load(t)) {
|
||||
t.Run(c.id(), func(t *testing.T) {
|
||||
raw := c.raw(t)
|
||||
|
||||
var err error
|
||||
if c.Kind == "agent" {
|
||||
err = definition.ValidateAgent(raw)
|
||||
} else {
|
||||
err = definition.ValidateSkill(raw)
|
||||
}
|
||||
accepted := err == nil
|
||||
|
||||
if reason, expected := knownDivergence[c.id()]; expected {
|
||||
seen[c.id()] = true
|
||||
if accepted == c.Accepted {
|
||||
t.Errorf("listed as a known divergence but the two now agree (%v).\n"+
|
||||
"Remove it from knownDivergence.\n reason on file: %s", accepted, reason)
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
if accepted != c.Accepted {
|
||||
t.Fatalf("acceptance differs: go=%v js=%v\n go said: %v\n js said: %v",
|
||||
accepted, c.Accepted, err, deref(c.Rejection))
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
for id := range knownDivergence {
|
||||
if !seen[id] {
|
||||
t.Errorf("knownDivergence names %q, which is not in the fixture", id)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Where both refuse a definition, they must refuse it for the same stated
|
||||
// reason. A parser that rejects the right definitions with the wrong messages
|
||||
// sends an author to the wrong line.
|
||||
func TestRejectionMessageParity(t *testing.T) {
|
||||
for _, c := range all(load(t)) {
|
||||
if c.Accepted || c.Rejection == nil {
|
||||
continue
|
||||
}
|
||||
if _, skip := knownDivergence[c.id()]; skip {
|
||||
continue
|
||||
}
|
||||
t.Run(c.id(), func(t *testing.T) {
|
||||
raw := c.raw(t)
|
||||
var err error
|
||||
if c.Kind == "agent" {
|
||||
err = definition.ValidateAgent(raw)
|
||||
} else {
|
||||
err = definition.ValidateSkill(raw)
|
||||
}
|
||||
if err == nil {
|
||||
t.Fatalf("JS rejected this; Go accepted it")
|
||||
}
|
||||
if r, ok := err.(*definition.Rejection); ok && r.BackendOnly {
|
||||
t.Fatalf("refused by a backend-only rule where JS refused it too: %q", r.Message)
|
||||
}
|
||||
if err.Error() != *c.Rejection {
|
||||
t.Errorf("rejection differs\n go %q\n js %q", err.Error(), *c.Rejection)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func deref(s *string) string {
|
||||
if s == nil {
|
||||
return "<accepted>"
|
||||
}
|
||||
return *s
|
||||
}
|
||||
|
||||
/* ── 5. Normalized projection parity ──────────────────────────────────────── */
|
||||
|
||||
// textCoercion is the complete list of fixture cases where the JavaScript
|
||||
// record holds a value that is NOT a string in a field migration 000005
|
||||
// projects into a `text` or `text[]` column, named field by field.
|
||||
//
|
||||
// The frontend can afford this and the backend cannot. `name: [a, b]` leaves a
|
||||
// JavaScript ARRAY on skill.name, and every consumer stringifies it at the
|
||||
// point of use — the trigger list on that very record reads "a,b", which is
|
||||
// String(["a","b"]). A text column has no such option: something must be
|
||||
// written, once, at the boundary. This package writes String(x), which is the
|
||||
// string the frontend's own consumers produce.
|
||||
//
|
||||
// Listing them rather than coercing everywhere is the point. Coercion is
|
||||
// applied ONLY to the fields named here, so a genuine difference between two
|
||||
// strings still fails; each entry is checked to be still necessary, so the list
|
||||
// cannot go stale; and an unlisted case that needs coercion fails outright, so
|
||||
// the list cannot quietly grow. It is the same closed-list discipline
|
||||
// knownDivergence has, for the same reason.
|
||||
var textCoercion = map[string][]string{
|
||||
"invalid-field-type-name-list": {"name"},
|
||||
"name-list-agent": {"name"},
|
||||
"name-numeric": {"name"},
|
||||
"name-boolean": {"name"},
|
||||
"description-list": {"description"},
|
||||
"description-numeric": {"description"},
|
||||
"pages-numeric-entry": {"pages"},
|
||||
"pages-mapping-entry": {"pages"},
|
||||
}
|
||||
|
||||
// The two shapes a text projection can have, named rather than inferred.
|
||||
//
|
||||
// Which one applies is a property of the COLUMN, not of what the author
|
||||
// happened to write. `name` is `text`, so a sequence written there becomes one
|
||||
// string — String(["a","b"]) is "a,b". `pages` is `text[]`, so a sequence stays
|
||||
// a sequence and each entry becomes a string of its own. Inferring the shape
|
||||
// from whatever Go produced would make the test agree with the parser by
|
||||
// construction, which is the one thing it must not do.
|
||||
var textScalarFields = map[string]bool{
|
||||
"name": true, "description": true, "category": true, "trigger": true, "prompt": true,
|
||||
}
|
||||
|
||||
var textListFields = map[string]bool{
|
||||
"pages": true, "actions": true, "triggers": true, "skills": true, "subagents": true,
|
||||
}
|
||||
|
||||
// jsText is String(x) for a value decoded from the fixture's JSON — the same
|
||||
// conversion jsvalue.go performs inside the parser, restated here so the test
|
||||
// does not have to reach into the package it is testing to check it.
|
||||
func jsText(t *testing.T, field string, v any) any {
|
||||
t.Helper()
|
||||
|
||||
switch {
|
||||
case textScalarFields[field]:
|
||||
return jsScalarText(v)
|
||||
case textListFields[field]:
|
||||
list, ok := v.([]any)
|
||||
if !ok {
|
||||
return jsScalarText(v)
|
||||
}
|
||||
out := make([]any, len(list))
|
||||
for i, item := range list {
|
||||
out[i] = jsScalarText(item)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
t.Fatalf("textCoercion names %q, which is not a text-projected field", field)
|
||||
return nil
|
||||
}
|
||||
|
||||
func jsScalarText(v any) string {
|
||||
switch x := v.(type) {
|
||||
case nil:
|
||||
return "null"
|
||||
case bool:
|
||||
if x {
|
||||
return "true"
|
||||
}
|
||||
return "false"
|
||||
case float64:
|
||||
if x == float64(int64(x)) {
|
||||
return strconv.FormatInt(int64(x), 10)
|
||||
}
|
||||
return strconv.FormatFloat(x, 'g', -1, 64)
|
||||
case string:
|
||||
return x
|
||||
case []any:
|
||||
// Array.prototype.toString: nil renders as the empty string, not
|
||||
// "null", which is the one place the two differ.
|
||||
parts := make([]string, len(x))
|
||||
for i, item := range x {
|
||||
if item == nil {
|
||||
continue
|
||||
}
|
||||
parts[i] = jsScalarText(item)
|
||||
}
|
||||
return strings.Join(parts, ",")
|
||||
case map[string]any:
|
||||
return "[object Object]"
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// The fields migration 000005 projects into columns, compared for every
|
||||
// definition both parsers accept. These are the values that reach the
|
||||
// database, so a difference here is a row the frontend would render wrongly.
|
||||
func TestProjectionParity(t *testing.T) {
|
||||
fixture := map[string]bool{}
|
||||
|
||||
for _, c := range all(load(t)) {
|
||||
fixture[c.id()] = true
|
||||
if c.Normalized == nil {
|
||||
continue // JS could not parse it; covered by the tree test
|
||||
}
|
||||
coerce := map[string]bool{}
|
||||
for _, field := range textCoercion[c.id()] {
|
||||
coerce[field] = true
|
||||
}
|
||||
|
||||
t.Run(c.id(), func(t *testing.T) {
|
||||
raw := c.raw(t)
|
||||
|
||||
if c.Kind == "agent" {
|
||||
agent, err := definition.ParseAgent(raw, definition.Options{})
|
||||
if err != nil {
|
||||
t.Fatalf("JS parsed this; Go refused it: %v", err)
|
||||
}
|
||||
compare(t, map[string]any{
|
||||
"id": agent.ID,
|
||||
"name": agent.Name,
|
||||
"description": agent.Description,
|
||||
"status": agent.Status,
|
||||
"version": agent.Version,
|
||||
"pages": agent.Pages,
|
||||
"icon": agent.Icon,
|
||||
"reasoning": agent.Reasoning,
|
||||
"trigger": agent.Trigger,
|
||||
"webSearch": agent.WebSearch,
|
||||
"skills": agent.Skills,
|
||||
"subagents": agent.Subagents,
|
||||
"starters": agent.Starters,
|
||||
"permissions": agent.Permissions,
|
||||
"errors": agent.Errors,
|
||||
}, c.Normalized, coerce)
|
||||
return
|
||||
}
|
||||
|
||||
skill, err := definition.ParseSkill(raw, definition.Options{})
|
||||
if err != nil {
|
||||
t.Fatalf("JS parsed this; Go refused it: %v", err)
|
||||
}
|
||||
compare(t, map[string]any{
|
||||
"id": skill.ID,
|
||||
"name": skill.Name,
|
||||
"description": skill.Description,
|
||||
"status": skill.Status,
|
||||
"pages": skill.Pages,
|
||||
"kind": skill.Kind,
|
||||
"category": skill.Category,
|
||||
"actions": skill.Actions,
|
||||
"triggers": skill.Triggers,
|
||||
"declaredTriggers": skill.DeclaredTriggers,
|
||||
"prompt": skill.Prompt,
|
||||
"skillId": skill.SkillID,
|
||||
}, c.Normalized, coerce)
|
||||
})
|
||||
}
|
||||
|
||||
for id := range textCoercion {
|
||||
if !fixture[id] {
|
||||
t.Errorf("textCoercion names %q, which is not in the fixture", id)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// compare checks every field Go produced against the JS record, field by field
|
||||
// so a failure names the field rather than dumping two objects.
|
||||
func compare(t *testing.T, got map[string]any, want map[string]any, coerce map[string]bool) {
|
||||
t.Helper()
|
||||
|
||||
keys := make([]string, 0, len(got))
|
||||
for k := range got {
|
||||
keys = append(keys, k)
|
||||
}
|
||||
sort.Strings(keys)
|
||||
|
||||
for _, k := range keys {
|
||||
wantValue, present := want[k]
|
||||
if !present {
|
||||
t.Errorf("%s: absent from the JS record", k)
|
||||
continue
|
||||
}
|
||||
if coerce[k] {
|
||||
// Listed in textCoercion. Check the entry is still earning its
|
||||
// place before honouring it: if the JS value is already the string
|
||||
// Go produced, the coercion is doing nothing and the list has gone
|
||||
// stale.
|
||||
coerced := jsText(t, k, normalizeTree(wantValue))
|
||||
if reflect.DeepEqual(normalizeTree(wantValue), normalizeTree(coerced)) {
|
||||
t.Errorf("%s: listed in textCoercion, but the JS value is already "+
|
||||
"a string. Remove the entry.", k)
|
||||
}
|
||||
wantValue = coerced
|
||||
}
|
||||
|
||||
g, w := normalizeTree(got[k]), normalizeTree(wantValue)
|
||||
// An empty list and a missing one are the same thing to both parsers.
|
||||
if isEmptyList(g) && isEmptyList(w) {
|
||||
continue
|
||||
}
|
||||
if !reflect.DeepEqual(g, w) {
|
||||
t.Errorf("%s differs\n go %s\n js %s", k, show(g), show(w))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func isEmptyList(v any) bool {
|
||||
if v == nil {
|
||||
return true
|
||||
}
|
||||
l, ok := v.([]any)
|
||||
return ok && len(l) == 0
|
||||
}
|
||||
|
||||
/* ── 6. The parser never rewrites what is stored ──────────────────────────── */
|
||||
|
||||
// Migration 000005 keeps `markdown` verbatim and derives every other column
|
||||
// from it. Parsing must therefore be a read: normalization exists to
|
||||
// INTERPRET a definition, never to rewrite it.
|
||||
func TestParsingDoesNotMutateSource(t *testing.T) {
|
||||
for _, c := range all(load(t)) {
|
||||
raw := c.raw(t)
|
||||
before := string(append([]byte{}, raw...))
|
||||
|
||||
_, _ = definition.ParseSkill(raw, definition.Options{})
|
||||
_, _ = definition.ParseAgent(raw, definition.Options{})
|
||||
_ = definition.ValidateSkill(raw)
|
||||
_ = definition.ValidateAgent(raw)
|
||||
|
||||
if raw != before {
|
||||
t.Fatalf("%s: the source changed under the parser", c.id())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Normalize is what the parser reads THROUGH; what it returns must never be
|
||||
// what gets stored. Asserted directly, because the whole separation rests on
|
||||
// it: the corpus contains definitions whose normalized form differs from their
|
||||
// stored form, and storing the normalized one would silently rewrite an
|
||||
// author's file.
|
||||
func TestNormalizationIsNotStorage(t *testing.T) {
|
||||
rewritten := 0
|
||||
for _, c := range all(load(t)) {
|
||||
raw := c.raw(t)
|
||||
if definition.Normalize(raw) != raw {
|
||||
rewritten++
|
||||
}
|
||||
}
|
||||
if rewritten == 0 {
|
||||
t.Fatal("no case in the corpus is changed by Normalize; " +
|
||||
"this test can no longer tell storage and interpretation apart")
|
||||
}
|
||||
t.Logf("%d of %d definitions normalize to something other than their stored bytes", rewritten, len(all(load(t))))
|
||||
}
|
||||
|
||||
/* ── 7. Adversarial coverage is real ──────────────────────────────────────── */
|
||||
|
||||
// The adversarial cases Phase 4D requires, each mapped to the fixture rows that
|
||||
// exercise it. A case list that drifts away from the requirement is a suite
|
||||
// that looks thorough and tests something else.
|
||||
func TestAdversarialCoverage(t *testing.T) {
|
||||
required := map[string][]string{
|
||||
"UTF-8 BOM": {"utf8-bom", "utf8-bom-agent", "bom-crlf-blankline"},
|
||||
"CRLF": {"crlf", "crlf-agent", "crlf-inside-frontmatter-only"},
|
||||
"CR": {"cr-only"},
|
||||
"leading blank line": {"leading-blank-line"},
|
||||
"multiple leading blank lines": {"multiple-leading-blank-lines", "leading-spaces-then-blank-lines"},
|
||||
"trailing spaces": {"trailing-spaces-on-values"},
|
||||
"trailing newline": {"many-trailing-newlines", "no-trailing-newline"},
|
||||
"trailing ws after fence": {"trailing-ws-after-open-fence", "trailing-tab-after-close-fence"},
|
||||
"quoted scalar": {"double-quoted-scalar", "doubled-quote-escape"},
|
||||
"single-quoted scalar": {"single-quoted-scalar"},
|
||||
"colon inside quoted string": {"colon-in-quoted-string", "colon-in-unquoted-string"},
|
||||
"hash inside quoted string": {"hash-in-quoted-string", "hash-unquoted-trailing-comment", "hash-unquoted-midword"},
|
||||
"empty scalar": {"empty-scalar", "tilde-scalar", "null-scalar"},
|
||||
"empty array": {"empty-array"},
|
||||
"inline array": {"inline-flow-array", "inline-flow-map"},
|
||||
"multiline scalar": {"block-scalar-literal", "block-scalar-folded"},
|
||||
"duplicate key": {"duplicate-key", "duplicate-key-array"},
|
||||
"malformed YAML": {"malformed-yaml-bare-line", "key-with-space", "ragged-indent"},
|
||||
"malformed opening fence": {"malformed-open-fence-two-dashes", "malformed-open-fence-four-dashes",
|
||||
"malformed-open-fence-indented", "malformed-open-fence-text-after"},
|
||||
"malformed closing fence": {"malformed-close-fence-two-dashes", "malformed-close-fence-missing",
|
||||
"malformed-close-fence-four-dashes"},
|
||||
"missing frontmatter": {"missing-frontmatter", "empty-fence-pair", "frontmatter-is-a-sequence"},
|
||||
"unsupported frontmatter field": {"unsupported-frontmatter-field", "unsupported-field-agent", "uppercase-key"},
|
||||
"invalid field type": {"invalid-field-type-pages-scalar", "invalid-field-type-pages-scalar-agent",
|
||||
"invalid-field-type-name-list"},
|
||||
"invalid definition id": {"invalid-definition-id-uppercase", "invalid-definition-id-leading-dash",
|
||||
"invalid-definition-id-underscore"},
|
||||
"invalid status": {"invalid-status-skill", "invalid-status-agent", "inactive-status-skill"},
|
||||
"invalid visibility": {"visibility-field-personal", "visibility-field-invalid"},
|
||||
"oversized markdown": {"oversized-markdown", "oversized-agent", "at-size-bound"},
|
||||
"empty markdown": {"empty-markdown", "whitespace-only-markdown"},
|
||||
}
|
||||
|
||||
present := map[string]bool{}
|
||||
for _, c := range load(t).Cases {
|
||||
present[c.Name] = true
|
||||
}
|
||||
|
||||
for requirement, names := range required {
|
||||
for _, n := range names {
|
||||
if !present[n] {
|
||||
t.Errorf("%q: the fixture has no case named %q", requirement, n)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* ── 8. Deferred blocks are reported, not assumed ─────────────────────────── */
|
||||
|
||||
// Every skill carrying a `ui:` or `owliver:` block must say so, because that is
|
||||
// the one part of validation this package does not do. A block that stopped
|
||||
// being reported would be a gap nobody could see.
|
||||
func TestDeferredBlocksAreReported(t *testing.T) {
|
||||
o := load(t)
|
||||
found := 0
|
||||
|
||||
for _, c := range append(append([]observation{}, o.Corpus...), o.Cases...) {
|
||||
if c.Kind != "skill" || !c.Frontmatter.OK {
|
||||
continue
|
||||
}
|
||||
want := []string{}
|
||||
for _, key := range []string{"ui", "owliver"} {
|
||||
if _, present := c.Frontmatter.Data[key]; present {
|
||||
want = append(want, key)
|
||||
}
|
||||
}
|
||||
|
||||
skill, err := definition.ParseSkill(c.raw(t), definition.Options{})
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
if len(want) == 0 {
|
||||
if len(skill.Deferred) != 0 {
|
||||
t.Errorf("%s: reported Deferred %v with no such block", c.id(), skill.Deferred)
|
||||
}
|
||||
continue
|
||||
}
|
||||
found++
|
||||
if !reflect.DeepEqual(skill.Deferred, want) {
|
||||
t.Errorf("%s: Deferred = %v, want %v", c.id(), skill.Deferred, want)
|
||||
}
|
||||
}
|
||||
|
||||
if found != 19 {
|
||||
t.Errorf("definitions carrying a deferred block: got %d, want 19", found)
|
||||
}
|
||||
}
|
||||
|
||||
/* ── 9. Mutation checks ───────────────────────────────────────────────────── */
|
||||
|
||||
// Tests that pass against a broken parser are not tests. Each mutation below
|
||||
// is a plausible mistake in this package; every one must be caught by a real
|
||||
// definition changing its meaning, not by an assertion written to notice it.
|
||||
func TestMutationsWouldBeCaught(t *testing.T) {
|
||||
base := strings.Join([]string{
|
||||
"---",
|
||||
"id: sample-skill",
|
||||
"name: Sample Skill",
|
||||
"description: A sample.",
|
||||
"pages:",
|
||||
" - candidates",
|
||||
"---",
|
||||
"",
|
||||
"# Sample Skill",
|
||||
}, "\n")
|
||||
|
||||
mutations := []struct {
|
||||
name string
|
||||
raw string
|
||||
check func(t *testing.T, s *definition.Skill, err error)
|
||||
}{
|
||||
{
|
||||
// Dropping the BOM strip: the fence stops matching and every field
|
||||
// empties out.
|
||||
name: "BOM before the fence still fences",
|
||||
raw: "\uFEFF" + base,
|
||||
check: func(t *testing.T, s *definition.Skill, err error) {
|
||||
if err != nil || s.ID != "sample-skill" || len(s.Pages) != 1 {
|
||||
t.Errorf("got id=%q pages=%v err=%v", s.ID, s.Pages, err)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
// Dropping CR normalization: `candidates\r` is not a page.
|
||||
name: "CRLF endings do not leak into values",
|
||||
raw: strings.ReplaceAll(base, "\n", "\r\n"),
|
||||
check: func(t *testing.T, s *definition.Skill, err error) {
|
||||
if err != nil || len(s.Pages) != 1 || s.Pages[0] != "candidates" {
|
||||
t.Errorf("got pages=%v err=%v", s.Pages, err)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
// Trimming the closing fence too eagerly, or not at all.
|
||||
name: "trailing tab after the closing fence still closes it",
|
||||
raw: strings.Replace(base, "\n---\n", "\n---\t\n", 1),
|
||||
check: func(t *testing.T, s *definition.Skill, err error) {
|
||||
if err != nil || s.Name != "Sample Skill" {
|
||||
t.Errorf("got name=%q err=%v", s.Name, err)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
// A greedy fence would swallow the second document and lose the id.
|
||||
name: "a second --- document is body, not frontmatter",
|
||||
raw: base + "\n\n---\nid: second\n---\n",
|
||||
check: func(t *testing.T, s *definition.Skill, err error) {
|
||||
if err != nil || s.ID != "sample-skill" {
|
||||
t.Errorf("got id=%q err=%v", s.ID, err)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
// Treating `#` as always starting a comment.
|
||||
name: "a hash inside a word is part of the word",
|
||||
raw: strings.Replace(base, "description: A sample.", "category: ops#1", 1),
|
||||
check: func(t *testing.T, s *definition.Skill, err error) {
|
||||
if err != nil || s.Category != "ops#1" {
|
||||
t.Errorf("got category=%q err=%v", s.Category, err)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
// Treating a spaced `#` as part of the value.
|
||||
name: "a spaced hash starts a comment",
|
||||
raw: strings.Replace(base, "description: A sample.", "category: ops # note", 1),
|
||||
check: func(t *testing.T, s *definition.Skill, err error) {
|
||||
if err != nil || s.Category != "ops" {
|
||||
t.Errorf("got category=%q err=%v", s.Category, err)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
// Splitting a quoted value on its colon.
|
||||
name: "a colon inside quotes stays in the value",
|
||||
raw: strings.Replace(base, "name: Sample Skill", `name: "Sample: Skill"`, 1),
|
||||
check: func(t *testing.T, s *definition.Skill, err error) {
|
||||
if err != nil || s.Name != "Sample: Skill" {
|
||||
t.Errorf("got name=%q err=%v", s.Name, err)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
// Keeping the first duplicate rather than the last.
|
||||
name: "a duplicate key takes the last value",
|
||||
raw: strings.Replace(base, "name: Sample Skill", "name: First\nname: Second", 1),
|
||||
check: func(t *testing.T, s *definition.Skill, err error) {
|
||||
if err != nil || s.Name != "Second" {
|
||||
t.Errorf("got name=%q err=%v", s.Name, err)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
// Accepting ragged indentation instead of refusing it.
|
||||
name: "ragged indentation is refused with its line",
|
||||
raw: strings.Replace(base, " - candidates", " - candidates\n - positions", 1),
|
||||
check: func(t *testing.T, s *definition.Skill, err error) {
|
||||
var pe *definition.Error
|
||||
if err == nil {
|
||||
t.Fatalf("accepted ragged indentation: %+v", s)
|
||||
}
|
||||
if !asError(err, &pe) || pe.Line != 6 {
|
||||
t.Errorf("got %v, want an *Error on line 6", err)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
// Canonicalising a skill's pages, which the frontend does not do.
|
||||
name: "a skill keeps the page name as written",
|
||||
raw: strings.Replace(base, " - candidates", " - Talent Pool", 1),
|
||||
check: func(t *testing.T, s *definition.Skill, err error) {
|
||||
if err != nil || len(s.Pages) != 1 || s.Pages[0] != "Talent Pool" {
|
||||
t.Errorf("got pages=%v err=%v", s.Pages, err)
|
||||
}
|
||||
if err := definition.ValidateSkill(strings.Replace(base, " - candidates", " - Talent Pool", 1)); err != nil {
|
||||
t.Errorf("an aliased page should still validate: %v", err)
|
||||
}
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
for _, m := range mutations {
|
||||
t.Run(m.name, func(t *testing.T) {
|
||||
skill, err := definition.ParseSkill(m.raw, definition.Options{})
|
||||
if skill == nil {
|
||||
skill = &definition.Skill{}
|
||||
}
|
||||
m.check(t, skill, err)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// asError is errors.As, spelled out for the one concrete type this package
|
||||
// returns.
|
||||
func asError(err error, target **definition.Error) bool {
|
||||
e, ok := err.(*definition.Error)
|
||||
if ok {
|
||||
*target = e
|
||||
}
|
||||
return ok
|
||||
}
|
||||
|
||||
/* ── 10. Bounds ───────────────────────────────────────────────────────────── */
|
||||
|
||||
// The size bound is the backend's, and both directions of it matter: a
|
||||
// definition at the limit must be storable and one character more must not.
|
||||
// The companion to TestSizeBound, for the other backend-only bound. The
|
||||
// fixture pins a version well above the bound and one exactly at it, which
|
||||
// leaves the step between them untested — an off-by-one there would refuse a
|
||||
// version PostgreSQL can store, or accept one it cannot. Both sides of the
|
||||
// step are named here so that cannot happen.
|
||||
func TestVersionBound(t *testing.T) {
|
||||
agent := func(version string) string {
|
||||
return "---\nid: sample-agent\nname: Sample Agent\npages:\n - candidates\n" +
|
||||
"version: " + version + "\n---\n\n# Sample Agent\n"
|
||||
}
|
||||
|
||||
at := strconv.Itoa(definition.MaxVersion)
|
||||
if err := definition.ValidateAgent(agent(at)); err != nil {
|
||||
t.Errorf("version %s, exactly at the bound, was refused: %v", at, err)
|
||||
}
|
||||
|
||||
over := strconv.FormatInt(int64(definition.MaxVersion)+1, 10)
|
||||
err := definition.ValidateAgent(agent(over))
|
||||
if err == nil {
|
||||
t.Fatalf("version %s, one past the bound, was accepted", over)
|
||||
}
|
||||
r, ok := err.(*definition.Rejection)
|
||||
if !ok || !r.BackendOnly {
|
||||
t.Errorf("the version bound should be reported as a backend-only rule, got %v", err)
|
||||
}
|
||||
|
||||
// The bound belongs to validation, not to parsing: a version the database
|
||||
// cannot store must still normalize to the number the author wrote, or the
|
||||
// record the editor shows and the record Go builds would disagree.
|
||||
parsed, err := definition.ParseAgent(agent(over), definition.Options{})
|
||||
if err != nil {
|
||||
t.Fatalf("parsing a too-large version failed: %v", err)
|
||||
}
|
||||
if got := strconv.Itoa(parsed.Version); got != over {
|
||||
t.Errorf("parse clamped the version to %s; it should carry %s", got, over)
|
||||
}
|
||||
if len(parsed.Errors) != 0 {
|
||||
t.Errorf("parse reported a backend-only bound as an authoring error: %v", parsed.Errors)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSizeBound(t *testing.T) {
|
||||
head := "---\nid: sample-skill\nname: Sample Skill\npages:\n - candidates\n---\n\n"
|
||||
|
||||
at := head + strings.Repeat("y", definition.MaxMarkdownLength-len(head))
|
||||
if n := len([]rune(at)); n != definition.MaxMarkdownLength {
|
||||
t.Fatalf("fixture is %d characters, wanted exactly %d", n, definition.MaxMarkdownLength)
|
||||
}
|
||||
if err := definition.ValidateSkill(at); err != nil {
|
||||
t.Errorf("a definition exactly at the bound was refused: %v", err)
|
||||
}
|
||||
|
||||
over := at + "y"
|
||||
err := definition.ValidateSkill(over)
|
||||
if err == nil {
|
||||
t.Fatal("a definition one character over the bound was accepted")
|
||||
}
|
||||
r, ok := err.(*definition.Rejection)
|
||||
if !ok || !r.BackendOnly {
|
||||
t.Errorf("the size bound should be reported as a backend-only rule, got %v", err)
|
||||
}
|
||||
}
|
||||
102
go-api/internal/definition/definition.go
Normal file
102
go-api/internal/definition/definition.go
Normal file
@@ -0,0 +1,102 @@
|
||||
// Package definition reads Krow agent and skill definitions — Markdown with
|
||||
// YAML frontmatter — the way the frontend reads them.
|
||||
//
|
||||
// # Why this exists
|
||||
//
|
||||
// A definition is authored in the browser and stored by the server, so two
|
||||
// parsers see it: the JavaScript in src/lib/skills and src/lib/agents, and
|
||||
// this one. If they disagree, one of two things happens, and both are silent:
|
||||
//
|
||||
// - A definition the editor accepts and this package rejects looks valid
|
||||
// while it is being written and fails when it is saved.
|
||||
// - A definition this package accepts and the editor rejects is stored and
|
||||
// then cannot be rendered by the product that owns it.
|
||||
//
|
||||
// Compatibility is therefore a contract rather than an aspiration, and it is
|
||||
// enforced by a conformance suite (conformance_test.go) that replays the
|
||||
// ACTUAL output of the JavaScript parser — captured from the real frontend
|
||||
// module graph — against this one, over all 37 shipped definitions and every
|
||||
// adversarial case in testdata/oracle.json. The corpus count is pinned by a
|
||||
// test; the case count is deliberately not restated here, because a number
|
||||
// kept in a comment is a number that goes stale.
|
||||
//
|
||||
// # What is in the contract
|
||||
//
|
||||
// - The document layer: byte-order mark, line endings, leading blank lines,
|
||||
// fence recognition, body extraction. See frontmatter.go.
|
||||
// - The YAML subset: block maps and sequences, scalars, quoting, comments.
|
||||
// A port of yaml.js, with no YAML dependency, deliberately — see yaml.go.
|
||||
// - Definition-level normalization and validation: id, name, description,
|
||||
// status, version, pages, icons, reasoning, permissions, starters,
|
||||
// knowledge, subagents.
|
||||
//
|
||||
// Those cover every column migration 000005 projects out of a definition:
|
||||
// definition_id, status, version, name, description, pages.
|
||||
//
|
||||
// # What is deferred, and why
|
||||
//
|
||||
// A skill may carry a `ui:` block (declarative page sections) or an `owliver:`
|
||||
// block (assistant capabilities). Validating those means reproducing roughly
|
||||
// 1,500 lines of closed vocabulary describing what the FRONTEND can render —
|
||||
// placements, data sources, section types, periods — none of which the backend
|
||||
// stores, projects, or acts on.
|
||||
//
|
||||
// This package therefore does not check them. It records their presence on
|
||||
// Skill.Deferred instead, so the gap is a value a caller can see rather than
|
||||
// an assumption. The one consequence is stated exactly:
|
||||
//
|
||||
// skill-examples/board-invalid-context.md is rejected by the frontend, on a
|
||||
// rule about which placement can supply which data source, and accepted
|
||||
// here. It is the only definition in the corpus where the two disagree, and
|
||||
// the conformance suite asserts that it stays the only one.
|
||||
//
|
||||
// # What is not the parser's job
|
||||
//
|
||||
// Normalization never rewrites what is stored. Migration 000005 keeps
|
||||
// `markdown` verbatim and every other column is derived from it; this package
|
||||
// only ever reads. The Markdown handed in is the Markdown that goes to the
|
||||
// database, byte for byte, and a test asserts it.
|
||||
//
|
||||
// Visibility (personal or organization) is deliberately absent. It is not a
|
||||
// frontmatter field — the frontend ignores `visibility:` in a definition
|
||||
// entirely — it is a storage tier chosen by the request and checked by the
|
||||
// visibility CHECK in migration 000005. A definition cannot name its own
|
||||
// tenancy.
|
||||
package definition
|
||||
|
||||
// MaxMarkdownLength is the markdown_size CHECK from migration 000005, in
|
||||
// CHARACTERS — `length()` in PostgreSQL counts characters, not bytes.
|
||||
//
|
||||
// The frontend does NOT enforce this, so a definition longer than this is one
|
||||
// the editor accepts and the database refuses. This package refuses it first,
|
||||
// which turns a constraint violation into a message an author can act on.
|
||||
const MaxMarkdownLength = 65536
|
||||
|
||||
// MaxVersion is the range of agent_definitions.version, a PostgreSQL
|
||||
// `integer`.
|
||||
//
|
||||
// The frontend accepts any integer of 1 or more, so a version above this is
|
||||
// another value the editor accepts and the database cannot store.
|
||||
const MaxVersion = 2147483647
|
||||
|
||||
// maxExactInteger is 2^53-1, the largest integer a float64 names exactly and so
|
||||
// the largest a JavaScript number carries without loss. It bounds the version
|
||||
// conversion in ParseAgent; it is not a rule about what may be stored, which is
|
||||
// MaxVersion's job.
|
||||
const maxExactInteger = 1<<53 - 1
|
||||
|
||||
// Rejection is a definition that parses but may not be stored.
|
||||
//
|
||||
// Message is the frontend's own wording wherever the rule is shared, so the
|
||||
// editor and the API describe the same problem the same way.
|
||||
type Rejection struct {
|
||||
Message string
|
||||
|
||||
// BackendOnly marks a rule the frontend does not have — a bound the
|
||||
// database imposes that the editor never checks. These are the only
|
||||
// messages that can differ from what an author would see in the browser,
|
||||
// and each one is listed in docs/phase-4d-parser-contract.md.
|
||||
BackendOnly bool
|
||||
}
|
||||
|
||||
func (r *Rejection) Error() string { return r.Message }
|
||||
332
go-api/internal/definition/frontmatter.go
Normal file
332
go-api/internal/definition/frontmatter.go
Normal file
@@ -0,0 +1,332 @@
|
||||
package definition
|
||||
|
||||
import (
|
||||
"strings"
|
||||
)
|
||||
|
||||
// The document layer: what is frontmatter, what is body, and what a `## Heading`
|
||||
// section contains.
|
||||
//
|
||||
// A port of the four exported readers in src/lib/skills/registry.js —
|
||||
// normalizeDefinition, hasFrontmatter, parseFrontmatter and the section
|
||||
// readers. Agent and skill definitions are read by the same code on the
|
||||
// frontend, deliberately, so that the two formats cannot drift; the same is
|
||||
// true here.
|
||||
//
|
||||
// The regular expressions the JavaScript uses are hand-rolled rather than
|
||||
// translated, because two of them rely on lookahead and lazy matching that RE2
|
||||
// does not have. Each is written out below with the JavaScript it reproduces.
|
||||
|
||||
// Normalize is the frontend's `normalizeDefinition`: a definition's text as the
|
||||
// parser needs to see it.
|
||||
//
|
||||
// Files arrive from editors, from Windows, from copy-paste and from downloads,
|
||||
// and four of the things they arrive with used to take the whole frontmatter
|
||||
// block down — a UTF-8 byte-order mark before the opening fence, blank lines
|
||||
// above it, CRLF endings, and trailing spaces after `---`. In each case the
|
||||
// fence did not match and the definition registered as untitled with no pages.
|
||||
//
|
||||
// This is NOT a lenient parser. The subset inside the fences is exactly as
|
||||
// strict as it was. This is only about recognising that a fence is a fence.
|
||||
//
|
||||
// String(raw ?? '')
|
||||
// .replace(/^\uFEFF/, '')
|
||||
// .replace(/\r\n?/g, '\n')
|
||||
// .replace(/^\s*\n+/, '')
|
||||
//
|
||||
// The order is load bearing: the BOM goes first so it cannot be counted as the
|
||||
// leading whitespace, and CR normalisation goes before the blank-line strip so
|
||||
// that a CRLF blank line is one.
|
||||
func Normalize(raw string) string {
|
||||
text := strings.TrimPrefix(raw, "\uFEFF")
|
||||
|
||||
// `\r\n?` → `\n`: a CRLF pair and a lone CR both become one newline.
|
||||
if strings.IndexByte(text, '\r') >= 0 {
|
||||
var b strings.Builder
|
||||
b.Grow(len(text))
|
||||
for i := 0; i < len(text); i++ {
|
||||
if text[i] != '\r' {
|
||||
b.WriteByte(text[i])
|
||||
continue
|
||||
}
|
||||
b.WriteByte('\n')
|
||||
if i+1 < len(text) && text[i+1] == '\n' {
|
||||
i++
|
||||
}
|
||||
}
|
||||
text = b.String()
|
||||
}
|
||||
|
||||
// `^\s*\n+` → ``. Greedy `\s*` then at least one newline: the effect is to
|
||||
// drop the leading whitespace run up to and including its LAST newline, and
|
||||
// to drop nothing at all when that run contains no newline. A definition
|
||||
// indented by one space is therefore still unfenced, which is what the
|
||||
// editor decides too.
|
||||
end, last := 0, -1
|
||||
for i, r := range text {
|
||||
if !jsIsSpace(r) {
|
||||
break
|
||||
}
|
||||
if r == '\n' {
|
||||
last = i
|
||||
}
|
||||
end = i + len(string(r))
|
||||
}
|
||||
_ = end
|
||||
if last >= 0 {
|
||||
text = text[last+1:]
|
||||
}
|
||||
|
||||
return text
|
||||
}
|
||||
|
||||
// fence locates the frontmatter block in already-normalized text.
|
||||
//
|
||||
// /^---[ \t]*\n([\s\S]*?)\n---[ \t]*(?=\n|$)/
|
||||
//
|
||||
// Returns the YAML source, the offset just past the closing fence, and whether
|
||||
// there was one. Lazy: the FIRST closing fence wins, which is why a definition
|
||||
// carrying a second `---` document keeps only the first and reads the rest as
|
||||
// body.
|
||||
func fence(text string) (yaml string, end int, ok bool) {
|
||||
if !strings.HasPrefix(text, "---") {
|
||||
return "", 0, false
|
||||
}
|
||||
i := 3
|
||||
for i < len(text) && (text[i] == ' ' || text[i] == '\t') {
|
||||
i++
|
||||
}
|
||||
if i >= len(text) || text[i] != '\n' {
|
||||
return "", 0, false
|
||||
}
|
||||
start := i + 1
|
||||
|
||||
for at := start - 1; at >= 0 && at < len(text); {
|
||||
nl := strings.IndexByte(text[at+1:], '\n')
|
||||
if nl < 0 {
|
||||
return "", 0, false
|
||||
}
|
||||
at = at + 1 + nl // index of the newline that must precede the fence
|
||||
|
||||
rest := text[at+1:]
|
||||
if !strings.HasPrefix(rest, "---") {
|
||||
continue
|
||||
}
|
||||
j := 3
|
||||
for j < len(rest) && (rest[j] == ' ' || rest[j] == '\t') {
|
||||
j++
|
||||
}
|
||||
// `(?=\n|$)` — end of the document, or the end of this line. `$` has no
|
||||
// multiline flag on the frontend either, so it means end of document.
|
||||
if j < len(rest) && rest[j] != '\n' {
|
||||
continue
|
||||
}
|
||||
return text[start:at], at + 1 + j, true
|
||||
}
|
||||
return "", 0, false
|
||||
}
|
||||
|
||||
// HasFrontmatter reports whether this text opens with a frontmatter block at
|
||||
// all. It does not say whether that block parses.
|
||||
func HasFrontmatter(raw string) bool {
|
||||
_, _, ok := fence(Normalize(raw))
|
||||
return ok
|
||||
}
|
||||
|
||||
// Document is a definition split into its two halves.
|
||||
type Document struct {
|
||||
// Data is the frontmatter as plain data. Always a mapping: a frontmatter
|
||||
// block that parses to a sequence is discarded, exactly as the frontend
|
||||
// discards it, because every reader downstream indexes it by key.
|
||||
Data map[string]any
|
||||
|
||||
// Body is everything after the closing fence, trimmed. A document with no
|
||||
// frontmatter is all body.
|
||||
Body string
|
||||
|
||||
// Fenced records whether a frontmatter block was found, which Data alone
|
||||
// cannot express — an empty fence pair and a missing one both give an
|
||||
// empty mapping.
|
||||
Fenced bool
|
||||
}
|
||||
|
||||
// ParseFrontmatter splits a definition and reads its frontmatter.
|
||||
//
|
||||
// Returns a *Error when the YAML subset refuses a line. A document with no
|
||||
// recognisable fence is NOT an error: it is a document with no frontmatter,
|
||||
// and what happens to it is the validator's decision — the same division the
|
||||
// frontend makes.
|
||||
func ParseFrontmatter(raw string) (Document, error) {
|
||||
text := Normalize(raw)
|
||||
|
||||
yaml, end, ok := fence(text)
|
||||
if !ok {
|
||||
return Document{Data: map[string]any{}, Body: text, Fenced: false}, nil
|
||||
}
|
||||
|
||||
value, err := ParseYAML(yaml)
|
||||
if err != nil {
|
||||
return Document{}, err
|
||||
}
|
||||
|
||||
data, _ := value.(map[string]any)
|
||||
if data == nil {
|
||||
data = map[string]any{}
|
||||
}
|
||||
|
||||
return Document{Data: data, Body: jsTrim(text[end:]), Fenced: true}, nil
|
||||
}
|
||||
|
||||
// sectionSource is the text under a `## Heading`, up to the next one.
|
||||
//
|
||||
// new RegExp(`##\\s+${escaped}\\s*\\n([\\s\\S]*?)(?=\\n##\\s|$)`, 'i')
|
||||
//
|
||||
// Case-insensitive, and deliberately not anchored to the start of a line —
|
||||
// that is what the frontend does. The heading is compared literally rather
|
||||
// than compiled into a pattern, which is the same protection the frontend gets
|
||||
// by escaping it: a heading containing regular-expression punctuation must
|
||||
// match the words it is built from.
|
||||
//
|
||||
// Returns ok=false for a section that is not there, which is a different thing
|
||||
// from a section that is there and empty.
|
||||
func sectionSource(body, heading string) (string, bool) {
|
||||
lower := strings.ToLower(body)
|
||||
want := strings.ToLower(heading)
|
||||
|
||||
for at := 0; ; {
|
||||
h := strings.Index(lower[at:], "##")
|
||||
if h < 0 {
|
||||
return "", false
|
||||
}
|
||||
h += at
|
||||
at = h + 2
|
||||
|
||||
// `##` then `\s+` then the heading.
|
||||
i := h + 2
|
||||
gap := i
|
||||
for i < len(body) {
|
||||
r, size := decodeRune(body[i:])
|
||||
if !jsIsSpace(r) {
|
||||
break
|
||||
}
|
||||
i += size
|
||||
}
|
||||
if i == gap {
|
||||
continue // `\s+` needs at least one
|
||||
}
|
||||
if !strings.HasPrefix(lower[i:], want) {
|
||||
continue
|
||||
}
|
||||
i += len(want)
|
||||
|
||||
// `\s*\n`: a whitespace run that ends in a newline.
|
||||
j, nl := i, -1
|
||||
for j < len(body) {
|
||||
r, size := decodeRune(body[j:])
|
||||
if !jsIsSpace(r) {
|
||||
break
|
||||
}
|
||||
if r == '\n' {
|
||||
nl = j
|
||||
break
|
||||
}
|
||||
j += size
|
||||
}
|
||||
if nl < 0 {
|
||||
continue
|
||||
}
|
||||
|
||||
start := nl + 1
|
||||
// `(?=\n##\s|$)`, lazily: the first following line that opens a new
|
||||
// `##` heading. `###` does not, because the character after `##` must
|
||||
// be whitespace.
|
||||
for k := start; ; {
|
||||
n := strings.Index(body[k:], "\n##")
|
||||
if n < 0 {
|
||||
return body[start:], true
|
||||
}
|
||||
n += k
|
||||
after := n + 3
|
||||
if after < len(body) {
|
||||
r, _ := decodeRune(body[after:])
|
||||
if jsIsSpace(r) {
|
||||
return body[start:n], true
|
||||
}
|
||||
}
|
||||
k = n + 1
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// decodeRune is utf8.DecodeRuneInString, kept local so the section reader has
|
||||
// one obvious way to step through the body.
|
||||
func decodeRune(s string) (rune, int) {
|
||||
for i, r := range s {
|
||||
_ = i
|
||||
return r, len(string(r))
|
||||
}
|
||||
return 0, 0
|
||||
}
|
||||
|
||||
// SectionText is the prose under a `## Heading`, with its bullets and blank
|
||||
// lines flattened to one line.
|
||||
//
|
||||
// Used for a workforce level's description, which is a sentence rather than a
|
||||
// list.
|
||||
func SectionText(body, heading string) string {
|
||||
source, ok := sectionSource(body, heading)
|
||||
if !ok {
|
||||
return ""
|
||||
}
|
||||
parts := []string{}
|
||||
for _, l := range strings.Split(source, "\n") {
|
||||
l = jsTrim(stripListMarker(l))
|
||||
if l == "" {
|
||||
continue
|
||||
}
|
||||
parts = append(parts, l)
|
||||
}
|
||||
return jsTrim(strings.Join(parts, " "))
|
||||
}
|
||||
|
||||
// stripListMarker removes a leading `-`, `*` or `1.` / `1)` bullet.
|
||||
//
|
||||
// /^\s*(?:[-*]|\d+[.)])\s+/
|
||||
func stripListMarker(l string) string {
|
||||
i := 0
|
||||
for i < len(l) {
|
||||
r, size := decodeRune(l[i:])
|
||||
if !jsIsSpace(r) {
|
||||
break
|
||||
}
|
||||
i += size
|
||||
}
|
||||
marker := i
|
||||
switch {
|
||||
case i < len(l) && (l[i] == '-' || l[i] == '*'):
|
||||
i++
|
||||
default:
|
||||
digits := i
|
||||
for i < len(l) && l[i] >= '0' && l[i] <= '9' {
|
||||
i++
|
||||
}
|
||||
if i == digits || i >= len(l) || (l[i] != '.' && l[i] != ')') {
|
||||
return l
|
||||
}
|
||||
i++
|
||||
}
|
||||
// `\s+` after the marker is required; without it there is no list item.
|
||||
space := i
|
||||
for i < len(l) {
|
||||
r, size := decodeRune(l[i:])
|
||||
if !jsIsSpace(r) {
|
||||
break
|
||||
}
|
||||
i += size
|
||||
}
|
||||
if i == space {
|
||||
return l
|
||||
}
|
||||
_ = marker
|
||||
return l[i:]
|
||||
}
|
||||
210
go-api/internal/definition/jsvalue.go
Normal file
210
go-api/internal/definition/jsvalue.go
Normal file
@@ -0,0 +1,210 @@
|
||||
package definition
|
||||
|
||||
import (
|
||||
"math"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// JavaScript value semantics, reproduced exactly.
|
||||
//
|
||||
// The frontend parser is JavaScript, and the compatibility contract is with
|
||||
// THAT parser, not with an idealised YAML. Three of its behaviours are load
|
||||
// bearing and none of them are Go's defaults:
|
||||
//
|
||||
// - `\s` and `String.prototype.trim` cover a different set of code points
|
||||
// than `unicode.IsSpace`. JS treats U+FEFF as whitespace and U+0085 as
|
||||
// not; Go is the other way round. A definition is trimmed on the way
|
||||
// through the parser at least four times, so the difference is reachable.
|
||||
// - Truthiness decides whether `id:` is used or derived, whether a name
|
||||
// falls back to `Untitled skill`, and what `prompt:` becomes. `0`, `false`
|
||||
// and `""` are falsy; `"0"` and `[]` are not.
|
||||
// - `String(x)` and `Number(x)` have defined results for every type, and the
|
||||
// validator interpolates them into messages an author reads. `[1,2]`
|
||||
// stringifies to `1,2`, an object to `[object Object]`.
|
||||
//
|
||||
// Reimplementing these is not gold-plating: each one is exercised by a case in
|
||||
// the conformance suite because each one is reachable from a definition an
|
||||
// author could write.
|
||||
|
||||
// jsIsSpace reports whether r is whitespace to JavaScript — the union of
|
||||
// WhiteSpace and LineTerminator in the specification.
|
||||
//
|
||||
// Deliberately NOT unicode.IsSpace: that set includes U+0085 (NEL), which JS
|
||||
// does not, and excludes U+FEFF, which JS does.
|
||||
func jsIsSpace(r rune) bool {
|
||||
switch r {
|
||||
case '\t', '\n', '\v', '\f', '\r', ' ',
|
||||
0x00A0, 0x1680, 0x2028, 0x2029, 0x202F, 0x205F, 0x3000, 0xFEFF:
|
||||
return true
|
||||
}
|
||||
return r >= 0x2000 && r <= 0x200A
|
||||
}
|
||||
|
||||
// jsTrim is String.prototype.trim.
|
||||
func jsTrim(s string) string { return strings.TrimFunc(s, jsIsSpace) }
|
||||
|
||||
// jsTrimStart is String.prototype.trimStart.
|
||||
func jsTrimStart(s string) string { return strings.TrimLeftFunc(s, jsIsSpace) }
|
||||
|
||||
// jsTruthy is the `!!x` of a parsed YAML value.
|
||||
//
|
||||
// The parser produces only nil, bool, float64, string, []any and
|
||||
// map[string]any, so those are the only cases that can arise. An empty array
|
||||
// and an empty object are both truthy in JavaScript, which is why they are not
|
||||
// listed alongside the empty string.
|
||||
func jsTruthy(v any) bool {
|
||||
switch x := v.(type) {
|
||||
case nil:
|
||||
return false
|
||||
case bool:
|
||||
return x
|
||||
case float64:
|
||||
return x != 0 && !math.IsNaN(x)
|
||||
case string:
|
||||
return x != ""
|
||||
default:
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
// jsNumberToString is JavaScript's Number → String conversion for the values
|
||||
// this parser can produce.
|
||||
//
|
||||
// `-0` prints as `0`, integers print without a decimal point, and everything
|
||||
// else takes the shortest representation that round-trips — which is what
|
||||
// strconv's 'g' with precision -1 gives, in the range a definition can reach.
|
||||
func jsNumberToString(f float64) string {
|
||||
switch {
|
||||
case math.IsNaN(f):
|
||||
return "NaN"
|
||||
case math.IsInf(f, 1):
|
||||
return "Infinity"
|
||||
case math.IsInf(f, -1):
|
||||
return "-Infinity"
|
||||
case f == 0:
|
||||
return "0" // collapses -0
|
||||
}
|
||||
if f == math.Trunc(f) && math.Abs(f) < 1e21 {
|
||||
return strconv.FormatFloat(f, 'f', -1, 64)
|
||||
}
|
||||
return strconv.FormatFloat(f, 'g', -1, 64)
|
||||
}
|
||||
|
||||
// jsString is the `String(x)` of a parsed YAML value.
|
||||
//
|
||||
// Arrays join on `,` with nil rendering as the empty string, which is
|
||||
// Array.prototype.toString; a mapping renders as `[object Object]`. Both are
|
||||
// reachable: `trigger:` may be written as a list, and the message an author
|
||||
// reads interpolates the result.
|
||||
func jsString(v any) string {
|
||||
switch x := v.(type) {
|
||||
case nil:
|
||||
return "null"
|
||||
case bool:
|
||||
if x {
|
||||
return "true"
|
||||
}
|
||||
return "false"
|
||||
case float64:
|
||||
return jsNumberToString(x)
|
||||
case string:
|
||||
return x
|
||||
case []any:
|
||||
parts := make([]string, len(x))
|
||||
for i, item := range x {
|
||||
if item == nil {
|
||||
parts[i] = ""
|
||||
continue
|
||||
}
|
||||
parts[i] = jsString(item)
|
||||
}
|
||||
return strings.Join(parts, ",")
|
||||
case map[string]any:
|
||||
return "[object Object]"
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// jsTrimmed is the frontend's `trimmed()` helper: String(value ?? empty).trim().
|
||||
//
|
||||
// The nullish coalescing matters: a nil renders as the empty string here,
|
||||
// where a bare String(null) would render as the four characters `null`.
|
||||
func jsTrimmed(v any) string {
|
||||
if v == nil {
|
||||
return ""
|
||||
}
|
||||
return jsTrim(jsString(v))
|
||||
}
|
||||
|
||||
// jsNumber is the `Number(x)` of a parsed YAML value, NaN where JavaScript
|
||||
// gives NaN.
|
||||
//
|
||||
// Only reached from `version:`, where the result is checked with
|
||||
// Number.isInteger. The string cases below are the ones a YAML scalar can
|
||||
// still be carrying at that point: a quoted `"3"` stays a string, and so does
|
||||
// anything the numeric patterns in toScalar declined.
|
||||
func jsNumber(v any) float64 {
|
||||
switch x := v.(type) {
|
||||
case nil:
|
||||
return 0
|
||||
case bool:
|
||||
if x {
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
case float64:
|
||||
return x
|
||||
case string:
|
||||
return jsNumberFromString(x)
|
||||
case []any:
|
||||
// Number([]) is 0 and Number([3]) is 3, via the same String()
|
||||
// conversion; anything longer stringifies with a comma and fails.
|
||||
if len(x) == 0 {
|
||||
return 0
|
||||
}
|
||||
if len(x) == 1 {
|
||||
return jsNumberFromString(jsString(x[0]))
|
||||
}
|
||||
}
|
||||
return math.NaN()
|
||||
}
|
||||
|
||||
func jsNumberFromString(s string) float64 {
|
||||
s = jsTrim(s)
|
||||
if s == "" {
|
||||
return 0
|
||||
}
|
||||
switch s {
|
||||
case "Infinity", "+Infinity":
|
||||
return math.Inf(1)
|
||||
case "-Infinity":
|
||||
return math.Inf(-1)
|
||||
}
|
||||
// The radix prefixes JavaScript accepts in a numeric string literal. Signs
|
||||
// are not permitted with them, which ParseUint enforces by rejecting the
|
||||
// leading character.
|
||||
if len(s) > 2 && s[0] == '0' {
|
||||
var base int
|
||||
switch s[1] {
|
||||
case 'x', 'X':
|
||||
base = 16
|
||||
case 'o', 'O':
|
||||
base = 8
|
||||
case 'b', 'B':
|
||||
base = 2
|
||||
}
|
||||
if base != 0 {
|
||||
n, err := strconv.ParseUint(s[2:], base, 64)
|
||||
if err != nil {
|
||||
return math.NaN()
|
||||
}
|
||||
return float64(n)
|
||||
}
|
||||
}
|
||||
f, err := strconv.ParseFloat(s, 64)
|
||||
if err != nil {
|
||||
return math.NaN()
|
||||
}
|
||||
return f
|
||||
}
|
||||
349
go-api/internal/definition/skill.go
Normal file
349
go-api/internal/definition/skill.go
Normal file
@@ -0,0 +1,349 @@
|
||||
package definition
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// One Markdown definition → one skill.
|
||||
//
|
||||
// A port of parseSkill and validateSkillSource in src/lib/skills/registry.js,
|
||||
// restricted to the definition contract — see the package documentation in
|
||||
// definition.go for exactly where that boundary is and why the `ui:` and
|
||||
// `owliver:` blocks are on the other side of it.
|
||||
|
||||
// Level is one rung of a workforce ladder, read from the body's own headings.
|
||||
type Level struct {
|
||||
Level string `json:"level"`
|
||||
Label string `json:"label"`
|
||||
Summary string `json:"summary"`
|
||||
}
|
||||
|
||||
// Skill is a definition as the backend reads it.
|
||||
//
|
||||
// The five fields migration 000005 projects into columns — ID, Name,
|
||||
// Description, Status, Pages — are the compatibility contract; the rest is
|
||||
// carried because it is free once the frontmatter is parsed and because the
|
||||
// conformance suite compares it.
|
||||
type Skill struct {
|
||||
ID string `json:"id"`
|
||||
Name string `json:"name"`
|
||||
Description string `json:"description"`
|
||||
Status string `json:"status"`
|
||||
|
||||
// Pages as the AUTHOR WROTE THEM, not canonicalised.
|
||||
//
|
||||
// This is not an oversight and must not be "fixed": parseSkill keeps the
|
||||
// declared strings, so a skill written against `Talent Pool` is registered
|
||||
// under `Talent Pool` and resolved through normalizeKey at every use.
|
||||
// Agents are the other way round — see Agent.Pages. Canonicalising here
|
||||
// would make the backend's projection disagree with the editor's.
|
||||
Pages []string `json:"pages"`
|
||||
|
||||
Kind string `json:"kind"`
|
||||
Category string `json:"category"`
|
||||
Actions []string `json:"actions"`
|
||||
Triggers []string `json:"triggers"`
|
||||
DeclaredTriggers bool `json:"declaredTriggers"`
|
||||
Prompt *string `json:"prompt"`
|
||||
SkillID *string `json:"skillId"`
|
||||
Levels []Level `json:"levels"`
|
||||
|
||||
// Body is the Markdown after the frontmatter, trimmed. The definition
|
||||
// itself is NEVER rewritten — see Definition.Markdown.
|
||||
Body string `json:"-"`
|
||||
|
||||
// Deferred names the frontmatter blocks whose semantics this package does
|
||||
// not check and the frontend does. Empty for every definition the backend
|
||||
// can fully validate on its own. See package documentation.
|
||||
Deferred []string `json:"deferred,omitempty"`
|
||||
}
|
||||
|
||||
// AuthoredPath is the origin an authored definition has when the caller names
|
||||
// none. It is a value rather than an absence for one reason: it is the
|
||||
// frontend's own default parameter.
|
||||
//
|
||||
// parseSkill(raw, { path = 'custom', custom = false } = {})
|
||||
// parseAgent(raw, { path = 'custom', custom = false } = {})
|
||||
//
|
||||
// validateSkillSource and validateAgentSource both call their parser with no
|
||||
// path, so every definition the EDITOR checks derives its last-resort id from
|
||||
// the literal string `custom`. That is the same call the backend is making — a
|
||||
// definition submitted to the API is authored, not shipped — so the backend
|
||||
// must derive the same id.
|
||||
//
|
||||
// The difference is reachable and it is not cosmetic. A definition with no
|
||||
// `id:`, no `name:` and a valid `pages:` list gets the id `custom` on the
|
||||
// frontend, passes the id-format check and is ACCEPTED. Deriving no id here
|
||||
// would refuse it with "The frontmatter needs an `id`." — a definition that
|
||||
// validates in the editor and fails on save, which is the exact failure mode
|
||||
// this package exists to prevent. Fixture case: id-omitted-unnamed.
|
||||
const AuthoredPath = "custom"
|
||||
|
||||
// Options carries what the caller knows that the definition does not.
|
||||
type Options struct {
|
||||
// Path is the definition's origin, used only as the last fallback for an
|
||||
// id. Leave it empty for anything authored rather than shipped — which is
|
||||
// what the backend always has — and it becomes AuthoredPath, exactly as the
|
||||
// frontend's default parameter does.
|
||||
Path string
|
||||
}
|
||||
|
||||
// path is the origin an id is derived from, with the frontend's default
|
||||
// applied.
|
||||
func (o Options) path() string {
|
||||
if o.Path == "" {
|
||||
return AuthoredPath
|
||||
}
|
||||
return o.Path
|
||||
}
|
||||
|
||||
// ParseSkill reads a skill definition.
|
||||
//
|
||||
// Returns a *Error when the frontmatter cannot be read. A definition with no
|
||||
// frontmatter at all is not an error here: it parses to a skill carrying the
|
||||
// derived id and no pages, and ValidateSkill is what refuses it — the same
|
||||
// division of labour the frontend has.
|
||||
func ParseSkill(raw string, opts Options) (*Skill, error) {
|
||||
doc, err := ParseFrontmatter(raw)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
data := doc.Data
|
||||
|
||||
declaredPages, _ := data["pages"].([]any)
|
||||
|
||||
// `id:` always wins. An explicit id is the address other definitions and
|
||||
// stored preferences refer to, and deriving over the top of one would
|
||||
// silently rename a skill. Slugging the name is what an author means by
|
||||
// leaving it out; the filename is right only for a file, which is why it
|
||||
// is last.
|
||||
id := jsTrim(jsString(data["id"]))
|
||||
if !jsTruthy(data["id"]) {
|
||||
id = slugify(data["name"])
|
||||
if id == "" {
|
||||
id = fileStem(opts.path())
|
||||
}
|
||||
}
|
||||
|
||||
levels := sectionLevels(doc.Body)
|
||||
|
||||
// Two things wear the same format. A definition that names a ladder is a
|
||||
// workforce skill; nothing else distinguishes them, so an author declares
|
||||
// one by writing one rather than by setting a flag.
|
||||
kind := jsTrim(jsString(data["kind"]))
|
||||
if !jsTruthy(data["kind"]) {
|
||||
kind = "assistant"
|
||||
if len(levels) > 0 {
|
||||
kind = "workforce"
|
||||
}
|
||||
}
|
||||
|
||||
skill := &Skill{
|
||||
ID: id,
|
||||
Kind: kind,
|
||||
Levels: levels,
|
||||
Body: doc.Body,
|
||||
Name: "Untitled skill",
|
||||
Pages: stringsOf(declaredPages),
|
||||
}
|
||||
|
||||
if jsTruthy(data["name"]) {
|
||||
skill.Name = jsString(data["name"])
|
||||
}
|
||||
if jsTruthy(data["description"]) {
|
||||
skill.Description = jsString(data["description"])
|
||||
}
|
||||
if s, ok := data["category"].(string); ok {
|
||||
skill.Category = jsTrim(s)
|
||||
}
|
||||
|
||||
// The whole of a skill's lifecycle, and deliberately a coercion rather
|
||||
// than a check: the frontend reads anything that is not `inactive` as
|
||||
// `active`, so `status: bogus` registers as active rather than being
|
||||
// refused. Reproduced, not corrected — see the divergence note in
|
||||
// docs/phase-4d-parser-contract.md.
|
||||
skill.Status = "active"
|
||||
if s, ok := data["status"].(string); ok && s == "inactive" {
|
||||
skill.Status = "inactive"
|
||||
}
|
||||
|
||||
if actions, ok := data["actions"].([]any); ok {
|
||||
skill.Actions = stringsOf(actions)
|
||||
} else {
|
||||
skill.Actions = []string{}
|
||||
}
|
||||
|
||||
// A skill with no declared triggers answers to its own name, so a
|
||||
// definition that omits the field is still reachable by asking for it.
|
||||
// Explicit triggers replace the fallback rather than adding to it.
|
||||
triggers, hasTriggers := data["triggers"].([]any)
|
||||
skill.DeclaredTriggers = hasTriggers && len(triggers) > 0
|
||||
skill.Triggers = []string{}
|
||||
if skill.DeclaredTriggers {
|
||||
for _, t := range triggers {
|
||||
skill.Triggers = append(skill.Triggers, strings.ToLower(jsString(t)))
|
||||
}
|
||||
} else if jsTruthy(data["name"]) {
|
||||
skill.Triggers = append(skill.Triggers, strings.ToLower(jsString(data["name"])))
|
||||
}
|
||||
|
||||
if jsTruthy(data["prompt"]) {
|
||||
p := jsString(data["prompt"])
|
||||
skill.Prompt = &p
|
||||
}
|
||||
|
||||
// The capability in the skill graph a workforce definition governs:
|
||||
// `skill:`, or the id with a `-training` suffix dropped and dashes swapped
|
||||
// for underscores.
|
||||
if kind == "workforce" {
|
||||
base := id
|
||||
if jsTruthy(data["skill"]) {
|
||||
base = jsString(data["skill"])
|
||||
} else {
|
||||
base = strings.TrimSuffix(base, "-training")
|
||||
}
|
||||
s := strings.ReplaceAll(base, "-", "_")
|
||||
skill.SkillID = &s
|
||||
}
|
||||
|
||||
skill.Deferred = deferredBlocks(data)
|
||||
return skill, nil
|
||||
}
|
||||
|
||||
// sectionLevels reads the ladder a workforce definition defines, in order, from
|
||||
// the body's own headings. A rung with no prose is not a rung.
|
||||
func sectionLevels(body string) []Level {
|
||||
out := []Level{}
|
||||
for _, heading := range levelHeadings {
|
||||
summary := SectionText(body, heading)
|
||||
if summary == "" {
|
||||
continue
|
||||
}
|
||||
out = append(out, Level{
|
||||
Level: strings.ToLower(heading),
|
||||
Label: heading,
|
||||
Summary: summary,
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// deferredBlocks names the frontmatter this package does not semantically
|
||||
// check. See the package documentation for why they are deferred rather than
|
||||
// validated or rejected.
|
||||
func deferredBlocks(data map[string]any) []string {
|
||||
out := []string{}
|
||||
for _, key := range []string{"ui", "owliver"} {
|
||||
if _, present := data[key]; present {
|
||||
out = append(out, key)
|
||||
}
|
||||
}
|
||||
if len(out) == 0 {
|
||||
return nil
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// stringsOf renders a parsed sequence as the strings the frontend would read
|
||||
// out of it. Non-string entries are stringified rather than dropped, because
|
||||
// that is what every consumer of `pages` and `actions` does with them.
|
||||
func stringsOf(list []any) []string {
|
||||
out := make([]string, 0, len(list))
|
||||
for _, v := range list {
|
||||
out = append(out, jsString(v))
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func fileStem(path string) string {
|
||||
if path == "" {
|
||||
return ""
|
||||
}
|
||||
if i := strings.LastIndexByte(path, '/'); i >= 0 {
|
||||
path = path[i+1:]
|
||||
}
|
||||
return strings.TrimSuffix(path, ".md")
|
||||
}
|
||||
|
||||
// ValidateSkill decides whether a skill definition may be stored.
|
||||
//
|
||||
// Returns nil when it may. The order is the order an author would fix things
|
||||
// in, and every message below is the frontend's message character for
|
||||
// character — an author who sees one in the editor and a different one from
|
||||
// the API is being told about two different problems.
|
||||
//
|
||||
// Two rules are the backend's own and are marked as such: the size bound and
|
||||
// the deferred-block rule. Both are explained in
|
||||
// docs/phase-4d-parser-contract.md.
|
||||
func ValidateSkill(raw string) error {
|
||||
if jsTrim(raw) == "" {
|
||||
return &Rejection{Message: "Paste or upload a Markdown definition."}
|
||||
}
|
||||
|
||||
// The backend's own rule, from migration 000005's markdown_size CHECK. The
|
||||
// editor does not enforce it, so a definition over the bound is one the
|
||||
// frontend accepts and the DATABASE refuses; refusing it here turns a
|
||||
// constraint violation into a message. See the contract document.
|
||||
if n := len([]rune(raw)); n > MaxMarkdownLength {
|
||||
return &Rejection{
|
||||
BackendOnly: true,
|
||||
Message: fmt.Sprintf(
|
||||
"That definition is %d characters. The limit is %d.", n, MaxMarkdownLength),
|
||||
}
|
||||
}
|
||||
|
||||
skill, err := ParseSkill(raw, Options{})
|
||||
if err != nil {
|
||||
// The subset reports the line it failed on, which is far more useful
|
||||
// than "could not be parsed".
|
||||
return &Rejection{Message: jsTrim("That definition could not be parsed. " + err.Error())}
|
||||
}
|
||||
|
||||
if skill.ID == "" {
|
||||
return &Rejection{Message: "The frontmatter needs an `id`."}
|
||||
}
|
||||
if !isDefinitionID(skill.ID) {
|
||||
return &Rejection{Message: "`id` must be lower-case letters, numbers and dashes."}
|
||||
}
|
||||
// Faithful to the frontend, where `name` has already fallen back to
|
||||
// `Untitled skill` and this check can therefore never fire. Kept so the
|
||||
// two validators have the same shape and the same order.
|
||||
if skill.Name == "" {
|
||||
return &Rejection{Message: "The frontmatter needs a `name`."}
|
||||
}
|
||||
// The backend's own rule, and it must be asked BEFORE the generic one
|
||||
// below. A `ui:` block declares the pages it draws on, and parseSkill falls
|
||||
// back to those pages when `pages:` is absent — a fallback this package
|
||||
// cannot compute, because it does not read the `ui:` vocabulary. Rather
|
||||
// than report an empty page list it never really established, say what is
|
||||
// actually missing. No shipped definition relies on the fallback: all
|
||||
// nineteen that carry a `ui:` or `owliver:` block also declare `pages:`.
|
||||
if len(skill.Pages) == 0 && len(skill.Deferred) > 0 {
|
||||
return &Rejection{
|
||||
BackendOnly: true,
|
||||
Message: "A definition with a `ui:` block needs an explicit `pages:` list.",
|
||||
}
|
||||
}
|
||||
if len(skill.Pages) == 0 {
|
||||
return &Rejection{Message: "The frontmatter needs at least one `pages` entry."}
|
||||
}
|
||||
|
||||
unknown := []string{}
|
||||
for _, p := range skill.Pages {
|
||||
if !SurfaceExists(p) {
|
||||
unknown = append(unknown, p)
|
||||
}
|
||||
}
|
||||
if len(unknown) > 0 {
|
||||
plural := ""
|
||||
if len(unknown) > 1 {
|
||||
plural = "s"
|
||||
}
|
||||
return &Rejection{Message: fmt.Sprintf(
|
||||
"Unsupported page%s: %s. Supported pages: %s.",
|
||||
plural, strings.Join(unknown, ", "), strings.Join(SupportedPages, ", "))}
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
9633
go-api/internal/definition/testdata/oracle.json
vendored
Normal file
9633
go-api/internal/definition/testdata/oracle.json
vendored
Normal file
File diff suppressed because one or more lines are too long
184
go-api/internal/definition/vocabulary.go
Normal file
184
go-api/internal/definition/vocabulary.go
Normal file
@@ -0,0 +1,184 @@
|
||||
package definition
|
||||
|
||||
import "strings"
|
||||
|
||||
// The closed vocabulary a definition is allowed to name.
|
||||
//
|
||||
// Every table here is a transcription of a frontend table, and the conformance
|
||||
// suite asserts each one against the vocabulary the JavaScript actually
|
||||
// exports (testdata/oracle.json, `vocabulary`) — so a page added to
|
||||
// surfaces.js or an icon added to vocabulary.js fails a test here rather than
|
||||
// silently making the two ends disagree about what is valid.
|
||||
//
|
||||
// Nothing is looked up dynamically and nothing is constructed: a definition
|
||||
// names a key and this file answers whether that key exists.
|
||||
|
||||
// pageSurfaces mirrors SKILL_SURFACES in src/lib/skills/surfaces.js — id first,
|
||||
// then the alternative spellings an author may use for it.
|
||||
var pageSurfaces = []struct {
|
||||
ID string
|
||||
Aliases []string
|
||||
}{
|
||||
{ID: "control-center"},
|
||||
{ID: "positions"},
|
||||
{ID: "create-position", Aliases: []string{"new-position"}},
|
||||
{ID: "candidates"},
|
||||
{ID: "hired-history", Aliases: []string{"hired"}},
|
||||
{ID: "talent-pool"},
|
||||
{ID: "krow-forge", Aliases: []string{"university", "forge"}},
|
||||
{ID: "analytics"},
|
||||
{ID: "activity"},
|
||||
{ID: "workspace-agent-configure"},
|
||||
{ID: "settings"},
|
||||
{ID: "workspace"},
|
||||
{ID: "workspace-agents"},
|
||||
{ID: "workspace-skills"},
|
||||
{ID: "workspace-skill-configure"},
|
||||
{ID: "skill-development"},
|
||||
{ID: "profile"},
|
||||
{ID: "candidates-analysis"},
|
||||
}
|
||||
|
||||
// surfaceByKey resolves every spelling — canonical or alias — to its canonical
|
||||
// id.
|
||||
var surfaceByKey = func() map[string]string {
|
||||
m := map[string]string{}
|
||||
for _, s := range pageSurfaces {
|
||||
m[s.ID] = s.ID
|
||||
for _, a := range s.Aliases {
|
||||
m[a] = s.ID
|
||||
}
|
||||
}
|
||||
return m
|
||||
}()
|
||||
|
||||
// SupportedPages is every canonical page id, in declaration order. The order is
|
||||
// the order the frontend lists them in when it refuses an unsupported page, and
|
||||
// that message is compared byte for byte.
|
||||
var SupportedPages = func() []string {
|
||||
out := make([]string, len(pageSurfaces))
|
||||
for i, s := range pageSurfaces {
|
||||
out[i] = s.ID
|
||||
}
|
||||
return out
|
||||
}()
|
||||
|
||||
// normalizeKey is surfaces.js's own: trimmed, lower-cased, with spaces and
|
||||
// underscores read as dashes. `Talent Pool` and `talent_pool` both reach
|
||||
// `talent-pool`; the canonical keys never widen, only what an author may type
|
||||
// to reach them.
|
||||
func normalizeKey(page any) string {
|
||||
s := strings.ToLower(jsTrim(jsString(page)))
|
||||
if page == nil {
|
||||
s = ""
|
||||
}
|
||||
return strings.Map(func(r rune) rune {
|
||||
if r == ' ' || r == '_' || jsIsSpace(r) {
|
||||
return '-'
|
||||
}
|
||||
return r
|
||||
}, s)
|
||||
}
|
||||
|
||||
// SurfaceExists reports whether a declared page name refers to a real surface.
|
||||
func SurfaceExists(page any) bool {
|
||||
_, ok := surfaceByKey[normalizeKey(page)]
|
||||
return ok
|
||||
}
|
||||
|
||||
// CanonicalPage is the canonical id a declared page name refers to, or "".
|
||||
func CanonicalPage(page any) string { return surfaceByKey[normalizeKey(page)] }
|
||||
|
||||
/* ── Agent vocabulary — src/lib/agents/vocabulary.js ─────────────────────── */
|
||||
|
||||
var (
|
||||
AgentStatuses = []string{"draft", "published", "archived"}
|
||||
ReasoningModes = []string{"fast", "balanced", "deep"}
|
||||
KnowledgeKinds = []string{"note", "link", "skill-reference"}
|
||||
AgentAccess = []string{"all", "specific"}
|
||||
PermissionRole = []string{"manager", "editor", "viewer"}
|
||||
AgentIcons = []string{
|
||||
"owliver", "sparkles", "briefcase", "users", "user-check",
|
||||
"layers", "graduation-cap", "bar-chart", "activity", "shield",
|
||||
}
|
||||
)
|
||||
|
||||
const (
|
||||
DefaultAgentStatus = "draft"
|
||||
DefaultReasoning = "balanced"
|
||||
DefaultAgentIcon = "owliver"
|
||||
DefaultKnowledgeKind = "note"
|
||||
DefaultAgentAccess = "all"
|
||||
DefaultPermission = "viewer"
|
||||
)
|
||||
|
||||
func contains(list []string, want string) bool {
|
||||
for _, v := range list {
|
||||
if v == want {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
/* ── Skill vocabulary ────────────────────────────────────────────────────── */
|
||||
|
||||
// SkillStatuses is the whole of a skill's lifecycle. There is no version and no
|
||||
// publish step: migration 000005 states the same two values.
|
||||
var SkillStatuses = []string{"active", "inactive"}
|
||||
|
||||
// levelHeadings are the rungs a workforce skill may define, and the reason a
|
||||
// definition is read as workforce rather than assistant. Order is the ladder's.
|
||||
var levelHeadings = []string{"Beginner", "Intermediate", "Advanced", "Expert"}
|
||||
|
||||
// slugify is uiConfig.js's, used to derive an id from a name.
|
||||
//
|
||||
// String(value || '').toLowerCase().trim()
|
||||
// .replace(/[^a-z0-9]+/g, '-').replace(/^-|-$/g, '')
|
||||
//
|
||||
// The lower-casing happens BEFORE the class replacement, so an upper-case
|
||||
// letter becomes itself rather than a dash.
|
||||
func slugify(value any) string {
|
||||
if !jsTruthy(value) {
|
||||
return ""
|
||||
}
|
||||
s := jsTrim(strings.ToLower(jsString(value)))
|
||||
|
||||
var b strings.Builder
|
||||
dash := false
|
||||
for _, r := range s {
|
||||
if (r >= 'a' && r <= 'z') || (r >= '0' && r <= '9') {
|
||||
b.WriteRune(r)
|
||||
dash = false
|
||||
continue
|
||||
}
|
||||
if !dash {
|
||||
b.WriteByte('-')
|
||||
dash = true
|
||||
}
|
||||
}
|
||||
out := b.String()
|
||||
out = strings.TrimPrefix(out, "-")
|
||||
out = strings.TrimSuffix(out, "-")
|
||||
return out
|
||||
}
|
||||
|
||||
// isDefinitionID is the id format the frontend validator enforces and the
|
||||
// definition_id CHECK in migration 000005 restates.
|
||||
func isDefinitionID(id string) bool {
|
||||
if id == "" {
|
||||
return false
|
||||
}
|
||||
for i, r := range id {
|
||||
lower := r >= 'a' && r <= 'z'
|
||||
digit := r >= '0' && r <= '9'
|
||||
if lower || digit {
|
||||
continue
|
||||
}
|
||||
if r == '-' && i > 0 {
|
||||
continue
|
||||
}
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
362
go-api/internal/definition/yaml.go
Normal file
362
go-api/internal/definition/yaml.go
Normal file
@@ -0,0 +1,362 @@
|
||||
package definition
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// The YAML subset a Krow definition is allowed to use.
|
||||
//
|
||||
// A line-for-line port of the frontend's src/lib/skills/yaml.js. That file is
|
||||
// the specification and this is the second implementation of it, so every
|
||||
// decision below is made because the JavaScript makes it — including the ones
|
||||
// a YAML library would make differently.
|
||||
//
|
||||
// Supported, and nothing else:
|
||||
//
|
||||
// - block maps and block sequences, nested to any depth
|
||||
// - scalars: strings, integers, floats, booleans, null
|
||||
// - quoted strings, for values containing `:` or `#`
|
||||
// - `- key: value`, a mapping whose first key sits on the dash
|
||||
// - `#` comments, and blank lines
|
||||
//
|
||||
// Anchors, aliases, merge keys, multi-document files, flow mappings, flow
|
||||
// sequences, block scalars and tags are NOT supported, and are not silently
|
||||
// half-read: an unparseable line is an error carrying the line number, so a
|
||||
// definition either means what it says or is refused with somewhere to look.
|
||||
//
|
||||
// No dependency. A general YAML library would accept a much larger language
|
||||
// than the frontend does, and every construct it accepted and the frontend did
|
||||
// not would be a definition the backend stores and the editor cannot read —
|
||||
// exactly the failure this package exists to prevent. The subset is small
|
||||
// enough to port exactly, so it is ported exactly.
|
||||
//
|
||||
// Nothing here evaluates anything. There is no reflection, no template, no
|
||||
// code path from a definition to execution of any kind: a definition is
|
||||
// configuration, and this is the boundary that keeps it configuration.
|
||||
|
||||
// Error is a definition that could not be read, carrying the line it failed on.
|
||||
//
|
||||
// Line is 1-based and counts lines of FRONTMATTER, not of the file — which is
|
||||
// what the JavaScript reports, because parseYaml is handed the fenced block
|
||||
// rather than the document. Reproduced rather than improved: an author who
|
||||
// sees one message in the editor and another from the API is being told about
|
||||
// two different problems.
|
||||
type Error struct {
|
||||
Line int
|
||||
Message string
|
||||
}
|
||||
|
||||
func (e *Error) Error() string { return e.Message }
|
||||
|
||||
// errIndent and errPair are the two failures the subset has, worded exactly as
|
||||
// the frontend words them.
|
||||
func errIndent(line int) *Error {
|
||||
return &Error{Line: line, Message: fmt.Sprintf("Unexpected indentation on line %d", line)}
|
||||
}
|
||||
|
||||
func errPair(line int, content string) *Error {
|
||||
return &Error{Line: line, Message: fmt.Sprintf("Line %d is not `key: value`: %s", line, content)}
|
||||
}
|
||||
|
||||
// keyPair matches `key: value` and is the only shape a mapping entry may take.
|
||||
// The key alphabet is the frontend's: letters, digits, underscore, dot, dash —
|
||||
// which is why `my key: value` is a refusal rather than a key with a space.
|
||||
// jsSpaceClass is the JavaScript `\s` character class. Go's own `\s` is
|
||||
// ASCII-only, and the difference is reachable: a non-breaking space after the
|
||||
// colon is whitespace to the editor's parser and would be part of the value
|
||||
// here.
|
||||
const jsSpaceClass = `[\t\n\v\f\r \x{00A0}\x{1680}\x{2000}-\x{200A}\x{2028}\x{2029}\x{202F}\x{205F}\x{3000}\x{FEFF}]`
|
||||
|
||||
var keyPair = regexp.MustCompile(`^([A-Za-z0-9_.-]+):` + jsSpaceClass + `*([\s\S]*)$`)
|
||||
|
||||
var (
|
||||
reInt = regexp.MustCompile(`^-?[0-9]+$`)
|
||||
reFloat = regexp.MustCompile(`^-?[0-9]*\.[0-9]+$`)
|
||||
)
|
||||
|
||||
// line is one significant line, reduced to what the parser needs to decide.
|
||||
type line struct {
|
||||
number int // 1-based, within the frontmatter block
|
||||
indent int // leading whitespace, tabs counted as two
|
||||
content string
|
||||
}
|
||||
|
||||
// readLines drops blank lines and whole-line comments, and measures what is
|
||||
// left.
|
||||
//
|
||||
// Indentation is counted in code points with a tab worth two spaces, which is
|
||||
// what the JavaScript does and is why a tab-indented sequence sits at the same
|
||||
// depth as a two-space one.
|
||||
func readLines(source string) []line {
|
||||
out := []line{}
|
||||
for i, text := range strings.Split(source, "\n") {
|
||||
trimmed := jsTrim(text)
|
||||
if trimmed == "" {
|
||||
continue
|
||||
}
|
||||
// A whole-line comment: `^\s*#`.
|
||||
if strings.HasPrefix(jsTrimStart(text), "#") {
|
||||
continue
|
||||
}
|
||||
indent := 0
|
||||
for _, r := range text {
|
||||
if !jsIsSpace(r) {
|
||||
break
|
||||
}
|
||||
if r == '\t' {
|
||||
indent += 2
|
||||
continue
|
||||
}
|
||||
indent++
|
||||
}
|
||||
out = append(out, line{number: i + 1, indent: indent, content: trimmed})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// quoted matches a scalar wrapped in one kind of quote, end to end.
|
||||
//
|
||||
// Greedy and anchored at both ends, as in the frontend: `"a" "b"` is therefore
|
||||
// ONE quoted string whose content is `a" "b`, not two. That is a strange
|
||||
// reading, and it is the reading the editor gives, so it is the reading here.
|
||||
func quotedScalar(value string) (string, bool) {
|
||||
if len(value) < 2 {
|
||||
return "", false
|
||||
}
|
||||
q := value[0]
|
||||
if q != '\'' && q != '"' {
|
||||
return "", false
|
||||
}
|
||||
if value[len(value)-1] != q {
|
||||
return "", false
|
||||
}
|
||||
inner := value[1 : len(value)-1]
|
||||
// The only escape the subset has: a doubled quote is one quote.
|
||||
return strings.ReplaceAll(inner, string([]byte{q, q}), string(q)), true
|
||||
}
|
||||
|
||||
// stripTrailingComment removes an unquoted trailing `#` comment.
|
||||
//
|
||||
// `\s+#.*$` applied once, leftmost — so `ops # a # b` loses everything from
|
||||
// the first spaced hash, and `ops#1` loses nothing, because a hash inside a
|
||||
// word is part of the word.
|
||||
func stripTrailingComment(value string) string {
|
||||
runes := []rune(value)
|
||||
for i := 0; i < len(runes); i++ {
|
||||
if !jsIsSpace(runes[i]) {
|
||||
continue
|
||||
}
|
||||
j := i
|
||||
for j < len(runes) && jsIsSpace(runes[j]) {
|
||||
j++
|
||||
}
|
||||
if j < len(runes) && runes[j] == '#' {
|
||||
return string(runes[:i])
|
||||
}
|
||||
i = j - 1
|
||||
}
|
||||
return value
|
||||
}
|
||||
|
||||
// toScalar reads one written value: `true`, `false`, `null`, a number, a
|
||||
// quoted string, or the string as written.
|
||||
func toScalar(raw string) any {
|
||||
value := jsTrim(raw)
|
||||
|
||||
switch value {
|
||||
case "", "~", "null":
|
||||
return nil
|
||||
case "true":
|
||||
return true
|
||||
case "false":
|
||||
return false
|
||||
}
|
||||
|
||||
// Quoted: taken literally, which is how a value containing `:` or `#` is
|
||||
// written. No escape processing beyond the doubled quote.
|
||||
if inner, ok := quotedScalar(value); ok {
|
||||
return inner
|
||||
}
|
||||
|
||||
if reInt.MatchString(value) || reFloat.MatchString(value) {
|
||||
if f, err := strconv.ParseFloat(value, 64); err == nil {
|
||||
return f
|
||||
}
|
||||
}
|
||||
|
||||
return jsTrim(stripTrailingComment(value))
|
||||
}
|
||||
|
||||
// cursor is shared down the recursion so a child consumes the lines it owns.
|
||||
type cursor struct{ i int }
|
||||
|
||||
// parseBlock reads one block at indent or deeper.
|
||||
//
|
||||
// Map or sequence depending on what the first line at this level is, which is
|
||||
// how YAML itself decides.
|
||||
func parseBlock(lines []line, c *cursor, indent int) (any, *Error) {
|
||||
if c.i >= len(lines) {
|
||||
return nil, nil
|
||||
}
|
||||
first := lines[c.i]
|
||||
if strings.HasPrefix(first.content, "- ") || first.content == "-" {
|
||||
return parseSequence(lines, c, indent)
|
||||
}
|
||||
return parseMapping(lines, c, indent)
|
||||
}
|
||||
|
||||
// dashPrefix is the `-` and the whitespace after it, as `^-\s*` consumes them.
|
||||
func dashPrefix(content string) int {
|
||||
if !strings.HasPrefix(content, "-") {
|
||||
return 0
|
||||
}
|
||||
n := 1
|
||||
for _, r := range content[1:] {
|
||||
if !jsIsSpace(r) {
|
||||
break
|
||||
}
|
||||
n += len(string(r))
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
func parseSequence(lines []line, c *cursor, indent int) (any, *Error) {
|
||||
out := []any{}
|
||||
|
||||
for c.i < len(lines) {
|
||||
cur := lines[c.i]
|
||||
if cur.indent < indent {
|
||||
break
|
||||
}
|
||||
if cur.indent > indent {
|
||||
return nil, errIndent(cur.number)
|
||||
}
|
||||
if !strings.HasPrefix(cur.content, "-") {
|
||||
break
|
||||
}
|
||||
|
||||
cut := dashPrefix(cur.content)
|
||||
rest := cur.content[cut:]
|
||||
c.i++
|
||||
|
||||
if rest == "" {
|
||||
// `-` alone: the item is the indented block beneath it.
|
||||
if c.i < len(lines) && lines[c.i].indent > indent {
|
||||
item, err := parseBlock(lines, c, lines[c.i].indent)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, item)
|
||||
continue
|
||||
}
|
||||
out = append(out, nil)
|
||||
continue
|
||||
}
|
||||
|
||||
// `- key: value` opens a mapping whose first key sits on the dash. The
|
||||
// remaining keys are indented to where that key started.
|
||||
if m := keyPair.FindStringSubmatch(rest); m != nil {
|
||||
keyIndent := indent + cut
|
||||
item := map[string]any{}
|
||||
key, value := m[1], m[2]
|
||||
|
||||
if value == "" && c.i < len(lines) && lines[c.i].indent > indent {
|
||||
block, err := parseBlock(lines, c, lines[c.i].indent)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
item[key] = block
|
||||
} else {
|
||||
item[key] = toScalar(value)
|
||||
}
|
||||
|
||||
for c.i < len(lines) && lines[c.i].indent == keyIndent &&
|
||||
!strings.HasPrefix(lines[c.i].content, "- ") {
|
||||
more, err := parseMapping(lines, c, keyIndent)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if m, ok := more.(map[string]any); ok {
|
||||
for k, v := range m {
|
||||
item[k] = v
|
||||
}
|
||||
}
|
||||
}
|
||||
out = append(out, item)
|
||||
continue
|
||||
}
|
||||
|
||||
out = append(out, toScalar(rest))
|
||||
}
|
||||
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func parseMapping(lines []line, c *cursor, indent int) (any, *Error) {
|
||||
out := map[string]any{}
|
||||
|
||||
for c.i < len(lines) {
|
||||
cur := lines[c.i]
|
||||
if cur.indent < indent {
|
||||
break
|
||||
}
|
||||
if cur.indent > indent {
|
||||
return nil, errIndent(cur.number)
|
||||
}
|
||||
if strings.HasPrefix(cur.content, "- ") {
|
||||
break
|
||||
}
|
||||
|
||||
m := keyPair.FindStringSubmatch(cur.content)
|
||||
if m == nil {
|
||||
return nil, errPair(cur.number, cur.content)
|
||||
}
|
||||
|
||||
key, value := m[1], m[2]
|
||||
c.i++
|
||||
|
||||
if value != "" {
|
||||
out[key] = toScalar(value)
|
||||
continue
|
||||
}
|
||||
|
||||
// An empty value means the value is the block below — or nothing.
|
||||
if c.i < len(lines) && lines[c.i].indent > indent {
|
||||
block, err := parseBlock(lines, c, lines[c.i].indent)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out[key] = block
|
||||
continue
|
||||
}
|
||||
out[key] = nil
|
||||
}
|
||||
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// ParseYAML reads one document of the subset as plain data.
|
||||
//
|
||||
// Returns map[string]any, []any, or the empty map for an empty document.
|
||||
// Anything it cannot read is an error rather than a guess, so a malformed
|
||||
// definition is reported to its author instead of being registered in a shape
|
||||
// nobody intended.
|
||||
func ParseYAML(source string) (any, error) {
|
||||
lines := readLines(source)
|
||||
if len(lines) == 0 {
|
||||
return map[string]any{}, nil
|
||||
}
|
||||
|
||||
c := &cursor{}
|
||||
value, err := parseBlock(lines, c, lines[0].indent)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if c.i < len(lines) {
|
||||
return nil, errIndent(lines[c.i].number)
|
||||
}
|
||||
return value, nil
|
||||
}
|
||||
Reference in New Issue
Block a user