first commit

This commit is contained in:
2026-08-24 13:06:29 +05:30
commit 7d12ebef3d
86 changed files with 39996 additions and 0 deletions

View File

@@ -0,0 +1,543 @@
package definition
import (
"fmt"
"math"
"strings"
)
// One Markdown definition → one agent.
//
// A port of parseAgent / validateAgentSource in src/lib/agents/registry.js and
// normalizeAgent in src/lib/agents/agentConfig.js.
//
// The contract normalizeAgent holds, and this holds with it:
//
// - Everything is optional. A definition declaring only an id and a name
// normalizes to a working agent with documented defaults.
// - Nothing unknown survives. Statuses, reasoning modes, pages, icons,
// knowledge kinds and permission roles are checked against the closed
// tables in vocabulary.go; an unrecognised value is a named error rather
// than a dropped key.
// - What validates is kept. One bad entry costs its author that entry and a
// message, never the rest of the file.
//
// One rule is deliberately absent, and it is absent on the frontend for the
// same reason: an agent with NO SKILLS is not refused. Five of this product's
// pages have no Owliver skills and answer from their own page responder, so
// refusing a skill-less agent would mean inventing placeholder skills to make
// those pages configurable.
// Starter is one conversation starter.
type Starter struct {
Label string `json:"label"`
Prompt string `json:"prompt"`
}
// Knowledge is one thing an agent has been told, as distinct from something it
// can do. Modelled as a document with an id and a body because that is the
// shape a retrieval layer reads.
type Knowledge struct {
ID string `json:"id"`
Label string `json:"label"`
Kind string `json:"kind"`
Body string `json:"body"`
URL string `json:"url"`
}
// Person is one named grant on an agent.
type Person struct {
User string `json:"user"`
Role string `json:"role"`
}
// Permissions is who owns an agent, who may reach it, and what they may do.
//
// Parsed and NOT enforced. Migration 000005 deliberately has no
// definition_permissions table: the block stays inside the Markdown until its
// semantics are defined.
type Permissions struct {
Owner string `json:"owner"`
Access string `json:"access"`
People []Person `json:"people"`
}
// Agent is a definition as the backend reads it.
type Agent struct {
ID string `json:"id"`
Name string `json:"name"`
Description string `json:"description"`
Status string `json:"status"`
Version int `json:"version"`
// Pages as CANONICAL surface ids.
//
// Unlike Skill.Pages, which keeps what the author wrote. The two are
// genuinely different on the frontend — normalizeAgent maps every page
// through canonicalPage and parseSkill does not — so agent_definitions.pages
// and skill_definitions.pages hold different vocabularies for the same
// concept. Reproduced rather than reconciled: making them agree here would
// make each one disagree with its own editor.
Pages []string `json:"pages"`
Icon string `json:"icon"`
Reasoning string `json:"reasoning"`
Trigger string `json:"trigger"`
WebSearch bool `json:"webSearch"`
Skills []string `json:"skills"`
Subagents []string `json:"subagents"`
Starters []Starter `json:"starters"`
Knowledge []Knowledge `json:"knowledge"`
Permissions Permissions `json:"permissions"`
// Instructions is the body's `## Instructions` section. Prose belongs under
// a heading where it can be written and read as prose, not in a
// frontmatter string.
Instructions string `json:"instructions"`
// Errors is what this definition lost on the way in, in the order
// normalizeAgent produces them. Carried on the record rather than thrown,
// so one bad entry costs its author that entry and a message.
Errors []string `json:"errors"`
Body string `json:"-"`
}
// asList is agentConfig.js's own coercion: an array stays an array, nothing
// becomes nothing, and anything else becomes a list of one.
//
// This is why `pages: candidates` is accepted for an AGENT and refused for a
// SKILL — parseSkill requires a real sequence and normalizeAgent coerces.
func asList(v any) []any {
switch x := v.(type) {
case []any:
return x
case nil:
return []any{}
case string:
if x == "" {
return []any{}
}
}
return []any{v}
}
// uniqueStrings keeps order and drops repeats; a blank entry is an error rather
// than a silent gap, because a blank id is an address that points nowhere.
func uniqueStrings(raw any, where, label string, errs *[]string) []string {
seen := map[string]bool{}
out := []string{}
for i, entry := range asList(raw) {
value := jsTrimmed(entry)
if value == "" {
*errs = append(*errs, fmt.Sprintf("%s[%d]: %s cannot be blank.", where, i, label))
continue
}
if seen[value] {
continue
}
seen[value] = true
out = append(out, value)
}
return out
}
// normalizePages resolves every declared page to a canonical surface key.
//
// Through CanonicalPage, so a definition may write an alias — `university` for
// `krow-forge` — exactly as a skill may. An unknown page is an error rather
// than a silently dropped entry, because a page nobody recognises is an agent
// that will never appear anywhere and give no reason why.
func normalizePages(raw any, errs *[]string) []string {
seen := map[string]bool{}
pages := []string{}
for i, entry := range asList(raw) {
written := jsTrimmed(entry)
if written == "" {
*errs = append(*errs, fmt.Sprintf("pages[%d]: a page cannot be blank.", i))
continue
}
canonical := CanonicalPage(written)
if canonical == "" {
*errs = append(*errs, fmt.Sprintf(
"pages[%d]: `%s` is not a page this product has.", i, written))
continue
}
if seen[canonical] {
continue
}
seen[canonical] = true
pages = append(pages, canonical)
}
return pages
}
// normalizeStarter reads one starter, in either the plain-string or the mapping
// form. A starter with no prompt of its own asks what it says.
func normalizeStarter(raw any, index int, errs *[]string) *Starter {
where := fmt.Sprintf("starters[%d]", index)
switch v := raw.(type) {
case string, float64:
label := jsTrimmed(v)
if label == "" {
*errs = append(*errs, where+": a starter needs text.")
return nil
}
return &Starter{Label: label, Prompt: label}
case map[string]any:
// `raw.label ?? raw.prompt` — nullish, so an absent or null label
// falls through to the prompt and a starter written as a bare prompt
// still has something to show.
source := v["label"]
if source == nil {
source = v["prompt"]
}
label := jsTrimmed(source)
if label == "" {
*errs = append(*errs, where+": a starter needs a `label`.")
return nil
}
prompt := jsTrimmed(v["prompt"])
if prompt == "" {
prompt = label
}
return &Starter{Label: label, Prompt: prompt}
}
*errs = append(*errs, where+": a starter must be a line of text, or a mapping of options.")
return nil
}
// normalizeKnowledge reads one knowledge entry.
func normalizeKnowledge(raw any, index int, errs *[]string) *Knowledge {
where := fmt.Sprintf("knowledge[%d]", index)
switch v := raw.(type) {
case string, float64:
body := jsTrimmed(v)
if body == "" {
*errs = append(*errs, where+": a knowledge entry needs text.")
return nil
}
id := slugify(runeSlice(body, 40))
if id == "" {
id = fmt.Sprintf("k%d", index+1)
}
return &Knowledge{
ID: id, Label: runeSlice(body, 60), Kind: DefaultKnowledgeKind, Body: body,
}
case map[string]any:
label := jsTrimmed(v["label"])
body := jsTrimmed(v["body"])
url := jsTrimmed(v["url"])
if label == "" && body == "" {
*errs = append(*errs, where+": a knowledge entry needs a `label` or a `body`.")
return nil
}
kind := jsTrimmed(v["kind"])
if kind == "" {
kind = DefaultKnowledgeKind
}
if !contains(KnowledgeKinds, kind) {
*errs = append(*errs, fmt.Sprintf(
"%s: `%s` is not a knowledge kind. Use one of %s.",
where, kind, strings.Join(KnowledgeKinds, ", ")))
return nil
}
if kind == "link" && url == "" {
*errs = append(*errs, where+": a `link` needs a `url`.")
return nil
}
id := jsTrimmed(v["id"])
if id == "" {
id = slugify(label)
}
if id == "" {
id = fmt.Sprintf("k%d", index+1)
}
if label == "" {
label = runeSlice(body, 60)
}
return &Knowledge{ID: id, Label: label, Kind: kind, Body: body, URL: url}
}
*errs = append(*errs, where+": a knowledge entry must be a line of text, or a mapping of options.")
return nil
}
// runeSlice is JavaScript's String.prototype.slice(0, n), which counts UTF-16
// units. Counting runes instead differs only for astral characters, and cutting
// a surrogate pair in half — which the frontend can do — would produce a label
// no comparison could match. Runes are used deliberately; the conformance suite
// carries no case that distinguishes them.
func runeSlice(s string, n int) string {
r := []rune(s)
if len(r) <= n {
return s
}
return string(r[:n])
}
// normalizePermissions reads the `permissions:` block.
func normalizePermissions(raw any, errs *[]string) Permissions {
none := Permissions{Access: DefaultAgentAccess, People: []Person{}}
if raw == nil {
return none
}
mapping, ok := raw.(map[string]any)
if !ok {
*errs = append(*errs, "permissions: must be a mapping of `owner`, `access` and `people`.")
return none
}
access := jsTrimmed(mapping["access"])
if access == "" {
access = DefaultAgentAccess
}
if !contains(AgentAccess, access) {
*errs = append(*errs, fmt.Sprintf(
"permissions.access: `%s` is not an access mode. Use one of %s.",
access, strings.Join(AgentAccess, ", ")))
}
people := []Person{}
for i, entry := range asList(mapping["people"]) {
where := fmt.Sprintf("permissions.people[%d]", i)
person, ok := entry.(map[string]any)
if !ok {
*errs = append(*errs, where+": must be a mapping of `user` and `role`.")
continue
}
user := jsTrimmed(person["user"])
if user == "" {
*errs = append(*errs, where+": needs a `user`.")
continue
}
role := jsTrimmed(person["role"])
if role == "" {
role = DefaultPermission
}
if !contains(PermissionRole, role) {
*errs = append(*errs, fmt.Sprintf(
"%s: `%s` is not a role. Use one of %s.",
where, role, strings.Join(PermissionRole, ", ")))
continue
}
people = append(people, Person{User: user, Role: role})
}
result := Permissions{Owner: jsTrimmed(mapping["owner"]), Access: access, People: people}
if !contains(AgentAccess, access) {
result.Access = DefaultAgentAccess
}
return result
}
// ParseAgent reads an agent definition.
//
// The order in which errors accumulate is part of the contract: validateAgent
// reports the FIRST one, so a definition with two problems must name the same
// one the editor names. That order is status, reasoning, icon, version,
// subagents, starters, knowledge, pages, skills, permissions — which is
// evaluation order in normalizeAgent, counting the object literal it returns.
func ParseAgent(raw string, opts Options) (*Agent, error) {
doc, err := ParseFrontmatter(raw)
if err != nil {
return nil, err
}
data := doc.Data
errs := []string{}
id := jsTrim(jsString(data["id"]))
if !jsTruthy(data["id"]) {
id = slugify(data["name"])
if id == "" {
id = fileStem(opts.path())
}
}
status := jsTrimmed(data["status"])
if status == "" {
status = DefaultAgentStatus
}
if !contains(AgentStatuses, status) {
errs = append(errs, fmt.Sprintf("status: `%s` is not a status. Use one of %s.",
status, strings.Join(AgentStatuses, ", ")))
}
reasoning := jsTrimmed(data["reasoning"])
if reasoning == "" {
reasoning = DefaultReasoning
}
if !contains(ReasoningModes, reasoning) {
errs = append(errs, fmt.Sprintf("reasoning: `%s` is not a reasoning mode. Use one of %s.",
reasoning, strings.Join(ReasoningModes, ", ")))
}
icon := jsTrimmed(data["icon"])
if icon == "" {
icon = DefaultAgentIcon
}
if !contains(AgentIcons, icon) {
errs = append(errs, fmt.Sprintf("icon: `%s` is not an icon this product has.", icon))
}
// A version is an integer that only ever goes up. Anything else is an
// authoring slip, and reading it as 1 is kinder than refusing the file —
// but it is still reported, because a definition that thinks it is v3 and
// registers as v1 will publish over something.
version := 1
if v, present := data["version"]; present && v != nil && v != "" {
parsed := jsNumber(v)
if math.IsNaN(parsed) || parsed != math.Trunc(parsed) || math.IsInf(parsed, 0) || parsed < 1 {
errs = append(errs, fmt.Sprintf(
"version: `%s` is not a whole number of 1 or more.", jsString(v)))
} else if parsed > maxExactInteger {
// Beyond 2^53-1 a float64 no longer names one integer, so there is
// no value to carry. Saturating keeps the conversion below defined,
// and ValidateAgent refuses everything above MaxVersion anyway, so
// a saturated version can never reach a column.
version = maxExactInteger
} else {
version = int(parsed)
}
}
subagents := uniqueStrings(data["subagents"], "subagents", "a subagent id", &errs)
kept := subagents[:0]
for _, s := range subagents {
if id != "" && s == id {
errs = append(errs, "subagents: an agent cannot be its own subagent.")
continue
}
kept = append(kept, s)
}
subagents = kept
starters := []Starter{}
for i, entry := range asList(data["starters"]) {
if s := normalizeStarter(entry, i, &errs); s != nil {
starters = append(starters, *s)
}
}
knowledge := []Knowledge{}
for i, entry := range asList(data["knowledge"]) {
if k := normalizeKnowledge(entry, i, &errs); k != nil {
knowledge = append(knowledge, *k)
}
}
// From here the order follows the object literal normalizeAgent returns.
pages := normalizePages(data["pages"], &errs)
skills := uniqueStrings(data["skills"], "skills", "a skill id", &errs)
permissions := normalizePermissions(data["permissions"], &errs)
instructions, _ := sectionSource(doc.Body, "Instructions")
agent := &Agent{
ID: id,
Name: "Untitled agent",
Status: status,
Version: version,
Pages: pages,
Icon: icon,
Reasoning: reasoning,
Trigger: jsTrimmed(data["trigger"]),
WebSearch: data["webSearch"] == true || data["web_search"] == true,
Skills: skills,
Subagents: subagents,
Starters: starters,
Knowledge: knowledge,
Permissions: permissions,
Instructions: jsTrim(instructions),
Errors: errs,
Body: doc.Body,
}
if jsTruthy(data["name"]) {
agent.Name = jsString(data["name"])
}
if jsTruthy(data["description"]) {
agent.Description = jsString(data["description"])
}
if !contains(AgentStatuses, status) {
agent.Status = DefaultAgentStatus
}
if !contains(ReasoningModes, reasoning) {
agent.Reasoning = DefaultReasoning
}
if !contains(AgentIcons, icon) {
agent.Icon = DefaultAgentIcon
}
return agent, nil
}
// ValidateAgent decides whether an agent definition may be stored.
//
// Returns nil when it may. The order is the order an author would fix things
// in, which is why it reads the same way ValidateSkill does. Note that the
// `id` message differs from the skill one by two words — that difference is
// the frontend's, and it is reproduced rather than tidied.
func ValidateAgent(raw string) error {
if jsTrim(raw) == "" {
return &Rejection{Message: "Paste or upload a Markdown definition."}
}
if n := len([]rune(raw)); n > MaxMarkdownLength {
return &Rejection{
BackendOnly: true,
Message: fmt.Sprintf(
"That definition is %d characters. The limit is %d.", n, MaxMarkdownLength),
}
}
agent, err := ParseAgent(raw, Options{})
if err != nil {
return &Rejection{Message: jsTrim("That definition could not be parsed. " + err.Error())}
}
if agent.ID == "" {
return &Rejection{Message: "The frontmatter needs an `id`."}
}
if !isDefinitionID(agent.ID) {
return &Rejection{Message: "The `id` must be lower-case letters, numbers and dashes."}
}
// `!raw.includes('name:') || agent.name === 'Untitled agent'` — the literal
// substring test is the frontend's, and it is why a definition whose name
// resolves to the fallback is refused even when some other key happens to
// spell `name:`.
if !strings.Contains(raw, "name:") || agent.Name == "Untitled agent" {
return &Rejection{Message: "The frontmatter needs a `name`."}
}
if len(agent.Pages) == 0 {
return &Rejection{Message: "An agent needs at least one `pages:` entry, or it can never be offered anywhere."}
}
if len(agent.Errors) > 0 {
return &Rejection{Message: agent.Errors[0]}
}
// The backend's own bound, the companion to the size rule above:
// agent_definitions.version is a PostgreSQL `integer`, and the frontend
// accepts any whole number of 1 or more. It is checked here rather than in
// ParseAgent so the normalized record stays identical to the frontend's for
// every definition the frontend accepts, and last among the rules so a
// definition the frontend also refuses is refused with the frontend's own
// message.
if agent.Version > MaxVersion {
return &Rejection{
BackendOnly: true,
Message: fmt.Sprintf(
"version: `%d` is larger than %d.", agent.Version, MaxVersion),
}
}
return nil
}

View File

@@ -0,0 +1,928 @@
package definition_test
import (
"encoding/base64"
"encoding/json"
"fmt"
"os"
"reflect"
"sort"
"strconv"
"strings"
"testing"
"github.com/krow/krow-backend/go-api/internal/definition"
)
// Phase 4D — JS/Go parser conformance.
//
// The fixture these tests read (testdata/oracle.json) is not written by hand.
// It is captured by scripts/oracle.mjs, which loads the REAL frontend module
// graph through Vite — import.meta.glob, the `@/` alias and raw Markdown
// loading all behave exactly as they do in the app — and records what the
// JavaScript parser did with every shipped definition and every adversarial
// case. So the assertion below is not "Go agrees with a description of the
// frontend"; it is "Go agrees with the frontend", replayed.
//
// Regenerate after any change to src/lib/skills or src/lib/agents:
//
// node scripts/oracle.mjs go-api/internal/definition/testdata/oracle.json
//
// A frontend change that alters parsing therefore fails these tests, which is
// the point: the contract cannot drift silently in either direction.
type oracle struct {
Vocabulary struct {
Pages []struct {
ID string `json:"id"`
Aliases []string `json:"aliases"`
} `json:"pages"`
AgentStatuses []string `json:"agentStatuses"`
Reasoning []string `json:"reasoning"`
Icons []string `json:"icons"`
KnowledgeKinds []string `json:"knowledgeKinds"`
Access []string `json:"access"`
Roles []string `json:"roles"`
} `json:"vocabulary"`
Corpus []observation `json:"corpus"`
Cases []observation `json:"cases"`
}
// observation is one definition as the JavaScript saw it, end to end.
type observation struct {
// Exactly one of these identifies the row.
Path string `json:"path"`
Name string `json:"name"`
Type string `json:"type"`
Kind string `json:"kind"` // agent | skill
RawBase64 string `json:"rawBase64"`
HasFrontmatter bool `json:"hasFrontmatter"`
Frontmatter struct {
OK bool `json:"ok"`
Data map[string]any `json:"data"`
Body string `json:"body"`
Error string `json:"error"`
} `json:"frontmatter"`
Parse struct {
OK bool `json:"ok"`
Error string `json:"error"`
} `json:"parse"`
Normalized map[string]any `json:"normalized"`
Accepted bool `json:"accepted"`
Rejection *string `json:"rejection"`
}
func (o observation) id() string {
if o.Path != "" {
return o.Path
}
return o.Name
}
func (o observation) raw(t *testing.T) string {
t.Helper()
b, err := base64.StdEncoding.DecodeString(o.RawBase64)
if err != nil {
t.Fatalf("%s: undecodable fixture: %v", o.id(), err)
}
return string(b)
}
func load(t *testing.T) *oracle {
t.Helper()
b, err := os.ReadFile("testdata/oracle.json")
if err != nil {
t.Fatalf("read fixture: %v", err)
}
var o oracle
if err := json.Unmarshal(b, &o); err != nil {
t.Fatalf("parse fixture: %v", err)
}
if len(o.Corpus) == 0 || len(o.Cases) == 0 {
t.Fatal("fixture is empty; regenerate with scripts/oracle.mjs")
}
return &o
}
func all(o *oracle) []observation { return append(append([]observation{}, o.Corpus...), o.Cases...) }
/* ── 1. The corpus is the corpus ──────────────────────────────────────────── */
// The shipped definition count, asserted rather than assumed. A definition
// added to or removed from the product without regenerating the fixture leaves
// these tests passing against a corpus that no longer exists, which is the one
// way this suite could quietly stop meaning anything.
func TestCorpusShape(t *testing.T) {
o := load(t)
counts := map[string]int{}
for _, c := range o.Corpus {
counts[c.Type]++
}
for _, want := range []struct {
kind string
n int
}{{"agent", 9}, {"skill", 23}, {"example", 5}} {
if counts[want.kind] != want.n {
t.Errorf("%s definitions: got %d, want %d", want.kind, counts[want.kind], want.n)
}
}
if len(o.Corpus) != 37 {
t.Errorf("shipped definitions: got %d, want 37", len(o.Corpus))
}
}
/* ── 2. The vocabulary has not drifted ────────────────────────────────────── */
// Every closed table in vocabulary.go, checked against the table the frontend
// actually exports. A page added to surfaces.js fails here rather than becoming
// a definition the editor accepts and the API rejects.
func TestVocabularyMatchesFrontend(t *testing.T) {
o := load(t)
wantPages := make([]string, len(o.Vocabulary.Pages))
for i, p := range o.Vocabulary.Pages {
wantPages[i] = p.ID
}
if !reflect.DeepEqual(definition.SupportedPages, wantPages) {
t.Errorf("supported pages differ\n go %v\n js %v", definition.SupportedPages, wantPages)
}
// Aliases resolve, and resolve to the same canonical id.
for _, p := range o.Vocabulary.Pages {
for _, alias := range append([]string{p.ID}, p.Aliases...) {
if got := definition.CanonicalPage(alias); got != p.ID {
t.Errorf("CanonicalPage(%q) = %q, want %q", alias, got, p.ID)
}
}
}
for _, table := range []struct {
name string
got []string
wanted []string
}{
{"agent statuses", definition.AgentStatuses, o.Vocabulary.AgentStatuses},
{"reasoning modes", definition.ReasoningModes, o.Vocabulary.Reasoning},
{"icons", definition.AgentIcons, o.Vocabulary.Icons},
{"knowledge kinds", definition.KnowledgeKinds, o.Vocabulary.KnowledgeKinds},
{"access modes", definition.AgentAccess, o.Vocabulary.Access},
{"permission roles", definition.PermissionRole, o.Vocabulary.Roles},
} {
if !reflect.DeepEqual(table.got, table.wanted) {
t.Errorf("%s differ\n go %v\n js %v", table.name, table.got, table.wanted)
}
}
}
/* ── 3. The frontmatter tree ──────────────────────────────────────────────── */
// The deepest parity check available: the YAML subset must produce the same
// data structure the JavaScript produced, for every definition and every
// adversarial case. Not a projection of it — the whole tree.
func TestFrontmatterTreeParity(t *testing.T) {
for _, c := range all(load(t)) {
t.Run(c.id(), func(t *testing.T) {
raw := c.raw(t)
doc, err := definition.ParseFrontmatter(raw)
if !c.Frontmatter.OK {
if err == nil {
t.Fatalf("JS refused this frontmatter (%s); Go accepted it", c.Frontmatter.Error)
}
if err.Error() != c.Frontmatter.Error {
t.Errorf("error text differs\n go %q\n js %q", err.Error(), c.Frontmatter.Error)
}
return
}
if err != nil {
t.Fatalf("JS read this frontmatter; Go refused it: %v", err)
}
if got, want := normalizeTree(doc.Data), normalizeTree(c.Frontmatter.Data); !reflect.DeepEqual(got, want) {
t.Errorf("frontmatter differs\n go %s\n js %s", show(got), show(want))
}
if doc.Body != c.Frontmatter.Body {
t.Errorf("body differs\n go %q\n js %q", doc.Body, c.Frontmatter.Body)
}
if got := definition.HasFrontmatter(raw); got != c.HasFrontmatter {
t.Errorf("HasFrontmatter = %v, JS said %v", got, c.HasFrontmatter)
}
})
}
}
// normalizeTree puts a parsed tree into the shape `encoding/json` would have
// produced, so the Go value and the value round-tripped through the fixture's
// JSON are comparable. Numbers become float64 on both sides, which is what
// JavaScript had in the first place.
func normalizeTree(v any) any {
b, err := json.Marshal(v)
if err != nil {
return fmt.Sprintf("unmarshalable: %v", err)
}
var out any
if err := json.Unmarshal(b, &out); err != nil {
return fmt.Sprintf("unmarshalable: %v", err)
}
return out
}
func show(v any) string {
b, _ := json.Marshal(v)
return string(b)
}
/* ── 4. Accept / reject parity ────────────────────────────────────────────── */
// knownDivergence is the complete list of definitions where the two parsers
// disagree, each with the reason. It is a CLOSED list: anything not on it that
// disagrees fails, and anything on it that stops disagreeing fails too, so the
// list cannot quietly grow and cannot quietly go stale.
//
// Every entry is a case where Go is stricter, except the first — and the first
// is the one asymmetry this package documents as deferred.
var knownDivergence = map[string]string{
"skill-examples/board-invalid-context.md": "" +
"JS rejects on `ui:` placement/source semantics, which this package defers " +
"to the frontend. Go accepts and reports Deferred: [ui].",
"oversized-markdown": "" +
"JS accepts; the database refuses it (markdown_size CHECK). Go refuses it " +
"first, so an author gets a message instead of a constraint violation.",
"oversized-agent": "" +
"JS accepts; the database refuses it (markdown_size CHECK). Go refuses it " +
"first, so an author gets a message instead of a constraint violation.",
"version-above-int32-agent": "" +
"JS accepts any whole number of 1 or more; agent_definitions.version is a " +
"PostgreSQL `integer`, so the database refuses this one. Go refuses it " +
"first, for the same reason as the size bound.",
}
func TestAcceptanceParity(t *testing.T) {
seen := map[string]bool{}
for _, c := range all(load(t)) {
t.Run(c.id(), func(t *testing.T) {
raw := c.raw(t)
var err error
if c.Kind == "agent" {
err = definition.ValidateAgent(raw)
} else {
err = definition.ValidateSkill(raw)
}
accepted := err == nil
if reason, expected := knownDivergence[c.id()]; expected {
seen[c.id()] = true
if accepted == c.Accepted {
t.Errorf("listed as a known divergence but the two now agree (%v).\n"+
"Remove it from knownDivergence.\n reason on file: %s", accepted, reason)
}
return
}
if accepted != c.Accepted {
t.Fatalf("acceptance differs: go=%v js=%v\n go said: %v\n js said: %v",
accepted, c.Accepted, err, deref(c.Rejection))
}
})
}
for id := range knownDivergence {
if !seen[id] {
t.Errorf("knownDivergence names %q, which is not in the fixture", id)
}
}
}
// Where both refuse a definition, they must refuse it for the same stated
// reason. A parser that rejects the right definitions with the wrong messages
// sends an author to the wrong line.
func TestRejectionMessageParity(t *testing.T) {
for _, c := range all(load(t)) {
if c.Accepted || c.Rejection == nil {
continue
}
if _, skip := knownDivergence[c.id()]; skip {
continue
}
t.Run(c.id(), func(t *testing.T) {
raw := c.raw(t)
var err error
if c.Kind == "agent" {
err = definition.ValidateAgent(raw)
} else {
err = definition.ValidateSkill(raw)
}
if err == nil {
t.Fatalf("JS rejected this; Go accepted it")
}
if r, ok := err.(*definition.Rejection); ok && r.BackendOnly {
t.Fatalf("refused by a backend-only rule where JS refused it too: %q", r.Message)
}
if err.Error() != *c.Rejection {
t.Errorf("rejection differs\n go %q\n js %q", err.Error(), *c.Rejection)
}
})
}
}
func deref(s *string) string {
if s == nil {
return "<accepted>"
}
return *s
}
/* ── 5. Normalized projection parity ──────────────────────────────────────── */
// textCoercion is the complete list of fixture cases where the JavaScript
// record holds a value that is NOT a string in a field migration 000005
// projects into a `text` or `text[]` column, named field by field.
//
// The frontend can afford this and the backend cannot. `name: [a, b]` leaves a
// JavaScript ARRAY on skill.name, and every consumer stringifies it at the
// point of use — the trigger list on that very record reads "a,b", which is
// String(["a","b"]). A text column has no such option: something must be
// written, once, at the boundary. This package writes String(x), which is the
// string the frontend's own consumers produce.
//
// Listing them rather than coercing everywhere is the point. Coercion is
// applied ONLY to the fields named here, so a genuine difference between two
// strings still fails; each entry is checked to be still necessary, so the list
// cannot go stale; and an unlisted case that needs coercion fails outright, so
// the list cannot quietly grow. It is the same closed-list discipline
// knownDivergence has, for the same reason.
var textCoercion = map[string][]string{
"invalid-field-type-name-list": {"name"},
"name-list-agent": {"name"},
"name-numeric": {"name"},
"name-boolean": {"name"},
"description-list": {"description"},
"description-numeric": {"description"},
"pages-numeric-entry": {"pages"},
"pages-mapping-entry": {"pages"},
}
// The two shapes a text projection can have, named rather than inferred.
//
// Which one applies is a property of the COLUMN, not of what the author
// happened to write. `name` is `text`, so a sequence written there becomes one
// string — String(["a","b"]) is "a,b". `pages` is `text[]`, so a sequence stays
// a sequence and each entry becomes a string of its own. Inferring the shape
// from whatever Go produced would make the test agree with the parser by
// construction, which is the one thing it must not do.
var textScalarFields = map[string]bool{
"name": true, "description": true, "category": true, "trigger": true, "prompt": true,
}
var textListFields = map[string]bool{
"pages": true, "actions": true, "triggers": true, "skills": true, "subagents": true,
}
// jsText is String(x) for a value decoded from the fixture's JSON — the same
// conversion jsvalue.go performs inside the parser, restated here so the test
// does not have to reach into the package it is testing to check it.
func jsText(t *testing.T, field string, v any) any {
t.Helper()
switch {
case textScalarFields[field]:
return jsScalarText(v)
case textListFields[field]:
list, ok := v.([]any)
if !ok {
return jsScalarText(v)
}
out := make([]any, len(list))
for i, item := range list {
out[i] = jsScalarText(item)
}
return out
}
t.Fatalf("textCoercion names %q, which is not a text-projected field", field)
return nil
}
func jsScalarText(v any) string {
switch x := v.(type) {
case nil:
return "null"
case bool:
if x {
return "true"
}
return "false"
case float64:
if x == float64(int64(x)) {
return strconv.FormatInt(int64(x), 10)
}
return strconv.FormatFloat(x, 'g', -1, 64)
case string:
return x
case []any:
// Array.prototype.toString: nil renders as the empty string, not
// "null", which is the one place the two differ.
parts := make([]string, len(x))
for i, item := range x {
if item == nil {
continue
}
parts[i] = jsScalarText(item)
}
return strings.Join(parts, ",")
case map[string]any:
return "[object Object]"
}
return ""
}
// The fields migration 000005 projects into columns, compared for every
// definition both parsers accept. These are the values that reach the
// database, so a difference here is a row the frontend would render wrongly.
func TestProjectionParity(t *testing.T) {
fixture := map[string]bool{}
for _, c := range all(load(t)) {
fixture[c.id()] = true
if c.Normalized == nil {
continue // JS could not parse it; covered by the tree test
}
coerce := map[string]bool{}
for _, field := range textCoercion[c.id()] {
coerce[field] = true
}
t.Run(c.id(), func(t *testing.T) {
raw := c.raw(t)
if c.Kind == "agent" {
agent, err := definition.ParseAgent(raw, definition.Options{})
if err != nil {
t.Fatalf("JS parsed this; Go refused it: %v", err)
}
compare(t, map[string]any{
"id": agent.ID,
"name": agent.Name,
"description": agent.Description,
"status": agent.Status,
"version": agent.Version,
"pages": agent.Pages,
"icon": agent.Icon,
"reasoning": agent.Reasoning,
"trigger": agent.Trigger,
"webSearch": agent.WebSearch,
"skills": agent.Skills,
"subagents": agent.Subagents,
"starters": agent.Starters,
"permissions": agent.Permissions,
"errors": agent.Errors,
}, c.Normalized, coerce)
return
}
skill, err := definition.ParseSkill(raw, definition.Options{})
if err != nil {
t.Fatalf("JS parsed this; Go refused it: %v", err)
}
compare(t, map[string]any{
"id": skill.ID,
"name": skill.Name,
"description": skill.Description,
"status": skill.Status,
"pages": skill.Pages,
"kind": skill.Kind,
"category": skill.Category,
"actions": skill.Actions,
"triggers": skill.Triggers,
"declaredTriggers": skill.DeclaredTriggers,
"prompt": skill.Prompt,
"skillId": skill.SkillID,
}, c.Normalized, coerce)
})
}
for id := range textCoercion {
if !fixture[id] {
t.Errorf("textCoercion names %q, which is not in the fixture", id)
}
}
}
// compare checks every field Go produced against the JS record, field by field
// so a failure names the field rather than dumping two objects.
func compare(t *testing.T, got map[string]any, want map[string]any, coerce map[string]bool) {
t.Helper()
keys := make([]string, 0, len(got))
for k := range got {
keys = append(keys, k)
}
sort.Strings(keys)
for _, k := range keys {
wantValue, present := want[k]
if !present {
t.Errorf("%s: absent from the JS record", k)
continue
}
if coerce[k] {
// Listed in textCoercion. Check the entry is still earning its
// place before honouring it: if the JS value is already the string
// Go produced, the coercion is doing nothing and the list has gone
// stale.
coerced := jsText(t, k, normalizeTree(wantValue))
if reflect.DeepEqual(normalizeTree(wantValue), normalizeTree(coerced)) {
t.Errorf("%s: listed in textCoercion, but the JS value is already "+
"a string. Remove the entry.", k)
}
wantValue = coerced
}
g, w := normalizeTree(got[k]), normalizeTree(wantValue)
// An empty list and a missing one are the same thing to both parsers.
if isEmptyList(g) && isEmptyList(w) {
continue
}
if !reflect.DeepEqual(g, w) {
t.Errorf("%s differs\n go %s\n js %s", k, show(g), show(w))
}
}
}
func isEmptyList(v any) bool {
if v == nil {
return true
}
l, ok := v.([]any)
return ok && len(l) == 0
}
/* ── 6. The parser never rewrites what is stored ──────────────────────────── */
// Migration 000005 keeps `markdown` verbatim and derives every other column
// from it. Parsing must therefore be a read: normalization exists to
// INTERPRET a definition, never to rewrite it.
func TestParsingDoesNotMutateSource(t *testing.T) {
for _, c := range all(load(t)) {
raw := c.raw(t)
before := string(append([]byte{}, raw...))
_, _ = definition.ParseSkill(raw, definition.Options{})
_, _ = definition.ParseAgent(raw, definition.Options{})
_ = definition.ValidateSkill(raw)
_ = definition.ValidateAgent(raw)
if raw != before {
t.Fatalf("%s: the source changed under the parser", c.id())
}
}
}
// Normalize is what the parser reads THROUGH; what it returns must never be
// what gets stored. Asserted directly, because the whole separation rests on
// it: the corpus contains definitions whose normalized form differs from their
// stored form, and storing the normalized one would silently rewrite an
// author's file.
func TestNormalizationIsNotStorage(t *testing.T) {
rewritten := 0
for _, c := range all(load(t)) {
raw := c.raw(t)
if definition.Normalize(raw) != raw {
rewritten++
}
}
if rewritten == 0 {
t.Fatal("no case in the corpus is changed by Normalize; " +
"this test can no longer tell storage and interpretation apart")
}
t.Logf("%d of %d definitions normalize to something other than their stored bytes", rewritten, len(all(load(t))))
}
/* ── 7. Adversarial coverage is real ──────────────────────────────────────── */
// The adversarial cases Phase 4D requires, each mapped to the fixture rows that
// exercise it. A case list that drifts away from the requirement is a suite
// that looks thorough and tests something else.
func TestAdversarialCoverage(t *testing.T) {
required := map[string][]string{
"UTF-8 BOM": {"utf8-bom", "utf8-bom-agent", "bom-crlf-blankline"},
"CRLF": {"crlf", "crlf-agent", "crlf-inside-frontmatter-only"},
"CR": {"cr-only"},
"leading blank line": {"leading-blank-line"},
"multiple leading blank lines": {"multiple-leading-blank-lines", "leading-spaces-then-blank-lines"},
"trailing spaces": {"trailing-spaces-on-values"},
"trailing newline": {"many-trailing-newlines", "no-trailing-newline"},
"trailing ws after fence": {"trailing-ws-after-open-fence", "trailing-tab-after-close-fence"},
"quoted scalar": {"double-quoted-scalar", "doubled-quote-escape"},
"single-quoted scalar": {"single-quoted-scalar"},
"colon inside quoted string": {"colon-in-quoted-string", "colon-in-unquoted-string"},
"hash inside quoted string": {"hash-in-quoted-string", "hash-unquoted-trailing-comment", "hash-unquoted-midword"},
"empty scalar": {"empty-scalar", "tilde-scalar", "null-scalar"},
"empty array": {"empty-array"},
"inline array": {"inline-flow-array", "inline-flow-map"},
"multiline scalar": {"block-scalar-literal", "block-scalar-folded"},
"duplicate key": {"duplicate-key", "duplicate-key-array"},
"malformed YAML": {"malformed-yaml-bare-line", "key-with-space", "ragged-indent"},
"malformed opening fence": {"malformed-open-fence-two-dashes", "malformed-open-fence-four-dashes",
"malformed-open-fence-indented", "malformed-open-fence-text-after"},
"malformed closing fence": {"malformed-close-fence-two-dashes", "malformed-close-fence-missing",
"malformed-close-fence-four-dashes"},
"missing frontmatter": {"missing-frontmatter", "empty-fence-pair", "frontmatter-is-a-sequence"},
"unsupported frontmatter field": {"unsupported-frontmatter-field", "unsupported-field-agent", "uppercase-key"},
"invalid field type": {"invalid-field-type-pages-scalar", "invalid-field-type-pages-scalar-agent",
"invalid-field-type-name-list"},
"invalid definition id": {"invalid-definition-id-uppercase", "invalid-definition-id-leading-dash",
"invalid-definition-id-underscore"},
"invalid status": {"invalid-status-skill", "invalid-status-agent", "inactive-status-skill"},
"invalid visibility": {"visibility-field-personal", "visibility-field-invalid"},
"oversized markdown": {"oversized-markdown", "oversized-agent", "at-size-bound"},
"empty markdown": {"empty-markdown", "whitespace-only-markdown"},
}
present := map[string]bool{}
for _, c := range load(t).Cases {
present[c.Name] = true
}
for requirement, names := range required {
for _, n := range names {
if !present[n] {
t.Errorf("%q: the fixture has no case named %q", requirement, n)
}
}
}
}
/* ── 8. Deferred blocks are reported, not assumed ─────────────────────────── */
// Every skill carrying a `ui:` or `owliver:` block must say so, because that is
// the one part of validation this package does not do. A block that stopped
// being reported would be a gap nobody could see.
func TestDeferredBlocksAreReported(t *testing.T) {
o := load(t)
found := 0
for _, c := range append(append([]observation{}, o.Corpus...), o.Cases...) {
if c.Kind != "skill" || !c.Frontmatter.OK {
continue
}
want := []string{}
for _, key := range []string{"ui", "owliver"} {
if _, present := c.Frontmatter.Data[key]; present {
want = append(want, key)
}
}
skill, err := definition.ParseSkill(c.raw(t), definition.Options{})
if err != nil {
continue
}
if len(want) == 0 {
if len(skill.Deferred) != 0 {
t.Errorf("%s: reported Deferred %v with no such block", c.id(), skill.Deferred)
}
continue
}
found++
if !reflect.DeepEqual(skill.Deferred, want) {
t.Errorf("%s: Deferred = %v, want %v", c.id(), skill.Deferred, want)
}
}
if found != 19 {
t.Errorf("definitions carrying a deferred block: got %d, want 19", found)
}
}
/* ── 9. Mutation checks ───────────────────────────────────────────────────── */
// Tests that pass against a broken parser are not tests. Each mutation below
// is a plausible mistake in this package; every one must be caught by a real
// definition changing its meaning, not by an assertion written to notice it.
func TestMutationsWouldBeCaught(t *testing.T) {
base := strings.Join([]string{
"---",
"id: sample-skill",
"name: Sample Skill",
"description: A sample.",
"pages:",
" - candidates",
"---",
"",
"# Sample Skill",
}, "\n")
mutations := []struct {
name string
raw string
check func(t *testing.T, s *definition.Skill, err error)
}{
{
// Dropping the BOM strip: the fence stops matching and every field
// empties out.
name: "BOM before the fence still fences",
raw: "\uFEFF" + base,
check: func(t *testing.T, s *definition.Skill, err error) {
if err != nil || s.ID != "sample-skill" || len(s.Pages) != 1 {
t.Errorf("got id=%q pages=%v err=%v", s.ID, s.Pages, err)
}
},
},
{
// Dropping CR normalization: `candidates\r` is not a page.
name: "CRLF endings do not leak into values",
raw: strings.ReplaceAll(base, "\n", "\r\n"),
check: func(t *testing.T, s *definition.Skill, err error) {
if err != nil || len(s.Pages) != 1 || s.Pages[0] != "candidates" {
t.Errorf("got pages=%v err=%v", s.Pages, err)
}
},
},
{
// Trimming the closing fence too eagerly, or not at all.
name: "trailing tab after the closing fence still closes it",
raw: strings.Replace(base, "\n---\n", "\n---\t\n", 1),
check: func(t *testing.T, s *definition.Skill, err error) {
if err != nil || s.Name != "Sample Skill" {
t.Errorf("got name=%q err=%v", s.Name, err)
}
},
},
{
// A greedy fence would swallow the second document and lose the id.
name: "a second --- document is body, not frontmatter",
raw: base + "\n\n---\nid: second\n---\n",
check: func(t *testing.T, s *definition.Skill, err error) {
if err != nil || s.ID != "sample-skill" {
t.Errorf("got id=%q err=%v", s.ID, err)
}
},
},
{
// Treating `#` as always starting a comment.
name: "a hash inside a word is part of the word",
raw: strings.Replace(base, "description: A sample.", "category: ops#1", 1),
check: func(t *testing.T, s *definition.Skill, err error) {
if err != nil || s.Category != "ops#1" {
t.Errorf("got category=%q err=%v", s.Category, err)
}
},
},
{
// Treating a spaced `#` as part of the value.
name: "a spaced hash starts a comment",
raw: strings.Replace(base, "description: A sample.", "category: ops # note", 1),
check: func(t *testing.T, s *definition.Skill, err error) {
if err != nil || s.Category != "ops" {
t.Errorf("got category=%q err=%v", s.Category, err)
}
},
},
{
// Splitting a quoted value on its colon.
name: "a colon inside quotes stays in the value",
raw: strings.Replace(base, "name: Sample Skill", `name: "Sample: Skill"`, 1),
check: func(t *testing.T, s *definition.Skill, err error) {
if err != nil || s.Name != "Sample: Skill" {
t.Errorf("got name=%q err=%v", s.Name, err)
}
},
},
{
// Keeping the first duplicate rather than the last.
name: "a duplicate key takes the last value",
raw: strings.Replace(base, "name: Sample Skill", "name: First\nname: Second", 1),
check: func(t *testing.T, s *definition.Skill, err error) {
if err != nil || s.Name != "Second" {
t.Errorf("got name=%q err=%v", s.Name, err)
}
},
},
{
// Accepting ragged indentation instead of refusing it.
name: "ragged indentation is refused with its line",
raw: strings.Replace(base, " - candidates", " - candidates\n - positions", 1),
check: func(t *testing.T, s *definition.Skill, err error) {
var pe *definition.Error
if err == nil {
t.Fatalf("accepted ragged indentation: %+v", s)
}
if !asError(err, &pe) || pe.Line != 6 {
t.Errorf("got %v, want an *Error on line 6", err)
}
},
},
{
// Canonicalising a skill's pages, which the frontend does not do.
name: "a skill keeps the page name as written",
raw: strings.Replace(base, " - candidates", " - Talent Pool", 1),
check: func(t *testing.T, s *definition.Skill, err error) {
if err != nil || len(s.Pages) != 1 || s.Pages[0] != "Talent Pool" {
t.Errorf("got pages=%v err=%v", s.Pages, err)
}
if err := definition.ValidateSkill(strings.Replace(base, " - candidates", " - Talent Pool", 1)); err != nil {
t.Errorf("an aliased page should still validate: %v", err)
}
},
},
}
for _, m := range mutations {
t.Run(m.name, func(t *testing.T) {
skill, err := definition.ParseSkill(m.raw, definition.Options{})
if skill == nil {
skill = &definition.Skill{}
}
m.check(t, skill, err)
})
}
}
// asError is errors.As, spelled out for the one concrete type this package
// returns.
func asError(err error, target **definition.Error) bool {
e, ok := err.(*definition.Error)
if ok {
*target = e
}
return ok
}
/* ── 10. Bounds ───────────────────────────────────────────────────────────── */
// The size bound is the backend's, and both directions of it matter: a
// definition at the limit must be storable and one character more must not.
// The companion to TestSizeBound, for the other backend-only bound. The
// fixture pins a version well above the bound and one exactly at it, which
// leaves the step between them untested — an off-by-one there would refuse a
// version PostgreSQL can store, or accept one it cannot. Both sides of the
// step are named here so that cannot happen.
func TestVersionBound(t *testing.T) {
agent := func(version string) string {
return "---\nid: sample-agent\nname: Sample Agent\npages:\n - candidates\n" +
"version: " + version + "\n---\n\n# Sample Agent\n"
}
at := strconv.Itoa(definition.MaxVersion)
if err := definition.ValidateAgent(agent(at)); err != nil {
t.Errorf("version %s, exactly at the bound, was refused: %v", at, err)
}
over := strconv.FormatInt(int64(definition.MaxVersion)+1, 10)
err := definition.ValidateAgent(agent(over))
if err == nil {
t.Fatalf("version %s, one past the bound, was accepted", over)
}
r, ok := err.(*definition.Rejection)
if !ok || !r.BackendOnly {
t.Errorf("the version bound should be reported as a backend-only rule, got %v", err)
}
// The bound belongs to validation, not to parsing: a version the database
// cannot store must still normalize to the number the author wrote, or the
// record the editor shows and the record Go builds would disagree.
parsed, err := definition.ParseAgent(agent(over), definition.Options{})
if err != nil {
t.Fatalf("parsing a too-large version failed: %v", err)
}
if got := strconv.Itoa(parsed.Version); got != over {
t.Errorf("parse clamped the version to %s; it should carry %s", got, over)
}
if len(parsed.Errors) != 0 {
t.Errorf("parse reported a backend-only bound as an authoring error: %v", parsed.Errors)
}
}
func TestSizeBound(t *testing.T) {
head := "---\nid: sample-skill\nname: Sample Skill\npages:\n - candidates\n---\n\n"
at := head + strings.Repeat("y", definition.MaxMarkdownLength-len(head))
if n := len([]rune(at)); n != definition.MaxMarkdownLength {
t.Fatalf("fixture is %d characters, wanted exactly %d", n, definition.MaxMarkdownLength)
}
if err := definition.ValidateSkill(at); err != nil {
t.Errorf("a definition exactly at the bound was refused: %v", err)
}
over := at + "y"
err := definition.ValidateSkill(over)
if err == nil {
t.Fatal("a definition one character over the bound was accepted")
}
r, ok := err.(*definition.Rejection)
if !ok || !r.BackendOnly {
t.Errorf("the size bound should be reported as a backend-only rule, got %v", err)
}
}

View File

@@ -0,0 +1,102 @@
// Package definition reads Krow agent and skill definitions — Markdown with
// YAML frontmatter — the way the frontend reads them.
//
// # Why this exists
//
// A definition is authored in the browser and stored by the server, so two
// parsers see it: the JavaScript in src/lib/skills and src/lib/agents, and
// this one. If they disagree, one of two things happens, and both are silent:
//
// - A definition the editor accepts and this package rejects looks valid
// while it is being written and fails when it is saved.
// - A definition this package accepts and the editor rejects is stored and
// then cannot be rendered by the product that owns it.
//
// Compatibility is therefore a contract rather than an aspiration, and it is
// enforced by a conformance suite (conformance_test.go) that replays the
// ACTUAL output of the JavaScript parser — captured from the real frontend
// module graph — against this one, over all 37 shipped definitions and every
// adversarial case in testdata/oracle.json. The corpus count is pinned by a
// test; the case count is deliberately not restated here, because a number
// kept in a comment is a number that goes stale.
//
// # What is in the contract
//
// - The document layer: byte-order mark, line endings, leading blank lines,
// fence recognition, body extraction. See frontmatter.go.
// - The YAML subset: block maps and sequences, scalars, quoting, comments.
// A port of yaml.js, with no YAML dependency, deliberately — see yaml.go.
// - Definition-level normalization and validation: id, name, description,
// status, version, pages, icons, reasoning, permissions, starters,
// knowledge, subagents.
//
// Those cover every column migration 000005 projects out of a definition:
// definition_id, status, version, name, description, pages.
//
// # What is deferred, and why
//
// A skill may carry a `ui:` block (declarative page sections) or an `owliver:`
// block (assistant capabilities). Validating those means reproducing roughly
// 1,500 lines of closed vocabulary describing what the FRONTEND can render —
// placements, data sources, section types, periods — none of which the backend
// stores, projects, or acts on.
//
// This package therefore does not check them. It records their presence on
// Skill.Deferred instead, so the gap is a value a caller can see rather than
// an assumption. The one consequence is stated exactly:
//
// skill-examples/board-invalid-context.md is rejected by the frontend, on a
// rule about which placement can supply which data source, and accepted
// here. It is the only definition in the corpus where the two disagree, and
// the conformance suite asserts that it stays the only one.
//
// # What is not the parser's job
//
// Normalization never rewrites what is stored. Migration 000005 keeps
// `markdown` verbatim and every other column is derived from it; this package
// only ever reads. The Markdown handed in is the Markdown that goes to the
// database, byte for byte, and a test asserts it.
//
// Visibility (personal or organization) is deliberately absent. It is not a
// frontmatter field — the frontend ignores `visibility:` in a definition
// entirely — it is a storage tier chosen by the request and checked by the
// visibility CHECK in migration 000005. A definition cannot name its own
// tenancy.
package definition
// MaxMarkdownLength is the markdown_size CHECK from migration 000005, in
// CHARACTERS — `length()` in PostgreSQL counts characters, not bytes.
//
// The frontend does NOT enforce this, so a definition longer than this is one
// the editor accepts and the database refuses. This package refuses it first,
// which turns a constraint violation into a message an author can act on.
const MaxMarkdownLength = 65536
// MaxVersion is the range of agent_definitions.version, a PostgreSQL
// `integer`.
//
// The frontend accepts any integer of 1 or more, so a version above this is
// another value the editor accepts and the database cannot store.
const MaxVersion = 2147483647
// maxExactInteger is 2^53-1, the largest integer a float64 names exactly and so
// the largest a JavaScript number carries without loss. It bounds the version
// conversion in ParseAgent; it is not a rule about what may be stored, which is
// MaxVersion's job.
const maxExactInteger = 1<<53 - 1
// Rejection is a definition that parses but may not be stored.
//
// Message is the frontend's own wording wherever the rule is shared, so the
// editor and the API describe the same problem the same way.
type Rejection struct {
Message string
// BackendOnly marks a rule the frontend does not have — a bound the
// database imposes that the editor never checks. These are the only
// messages that can differ from what an author would see in the browser,
// and each one is listed in docs/phase-4d-parser-contract.md.
BackendOnly bool
}
func (r *Rejection) Error() string { return r.Message }

View File

@@ -0,0 +1,332 @@
package definition
import (
"strings"
)
// The document layer: what is frontmatter, what is body, and what a `## Heading`
// section contains.
//
// A port of the four exported readers in src/lib/skills/registry.js —
// normalizeDefinition, hasFrontmatter, parseFrontmatter and the section
// readers. Agent and skill definitions are read by the same code on the
// frontend, deliberately, so that the two formats cannot drift; the same is
// true here.
//
// The regular expressions the JavaScript uses are hand-rolled rather than
// translated, because two of them rely on lookahead and lazy matching that RE2
// does not have. Each is written out below with the JavaScript it reproduces.
// Normalize is the frontend's `normalizeDefinition`: a definition's text as the
// parser needs to see it.
//
// Files arrive from editors, from Windows, from copy-paste and from downloads,
// and four of the things they arrive with used to take the whole frontmatter
// block down — a UTF-8 byte-order mark before the opening fence, blank lines
// above it, CRLF endings, and trailing spaces after `---`. In each case the
// fence did not match and the definition registered as untitled with no pages.
//
// This is NOT a lenient parser. The subset inside the fences is exactly as
// strict as it was. This is only about recognising that a fence is a fence.
//
// String(raw ?? '')
// .replace(/^\uFEFF/, '')
// .replace(/\r\n?/g, '\n')
// .replace(/^\s*\n+/, '')
//
// The order is load bearing: the BOM goes first so it cannot be counted as the
// leading whitespace, and CR normalisation goes before the blank-line strip so
// that a CRLF blank line is one.
func Normalize(raw string) string {
text := strings.TrimPrefix(raw, "\uFEFF")
// `\r\n?` → `\n`: a CRLF pair and a lone CR both become one newline.
if strings.IndexByte(text, '\r') >= 0 {
var b strings.Builder
b.Grow(len(text))
for i := 0; i < len(text); i++ {
if text[i] != '\r' {
b.WriteByte(text[i])
continue
}
b.WriteByte('\n')
if i+1 < len(text) && text[i+1] == '\n' {
i++
}
}
text = b.String()
}
// `^\s*\n+` → ``. Greedy `\s*` then at least one newline: the effect is to
// drop the leading whitespace run up to and including its LAST newline, and
// to drop nothing at all when that run contains no newline. A definition
// indented by one space is therefore still unfenced, which is what the
// editor decides too.
end, last := 0, -1
for i, r := range text {
if !jsIsSpace(r) {
break
}
if r == '\n' {
last = i
}
end = i + len(string(r))
}
_ = end
if last >= 0 {
text = text[last+1:]
}
return text
}
// fence locates the frontmatter block in already-normalized text.
//
// /^---[ \t]*\n([\s\S]*?)\n---[ \t]*(?=\n|$)/
//
// Returns the YAML source, the offset just past the closing fence, and whether
// there was one. Lazy: the FIRST closing fence wins, which is why a definition
// carrying a second `---` document keeps only the first and reads the rest as
// body.
func fence(text string) (yaml string, end int, ok bool) {
if !strings.HasPrefix(text, "---") {
return "", 0, false
}
i := 3
for i < len(text) && (text[i] == ' ' || text[i] == '\t') {
i++
}
if i >= len(text) || text[i] != '\n' {
return "", 0, false
}
start := i + 1
for at := start - 1; at >= 0 && at < len(text); {
nl := strings.IndexByte(text[at+1:], '\n')
if nl < 0 {
return "", 0, false
}
at = at + 1 + nl // index of the newline that must precede the fence
rest := text[at+1:]
if !strings.HasPrefix(rest, "---") {
continue
}
j := 3
for j < len(rest) && (rest[j] == ' ' || rest[j] == '\t') {
j++
}
// `(?=\n|$)` — end of the document, or the end of this line. `$` has no
// multiline flag on the frontend either, so it means end of document.
if j < len(rest) && rest[j] != '\n' {
continue
}
return text[start:at], at + 1 + j, true
}
return "", 0, false
}
// HasFrontmatter reports whether this text opens with a frontmatter block at
// all. It does not say whether that block parses.
func HasFrontmatter(raw string) bool {
_, _, ok := fence(Normalize(raw))
return ok
}
// Document is a definition split into its two halves.
type Document struct {
// Data is the frontmatter as plain data. Always a mapping: a frontmatter
// block that parses to a sequence is discarded, exactly as the frontend
// discards it, because every reader downstream indexes it by key.
Data map[string]any
// Body is everything after the closing fence, trimmed. A document with no
// frontmatter is all body.
Body string
// Fenced records whether a frontmatter block was found, which Data alone
// cannot express — an empty fence pair and a missing one both give an
// empty mapping.
Fenced bool
}
// ParseFrontmatter splits a definition and reads its frontmatter.
//
// Returns a *Error when the YAML subset refuses a line. A document with no
// recognisable fence is NOT an error: it is a document with no frontmatter,
// and what happens to it is the validator's decision — the same division the
// frontend makes.
func ParseFrontmatter(raw string) (Document, error) {
text := Normalize(raw)
yaml, end, ok := fence(text)
if !ok {
return Document{Data: map[string]any{}, Body: text, Fenced: false}, nil
}
value, err := ParseYAML(yaml)
if err != nil {
return Document{}, err
}
data, _ := value.(map[string]any)
if data == nil {
data = map[string]any{}
}
return Document{Data: data, Body: jsTrim(text[end:]), Fenced: true}, nil
}
// sectionSource is the text under a `## Heading`, up to the next one.
//
// new RegExp(`##\\s+${escaped}\\s*\\n([\\s\\S]*?)(?=\\n##\\s|$)`, 'i')
//
// Case-insensitive, and deliberately not anchored to the start of a line —
// that is what the frontend does. The heading is compared literally rather
// than compiled into a pattern, which is the same protection the frontend gets
// by escaping it: a heading containing regular-expression punctuation must
// match the words it is built from.
//
// Returns ok=false for a section that is not there, which is a different thing
// from a section that is there and empty.
func sectionSource(body, heading string) (string, bool) {
lower := strings.ToLower(body)
want := strings.ToLower(heading)
for at := 0; ; {
h := strings.Index(lower[at:], "##")
if h < 0 {
return "", false
}
h += at
at = h + 2
// `##` then `\s+` then the heading.
i := h + 2
gap := i
for i < len(body) {
r, size := decodeRune(body[i:])
if !jsIsSpace(r) {
break
}
i += size
}
if i == gap {
continue // `\s+` needs at least one
}
if !strings.HasPrefix(lower[i:], want) {
continue
}
i += len(want)
// `\s*\n`: a whitespace run that ends in a newline.
j, nl := i, -1
for j < len(body) {
r, size := decodeRune(body[j:])
if !jsIsSpace(r) {
break
}
if r == '\n' {
nl = j
break
}
j += size
}
if nl < 0 {
continue
}
start := nl + 1
// `(?=\n##\s|$)`, lazily: the first following line that opens a new
// `##` heading. `###` does not, because the character after `##` must
// be whitespace.
for k := start; ; {
n := strings.Index(body[k:], "\n##")
if n < 0 {
return body[start:], true
}
n += k
after := n + 3
if after < len(body) {
r, _ := decodeRune(body[after:])
if jsIsSpace(r) {
return body[start:n], true
}
}
k = n + 1
}
}
}
// decodeRune is utf8.DecodeRuneInString, kept local so the section reader has
// one obvious way to step through the body.
func decodeRune(s string) (rune, int) {
for i, r := range s {
_ = i
return r, len(string(r))
}
return 0, 0
}
// SectionText is the prose under a `## Heading`, with its bullets and blank
// lines flattened to one line.
//
// Used for a workforce level's description, which is a sentence rather than a
// list.
func SectionText(body, heading string) string {
source, ok := sectionSource(body, heading)
if !ok {
return ""
}
parts := []string{}
for _, l := range strings.Split(source, "\n") {
l = jsTrim(stripListMarker(l))
if l == "" {
continue
}
parts = append(parts, l)
}
return jsTrim(strings.Join(parts, " "))
}
// stripListMarker removes a leading `-`, `*` or `1.` / `1)` bullet.
//
// /^\s*(?:[-*]|\d+[.)])\s+/
func stripListMarker(l string) string {
i := 0
for i < len(l) {
r, size := decodeRune(l[i:])
if !jsIsSpace(r) {
break
}
i += size
}
marker := i
switch {
case i < len(l) && (l[i] == '-' || l[i] == '*'):
i++
default:
digits := i
for i < len(l) && l[i] >= '0' && l[i] <= '9' {
i++
}
if i == digits || i >= len(l) || (l[i] != '.' && l[i] != ')') {
return l
}
i++
}
// `\s+` after the marker is required; without it there is no list item.
space := i
for i < len(l) {
r, size := decodeRune(l[i:])
if !jsIsSpace(r) {
break
}
i += size
}
if i == space {
return l
}
_ = marker
return l[i:]
}

View File

@@ -0,0 +1,210 @@
package definition
import (
"math"
"strconv"
"strings"
)
// JavaScript value semantics, reproduced exactly.
//
// The frontend parser is JavaScript, and the compatibility contract is with
// THAT parser, not with an idealised YAML. Three of its behaviours are load
// bearing and none of them are Go's defaults:
//
// - `\s` and `String.prototype.trim` cover a different set of code points
// than `unicode.IsSpace`. JS treats U+FEFF as whitespace and U+0085 as
// not; Go is the other way round. A definition is trimmed on the way
// through the parser at least four times, so the difference is reachable.
// - Truthiness decides whether `id:` is used or derived, whether a name
// falls back to `Untitled skill`, and what `prompt:` becomes. `0`, `false`
// and `""` are falsy; `"0"` and `[]` are not.
// - `String(x)` and `Number(x)` have defined results for every type, and the
// validator interpolates them into messages an author reads. `[1,2]`
// stringifies to `1,2`, an object to `[object Object]`.
//
// Reimplementing these is not gold-plating: each one is exercised by a case in
// the conformance suite because each one is reachable from a definition an
// author could write.
// jsIsSpace reports whether r is whitespace to JavaScript — the union of
// WhiteSpace and LineTerminator in the specification.
//
// Deliberately NOT unicode.IsSpace: that set includes U+0085 (NEL), which JS
// does not, and excludes U+FEFF, which JS does.
func jsIsSpace(r rune) bool {
switch r {
case '\t', '\n', '\v', '\f', '\r', ' ',
0x00A0, 0x1680, 0x2028, 0x2029, 0x202F, 0x205F, 0x3000, 0xFEFF:
return true
}
return r >= 0x2000 && r <= 0x200A
}
// jsTrim is String.prototype.trim.
func jsTrim(s string) string { return strings.TrimFunc(s, jsIsSpace) }
// jsTrimStart is String.prototype.trimStart.
func jsTrimStart(s string) string { return strings.TrimLeftFunc(s, jsIsSpace) }
// jsTruthy is the `!!x` of a parsed YAML value.
//
// The parser produces only nil, bool, float64, string, []any and
// map[string]any, so those are the only cases that can arise. An empty array
// and an empty object are both truthy in JavaScript, which is why they are not
// listed alongside the empty string.
func jsTruthy(v any) bool {
switch x := v.(type) {
case nil:
return false
case bool:
return x
case float64:
return x != 0 && !math.IsNaN(x)
case string:
return x != ""
default:
return true
}
}
// jsNumberToString is JavaScript's Number → String conversion for the values
// this parser can produce.
//
// `-0` prints as `0`, integers print without a decimal point, and everything
// else takes the shortest representation that round-trips — which is what
// strconv's 'g' with precision -1 gives, in the range a definition can reach.
func jsNumberToString(f float64) string {
switch {
case math.IsNaN(f):
return "NaN"
case math.IsInf(f, 1):
return "Infinity"
case math.IsInf(f, -1):
return "-Infinity"
case f == 0:
return "0" // collapses -0
}
if f == math.Trunc(f) && math.Abs(f) < 1e21 {
return strconv.FormatFloat(f, 'f', -1, 64)
}
return strconv.FormatFloat(f, 'g', -1, 64)
}
// jsString is the `String(x)` of a parsed YAML value.
//
// Arrays join on `,` with nil rendering as the empty string, which is
// Array.prototype.toString; a mapping renders as `[object Object]`. Both are
// reachable: `trigger:` may be written as a list, and the message an author
// reads interpolates the result.
func jsString(v any) string {
switch x := v.(type) {
case nil:
return "null"
case bool:
if x {
return "true"
}
return "false"
case float64:
return jsNumberToString(x)
case string:
return x
case []any:
parts := make([]string, len(x))
for i, item := range x {
if item == nil {
parts[i] = ""
continue
}
parts[i] = jsString(item)
}
return strings.Join(parts, ",")
case map[string]any:
return "[object Object]"
}
return ""
}
// jsTrimmed is the frontend's `trimmed()` helper: String(value ?? empty).trim().
//
// The nullish coalescing matters: a nil renders as the empty string here,
// where a bare String(null) would render as the four characters `null`.
func jsTrimmed(v any) string {
if v == nil {
return ""
}
return jsTrim(jsString(v))
}
// jsNumber is the `Number(x)` of a parsed YAML value, NaN where JavaScript
// gives NaN.
//
// Only reached from `version:`, where the result is checked with
// Number.isInteger. The string cases below are the ones a YAML scalar can
// still be carrying at that point: a quoted `"3"` stays a string, and so does
// anything the numeric patterns in toScalar declined.
func jsNumber(v any) float64 {
switch x := v.(type) {
case nil:
return 0
case bool:
if x {
return 1
}
return 0
case float64:
return x
case string:
return jsNumberFromString(x)
case []any:
// Number([]) is 0 and Number([3]) is 3, via the same String()
// conversion; anything longer stringifies with a comma and fails.
if len(x) == 0 {
return 0
}
if len(x) == 1 {
return jsNumberFromString(jsString(x[0]))
}
}
return math.NaN()
}
func jsNumberFromString(s string) float64 {
s = jsTrim(s)
if s == "" {
return 0
}
switch s {
case "Infinity", "+Infinity":
return math.Inf(1)
case "-Infinity":
return math.Inf(-1)
}
// The radix prefixes JavaScript accepts in a numeric string literal. Signs
// are not permitted with them, which ParseUint enforces by rejecting the
// leading character.
if len(s) > 2 && s[0] == '0' {
var base int
switch s[1] {
case 'x', 'X':
base = 16
case 'o', 'O':
base = 8
case 'b', 'B':
base = 2
}
if base != 0 {
n, err := strconv.ParseUint(s[2:], base, 64)
if err != nil {
return math.NaN()
}
return float64(n)
}
}
f, err := strconv.ParseFloat(s, 64)
if err != nil {
return math.NaN()
}
return f
}

View File

@@ -0,0 +1,349 @@
package definition
import (
"fmt"
"strings"
)
// One Markdown definition → one skill.
//
// A port of parseSkill and validateSkillSource in src/lib/skills/registry.js,
// restricted to the definition contract — see the package documentation in
// definition.go for exactly where that boundary is and why the `ui:` and
// `owliver:` blocks are on the other side of it.
// Level is one rung of a workforce ladder, read from the body's own headings.
type Level struct {
Level string `json:"level"`
Label string `json:"label"`
Summary string `json:"summary"`
}
// Skill is a definition as the backend reads it.
//
// The five fields migration 000005 projects into columns — ID, Name,
// Description, Status, Pages — are the compatibility contract; the rest is
// carried because it is free once the frontmatter is parsed and because the
// conformance suite compares it.
type Skill struct {
ID string `json:"id"`
Name string `json:"name"`
Description string `json:"description"`
Status string `json:"status"`
// Pages as the AUTHOR WROTE THEM, not canonicalised.
//
// This is not an oversight and must not be "fixed": parseSkill keeps the
// declared strings, so a skill written against `Talent Pool` is registered
// under `Talent Pool` and resolved through normalizeKey at every use.
// Agents are the other way round — see Agent.Pages. Canonicalising here
// would make the backend's projection disagree with the editor's.
Pages []string `json:"pages"`
Kind string `json:"kind"`
Category string `json:"category"`
Actions []string `json:"actions"`
Triggers []string `json:"triggers"`
DeclaredTriggers bool `json:"declaredTriggers"`
Prompt *string `json:"prompt"`
SkillID *string `json:"skillId"`
Levels []Level `json:"levels"`
// Body is the Markdown after the frontmatter, trimmed. The definition
// itself is NEVER rewritten — see Definition.Markdown.
Body string `json:"-"`
// Deferred names the frontmatter blocks whose semantics this package does
// not check and the frontend does. Empty for every definition the backend
// can fully validate on its own. See package documentation.
Deferred []string `json:"deferred,omitempty"`
}
// AuthoredPath is the origin an authored definition has when the caller names
// none. It is a value rather than an absence for one reason: it is the
// frontend's own default parameter.
//
// parseSkill(raw, { path = 'custom', custom = false } = {})
// parseAgent(raw, { path = 'custom', custom = false } = {})
//
// validateSkillSource and validateAgentSource both call their parser with no
// path, so every definition the EDITOR checks derives its last-resort id from
// the literal string `custom`. That is the same call the backend is making — a
// definition submitted to the API is authored, not shipped — so the backend
// must derive the same id.
//
// The difference is reachable and it is not cosmetic. A definition with no
// `id:`, no `name:` and a valid `pages:` list gets the id `custom` on the
// frontend, passes the id-format check and is ACCEPTED. Deriving no id here
// would refuse it with "The frontmatter needs an `id`." — a definition that
// validates in the editor and fails on save, which is the exact failure mode
// this package exists to prevent. Fixture case: id-omitted-unnamed.
const AuthoredPath = "custom"
// Options carries what the caller knows that the definition does not.
type Options struct {
// Path is the definition's origin, used only as the last fallback for an
// id. Leave it empty for anything authored rather than shipped — which is
// what the backend always has — and it becomes AuthoredPath, exactly as the
// frontend's default parameter does.
Path string
}
// path is the origin an id is derived from, with the frontend's default
// applied.
func (o Options) path() string {
if o.Path == "" {
return AuthoredPath
}
return o.Path
}
// ParseSkill reads a skill definition.
//
// Returns a *Error when the frontmatter cannot be read. A definition with no
// frontmatter at all is not an error here: it parses to a skill carrying the
// derived id and no pages, and ValidateSkill is what refuses it — the same
// division of labour the frontend has.
func ParseSkill(raw string, opts Options) (*Skill, error) {
doc, err := ParseFrontmatter(raw)
if err != nil {
return nil, err
}
data := doc.Data
declaredPages, _ := data["pages"].([]any)
// `id:` always wins. An explicit id is the address other definitions and
// stored preferences refer to, and deriving over the top of one would
// silently rename a skill. Slugging the name is what an author means by
// leaving it out; the filename is right only for a file, which is why it
// is last.
id := jsTrim(jsString(data["id"]))
if !jsTruthy(data["id"]) {
id = slugify(data["name"])
if id == "" {
id = fileStem(opts.path())
}
}
levels := sectionLevels(doc.Body)
// Two things wear the same format. A definition that names a ladder is a
// workforce skill; nothing else distinguishes them, so an author declares
// one by writing one rather than by setting a flag.
kind := jsTrim(jsString(data["kind"]))
if !jsTruthy(data["kind"]) {
kind = "assistant"
if len(levels) > 0 {
kind = "workforce"
}
}
skill := &Skill{
ID: id,
Kind: kind,
Levels: levels,
Body: doc.Body,
Name: "Untitled skill",
Pages: stringsOf(declaredPages),
}
if jsTruthy(data["name"]) {
skill.Name = jsString(data["name"])
}
if jsTruthy(data["description"]) {
skill.Description = jsString(data["description"])
}
if s, ok := data["category"].(string); ok {
skill.Category = jsTrim(s)
}
// The whole of a skill's lifecycle, and deliberately a coercion rather
// than a check: the frontend reads anything that is not `inactive` as
// `active`, so `status: bogus` registers as active rather than being
// refused. Reproduced, not corrected — see the divergence note in
// docs/phase-4d-parser-contract.md.
skill.Status = "active"
if s, ok := data["status"].(string); ok && s == "inactive" {
skill.Status = "inactive"
}
if actions, ok := data["actions"].([]any); ok {
skill.Actions = stringsOf(actions)
} else {
skill.Actions = []string{}
}
// A skill with no declared triggers answers to its own name, so a
// definition that omits the field is still reachable by asking for it.
// Explicit triggers replace the fallback rather than adding to it.
triggers, hasTriggers := data["triggers"].([]any)
skill.DeclaredTriggers = hasTriggers && len(triggers) > 0
skill.Triggers = []string{}
if skill.DeclaredTriggers {
for _, t := range triggers {
skill.Triggers = append(skill.Triggers, strings.ToLower(jsString(t)))
}
} else if jsTruthy(data["name"]) {
skill.Triggers = append(skill.Triggers, strings.ToLower(jsString(data["name"])))
}
if jsTruthy(data["prompt"]) {
p := jsString(data["prompt"])
skill.Prompt = &p
}
// The capability in the skill graph a workforce definition governs:
// `skill:`, or the id with a `-training` suffix dropped and dashes swapped
// for underscores.
if kind == "workforce" {
base := id
if jsTruthy(data["skill"]) {
base = jsString(data["skill"])
} else {
base = strings.TrimSuffix(base, "-training")
}
s := strings.ReplaceAll(base, "-", "_")
skill.SkillID = &s
}
skill.Deferred = deferredBlocks(data)
return skill, nil
}
// sectionLevels reads the ladder a workforce definition defines, in order, from
// the body's own headings. A rung with no prose is not a rung.
func sectionLevels(body string) []Level {
out := []Level{}
for _, heading := range levelHeadings {
summary := SectionText(body, heading)
if summary == "" {
continue
}
out = append(out, Level{
Level: strings.ToLower(heading),
Label: heading,
Summary: summary,
})
}
return out
}
// deferredBlocks names the frontmatter this package does not semantically
// check. See the package documentation for why they are deferred rather than
// validated or rejected.
func deferredBlocks(data map[string]any) []string {
out := []string{}
for _, key := range []string{"ui", "owliver"} {
if _, present := data[key]; present {
out = append(out, key)
}
}
if len(out) == 0 {
return nil
}
return out
}
// stringsOf renders a parsed sequence as the strings the frontend would read
// out of it. Non-string entries are stringified rather than dropped, because
// that is what every consumer of `pages` and `actions` does with them.
func stringsOf(list []any) []string {
out := make([]string, 0, len(list))
for _, v := range list {
out = append(out, jsString(v))
}
return out
}
func fileStem(path string) string {
if path == "" {
return ""
}
if i := strings.LastIndexByte(path, '/'); i >= 0 {
path = path[i+1:]
}
return strings.TrimSuffix(path, ".md")
}
// ValidateSkill decides whether a skill definition may be stored.
//
// Returns nil when it may. The order is the order an author would fix things
// in, and every message below is the frontend's message character for
// character — an author who sees one in the editor and a different one from
// the API is being told about two different problems.
//
// Two rules are the backend's own and are marked as such: the size bound and
// the deferred-block rule. Both are explained in
// docs/phase-4d-parser-contract.md.
func ValidateSkill(raw string) error {
if jsTrim(raw) == "" {
return &Rejection{Message: "Paste or upload a Markdown definition."}
}
// The backend's own rule, from migration 000005's markdown_size CHECK. The
// editor does not enforce it, so a definition over the bound is one the
// frontend accepts and the DATABASE refuses; refusing it here turns a
// constraint violation into a message. See the contract document.
if n := len([]rune(raw)); n > MaxMarkdownLength {
return &Rejection{
BackendOnly: true,
Message: fmt.Sprintf(
"That definition is %d characters. The limit is %d.", n, MaxMarkdownLength),
}
}
skill, err := ParseSkill(raw, Options{})
if err != nil {
// The subset reports the line it failed on, which is far more useful
// than "could not be parsed".
return &Rejection{Message: jsTrim("That definition could not be parsed. " + err.Error())}
}
if skill.ID == "" {
return &Rejection{Message: "The frontmatter needs an `id`."}
}
if !isDefinitionID(skill.ID) {
return &Rejection{Message: "`id` must be lower-case letters, numbers and dashes."}
}
// Faithful to the frontend, where `name` has already fallen back to
// `Untitled skill` and this check can therefore never fire. Kept so the
// two validators have the same shape and the same order.
if skill.Name == "" {
return &Rejection{Message: "The frontmatter needs a `name`."}
}
// The backend's own rule, and it must be asked BEFORE the generic one
// below. A `ui:` block declares the pages it draws on, and parseSkill falls
// back to those pages when `pages:` is absent — a fallback this package
// cannot compute, because it does not read the `ui:` vocabulary. Rather
// than report an empty page list it never really established, say what is
// actually missing. No shipped definition relies on the fallback: all
// nineteen that carry a `ui:` or `owliver:` block also declare `pages:`.
if len(skill.Pages) == 0 && len(skill.Deferred) > 0 {
return &Rejection{
BackendOnly: true,
Message: "A definition with a `ui:` block needs an explicit `pages:` list.",
}
}
if len(skill.Pages) == 0 {
return &Rejection{Message: "The frontmatter needs at least one `pages` entry."}
}
unknown := []string{}
for _, p := range skill.Pages {
if !SurfaceExists(p) {
unknown = append(unknown, p)
}
}
if len(unknown) > 0 {
plural := ""
if len(unknown) > 1 {
plural = "s"
}
return &Rejection{Message: fmt.Sprintf(
"Unsupported page%s: %s. Supported pages: %s.",
plural, strings.Join(unknown, ", "), strings.Join(SupportedPages, ", "))}
}
return nil
}

File diff suppressed because one or more lines are too long

View File

@@ -0,0 +1,184 @@
package definition
import "strings"
// The closed vocabulary a definition is allowed to name.
//
// Every table here is a transcription of a frontend table, and the conformance
// suite asserts each one against the vocabulary the JavaScript actually
// exports (testdata/oracle.json, `vocabulary`) — so a page added to
// surfaces.js or an icon added to vocabulary.js fails a test here rather than
// silently making the two ends disagree about what is valid.
//
// Nothing is looked up dynamically and nothing is constructed: a definition
// names a key and this file answers whether that key exists.
// pageSurfaces mirrors SKILL_SURFACES in src/lib/skills/surfaces.js — id first,
// then the alternative spellings an author may use for it.
var pageSurfaces = []struct {
ID string
Aliases []string
}{
{ID: "control-center"},
{ID: "positions"},
{ID: "create-position", Aliases: []string{"new-position"}},
{ID: "candidates"},
{ID: "hired-history", Aliases: []string{"hired"}},
{ID: "talent-pool"},
{ID: "krow-forge", Aliases: []string{"university", "forge"}},
{ID: "analytics"},
{ID: "activity"},
{ID: "workspace-agent-configure"},
{ID: "settings"},
{ID: "workspace"},
{ID: "workspace-agents"},
{ID: "workspace-skills"},
{ID: "workspace-skill-configure"},
{ID: "skill-development"},
{ID: "profile"},
{ID: "candidates-analysis"},
}
// surfaceByKey resolves every spelling — canonical or alias — to its canonical
// id.
var surfaceByKey = func() map[string]string {
m := map[string]string{}
for _, s := range pageSurfaces {
m[s.ID] = s.ID
for _, a := range s.Aliases {
m[a] = s.ID
}
}
return m
}()
// SupportedPages is every canonical page id, in declaration order. The order is
// the order the frontend lists them in when it refuses an unsupported page, and
// that message is compared byte for byte.
var SupportedPages = func() []string {
out := make([]string, len(pageSurfaces))
for i, s := range pageSurfaces {
out[i] = s.ID
}
return out
}()
// normalizeKey is surfaces.js's own: trimmed, lower-cased, with spaces and
// underscores read as dashes. `Talent Pool` and `talent_pool` both reach
// `talent-pool`; the canonical keys never widen, only what an author may type
// to reach them.
func normalizeKey(page any) string {
s := strings.ToLower(jsTrim(jsString(page)))
if page == nil {
s = ""
}
return strings.Map(func(r rune) rune {
if r == ' ' || r == '_' || jsIsSpace(r) {
return '-'
}
return r
}, s)
}
// SurfaceExists reports whether a declared page name refers to a real surface.
func SurfaceExists(page any) bool {
_, ok := surfaceByKey[normalizeKey(page)]
return ok
}
// CanonicalPage is the canonical id a declared page name refers to, or "".
func CanonicalPage(page any) string { return surfaceByKey[normalizeKey(page)] }
/* ── Agent vocabulary — src/lib/agents/vocabulary.js ─────────────────────── */
var (
AgentStatuses = []string{"draft", "published", "archived"}
ReasoningModes = []string{"fast", "balanced", "deep"}
KnowledgeKinds = []string{"note", "link", "skill-reference"}
AgentAccess = []string{"all", "specific"}
PermissionRole = []string{"manager", "editor", "viewer"}
AgentIcons = []string{
"owliver", "sparkles", "briefcase", "users", "user-check",
"layers", "graduation-cap", "bar-chart", "activity", "shield",
}
)
const (
DefaultAgentStatus = "draft"
DefaultReasoning = "balanced"
DefaultAgentIcon = "owliver"
DefaultKnowledgeKind = "note"
DefaultAgentAccess = "all"
DefaultPermission = "viewer"
)
func contains(list []string, want string) bool {
for _, v := range list {
if v == want {
return true
}
}
return false
}
/* ── Skill vocabulary ────────────────────────────────────────────────────── */
// SkillStatuses is the whole of a skill's lifecycle. There is no version and no
// publish step: migration 000005 states the same two values.
var SkillStatuses = []string{"active", "inactive"}
// levelHeadings are the rungs a workforce skill may define, and the reason a
// definition is read as workforce rather than assistant. Order is the ladder's.
var levelHeadings = []string{"Beginner", "Intermediate", "Advanced", "Expert"}
// slugify is uiConfig.js's, used to derive an id from a name.
//
// String(value || '').toLowerCase().trim()
// .replace(/[^a-z0-9]+/g, '-').replace(/^-|-$/g, '')
//
// The lower-casing happens BEFORE the class replacement, so an upper-case
// letter becomes itself rather than a dash.
func slugify(value any) string {
if !jsTruthy(value) {
return ""
}
s := jsTrim(strings.ToLower(jsString(value)))
var b strings.Builder
dash := false
for _, r := range s {
if (r >= 'a' && r <= 'z') || (r >= '0' && r <= '9') {
b.WriteRune(r)
dash = false
continue
}
if !dash {
b.WriteByte('-')
dash = true
}
}
out := b.String()
out = strings.TrimPrefix(out, "-")
out = strings.TrimSuffix(out, "-")
return out
}
// isDefinitionID is the id format the frontend validator enforces and the
// definition_id CHECK in migration 000005 restates.
func isDefinitionID(id string) bool {
if id == "" {
return false
}
for i, r := range id {
lower := r >= 'a' && r <= 'z'
digit := r >= '0' && r <= '9'
if lower || digit {
continue
}
if r == '-' && i > 0 {
continue
}
return false
}
return true
}

View File

@@ -0,0 +1,362 @@
package definition
import (
"fmt"
"regexp"
"strconv"
"strings"
)
// The YAML subset a Krow definition is allowed to use.
//
// A line-for-line port of the frontend's src/lib/skills/yaml.js. That file is
// the specification and this is the second implementation of it, so every
// decision below is made because the JavaScript makes it — including the ones
// a YAML library would make differently.
//
// Supported, and nothing else:
//
// - block maps and block sequences, nested to any depth
// - scalars: strings, integers, floats, booleans, null
// - quoted strings, for values containing `:` or `#`
// - `- key: value`, a mapping whose first key sits on the dash
// - `#` comments, and blank lines
//
// Anchors, aliases, merge keys, multi-document files, flow mappings, flow
// sequences, block scalars and tags are NOT supported, and are not silently
// half-read: an unparseable line is an error carrying the line number, so a
// definition either means what it says or is refused with somewhere to look.
//
// No dependency. A general YAML library would accept a much larger language
// than the frontend does, and every construct it accepted and the frontend did
// not would be a definition the backend stores and the editor cannot read —
// exactly the failure this package exists to prevent. The subset is small
// enough to port exactly, so it is ported exactly.
//
// Nothing here evaluates anything. There is no reflection, no template, no
// code path from a definition to execution of any kind: a definition is
// configuration, and this is the boundary that keeps it configuration.
// Error is a definition that could not be read, carrying the line it failed on.
//
// Line is 1-based and counts lines of FRONTMATTER, not of the file — which is
// what the JavaScript reports, because parseYaml is handed the fenced block
// rather than the document. Reproduced rather than improved: an author who
// sees one message in the editor and another from the API is being told about
// two different problems.
type Error struct {
Line int
Message string
}
func (e *Error) Error() string { return e.Message }
// errIndent and errPair are the two failures the subset has, worded exactly as
// the frontend words them.
func errIndent(line int) *Error {
return &Error{Line: line, Message: fmt.Sprintf("Unexpected indentation on line %d", line)}
}
func errPair(line int, content string) *Error {
return &Error{Line: line, Message: fmt.Sprintf("Line %d is not `key: value`: %s", line, content)}
}
// keyPair matches `key: value` and is the only shape a mapping entry may take.
// The key alphabet is the frontend's: letters, digits, underscore, dot, dash —
// which is why `my key: value` is a refusal rather than a key with a space.
// jsSpaceClass is the JavaScript `\s` character class. Go's own `\s` is
// ASCII-only, and the difference is reachable: a non-breaking space after the
// colon is whitespace to the editor's parser and would be part of the value
// here.
const jsSpaceClass = `[\t\n\v\f\r \x{00A0}\x{1680}\x{2000}-\x{200A}\x{2028}\x{2029}\x{202F}\x{205F}\x{3000}\x{FEFF}]`
var keyPair = regexp.MustCompile(`^([A-Za-z0-9_.-]+):` + jsSpaceClass + `*([\s\S]*)$`)
var (
reInt = regexp.MustCompile(`^-?[0-9]+$`)
reFloat = regexp.MustCompile(`^-?[0-9]*\.[0-9]+$`)
)
// line is one significant line, reduced to what the parser needs to decide.
type line struct {
number int // 1-based, within the frontmatter block
indent int // leading whitespace, tabs counted as two
content string
}
// readLines drops blank lines and whole-line comments, and measures what is
// left.
//
// Indentation is counted in code points with a tab worth two spaces, which is
// what the JavaScript does and is why a tab-indented sequence sits at the same
// depth as a two-space one.
func readLines(source string) []line {
out := []line{}
for i, text := range strings.Split(source, "\n") {
trimmed := jsTrim(text)
if trimmed == "" {
continue
}
// A whole-line comment: `^\s*#`.
if strings.HasPrefix(jsTrimStart(text), "#") {
continue
}
indent := 0
for _, r := range text {
if !jsIsSpace(r) {
break
}
if r == '\t' {
indent += 2
continue
}
indent++
}
out = append(out, line{number: i + 1, indent: indent, content: trimmed})
}
return out
}
// quoted matches a scalar wrapped in one kind of quote, end to end.
//
// Greedy and anchored at both ends, as in the frontend: `"a" "b"` is therefore
// ONE quoted string whose content is `a" "b`, not two. That is a strange
// reading, and it is the reading the editor gives, so it is the reading here.
func quotedScalar(value string) (string, bool) {
if len(value) < 2 {
return "", false
}
q := value[0]
if q != '\'' && q != '"' {
return "", false
}
if value[len(value)-1] != q {
return "", false
}
inner := value[1 : len(value)-1]
// The only escape the subset has: a doubled quote is one quote.
return strings.ReplaceAll(inner, string([]byte{q, q}), string(q)), true
}
// stripTrailingComment removes an unquoted trailing `#` comment.
//
// `\s+#.*$` applied once, leftmost — so `ops # a # b` loses everything from
// the first spaced hash, and `ops#1` loses nothing, because a hash inside a
// word is part of the word.
func stripTrailingComment(value string) string {
runes := []rune(value)
for i := 0; i < len(runes); i++ {
if !jsIsSpace(runes[i]) {
continue
}
j := i
for j < len(runes) && jsIsSpace(runes[j]) {
j++
}
if j < len(runes) && runes[j] == '#' {
return string(runes[:i])
}
i = j - 1
}
return value
}
// toScalar reads one written value: `true`, `false`, `null`, a number, a
// quoted string, or the string as written.
func toScalar(raw string) any {
value := jsTrim(raw)
switch value {
case "", "~", "null":
return nil
case "true":
return true
case "false":
return false
}
// Quoted: taken literally, which is how a value containing `:` or `#` is
// written. No escape processing beyond the doubled quote.
if inner, ok := quotedScalar(value); ok {
return inner
}
if reInt.MatchString(value) || reFloat.MatchString(value) {
if f, err := strconv.ParseFloat(value, 64); err == nil {
return f
}
}
return jsTrim(stripTrailingComment(value))
}
// cursor is shared down the recursion so a child consumes the lines it owns.
type cursor struct{ i int }
// parseBlock reads one block at indent or deeper.
//
// Map or sequence depending on what the first line at this level is, which is
// how YAML itself decides.
func parseBlock(lines []line, c *cursor, indent int) (any, *Error) {
if c.i >= len(lines) {
return nil, nil
}
first := lines[c.i]
if strings.HasPrefix(first.content, "- ") || first.content == "-" {
return parseSequence(lines, c, indent)
}
return parseMapping(lines, c, indent)
}
// dashPrefix is the `-` and the whitespace after it, as `^-\s*` consumes them.
func dashPrefix(content string) int {
if !strings.HasPrefix(content, "-") {
return 0
}
n := 1
for _, r := range content[1:] {
if !jsIsSpace(r) {
break
}
n += len(string(r))
}
return n
}
func parseSequence(lines []line, c *cursor, indent int) (any, *Error) {
out := []any{}
for c.i < len(lines) {
cur := lines[c.i]
if cur.indent < indent {
break
}
if cur.indent > indent {
return nil, errIndent(cur.number)
}
if !strings.HasPrefix(cur.content, "-") {
break
}
cut := dashPrefix(cur.content)
rest := cur.content[cut:]
c.i++
if rest == "" {
// `-` alone: the item is the indented block beneath it.
if c.i < len(lines) && lines[c.i].indent > indent {
item, err := parseBlock(lines, c, lines[c.i].indent)
if err != nil {
return nil, err
}
out = append(out, item)
continue
}
out = append(out, nil)
continue
}
// `- key: value` opens a mapping whose first key sits on the dash. The
// remaining keys are indented to where that key started.
if m := keyPair.FindStringSubmatch(rest); m != nil {
keyIndent := indent + cut
item := map[string]any{}
key, value := m[1], m[2]
if value == "" && c.i < len(lines) && lines[c.i].indent > indent {
block, err := parseBlock(lines, c, lines[c.i].indent)
if err != nil {
return nil, err
}
item[key] = block
} else {
item[key] = toScalar(value)
}
for c.i < len(lines) && lines[c.i].indent == keyIndent &&
!strings.HasPrefix(lines[c.i].content, "- ") {
more, err := parseMapping(lines, c, keyIndent)
if err != nil {
return nil, err
}
if m, ok := more.(map[string]any); ok {
for k, v := range m {
item[k] = v
}
}
}
out = append(out, item)
continue
}
out = append(out, toScalar(rest))
}
return out, nil
}
func parseMapping(lines []line, c *cursor, indent int) (any, *Error) {
out := map[string]any{}
for c.i < len(lines) {
cur := lines[c.i]
if cur.indent < indent {
break
}
if cur.indent > indent {
return nil, errIndent(cur.number)
}
if strings.HasPrefix(cur.content, "- ") {
break
}
m := keyPair.FindStringSubmatch(cur.content)
if m == nil {
return nil, errPair(cur.number, cur.content)
}
key, value := m[1], m[2]
c.i++
if value != "" {
out[key] = toScalar(value)
continue
}
// An empty value means the value is the block below — or nothing.
if c.i < len(lines) && lines[c.i].indent > indent {
block, err := parseBlock(lines, c, lines[c.i].indent)
if err != nil {
return nil, err
}
out[key] = block
continue
}
out[key] = nil
}
return out, nil
}
// ParseYAML reads one document of the subset as plain data.
//
// Returns map[string]any, []any, or the empty map for an empty document.
// Anything it cannot read is an error rather than a guess, so a malformed
// definition is reported to its author instead of being registered in a shape
// nobody intended.
func ParseYAML(source string) (any, error) {
lines := readLines(source)
if len(lines) == 0 {
return map[string]any{}, nil
}
c := &cursor{}
value, err := parseBlock(lines, c, lines[0].indent)
if err != nil {
return nil, err
}
if c.i < len(lines) {
return nil, errIndent(lines[c.i].number)
}
return value, nil
}