Files
krow_backend/go-api/internal/definition/yaml.go
2026-08-24 13:06:29 +05:30

363 lines
10 KiB
Go

package definition
import (
"fmt"
"regexp"
"strconv"
"strings"
)
// The YAML subset a Krow definition is allowed to use.
//
// A line-for-line port of the frontend's src/lib/skills/yaml.js. That file is
// the specification and this is the second implementation of it, so every
// decision below is made because the JavaScript makes it — including the ones
// a YAML library would make differently.
//
// Supported, and nothing else:
//
// - block maps and block sequences, nested to any depth
// - scalars: strings, integers, floats, booleans, null
// - quoted strings, for values containing `:` or `#`
// - `- key: value`, a mapping whose first key sits on the dash
// - `#` comments, and blank lines
//
// Anchors, aliases, merge keys, multi-document files, flow mappings, flow
// sequences, block scalars and tags are NOT supported, and are not silently
// half-read: an unparseable line is an error carrying the line number, so a
// definition either means what it says or is refused with somewhere to look.
//
// No dependency. A general YAML library would accept a much larger language
// than the frontend does, and every construct it accepted and the frontend did
// not would be a definition the backend stores and the editor cannot read —
// exactly the failure this package exists to prevent. The subset is small
// enough to port exactly, so it is ported exactly.
//
// Nothing here evaluates anything. There is no reflection, no template, no
// code path from a definition to execution of any kind: a definition is
// configuration, and this is the boundary that keeps it configuration.
// Error is a definition that could not be read, carrying the line it failed on.
//
// Line is 1-based and counts lines of FRONTMATTER, not of the file — which is
// what the JavaScript reports, because parseYaml is handed the fenced block
// rather than the document. Reproduced rather than improved: an author who
// sees one message in the editor and another from the API is being told about
// two different problems.
type Error struct {
Line int
Message string
}
func (e *Error) Error() string { return e.Message }
// errIndent and errPair are the two failures the subset has, worded exactly as
// the frontend words them.
func errIndent(line int) *Error {
return &Error{Line: line, Message: fmt.Sprintf("Unexpected indentation on line %d", line)}
}
func errPair(line int, content string) *Error {
return &Error{Line: line, Message: fmt.Sprintf("Line %d is not `key: value`: %s", line, content)}
}
// keyPair matches `key: value` and is the only shape a mapping entry may take.
// The key alphabet is the frontend's: letters, digits, underscore, dot, dash —
// which is why `my key: value` is a refusal rather than a key with a space.
// jsSpaceClass is the JavaScript `\s` character class. Go's own `\s` is
// ASCII-only, and the difference is reachable: a non-breaking space after the
// colon is whitespace to the editor's parser and would be part of the value
// here.
const jsSpaceClass = `[\t\n\v\f\r \x{00A0}\x{1680}\x{2000}-\x{200A}\x{2028}\x{2029}\x{202F}\x{205F}\x{3000}\x{FEFF}]`
var keyPair = regexp.MustCompile(`^([A-Za-z0-9_.-]+):` + jsSpaceClass + `*([\s\S]*)$`)
var (
reInt = regexp.MustCompile(`^-?[0-9]+$`)
reFloat = regexp.MustCompile(`^-?[0-9]*\.[0-9]+$`)
)
// line is one significant line, reduced to what the parser needs to decide.
type line struct {
number int // 1-based, within the frontmatter block
indent int // leading whitespace, tabs counted as two
content string
}
// readLines drops blank lines and whole-line comments, and measures what is
// left.
//
// Indentation is counted in code points with a tab worth two spaces, which is
// what the JavaScript does and is why a tab-indented sequence sits at the same
// depth as a two-space one.
func readLines(source string) []line {
out := []line{}
for i, text := range strings.Split(source, "\n") {
trimmed := jsTrim(text)
if trimmed == "" {
continue
}
// A whole-line comment: `^\s*#`.
if strings.HasPrefix(jsTrimStart(text), "#") {
continue
}
indent := 0
for _, r := range text {
if !jsIsSpace(r) {
break
}
if r == '\t' {
indent += 2
continue
}
indent++
}
out = append(out, line{number: i + 1, indent: indent, content: trimmed})
}
return out
}
// quoted matches a scalar wrapped in one kind of quote, end to end.
//
// Greedy and anchored at both ends, as in the frontend: `"a" "b"` is therefore
// ONE quoted string whose content is `a" "b`, not two. That is a strange
// reading, and it is the reading the editor gives, so it is the reading here.
func quotedScalar(value string) (string, bool) {
if len(value) < 2 {
return "", false
}
q := value[0]
if q != '\'' && q != '"' {
return "", false
}
if value[len(value)-1] != q {
return "", false
}
inner := value[1 : len(value)-1]
// The only escape the subset has: a doubled quote is one quote.
return strings.ReplaceAll(inner, string([]byte{q, q}), string(q)), true
}
// stripTrailingComment removes an unquoted trailing `#` comment.
//
// `\s+#.*$` applied once, leftmost — so `ops # a # b` loses everything from
// the first spaced hash, and `ops#1` loses nothing, because a hash inside a
// word is part of the word.
func stripTrailingComment(value string) string {
runes := []rune(value)
for i := 0; i < len(runes); i++ {
if !jsIsSpace(runes[i]) {
continue
}
j := i
for j < len(runes) && jsIsSpace(runes[j]) {
j++
}
if j < len(runes) && runes[j] == '#' {
return string(runes[:i])
}
i = j - 1
}
return value
}
// toScalar reads one written value: `true`, `false`, `null`, a number, a
// quoted string, or the string as written.
func toScalar(raw string) any {
value := jsTrim(raw)
switch value {
case "", "~", "null":
return nil
case "true":
return true
case "false":
return false
}
// Quoted: taken literally, which is how a value containing `:` or `#` is
// written. No escape processing beyond the doubled quote.
if inner, ok := quotedScalar(value); ok {
return inner
}
if reInt.MatchString(value) || reFloat.MatchString(value) {
if f, err := strconv.ParseFloat(value, 64); err == nil {
return f
}
}
return jsTrim(stripTrailingComment(value))
}
// cursor is shared down the recursion so a child consumes the lines it owns.
type cursor struct{ i int }
// parseBlock reads one block at indent or deeper.
//
// Map or sequence depending on what the first line at this level is, which is
// how YAML itself decides.
func parseBlock(lines []line, c *cursor, indent int) (any, *Error) {
if c.i >= len(lines) {
return nil, nil
}
first := lines[c.i]
if strings.HasPrefix(first.content, "- ") || first.content == "-" {
return parseSequence(lines, c, indent)
}
return parseMapping(lines, c, indent)
}
// dashPrefix is the `-` and the whitespace after it, as `^-\s*` consumes them.
func dashPrefix(content string) int {
if !strings.HasPrefix(content, "-") {
return 0
}
n := 1
for _, r := range content[1:] {
if !jsIsSpace(r) {
break
}
n += len(string(r))
}
return n
}
func parseSequence(lines []line, c *cursor, indent int) (any, *Error) {
out := []any{}
for c.i < len(lines) {
cur := lines[c.i]
if cur.indent < indent {
break
}
if cur.indent > indent {
return nil, errIndent(cur.number)
}
if !strings.HasPrefix(cur.content, "-") {
break
}
cut := dashPrefix(cur.content)
rest := cur.content[cut:]
c.i++
if rest == "" {
// `-` alone: the item is the indented block beneath it.
if c.i < len(lines) && lines[c.i].indent > indent {
item, err := parseBlock(lines, c, lines[c.i].indent)
if err != nil {
return nil, err
}
out = append(out, item)
continue
}
out = append(out, nil)
continue
}
// `- key: value` opens a mapping whose first key sits on the dash. The
// remaining keys are indented to where that key started.
if m := keyPair.FindStringSubmatch(rest); m != nil {
keyIndent := indent + cut
item := map[string]any{}
key, value := m[1], m[2]
if value == "" && c.i < len(lines) && lines[c.i].indent > indent {
block, err := parseBlock(lines, c, lines[c.i].indent)
if err != nil {
return nil, err
}
item[key] = block
} else {
item[key] = toScalar(value)
}
for c.i < len(lines) && lines[c.i].indent == keyIndent &&
!strings.HasPrefix(lines[c.i].content, "- ") {
more, err := parseMapping(lines, c, keyIndent)
if err != nil {
return nil, err
}
if m, ok := more.(map[string]any); ok {
for k, v := range m {
item[k] = v
}
}
}
out = append(out, item)
continue
}
out = append(out, toScalar(rest))
}
return out, nil
}
func parseMapping(lines []line, c *cursor, indent int) (any, *Error) {
out := map[string]any{}
for c.i < len(lines) {
cur := lines[c.i]
if cur.indent < indent {
break
}
if cur.indent > indent {
return nil, errIndent(cur.number)
}
if strings.HasPrefix(cur.content, "- ") {
break
}
m := keyPair.FindStringSubmatch(cur.content)
if m == nil {
return nil, errPair(cur.number, cur.content)
}
key, value := m[1], m[2]
c.i++
if value != "" {
out[key] = toScalar(value)
continue
}
// An empty value means the value is the block below — or nothing.
if c.i < len(lines) && lines[c.i].indent > indent {
block, err := parseBlock(lines, c, lines[c.i].indent)
if err != nil {
return nil, err
}
out[key] = block
continue
}
out[key] = nil
}
return out, nil
}
// ParseYAML reads one document of the subset as plain data.
//
// Returns map[string]any, []any, or the empty map for an empty document.
// Anything it cannot read is an error rather than a guess, so a malformed
// definition is reported to its author instead of being registered in a shape
// nobody intended.
func ParseYAML(source string) (any, error) {
lines := readLines(source)
if len(lines) == 0 {
return map[string]any{}, nil
}
c := &cursor{}
value, err := parseBlock(lines, c, lines[0].indent)
if err != nil {
return nil, err
}
if c.i < len(lines) {
return nil, errIndent(lines[c.i].number)
}
return value, nil
}