first commit

This commit is contained in:
2026-08-24 13:06:29 +05:30
commit 7d12ebef3d
86 changed files with 39996 additions and 0 deletions

268
scripts/cases.mjs Normal file
View File

@@ -0,0 +1,268 @@
/**
* Adversarial case corpus for Phase 4D parser conformance.
*
* Each case is raw bytes as an author could actually produce them. Nothing here
* is normalized on the way in: the point is what the two parsers do with the
* awkward form, so the awkward form is what is stored.
*/
const SKILL = [
'---',
'id: sample-skill',
'name: Sample Skill',
'description: A sample skill.',
'pages:',
' - candidates',
'status: active',
'actions:',
' - navigate_to_candidates',
'---',
'',
'# Sample Skill',
'',
'## Purpose',
'',
'- Read the pipeline.',
'',
'## Capabilities',
'',
'- Summarize candidates.',
'',
].join('\n');
const AGENT = [
'---',
'id: sample-agent',
'name: Sample Agent',
'description: A sample agent.',
'icon: users',
'status: published',
'version: 2',
'reasoning: balanced',
'trigger: Use on candidates.',
'pages:',
' - candidates',
'skills:',
' - candidate-search',
'starters:',
' - label: Who is waiting?',
' prompt: Who is waiting on a decision?',
'permissions:',
' owner: demo@krow.app',
' access: all',
'---',
'',
'# Sample Agent',
'',
'## Instructions',
'',
'Answer about candidates.',
'',
'## Purpose',
'',
'- Report the pipeline.',
'',
].join('\n');
/** A skill body, with the frontmatter lines replaced wholesale. */
const skillWith = (fmLines) => ['---', ...fmLines, '---', '', '# Sample Skill', '', '## Purpose', '', '- Read the pipeline.', ''].join('\n');
const agentWith = (fmLines) => ['---', ...fmLines, '---', '', '# Sample Agent', '', '## Instructions', '', 'Answer.', ''].join('\n');
const BASE_SKILL_FM = [
'id: sample-skill',
'name: Sample Skill',
'description: A sample skill.',
'pages:',
' - candidates',
];
const BASE_AGENT_FM = [
'id: sample-agent',
'name: Sample Agent',
'description: A sample agent.',
'pages:',
' - candidates',
];
/** BASE_SKILL_FM with one line appended. */
const skillPlus = (...extra) => skillWith([...BASE_SKILL_FM, ...extra]);
const agentPlus = (...extra) => agentWith([...BASE_AGENT_FM, ...extra]);
export const CASES = [
/* ── Baselines ────────────────────────────────────────────────────────── */
{ name: 'baseline-skill', kind: 'skill', raw: SKILL },
{ name: 'baseline-agent', kind: 'agent', raw: AGENT },
/* ── Byte-level shape ─────────────────────────────────────────────────── */
{ name: 'utf8-bom', kind: 'skill', raw: '' + SKILL },
{ name: 'utf8-bom-agent', kind: 'agent', raw: '' + AGENT },
{ name: 'crlf', kind: 'skill', raw: SKILL.replace(/\n/g, '\r\n') },
{ name: 'crlf-agent', kind: 'agent', raw: AGENT.replace(/\n/g, '\r\n') },
{ name: 'cr-only', kind: 'skill', raw: SKILL.replace(/\n/g, '\r') },
{ name: 'bom-crlf-blankline', kind: 'skill', raw: '\r\n' + SKILL.replace(/\n/g, '\r\n') },
{ name: 'leading-blank-line', kind: 'skill', raw: '\n' + SKILL },
{ name: 'multiple-leading-blank-lines', kind: 'skill', raw: '\n\n\n' + SKILL },
{ name: 'leading-spaces-then-blank-lines', kind: 'skill', raw: ' \n \n' + SKILL },
{ name: 'no-trailing-newline', kind: 'skill', raw: SKILL.replace(/\n+$/, '') },
{ name: 'many-trailing-newlines', kind: 'skill', raw: SKILL + '\n\n\n' },
{ name: 'trailing-spaces-on-values', kind: 'skill', raw: skillWith(BASE_SKILL_FM.map((l) => l + ' ')) },
{ name: 'trailing-ws-after-open-fence', kind: 'skill', raw: SKILL.replace(/^---/, '--- ') },
{ name: 'trailing-tab-after-close-fence', kind: 'skill', raw: SKILL.replace(/\n---\n/, '\n---\t\n') },
{ name: 'frontmatter-only-no-body', kind: 'skill', raw: '---\n' + BASE_SKILL_FM.join('\n') + '\n---' },
{ name: 'frontmatter-only-trailing-newline', kind: 'skill', raw: '---\n' + BASE_SKILL_FM.join('\n') + '\n---\n' },
/* ── Scalars ──────────────────────────────────────────────────────────── */
{ name: 'double-quoted-scalar', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: "Sample Skill"', 'description: "A sample."', 'pages:', ' - candidates']) },
{ name: 'single-quoted-scalar', kind: 'skill', raw: skillWith(["id: sample-skill", "name: 'Sample Skill'", "description: 'A sample.'", 'pages:', ' - candidates']) },
{ name: 'colon-in-quoted-string', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: "Sample: Skill"', 'description: "Note: read this."', 'pages:', ' - candidates']) },
{ name: 'colon-in-unquoted-string', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description: Note: read this.', 'pages:', ' - candidates']) },
{ name: 'hash-in-quoted-string', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: "Shift #1"', 'description: "Tag #ops"', 'pages:', ' - candidates']) },
{ name: 'hash-unquoted-trailing-comment', kind: 'skill', raw: skillPlus('category: ops # a trailing comment') },
{ name: 'hash-unquoted-midword', kind: 'skill', raw: skillPlus('category: ops#1') },
{ name: 'doubled-quote-escape', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: "She said ""go"""', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'full-line-comment', kind: 'skill', raw: skillWith(['# a comment line', ...BASE_SKILL_FM]) },
{ name: 'empty-scalar', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description:', 'pages:', ' - candidates']) },
{ name: 'tilde-scalar', kind: 'skill', raw: skillPlus('category: ~') },
{ name: 'null-scalar', kind: 'skill', raw: skillPlus('category: null') },
{ name: 'boolean-scalar', kind: 'agent', raw: agentPlus('webSearch: true') },
{ name: 'integer-scalar', kind: 'agent', raw: agentPlus('version: 3') },
{ name: 'float-scalar', kind: 'agent', raw: agentPlus('version: 1.5') },
{ name: 'negative-integer', kind: 'agent', raw: agentPlus('version: -2') },
/* ── Collections ──────────────────────────────────────────────────────── */
{ name: 'empty-array', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description: A sample.', 'pages:', ' - candidates', 'actions:']) },
{ name: 'inline-flow-array', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description: A sample.', 'pages: [candidates]']) },
{ name: 'inline-flow-map', kind: 'agent', raw: agentPlus('permissions: {owner: a, access: all}') },
{ name: 'sequence-of-mappings', kind: 'agent', raw: AGENT },
{ name: 'nested-mapping', kind: 'agent', raw: agentWith([...BASE_AGENT_FM, 'permissions:', ' owner: demo@krow.app', ' access: specific', ' people:', ' - user: a@b.c', ' role: editor']) },
{ name: 'dash-alone-nested-block', kind: 'agent', raw: agentWith([...BASE_AGENT_FM, 'starters:', ' -', ' label: Hello', ' prompt: Hello there']) },
{ name: 'tab-indented-sequence', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description: A sample.', 'pages:', '\t- candidates']) },
{ name: 'four-space-indent', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'ragged-indent', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description: A sample.', 'pages:', ' - candidates', ' - positions']) },
/* ── Multiline / unsupported YAML ─────────────────────────────────────── */
{ name: 'block-scalar-literal', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description: |', ' Line one.', ' Line two.', 'pages:', ' - candidates']) },
{ name: 'block-scalar-folded', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description: >', ' Line one.', 'pages:', ' - candidates']) },
{ name: 'anchor-and-alias', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: &n Sample Skill', 'description: *n', 'pages:', ' - candidates']) },
{ name: 'multi-document', kind: 'skill', raw: skillWith([...BASE_SKILL_FM, '---', 'id: second']) },
/* ── Keys ─────────────────────────────────────────────────────────────── */
{ name: 'duplicate-key', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: First Name', 'name: Second Name', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'duplicate-key-array', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description: A sample.', 'pages:', ' - candidates', 'pages:', ' - positions']) },
{ name: 'unsupported-frontmatter-field', kind: 'skill', raw: skillPlus('unknownField: whatever') },
{ name: 'unsupported-field-agent', kind: 'agent', raw: agentPlus('nonsense: 1') },
{ name: 'key-with-space', kind: 'skill', raw: skillWith(['id: sample-skill', 'my key: value', 'name: Sample Skill', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'uppercase-key', kind: 'skill', raw: skillPlus('Category: Ops') },
/* ── Malformed ────────────────────────────────────────────────────────── */
{ name: 'malformed-yaml-bare-line', kind: 'skill', raw: skillWith(['id: sample-skill', 'this is not a pair', 'name: Sample Skill', 'pages:', ' - candidates']) },
{ name: 'malformed-open-fence-two-dashes', kind: 'skill', raw: SKILL.replace(/^---/, '--') },
{ name: 'malformed-open-fence-four-dashes', kind: 'skill', raw: SKILL.replace(/^---/, '----') },
{ name: 'malformed-open-fence-indented', kind: 'skill', raw: ' ' + SKILL },
{ name: 'malformed-open-fence-text-after', kind: 'skill', raw: SKILL.replace(/^---/, '---yaml') },
{ name: 'malformed-close-fence-two-dashes', kind: 'skill', raw: '---\n' + BASE_SKILL_FM.join('\n') + '\n--\n\n# Body\n' },
{ name: 'malformed-close-fence-missing', kind: 'skill', raw: '---\n' + BASE_SKILL_FM.join('\n') + '\n\n# Body\n' },
{ name: 'malformed-close-fence-four-dashes', kind: 'skill', raw: '---\n' + BASE_SKILL_FM.join('\n') + '\n----\n\n# Body\n' },
{ name: 'empty-fence-pair', kind: 'skill', raw: '---\n---\n\n# Body\n' },
{ name: 'empty-fence-with-blank', kind: 'skill', raw: '---\n\n---\n\n# Body\n' },
{ name: 'missing-frontmatter', kind: 'skill', raw: '# Sample Skill\n\n## Purpose\n\n- Read the pipeline.\n' },
{ name: 'empty-markdown', kind: 'skill', raw: '' },
{ name: 'whitespace-only-markdown', kind: 'skill', raw: ' \n\n\t\n' },
{ name: 'frontmatter-is-a-sequence', kind: 'skill', raw: '---\n- one\n- two\n---\n\n# Body\n' },
/* ── Field-level validity ─────────────────────────────────────────────── */
{ name: 'invalid-definition-id-uppercase', kind: 'skill', raw: skillWith(['id: Sample_Skill', 'name: Sample Skill', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'invalid-definition-id-leading-dash', kind: 'skill', raw: skillWith(['id: -sample', 'name: Sample Skill', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'invalid-definition-id-underscore', kind: 'skill', raw: skillWith(['id: sample_skill', 'name: Sample Skill', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'missing-id-derives-from-name', kind: 'skill', raw: skillWith(['name: Sample Skill', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'missing-name', kind: 'skill', raw: skillWith(['id: sample-skill', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'missing-pages', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description: A sample.']) },
{ name: 'unknown-page', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description: A sample.', 'pages:', ' - nowhere']) },
{ name: 'page-alias-uppercase', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description: A sample.', 'pages:', ' - Candidates']) },
{ name: 'page-alias-underscore', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description: A sample.', 'pages:', ' - talent_pool']) },
{ name: 'invalid-status-skill', kind: 'skill', raw: skillPlus('status: bogus') },
{ name: 'inactive-status-skill', kind: 'skill', raw: skillPlus('status: inactive') },
{ name: 'invalid-status-agent', kind: 'agent', raw: agentPlus('status: bogus') },
{ name: 'invalid-reasoning-agent', kind: 'agent', raw: agentPlus('reasoning: turbo') },
{ name: 'invalid-icon-agent', kind: 'agent', raw: agentPlus('icon: rocket') },
{ name: 'invalid-version-agent', kind: 'agent', raw: agentPlus('version: zero') },
{ name: 'invalid-field-type-pages-scalar', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description: A sample.', 'pages: candidates']) },
{ name: 'invalid-field-type-pages-scalar-agent', kind: 'agent', raw: agentWith(['id: sample-agent', 'name: Sample Agent', 'description: A sample.', 'pages: candidates']) },
{ name: 'invalid-field-type-name-list', kind: 'skill', raw: skillWith(['id: sample-skill', 'name:', ' - a', ' - b', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'invalid-permissions-access', kind: 'agent', raw: agentWith([...BASE_AGENT_FM, 'permissions:', ' access: nobody']) },
{ name: 'self-subagent', kind: 'agent', raw: agentWith([...BASE_AGENT_FM, 'subagents:', ' - sample-agent']) },
/* ── Size ─────────────────────────────────────────────────────────────── */
{ name: 'oversized-markdown', kind: 'skill', raw: skillPlus() + '\n' + 'x'.repeat(65600) },
{ name: 'at-size-bound', kind: 'skill', raw: (() => { const b = skillPlus(); return b + 'y'.repeat(65536 - b.length); })() },
/* ── Gaps closed after the first oracle run ───────────────────────────── */
{ name: 'missing-name-agent', kind: 'agent', raw: agentWith(['id: sample-agent', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'missing-pages-agent', kind: 'agent', raw: agentWith(['id: sample-agent', 'name: Sample Agent', 'description: A sample.']) },
{ name: 'unknown-page-agent', kind: 'agent', raw: agentWith(['id: sample-agent', 'name: Sample Agent', 'description: A sample.', 'pages:', ' - nowhere']) },
{ name: 'oversized-agent', kind: 'agent', raw: agentPlus() + '\n' + 'x'.repeat(65600) },
{ name: 'visibility-field-personal', kind: 'skill', raw: skillPlus('visibility: personal') },
{ name: 'visibility-field-invalid', kind: 'skill', raw: skillPlus('visibility: nobody') },
{ name: 'blank-page-entry-agent', kind: 'agent', raw: agentWith(['id: sample-agent', 'name: Sample Agent', 'description: A sample.', 'pages:', ' - candidates', ' - ""']) },
{ name: 'duplicate-page-entry-agent', kind: 'agent', raw: agentWith(['id: sample-agent', 'name: Sample Agent', 'description: A sample.', 'pages:', ' - candidates', ' - candidates']) },
{ name: 'id-with-trailing-space', kind: 'skill', raw: skillWith(['id: sample-skill ', 'name: Sample Skill', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'id-quoted', kind: 'skill', raw: skillWith(['id: "sample-skill"', 'name: Sample Skill', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'starters-plain-strings-agent', kind: 'agent', raw: agentWith([...BASE_AGENT_FM, 'starters:', ' - Who is waiting?', ' - Where are the gaps?']) },
{ name: 'starter-missing-label-agent', kind: 'agent', raw: agentWith([...BASE_AGENT_FM, 'starters:', ' - prompt: Only a prompt']) },
{ name: 'knowledge-note-agent', kind: 'agent', raw: agentWith([...BASE_AGENT_FM, 'knowledge:', ' - label: A note', ' body: The body of the note.']) },
{ name: 'knowledge-bad-kind-agent', kind: 'agent', raw: agentWith([...BASE_AGENT_FM, 'knowledge:', ' - label: A note', ' kind: rumour', ' body: x']) },
{ name: 'knowledge-link-without-url-agent', kind: 'agent', raw: agentWith([...BASE_AGENT_FM, 'knowledge:', ' - label: A link', ' kind: link']) },
{ name: 'skills-scalar-coerced-agent', kind: 'agent', raw: agentPlus('skills: candidate-search') },
{ name: 'skills-blank-entry-agent', kind: 'agent', raw: agentWith([...BASE_AGENT_FM, 'skills:', ' - candidate-search', ' - ""']) },
{ name: 'web-search-snake-case-agent', kind: 'agent', raw: agentPlus('web_search: true') },
{ name: 'web-search-string-true-agent', kind: 'agent', raw: agentPlus('webSearch: "true"') },
{ name: 'version-quoted-integer-agent', kind: 'agent', raw: agentPlus('version: "3"') },
{ name: 'version-zero-agent', kind: 'agent', raw: agentPlus('version: 0') },
{ name: 'version-empty-agent', kind: 'agent', raw: agentPlus('version:') },
{ name: 'crlf-inside-frontmatter-only', kind: 'skill', raw: (() => { const i = SKILL.indexOf('\n---\n', 3); return SKILL.slice(0, i).replace(/\n/g, '\r\n') + SKILL.slice(i); })() },
{ name: 'body-whitespace-only-after-fence', kind: 'skill', raw: '---\n' + BASE_SKILL_FM.join('\n') + '\n---\n \n\t\n' },
{ name: 'description-whitespace-only', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description: " "', 'pages:', ' - candidates']) },
{ name: 'permissions-null-agent', kind: 'agent', raw: agentPlus('permissions:') },
{ name: 'permissions-people-bad-role-agent', kind: 'agent', raw: agentWith([...BASE_AGENT_FM, 'permissions:', ' access: specific', ' people:', ' - user: a@b.c', ' role: overlord']) },
/* ── Gaps closed after the second oracle run ──────────────────────────────
*
* Two rules had no case that could tell a correct implementation from a
* plausible wrong one, so both were being asserted by accident:
*
* The id fallback chain. Every earlier case either declares `id:` or
* declares a `name:` to slug, so the THIRD link — the default path, which
* is the literal `custom` on both sides — was never reached by a case that
* is otherwise valid. `id-omitted-unnamed` is that case: the frontend
* accepts it as the skill `custom`, and a backend deriving no id would
* refuse it on save.
*
* Text coercion. `name`, `description`, `category`, `trigger` and the
* entries of `pages` are text columns in migration 000005, and a definition
* may write any of them as a list, a number or a boolean. The frontend
* keeps the raw value on the record and stringifies it at each use; the
* backend has to choose the string at the boundary. These cases pin which
* string it chooses.
*/
{ name: 'id-omitted-unnamed', kind: 'skill', raw: skillWith(['description: A sample.', 'pages:', ' - candidates']) },
{ name: 'id-omitted-named', kind: 'skill', raw: skillWith(['name: Sample Skill', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'id-omitted-unnamed-agent', kind: 'agent', raw: agentWith(['description: A sample.', 'pages:', ' - candidates']) },
{ name: 'id-omitted-named-agent', kind: 'agent', raw: agentWith(['name: Sample Agent', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'id-omitted-name-unsluggable', kind: 'skill', raw: skillWith(['name: "!!!"', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'name-numeric', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: 42', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'name-boolean', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: true', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'description-list', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description:', ' - one', ' - two', 'pages:', ' - candidates']) },
{ name: 'description-numeric', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description: 7', 'pages:', ' - candidates']) },
{ name: 'pages-numeric-entry', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description: A sample.', 'pages:', ' - candidates', ' - 5']) },
{ name: 'pages-mapping-entry', kind: 'skill', raw: skillWith(['id: sample-skill', 'name: Sample Skill', 'description: A sample.', 'pages:', ' - page: candidates']) },
{ name: 'trigger-list-agent', kind: 'agent', raw: agentPlus('trigger:', ' - one', ' - two') },
{ name: 'name-list-agent', kind: 'agent', raw: agentWith(['id: sample-agent', 'name:', ' - a', ' - b', 'description: A sample.', 'pages:', ' - candidates']) },
{ name: 'category-numeric', kind: 'skill', raw: skillPlus('category: 3') },
/* The other backend-only bound. agent_definitions.version is a PostgreSQL
`integer`; the frontend accepts any whole number of 1 or more, so a version
above 2^31-1 is one the editor takes and the database cannot store. It had
no case at all, which left the rule asserted by nobody. */
{ name: 'version-above-int32-agent', kind: 'agent', raw: agentPlus('version: 3000000000') },
{ name: 'version-at-int32-agent', kind: 'agent', raw: agentPlus('version: 2147483647') },
];

161
scripts/gen_resources.py Executable file
View File

@@ -0,0 +1,161 @@
#!/usr/bin/env python3
"""Emit go-api/internal/domain/resources_gen.go from the live PostgreSQL schema.
Column names, types, enum values and nullability are read out of
information_schema so they can never drift from the migrations. The per-resource
metadata below (path, default sort, default limit, operations, required fields)
comes from docs/api-contract.md and is the only hand-maintained part.
Usage: make gen-resources
"""
import collections
import json
import os
import subprocess
import sys
DB = os.environ.get("DATABASE_NAME", "Krow-force")
HOST = os.environ.get("DATABASE_HOST", "127.0.0.1")
PORT = os.environ.get("DATABASE_PORT", "5432")
USER = os.environ.get("DATABASE_USER", "postgres")
def q(sql):
out = subprocess.run(
["psql", "-h", HOST, "-p", PORT, "-U", USER, "-d", DB, "-tAF", "\x1f", "-c", sql],
capture_output=True, text=True)
if out.returncode:
sys.exit(out.stderr)
return [l.split("\x1f") for l in out.stdout.strip().split("\n") if l.strip()]
# ops='' means the resource has a table and is seeded, but serves no HTTP
# endpoint (see api-contract.md §2). It still needs a descriptor so the seeder
# can write it.
META = {
'job_postings': dict(name='JobPosting', path='job-postings', sort='-created_date', limit=100, ops='List|Get|Create|Update', req=['title']),
'job_applications':dict(name='JobApplication',path='job-applications',sort='-ai_score', limit=200, ops='List|Create|Update|Delete', req=['job_posting_id','applicant_name','email']),
'ai_interviews': dict(name='AIInterview', path='ai-interviews', sort='-created_date', limit=100, ops='List|Create', req=['application_id','job_posting_id']),
'staff': dict(name='Staff', path='staff', sort='-created_date', limit=100, ops='List|Create|Update', req=['name','email','hire_date']),
'worker_profiles': dict(name='WorkerProfile', path='worker-profiles', sort='-krow_score', limit=500, ops='List|Create|Update', req=['full_name','email']),
'courses': dict(name='Course', path='courses', sort='-created_date', limit=200, ops='List|Get|Create|Update', req=['title'], orgnull=True),
'learning_paths': dict(name='LearningPath', path='learning-paths', sort='-created_date', limit=100, ops='List', req=['name'], orgnull=True),
'role_categories': dict(name='RoleCategory', path='role-categories', sort='-created_date', limit=100, ops='List|Create', req=['name']),
'certifications': dict(name='Certification', path='certifications', sort='-created_date', limit=200, ops='List|Create|Delete', req=['name']),
'user_activity': dict(name='UserActivity', path='user-activity', sort='-created_date', limit=500, ops='List|Create', req=['event_type']),
'evidence': dict(name='Evidence', path='evidence', sort='-created_date', limit=200, ops='List|Create|Update', req=['type','worker_email']),
'assignments': dict(name='Assignment', path='assignments', sort='-created_date', limit=500, ops='List|Create', req=['job_posting_id','worker_email','starts_at']),
'shift_records': dict(name='ShiftRecord', path='shift-records', sort='-created_date', limit=500, ops='List', req=[]),
'badges': dict(name='Badge', path='badges', sort='-created_date', limit=200, ops='', req=['name']),
}
ORDER = list(META)
# Columns the API never accepts from a request body, on every table that has
# them. The row's identity, its tenant and its timestamps are the server's.
READONLY = {'id', 'org_id', 'created_date', 'updated_date', 'legacy_id'}
# Columns that are server-owned on ONE table only.
#
# These name a *person*, and before Phase 3D a client could set them freely —
# which meant any authorization rule written on top of them could be defeated by
# the same request the rule was meant to constrain. A caller could reassign a
# worker profile to somebody else, or write an audit-log entry attributed to
# anyone in the organization.
#
# They are read-only here and filled in from the authenticated session instead;
# see the Derived rules in internal/domain/policy.go for which value each one
# receives and when.
SERVER_OWNED = {
'worker_profiles': {'user_id'},
'user_activity': {'user_id', 'user_email', 'user_name', 'account_type'},
'job_postings': {'created_by'},
}
def kind(dt, udt, enums):
"""Classify a column.
USER-DEFINED covers both enums and extension types: citext reports as
USER-DEFINED too. Enum-ness is decided by whether pg_enum actually has
labels for the type, not by data_type alone — classifying citext as an
enum with no permitted values rejects every email the API is sent.
"""
if udt == 'uuid': return 'KindUUID', 'uuid'
if dt == 'ARRAY': return 'KindTextArray', 'text[]'
if dt == 'jsonb': return 'KindJSON', 'jsonb'
if dt in ('integer', 'smallint'): return 'KindInt', 'int'
if dt == 'bigint': return 'KindInt', 'bigint'
if dt == 'numeric': return 'KindFloat', 'numeric'
if dt == 'boolean': return 'KindBool', 'boolean'
if dt == 'timestamp with time zone': return 'KindTimestamp', 'timestamptz'
if dt == 'date': return 'KindDate', 'date'
if dt == 'USER-DEFINED' and enums.get(udt):
return 'KindEnum', udt
if udt == 'citext': return 'KindString', 'citext'
return 'KindString', 'text'
def main():
cols = q("""SELECT table_name, column_name, data_type, udt_name, is_nullable
FROM information_schema.columns
WHERE table_schema='public' AND table_name<>'schema_migrations'
ORDER BY table_name, ordinal_position""")
enums = collections.defaultdict(list)
for t, v in q("""SELECT t.typname, e.enumlabel FROM pg_type t
JOIN pg_enum e ON e.enumtypid=t.oid
JOIN pg_namespace n ON n.oid=t.typnamespace
WHERE n.nspname='public' ORDER BY t.typname, e.enumsortorder"""):
enums[t].append(v)
by_table = collections.defaultdict(list)
for t, c, dt, udt, nul in cols:
by_table[t].append((c, dt, udt, nul == 'YES'))
o = []
o.append('// Code generated by scripts/gen_resources.py. DO NOT EDIT BY HAND.')
o.append('// Regenerate with: make gen-resources')
o.append('//')
o.append('// Column names, types, enum values and nullability are read out of')
o.append('// information_schema so they cannot drift from the migrations. The')
o.append('// per-resource metadata (path, default sort, default limit, supported')
o.append('// operations, required fields) comes from docs/api-contract.md.')
o.append('')
o.append('package domain')
o.append('')
o.append('// AllResources is every resource the API serves.')
o.append('var AllResources = []*Resource{')
for tbl in ORDER:
m = META[tbl]
if m['ops']:
ops = ' | '.join('Op' + x for x in m['ops'].split('|'))
else:
# No operations means no routes are registered for this resource.
o.append(f'\t// {m["name"]} serves NO endpoint: useBadges has zero consumers and every')
o.append('\t// badge the UI renders comes from worker_profiles.earned_badges. The')
o.append('\t// descriptor exists so the seeder can write the table. api-contract.md §2.')
ops = '0'
o.append('\t{')
o.append(f'\t\tName: {json.dumps(m["name"])}, Path: {json.dumps(m["path"])}, Table: {json.dumps(tbl)},')
o.append(f'\t\tDefaultSort: {json.dumps(m["sort"])}, DefaultLimit: {m["limit"]},')
o.append(f'\t\tOps: {ops},')
if m.get('orgnull'):
o.append('\t\tOrgNullable: true,')
o.append('\t\tColumns: []Column{')
for c, dt, udt, nullable in by_table[tbl]:
k, cast = kind(dt, udt, enums)
p = [f'Name: {json.dumps(c)}', f'Kind: {k}', f'PGType: {json.dumps(cast)}']
if not nullable: p.append('NotNull: true')
if c in READONLY or c in SERVER_OWNED.get(tbl, ()):
p.append('ReadOnly: true')
if c in m['req']: p.append('Required: true')
if k == 'KindEnum':
p.append('Enum: []string{' + ', '.join(json.dumps(v) for v in enums[udt]) + '}')
o.append('\t\t\t{' + ', '.join(p) + '},')
o.append('\t\t},')
o.append('\t},')
o.append('}')
sys.stdout.write('\n'.join(o) + '\n')
if __name__ == '__main__':
main()

187
scripts/oracle.mjs Normal file
View File

@@ -0,0 +1,187 @@
/**
* The JS parser, as an oracle.
*
* Runs the REAL frontend module graph through Vite — `import.meta.glob`, the
* `@/` alias and raw Markdown loading behave exactly as they do in the app, the
* same technique `scripts/skill-check.mjs` uses. A mock of the registry would
* reproduce none of the behaviour this file exists to capture.
*
* Emits one JSON document: for every shipped definition and every adversarial
* case, what the JS parser did with it. That document is the fixture the Go
* conformance suite asserts against, so "the Go parser agrees with the JS
* parser" is a comparison against the JS parser's actual output rather than
* against anybody's description of it.
*/
import { readFileSync, readdirSync, writeFileSync, statSync } from 'node:fs';
import { join, relative } from 'node:path';
/* The frontend checkout. Overridable so this runs anywhere the two repos are
checked out side by side, which is the layout it defaults to. */
const FRONTEND = process.env.KROW_FRONTEND
|| new URL('../../krow-demo', import.meta.url).pathname;
const { createServer } = await import(join(FRONTEND, 'node_modules/vite/dist/node/index.js'));
const { CASES } = await import(new URL('./cases.mjs', import.meta.url).href);
const server = await createServer({
root: FRONTEND, server: { middlewareMode: true }, appType: 'custom', logLevel: 'error',
});
const skillReg = await server.ssrLoadModule('/src/lib/skills/registry.js');
const agentReg = await server.ssrLoadModule('/src/lib/agents/registry.js');
/** Every .md under a directory, recursively, repo-relative. */
function walk(dir) {
const out = [];
for (const entry of readdirSync(dir)) {
const full = join(dir, entry);
if (statSync(full).isDirectory()) out.push(...walk(full));
else if (entry.endsWith('.md')) out.push(full);
}
return out.sort();
}
/** The fields the backend contract actually projects out of a definition. */
const projectSkill = (s) => ({
id: s.id,
name: s.name,
description: s.description,
status: s.status,
pages: s.pages,
kind: s.kind,
category: s.category,
actions: s.actions,
triggers: s.triggers,
declaredTriggers: s.declaredTriggers,
prompt: s.prompt ?? null,
facets: s.facets,
skillId: s.skillId ?? null,
});
const projectAgent = (a) => ({
id: a.id,
name: a.name,
description: a.description,
status: a.status,
version: a.version,
pages: a.pages,
icon: a.icon,
reasoning: a.reasoning,
trigger: a.trigger,
webSearch: a.webSearch,
skills: a.skills,
subagents: a.subagents,
starters: a.starters,
permissions: a.permissions,
errors: a.errors,
});
/** One definition, as the JS side sees it end to end. */
function observe(raw, kind) {
const out = { kind };
/* Layer 1 — the fence. */
try {
out.hasFrontmatter = skillReg.hasFrontmatter(raw);
const { data, body } = skillReg.parseFrontmatter(raw);
out.frontmatter = { ok: true, data, body };
} catch (error) {
out.hasFrontmatter = (() => { try { return skillReg.hasFrontmatter(raw); } catch { return null; } })();
out.frontmatter = { ok: false, error: String(error?.message ?? error) };
}
/* Layer 2 — the definition. */
try {
const parsed = kind === 'agent'
? agentReg.parseAgent(raw, { custom: true })
: skillReg.parseSkill(raw, { custom: true });
out.parse = { ok: true };
out.normalized = kind === 'agent' ? projectAgent(parsed) : projectSkill(parsed);
out.markdownVerbatim = parsed.markdown === raw;
} catch (error) {
out.parse = { ok: false, error: String(error?.message ?? error) };
out.normalized = null;
out.markdownVerbatim = null;
}
/* Layer 3 — the save gate. This is the accept/reject contract. */
const problem = kind === 'agent'
? agentReg.validateAgentSource(raw)
: skillReg.validateSkillSource(raw);
out.accepted = problem === null;
out.rejection = problem;
return out;
}
/* ── The shipped corpus ───────────────────────────────────────────────────── */
const corpus = [];
const groups = [
{ dir: join(FRONTEND, 'src/agents'), type: 'agent', kind: 'agent' },
{ dir: join(FRONTEND, 'src/skills'), type: 'skill', kind: 'skill' },
{ dir: join(FRONTEND, 'skill-examples'), type: 'example', kind: 'skill' },
];
for (const { dir, type, kind } of groups) {
for (const file of walk(dir)) {
const raw = readFileSync(file, 'utf8');
corpus.push({
path: relative(FRONTEND, file),
type,
rawBase64: Buffer.from(raw, 'utf8').toString('base64'),
bytes: Buffer.byteLength(raw, 'utf8'),
...observe(raw, kind),
});
}
}
/* ── The adversarial cases ────────────────────────────────────────────────── */
const cases = CASES.map((c) => ({
name: c.name,
rawBase64: Buffer.from(c.raw, 'utf8').toString('base64'),
bytes: Buffer.byteLength(c.raw, 'utf8'),
...observe(c.raw, c.kind),
}));
/* ── The closed vocabularies, read out of the JS tables themselves ────────── */
const surfaces = await server.ssrLoadModule('/src/lib/skills/surfaces.js');
const agentVocab = await server.ssrLoadModule('/src/lib/agents/vocabulary.js');
const vocabulary = {
pages: surfaces.SKILL_SURFACES.map((s) => ({ id: s.id, aliases: s.aliases || [] })),
agentStatuses: agentVocab.AGENT_STATUSES,
defaultAgentStatus: agentVocab.DEFAULT_AGENT_STATUS,
reasoning: agentVocab.SUPPORTED_REASONING,
defaultReasoning: agentVocab.DEFAULT_REASONING,
icons: agentVocab.AGENT_ICONS,
defaultIcon: agentVocab.DEFAULT_AGENT_ICON,
knowledgeKinds: agentVocab.KNOWLEDGE_KINDS,
defaultKnowledgeKind: agentVocab.DEFAULT_KNOWLEDGE_KIND,
access: agentVocab.AGENT_ACCESS,
defaultAccess: agentVocab.DEFAULT_AGENT_ACCESS,
roles: agentVocab.PERMISSION_ROLES,
defaultRole: agentVocab.DEFAULT_PERMISSION_ROLE,
};
await server.close();
const doc = {
generatedBy: 'scripts/oracle.mjs against the frontend module graph',
frontendParser: {
yaml: 'src/lib/skills/yaml.js (hand-written YAML subset, no dependency)',
frontmatter: 'src/lib/skills/registry.js — normalizeDefinition / hasFrontmatter / parseFrontmatter',
skill: 'src/lib/skills/registry.js — parseSkill / validateSkillSource',
agent: 'src/lib/agents/registry.js — parseAgent / validateAgentSource',
},
vocabulary,
corpus,
cases,
};
const target = process.argv[2];
writeFileSync(target, JSON.stringify(doc, null, 2) + '\n');
const acc = (xs) => xs.filter((x) => x.accepted).length;
console.log(`corpus ${corpus.length} files — ${acc(corpus)} accepted, ${corpus.length - acc(corpus)} rejected`);
console.log(`cases ${cases.length} — ${acc(cases)} accepted, ${cases.length - acc(cases)} rejected`);
console.log(`markdown verbatim: ${[...corpus, ...cases].every((x) => x.markdownVerbatim !== false)}`);
console.log(`written: ${target}`);

66
scripts/verify_schema.sql Normal file
View File

@@ -0,0 +1,66 @@
-- Reports what actually exists in the application schema.
-- Read-only. Uses information_schema and pg_indexes, both standard views;
-- no system catalog is written and no object is modified.
\echo '── Applied migration ─────────────────────────────────────────────────'
SELECT version, dirty FROM schema_migrations;
\echo ''
\echo '── Tables in the application schema ──────────────────────────────────'
SELECT table_name
FROM information_schema.tables
WHERE table_schema = current_schema() AND table_type = 'BASE TABLE'
AND table_name <> 'schema_migrations'
ORDER BY table_name;
\echo ''
\echo '── Enum types and their values ───────────────────────────────────────'
SELECT t.typname AS enum_type,
string_agg(e.enumlabel, ', ' ORDER BY e.enumsortorder) AS values
FROM pg_type t
JOIN pg_enum e ON e.enumtypid = t.oid
JOIN pg_namespace n ON n.oid = t.typnamespace
WHERE n.nspname = current_schema()
GROUP BY t.typname
ORDER BY t.typname;
\echo ''
\echo '── Constraint counts by type ─────────────────────────────────────────'
SELECT CASE contype
WHEN 'p' THEN 'primary key'
WHEN 'f' THEN 'foreign key'
WHEN 'u' THEN 'unique'
WHEN 'c' THEN 'check'
END AS constraint_type,
count(*) AS total
FROM pg_constraint c
JOIN pg_namespace n ON n.oid = c.connamespace
WHERE n.nspname = current_schema() AND contype IN ('p','f','u','c')
GROUP BY contype
ORDER BY contype;
\echo ''
\echo '── Foreign keys ──────────────────────────────────────────────────────'
SELECT conrelid::regclass::text AS child,
confrelid::regclass::text AS parent,
conname AS constraint_name
FROM pg_constraint c
JOIN pg_namespace n ON n.oid = c.connamespace
WHERE n.nspname = current_schema() AND contype = 'f'
ORDER BY child, parent, conname;
\echo ''
\echo '── Indexes ───────────────────────────────────────────────────────────'
SELECT tablename, indexname
FROM pg_indexes
WHERE schemaname = current_schema() AND tablename <> 'schema_migrations'
ORDER BY tablename, indexname;
\echo ''
\echo '── Objects outside the application schema created by this migration ──'
\echo '(expected: zero rows)'
SELECT n.nspname AS schema, c.relname AS object
FROM pg_class c
JOIN pg_namespace n ON n.oid = c.relnamespace
WHERE n.nspname NOT IN ('pg_catalog','information_schema','pg_toast', current_schema())
AND c.relkind IN ('r','i','S');