agent build
This commit is contained in:
@@ -24,6 +24,15 @@ import (
|
||||
"github.com/krow/krow-backend/go-api/internal/httpserver"
|
||||
)
|
||||
|
||||
// version is stamped at link time:
|
||||
//
|
||||
// go build -ldflags="-X main.version=$(git rev-parse --short HEAD)"
|
||||
//
|
||||
// The Dockerfile passes its VERSION build arg through to this. "dev" is what an
|
||||
// unstamped local build reports, which is honest — it says the binary was not
|
||||
// built by the release path rather than inventing a number.
|
||||
var version = "dev"
|
||||
|
||||
func main() {
|
||||
if err := run(); err != nil {
|
||||
slog.Error("fatal", "error", err)
|
||||
@@ -38,7 +47,8 @@ func run() error {
|
||||
}
|
||||
|
||||
log := newLogger(cfg.Log.Level)
|
||||
log.Info("starting krow-api", "env", cfg.AppEnv, "database", cfg.DB.Redacted())
|
||||
log.Info("starting krow-api", "version", version,
|
||||
"env", cfg.AppEnv, "database", cfg.DB.Redacted())
|
||||
|
||||
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||
defer stop()
|
||||
@@ -50,7 +60,7 @@ func run() error {
|
||||
defer database.Close()
|
||||
log.Info("database connected", "schema", cfg.DB.Schema)
|
||||
|
||||
server, err := httpserver.New(cfg, database, log)
|
||||
server, err := httpserver.New(cfg, database, log, httpserver.WithBuildVersion(version))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
386
go-api/cmd/importagents/main.go
Normal file
386
go-api/cmd/importagents/main.go
Normal file
@@ -0,0 +1,386 @@
|
||||
// Command importagents publishes the agent specs in agents/ into a tenant.
|
||||
//
|
||||
// §7 says adding an agent is a data change: write the spec, validate it,
|
||||
// publish it. This is the publish step, and it is a command rather than a
|
||||
// migration because agents are TENANT data — a migration would either hardcode
|
||||
// one organization or run for none.
|
||||
//
|
||||
// What it does NOT do, deliberately:
|
||||
//
|
||||
// - It does not create versions. §3 says specs are immutable once published
|
||||
// and editing publishes a new version; this re-publishes in place, which is
|
||||
// right for a curated set shipped with the deployment and wrong for
|
||||
// authored ones. Version immutability is Phase 3's, and this command is the
|
||||
// thing that makes Phase 3 worth doing rather than a substitute for it.
|
||||
// - It does not validate tool names against the registry. §3 wants an unknown
|
||||
// tool to fail at publish; today the runtime records and drops one. The
|
||||
// check is cheap to add and belongs here — see the note in run().
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/jackc/pgx/v5"
|
||||
"github.com/jackc/pgx/v5/pgxpool"
|
||||
|
||||
"github.com/krow/krow-backend/go-api/internal/config"
|
||||
"github.com/krow/krow-backend/go-api/internal/db"
|
||||
"github.com/krow/krow-backend/go-api/internal/definition"
|
||||
)
|
||||
|
||||
func main() {
|
||||
var (
|
||||
dir = flag.String("dir", "./agents", "directory holding the agent specs")
|
||||
skillDir = flag.String("skills", "./skills", "directory holding the skill definitions")
|
||||
org = flag.String("org", "", "organization slug to publish into (required)")
|
||||
dryRun = flag.Bool("dry-run", false, "parse and report, write nothing")
|
||||
timeout = flag.Duration("timeout", 30*time.Second, "overall timeout")
|
||||
)
|
||||
flag.Parse()
|
||||
|
||||
if err := run(*dir, *skillDir, *org, *dryRun, *timeout); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "import-agents: %v\n", err)
|
||||
os.Exit(1)
|
||||
}
|
||||
}
|
||||
|
||||
func run(dir, skillDir, orgSlug string, dryRun bool, timeout time.Duration) error {
|
||||
if strings.TrimSpace(orgSlug) == "" {
|
||||
return errors.New("an organization is required: --org=<slug>")
|
||||
}
|
||||
|
||||
specs, err := loadSpecs(dir)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if len(specs) == 0 {
|
||||
return fmt.Errorf("no agent specs found in %s", dir)
|
||||
}
|
||||
|
||||
// Skills come with the agents, in the same transaction.
|
||||
//
|
||||
// Not optional and not a separate command: an agent whose spec names a
|
||||
// skill will not LOAD without it — the runtime refuses with
|
||||
// ErrDependencyMissing rather than running a degraded agent, which is the
|
||||
// right call and means a half-import produces agents that 422 instead of
|
||||
// answering. They belong to one operation because they fail as one.
|
||||
skills, err := loadSkills(skillDir)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
// Parsed before anything is opened, so a malformed spec is a message rather
|
||||
// than a half-finished import. Every spec, not the first failure: an
|
||||
// operator fixing five typos should see five, not one per run.
|
||||
var problems []string
|
||||
for _, s := range specs {
|
||||
if len(s.parsed.Errors) > 0 {
|
||||
problems = append(problems, fmt.Sprintf(" %s: %s",
|
||||
s.name, strings.Join(s.parsed.Errors, "; ")))
|
||||
}
|
||||
if s.parsed.Status != "published" {
|
||||
problems = append(problems, fmt.Sprintf(
|
||||
" %s: status is %q; only a published spec can be imported",
|
||||
s.name, s.parsed.Status))
|
||||
}
|
||||
}
|
||||
if len(problems) > 0 {
|
||||
return fmt.Errorf("%d spec(s) will not import:\n%s",
|
||||
len(problems), strings.Join(problems, "\n"))
|
||||
}
|
||||
|
||||
for _, s := range specs {
|
||||
fmt.Printf(" agent %-24s v%d %d tool(s) %d source(s) %d skill(s)\n",
|
||||
s.parsed.ID, s.parsed.Version, len(s.parsed.Tools), len(s.parsed.Sources),
|
||||
len(s.parsed.Skills))
|
||||
}
|
||||
fmt.Printf(" skills %d definition(s)\n", len(skills))
|
||||
|
||||
// Every skill an agent names must be present, checked before anything is
|
||||
// written. The runtime refuses to load an agent with a missing dependency,
|
||||
// so importing one without its skills produces an agent that exists and
|
||||
// cannot run — a failure that surfaces per request instead of here.
|
||||
if missing := missingSkills(specs, skills); len(missing) > 0 {
|
||||
return fmt.Errorf("%d skill(s) named by an agent are not in %s: %s",
|
||||
len(missing), skillDir, strings.Join(missing, ", "))
|
||||
}
|
||||
|
||||
if dryRun {
|
||||
fmt.Printf("\n%d agent(s) and %d skill(s) parsed; nothing written (--dry-run)\n",
|
||||
len(specs), len(skills))
|
||||
return nil
|
||||
}
|
||||
|
||||
cfg, err := config.Load()
|
||||
if err != nil {
|
||||
return fmt.Errorf("load configuration: %w", err)
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), timeout)
|
||||
defer cancel()
|
||||
|
||||
database, err := db.Open(ctx, cfg.DB)
|
||||
if err != nil {
|
||||
return fmt.Errorf("connect: %w", err)
|
||||
}
|
||||
defer database.Close()
|
||||
pool := database.Pool
|
||||
|
||||
orgID, err := resolveOrg(ctx, pool, orgSlug)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
author, err := resolveAuthor(ctx, pool, orgID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
// One transaction for the whole set. A partially-imported registry is a
|
||||
// deployment where some agents answer and others 404, which is harder to
|
||||
// diagnose than none of them working.
|
||||
tx, err := pool.Begin(ctx)
|
||||
if err != nil {
|
||||
return fmt.Errorf("begin: %w", err)
|
||||
}
|
||||
defer tx.Rollback(ctx) //nolint:errcheck // rolled back unless committed below
|
||||
|
||||
// Skills first. An agent row that lands before its dependencies exist is
|
||||
// briefly unloadable, and inside one transaction that is invisible — but
|
||||
// ordering them correctly costs nothing and means a future non-transactional
|
||||
// path is not silently broken.
|
||||
skillsWritten := 0
|
||||
for _, sk := range skills {
|
||||
if err := upsertSkill(ctx, tx, orgID, author, sk); err != nil {
|
||||
return fmt.Errorf("%s: %w", sk.name, err)
|
||||
}
|
||||
skillsWritten++
|
||||
}
|
||||
|
||||
inserted, updated := 0, 0
|
||||
for _, s := range specs {
|
||||
wasNew, err := upsert(ctx, tx, orgID, author, s)
|
||||
if err != nil {
|
||||
return fmt.Errorf("%s: %w", s.name, err)
|
||||
}
|
||||
if wasNew {
|
||||
inserted++
|
||||
} else {
|
||||
updated++
|
||||
}
|
||||
}
|
||||
if err := tx.Commit(ctx); err != nil {
|
||||
return fmt.Errorf("commit: %w", err)
|
||||
}
|
||||
|
||||
fmt.Printf("\n%d agent(s) published, %d updated, %d skill(s) written, into %s\n",
|
||||
inserted, updated, skillsWritten, orgSlug)
|
||||
return nil
|
||||
}
|
||||
|
||||
/* ── Reading the directory ──────────────────────────────────────────────── */
|
||||
|
||||
type spec struct {
|
||||
name string
|
||||
raw string
|
||||
parsed *definition.Agent
|
||||
}
|
||||
|
||||
func loadSpecs(dir string) ([]spec, error) {
|
||||
entries, err := os.ReadDir(dir)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read %s: %w", dir, err)
|
||||
}
|
||||
|
||||
var out []spec
|
||||
for _, e := range entries {
|
||||
name := e.Name()
|
||||
// README.md is documentation, not a spec. Skipped by name rather than
|
||||
// by trying to parse it and ignoring the failure — a parse error should
|
||||
// always mean something is wrong.
|
||||
if e.IsDir() || !strings.HasSuffix(name, ".md") || name == "README.md" {
|
||||
continue
|
||||
}
|
||||
raw, err := os.ReadFile(filepath.Join(dir, name))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read %s: %w", name, err)
|
||||
}
|
||||
parsed, err := definition.ParseAgent(string(raw), definition.Options{})
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", name, err)
|
||||
}
|
||||
out = append(out, spec{name: name, raw: string(raw), parsed: parsed})
|
||||
}
|
||||
|
||||
// Sorted so a run's output is stable and two runs are diffable.
|
||||
sort.Slice(out, func(a, b int) bool { return out[a].name < out[b].name })
|
||||
return out, nil
|
||||
}
|
||||
|
||||
/* ── Resolving the tenant ───────────────────────────────────────────────── */
|
||||
|
||||
func resolveOrg(ctx context.Context, pool *pgxpool.Pool, slug string) (string, error) {
|
||||
var id string
|
||||
err := pool.QueryRow(ctx,
|
||||
`SELECT id::text FROM organizations WHERE slug = $1`, slug).Scan(&id)
|
||||
if errors.Is(err, pgx.ErrNoRows) {
|
||||
return "", fmt.Errorf("no organization with slug %q", slug)
|
||||
}
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("resolve organization: %w", err)
|
||||
}
|
||||
return id, nil
|
||||
}
|
||||
|
||||
// resolveAuthor picks the user a shipped spec is attributed to.
|
||||
//
|
||||
// created_by is NOT NULL-able in spirit if not in schema, and attributing a
|
||||
// curated definition to whichever admin happens to sort first is honest: these
|
||||
// specs were shipped with the deployment, not authored by anyone in the tenant.
|
||||
// The alternative — a synthetic system user — is a row that then needs its own
|
||||
// permissions story.
|
||||
func resolveAuthor(ctx context.Context, pool *pgxpool.Pool, orgID string) (string, error) {
|
||||
var id string
|
||||
err := pool.QueryRow(ctx, `
|
||||
SELECT id::text FROM users
|
||||
WHERE org_id = $1::uuid AND role = 'admin' AND status = 'active'
|
||||
ORDER BY created_date ASC LIMIT 1`, orgID).Scan(&id)
|
||||
if errors.Is(err, pgx.ErrNoRows) {
|
||||
return "", errors.New("this organization has no active admin to attribute the specs to")
|
||||
}
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("resolve author: %w", err)
|
||||
}
|
||||
return id, nil
|
||||
}
|
||||
|
||||
/* ── Writing ────────────────────────────────────────────────────────────── */
|
||||
|
||||
// upsert publishes one spec, reporting whether it was new.
|
||||
//
|
||||
// `organization` visibility, always. A curated spec belongs to the tenant, not
|
||||
// to the admin whose id happens to be on it — publishing these as `private`
|
||||
// would make them invisible to everybody except that one person.
|
||||
func upsert(ctx context.Context, tx pgx.Tx, orgID, author string, s spec) (bool, error) {
|
||||
var existed bool
|
||||
err := tx.QueryRow(ctx, `
|
||||
INSERT INTO agent_definitions
|
||||
(definition_id, org_id, visibility, created_by, markdown,
|
||||
status, version, name, description, pages)
|
||||
VALUES ($1::text, $2::uuid, 'organization', $3::uuid, $4::text,
|
||||
$5::text, $6::integer, $7::text, $8::text, $9::text[])
|
||||
-- The uniqueness rule here is a PARTIAL index — agent_definitions is
|
||||
-- keyed on (owner_user_id, definition_id) for personal specs and on
|
||||
-- (org_id, definition_id) for organization ones — so the conflict target
|
||||
-- has to carry the same predicate. Without the WHERE, Postgres cannot
|
||||
-- match a partial index and refuses the statement outright, which is the
|
||||
-- friendly failure: silently matching the wrong index would let a
|
||||
-- curated spec collide with somebody's personal one.
|
||||
ON CONFLICT (org_id, definition_id) WHERE visibility = 'organization' DO UPDATE
|
||||
SET markdown = EXCLUDED.markdown,
|
||||
status = EXCLUDED.status,
|
||||
version = EXCLUDED.version,
|
||||
name = EXCLUDED.name,
|
||||
description = EXCLUDED.description,
|
||||
pages = EXCLUDED.pages,
|
||||
updated_date = now()
|
||||
RETURNING (xmax = 0)`,
|
||||
s.parsed.ID, orgID, author, s.raw,
|
||||
s.parsed.Status, s.parsed.Version, s.parsed.Name, s.parsed.Description,
|
||||
s.parsed.Pages,
|
||||
).Scan(&existed)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
return existed, nil
|
||||
}
|
||||
|
||||
/* ── Skills ─────────────────────────────────────────────────────────────── */
|
||||
|
||||
type skillSpec struct {
|
||||
name string
|
||||
raw string
|
||||
parsed *definition.Skill
|
||||
}
|
||||
|
||||
func loadSkills(dir string) ([]skillSpec, error) {
|
||||
entries, err := os.ReadDir(dir)
|
||||
if err != nil {
|
||||
if os.IsNotExist(err) {
|
||||
// A deployment may legitimately ship agents that name no skills.
|
||||
// Only the dependency check below decides whether that is a problem.
|
||||
return nil, nil
|
||||
}
|
||||
return nil, fmt.Errorf("read %s: %w", dir, err)
|
||||
}
|
||||
|
||||
var out []skillSpec
|
||||
for _, e := range entries {
|
||||
name := e.Name()
|
||||
if e.IsDir() || !strings.HasSuffix(name, ".md") || name == "README.md" {
|
||||
continue
|
||||
}
|
||||
raw, err := os.ReadFile(filepath.Join(dir, name))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read %s: %w", name, err)
|
||||
}
|
||||
parsed, err := definition.ParseSkill(string(raw), definition.Options{})
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", name, err)
|
||||
}
|
||||
out = append(out, skillSpec{name: name, raw: string(raw), parsed: parsed})
|
||||
}
|
||||
sort.Slice(out, func(a, b int) bool { return out[a].name < out[b].name })
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// missingSkills reports skills an agent names that no file provides.
|
||||
//
|
||||
// Checked here rather than discovered at run time, because §3's rule is that an
|
||||
// unknown reference fails at PUBLISH. This is the publish step, so this is where
|
||||
// it belongs — and the failure names every missing id at once, so an operator
|
||||
// fixing five sees five.
|
||||
func missingSkills(specs []spec, skills []skillSpec) []string {
|
||||
have := make(map[string]bool, len(skills))
|
||||
for _, sk := range skills {
|
||||
have[sk.parsed.ID] = true
|
||||
}
|
||||
seen := map[string]bool{}
|
||||
var missing []string
|
||||
for _, s := range specs {
|
||||
for _, id := range s.parsed.Skills {
|
||||
if !have[id] && !seen[id] {
|
||||
seen[id] = true
|
||||
missing = append(missing, id)
|
||||
}
|
||||
}
|
||||
}
|
||||
sort.Strings(missing)
|
||||
return missing
|
||||
}
|
||||
|
||||
func upsertSkill(ctx context.Context, tx pgx.Tx, orgID, author string, sk skillSpec) error {
|
||||
_, err := tx.Exec(ctx, `
|
||||
INSERT INTO skill_definitions
|
||||
(definition_id, org_id, visibility, created_by, markdown,
|
||||
status, name, description, pages)
|
||||
VALUES ($1::text, $2::uuid, 'organization', $3::uuid, $4::text,
|
||||
$5::text, $6::text, $7::text, $8::text[])
|
||||
ON CONFLICT (org_id, definition_id) WHERE visibility = 'organization' DO UPDATE
|
||||
SET markdown = EXCLUDED.markdown,
|
||||
status = EXCLUDED.status,
|
||||
name = EXCLUDED.name,
|
||||
description = EXCLUDED.description,
|
||||
pages = EXCLUDED.pages,
|
||||
updated_date = now()`,
|
||||
sk.parsed.ID, orgID, author, sk.raw,
|
||||
sk.parsed.Status, sk.parsed.Name, sk.parsed.Description, sk.parsed.Pages)
|
||||
return err
|
||||
}
|
||||
257
go-api/cmd/ingest/main.go
Normal file
257
go-api/cmd/ingest/main.go
Normal file
@@ -0,0 +1,257 @@
|
||||
// Command ingest puts Markdown documents into a tenant's knowledge corpus.
|
||||
//
|
||||
// The knowledge layer had an Ingester and no way to reach it — everything that
|
||||
// had ever been ingested was ingested by a test. This is the missing half.
|
||||
//
|
||||
// make ingest ORG=<slug>
|
||||
//
|
||||
// Each document declares its own audience in front matter, and a document that
|
||||
// declares none is REFUSED rather than defaulted. Both directions of a default
|
||||
// are wrong and neither raises: tenant-wide over-shares something somebody
|
||||
// meant to restrict, and empty indexes it into invisibility. See §5.
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/jackc/pgx/v5"
|
||||
|
||||
"github.com/krow/krow-backend/go-api/internal/config"
|
||||
"github.com/krow/krow-backend/go-api/internal/db"
|
||||
"github.com/krow/krow-backend/go-api/internal/domain"
|
||||
"github.com/krow/krow-backend/go-api/internal/knowledge"
|
||||
"github.com/krow/krow-backend/go-api/internal/runtime"
|
||||
)
|
||||
|
||||
func main() {
|
||||
var (
|
||||
dir = flag.String("dir", "./knowledge", "directory of Markdown documents")
|
||||
org = flag.String("org", "", "organization slug to ingest into (required)")
|
||||
dryRun = flag.Bool("dry-run", false, "parse and report, write nothing")
|
||||
timeout = flag.Duration("timeout", 15*time.Minute, "overall timeout")
|
||||
)
|
||||
flag.Parse()
|
||||
|
||||
if err := run(*dir, *org, *dryRun, *timeout); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "ingest: %v\n", err)
|
||||
os.Exit(1)
|
||||
}
|
||||
}
|
||||
|
||||
// parsed is one document, read and validated before anything is opened.
|
||||
type parsed struct {
|
||||
file string
|
||||
doc knowledge.Document
|
||||
}
|
||||
|
||||
func run(dir, orgSlug string, dryRun bool, timeout time.Duration) error {
|
||||
if strings.TrimSpace(orgSlug) == "" {
|
||||
return errors.New("an organization is required: --org=<slug>")
|
||||
}
|
||||
|
||||
docs, err := readAll(dir)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if len(docs) == 0 {
|
||||
return fmt.Errorf("no documents found in %s", dir)
|
||||
}
|
||||
|
||||
for _, d := range docs {
|
||||
tags, err := knowledge.TagsFor(d.doc.Audience)
|
||||
if err != nil {
|
||||
return fmt.Errorf("%s: %w", d.file, err)
|
||||
}
|
||||
fmt.Printf(" %-28s %-14s %s\n", d.doc.ExternalID, d.doc.Source, strings.Join(tags, " "))
|
||||
}
|
||||
if dryRun {
|
||||
fmt.Printf("\n%d document(s) parsed; nothing written (--dry-run)\n", len(docs))
|
||||
return nil
|
||||
}
|
||||
|
||||
cfg, err := config.Load()
|
||||
if err != nil {
|
||||
return fmt.Errorf("load configuration: %w", err)
|
||||
}
|
||||
|
||||
embedder := runtime.NewEmbedder(*cfg)
|
||||
if embedder == nil {
|
||||
// Not fatal. Chunks are written and left unembedded for `make reembed`,
|
||||
// so a corpus is keyword-searchable immediately and dense-searchable
|
||||
// once a model exists. Said out loud because a silently keyword-only
|
||||
// corpus is a retrieval problem that surfaces months later as "the
|
||||
// agent seems worse than it was".
|
||||
fmt.Println("\nno embedding model configured — documents will be keyword-searchable only")
|
||||
fmt.Println("set EMBED_PROVIDER and run `make reembed` to finish them")
|
||||
} else {
|
||||
fmt.Printf("\nembedding with %s\n", embedder.Model())
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), timeout)
|
||||
defer cancel()
|
||||
|
||||
database, err := db.Open(ctx, cfg.DB)
|
||||
if err != nil {
|
||||
return fmt.Errorf("connect: %w", err)
|
||||
}
|
||||
defer database.Close()
|
||||
|
||||
var orgID string
|
||||
err = database.Pool.QueryRow(ctx,
|
||||
`SELECT id::text FROM organizations WHERE slug = $1`, orgSlug).Scan(&orgID)
|
||||
if errors.Is(err, pgx.ErrNoRows) {
|
||||
return fmt.Errorf("no organization with slug %q", orgSlug)
|
||||
}
|
||||
if err != nil {
|
||||
return fmt.Errorf("resolve organization: %w", err)
|
||||
}
|
||||
|
||||
ing := knowledge.NewIngester(database.Pool, embedder)
|
||||
var chunks, unchanged int
|
||||
for _, d := range docs {
|
||||
res, err := ing.Ingest(ctx, orgID, d.doc)
|
||||
if err != nil {
|
||||
return fmt.Errorf("%s: %w", d.file, err)
|
||||
}
|
||||
chunks += res.Chunks
|
||||
if res.Unchanged {
|
||||
unchanged++
|
||||
fmt.Printf(" %-28s unchanged (%d chunks)\n", d.doc.ExternalID, res.Chunks)
|
||||
continue
|
||||
}
|
||||
note := ""
|
||||
if res.EmbeddingDeferred {
|
||||
note = " [not embedded]"
|
||||
}
|
||||
fmt.Printf(" %-28s %d chunks%s\n", d.doc.ExternalID, res.Chunks, note)
|
||||
}
|
||||
|
||||
fmt.Printf("\n%d document(s), %d chunk(s), %d unchanged, into %s\n",
|
||||
len(docs), chunks, unchanged, orgSlug)
|
||||
return nil
|
||||
}
|
||||
|
||||
/* ── Reading the directory ──────────────────────────────────────────────── */
|
||||
|
||||
func readAll(dir string) ([]parsed, error) {
|
||||
entries, err := os.ReadDir(dir)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read %s: %w", dir, err)
|
||||
}
|
||||
|
||||
var out []parsed
|
||||
for _, e := range entries {
|
||||
name := e.Name()
|
||||
if e.IsDir() || !strings.HasSuffix(name, ".md") || name == "README.md" {
|
||||
continue
|
||||
}
|
||||
raw, err := os.ReadFile(filepath.Join(dir, name))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read %s: %w", name, err)
|
||||
}
|
||||
doc, err := parse(name, string(raw))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", name, err)
|
||||
}
|
||||
out = append(out, parsed{file: name, doc: *doc})
|
||||
}
|
||||
sort.Slice(out, func(a, b int) bool { return out[a].file < out[b].file })
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// parse reads a document's front matter and body.
|
||||
//
|
||||
// A small reader rather than a YAML library: the front matter here is four flat
|
||||
// keys, and the value of a real parser is handling shapes this format does not
|
||||
// have. What matters is that a malformed audience is an error rather than a
|
||||
// silent default.
|
||||
func parse(file, raw string) (*knowledge.Document, error) {
|
||||
body := strings.ReplaceAll(raw, "\r\n", "\n")
|
||||
if !strings.HasPrefix(body, "---\n") {
|
||||
return nil, errors.New("no front matter; a document must declare its source and audience")
|
||||
}
|
||||
end := strings.Index(body[4:], "\n---")
|
||||
if end < 0 {
|
||||
return nil, errors.New("front matter is not closed")
|
||||
}
|
||||
head := body[4 : 4+end]
|
||||
rest := strings.TrimLeft(body[4+end+4:], "\n")
|
||||
|
||||
fields := map[string]string{}
|
||||
for _, line := range strings.Split(head, "\n") {
|
||||
k, v, ok := strings.Cut(line, ":")
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
fields[strings.TrimSpace(k)] = strings.TrimSpace(v)
|
||||
}
|
||||
|
||||
source := fields["source"]
|
||||
if source == "" {
|
||||
return nil, errors.New("no `source`; an agent's spec names the corpora it may read")
|
||||
}
|
||||
|
||||
audience, err := parseAudience(fields["audience"])
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
// The filename is the external id, so re-ingesting the same file updates
|
||||
// rather than duplicating. Stable, obvious, and something a person can
|
||||
// point at.
|
||||
id := strings.TrimSuffix(file, ".md")
|
||||
|
||||
title := fields["title"]
|
||||
if title == "" {
|
||||
title = id
|
||||
}
|
||||
|
||||
return &knowledge.Document{
|
||||
Source: source, ExternalID: id, Title: title,
|
||||
URI: fields["uri"], Body: rest, Audience: audience,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// parseAudience turns the declared audience into the one the ingester takes.
|
||||
//
|
||||
// An empty or unrecognised value is an ERROR. That is the whole point: §5
|
||||
// refuses a document that reaches nobody, and a typo'd role silently producing
|
||||
// a tag no principal holds is the same failure wearing better clothes.
|
||||
func parseAudience(raw string) (knowledge.Audience, error) {
|
||||
raw = strings.TrimSpace(raw)
|
||||
if raw == "" {
|
||||
return knowledge.Audience{}, errors.New(
|
||||
"no `audience`; a document that declares none is unreachable, not private")
|
||||
}
|
||||
|
||||
var a knowledge.Audience
|
||||
for _, part := range strings.Split(raw, ",") {
|
||||
part = strings.TrimSpace(part)
|
||||
switch {
|
||||
case part == "tenant":
|
||||
a.Tenant = true
|
||||
case strings.HasPrefix(part, "role:"):
|
||||
name := strings.TrimPrefix(part, "role:")
|
||||
role, ok := domain.ParseRole(name)
|
||||
if !ok {
|
||||
return knowledge.Audience{}, fmt.Errorf(
|
||||
"%q is not a role; use admin, employer or talent", name)
|
||||
}
|
||||
a.Roles = append(a.Roles, role)
|
||||
case strings.HasPrefix(part, "email:"):
|
||||
a.Emails = append(a.Emails, strings.TrimPrefix(part, "email:"))
|
||||
default:
|
||||
return knowledge.Audience{}, fmt.Errorf(
|
||||
"%q is not an audience; use tenant, role:<name> or email:<address>", part)
|
||||
}
|
||||
}
|
||||
return a, nil
|
||||
}
|
||||
107
go-api/cmd/reembed/main.go
Normal file
107
go-api/cmd/reembed/main.go
Normal file
@@ -0,0 +1,107 @@
|
||||
// Command reembed gives every chunk in a tenant a vector from the current
|
||||
// embedding model.
|
||||
//
|
||||
// Run it after changing EMBED_PROVIDER or EMBED_MODEL. The reason it is a
|
||||
// command and not something that happens automatically is that it costs
|
||||
// real time and, on a hosted provider, real money — and doing that silently on
|
||||
// a config change is how a deployment surprises somebody with a bill.
|
||||
//
|
||||
// The reason it EXISTS is that the alternative is silent too, in the worse
|
||||
// direction: vectors from two models are not comparable, so after a switch the
|
||||
// old ones simply stop being searched. Retrieval keeps working, keeps citing,
|
||||
// and quietly halves its own recall. Nothing errors.
|
||||
//
|
||||
// make reembed ORG=<slug>
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"os"
|
||||
"time"
|
||||
|
||||
"github.com/jackc/pgx/v5"
|
||||
|
||||
"github.com/krow/krow-backend/go-api/internal/config"
|
||||
"github.com/krow/krow-backend/go-api/internal/db"
|
||||
"github.com/krow/krow-backend/go-api/internal/knowledge"
|
||||
"github.com/krow/krow-backend/go-api/internal/runtime"
|
||||
)
|
||||
|
||||
func main() {
|
||||
var (
|
||||
org = flag.String("org", "", "organization slug to re-embed (required)")
|
||||
batch = flag.Int("batch", 32, "chunks per request to the embedding model")
|
||||
timeout = flag.Duration("timeout", 30*time.Minute, "overall timeout")
|
||||
)
|
||||
flag.Parse()
|
||||
|
||||
if err := run(*org, *batch, *timeout); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "reembed: %v\n", err)
|
||||
os.Exit(1)
|
||||
}
|
||||
}
|
||||
|
||||
func run(orgSlug string, batch int, timeout time.Duration) error {
|
||||
if orgSlug == "" {
|
||||
return errors.New("an organization is required: --org=<slug>")
|
||||
}
|
||||
|
||||
cfg, err := config.Load()
|
||||
if err != nil {
|
||||
return fmt.Errorf("load configuration: %w", err)
|
||||
}
|
||||
|
||||
embedder := runtime.NewEmbedder(*cfg)
|
||||
if embedder == nil {
|
||||
return errors.New("no embedding model is configured; set EMBED_PROVIDER " +
|
||||
"(and its model) before re-embedding")
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), timeout)
|
||||
defer cancel()
|
||||
|
||||
database, err := db.Open(ctx, cfg.DB)
|
||||
if err != nil {
|
||||
return fmt.Errorf("connect: %w", err)
|
||||
}
|
||||
defer database.Close()
|
||||
|
||||
var orgID string
|
||||
err = database.Pool.QueryRow(ctx,
|
||||
`SELECT id::text FROM organizations WHERE slug = $1`, orgSlug).Scan(&orgID)
|
||||
if errors.Is(err, pgx.ErrNoRows) {
|
||||
return fmt.Errorf("no organization with slug %q", orgSlug)
|
||||
}
|
||||
if err != nil {
|
||||
return fmt.Errorf("resolve organization: %w", err)
|
||||
}
|
||||
|
||||
fmt.Printf("re-embedding %s with %s\n", orgSlug, embedder.Model())
|
||||
|
||||
started := time.Now()
|
||||
last := 0
|
||||
done, err := knowledge.NewIngester(database.Pool, embedder).
|
||||
Reembed(ctx, orgID, batch, func(d, total int) {
|
||||
// Reported as it goes. A corpus takes long enough that a silent
|
||||
// command is one somebody kills halfway, which is the worst place
|
||||
// to stop.
|
||||
if d-last >= batch || d == total {
|
||||
fmt.Printf(" %d/%d chunks (%s elapsed)\n", d, total,
|
||||
time.Since(started).Round(time.Second))
|
||||
last = d
|
||||
}
|
||||
})
|
||||
if err != nil {
|
||||
return fmt.Errorf("after %d chunk(s): %w", done, err)
|
||||
}
|
||||
|
||||
if done == 0 {
|
||||
fmt.Println("nothing to do — every chunk already carries this model's vectors")
|
||||
return nil
|
||||
}
|
||||
fmt.Printf("\n%d chunk(s) re-embedded in %s\n", done, time.Since(started).Round(time.Second))
|
||||
return nil
|
||||
}
|
||||
Reference in New Issue
Block a user