Files
backend_fiesta/scratch/cataloguedims/main.go
Suriyakumarvijayanayagam 72907dae74 Scan-to-order: label from the customer's camera to "buy it here"
POST /v1/mob/scan/lookup   label + customer → catalogue match, sizes, and
                           every registered store that sells it with live
                           stock, in-stock first / nearest first, one
                           recommended
POST /v1/mob/scan/confirm  chosen store + size + qty → re-read the ledger;
                           ok, or the next-nearest store with enough of the
                           same product
GET  /v1/mob/scan/stores   registered stores nearest first

Recognition is pgvector cosine search over every brand_* table (each
with its own index, merged) plus a word match that settles near-ties
and works alone when no model is configured. The embedder is chosen by
EMBEDDING_PROVIDER (OpenAI-compatible or Gemini) and must be the model
that indexed the catalogue: verified 2026-09-15 as all-MiniLM-L6-v2 over
search_query, served by the cluster's Ollama as `all-minilm`; the first
search refuses a width mismatch by name.

Customer, stores and catalogue are read concurrently under a 5 s cap; a
slow model degrades to a text answer. Vectors and ranked hits are cached
in Redis and in-process; live stock never is. Availability uses the same
rules as the customer catalogue (approve, publishedat, ledger balance,
outlet price else retail). No stock reservation: confirm re-reads.

scratch/cataloguedims reports the catalogue's embedding width and fill.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-09-15 17:04:34 +05:30

90 lines
2.9 KiB
Go

// Reports how the catalogue's `embedding` columns are shaped — width, how
// many rows are filled, and a sample norm — so the embedding model Fiesta
// calls can be matched to the one that indexed the catalogue. Metadata and
// counts only, on a read-only transaction; it never writes.
//
// go run ./scratch/cataloguedims # reads CATALOGUE_DB_* from .env.production
package main
import (
"flag"
"fmt"
"log"
"net/url"
"os"
"github.com/joho/godotenv"
"gorm.io/driver/postgres"
"gorm.io/gorm"
)
func main() {
sample := flag.String("sample", "", "print one row's texts and stored vector from this table, to check which model produced it")
flag.Parse()
_ = godotenv.Load(".env.production")
dsn := url.URL{
Scheme: "postgres",
User: url.UserPassword(os.Getenv("CATALOGUE_DB_USER"), os.Getenv("CATALOGUE_DB_PASSWORD")),
Host: os.Getenv("CATALOGUE_DB_HOST") + ":" + os.Getenv("CATALOGUE_DB_PORT"),
Path: "/" + os.Getenv("CATALOGUE_DB_NAME"),
}
q := dsn.Query()
q.Set("sslmode", "disable")
q.Set("default_transaction_read_only", "on")
dsn.RawQuery = q.Encode()
db, err := gorm.Open(postgres.Open(dsn.String()), &gorm.Config{})
if err != nil {
log.Fatal(err)
}
if *sample != "" {
var row struct {
ProductName string
Title string
SearchQuery string
Embedding string
}
db.Raw(fmt.Sprintf(`SELECT product_name, COALESCE(title, '') AS title, COALESCE(search_query, '') AS search_query,
embedding::text AS embedding FROM %s WHERE embedding IS NOT NULL ORDER BY id LIMIT 1`, *sample)).Scan(&row)
fmt.Printf("product_name: %s\ntitle: %s\nsearch_query: %s\nembedding: %s\n", row.ProductName, row.Title, row.SearchQuery, row.Embedding)
return
}
var cols []struct {
Relname string
Typname string
Atttypmod int
}
if err := db.Raw(`
SELECT c.relname, t.typname, a.atttypmod
FROM pg_attribute a
JOIN pg_class c ON c.oid = a.attrelid
JOIN pg_type t ON t.oid = a.atttypid
WHERE a.attname = 'embedding' AND c.relname LIKE 'brand\_%'
ORDER BY c.relname`).Scan(&cols).Error; err != nil {
log.Fatal(err)
}
if len(cols) == 0 {
fmt.Println("no brand_* table has an embedding column")
return
}
fmt.Printf("%-28s %-8s %5s %5s %5s\n", "table", "type", "dims", "rows", "embd")
for _, c := range cols {
var total, filled int64
db.Raw(fmt.Sprintf(`SELECT COUNT(1) FROM %s`, c.Relname)).Scan(&total)
db.Raw(fmt.Sprintf(`SELECT COUNT(1) FROM %s WHERE embedding IS NOT NULL`, c.Relname)).Scan(&filled)
fmt.Printf("%-28s %-8s %5d %5d %5d\n", c.Relname, c.Typname, c.Atttypmod, total, filled)
}
// nomic/bge emit unit vectors; a norm far from 1 means another pipeline.
for _, c := range cols {
var norm float64
db.Raw(fmt.Sprintf(`SELECT vector_norm(embedding) FROM %s WHERE embedding IS NOT NULL LIMIT 1`, c.Relname)).Scan(&norm)
if norm > 0 {
fmt.Printf("sample vector norm (%s): %.4f\n", c.Relname, norm)
break
}
}
}