`"britannia"` is a substring of all 258 Britannia product names, and textScore returned 0.95 for any product whose name contained the label. So every one of them tied, the tie broke alphabetically, and the customer was shown one arbitrary biscuit with "confidence": 0.95 and a price. Lens hands back a bare wordmark often — it is usually the biggest thing printed on a packet — so this was the common case, not an edge one. Found via the example request in the mobile team's own proposal. Scoring now asks both questions. A hit carries `score` (ranks) and `text` (how specifically the label names THIS product: the harmonic mean of how much of the label the product explains and how much of the product's name the label explains, pack sizes dropped from both sides). A brand name scores its products ~0.33 equally instead of 0.95 arbitrarily. The "vector and text agree" bonus is now proportional to the text score, so a weak match can no longer inflate a whole brand. isAmbiguous reads that: the leader is a guess if anything is level with it (margin) or if the label names no one product (specificity), and then the response carries `ambiguous: true` with `candidates` — distinct products, not pack sizes, at most ten, each marked with whether one of the customer's stores has it in stock, available ones first. `match` is nil and `stores` empty on that path: no price for a product nobody chose. Erring towards asking is deliberate — a tap versus the wrong biscuit. To act on a pick, /lookup now accepts `brand` + `catalogueid` instead of a label and skips recognition entirely (also serves deep links and re-order). New: ScanRepository.CatalogueRef, resolving via the brand tables discovered from information_schema, never a name built from the request. Also: scratch/cataloguedims now reports every vector column, not just `embedding` — which is how we learned the catalogue also carries img_vector(1024), filled on 1885 of 2124 rows. SCAN_TO_ORDER.md records why that column stays unread for now and what would change it, alongside why the app is not asked to compute vectors on the phone. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
91 lines
2.9 KiB
Go
91 lines
2.9 KiB
Go
// Reports how the catalogue's `embedding` columns are shaped — width, how
|
|
// many rows are filled, and a sample norm — so the embedding model Fiesta
|
|
// calls can be matched to the one that indexed the catalogue. Metadata and
|
|
// counts only, on a read-only transaction; it never writes.
|
|
//
|
|
// go run ./scratch/cataloguedims # reads CATALOGUE_DB_* from .env.production
|
|
package main
|
|
|
|
import (
|
|
"flag"
|
|
"fmt"
|
|
"log"
|
|
"net/url"
|
|
"os"
|
|
|
|
"github.com/joho/godotenv"
|
|
"gorm.io/driver/postgres"
|
|
"gorm.io/gorm"
|
|
)
|
|
|
|
func main() {
|
|
sample := flag.String("sample", "", "print one row's texts and stored vector from this table, to check which model produced it")
|
|
flag.Parse()
|
|
_ = godotenv.Load(".env.production")
|
|
dsn := url.URL{
|
|
Scheme: "postgres",
|
|
User: url.UserPassword(os.Getenv("CATALOGUE_DB_USER"), os.Getenv("CATALOGUE_DB_PASSWORD")),
|
|
Host: os.Getenv("CATALOGUE_DB_HOST") + ":" + os.Getenv("CATALOGUE_DB_PORT"),
|
|
Path: "/" + os.Getenv("CATALOGUE_DB_NAME"),
|
|
}
|
|
q := dsn.Query()
|
|
q.Set("sslmode", "disable")
|
|
q.Set("default_transaction_read_only", "on")
|
|
dsn.RawQuery = q.Encode()
|
|
|
|
db, err := gorm.Open(postgres.Open(dsn.String()), &gorm.Config{})
|
|
if err != nil {
|
|
log.Fatal(err)
|
|
}
|
|
|
|
if *sample != "" {
|
|
var row struct {
|
|
ProductName string
|
|
Title string
|
|
SearchQuery string
|
|
Embedding string
|
|
}
|
|
db.Raw(fmt.Sprintf(`SELECT product_name, COALESCE(title, '') AS title, COALESCE(search_query, '') AS search_query,
|
|
embedding::text AS embedding FROM %s WHERE embedding IS NOT NULL ORDER BY id LIMIT 1`, *sample)).Scan(&row)
|
|
fmt.Printf("product_name: %s\ntitle: %s\nsearch_query: %s\nembedding: %s\n", row.ProductName, row.Title, row.SearchQuery, row.Embedding)
|
|
return
|
|
}
|
|
|
|
var cols []struct {
|
|
Relname string
|
|
Attname string
|
|
Typname string
|
|
Atttypmod int
|
|
}
|
|
if err := db.Raw(`
|
|
SELECT c.relname, a.attname, t.typname, a.atttypmod
|
|
FROM pg_attribute a
|
|
JOIN pg_class c ON c.oid = a.attrelid
|
|
JOIN pg_type t ON t.oid = a.atttypid
|
|
WHERE t.typname = 'vector' AND a.attnum > 0 AND c.relname LIKE 'brand\_%'
|
|
ORDER BY c.relname, a.attname`).Scan(&cols).Error; err != nil {
|
|
log.Fatal(err)
|
|
}
|
|
if len(cols) == 0 {
|
|
fmt.Println("no brand_* table has a vector column")
|
|
return
|
|
}
|
|
fmt.Printf("%-24s %-16s %-8s %5s %5s %5s\n", "table", "column", "type", "dims", "rows", "filled")
|
|
for _, c := range cols {
|
|
var total, filled int64
|
|
db.Raw(fmt.Sprintf(`SELECT COUNT(1) FROM %s`, c.Relname)).Scan(&total)
|
|
db.Raw(fmt.Sprintf(`SELECT COUNT(1) FROM %s WHERE %s IS NOT NULL`, c.Relname, c.Attname)).Scan(&filled)
|
|
fmt.Printf("%-24s %-16s %-8s %5d %5d %5d\n", c.Relname, c.Attname, c.Typname, c.Atttypmod, total, filled)
|
|
}
|
|
|
|
// nomic/bge emit unit vectors; a norm far from 1 means another pipeline.
|
|
for _, c := range cols {
|
|
var norm float64
|
|
db.Raw(fmt.Sprintf(`SELECT vector_norm(embedding) FROM %s WHERE embedding IS NOT NULL LIMIT 1`, c.Relname)).Scan(&norm)
|
|
if norm > 0 {
|
|
fmt.Printf("sample vector norm (%s): %.4f\n", c.Relname, norm)
|
|
break
|
|
}
|
|
}
|
|
}
|