Files
backend_fiesta/scratch/cataloguedims/main.go
Suriyakumarvijayanayagam 01bc89ab77 Ask instead of guessing when a label fits several products
`"britannia"` is a substring of all 258 Britannia product names, and
textScore returned 0.95 for any product whose name contained the label.
So every one of them tied, the tie broke alphabetically, and the customer
was shown one arbitrary biscuit with "confidence": 0.95 and a price. Lens
hands back a bare wordmark often — it is usually the biggest thing printed
on a packet — so this was the common case, not an edge one. Found via the
example request in the mobile team's own proposal.

Scoring now asks both questions. A hit carries `score` (ranks) and `text`
(how specifically the label names THIS product: the harmonic mean of how
much of the label the product explains and how much of the product's name
the label explains, pack sizes dropped from both sides). A brand name
scores its products ~0.33 equally instead of 0.95 arbitrarily. The
"vector and text agree" bonus is now proportional to the text score, so a
weak match can no longer inflate a whole brand.

isAmbiguous reads that: the leader is a guess if anything is level with it
(margin) or if the label names no one product (specificity), and then the
response carries `ambiguous: true` with `candidates` — distinct products,
not pack sizes, at most ten, each marked with whether one of the
customer's stores has it in stock, available ones first. `match` is nil
and `stores` empty on that path: no price for a product nobody chose.
Erring towards asking is deliberate — a tap versus the wrong biscuit.

To act on a pick, /lookup now accepts `brand` + `catalogueid` instead of a
label and skips recognition entirely (also serves deep links and re-order).
New: ScanRepository.CatalogueRef, resolving via the brand tables discovered
from information_schema, never a name built from the request.

Also: scratch/cataloguedims now reports every vector column, not just
`embedding` — which is how we learned the catalogue also carries
img_vector(1024), filled on 1885 of 2124 rows. SCAN_TO_ORDER.md records
why that column stays unread for now and what would change it, alongside
why the app is not asked to compute vectors on the phone.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-09-23 10:56:02 +05:30

91 lines
2.9 KiB
Go

// Reports how the catalogue's `embedding` columns are shaped — width, how
// many rows are filled, and a sample norm — so the embedding model Fiesta
// calls can be matched to the one that indexed the catalogue. Metadata and
// counts only, on a read-only transaction; it never writes.
//
// go run ./scratch/cataloguedims # reads CATALOGUE_DB_* from .env.production
package main
import (
"flag"
"fmt"
"log"
"net/url"
"os"
"github.com/joho/godotenv"
"gorm.io/driver/postgres"
"gorm.io/gorm"
)
func main() {
sample := flag.String("sample", "", "print one row's texts and stored vector from this table, to check which model produced it")
flag.Parse()
_ = godotenv.Load(".env.production")
dsn := url.URL{
Scheme: "postgres",
User: url.UserPassword(os.Getenv("CATALOGUE_DB_USER"), os.Getenv("CATALOGUE_DB_PASSWORD")),
Host: os.Getenv("CATALOGUE_DB_HOST") + ":" + os.Getenv("CATALOGUE_DB_PORT"),
Path: "/" + os.Getenv("CATALOGUE_DB_NAME"),
}
q := dsn.Query()
q.Set("sslmode", "disable")
q.Set("default_transaction_read_only", "on")
dsn.RawQuery = q.Encode()
db, err := gorm.Open(postgres.Open(dsn.String()), &gorm.Config{})
if err != nil {
log.Fatal(err)
}
if *sample != "" {
var row struct {
ProductName string
Title string
SearchQuery string
Embedding string
}
db.Raw(fmt.Sprintf(`SELECT product_name, COALESCE(title, '') AS title, COALESCE(search_query, '') AS search_query,
embedding::text AS embedding FROM %s WHERE embedding IS NOT NULL ORDER BY id LIMIT 1`, *sample)).Scan(&row)
fmt.Printf("product_name: %s\ntitle: %s\nsearch_query: %s\nembedding: %s\n", row.ProductName, row.Title, row.SearchQuery, row.Embedding)
return
}
var cols []struct {
Relname string
Attname string
Typname string
Atttypmod int
}
if err := db.Raw(`
SELECT c.relname, a.attname, t.typname, a.atttypmod
FROM pg_attribute a
JOIN pg_class c ON c.oid = a.attrelid
JOIN pg_type t ON t.oid = a.atttypid
WHERE t.typname = 'vector' AND a.attnum > 0 AND c.relname LIKE 'brand\_%'
ORDER BY c.relname, a.attname`).Scan(&cols).Error; err != nil {
log.Fatal(err)
}
if len(cols) == 0 {
fmt.Println("no brand_* table has a vector column")
return
}
fmt.Printf("%-24s %-16s %-8s %5s %5s %5s\n", "table", "column", "type", "dims", "rows", "filled")
for _, c := range cols {
var total, filled int64
db.Raw(fmt.Sprintf(`SELECT COUNT(1) FROM %s`, c.Relname)).Scan(&total)
db.Raw(fmt.Sprintf(`SELECT COUNT(1) FROM %s WHERE %s IS NOT NULL`, c.Relname, c.Attname)).Scan(&filled)
fmt.Printf("%-24s %-16s %-8s %5d %5d %5d\n", c.Relname, c.Attname, c.Typname, c.Atttypmod, total, filled)
}
// nomic/bge emit unit vectors; a norm far from 1 means another pipeline.
for _, c := range cols {
var norm float64
db.Raw(fmt.Sprintf(`SELECT vector_norm(embedding) FROM %s WHERE embedding IS NOT NULL LIMIT 1`, c.Relname)).Scan(&norm)
if norm > 0 {
fmt.Printf("sample vector norm (%s): %.4f\n", c.Relname, norm)
break
}
}
}