Ask instead of guessing when a label fits several products

`"britannia"` is a substring of all 258 Britannia product names, and
textScore returned 0.95 for any product whose name contained the label.
So every one of them tied, the tie broke alphabetically, and the customer
was shown one arbitrary biscuit with "confidence": 0.95 and a price. Lens
hands back a bare wordmark often — it is usually the biggest thing printed
on a packet — so this was the common case, not an edge one. Found via the
example request in the mobile team's own proposal.

Scoring now asks both questions. A hit carries `score` (ranks) and `text`
(how specifically the label names THIS product: the harmonic mean of how
much of the label the product explains and how much of the product's name
the label explains, pack sizes dropped from both sides). A brand name
scores its products ~0.33 equally instead of 0.95 arbitrarily. The
"vector and text agree" bonus is now proportional to the text score, so a
weak match can no longer inflate a whole brand.

isAmbiguous reads that: the leader is a guess if anything is level with it
(margin) or if the label names no one product (specificity), and then the
response carries `ambiguous: true` with `candidates` — distinct products,
not pack sizes, at most ten, each marked with whether one of the
customer's stores has it in stock, available ones first. `match` is nil
and `stores` empty on that path: no price for a product nobody chose.
Erring towards asking is deliberate — a tap versus the wrong biscuit.

To act on a pick, /lookup now accepts `brand` + `catalogueid` instead of a
label and skips recognition entirely (also serves deep links and re-order).
New: ScanRepository.CatalogueRef, resolving via the brand tables discovered
from information_schema, never a name built from the request.

Also: scratch/cataloguedims now reports every vector column, not just
`embedding` — which is how we learned the catalogue also carries
img_vector(1024), filled on 1885 of 2124 rows. SCAN_TO_ORDER.md records
why that column stays unread for now and what would change it, alongside
why the app is not asked to compute vectors on the phone.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-09-23 10:56:02 +05:30
parent aaea1bfc00
commit 01bc89ab77
6 changed files with 821 additions and 75 deletions

View File

@@ -10,6 +10,7 @@ import (
"nearle/repositories"
"nearle/utils"
"sort"
"strconv"
"strings"
"sync"
"time"
@@ -49,6 +50,16 @@ const (
scanCatalogueTopK = 15
// Below this the best hit is not shown as a match at all.
scanMinScore = 0.30
// How close the runner-up has to be before the leader stops being an
// answer and the two become a question. See isAmbiguous.
scanAmbiguityMargin = 0.06
// A "did you mean?" list longer than this is not a choice, it is a
// catalogue — the customer is standing in a shop holding a packet.
scanMaxCandidates = 10
// How much of the winning product's name the label has to account for
// before it counts as having identified it. See isAmbiguous and
// textScore.
scanSpecificEnough = 0.55
)
// ScanErrors the controller maps to statuses. Everything else is a 500.
@@ -77,11 +88,15 @@ func NewScanService(repo repositories.ScanRepository, embedder utils.Embedder) S
func (s *scanService) Lookup(ctx context.Context, req models.ScanLookupRequest) (*models.ScanLookupResponse, error) {
label := strings.TrimSpace(req.Label)
// The caller can name the product outright instead of describing it —
// how the app resolves a candidate the customer picked.
direct := strings.TrimSpace(req.Brand) != "" && req.Catalogueid > 0
if req.Customerid <= 0 {
return nil, fmt.Errorf("%w: customerid is required", ErrScanBadRequest)
}
if label == "" {
return nil, fmt.Errorf("%w: label is required", ErrScanBadRequest)
if label == "" && !direct {
return nil, fmt.Errorf("%w: label, or brand and catalogueid, is required", ErrScanBadRequest)
}
if len(label) > scanMaxLabelLen {
label = label[:scanMaxLabelLen]
@@ -121,6 +136,10 @@ func (s *scanService) Lookup(ctx context.Context, req models.ScanLookupRequest)
}()
go func() {
defer wg.Done()
if direct {
hits, method, matchErr = s.resolveRef(ctx, req.Brand, req.Catalogueid)
return
}
hits, method, matchErr = s.searchCatalogue(ctx, label)
}()
wg.Wait()
@@ -141,21 +160,35 @@ func (s *scanService) Lookup(ctx context.Context, req models.ScanLookupRequest)
}
resp := &models.ScanLookupResponse{
Label: label,
Stores: []models.ScanStoreOffer{},
Variants: []models.ScanCatalogueMatch{},
Label: label,
Stores: []models.ScanStoreOffer{},
Variants: []models.ScanCatalogueMatch{},
Candidates: []models.ScanCatalogueMatch{},
}
// Verify the app's idea of the customer's tenants against the truth.
stores, resp.UnregisteredTenantids = restrictToTenants(stores, req.Tenantids)
if len(hits) == 0 || hits[0].score < scanMinScore {
switch {
case len(hits) == 0 && direct:
resp.Message = "That product is no longer in the catalogue."
return resp, nil
case len(hits) == 0, hits[0].score < scanMinScore:
resp.Message = "We couldn't recognise that product. Try a clearer photo of the front of the pack."
return resp, nil
}
best := hits[0]
family := catalogueFamily(hits)
distinct := distinctProducts(hits)
// Several products fit and none of them clearly wins — a bare brand name
// or a generic word. Ask rather than guess: naming one of them would put
// a confident price on a product the customer did not photograph.
if !direct && isAmbiguous(distinct) {
return s.candidatesResponse(ctx, resp, hits, distinct, method, stores)
}
best := distinct[0]
family := catalogueFamily(hits, best)
resp.Match = ptr(best.toMatch(method))
resp.Confidence = round3(best.score)
for _, h := range family {
@@ -174,11 +207,7 @@ func (s *scanService) Lookup(ctx context.Context, req models.ScanLookupRequest)
keys = append(keys, repositories.CatalogueKey{Brand: h.Brand, Catalogueid: h.ID, Imageid: h.ImageID})
names = append(names, h.ProductName)
}
locationids := make([]int, 0, len(stores))
for _, st := range stores {
locationids = append(locationids, st.Locationid)
}
rows, err := s.repo.StoreOptions(ctx, locationids, keys, names)
rows, err := s.repo.StoreOptions(ctx, locationIDs(stores), keys, names)
if err != nil {
return nil, err
}
@@ -208,6 +237,137 @@ func (s *scanService) Lookup(ctx context.Context, req models.ScanLookupRequest)
return resp, nil
}
// candidatesResponse answers an ambiguous label with the products to choose
// between, marking which of them the customer can actually buy right now.
//
// The availability read is the same StoreOptions query the confident path
// runs, widened to every candidate — so a "did you mean?" list can put the
// three that are in stock above the five that are not, instead of sending
// somebody to a shelf that has none of them.
func (s *scanService) candidatesResponse(ctx context.Context, resp *models.ScanLookupResponse,
hits, distinct []scoredHit, method string, stores []models.ScanStore) (*models.ScanLookupResponse, error) {
candidates := distinct
if len(candidates) > scanMaxCandidates {
candidates = candidates[:scanMaxCandidates]
}
resp.Ambiguous = true
resp.Confidence = round3(distinct[0].score)
// Every catalogue row belonging to a candidate, and a way back from what
// a tenant's product row carries to the candidate it stands for.
inCandidates := make(map[string]bool, len(candidates))
for _, c := range candidates {
inCandidates[c.productKey()] = true
}
var keys []repositories.CatalogueKey
var names []string
byImage := make(map[string]string)
byRef := make(map[string]string)
byName := make(map[string]string)
for _, h := range hits {
key := h.productKey()
if !inCandidates[key] {
continue
}
keys = append(keys, repositories.CatalogueKey{Brand: h.Brand, Catalogueid: h.ID, Imageid: h.ImageID})
names = append(names, h.ProductName)
if h.ImageID != "" {
byImage[h.ImageID] = key
}
byRef[refKey(h.Brand, h.ID)] = key
if n := strings.ToLower(strings.TrimSpace(h.ProductName)); n != "" {
byName[n] = key
}
}
stocked := make(map[string]bool)
if len(stores) > 0 && len(keys) > 0 {
rows, err := s.repo.StoreOptions(ctx, locationIDs(stores), keys, names)
if err != nil {
return nil, err
}
for _, row := range rows {
if row.Stock <= 0 {
continue
}
// Same precedence as optionFromRow: the stable key first.
if row.Imageid != "" {
if key, ok := byImage[row.Imageid]; ok {
stocked[key] = true
continue
}
}
if row.Catalogueid > 0 {
if key, ok := byRef[refKey(row.Productbrand, row.Catalogueid)]; ok {
stocked[key] = true
continue
}
}
if key, ok := byName[strings.ToLower(strings.TrimSpace(row.Productname))]; ok {
stocked[key] = true
}
}
}
for _, c := range candidates {
m := c.toMatch(method)
m.Available = stocked[c.productKey()]
resp.Candidates = append(resp.Candidates, m)
}
// Buyable first; within each group the search's own ranking stands.
sort.SliceStable(resp.Candidates, func(i, j int) bool {
return resp.Candidates[i].Available && !resp.Candidates[j].Available
})
available := 0
for _, c := range resp.Candidates {
if c.Available {
available++
}
}
if available > 0 {
resp.Message = fmt.Sprintf("Which one is it? %d of these %d are in stock near you.",
available, len(resp.Candidates))
} else {
resp.Message = fmt.Sprintf("Which one is it? We found %d products that could match.",
len(resp.Candidates))
}
return resp, nil
}
// resolveRef reads the product the caller named, with its pack sizes. No
// recognition, so every row scores 1 and the method says so.
func (s *scanService) resolveRef(ctx context.Context, brand string, id int64) ([]scoredHit, string, error) {
rows, err := s.repo.CatalogueRef(ctx, brand, id)
if err != nil {
switch {
case errors.Is(err, repositories.ErrCatalogueDBUnavailable):
return nil, "direct", ErrScanCatalogueDown
case errors.Is(err, repositories.ErrUnknownBrand):
return nil, "direct", fmt.Errorf("%w: unknown brand %q", ErrScanBadRequest, brand)
}
return nil, "direct", err
}
hits := make([]scoredHit, 0, len(rows))
for _, row := range rows {
hits = append(hits, scoredHit{CatalogueHit: row, score: 1})
}
return hits, "direct", nil
}
func refKey(brand string, id int64) string {
return strings.ToLower(strings.TrimSpace(brand)) + "#" + strconv.FormatInt(id, 10)
}
func locationIDs(stores []models.ScanStore) []int {
ids := make([]int, 0, len(stores))
for _, st := range stores {
ids = append(ids, st.Locationid)
}
return ids
}
// ── Confirm ─────────────────────────────────────────────────────────────────
func (s *scanService) Confirm(ctx context.Context, req models.ScanConfirmRequest) (*models.ScanConfirmResponse, error) {
@@ -356,7 +516,12 @@ func (s *scanService) Stores(ctx context.Context, customerid int, latStr, lngStr
type scoredHit struct {
repositories.CatalogueHit
// score ranks; text says how specifically the label names THIS product.
// Kept apart because they answer different questions: a vector neighbour
// can rank first while the label ("britannia") names no one product, and
// only the second number knows that.
score float64
text float64
}
func (h scoredHit) toMatch(method string) models.ScanCatalogueMatch {
@@ -389,11 +554,12 @@ func (s *scanService) searchCatalogue(ctx context.Context, label string) ([]scor
method = "vector+text"
}
tokens := utils.SearchTokens(label)
if cached, ok := s.repo.CachedHits(ctx, method+":"+s.modelName(), label); ok {
return scoreCachedHits(cached), method, nil
return scoreCachedHits(cached, label, tokens), method, nil
}
tokens := utils.SearchTokens(label)
byKey := make(map[string]*scoredHit)
keyOf := func(h repositories.CatalogueHit) string { return h.Brand + "#" + fmt.Sprint(h.ID) }
@@ -433,10 +599,15 @@ func (s *scanService) searchCatalogue(ctx context.Context, label string) ([]scor
for _, h := range thits {
ts := textScore(h, label, tokens)
if existing, ok := byKey[keyOf(h)]; ok {
existing.score = math.Min(1, math.Max(existing.score, ts)+0.10)
// The bonus is proportional: only a text match that actually
// names the product confirms a vector hit. A flat +0.10 let a
// bare brand name — which matches every one of that brand's
// products weakly — inflate all of them equally.
existing.score = math.Min(1, math.Max(existing.score, ts)+0.10*ts)
existing.text = ts
continue
}
byKey[keyOf(h)] = &scoredHit{CatalogueHit: h, score: ts}
byKey[keyOf(h)] = &scoredHit{CatalogueHit: h, score: ts, text: ts}
}
hits := make([]scoredHit, 0, len(byKey))
@@ -458,10 +629,18 @@ func (s *scanService) searchCatalogue(ctx context.Context, label string) ([]scor
return hits, method, nil
}
func scoreCachedHits(cached []repositories.CatalogueHit) []scoredHit {
// scoreCachedHits restores the ranking the cache holds, and recomputes the
// text score from the row itself — the cache carries one number per row, and
// recomputing costs nothing while leaving out the specificity signal would
// make every cached lookup read as ambiguous.
func scoreCachedHits(cached []repositories.CatalogueHit, label string, tokens []string) []scoredHit {
hits := make([]scoredHit, 0, len(cached))
for _, c := range cached {
hits = append(hits, scoredHit{CatalogueHit: c, score: 1 - c.Distance})
hits = append(hits, scoredHit{
CatalogueHit: c,
score: 1 - c.Distance,
text: textScore(c, label, tokens),
})
}
sortHits(hits)
return hits
@@ -495,51 +674,149 @@ func (s *scanService) embed(ctx context.Context, label string) ([]float32, error
return v, nil
}
// textScore is how well a catalogue row's name matches the words Lens read.
// The whole label as a substring of the name is near-certain; otherwise the
// share of label words found in name+title, scaled so that "all of them"
// stops short of the substring case.
// textScore is how well a catalogue row matches the words Lens read.
//
// Both directions count, and that is the whole point:
//
// - labelCoverage — how much of what the customer said this product
// accounts for. "Milk Bikis" against "Milk Bikis 100g" is all of it.
// - nameCoverage — how much of the product the label accounts for, which
// is what makes the match SPECIFIC. "britannia" explains one word of
// "Britannia Good Day Cashew Cookies", so it does not identify it.
//
// The score is their harmonic mean, so a high score needs both.
//
// This replaces `strings.Contains(name, label) → 0.95`, which asked only the
// first question. A bare brand name is a substring of every one of that
// brand's products, so all 258 Britannia rows scored 0.95, the tie broke
// alphabetically, and the customer was shown one arbitrary biscuit with
// "confidence": 0.95. Lens returns a bare wordmark often — it is usually the
// most legible thing on a packet — so that was not an edge case.
//
// Now those rows score ~0.33 and, crucially, score it EQUALLY, which is what
// isAmbiguous reads to answer "did you mean?" instead of guessing.
func textScore(h repositories.CatalogueHit, label string, tokens []string) float64 {
name := strings.ToLower(h.ProductName)
hay := name + " " + strings.ToLower(h.Title)
label = strings.ToLower(strings.TrimSpace(label))
if label != "" && strings.Contains(name, label) {
return 0.95
}
if len(tokens) == 0 {
return 0
}
hay := strings.ToLower(h.ProductName + " " + h.Title)
found := 0
for _, t := range tokens {
if strings.Contains(hay, t) {
found++
}
}
return 0.8 * float64(found) / float64(len(tokens))
if found == 0 {
return 0
}
labelCoverage := float64(found) / float64(len(tokens))
// Pack sizes are dropped from both sides (SearchTokens), so "100g" never
// counts as a word the label failed to explain.
nameTokens := utils.SearchTokens(h.ProductName)
if len(nameTokens) == 0 {
return 0.5 * labelCoverage
}
explained := 0
for _, n := range nameTokens {
for _, t := range tokens {
if tokenMatch(n, t) {
explained++
break
}
}
}
if explained == 0 {
// Matched the title but not the name. Weak, not zero.
return 0.4 * labelCoverage
}
nameCoverage := float64(explained) / float64(len(nameTokens))
return 2 * labelCoverage * nameCoverage / (labelCoverage + nameCoverage)
}
// catalogueFamily is the best hit and its other pack sizes: same brand, and
// the same variant_key when the catalogue assigned one, else the same name.
// Every member is a separate catalogue row a shop may have imported.
func catalogueFamily(hits []scoredHit) []scoredHit {
if len(hits) == 0 {
return nil
// tokenMatch is equality, plus containment for words long enough that a
// shared prefix means something ("cookie"/"cookies", "chocolate"/"choco").
// Short tokens must match exactly, or "day" would match "daybreak".
func tokenMatch(a, b string) bool {
if a == b {
return true
}
best := hits[0]
family := []scoredHit{best}
for _, h := range hits[1:] {
if h.Brand != best.Brand {
if len(a) >= 5 && strings.Contains(b, a) {
return true
}
return len(b) >= 5 && strings.Contains(a, b)
}
// productKey identifies a product across its pack sizes: the catalogue's own
// variant_key where it assigned one, the name otherwise, always within a
// brand. Two rows sharing it are 100 g and 200 g of one thing; two rows that
// do not are different products to choose between.
func productKey(h repositories.CatalogueHit) string {
if k := strings.TrimSpace(h.VariantKey); k != "" {
return h.Brand + "/" + strings.ToLower(k)
}
return h.Brand + "/" + strings.ToLower(strings.TrimSpace(h.ProductName))
}
// productKey of a scored hit — scoredHit embeds the row it scored.
func (h scoredHit) productKey() string { return productKey(h.CatalogueHit) }
// distinctProducts keeps the best-scoring row of each product, in rank
// order — the list of things the customer could actually be shown to choose
// between, as opposed to the same product listed four times in four sizes.
func distinctProducts(hits []scoredHit) []scoredHit {
seen := make(map[string]bool, len(hits))
out := make([]scoredHit, 0, len(hits))
for _, h := range hits {
key := h.productKey()
if seen[key] {
continue
}
switch {
case best.VariantKey != "" && h.VariantKey != "":
if h.VariantKey == best.VariantKey {
family = append(family, h)
}
case strings.EqualFold(strings.TrimSpace(h.ProductName), strings.TrimSpace(best.ProductName)):
seen[key] = true
out = append(out, h)
}
return out
}
// isAmbiguous reports that naming the leader as THE match would be a guess
// dressed up as an answer. Two ways that happens:
//
// 1. Something else is level with it. A margin rather than an absolute
// threshold, because what matters is not how high the best score is but
// whether anything is tied with it.
// 2. Nothing is level, but the label does not actually name a product —
// a bare brand, a generic word, or a spelling the catalogue does not
// carry. The leader may still rank first on vector similarity, and
// ranking first among vague matches is not identification.
//
// Erring towards asking is deliberate. Asking costs the customer one tap on
// a picture; guessing wrong costs them the wrong biscuit and costs us the
// belief that the scanner works. An exact product name still scores ~1.0 on
// specificity, so the common case is unaffected.
func isAmbiguous(distinct []scoredHit) bool {
if len(distinct) < 2 {
return false
}
if distinct[1].score >= distinct[0].score-scanAmbiguityMargin {
return true
}
return distinct[0].text < scanSpecificEnough
}
// catalogueFamily is `of` and its other pack sizes, drawn from hits.
func catalogueFamily(hits []scoredHit, of scoredHit) []scoredHit {
key := of.productKey()
family := make([]scoredHit, 0, 4)
for _, h := range hits {
if h.productKey() == key {
family = append(family, h)
}
}
if len(family) == 0 {
return []scoredHit{of}
}
return family
}