Ask instead of guessing when a label fits several products
`"britannia"` is a substring of all 258 Britannia product names, and textScore returned 0.95 for any product whose name contained the label. So every one of them tied, the tie broke alphabetically, and the customer was shown one arbitrary biscuit with "confidence": 0.95 and a price. Lens hands back a bare wordmark often — it is usually the biggest thing printed on a packet — so this was the common case, not an edge one. Found via the example request in the mobile team's own proposal. Scoring now asks both questions. A hit carries `score` (ranks) and `text` (how specifically the label names THIS product: the harmonic mean of how much of the label the product explains and how much of the product's name the label explains, pack sizes dropped from both sides). A brand name scores its products ~0.33 equally instead of 0.95 arbitrarily. The "vector and text agree" bonus is now proportional to the text score, so a weak match can no longer inflate a whole brand. isAmbiguous reads that: the leader is a guess if anything is level with it (margin) or if the label names no one product (specificity), and then the response carries `ambiguous: true` with `candidates` — distinct products, not pack sizes, at most ten, each marked with whether one of the customer's stores has it in stock, available ones first. `match` is nil and `stores` empty on that path: no price for a product nobody chose. Erring towards asking is deliberate — a tap versus the wrong biscuit. To act on a pick, /lookup now accepts `brand` + `catalogueid` instead of a label and skips recognition entirely (also serves deep links and re-order). New: ScanRepository.CatalogueRef, resolving via the brand tables discovered from information_schema, never a name built from the request. Also: scratch/cataloguedims now reports every vector column, not just `embedding` — which is how we learned the catalogue also carries img_vector(1024), filled on 1885 of 2124 rows. SCAN_TO_ORDER.md records why that column stays unread for now and what would change it, alongside why the app is not asked to compute vectors on the phone. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -10,6 +10,7 @@ import (
|
||||
"nearle/repositories"
|
||||
"nearle/utils"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
@@ -49,6 +50,16 @@ const (
|
||||
scanCatalogueTopK = 15
|
||||
// Below this the best hit is not shown as a match at all.
|
||||
scanMinScore = 0.30
|
||||
// How close the runner-up has to be before the leader stops being an
|
||||
// answer and the two become a question. See isAmbiguous.
|
||||
scanAmbiguityMargin = 0.06
|
||||
// A "did you mean?" list longer than this is not a choice, it is a
|
||||
// catalogue — the customer is standing in a shop holding a packet.
|
||||
scanMaxCandidates = 10
|
||||
// How much of the winning product's name the label has to account for
|
||||
// before it counts as having identified it. See isAmbiguous and
|
||||
// textScore.
|
||||
scanSpecificEnough = 0.55
|
||||
)
|
||||
|
||||
// ScanErrors the controller maps to statuses. Everything else is a 500.
|
||||
@@ -77,11 +88,15 @@ func NewScanService(repo repositories.ScanRepository, embedder utils.Embedder) S
|
||||
|
||||
func (s *scanService) Lookup(ctx context.Context, req models.ScanLookupRequest) (*models.ScanLookupResponse, error) {
|
||||
label := strings.TrimSpace(req.Label)
|
||||
// The caller can name the product outright instead of describing it —
|
||||
// how the app resolves a candidate the customer picked.
|
||||
direct := strings.TrimSpace(req.Brand) != "" && req.Catalogueid > 0
|
||||
|
||||
if req.Customerid <= 0 {
|
||||
return nil, fmt.Errorf("%w: customerid is required", ErrScanBadRequest)
|
||||
}
|
||||
if label == "" {
|
||||
return nil, fmt.Errorf("%w: label is required", ErrScanBadRequest)
|
||||
if label == "" && !direct {
|
||||
return nil, fmt.Errorf("%w: label, or brand and catalogueid, is required", ErrScanBadRequest)
|
||||
}
|
||||
if len(label) > scanMaxLabelLen {
|
||||
label = label[:scanMaxLabelLen]
|
||||
@@ -121,6 +136,10 @@ func (s *scanService) Lookup(ctx context.Context, req models.ScanLookupRequest)
|
||||
}()
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
if direct {
|
||||
hits, method, matchErr = s.resolveRef(ctx, req.Brand, req.Catalogueid)
|
||||
return
|
||||
}
|
||||
hits, method, matchErr = s.searchCatalogue(ctx, label)
|
||||
}()
|
||||
wg.Wait()
|
||||
@@ -141,21 +160,35 @@ func (s *scanService) Lookup(ctx context.Context, req models.ScanLookupRequest)
|
||||
}
|
||||
|
||||
resp := &models.ScanLookupResponse{
|
||||
Label: label,
|
||||
Stores: []models.ScanStoreOffer{},
|
||||
Variants: []models.ScanCatalogueMatch{},
|
||||
Label: label,
|
||||
Stores: []models.ScanStoreOffer{},
|
||||
Variants: []models.ScanCatalogueMatch{},
|
||||
Candidates: []models.ScanCatalogueMatch{},
|
||||
}
|
||||
|
||||
// Verify the app's idea of the customer's tenants against the truth.
|
||||
stores, resp.UnregisteredTenantids = restrictToTenants(stores, req.Tenantids)
|
||||
|
||||
if len(hits) == 0 || hits[0].score < scanMinScore {
|
||||
switch {
|
||||
case len(hits) == 0 && direct:
|
||||
resp.Message = "That product is no longer in the catalogue."
|
||||
return resp, nil
|
||||
case len(hits) == 0, hits[0].score < scanMinScore:
|
||||
resp.Message = "We couldn't recognise that product. Try a clearer photo of the front of the pack."
|
||||
return resp, nil
|
||||
}
|
||||
|
||||
best := hits[0]
|
||||
family := catalogueFamily(hits)
|
||||
distinct := distinctProducts(hits)
|
||||
|
||||
// Several products fit and none of them clearly wins — a bare brand name
|
||||
// or a generic word. Ask rather than guess: naming one of them would put
|
||||
// a confident price on a product the customer did not photograph.
|
||||
if !direct && isAmbiguous(distinct) {
|
||||
return s.candidatesResponse(ctx, resp, hits, distinct, method, stores)
|
||||
}
|
||||
|
||||
best := distinct[0]
|
||||
family := catalogueFamily(hits, best)
|
||||
resp.Match = ptr(best.toMatch(method))
|
||||
resp.Confidence = round3(best.score)
|
||||
for _, h := range family {
|
||||
@@ -174,11 +207,7 @@ func (s *scanService) Lookup(ctx context.Context, req models.ScanLookupRequest)
|
||||
keys = append(keys, repositories.CatalogueKey{Brand: h.Brand, Catalogueid: h.ID, Imageid: h.ImageID})
|
||||
names = append(names, h.ProductName)
|
||||
}
|
||||
locationids := make([]int, 0, len(stores))
|
||||
for _, st := range stores {
|
||||
locationids = append(locationids, st.Locationid)
|
||||
}
|
||||
rows, err := s.repo.StoreOptions(ctx, locationids, keys, names)
|
||||
rows, err := s.repo.StoreOptions(ctx, locationIDs(stores), keys, names)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -208,6 +237,137 @@ func (s *scanService) Lookup(ctx context.Context, req models.ScanLookupRequest)
|
||||
return resp, nil
|
||||
}
|
||||
|
||||
// candidatesResponse answers an ambiguous label with the products to choose
|
||||
// between, marking which of them the customer can actually buy right now.
|
||||
//
|
||||
// The availability read is the same StoreOptions query the confident path
|
||||
// runs, widened to every candidate — so a "did you mean?" list can put the
|
||||
// three that are in stock above the five that are not, instead of sending
|
||||
// somebody to a shelf that has none of them.
|
||||
func (s *scanService) candidatesResponse(ctx context.Context, resp *models.ScanLookupResponse,
|
||||
hits, distinct []scoredHit, method string, stores []models.ScanStore) (*models.ScanLookupResponse, error) {
|
||||
|
||||
candidates := distinct
|
||||
if len(candidates) > scanMaxCandidates {
|
||||
candidates = candidates[:scanMaxCandidates]
|
||||
}
|
||||
resp.Ambiguous = true
|
||||
resp.Confidence = round3(distinct[0].score)
|
||||
|
||||
// Every catalogue row belonging to a candidate, and a way back from what
|
||||
// a tenant's product row carries to the candidate it stands for.
|
||||
inCandidates := make(map[string]bool, len(candidates))
|
||||
for _, c := range candidates {
|
||||
inCandidates[c.productKey()] = true
|
||||
}
|
||||
var keys []repositories.CatalogueKey
|
||||
var names []string
|
||||
byImage := make(map[string]string)
|
||||
byRef := make(map[string]string)
|
||||
byName := make(map[string]string)
|
||||
for _, h := range hits {
|
||||
key := h.productKey()
|
||||
if !inCandidates[key] {
|
||||
continue
|
||||
}
|
||||
keys = append(keys, repositories.CatalogueKey{Brand: h.Brand, Catalogueid: h.ID, Imageid: h.ImageID})
|
||||
names = append(names, h.ProductName)
|
||||
if h.ImageID != "" {
|
||||
byImage[h.ImageID] = key
|
||||
}
|
||||
byRef[refKey(h.Brand, h.ID)] = key
|
||||
if n := strings.ToLower(strings.TrimSpace(h.ProductName)); n != "" {
|
||||
byName[n] = key
|
||||
}
|
||||
}
|
||||
|
||||
stocked := make(map[string]bool)
|
||||
if len(stores) > 0 && len(keys) > 0 {
|
||||
rows, err := s.repo.StoreOptions(ctx, locationIDs(stores), keys, names)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
for _, row := range rows {
|
||||
if row.Stock <= 0 {
|
||||
continue
|
||||
}
|
||||
// Same precedence as optionFromRow: the stable key first.
|
||||
if row.Imageid != "" {
|
||||
if key, ok := byImage[row.Imageid]; ok {
|
||||
stocked[key] = true
|
||||
continue
|
||||
}
|
||||
}
|
||||
if row.Catalogueid > 0 {
|
||||
if key, ok := byRef[refKey(row.Productbrand, row.Catalogueid)]; ok {
|
||||
stocked[key] = true
|
||||
continue
|
||||
}
|
||||
}
|
||||
if key, ok := byName[strings.ToLower(strings.TrimSpace(row.Productname))]; ok {
|
||||
stocked[key] = true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for _, c := range candidates {
|
||||
m := c.toMatch(method)
|
||||
m.Available = stocked[c.productKey()]
|
||||
resp.Candidates = append(resp.Candidates, m)
|
||||
}
|
||||
// Buyable first; within each group the search's own ranking stands.
|
||||
sort.SliceStable(resp.Candidates, func(i, j int) bool {
|
||||
return resp.Candidates[i].Available && !resp.Candidates[j].Available
|
||||
})
|
||||
|
||||
available := 0
|
||||
for _, c := range resp.Candidates {
|
||||
if c.Available {
|
||||
available++
|
||||
}
|
||||
}
|
||||
if available > 0 {
|
||||
resp.Message = fmt.Sprintf("Which one is it? %d of these %d are in stock near you.",
|
||||
available, len(resp.Candidates))
|
||||
} else {
|
||||
resp.Message = fmt.Sprintf("Which one is it? We found %d products that could match.",
|
||||
len(resp.Candidates))
|
||||
}
|
||||
return resp, nil
|
||||
}
|
||||
|
||||
// resolveRef reads the product the caller named, with its pack sizes. No
|
||||
// recognition, so every row scores 1 and the method says so.
|
||||
func (s *scanService) resolveRef(ctx context.Context, brand string, id int64) ([]scoredHit, string, error) {
|
||||
rows, err := s.repo.CatalogueRef(ctx, brand, id)
|
||||
if err != nil {
|
||||
switch {
|
||||
case errors.Is(err, repositories.ErrCatalogueDBUnavailable):
|
||||
return nil, "direct", ErrScanCatalogueDown
|
||||
case errors.Is(err, repositories.ErrUnknownBrand):
|
||||
return nil, "direct", fmt.Errorf("%w: unknown brand %q", ErrScanBadRequest, brand)
|
||||
}
|
||||
return nil, "direct", err
|
||||
}
|
||||
hits := make([]scoredHit, 0, len(rows))
|
||||
for _, row := range rows {
|
||||
hits = append(hits, scoredHit{CatalogueHit: row, score: 1})
|
||||
}
|
||||
return hits, "direct", nil
|
||||
}
|
||||
|
||||
func refKey(brand string, id int64) string {
|
||||
return strings.ToLower(strings.TrimSpace(brand)) + "#" + strconv.FormatInt(id, 10)
|
||||
}
|
||||
|
||||
func locationIDs(stores []models.ScanStore) []int {
|
||||
ids := make([]int, 0, len(stores))
|
||||
for _, st := range stores {
|
||||
ids = append(ids, st.Locationid)
|
||||
}
|
||||
return ids
|
||||
}
|
||||
|
||||
// ── Confirm ─────────────────────────────────────────────────────────────────
|
||||
|
||||
func (s *scanService) Confirm(ctx context.Context, req models.ScanConfirmRequest) (*models.ScanConfirmResponse, error) {
|
||||
@@ -356,7 +516,12 @@ func (s *scanService) Stores(ctx context.Context, customerid int, latStr, lngStr
|
||||
|
||||
type scoredHit struct {
|
||||
repositories.CatalogueHit
|
||||
// score ranks; text says how specifically the label names THIS product.
|
||||
// Kept apart because they answer different questions: a vector neighbour
|
||||
// can rank first while the label ("britannia") names no one product, and
|
||||
// only the second number knows that.
|
||||
score float64
|
||||
text float64
|
||||
}
|
||||
|
||||
func (h scoredHit) toMatch(method string) models.ScanCatalogueMatch {
|
||||
@@ -389,11 +554,12 @@ func (s *scanService) searchCatalogue(ctx context.Context, label string) ([]scor
|
||||
method = "vector+text"
|
||||
}
|
||||
|
||||
tokens := utils.SearchTokens(label)
|
||||
|
||||
if cached, ok := s.repo.CachedHits(ctx, method+":"+s.modelName(), label); ok {
|
||||
return scoreCachedHits(cached), method, nil
|
||||
return scoreCachedHits(cached, label, tokens), method, nil
|
||||
}
|
||||
|
||||
tokens := utils.SearchTokens(label)
|
||||
byKey := make(map[string]*scoredHit)
|
||||
keyOf := func(h repositories.CatalogueHit) string { return h.Brand + "#" + fmt.Sprint(h.ID) }
|
||||
|
||||
@@ -433,10 +599,15 @@ func (s *scanService) searchCatalogue(ctx context.Context, label string) ([]scor
|
||||
for _, h := range thits {
|
||||
ts := textScore(h, label, tokens)
|
||||
if existing, ok := byKey[keyOf(h)]; ok {
|
||||
existing.score = math.Min(1, math.Max(existing.score, ts)+0.10)
|
||||
// The bonus is proportional: only a text match that actually
|
||||
// names the product confirms a vector hit. A flat +0.10 let a
|
||||
// bare brand name — which matches every one of that brand's
|
||||
// products weakly — inflate all of them equally.
|
||||
existing.score = math.Min(1, math.Max(existing.score, ts)+0.10*ts)
|
||||
existing.text = ts
|
||||
continue
|
||||
}
|
||||
byKey[keyOf(h)] = &scoredHit{CatalogueHit: h, score: ts}
|
||||
byKey[keyOf(h)] = &scoredHit{CatalogueHit: h, score: ts, text: ts}
|
||||
}
|
||||
|
||||
hits := make([]scoredHit, 0, len(byKey))
|
||||
@@ -458,10 +629,18 @@ func (s *scanService) searchCatalogue(ctx context.Context, label string) ([]scor
|
||||
return hits, method, nil
|
||||
}
|
||||
|
||||
func scoreCachedHits(cached []repositories.CatalogueHit) []scoredHit {
|
||||
// scoreCachedHits restores the ranking the cache holds, and recomputes the
|
||||
// text score from the row itself — the cache carries one number per row, and
|
||||
// recomputing costs nothing while leaving out the specificity signal would
|
||||
// make every cached lookup read as ambiguous.
|
||||
func scoreCachedHits(cached []repositories.CatalogueHit, label string, tokens []string) []scoredHit {
|
||||
hits := make([]scoredHit, 0, len(cached))
|
||||
for _, c := range cached {
|
||||
hits = append(hits, scoredHit{CatalogueHit: c, score: 1 - c.Distance})
|
||||
hits = append(hits, scoredHit{
|
||||
CatalogueHit: c,
|
||||
score: 1 - c.Distance,
|
||||
text: textScore(c, label, tokens),
|
||||
})
|
||||
}
|
||||
sortHits(hits)
|
||||
return hits
|
||||
@@ -495,51 +674,149 @@ func (s *scanService) embed(ctx context.Context, label string) ([]float32, error
|
||||
return v, nil
|
||||
}
|
||||
|
||||
// textScore is how well a catalogue row's name matches the words Lens read.
|
||||
// The whole label as a substring of the name is near-certain; otherwise the
|
||||
// share of label words found in name+title, scaled so that "all of them"
|
||||
// stops short of the substring case.
|
||||
// textScore is how well a catalogue row matches the words Lens read.
|
||||
//
|
||||
// Both directions count, and that is the whole point:
|
||||
//
|
||||
// - labelCoverage — how much of what the customer said this product
|
||||
// accounts for. "Milk Bikis" against "Milk Bikis 100g" is all of it.
|
||||
// - nameCoverage — how much of the product the label accounts for, which
|
||||
// is what makes the match SPECIFIC. "britannia" explains one word of
|
||||
// "Britannia Good Day Cashew Cookies", so it does not identify it.
|
||||
//
|
||||
// The score is their harmonic mean, so a high score needs both.
|
||||
//
|
||||
// This replaces `strings.Contains(name, label) → 0.95`, which asked only the
|
||||
// first question. A bare brand name is a substring of every one of that
|
||||
// brand's products, so all 258 Britannia rows scored 0.95, the tie broke
|
||||
// alphabetically, and the customer was shown one arbitrary biscuit with
|
||||
// "confidence": 0.95. Lens returns a bare wordmark often — it is usually the
|
||||
// most legible thing on a packet — so that was not an edge case.
|
||||
//
|
||||
// Now those rows score ~0.33 and, crucially, score it EQUALLY, which is what
|
||||
// isAmbiguous reads to answer "did you mean?" instead of guessing.
|
||||
func textScore(h repositories.CatalogueHit, label string, tokens []string) float64 {
|
||||
name := strings.ToLower(h.ProductName)
|
||||
hay := name + " " + strings.ToLower(h.Title)
|
||||
label = strings.ToLower(strings.TrimSpace(label))
|
||||
if label != "" && strings.Contains(name, label) {
|
||||
return 0.95
|
||||
}
|
||||
if len(tokens) == 0 {
|
||||
return 0
|
||||
}
|
||||
hay := strings.ToLower(h.ProductName + " " + h.Title)
|
||||
|
||||
found := 0
|
||||
for _, t := range tokens {
|
||||
if strings.Contains(hay, t) {
|
||||
found++
|
||||
}
|
||||
}
|
||||
return 0.8 * float64(found) / float64(len(tokens))
|
||||
if found == 0 {
|
||||
return 0
|
||||
}
|
||||
labelCoverage := float64(found) / float64(len(tokens))
|
||||
|
||||
// Pack sizes are dropped from both sides (SearchTokens), so "100g" never
|
||||
// counts as a word the label failed to explain.
|
||||
nameTokens := utils.SearchTokens(h.ProductName)
|
||||
if len(nameTokens) == 0 {
|
||||
return 0.5 * labelCoverage
|
||||
}
|
||||
explained := 0
|
||||
for _, n := range nameTokens {
|
||||
for _, t := range tokens {
|
||||
if tokenMatch(n, t) {
|
||||
explained++
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
if explained == 0 {
|
||||
// Matched the title but not the name. Weak, not zero.
|
||||
return 0.4 * labelCoverage
|
||||
}
|
||||
nameCoverage := float64(explained) / float64(len(nameTokens))
|
||||
|
||||
return 2 * labelCoverage * nameCoverage / (labelCoverage + nameCoverage)
|
||||
}
|
||||
|
||||
// catalogueFamily is the best hit and its other pack sizes: same brand, and
|
||||
// the same variant_key when the catalogue assigned one, else the same name.
|
||||
// Every member is a separate catalogue row a shop may have imported.
|
||||
func catalogueFamily(hits []scoredHit) []scoredHit {
|
||||
if len(hits) == 0 {
|
||||
return nil
|
||||
// tokenMatch is equality, plus containment for words long enough that a
|
||||
// shared prefix means something ("cookie"/"cookies", "chocolate"/"choco").
|
||||
// Short tokens must match exactly, or "day" would match "daybreak".
|
||||
func tokenMatch(a, b string) bool {
|
||||
if a == b {
|
||||
return true
|
||||
}
|
||||
best := hits[0]
|
||||
family := []scoredHit{best}
|
||||
for _, h := range hits[1:] {
|
||||
if h.Brand != best.Brand {
|
||||
if len(a) >= 5 && strings.Contains(b, a) {
|
||||
return true
|
||||
}
|
||||
return len(b) >= 5 && strings.Contains(a, b)
|
||||
}
|
||||
|
||||
// productKey identifies a product across its pack sizes: the catalogue's own
|
||||
// variant_key where it assigned one, the name otherwise, always within a
|
||||
// brand. Two rows sharing it are 100 g and 200 g of one thing; two rows that
|
||||
// do not are different products to choose between.
|
||||
func productKey(h repositories.CatalogueHit) string {
|
||||
if k := strings.TrimSpace(h.VariantKey); k != "" {
|
||||
return h.Brand + "/" + strings.ToLower(k)
|
||||
}
|
||||
return h.Brand + "/" + strings.ToLower(strings.TrimSpace(h.ProductName))
|
||||
}
|
||||
|
||||
// productKey of a scored hit — scoredHit embeds the row it scored.
|
||||
func (h scoredHit) productKey() string { return productKey(h.CatalogueHit) }
|
||||
|
||||
// distinctProducts keeps the best-scoring row of each product, in rank
|
||||
// order — the list of things the customer could actually be shown to choose
|
||||
// between, as opposed to the same product listed four times in four sizes.
|
||||
func distinctProducts(hits []scoredHit) []scoredHit {
|
||||
seen := make(map[string]bool, len(hits))
|
||||
out := make([]scoredHit, 0, len(hits))
|
||||
for _, h := range hits {
|
||||
key := h.productKey()
|
||||
if seen[key] {
|
||||
continue
|
||||
}
|
||||
switch {
|
||||
case best.VariantKey != "" && h.VariantKey != "":
|
||||
if h.VariantKey == best.VariantKey {
|
||||
family = append(family, h)
|
||||
}
|
||||
case strings.EqualFold(strings.TrimSpace(h.ProductName), strings.TrimSpace(best.ProductName)):
|
||||
seen[key] = true
|
||||
out = append(out, h)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// isAmbiguous reports that naming the leader as THE match would be a guess
|
||||
// dressed up as an answer. Two ways that happens:
|
||||
//
|
||||
// 1. Something else is level with it. A margin rather than an absolute
|
||||
// threshold, because what matters is not how high the best score is but
|
||||
// whether anything is tied with it.
|
||||
// 2. Nothing is level, but the label does not actually name a product —
|
||||
// a bare brand, a generic word, or a spelling the catalogue does not
|
||||
// carry. The leader may still rank first on vector similarity, and
|
||||
// ranking first among vague matches is not identification.
|
||||
//
|
||||
// Erring towards asking is deliberate. Asking costs the customer one tap on
|
||||
// a picture; guessing wrong costs them the wrong biscuit and costs us the
|
||||
// belief that the scanner works. An exact product name still scores ~1.0 on
|
||||
// specificity, so the common case is unaffected.
|
||||
func isAmbiguous(distinct []scoredHit) bool {
|
||||
if len(distinct) < 2 {
|
||||
return false
|
||||
}
|
||||
if distinct[1].score >= distinct[0].score-scanAmbiguityMargin {
|
||||
return true
|
||||
}
|
||||
return distinct[0].text < scanSpecificEnough
|
||||
}
|
||||
|
||||
// catalogueFamily is `of` and its other pack sizes, drawn from hits.
|
||||
func catalogueFamily(hits []scoredHit, of scoredHit) []scoredHit {
|
||||
key := of.productKey()
|
||||
family := make([]scoredHit, 0, 4)
|
||||
for _, h := range hits {
|
||||
if h.productKey() == key {
|
||||
family = append(family, h)
|
||||
}
|
||||
}
|
||||
if len(family) == 0 {
|
||||
return []scoredHit{of}
|
||||
}
|
||||
return family
|
||||
}
|
||||
|
||||
|
||||
@@ -3,10 +3,13 @@ package services
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"nearle/models"
|
||||
"nearle/repositories"
|
||||
"nearle/utils"
|
||||
)
|
||||
|
||||
/*
|
||||
@@ -30,9 +33,13 @@ type fakeScanRepo struct {
|
||||
options []repositories.StoreOptionRow
|
||||
at map[int]*repositories.StoreOptionRow // productid → row
|
||||
|
||||
ref []repositories.CatalogueHit
|
||||
refErr error
|
||||
|
||||
askedKeys []repositories.CatalogueKey
|
||||
askedNames []string
|
||||
askedLocs []int
|
||||
askedRef string
|
||||
cachedHits map[string][]repositories.CatalogueHit
|
||||
}
|
||||
|
||||
@@ -69,6 +76,10 @@ func (f *fakeScanRepo) TextSearch(context.Context, string, int) ([]repositories.
|
||||
return f.text, nil
|
||||
}
|
||||
func (f *fakeScanRepo) VectorSearchAvailable() bool { return f.hasVec }
|
||||
func (f *fakeScanRepo) CatalogueRef(_ context.Context, brand string, id int64) ([]repositories.CatalogueHit, error) {
|
||||
f.askedRef = fmt.Sprintf("%s#%d", brand, id)
|
||||
return f.ref, f.refErr
|
||||
}
|
||||
func (f *fakeScanRepo) CachedVector(context.Context, string, string) ([]float32, bool) {
|
||||
return nil, false
|
||||
}
|
||||
@@ -423,7 +434,7 @@ func TestCatalogueFamilyGroupsByVariantKeyThenName(t *testing.T) {
|
||||
{CatalogueHit: milkBikis200, score: 0.88},
|
||||
{CatalogueHit: repositories.CatalogueHit{Brand: "parle", ProductName: "Milk Bikis", VariantKey: "milk_bikis"}, score: 0.5},
|
||||
}
|
||||
family := catalogueFamily(hits)
|
||||
family := catalogueFamily(hits, hits[0])
|
||||
if len(family) != 2 || family[1].ID != 8 {
|
||||
t.Fatalf("family should be the two britannia sizes, got %+v", family)
|
||||
}
|
||||
@@ -432,7 +443,217 @@ func TestCatalogueFamilyGroupsByVariantKeyThenName(t *testing.T) {
|
||||
a := scoredHit{CatalogueHit: repositories.CatalogueHit{Brand: "b", ID: 1, ProductName: "Honey"}}
|
||||
b := scoredHit{CatalogueHit: repositories.CatalogueHit{Brand: "b", ID: 2, ProductName: "honey "}}
|
||||
c := scoredHit{CatalogueHit: repositories.CatalogueHit{Brand: "b", ID: 3, ProductName: "Honey Lite"}}
|
||||
if family := catalogueFamily([]scoredHit{a, b, c}); len(family) != 2 {
|
||||
if family := catalogueFamily([]scoredHit{a, b, c}, a); len(family) != 2 {
|
||||
t.Errorf("name match should join 1 and 2 only, got %+v", family)
|
||||
}
|
||||
}
|
||||
|
||||
/*
|
||||
Ambiguity.
|
||||
|
||||
Lens hands back whatever was most legible on the packet, and on a packet that
|
||||
is very often the brand wordmark alone. "britannia" fits 258 catalogue rows
|
||||
equally well, so there is no best one — and the old scoring said otherwise:
|
||||
every product whose name contained the label scored 0.95, the tie broke
|
||||
alphabetically, and the customer was shown one arbitrary biscuit with
|
||||
"confidence": 0.95 and a price. These tests are the contract that it asks
|
||||
instead.
|
||||
*/
|
||||
|
||||
// Three different Britannia products, of which the customer's stores stock
|
||||
// one. Text-only: no embedder, which is also how production runs until the
|
||||
// model is configured.
|
||||
func newBrandLabelFixture() *fakeScanRepo {
|
||||
cashew := repositories.CatalogueHit{Brand: "britannia", ID: 21, ProductName: "Britannia Good Day Cashew Cookies", VariantKey: "good_day_cashew", ImageID: "britannia_good_day_cashew"}
|
||||
butter := repositories.CatalogueHit{Brand: "britannia", ID: 22, ProductName: "Britannia Good Day Butter Cookies", VariantKey: "good_day_butter", ImageID: "britannia_good_day_butter"}
|
||||
marie := repositories.CatalogueHit{Brand: "britannia", ID: 23, ProductName: "Britannia Marie Gold", VariantKey: "marie_gold", ImageID: "britannia_marie_gold"}
|
||||
|
||||
return &fakeScanRepo{
|
||||
exists: true,
|
||||
stores: fixtureStores(),
|
||||
text: []repositories.CatalogueHit{cashew, butter, marie},
|
||||
options: []repositories.StoreOptionRow{
|
||||
{Tenantid: 2, Locationid: 20, Productid: 220, Productname: "Britannia Marie Gold",
|
||||
Productbrand: "britannia", Catalogueid: 23, Imageid: "britannia_marie_gold", Price: 30, Stock: 4},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func TestABareBrandNameAsksInsteadOfGuessing(t *testing.T) {
|
||||
repo := newBrandLabelFixture()
|
||||
svc := NewScanService(repo, nil)
|
||||
|
||||
resp, err := svc.Lookup(context.Background(), models.ScanLookupRequest{
|
||||
Customerid: 5, Label: "britannia", Latitude: "11.035", Longitude: "77.035",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if !resp.Ambiguous {
|
||||
t.Fatalf("a bare brand name must not resolve to one product, got match %+v", resp.Match)
|
||||
}
|
||||
if resp.Match != nil {
|
||||
t.Errorf("Match must be nil while ambiguous, got %+v", resp.Match)
|
||||
}
|
||||
if len(resp.Stores) != 0 {
|
||||
t.Errorf("no store or price may be quoted for a product the customer has not chosen, got %d offers", len(resp.Stores))
|
||||
}
|
||||
if len(resp.Variants) != 0 {
|
||||
t.Errorf("pack sizes belong to a chosen product, got %+v", resp.Variants)
|
||||
}
|
||||
if len(resp.Candidates) != 3 {
|
||||
t.Fatalf("want the three distinct Britannia products, got %d: %+v", len(resp.Candidates), resp.Candidates)
|
||||
}
|
||||
|
||||
// The one the customer can actually buy is offered first.
|
||||
if !resp.Candidates[0].Available || resp.Candidates[0].Catalogueid != 23 {
|
||||
t.Errorf("the stocked product should lead the list, got %+v", resp.Candidates[0])
|
||||
}
|
||||
for _, c := range resp.Candidates[1:] {
|
||||
if c.Available {
|
||||
t.Errorf("only Marie Gold is stocked, but %s reports available", c.ProductName)
|
||||
}
|
||||
}
|
||||
if resp.Confidence >= 0.55 {
|
||||
t.Errorf("confidence should record how weak the identification was, got %v", resp.Confidence)
|
||||
}
|
||||
if !strings.Contains(resp.Message, "Which one") {
|
||||
t.Errorf("the message should ask, got %q", resp.Message)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnAmbiguousLabelWithNoStockStillLists(t *testing.T) {
|
||||
repo := newBrandLabelFixture()
|
||||
repo.options = nil
|
||||
svc := NewScanService(repo, nil)
|
||||
|
||||
resp, err := svc.Lookup(context.Background(), models.ScanLookupRequest{Customerid: 5, Label: "britannia"})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !resp.Ambiguous || len(resp.Candidates) != 3 {
|
||||
t.Fatalf("want three candidates, got ambiguous=%v %d", resp.Ambiguous, len(resp.Candidates))
|
||||
}
|
||||
for _, c := range resp.Candidates {
|
||||
if c.Available {
|
||||
t.Errorf("%s cannot be available with no stock anywhere", c.ProductName)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The other half of the contract: a label that does name a product must not
|
||||
// start asking questions.
|
||||
func TestASpecificLabelStillWinsOutright(t *testing.T) {
|
||||
repo := newBrandLabelFixture()
|
||||
svc := NewScanService(repo, nil)
|
||||
|
||||
resp, err := svc.Lookup(context.Background(), models.ScanLookupRequest{
|
||||
Customerid: 5, Label: "good day cashew",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if resp.Ambiguous {
|
||||
t.Fatalf("a label naming one product should resolve, got candidates %+v", resp.Candidates)
|
||||
}
|
||||
if resp.Match == nil || resp.Match.Catalogueid != 21 {
|
||||
t.Fatalf("want the cashew cookies, got %+v", resp.Match)
|
||||
}
|
||||
if len(resp.Candidates) != 0 {
|
||||
t.Errorf("candidates belong to an ambiguous answer, got %+v", resp.Candidates)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTextScoreRewardsSpecificityNotJustOverlap(t *testing.T) {
|
||||
cashew := repositories.CatalogueHit{ProductName: "Britannia Good Day Cashew Cookies 200g"}
|
||||
butter := repositories.CatalogueHit{ProductName: "Britannia Good Day Butter Cookies 100g"}
|
||||
|
||||
brandOnly := textScore(cashew, "britannia", utils.SearchTokens("britannia"))
|
||||
if brandOnly > 0.45 {
|
||||
t.Errorf("a brand name explains one word of five and must not score as an identification, got %.3f", brandOnly)
|
||||
}
|
||||
if brandOnly != textScore(butter, "britannia", utils.SearchTokens("britannia")) {
|
||||
t.Error("a brand must score its products equally — that tie is what makes the label read as ambiguous")
|
||||
}
|
||||
|
||||
full := textScore(cashew, "Britannia Good Day Cashew Cookies", utils.SearchTokens("Britannia Good Day Cashew Cookies"))
|
||||
if full < 0.95 {
|
||||
t.Errorf("the product's own name should be near-certain, got %.3f", full)
|
||||
}
|
||||
|
||||
// A pack size on either side is not a word the label failed to explain.
|
||||
sized := textScore(repositories.CatalogueHit{ProductName: "Milk Bikis 100g"}, "Milk Bikis", utils.SearchTokens("Milk Bikis"))
|
||||
if sized < 0.95 {
|
||||
t.Errorf("pack sizes must not count against the match, got %.3f", sized)
|
||||
}
|
||||
|
||||
if none := textScore(cashew, "dabur honey", utils.SearchTokens("dabur honey")); none != 0 {
|
||||
t.Errorf("nothing in common should score 0, got %.3f", none)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDistinctProductsCollapsesPackSizes(t *testing.T) {
|
||||
hits := []scoredHit{
|
||||
{CatalogueHit: milkBikis, score: 1},
|
||||
{CatalogueHit: milkBikis200, score: 0.9},
|
||||
{CatalogueHit: goodDay, score: 0.5},
|
||||
}
|
||||
distinct := distinctProducts(hits)
|
||||
if len(distinct) != 2 || distinct[0].ID != 7 || distinct[1].ID != 9 {
|
||||
t.Fatalf("two sizes of one product are one choice, got %+v", distinct)
|
||||
}
|
||||
}
|
||||
|
||||
// Picking a candidate: the app sends the key instead of a description, and
|
||||
// nothing is recognised at all.
|
||||
func TestNamingTheProductSkipsRecognition(t *testing.T) {
|
||||
repo := newLookupFixture()
|
||||
repo.ref = []repositories.CatalogueHit{milkBikis, milkBikis200}
|
||||
// If recognition ran, these would decide the answer instead.
|
||||
repo.text = []repositories.CatalogueHit{goodDay}
|
||||
repo.vector = []repositories.CatalogueHit{goodDay}
|
||||
svc := NewScanService(repo, fakeEmbedder{vec: []float32{0.1}})
|
||||
|
||||
resp, err := svc.Lookup(context.Background(), models.ScanLookupRequest{
|
||||
Customerid: 5, Brand: "britannia", Catalogueid: 7,
|
||||
Latitude: "11.035", Longitude: "77.035",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if repo.askedRef != "britannia#7" {
|
||||
t.Fatalf("the catalogue should have been asked for that exact product, got %q", repo.askedRef)
|
||||
}
|
||||
if resp.Match == nil || resp.Match.Catalogueid != 7 || resp.Match.Method != "direct" {
|
||||
t.Fatalf("want a direct match on 7, got %+v", resp.Match)
|
||||
}
|
||||
if resp.Ambiguous || resp.Confidence != 1 {
|
||||
t.Errorf("a named product is not a guess: ambiguous=%v confidence=%v", resp.Ambiguous, resp.Confidence)
|
||||
}
|
||||
if len(resp.Variants) != 2 {
|
||||
t.Errorf("its pack sizes should come with it, got %+v", resp.Variants)
|
||||
}
|
||||
if !resp.Available || resp.RecommendedLocationid != 20 {
|
||||
t.Errorf("stores are resolved exactly as for a recognised product, got %+v", resp.Stores)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAMissingCatalogueRefIsNotAMatch(t *testing.T) {
|
||||
repo := newLookupFixture()
|
||||
repo.ref = nil
|
||||
svc := NewScanService(repo, nil)
|
||||
|
||||
resp, err := svc.Lookup(context.Background(), models.ScanLookupRequest{
|
||||
Customerid: 5, Brand: "britannia", Catalogueid: 999,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if resp.Match != nil || resp.Ambiguous || len(resp.Stores) != 0 {
|
||||
t.Fatalf("a product that is gone is not a match, got %+v", resp)
|
||||
}
|
||||
if !strings.Contains(resp.Message, "no longer") {
|
||||
t.Errorf("the message should say the product is gone, got %q", resp.Message)
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user