Merge origin/main: keep the substring rule, read its tie

main had moved on with retrieval work validated against real queries —
minTokenHits (the word match needs two thirds of the label, not all of
it), separator folding so "Parle G"/"Parle-G"/"ParleG" all reach Parle-G,
the floor at 0.50 after "Paracetamol" came back as "Paneer Makhni 500ml"
at 0.304, and ties broken on cosine distance instead of name. All of that
is kept exactly as it was.

The conflict was in textScore: this branch replaced the substring rule
with a coverage formula to stop a bare brand name resolving to one
arbitrary product. That is the wrong half to change. The substring rule
scores every product of a brand 0.95 IDENTICALLY, and that tie is not the
bug — it is the signal. isAmbiguous reads it, so the branch's coverage
rewrite is dropped and the ambiguity layer alone does the work:

  "britannia" → all 258 rows tie at 0.95 → ambiguous: true + candidates
  "Parle G"   → folding and the single-character token still land it
  a real name → runner-up far behind → match, unchanged

Dropped with it: scanSpecificEnough, the per-hit text score, and the
proportional confirmation bonus — the flat +0.10 is back. Simpler, and it
leaves main's tuning untouched.

TestTextScoreRewardsSpecificityNotJustOverlap tested the removed formula
and is replaced by TestABrandNameScoresItsProductsIdentically, which
guards the tie itself: a formula that broke it on name length or word
count would bring the bug back.

Docs carry both rationales, and now say plainly that confidence stays
high on the ambiguous path — gate on `ambiguous`, never on `confidence`.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-09-23 11:05:03 +05:30
25 changed files with 1661 additions and 228 deletions

View File

@@ -48,18 +48,19 @@ const (
scanLookupTimeout = 5 * time.Second
scanMaxLabelLen = 200
scanCatalogueTopK = 15
// Below this the best hit is not shown as a match at all.
scanMinScore = 0.30
// How close the runner-up has to be before the leader stops being an
// answer and the two become a question. See isAmbiguous.
// Below this the best hit is not shown as a match at all. A correct label
// scores ~0.92 against its own product's vector and ~0.23 against an
// unrelated one, so the floor sits in the empty middle of that split
// rather than just above the unrelated band: at 0.30, "Paracetamol"
// came back as "Paneer Makhni 500ml" (0.304) — a near-miss on an
// unrelated row clears a floor set that close to the noise.
scanMinScore = 0.50
// How close the runner-up may be before the leader stops being an answer
// and the two become a question. See isAmbiguous.
scanAmbiguityMargin = 0.06
// A "did you mean?" list longer than this is not a choice, it is a
// catalogue — the customer is standing in a shop holding a packet.
scanMaxCandidates = 10
// How much of the winning product's name the label has to account for
// before it counts as having identified it. See isAmbiguous and
// textScore.
scanSpecificEnough = 0.55
)
// ScanErrors the controller maps to statuses. Everything else is a 500.
@@ -401,6 +402,14 @@ func (s *scanService) Confirm(ctx context.Context, req models.ScanConfirmRequest
}
resp.Store = chosen
// Distance on the store they tapped, for every outcome and not just the
// out-of-stock one below: the app renders this store from the reply it
// gets. The phone's fix is free to parse; the saved address costs a
// query, so it is only reached for on the path that also ranks other
// outlets. Without either, distanceKm leaves the -1 the repository set.
lat, lng, hasPos := utils.ParseLatLng(string(req.Latitude), string(req.Longitude))
chosen.DistanceKm = distanceKm(*chosen, lat, lng, hasPos)
row, err := s.repo.ProductAt(ctx, req.Tenantid, req.Locationid, req.Productid)
if err != nil {
return nil, err
@@ -428,10 +437,10 @@ func (s *scanService) Confirm(ctx context.Context, req models.ScanConfirmRequest
}
// The same product elsewhere, nearest first, with enough of it.
lat, lng, hasPos := utils.ParseLatLng(string(req.Latitude), string(req.Longitude))
if !hasPos {
if hl, hg, ok, err := s.repo.CustomerHome(ctx, req.Customerid); err == nil && ok {
lat, lng, hasPos = hl, hg, true
chosen.DistanceKm = distanceKm(*chosen, lat, lng, hasPos)
}
}
others := make([]models.ScanStore, 0, len(stores))
@@ -516,12 +525,7 @@ func (s *scanService) Stores(ctx context.Context, customerid int, latStr, lngStr
type scoredHit struct {
repositories.CatalogueHit
// score ranks; text says how specifically the label names THIS product.
// Kept apart because they answer different questions: a vector neighbour
// can rank first while the label ("britannia") names no one product, and
// only the second number knows that.
score float64
text float64
}
func (h scoredHit) toMatch(method string) models.ScanCatalogueMatch {
@@ -554,12 +558,11 @@ func (s *scanService) searchCatalogue(ctx context.Context, label string) ([]scor
method = "vector+text"
}
tokens := utils.SearchTokens(label)
if cached, ok := s.repo.CachedHits(ctx, method+":"+s.modelName(), label); ok {
return scoreCachedHits(cached, label, tokens), method, nil
return scoreCachedHits(cached), method, nil
}
tokens := utils.SearchTokens(label)
byKey := make(map[string]*scoredHit)
keyOf := func(h repositories.CatalogueHit) string { return h.Brand + "#" + fmt.Sprint(h.ID) }
@@ -599,15 +602,10 @@ func (s *scanService) searchCatalogue(ctx context.Context, label string) ([]scor
for _, h := range thits {
ts := textScore(h, label, tokens)
if existing, ok := byKey[keyOf(h)]; ok {
// The bonus is proportional: only a text match that actually
// names the product confirms a vector hit. A flat +0.10 let a
// bare brand name — which matches every one of that brand's
// products weakly — inflate all of them equally.
existing.score = math.Min(1, math.Max(existing.score, ts)+0.10*ts)
existing.text = ts
existing.score = math.Min(1, math.Max(existing.score, ts)+0.10)
continue
}
byKey[keyOf(h)] = &scoredHit{CatalogueHit: h, score: ts, text: ts}
byKey[keyOf(h)] = &scoredHit{CatalogueHit: h, score: ts}
}
hits := make([]scoredHit, 0, len(byKey))
@@ -629,32 +627,42 @@ func (s *scanService) searchCatalogue(ctx context.Context, label string) ([]scor
return hits, method, nil
}
// scoreCachedHits restores the ranking the cache holds, and recomputes the
// text score from the row itself — the cache carries one number per row, and
// recomputing costs nothing while leaving out the specificity signal would
// make every cached lookup read as ambiguous.
func scoreCachedHits(cached []repositories.CatalogueHit, label string, tokens []string) []scoredHit {
func scoreCachedHits(cached []repositories.CatalogueHit) []scoredHit {
hits := make([]scoredHit, 0, len(cached))
for _, c := range cached {
hits = append(hits, scoredHit{
CatalogueHit: c,
score: 1 - c.Distance,
text: textScore(c, label, tokens),
})
hits = append(hits, scoredHit{CatalogueHit: c, score: 1 - c.Distance})
}
sortHits(hits)
return hits
}
// sortHits ranks by blended score, then by the model's own similarity, and
// only then by name. Name alone used to break every tie, which quietly made
// punctuation decide relevance: "Parle Monaco Classic" sorts above "Parle-G
// Original …" because a space precedes a hyphen in ASCII, so equal-scoring
// crackers beat the biscuit that was actually scanned.
func sortHits(hits []scoredHit) {
sort.SliceStable(hits, func(i, j int) bool {
if hits[i].score != hits[j].score {
return hits[i].score > hits[j].score
}
di, dj := vectorRank(hits[i].Distance), vectorRank(hits[j].Distance)
if di != dj {
return di < dj
}
return hits[i].ProductName < hits[j].ProductName
})
}
// vectorRank orders by cosine distance, nearest first, with a row the model
// never saw (-1, text-only) sorting behind every row it did.
func vectorRank(d float64) float64 {
if d < 0 {
return math.MaxFloat64
}
return d
}
func (s *scanService) modelName() string {
if s.embedder == nil {
return "none"
@@ -674,79 +682,44 @@ func (s *scanService) embed(ctx context.Context, label string) ([]float32, error
return v, nil
}
// textScore is how well a catalogue row matches the words Lens read.
// textScore is how well a catalogue row's name matches the words Lens read.
// The whole label as a substring of the name is near-certain; otherwise the
// share of label words found in name+title, scaled so that "all of them"
// stops short of the substring case.
//
// Both directions count, and that is the whole point:
//
// - labelCoverage — how much of what the customer said this product
// accounts for. "Milk Bikis" against "Milk Bikis 100g" is all of it.
// - nameCoverage — how much of the product the label accounts for, which
// is what makes the match SPECIFIC. "britannia" explains one word of
// "Britannia Good Day Cashew Cookies", so it does not identify it.
//
// The score is their harmonic mean, so a high score needs both.
//
// This replaces `strings.Contains(name, label) → 0.95`, which asked only the
// first question. A bare brand name is a substring of every one of that
// brand's products, so all 258 Britannia rows scored 0.95, the tie broke
// alphabetically, and the customer was shown one arbitrary biscuit with
// "confidence": 0.95. Lens returns a bare wordmark often — it is usually the
// most legible thing on a packet — so that was not an edge case.
//
// Now those rows score ~0.33 and, crucially, score it EQUALLY, which is what
// isAmbiguous reads to answer "did you mean?" instead of guessing.
// A label that is a substring of MANY names — a bare brand, "britannia" —
// therefore scores them all 0.95, identically. That tie is not a flaw to
// score around: it is the signal, and isAmbiguous reads it to answer "did
// you mean?" rather than letting the sort order pick a winner.
func textScore(h repositories.CatalogueHit, label string, tokens []string) float64 {
name := strings.ToLower(h.ProductName)
hay := name + " " + strings.ToLower(h.Title)
label = strings.ToLower(strings.TrimSpace(label))
// The substring test compares separator-folded forms, so the brand's own
// punctuation does not decide the match: "Parle G", "Parle-G" and
// "ParleG" all have to reach "Parle-G Original Glucose Biscuits".
if label != "" {
foldedName, foldedLabel := utils.FoldSeparators(name), utils.FoldSeparators(label)
if foldedLabel != "" && strings.Contains(foldedName, foldedLabel) {
return 0.95
}
// Separators dropped rather than folded. Only for a label long enough
// that a run of letters means something — "lay" inside "malayalam" is
// not a match anyone wants.
if tight := utils.TightenLabel(label); len(tight) >= 4 && strings.Contains(utils.TightenLabel(name), tight) {
return 0.95
}
}
if len(tokens) == 0 {
return 0
}
hay := strings.ToLower(h.ProductName + " " + h.Title)
found := 0
for _, t := range tokens {
if strings.Contains(hay, t) {
found++
}
}
if found == 0 {
return 0
}
labelCoverage := float64(found) / float64(len(tokens))
// Pack sizes are dropped from both sides (SearchTokens), so "100g" never
// counts as a word the label failed to explain.
nameTokens := utils.SearchTokens(h.ProductName)
if len(nameTokens) == 0 {
return 0.5 * labelCoverage
}
explained := 0
for _, n := range nameTokens {
for _, t := range tokens {
if tokenMatch(n, t) {
explained++
break
}
}
}
if explained == 0 {
// Matched the title but not the name. Weak, not zero.
return 0.4 * labelCoverage
}
nameCoverage := float64(explained) / float64(len(nameTokens))
return 2 * labelCoverage * nameCoverage / (labelCoverage + nameCoverage)
}
// tokenMatch is equality, plus containment for words long enough that a
// shared prefix means something ("cookie"/"cookies", "chocolate"/"choco").
// Short tokens must match exactly, or "day" would match "daybreak".
func tokenMatch(a, b string) bool {
if a == b {
return true
}
if len(a) >= 5 && strings.Contains(b, a) {
return true
}
return len(b) >= 5 && strings.Contains(a, b)
return 0.8 * float64(found) / float64(len(tokens))
}
// productKey identifies a product across its pack sizes: the catalogue's own
@@ -781,28 +754,19 @@ func distinctProducts(hits []scoredHit) []scoredHit {
}
// isAmbiguous reports that naming the leader as THE match would be a guess
// dressed up as an answer. Two ways that happens:
// dressed up as an answer, because something else is level with it.
//
// 1. Something else is level with it. A margin rather than an absolute
// threshold, because what matters is not how high the best score is but
// whether anything is tied with it.
// 2. Nothing is level, but the label does not actually name a product —
// a bare brand, a generic word, or a spelling the catalogue does not
// carry. The leader may still rank first on vector similarity, and
// ranking first among vague matches is not identification.
// A margin rather than an absolute threshold: what matters is not how high
// the best score is but whether anything is tied with it. A bare brand name
// is a substring of every one of that brand's names, so textScore gives them
// all 0.95 — a perfect tie at a HIGH score, which no floor would catch.
//
// Erring towards asking is deliberate. Asking costs the customer one tap on
// a picture; guessing wrong costs them the wrong biscuit and costs us the
// belief that the scanner works. An exact product name still scores ~1.0 on
// specificity, so the common case is unaffected.
// belief that the scanner works. A label that names one product leaves the
// runner-up far behind, so the common case is unaffected.
func isAmbiguous(distinct []scoredHit) bool {
if len(distinct) < 2 {
return false
}
if distinct[1].score >= distinct[0].score-scanAmbiguityMargin {
return true
}
return distinct[0].text < scanSpecificEnough
return len(distinct) >= 2 && distinct[1].score >= distinct[0].score-scanAmbiguityMargin
}
// catalogueFamily is `of` and its other pack sizes, drawn from hits.