Merge origin/main: keep the substring rule, read its tie

main had moved on with retrieval work validated against real queries —
minTokenHits (the word match needs two thirds of the label, not all of
it), separator folding so "Parle G"/"Parle-G"/"ParleG" all reach Parle-G,
the floor at 0.50 after "Paracetamol" came back as "Paneer Makhni 500ml"
at 0.304, and ties broken on cosine distance instead of name. All of that
is kept exactly as it was.

The conflict was in textScore: this branch replaced the substring rule
with a coverage formula to stop a bare brand name resolving to one
arbitrary product. That is the wrong half to change. The substring rule
scores every product of a brand 0.95 IDENTICALLY, and that tie is not the
bug — it is the signal. isAmbiguous reads it, so the branch's coverage
rewrite is dropped and the ambiguity layer alone does the work:

  "britannia" → all 258 rows tie at 0.95 → ambiguous: true + candidates
  "Parle G"   → folding and the single-character token still land it
  a real name → runner-up far behind → match, unchanged

Dropped with it: scanSpecificEnough, the per-hit text score, and the
proportional confirmation bonus — the flat +0.10 is back. Simpler, and it
leaves main's tuning untouched.

TestTextScoreRewardsSpecificityNotJustOverlap tested the removed formula
and is replaced by TestABrandNameScoresItsProductsIdentically, which
guards the tie itself: a formula that broke it on name length or word
count would bring the bug back.

Docs carry both rationales, and now say plainly that confidence stays
high on the ambiguous path — gate on `ambiguous`, never on `confidence`.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-09-23 11:05:03 +05:30
25 changed files with 1661 additions and 228 deletions

View File

@@ -453,11 +453,11 @@ Ambiguity.
Lens hands back whatever was most legible on the packet, and on a packet that
is very often the brand wordmark alone. "britannia" fits 258 catalogue rows
equally well, so there is no best one — and the old scoring said otherwise:
every product whose name contained the label scored 0.95, the tie broke
alphabetically, and the customer was shown one arbitrary biscuit with
"confidence": 0.95 and a price. These tests are the contract that it asks
instead.
equally well, so there is no best one. textScore gives all of them 0.95 —
correctly, the label IS in every one of those names — and with nothing to
read that tie, the sort order picked a winner and the customer was shown one
arbitrary biscuit with "confidence": 0.95 and a price. These tests are the
contract that it asks instead.
*/
// Three different Britannia products, of which the customer's stores stock
@@ -515,8 +515,12 @@ func TestABareBrandNameAsksInsteadOfGuessing(t *testing.T) {
t.Errorf("only Marie Gold is stocked, but %s reports available", c.ProductName)
}
}
if resp.Confidence >= 0.55 {
t.Errorf("confidence should record how weak the identification was, got %v", resp.Confidence)
// Note what confidence does NOT say here. The label appears verbatim in
// all three names, so relevance is high — and the answer is still a
// question. An app that gated on `confidence` instead of `ambiguous`
// would show a price for the wrong biscuit, which is the whole bug.
if resp.Confidence < 0.9 {
t.Errorf("a verbatim brand match scores high; %v suggests the scoring changed", resp.Confidence)
}
if !strings.Contains(resp.Message, "Which one") {
t.Errorf("the message should ask, got %q", resp.Message)
@@ -565,27 +569,31 @@ func TestASpecificLabelStillWinsOutright(t *testing.T) {
}
}
func TestTextScoreRewardsSpecificityNotJustOverlap(t *testing.T) {
// The property the ambiguity check rests on: a label that is a substring of
// several names scores them EQUALLY. Nothing downstream can tell "did you
// mean?" from "found it" if a formula breaks that tie on name length, word
// count or anything else incidental — which is how one arbitrary Britannia
// biscuit used to come back with a price on it.
func TestABrandNameScoresItsProductsIdentically(t *testing.T) {
cashew := repositories.CatalogueHit{ProductName: "Britannia Good Day Cashew Cookies 200g"}
butter := repositories.CatalogueHit{ProductName: "Britannia Good Day Butter Cookies 100g"}
// Deliberately a much shorter name: length must not become a tie-breaker.
marie := repositories.CatalogueHit{ProductName: "Britannia Marie Gold"}
brandOnly := textScore(cashew, "britannia", utils.SearchTokens("britannia"))
if brandOnly > 0.45 {
t.Errorf("a brand name explains one word of five and must not score as an identification, got %.3f", brandOnly)
tokens := utils.SearchTokens("britannia")
a, b, c := textScore(cashew, "britannia", tokens), textScore(butter, "britannia", tokens), textScore(marie, "britannia", tokens)
if a != b || b != c {
t.Fatalf("a brand must score its products equally, got %.3f / %.3f / %.3f", a, b, c)
}
if brandOnly != textScore(butter, "britannia", utils.SearchTokens("britannia")) {
t.Error("a brand must score its products equally — that tie is what makes the label read as ambiguous")
if a == 0 {
t.Fatal("the brand name is in every one of those names; scoring it 0 would hide them all")
}
full := textScore(cashew, "Britannia Good Day Cashew Cookies", utils.SearchTokens("Britannia Good Day Cashew Cookies"))
if full < 0.95 {
t.Errorf("the product's own name should be near-certain, got %.3f", full)
}
// A pack size on either side is not a word the label failed to explain.
sized := textScore(repositories.CatalogueHit{ProductName: "Milk Bikis 100g"}, "Milk Bikis", utils.SearchTokens("Milk Bikis"))
if sized < 0.95 {
t.Errorf("pack sizes must not count against the match, got %.3f", sized)
// And a label that does name a product must NOT tie with its siblings,
// or everything would be a question.
specific := utils.SearchTokens("good day cashew")
if textScore(cashew, "good day cashew", specific) <= textScore(butter, "good day cashew", specific) {
t.Error("a label naming one product must outscore its siblings")
}
if none := textScore(cashew, "dabur honey", utils.SearchTokens("dabur honey")); none != 0 {
@@ -657,3 +665,120 @@ func TestAMissingCatalogueRefIsNotAMatch(t *testing.T) {
t.Errorf("the message should say the product is gone, got %q", resp.Message)
}
}
// A vector neighbour that is merely not-quite-unrelated used to clear the old
// 0.30 floor: in production "Paracetamol" came back as "Paneer Makhni 500ml"
// on a 0.304 similarity. Correct labels land near 0.92, so nothing this weak
// is a match.
func TestLookupRefusesANearMissAboveTheOldFloor(t *testing.T) {
repo := newLookupFixture()
repo.vector = []repositories.CatalogueHit{{Brand: "amul", ID: 4, ProductName: "Paneer Makhni 500ml", Distance: 0.696}} // score 0.304
repo.text = nil
svc := NewScanService(repo, fakeEmbedder{vec: []float32{0.1}})
resp, err := svc.Lookup(context.Background(), models.ScanLookupRequest{Customerid: 5, Label: "Paracetamol"})
if err != nil {
t.Fatal(err)
}
if resp.Match != nil {
t.Fatalf("0.304 is a near-miss, not a match; got %+v", resp.Match)
}
if resp.Available || len(resp.Stores) != 0 {
t.Fatalf("nothing should be offered without a match; got %+v", resp)
}
}
// Confirm answers about the store the customer tapped, so that store carries a
// distance on every outcome — not only on the out-of-stock path that ranks
// alternatives. Absent any position it stays -1, the documented "unknown".
func TestConfirmReportsDistanceToTheChosenStore(t *testing.T) {
repo := newLookupFixture()
repo.at = map[int]*repositories.StoreOptionRow{200: &repo.options[1]}
svc := NewScanService(repo, nil)
req := models.ScanConfirmRequest{Customerid: 5, Tenantid: 2, Locationid: 20, Productid: 200, Quantity: 4}
withPos := req
withPos.Latitude, withPos.Longitude = "11.035", "77.035"
resp, err := svc.Confirm(context.Background(), withPos)
if err != nil {
t.Fatal(err)
}
if !resp.Ok || resp.Store == nil {
t.Fatalf("expected the in-stock answer, got %+v", resp)
}
if resp.Store.DistanceKm <= 0 {
t.Fatalf("the phone sent a fix, so the tapped store has a distance; got %v", resp.Store.DistanceKm)
}
// No fix from the phone, but a saved address on file.
repo.homeLat, repo.homeLng, repo.homeOK = 11.035, 77.035, true
resp, err = svc.Confirm(context.Background(), req)
if err != nil {
t.Fatal(err)
}
if resp.Store == nil || resp.Store.DistanceKm != -1 {
t.Fatalf("in stock is answered without reaching for the saved address; got %v", resp.Store)
}
// Neither: unknown, and the app sorts it last.
repo.homeOK = false
resp, err = svc.Confirm(context.Background(), req)
if err != nil {
t.Fatal(err)
}
if resp.Store == nil || resp.Store.DistanceKm != -1 {
t.Fatalf("no position at all is -1; got %v", resp.Store)
}
}
var parleG = repositories.CatalogueHit{Brand: "parle", ID: 1, ProductName: "Parle-G Original Glucose Biscuits 250g", Title: "Parle-G", VariantKey: "parle_g", ImageID: "parle_parle_g_250g", Distance: 0.20}
var monaco = repositories.CatalogueHit{Brand: "parle", ID: 2, ProductName: "Parle Monaco Classic Regular 200g", Title: "Monaco", VariantKey: "monaco", ImageID: "parle_monaco_200g", Distance: 0.20}
// Lens reads "Parle-G" off the packet and the customer types "Parle G". Both
// spellings, and the run-together one, have to reach the biscuit — not the
// salted cracker that merely shares a brand. In production "Parle G" returned
// "Parle Monaco Classic Regular 200g" at a confident 0.9.
func TestLookupMatchesAHyphenatedNameHoweverItIsWritten(t *testing.T) {
for _, label := range []string{"Parle G", "Parle-G", "ParleG", "parle g"} {
repo := newLookupFixture()
repo.vector = []repositories.CatalogueHit{monaco, parleG} // model puts the cracker first
repo.text = []repositories.CatalogueHit{monaco, parleG}
svc := NewScanService(repo, fakeEmbedder{vec: []float32{0.1}})
resp, err := svc.Lookup(context.Background(), models.ScanLookupRequest{Customerid: 5, Label: label})
if err != nil {
t.Fatal(err)
}
if resp.Match == nil {
t.Fatalf("%q: a stocked product went unrecognised", label)
}
if resp.Match.Catalogueid != parleG.ID {
t.Fatalf("%q: matched %q (%.3f), want Parle-G", label, resp.Match.ProductName, resp.Match.Score)
}
}
}
// Equal blended scores used to be settled by product name, which let ASCII
// decide relevance: a space sorts before a hyphen, so "Parle Monaco …" beat
// "Parle-G …". The model's own similarity settles it instead.
func TestSortHitsBreaksTiesOnSimilarityNotPunctuation(t *testing.T) {
near := parleG
near.Distance = 0.10 // the model is surer about this one
far := monaco
far.Distance = 0.40
hits := []scoredHit{{CatalogueHit: far, score: 0.9}, {CatalogueHit: near, score: 0.9}}
sortHits(hits)
if hits[0].ID != near.ID {
t.Fatalf("the nearer vector should win a tie, got %q", hits[0].ProductName)
}
// A row the model never scored (-1, text-only) ranks behind one it did.
textOnly := parleG
textOnly.Distance = -1
hits = []scoredHit{{CatalogueHit: textOnly, score: 0.9}, {CatalogueHit: far, score: 0.9}}
sortHits(hits)
if hits[0].ID != far.ID {
t.Fatalf("a scored row outranks an unscored one, got %q", hits[0].ProductName)
}
}