From 28af3e05f23168adb4870ad62b706190b9c67eec Mon Sep 17 00:00:00 2001 From: abhishek Date: Wed, 16 Sep 2026 11:52:35 +0530 Subject: [PATCH] image search test --- docs/SCAN_TO_ORDER.md | 18 +++++++++++++- services/scanService.go | 34 +++++++++++++++++++++++++-- services/scan_test.go | 52 +++++++++++++++++++++++++++++++++++++++++ utils/geo.go | 44 +++++++++++++++++++++++++++++++--- utils/geo_test.go | 39 +++++++++++++++++++++++++++++++ 5 files changed, 181 insertions(+), 6 deletions(-) diff --git a/docs/SCAN_TO_ORDER.md b/docs/SCAN_TO_ORDER.md index 452a0d2..030006b 100644 --- a/docs/SCAN_TO_ORDER.md +++ b/docs/SCAN_TO_ORDER.md @@ -244,7 +244,23 @@ you change ranking — it is the spec. Scores: vector = `1 − cosine distance`; text = 0.95 for the whole label inside the name, else `0.8 × (label words found / label words)`; combined = -`max(vector, text) + 0.10` when both hit, capped at 1. +`max(vector, text) + 0.10` when both hit, capped at 1. Ties are broken by +cosine distance — nearest first, a text-only row last — and only then by +name. + +The label and the product name are both separator-folded before that +substring test (`utils.FoldSeparators`), and compared again with separators +removed (`utils.TightenLabel`, labels of 4+ characters), so the brand's own +punctuation does not decide the match: "Parle G", "Parle-G" and "ParleG" all +reach *Parle-G Original Glucose Biscuits*. A single-character token survives +tokenising when it follows a word, because it is often the whole name — the +"G" of Parle-G, the "K" of Special K. It is still dropped when it stands +alone or is a pack multiplier. + +All three mattered at once: before this, "Parle G" tied with *Parle Monaco +Classic* at 0.9 (the "G" was dropped, so only "parle" matched either row), +and the name tie-break handed it to Monaco because a space precedes a hyphen +in ASCII. A confident, wrong answer — the kind no score floor can catch. ### Changing the embedding model diff --git a/services/scanService.go b/services/scanService.go index da3b837..954fa8c 100644 --- a/services/scanService.go +++ b/services/scanService.go @@ -480,15 +480,33 @@ func scoreCachedHits(cached []repositories.CatalogueHit) []scoredHit { return hits } +// sortHits ranks by blended score, then by the model's own similarity, and +// only then by name. Name alone used to break every tie, which quietly made +// punctuation decide relevance: "Parle Monaco Classic" sorts above "Parle-G +// Original …" because a space precedes a hyphen in ASCII, so equal-scoring +// crackers beat the biscuit that was actually scanned. func sortHits(hits []scoredHit) { sort.SliceStable(hits, func(i, j int) bool { if hits[i].score != hits[j].score { return hits[i].score > hits[j].score } + di, dj := vectorRank(hits[i].Distance), vectorRank(hits[j].Distance) + if di != dj { + return di < dj + } return hits[i].ProductName < hits[j].ProductName }) } +// vectorRank orders by cosine distance, nearest first, with a row the model +// never saw (-1, text-only) sorting behind every row it did. +func vectorRank(d float64) float64 { + if d < 0 { + return math.MaxFloat64 + } + return d +} + func (s *scanService) modelName() string { if s.embedder == nil { return "none" @@ -516,8 +534,20 @@ func textScore(h repositories.CatalogueHit, label string, tokens []string) float name := strings.ToLower(h.ProductName) hay := name + " " + strings.ToLower(h.Title) label = strings.ToLower(strings.TrimSpace(label)) - if label != "" && strings.Contains(name, label) { - return 0.95 + // The substring test compares separator-folded forms, so the brand's own + // punctuation does not decide the match: "Parle G", "Parle-G" and + // "ParleG" all have to reach "Parle-G Original Glucose Biscuits". + if label != "" { + foldedName, foldedLabel := utils.FoldSeparators(name), utils.FoldSeparators(label) + if foldedLabel != "" && strings.Contains(foldedName, foldedLabel) { + return 0.95 + } + // Separators dropped rather than folded. Only for a label long enough + // that a run of letters means something — "lay" inside "malayalam" is + // not a match anyone wants. + if tight := utils.TightenLabel(label); len(tight) >= 4 && strings.Contains(utils.TightenLabel(name), tight) { + return 0.95 + } } if len(tokens) == 0 { return 0 diff --git a/services/scan_test.go b/services/scan_test.go index 7df6144..7a41043 100644 --- a/services/scan_test.go +++ b/services/scan_test.go @@ -501,3 +501,55 @@ func TestConfirmReportsDistanceToTheChosenStore(t *testing.T) { t.Fatalf("no position at all is -1; got %v", resp.Store) } } + +var parleG = repositories.CatalogueHit{Brand: "parle", ID: 1, ProductName: "Parle-G Original Glucose Biscuits 250g", Title: "Parle-G", VariantKey: "parle_g", ImageID: "parle_parle_g_250g", Distance: 0.20} +var monaco = repositories.CatalogueHit{Brand: "parle", ID: 2, ProductName: "Parle Monaco Classic Regular 200g", Title: "Monaco", VariantKey: "monaco", ImageID: "parle_monaco_200g", Distance: 0.20} + +// Lens reads "Parle-G" off the packet and the customer types "Parle G". Both +// spellings, and the run-together one, have to reach the biscuit — not the +// salted cracker that merely shares a brand. In production "Parle G" returned +// "Parle Monaco Classic Regular 200g" at a confident 0.9. +func TestLookupMatchesAHyphenatedNameHoweverItIsWritten(t *testing.T) { + for _, label := range []string{"Parle G", "Parle-G", "ParleG", "parle g"} { + repo := newLookupFixture() + repo.vector = []repositories.CatalogueHit{monaco, parleG} // model puts the cracker first + repo.text = []repositories.CatalogueHit{monaco, parleG} + svc := NewScanService(repo, fakeEmbedder{vec: []float32{0.1}}) + + resp, err := svc.Lookup(context.Background(), models.ScanLookupRequest{Customerid: 5, Label: label}) + if err != nil { + t.Fatal(err) + } + if resp.Match == nil { + t.Fatalf("%q: a stocked product went unrecognised", label) + } + if resp.Match.Catalogueid != parleG.ID { + t.Fatalf("%q: matched %q (%.3f), want Parle-G", label, resp.Match.ProductName, resp.Match.Score) + } + } +} + +// Equal blended scores used to be settled by product name, which let ASCII +// decide relevance: a space sorts before a hyphen, so "Parle Monaco …" beat +// "Parle-G …". The model's own similarity settles it instead. +func TestSortHitsBreaksTiesOnSimilarityNotPunctuation(t *testing.T) { + near := parleG + near.Distance = 0.10 // the model is surer about this one + far := monaco + far.Distance = 0.40 + + hits := []scoredHit{{CatalogueHit: far, score: 0.9}, {CatalogueHit: near, score: 0.9}} + sortHits(hits) + if hits[0].ID != near.ID { + t.Fatalf("the nearer vector should win a tie, got %q", hits[0].ProductName) + } + + // A row the model never scored (-1, text-only) ranks behind one it did. + textOnly := parleG + textOnly.Distance = -1 + hits = []scoredHit{{CatalogueHit: textOnly, score: 0.9}, {CatalogueHit: far, score: 0.9}} + sortHits(hits) + if hits[0].ID != far.ID { + t.Fatalf("a scored row outranks an unscored one, got %q", hits[0].ProductName) + } +} diff --git a/utils/geo.go b/utils/geo.go index cc0a44c..3658e60 100644 --- a/utils/geo.go +++ b/utils/geo.go @@ -69,23 +69,61 @@ func parseClock(s string) (int, bool) { } // SearchTokens splits a label into the words worth matching on: lowercased, -// punctuation stripped, single characters and pack-size noise dropped. "Milk -// Bikis 100g" → ["milk", "bikis"]; the size is matched separately, if at all. +// punctuation stripped, pack-size noise dropped. "Milk Bikis 100g" → +// ["milk", "bikis"]; the size is matched separately, if at all. +// +// A single character is kept when it follows a word, because in this +// catalogue that character is often the whole product: the "G" in "Parle G", +// the "K" in "Special K". Dropping it made "Parle G" score the same against +// "Parle-G Original Glucose Biscuits" as against "Parle Monaco Classic", and +// the tie went to Monaco. It is still dropped when it stands alone — a +// one-letter label is not a search — and bare multipliers ("2 x 50gm") are +// never words. func SearchTokens(label string) []string { var tokens []string seen := make(map[string]bool) + kept := 0 // multi-character tokens so far: a lone letter needs one for _, raw := range strings.FieldsFunc(strings.ToLower(label), func(r rune) bool { return !(r >= 'a' && r <= 'z' || r >= '0' && r <= '9') }) { - if len(raw) < 2 || isPackSize(raw) || seen[raw] { + if isPackSize(raw) || seen[raw] { continue } + if len(raw) < 2 && (kept == 0 || isMultiplier(raw)) { + continue + } + if len(raw) > 1 { + kept++ + } seen[raw] = true tokens = append(tokens, raw) } return tokens } +// isMultiplier is the "x" of "2 x 50gm" and the "n" of a multipack — a single +// character that joins sizes rather than naming a product. +func isMultiplier(tok string) bool { + return tok == "x" || tok == "n" +} + +// FoldSeparators turns every run of punctuation into one space, so a label +// typed without the brand's own punctuation still matches it: "Parle G" and +// "Parle-G" both fold to "parle g". Lens reads letterforms off a packet, and +// people type what they see, so the hyphen is not reliably either present or +// absent on the way in. +func FoldSeparators(s string) string { + return strings.Join(strings.FieldsFunc(strings.ToLower(s), func(r rune) bool { + return !(r >= 'a' && r <= 'z' || r >= '0' && r <= '9') + }), " ") +} + +// TightenLabel removes separators outright rather than folding them, catching +// the other way people write a hyphenated name: "ParleG" against "Parle-G". +func TightenLabel(s string) string { + return strings.ReplaceAll(FoldSeparators(s), " ", "") +} + // isPackSize is "100g", "1kg", "500ml", "2l", "250gm" — a number with a unit // glued on, or a bare number. func isPackSize(tok string) bool { diff --git a/utils/geo_test.go b/utils/geo_test.go index eb7c012..6740235 100644 --- a/utils/geo_test.go +++ b/utils/geo_test.go @@ -63,3 +63,42 @@ func TestSearchTokens(t *testing.T) { t.Error("pack sizes alone are not searchable") } } + +// A single letter is often the whole product name in this catalogue, so it +// survives when it follows a word — but not when it stands alone, and not +// when it is the multiplier in a pack size. +func TestSearchTokensKeepsALetterThatFollowsAWord(t *testing.T) { + for _, tc := range []struct { + label string + want []string + }{ + {"Parle G", []string{"parle", "g"}}, + {"Parle-G", []string{"parle", "g"}}, + {"Special K Original", []string{"special", "k", "original"}}, + {"G", nil}, // a letter alone is not a search + {"2 x 50gm", nil}, // multiplier and pack size, no words + {"Milk Bikis 100g, Britannia (2 x 50gm)", []string{"milk", "bikis", "britannia"}}, + } { + got := SearchTokens(tc.label) + if len(got) != len(tc.want) { + t.Fatalf("%q: got %v want %v", tc.label, got, tc.want) + } + for i := range tc.want { + if got[i] != tc.want[i] { + t.Fatalf("%q: got %v want %v", tc.label, got, tc.want) + } + } + } +} + +func TestFoldAndTightenSeparators(t *testing.T) { + if got := FoldSeparators("Parle-G Original"); got != "parle g original" { + t.Fatalf("fold: got %q", got) + } + if FoldSeparators("Parle-G") != FoldSeparators("Parle G") { + t.Error("a hyphen and a space are the same separator to us") + } + if got := TightenLabel("Parle-G"); got != "parleg" { + t.Fatalf("tighten: got %q", got) + } +}