image search test

This commit is contained in:
2026-09-16 11:52:35 +05:30
parent 42ea007fe7
commit 28af3e05f2
5 changed files with 181 additions and 6 deletions

View File

@@ -244,7 +244,23 @@ you change ranking — it is the spec.
Scores: vector = `1 − cosine distance`; text = 0.95 for the whole label Scores: vector = `1 − cosine distance`; text = 0.95 for the whole label
inside the name, else `0.8 × (label words found / label words)`; combined = inside the name, else `0.8 × (label words found / label words)`; combined =
`max(vector, text) + 0.10` when both hit, capped at 1. `max(vector, text) + 0.10` when both hit, capped at 1. Ties are broken by
cosine distance — nearest first, a text-only row last — and only then by
name.
The label and the product name are both separator-folded before that
substring test (`utils.FoldSeparators`), and compared again with separators
removed (`utils.TightenLabel`, labels of 4+ characters), so the brand's own
punctuation does not decide the match: "Parle G", "Parle-G" and "ParleG" all
reach *Parle-G Original Glucose Biscuits*. A single-character token survives
tokenising when it follows a word, because it is often the whole name — the
"G" of Parle-G, the "K" of Special K. It is still dropped when it stands
alone or is a pack multiplier.
All three mattered at once: before this, "Parle G" tied with *Parle Monaco
Classic* at 0.9 (the "G" was dropped, so only "parle" matched either row),
and the name tie-break handed it to Monaco because a space precedes a hyphen
in ASCII. A confident, wrong answer — the kind no score floor can catch.
### Changing the embedding model ### Changing the embedding model

View File

@@ -480,15 +480,33 @@ func scoreCachedHits(cached []repositories.CatalogueHit) []scoredHit {
return hits return hits
} }
// sortHits ranks by blended score, then by the model's own similarity, and
// only then by name. Name alone used to break every tie, which quietly made
// punctuation decide relevance: "Parle Monaco Classic" sorts above "Parle-G
// Original …" because a space precedes a hyphen in ASCII, so equal-scoring
// crackers beat the biscuit that was actually scanned.
func sortHits(hits []scoredHit) { func sortHits(hits []scoredHit) {
sort.SliceStable(hits, func(i, j int) bool { sort.SliceStable(hits, func(i, j int) bool {
if hits[i].score != hits[j].score { if hits[i].score != hits[j].score {
return hits[i].score > hits[j].score return hits[i].score > hits[j].score
} }
di, dj := vectorRank(hits[i].Distance), vectorRank(hits[j].Distance)
if di != dj {
return di < dj
}
return hits[i].ProductName < hits[j].ProductName return hits[i].ProductName < hits[j].ProductName
}) })
} }
// vectorRank orders by cosine distance, nearest first, with a row the model
// never saw (-1, text-only) sorting behind every row it did.
func vectorRank(d float64) float64 {
if d < 0 {
return math.MaxFloat64
}
return d
}
func (s *scanService) modelName() string { func (s *scanService) modelName() string {
if s.embedder == nil { if s.embedder == nil {
return "none" return "none"
@@ -516,8 +534,20 @@ func textScore(h repositories.CatalogueHit, label string, tokens []string) float
name := strings.ToLower(h.ProductName) name := strings.ToLower(h.ProductName)
hay := name + " " + strings.ToLower(h.Title) hay := name + " " + strings.ToLower(h.Title)
label = strings.ToLower(strings.TrimSpace(label)) label = strings.ToLower(strings.TrimSpace(label))
if label != "" && strings.Contains(name, label) { // The substring test compares separator-folded forms, so the brand's own
return 0.95 // punctuation does not decide the match: "Parle G", "Parle-G" and
// "ParleG" all have to reach "Parle-G Original Glucose Biscuits".
if label != "" {
foldedName, foldedLabel := utils.FoldSeparators(name), utils.FoldSeparators(label)
if foldedLabel != "" && strings.Contains(foldedName, foldedLabel) {
return 0.95
}
// Separators dropped rather than folded. Only for a label long enough
// that a run of letters means something — "lay" inside "malayalam" is
// not a match anyone wants.
if tight := utils.TightenLabel(label); len(tight) >= 4 && strings.Contains(utils.TightenLabel(name), tight) {
return 0.95
}
} }
if len(tokens) == 0 { if len(tokens) == 0 {
return 0 return 0

View File

@@ -501,3 +501,55 @@ func TestConfirmReportsDistanceToTheChosenStore(t *testing.T) {
t.Fatalf("no position at all is -1; got %v", resp.Store) t.Fatalf("no position at all is -1; got %v", resp.Store)
} }
} }
var parleG = repositories.CatalogueHit{Brand: "parle", ID: 1, ProductName: "Parle-G Original Glucose Biscuits 250g", Title: "Parle-G", VariantKey: "parle_g", ImageID: "parle_parle_g_250g", Distance: 0.20}
var monaco = repositories.CatalogueHit{Brand: "parle", ID: 2, ProductName: "Parle Monaco Classic Regular 200g", Title: "Monaco", VariantKey: "monaco", ImageID: "parle_monaco_200g", Distance: 0.20}
// Lens reads "Parle-G" off the packet and the customer types "Parle G". Both
// spellings, and the run-together one, have to reach the biscuit — not the
// salted cracker that merely shares a brand. In production "Parle G" returned
// "Parle Monaco Classic Regular 200g" at a confident 0.9.
func TestLookupMatchesAHyphenatedNameHoweverItIsWritten(t *testing.T) {
for _, label := range []string{"Parle G", "Parle-G", "ParleG", "parle g"} {
repo := newLookupFixture()
repo.vector = []repositories.CatalogueHit{monaco, parleG} // model puts the cracker first
repo.text = []repositories.CatalogueHit{monaco, parleG}
svc := NewScanService(repo, fakeEmbedder{vec: []float32{0.1}})
resp, err := svc.Lookup(context.Background(), models.ScanLookupRequest{Customerid: 5, Label: label})
if err != nil {
t.Fatal(err)
}
if resp.Match == nil {
t.Fatalf("%q: a stocked product went unrecognised", label)
}
if resp.Match.Catalogueid != parleG.ID {
t.Fatalf("%q: matched %q (%.3f), want Parle-G", label, resp.Match.ProductName, resp.Match.Score)
}
}
}
// Equal blended scores used to be settled by product name, which let ASCII
// decide relevance: a space sorts before a hyphen, so "Parle Monaco …" beat
// "Parle-G …". The model's own similarity settles it instead.
func TestSortHitsBreaksTiesOnSimilarityNotPunctuation(t *testing.T) {
near := parleG
near.Distance = 0.10 // the model is surer about this one
far := monaco
far.Distance = 0.40
hits := []scoredHit{{CatalogueHit: far, score: 0.9}, {CatalogueHit: near, score: 0.9}}
sortHits(hits)
if hits[0].ID != near.ID {
t.Fatalf("the nearer vector should win a tie, got %q", hits[0].ProductName)
}
// A row the model never scored (-1, text-only) ranks behind one it did.
textOnly := parleG
textOnly.Distance = -1
hits = []scoredHit{{CatalogueHit: textOnly, score: 0.9}, {CatalogueHit: far, score: 0.9}}
sortHits(hits)
if hits[0].ID != far.ID {
t.Fatalf("a scored row outranks an unscored one, got %q", hits[0].ProductName)
}
}

View File

@@ -69,23 +69,61 @@ func parseClock(s string) (int, bool) {
} }
// SearchTokens splits a label into the words worth matching on: lowercased, // SearchTokens splits a label into the words worth matching on: lowercased,
// punctuation stripped, single characters and pack-size noise dropped. "Milk // punctuation stripped, pack-size noise dropped. "Milk Bikis 100g" →
// Bikis 100g" → ["milk", "bikis"]; the size is matched separately, if at all. // ["milk", "bikis"]; the size is matched separately, if at all.
//
// A single character is kept when it follows a word, because in this
// catalogue that character is often the whole product: the "G" in "Parle G",
// the "K" in "Special K". Dropping it made "Parle G" score the same against
// "Parle-G Original Glucose Biscuits" as against "Parle Monaco Classic", and
// the tie went to Monaco. It is still dropped when it stands alone — a
// one-letter label is not a search — and bare multipliers ("2 x 50gm") are
// never words.
func SearchTokens(label string) []string { func SearchTokens(label string) []string {
var tokens []string var tokens []string
seen := make(map[string]bool) seen := make(map[string]bool)
kept := 0 // multi-character tokens so far: a lone letter needs one
for _, raw := range strings.FieldsFunc(strings.ToLower(label), func(r rune) bool { for _, raw := range strings.FieldsFunc(strings.ToLower(label), func(r rune) bool {
return !(r >= 'a' && r <= 'z' || r >= '0' && r <= '9') return !(r >= 'a' && r <= 'z' || r >= '0' && r <= '9')
}) { }) {
if len(raw) < 2 || isPackSize(raw) || seen[raw] { if isPackSize(raw) || seen[raw] {
continue continue
} }
if len(raw) < 2 && (kept == 0 || isMultiplier(raw)) {
continue
}
if len(raw) > 1 {
kept++
}
seen[raw] = true seen[raw] = true
tokens = append(tokens, raw) tokens = append(tokens, raw)
} }
return tokens return tokens
} }
// isMultiplier is the "x" of "2 x 50gm" and the "n" of a multipack — a single
// character that joins sizes rather than naming a product.
func isMultiplier(tok string) bool {
return tok == "x" || tok == "n"
}
// FoldSeparators turns every run of punctuation into one space, so a label
// typed without the brand's own punctuation still matches it: "Parle G" and
// "Parle-G" both fold to "parle g". Lens reads letterforms off a packet, and
// people type what they see, so the hyphen is not reliably either present or
// absent on the way in.
func FoldSeparators(s string) string {
return strings.Join(strings.FieldsFunc(strings.ToLower(s), func(r rune) bool {
return !(r >= 'a' && r <= 'z' || r >= '0' && r <= '9')
}), " ")
}
// TightenLabel removes separators outright rather than folding them, catching
// the other way people write a hyphenated name: "ParleG" against "Parle-G".
func TightenLabel(s string) string {
return strings.ReplaceAll(FoldSeparators(s), " ", "")
}
// isPackSize is "100g", "1kg", "500ml", "2l", "250gm" — a number with a unit // isPackSize is "100g", "1kg", "500ml", "2l", "250gm" — a number with a unit
// glued on, or a bare number. // glued on, or a bare number.
func isPackSize(tok string) bool { func isPackSize(tok string) bool {

View File

@@ -63,3 +63,42 @@ func TestSearchTokens(t *testing.T) {
t.Error("pack sizes alone are not searchable") t.Error("pack sizes alone are not searchable")
} }
} }
// A single letter is often the whole product name in this catalogue, so it
// survives when it follows a word — but not when it stands alone, and not
// when it is the multiplier in a pack size.
func TestSearchTokensKeepsALetterThatFollowsAWord(t *testing.T) {
for _, tc := range []struct {
label string
want []string
}{
{"Parle G", []string{"parle", "g"}},
{"Parle-G", []string{"parle", "g"}},
{"Special K Original", []string{"special", "k", "original"}},
{"G", nil}, // a letter alone is not a search
{"2 x 50gm", nil}, // multiplier and pack size, no words
{"Milk Bikis 100g, Britannia (2 x 50gm)", []string{"milk", "bikis", "britannia"}},
} {
got := SearchTokens(tc.label)
if len(got) != len(tc.want) {
t.Fatalf("%q: got %v want %v", tc.label, got, tc.want)
}
for i := range tc.want {
if got[i] != tc.want[i] {
t.Fatalf("%q: got %v want %v", tc.label, got, tc.want)
}
}
}
}
func TestFoldAndTightenSeparators(t *testing.T) {
if got := FoldSeparators("Parle-G Original"); got != "parle g original" {
t.Fatalf("fold: got %q", got)
}
if FoldSeparators("Parle-G") != FoldSeparators("Parle G") {
t.Error("a hyphen and a space are the same separator to us")
}
if got := TightenLabel("Parle-G"); got != "parleg" {
t.Fatalf("tighten: got %q", got)
}
}