image search test
This commit is contained in:
@@ -244,7 +244,23 @@ you change ranking — it is the spec.
|
|||||||
|
|
||||||
Scores: vector = `1 − cosine distance`; text = 0.95 for the whole label
|
Scores: vector = `1 − cosine distance`; text = 0.95 for the whole label
|
||||||
inside the name, else `0.8 × (label words found / label words)`; combined =
|
inside the name, else `0.8 × (label words found / label words)`; combined =
|
||||||
`max(vector, text) + 0.10` when both hit, capped at 1.
|
`max(vector, text) + 0.10` when both hit, capped at 1. Ties are broken by
|
||||||
|
cosine distance — nearest first, a text-only row last — and only then by
|
||||||
|
name.
|
||||||
|
|
||||||
|
The label and the product name are both separator-folded before that
|
||||||
|
substring test (`utils.FoldSeparators`), and compared again with separators
|
||||||
|
removed (`utils.TightenLabel`, labels of 4+ characters), so the brand's own
|
||||||
|
punctuation does not decide the match: "Parle G", "Parle-G" and "ParleG" all
|
||||||
|
reach *Parle-G Original Glucose Biscuits*. A single-character token survives
|
||||||
|
tokenising when it follows a word, because it is often the whole name — the
|
||||||
|
"G" of Parle-G, the "K" of Special K. It is still dropped when it stands
|
||||||
|
alone or is a pack multiplier.
|
||||||
|
|
||||||
|
All three mattered at once: before this, "Parle G" tied with *Parle Monaco
|
||||||
|
Classic* at 0.9 (the "G" was dropped, so only "parle" matched either row),
|
||||||
|
and the name tie-break handed it to Monaco because a space precedes a hyphen
|
||||||
|
in ASCII. A confident, wrong answer — the kind no score floor can catch.
|
||||||
|
|
||||||
### Changing the embedding model
|
### Changing the embedding model
|
||||||
|
|
||||||
|
|||||||
@@ -480,15 +480,33 @@ func scoreCachedHits(cached []repositories.CatalogueHit) []scoredHit {
|
|||||||
return hits
|
return hits
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// sortHits ranks by blended score, then by the model's own similarity, and
|
||||||
|
// only then by name. Name alone used to break every tie, which quietly made
|
||||||
|
// punctuation decide relevance: "Parle Monaco Classic" sorts above "Parle-G
|
||||||
|
// Original …" because a space precedes a hyphen in ASCII, so equal-scoring
|
||||||
|
// crackers beat the biscuit that was actually scanned.
|
||||||
func sortHits(hits []scoredHit) {
|
func sortHits(hits []scoredHit) {
|
||||||
sort.SliceStable(hits, func(i, j int) bool {
|
sort.SliceStable(hits, func(i, j int) bool {
|
||||||
if hits[i].score != hits[j].score {
|
if hits[i].score != hits[j].score {
|
||||||
return hits[i].score > hits[j].score
|
return hits[i].score > hits[j].score
|
||||||
}
|
}
|
||||||
|
di, dj := vectorRank(hits[i].Distance), vectorRank(hits[j].Distance)
|
||||||
|
if di != dj {
|
||||||
|
return di < dj
|
||||||
|
}
|
||||||
return hits[i].ProductName < hits[j].ProductName
|
return hits[i].ProductName < hits[j].ProductName
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// vectorRank orders by cosine distance, nearest first, with a row the model
|
||||||
|
// never saw (-1, text-only) sorting behind every row it did.
|
||||||
|
func vectorRank(d float64) float64 {
|
||||||
|
if d < 0 {
|
||||||
|
return math.MaxFloat64
|
||||||
|
}
|
||||||
|
return d
|
||||||
|
}
|
||||||
|
|
||||||
func (s *scanService) modelName() string {
|
func (s *scanService) modelName() string {
|
||||||
if s.embedder == nil {
|
if s.embedder == nil {
|
||||||
return "none"
|
return "none"
|
||||||
@@ -516,8 +534,20 @@ func textScore(h repositories.CatalogueHit, label string, tokens []string) float
|
|||||||
name := strings.ToLower(h.ProductName)
|
name := strings.ToLower(h.ProductName)
|
||||||
hay := name + " " + strings.ToLower(h.Title)
|
hay := name + " " + strings.ToLower(h.Title)
|
||||||
label = strings.ToLower(strings.TrimSpace(label))
|
label = strings.ToLower(strings.TrimSpace(label))
|
||||||
if label != "" && strings.Contains(name, label) {
|
// The substring test compares separator-folded forms, so the brand's own
|
||||||
return 0.95
|
// punctuation does not decide the match: "Parle G", "Parle-G" and
|
||||||
|
// "ParleG" all have to reach "Parle-G Original Glucose Biscuits".
|
||||||
|
if label != "" {
|
||||||
|
foldedName, foldedLabel := utils.FoldSeparators(name), utils.FoldSeparators(label)
|
||||||
|
if foldedLabel != "" && strings.Contains(foldedName, foldedLabel) {
|
||||||
|
return 0.95
|
||||||
|
}
|
||||||
|
// Separators dropped rather than folded. Only for a label long enough
|
||||||
|
// that a run of letters means something — "lay" inside "malayalam" is
|
||||||
|
// not a match anyone wants.
|
||||||
|
if tight := utils.TightenLabel(label); len(tight) >= 4 && strings.Contains(utils.TightenLabel(name), tight) {
|
||||||
|
return 0.95
|
||||||
|
}
|
||||||
}
|
}
|
||||||
if len(tokens) == 0 {
|
if len(tokens) == 0 {
|
||||||
return 0
|
return 0
|
||||||
|
|||||||
@@ -501,3 +501,55 @@ func TestConfirmReportsDistanceToTheChosenStore(t *testing.T) {
|
|||||||
t.Fatalf("no position at all is -1; got %v", resp.Store)
|
t.Fatalf("no position at all is -1; got %v", resp.Store)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
var parleG = repositories.CatalogueHit{Brand: "parle", ID: 1, ProductName: "Parle-G Original Glucose Biscuits 250g", Title: "Parle-G", VariantKey: "parle_g", ImageID: "parle_parle_g_250g", Distance: 0.20}
|
||||||
|
var monaco = repositories.CatalogueHit{Brand: "parle", ID: 2, ProductName: "Parle Monaco Classic Regular 200g", Title: "Monaco", VariantKey: "monaco", ImageID: "parle_monaco_200g", Distance: 0.20}
|
||||||
|
|
||||||
|
// Lens reads "Parle-G" off the packet and the customer types "Parle G". Both
|
||||||
|
// spellings, and the run-together one, have to reach the biscuit — not the
|
||||||
|
// salted cracker that merely shares a brand. In production "Parle G" returned
|
||||||
|
// "Parle Monaco Classic Regular 200g" at a confident 0.9.
|
||||||
|
func TestLookupMatchesAHyphenatedNameHoweverItIsWritten(t *testing.T) {
|
||||||
|
for _, label := range []string{"Parle G", "Parle-G", "ParleG", "parle g"} {
|
||||||
|
repo := newLookupFixture()
|
||||||
|
repo.vector = []repositories.CatalogueHit{monaco, parleG} // model puts the cracker first
|
||||||
|
repo.text = []repositories.CatalogueHit{monaco, parleG}
|
||||||
|
svc := NewScanService(repo, fakeEmbedder{vec: []float32{0.1}})
|
||||||
|
|
||||||
|
resp, err := svc.Lookup(context.Background(), models.ScanLookupRequest{Customerid: 5, Label: label})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if resp.Match == nil {
|
||||||
|
t.Fatalf("%q: a stocked product went unrecognised", label)
|
||||||
|
}
|
||||||
|
if resp.Match.Catalogueid != parleG.ID {
|
||||||
|
t.Fatalf("%q: matched %q (%.3f), want Parle-G", label, resp.Match.ProductName, resp.Match.Score)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Equal blended scores used to be settled by product name, which let ASCII
|
||||||
|
// decide relevance: a space sorts before a hyphen, so "Parle Monaco …" beat
|
||||||
|
// "Parle-G …". The model's own similarity settles it instead.
|
||||||
|
func TestSortHitsBreaksTiesOnSimilarityNotPunctuation(t *testing.T) {
|
||||||
|
near := parleG
|
||||||
|
near.Distance = 0.10 // the model is surer about this one
|
||||||
|
far := monaco
|
||||||
|
far.Distance = 0.40
|
||||||
|
|
||||||
|
hits := []scoredHit{{CatalogueHit: far, score: 0.9}, {CatalogueHit: near, score: 0.9}}
|
||||||
|
sortHits(hits)
|
||||||
|
if hits[0].ID != near.ID {
|
||||||
|
t.Fatalf("the nearer vector should win a tie, got %q", hits[0].ProductName)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A row the model never scored (-1, text-only) ranks behind one it did.
|
||||||
|
textOnly := parleG
|
||||||
|
textOnly.Distance = -1
|
||||||
|
hits = []scoredHit{{CatalogueHit: textOnly, score: 0.9}, {CatalogueHit: far, score: 0.9}}
|
||||||
|
sortHits(hits)
|
||||||
|
if hits[0].ID != far.ID {
|
||||||
|
t.Fatalf("a scored row outranks an unscored one, got %q", hits[0].ProductName)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
44
utils/geo.go
44
utils/geo.go
@@ -69,23 +69,61 @@ func parseClock(s string) (int, bool) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// SearchTokens splits a label into the words worth matching on: lowercased,
|
// SearchTokens splits a label into the words worth matching on: lowercased,
|
||||||
// punctuation stripped, single characters and pack-size noise dropped. "Milk
|
// punctuation stripped, pack-size noise dropped. "Milk Bikis 100g" →
|
||||||
// Bikis 100g" → ["milk", "bikis"]; the size is matched separately, if at all.
|
// ["milk", "bikis"]; the size is matched separately, if at all.
|
||||||
|
//
|
||||||
|
// A single character is kept when it follows a word, because in this
|
||||||
|
// catalogue that character is often the whole product: the "G" in "Parle G",
|
||||||
|
// the "K" in "Special K". Dropping it made "Parle G" score the same against
|
||||||
|
// "Parle-G Original Glucose Biscuits" as against "Parle Monaco Classic", and
|
||||||
|
// the tie went to Monaco. It is still dropped when it stands alone — a
|
||||||
|
// one-letter label is not a search — and bare multipliers ("2 x 50gm") are
|
||||||
|
// never words.
|
||||||
func SearchTokens(label string) []string {
|
func SearchTokens(label string) []string {
|
||||||
var tokens []string
|
var tokens []string
|
||||||
seen := make(map[string]bool)
|
seen := make(map[string]bool)
|
||||||
|
kept := 0 // multi-character tokens so far: a lone letter needs one
|
||||||
for _, raw := range strings.FieldsFunc(strings.ToLower(label), func(r rune) bool {
|
for _, raw := range strings.FieldsFunc(strings.ToLower(label), func(r rune) bool {
|
||||||
return !(r >= 'a' && r <= 'z' || r >= '0' && r <= '9')
|
return !(r >= 'a' && r <= 'z' || r >= '0' && r <= '9')
|
||||||
}) {
|
}) {
|
||||||
if len(raw) < 2 || isPackSize(raw) || seen[raw] {
|
if isPackSize(raw) || seen[raw] {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
if len(raw) < 2 && (kept == 0 || isMultiplier(raw)) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if len(raw) > 1 {
|
||||||
|
kept++
|
||||||
|
}
|
||||||
seen[raw] = true
|
seen[raw] = true
|
||||||
tokens = append(tokens, raw)
|
tokens = append(tokens, raw)
|
||||||
}
|
}
|
||||||
return tokens
|
return tokens
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// isMultiplier is the "x" of "2 x 50gm" and the "n" of a multipack — a single
|
||||||
|
// character that joins sizes rather than naming a product.
|
||||||
|
func isMultiplier(tok string) bool {
|
||||||
|
return tok == "x" || tok == "n"
|
||||||
|
}
|
||||||
|
|
||||||
|
// FoldSeparators turns every run of punctuation into one space, so a label
|
||||||
|
// typed without the brand's own punctuation still matches it: "Parle G" and
|
||||||
|
// "Parle-G" both fold to "parle g". Lens reads letterforms off a packet, and
|
||||||
|
// people type what they see, so the hyphen is not reliably either present or
|
||||||
|
// absent on the way in.
|
||||||
|
func FoldSeparators(s string) string {
|
||||||
|
return strings.Join(strings.FieldsFunc(strings.ToLower(s), func(r rune) bool {
|
||||||
|
return !(r >= 'a' && r <= 'z' || r >= '0' && r <= '9')
|
||||||
|
}), " ")
|
||||||
|
}
|
||||||
|
|
||||||
|
// TightenLabel removes separators outright rather than folding them, catching
|
||||||
|
// the other way people write a hyphenated name: "ParleG" against "Parle-G".
|
||||||
|
func TightenLabel(s string) string {
|
||||||
|
return strings.ReplaceAll(FoldSeparators(s), " ", "")
|
||||||
|
}
|
||||||
|
|
||||||
// isPackSize is "100g", "1kg", "500ml", "2l", "250gm" — a number with a unit
|
// isPackSize is "100g", "1kg", "500ml", "2l", "250gm" — a number with a unit
|
||||||
// glued on, or a bare number.
|
// glued on, or a bare number.
|
||||||
func isPackSize(tok string) bool {
|
func isPackSize(tok string) bool {
|
||||||
|
|||||||
@@ -63,3 +63,42 @@ func TestSearchTokens(t *testing.T) {
|
|||||||
t.Error("pack sizes alone are not searchable")
|
t.Error("pack sizes alone are not searchable")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// A single letter is often the whole product name in this catalogue, so it
|
||||||
|
// survives when it follows a word — but not when it stands alone, and not
|
||||||
|
// when it is the multiplier in a pack size.
|
||||||
|
func TestSearchTokensKeepsALetterThatFollowsAWord(t *testing.T) {
|
||||||
|
for _, tc := range []struct {
|
||||||
|
label string
|
||||||
|
want []string
|
||||||
|
}{
|
||||||
|
{"Parle G", []string{"parle", "g"}},
|
||||||
|
{"Parle-G", []string{"parle", "g"}},
|
||||||
|
{"Special K Original", []string{"special", "k", "original"}},
|
||||||
|
{"G", nil}, // a letter alone is not a search
|
||||||
|
{"2 x 50gm", nil}, // multiplier and pack size, no words
|
||||||
|
{"Milk Bikis 100g, Britannia (2 x 50gm)", []string{"milk", "bikis", "britannia"}},
|
||||||
|
} {
|
||||||
|
got := SearchTokens(tc.label)
|
||||||
|
if len(got) != len(tc.want) {
|
||||||
|
t.Fatalf("%q: got %v want %v", tc.label, got, tc.want)
|
||||||
|
}
|
||||||
|
for i := range tc.want {
|
||||||
|
if got[i] != tc.want[i] {
|
||||||
|
t.Fatalf("%q: got %v want %v", tc.label, got, tc.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestFoldAndTightenSeparators(t *testing.T) {
|
||||||
|
if got := FoldSeparators("Parle-G Original"); got != "parle g original" {
|
||||||
|
t.Fatalf("fold: got %q", got)
|
||||||
|
}
|
||||||
|
if FoldSeparators("Parle-G") != FoldSeparators("Parle G") {
|
||||||
|
t.Error("a hyphen and a space are the same separator to us")
|
||||||
|
}
|
||||||
|
if got := TightenLabel("Parle-G"); got != "parleg" {
|
||||||
|
t.Fatalf("tighten: got %q", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user