From c06b029cb29d2bd362709f34fe3237ac2fb88742 Mon Sep 17 00:00:00 2001 From: abhishek Date: Wed, 16 Sep 2026 12:17:39 +0530 Subject: [PATCH] text search --- docs/SCAN_TO_ORDER.md | 12 +++++++ repositories/catalogueColumns_test.go | 32 ++++++++++++++++++ repositories/scanRepository.go | 47 +++++++++++++++++++++------ utils/geo.go | 27 ++++++++++++++- utils/geo_test.go | 26 +++++++++++++++ 5 files changed, 133 insertions(+), 11 deletions(-) diff --git a/docs/SCAN_TO_ORDER.md b/docs/SCAN_TO_ORDER.md index 030006b..25bb4f5 100644 --- a/docs/SCAN_TO_ORDER.md +++ b/docs/SCAN_TO_ORDER.md @@ -158,6 +158,18 @@ Same `ScanStore` shape as inside `stores[]` above, without options. `EMBEDDING_PROVIDER/MODEL/API_KEY` and **must** be the one that indexed the catalogue — the first search checks the vector width and refuses a mismatch by name. +- **The word match asks for most of the label, not all of it** + (`minTokenHits`: two thirds, rounded up, and both of a two-word label). + Requiring every word meant one word the catalogue does not use took the + right product out of the running entirely — "Dettol bottle pack" retrieved + no Dettol, "Parle G biscuit pack" retrieved no Parle-G — and the vector + search then answered alone, confidently and wrongly, at a score the floor + could not catch. Each brand's rows are ordered by how much of the label + they carry (the whole label as a substring outranks any number of loose + words) so that the per-brand `LIMIT` keeps the best rows and not merely the + first ones the planner reached. Packaging words — "pack", "bottle", "jar", + "sachet" and friends, see `utils.isPackaging` — are dropped before any of + this, like pack sizes, unless the label is nothing else. - **The catalogue's model** (verified 2026-09-15 by cosine against a stored row: 1.0000): `all-MiniLM-L6-v2`, 384-d, unit-normalised, embedding the `search_query` column (brand + name + category + blurb + price range). diff --git a/repositories/catalogueColumns_test.go b/repositories/catalogueColumns_test.go index dea15c4..b9f1c1e 100644 --- a/repositories/catalogueColumns_test.go +++ b/repositories/catalogueColumns_test.go @@ -208,3 +208,35 @@ func TestNormaliseBrandKeyRefusesToInventAKey(t *testing.T) { } } } + +// Every word of the label was once required, so one word the catalogue does +// not use ("Parle G biscuit pack") kept the right product out of the result +// altogether and left the vector search to answer alone. +func TestMinTokenHitsAsksForMostWordsNotAllOfThem(t *testing.T) { + for _, tc := range []struct{ tokens, want int }{ + {1, 1}, // one word: it has to be there + {2, 2}, // "Parle G" — both, and both are in Parle-G + {3, 2}, // "Milk Bikis pack" — the pack is allowed to be missing + {4, 3}, // "Parle G biscuit pack" + {5, 4}, + {6, 4}, + } { + if got := minTokenHits(tc.tokens); got != tc.want { + t.Errorf("%d tokens: need %d, want %d", tc.tokens, got, tc.want) + } + } +} + +// A threshold that could fall to 1 would let any single common word drag in +// whole brand tables; one that stayed at n would be the bug all over again. +func TestMinTokenHitsStaysBetweenTwoAndAll(t *testing.T) { + for n := 3; n <= 30; n++ { + got := minTokenHits(n) + if got < 2 { + t.Fatalf("%d tokens: %d is too loose", n, got) + } + if got >= n { + t.Fatalf("%d tokens: %d still demands every word", n, got) + } + } +} diff --git a/repositories/scanRepository.go b/repositories/scanRepository.go index 7e52cba..b76b375 100644 --- a/repositories/scanRepository.go +++ b/repositories/scanRepository.go @@ -467,9 +467,22 @@ func (r *scanRepository) VectorSearch(ctx context.Context, vector []float32, lim return hits, nil } +// minTokenHits is how many of the label's words a row must carry to be worth +// looking at. Every word was once required, which meant a single word the +// catalogue does not use — "Parle G biscuit pack", "Milk Bikis pack" — kept +// the right product out of the result entirely, leaving the vector search to +// answer alone and confidently wrong. Most of them is enough; scoring sorts +// out the rest. +func minTokenHits(n int) int { + if n <= 2 { + return n + } + return (n*2 + 2) / 3 // two thirds, rounded up; never below 2 for n >= 3 +} + // TextSearch is the fallback when there is no embedder, and the tie-breaker // beside it when there is: rows whose name or title contains the label, or -// contains every word of it. +// carry most of its words. func (r *scanRepository) TextSearch(ctx context.Context, label string, limit int) ([]CatalogueHit, error) { tables, err := r.brandTables(ctx) if err != nil { @@ -495,18 +508,32 @@ func (r *scanRepository) TextSearch(ctx context.Context, label string, limit int hay = "LOWER(COALESCE(product_name, '') || ' ' || COALESCE(title, '') || ' ' || COALESCE(search_query, ''))" } - conds := []string{hay + " LIKE ?"} - args = append(args, "%"+label+"%") - all := make([]string, 0, len(tokens)) - for _, tok := range tokens { - all = append(all, hay+" LIKE ?") - args = append(args, "%"+tok+"%") + // How well a row matches, as a number: the whole label as a substring + // outweighs any number of loose words, then one point per word found. + hits := make([]string, 0, len(tokens)+1) + hits = append(hits, "(CASE WHEN "+hay+" LIKE ? THEN 100 ELSE 0 END)") + for range tokens { + hits = append(hits, "(CASE WHEN "+hay+" LIKE ? THEN 1 ELSE 0 END)") } - conds = append(conds, "("+strings.Join(all, " AND ")+")") + rank := strings.Join(hits, " + ") + // The expression appears twice in the SQL — once to filter, once to + // order — so its arguments are bound twice, in that order. + bind := func() { + args = append(args, "%"+label+"%") + for _, tok := range tokens { + args = append(args, "%"+tok+"%") + } + } + bind() + bind() + + // Ordering matters as much as the threshold: a looser WHERE lets more + // rows qualify, and an unordered LIMIT would then be free to return + // the wrong ones. Best match per brand first, id to keep it stable. branches = append(branches, fmt.Sprintf( - `(SELECT %s, -1::float8 AS distance FROM %s WHERE %s LIMIT %d)`, - hitColumns(brand, cols), table, strings.Join(conds, " OR "), limit)) + `(SELECT %s, -1::float8 AS distance FROM %s WHERE (%s) >= %d ORDER BY (%s) DESC, id LIMIT %d)`, + hitColumns(brand, cols), table, rank, minTokenHits(len(tokens)), rank, limit)) } if len(branches) == 0 { return nil, nil diff --git a/utils/geo.go b/utils/geo.go index 3658e60..1c02231 100644 --- a/utils/geo.go +++ b/utils/geo.go @@ -83,6 +83,7 @@ func SearchTokens(label string) []string { var tokens []string seen := make(map[string]bool) kept := 0 // multi-character tokens so far: a lone letter needs one + var filler []string // packaging words, kept only if nothing else survives for _, raw := range strings.FieldsFunc(strings.ToLower(label), func(r rune) bool { return !(r >= 'a' && r <= 'z' || r >= '0' && r <= '9') }) { @@ -92,15 +93,39 @@ func SearchTokens(label string) []string { if len(raw) < 2 && (kept == 0 || isMultiplier(raw)) { continue } + seen[raw] = true + if isPackaging(raw) { + filler = append(filler, raw) + continue + } if len(raw) > 1 { kept++ } - seen[raw] = true tokens = append(tokens, raw) } + // "Dettol bottle pack" is a scan of Dettol. Only when the label is nothing + // but packaging does that packaging become the search. + if len(tokens) == 0 { + return filler + } return tokens } +// isPackaging is what the label says about the wrapper rather than the +// product: "Dettol bottle pack", "Parle G biscuit pack". Lens reads these off +// the packet and no catalogue name carries them, so every one of them used to +// be a word the row had to contain — and "Dettol bottle pack" found no Dettol +// at all. Treated like pack sizes: real words, just not the product's name. +func isPackaging(tok string) bool { + switch tok { + case "pack", "packs", "packet", "packets", "bottle", "bottles", + "box", "boxes", "jar", "jars", "tin", "tins", "pouch", "pouches", + "carton", "cartons", "sachet", "sachets", "combo", "refill": + return true + } + return false +} + // isMultiplier is the "x" of "2 x 50gm" and the "n" of a multipack — a single // character that joins sizes rather than naming a product. func isMultiplier(tok string) bool { diff --git a/utils/geo_test.go b/utils/geo_test.go index 6740235..124092c 100644 --- a/utils/geo_test.go +++ b/utils/geo_test.go @@ -102,3 +102,29 @@ func TestFoldAndTightenSeparators(t *testing.T) { t.Fatalf("tighten: got %q", got) } } + +// Lens reads the wrapper as well as the product. No catalogue name carries +// "bottle" or "pack", so requiring them found no Dettol at all. +func TestSearchTokensDropsPackagingWords(t *testing.T) { + for _, tc := range []struct { + label string + want []string + }{ + {"Dettol bottle pack", []string{"dettol"}}, + {"Parle G biscuit pack", []string{"parle", "g", "biscuit"}}, + {"Milk Bikis pack", []string{"milk", "bikis"}}, + {"Nescafe jar 50g", []string{"nescafe"}}, + {"pack", []string{"pack"}}, // nothing else: the wrapper is the search + {"combo pack", []string{"combo", "pack"}}, // ditto, both kept + } { + got := SearchTokens(tc.label) + if len(got) != len(tc.want) { + t.Fatalf("%q: got %v want %v", tc.label, got, tc.want) + } + for i := range tc.want { + if got[i] != tc.want[i] { + t.Fatalf("%q: got %v want %v", tc.label, got, tc.want) + } + } + } +}