text search

This commit is contained in:
2026-09-16 12:17:39 +05:30
parent 28af3e05f2
commit c06b029cb2
5 changed files with 133 additions and 11 deletions

View File

@@ -158,6 +158,18 @@ Same `ScanStore` shape as inside `stores[]` above, without options.
`EMBEDDING_PROVIDER/MODEL/API_KEY` and **must** be the one that indexed `EMBEDDING_PROVIDER/MODEL/API_KEY` and **must** be the one that indexed
the catalogue — the first search checks the vector width and refuses a the catalogue — the first search checks the vector width and refuses a
mismatch by name. mismatch by name.
- **The word match asks for most of the label, not all of it**
(`minTokenHits`: two thirds, rounded up, and both of a two-word label).
Requiring every word meant one word the catalogue does not use took the
right product out of the running entirely — "Dettol bottle pack" retrieved
no Dettol, "Parle G biscuit pack" retrieved no Parle-G — and the vector
search then answered alone, confidently and wrongly, at a score the floor
could not catch. Each brand's rows are ordered by how much of the label
they carry (the whole label as a substring outranks any number of loose
words) so that the per-brand `LIMIT` keeps the best rows and not merely the
first ones the planner reached. Packaging words — "pack", "bottle", "jar",
"sachet" and friends, see `utils.isPackaging` — are dropped before any of
this, like pack sizes, unless the label is nothing else.
- **The catalogue's model** (verified 2026-09-15 by cosine against a stored - **The catalogue's model** (verified 2026-09-15 by cosine against a stored
row: 1.0000): `all-MiniLM-L6-v2`, 384-d, unit-normalised, embedding the row: 1.0000): `all-MiniLM-L6-v2`, 384-d, unit-normalised, embedding the
`search_query` column (brand + name + category + blurb + price range). `search_query` column (brand + name + category + blurb + price range).

View File

@@ -208,3 +208,35 @@ func TestNormaliseBrandKeyRefusesToInventAKey(t *testing.T) {
} }
} }
} }
// Every word of the label was once required, so one word the catalogue does
// not use ("Parle G biscuit pack") kept the right product out of the result
// altogether and left the vector search to answer alone.
func TestMinTokenHitsAsksForMostWordsNotAllOfThem(t *testing.T) {
for _, tc := range []struct{ tokens, want int }{
{1, 1}, // one word: it has to be there
{2, 2}, // "Parle G" — both, and both are in Parle-G
{3, 2}, // "Milk Bikis pack" — the pack is allowed to be missing
{4, 3}, // "Parle G biscuit pack"
{5, 4},
{6, 4},
} {
if got := minTokenHits(tc.tokens); got != tc.want {
t.Errorf("%d tokens: need %d, want %d", tc.tokens, got, tc.want)
}
}
}
// A threshold that could fall to 1 would let any single common word drag in
// whole brand tables; one that stayed at n would be the bug all over again.
func TestMinTokenHitsStaysBetweenTwoAndAll(t *testing.T) {
for n := 3; n <= 30; n++ {
got := minTokenHits(n)
if got < 2 {
t.Fatalf("%d tokens: %d is too loose", n, got)
}
if got >= n {
t.Fatalf("%d tokens: %d still demands every word", n, got)
}
}
}

View File

@@ -467,9 +467,22 @@ func (r *scanRepository) VectorSearch(ctx context.Context, vector []float32, lim
return hits, nil return hits, nil
} }
// minTokenHits is how many of the label's words a row must carry to be worth
// looking at. Every word was once required, which meant a single word the
// catalogue does not use — "Parle G biscuit pack", "Milk Bikis pack" — kept
// the right product out of the result entirely, leaving the vector search to
// answer alone and confidently wrong. Most of them is enough; scoring sorts
// out the rest.
func minTokenHits(n int) int {
if n <= 2 {
return n
}
return (n*2 + 2) / 3 // two thirds, rounded up; never below 2 for n >= 3
}
// TextSearch is the fallback when there is no embedder, and the tie-breaker // TextSearch is the fallback when there is no embedder, and the tie-breaker
// beside it when there is: rows whose name or title contains the label, or // beside it when there is: rows whose name or title contains the label, or
// contains every word of it. // carry most of its words.
func (r *scanRepository) TextSearch(ctx context.Context, label string, limit int) ([]CatalogueHit, error) { func (r *scanRepository) TextSearch(ctx context.Context, label string, limit int) ([]CatalogueHit, error) {
tables, err := r.brandTables(ctx) tables, err := r.brandTables(ctx)
if err != nil { if err != nil {
@@ -495,18 +508,32 @@ func (r *scanRepository) TextSearch(ctx context.Context, label string, limit int
hay = "LOWER(COALESCE(product_name, '') || ' ' || COALESCE(title, '') || ' ' || COALESCE(search_query, ''))" hay = "LOWER(COALESCE(product_name, '') || ' ' || COALESCE(title, '') || ' ' || COALESCE(search_query, ''))"
} }
conds := []string{hay + " LIKE ?"} // How well a row matches, as a number: the whole label as a substring
args = append(args, "%"+label+"%") // outweighs any number of loose words, then one point per word found.
all := make([]string, 0, len(tokens)) hits := make([]string, 0, len(tokens)+1)
for _, tok := range tokens { hits = append(hits, "(CASE WHEN "+hay+" LIKE ? THEN 100 ELSE 0 END)")
all = append(all, hay+" LIKE ?") for range tokens {
args = append(args, "%"+tok+"%") hits = append(hits, "(CASE WHEN "+hay+" LIKE ? THEN 1 ELSE 0 END)")
} }
conds = append(conds, "("+strings.Join(all, " AND ")+")") rank := strings.Join(hits, " + ")
// The expression appears twice in the SQL — once to filter, once to
// order — so its arguments are bound twice, in that order.
bind := func() {
args = append(args, "%"+label+"%")
for _, tok := range tokens {
args = append(args, "%"+tok+"%")
}
}
bind()
bind()
// Ordering matters as much as the threshold: a looser WHERE lets more
// rows qualify, and an unordered LIMIT would then be free to return
// the wrong ones. Best match per brand first, id to keep it stable.
branches = append(branches, fmt.Sprintf( branches = append(branches, fmt.Sprintf(
`(SELECT %s, -1::float8 AS distance FROM %s WHERE %s LIMIT %d)`, `(SELECT %s, -1::float8 AS distance FROM %s WHERE (%s) >= %d ORDER BY (%s) DESC, id LIMIT %d)`,
hitColumns(brand, cols), table, strings.Join(conds, " OR "), limit)) hitColumns(brand, cols), table, rank, minTokenHits(len(tokens)), rank, limit))
} }
if len(branches) == 0 { if len(branches) == 0 {
return nil, nil return nil, nil

View File

@@ -83,6 +83,7 @@ func SearchTokens(label string) []string {
var tokens []string var tokens []string
seen := make(map[string]bool) seen := make(map[string]bool)
kept := 0 // multi-character tokens so far: a lone letter needs one kept := 0 // multi-character tokens so far: a lone letter needs one
var filler []string // packaging words, kept only if nothing else survives
for _, raw := range strings.FieldsFunc(strings.ToLower(label), func(r rune) bool { for _, raw := range strings.FieldsFunc(strings.ToLower(label), func(r rune) bool {
return !(r >= 'a' && r <= 'z' || r >= '0' && r <= '9') return !(r >= 'a' && r <= 'z' || r >= '0' && r <= '9')
}) { }) {
@@ -92,15 +93,39 @@ func SearchTokens(label string) []string {
if len(raw) < 2 && (kept == 0 || isMultiplier(raw)) { if len(raw) < 2 && (kept == 0 || isMultiplier(raw)) {
continue continue
} }
seen[raw] = true
if isPackaging(raw) {
filler = append(filler, raw)
continue
}
if len(raw) > 1 { if len(raw) > 1 {
kept++ kept++
} }
seen[raw] = true
tokens = append(tokens, raw) tokens = append(tokens, raw)
} }
// "Dettol bottle pack" is a scan of Dettol. Only when the label is nothing
// but packaging does that packaging become the search.
if len(tokens) == 0 {
return filler
}
return tokens return tokens
} }
// isPackaging is what the label says about the wrapper rather than the
// product: "Dettol bottle pack", "Parle G biscuit pack". Lens reads these off
// the packet and no catalogue name carries them, so every one of them used to
// be a word the row had to contain — and "Dettol bottle pack" found no Dettol
// at all. Treated like pack sizes: real words, just not the product's name.
func isPackaging(tok string) bool {
switch tok {
case "pack", "packs", "packet", "packets", "bottle", "bottles",
"box", "boxes", "jar", "jars", "tin", "tins", "pouch", "pouches",
"carton", "cartons", "sachet", "sachets", "combo", "refill":
return true
}
return false
}
// isMultiplier is the "x" of "2 x 50gm" and the "n" of a multipack — a single // isMultiplier is the "x" of "2 x 50gm" and the "n" of a multipack — a single
// character that joins sizes rather than naming a product. // character that joins sizes rather than naming a product.
func isMultiplier(tok string) bool { func isMultiplier(tok string) bool {

View File

@@ -102,3 +102,29 @@ func TestFoldAndTightenSeparators(t *testing.T) {
t.Fatalf("tighten: got %q", got) t.Fatalf("tighten: got %q", got)
} }
} }
// Lens reads the wrapper as well as the product. No catalogue name carries
// "bottle" or "pack", so requiring them found no Dettol at all.
func TestSearchTokensDropsPackagingWords(t *testing.T) {
for _, tc := range []struct {
label string
want []string
}{
{"Dettol bottle pack", []string{"dettol"}},
{"Parle G biscuit pack", []string{"parle", "g", "biscuit"}},
{"Milk Bikis pack", []string{"milk", "bikis"}},
{"Nescafe jar 50g", []string{"nescafe"}},
{"pack", []string{"pack"}}, // nothing else: the wrapper is the search
{"combo pack", []string{"combo", "pack"}}, // ditto, both kept
} {
got := SearchTokens(tc.label)
if len(got) != len(tc.want) {
t.Fatalf("%q: got %v want %v", tc.label, got, tc.want)
}
for i := range tc.want {
if got[i] != tc.want[i] {
t.Fatalf("%q: got %v want %v", tc.label, got, tc.want)
}
}
}
}