text search
This commit is contained in:
@@ -158,6 +158,18 @@ Same `ScanStore` shape as inside `stores[]` above, without options.
|
||||
`EMBEDDING_PROVIDER/MODEL/API_KEY` and **must** be the one that indexed
|
||||
the catalogue — the first search checks the vector width and refuses a
|
||||
mismatch by name.
|
||||
- **The word match asks for most of the label, not all of it**
|
||||
(`minTokenHits`: two thirds, rounded up, and both of a two-word label).
|
||||
Requiring every word meant one word the catalogue does not use took the
|
||||
right product out of the running entirely — "Dettol bottle pack" retrieved
|
||||
no Dettol, "Parle G biscuit pack" retrieved no Parle-G — and the vector
|
||||
search then answered alone, confidently and wrongly, at a score the floor
|
||||
could not catch. Each brand's rows are ordered by how much of the label
|
||||
they carry (the whole label as a substring outranks any number of loose
|
||||
words) so that the per-brand `LIMIT` keeps the best rows and not merely the
|
||||
first ones the planner reached. Packaging words — "pack", "bottle", "jar",
|
||||
"sachet" and friends, see `utils.isPackaging` — are dropped before any of
|
||||
this, like pack sizes, unless the label is nothing else.
|
||||
- **The catalogue's model** (verified 2026-09-15 by cosine against a stored
|
||||
row: 1.0000): `all-MiniLM-L6-v2`, 384-d, unit-normalised, embedding the
|
||||
`search_query` column (brand + name + category + blurb + price range).
|
||||
|
||||
@@ -208,3 +208,35 @@ func TestNormaliseBrandKeyRefusesToInventAKey(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Every word of the label was once required, so one word the catalogue does
|
||||
// not use ("Parle G biscuit pack") kept the right product out of the result
|
||||
// altogether and left the vector search to answer alone.
|
||||
func TestMinTokenHitsAsksForMostWordsNotAllOfThem(t *testing.T) {
|
||||
for _, tc := range []struct{ tokens, want int }{
|
||||
{1, 1}, // one word: it has to be there
|
||||
{2, 2}, // "Parle G" — both, and both are in Parle-G
|
||||
{3, 2}, // "Milk Bikis pack" — the pack is allowed to be missing
|
||||
{4, 3}, // "Parle G biscuit pack"
|
||||
{5, 4},
|
||||
{6, 4},
|
||||
} {
|
||||
if got := minTokenHits(tc.tokens); got != tc.want {
|
||||
t.Errorf("%d tokens: need %d, want %d", tc.tokens, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A threshold that could fall to 1 would let any single common word drag in
|
||||
// whole brand tables; one that stayed at n would be the bug all over again.
|
||||
func TestMinTokenHitsStaysBetweenTwoAndAll(t *testing.T) {
|
||||
for n := 3; n <= 30; n++ {
|
||||
got := minTokenHits(n)
|
||||
if got < 2 {
|
||||
t.Fatalf("%d tokens: %d is too loose", n, got)
|
||||
}
|
||||
if got >= n {
|
||||
t.Fatalf("%d tokens: %d still demands every word", n, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -467,9 +467,22 @@ func (r *scanRepository) VectorSearch(ctx context.Context, vector []float32, lim
|
||||
return hits, nil
|
||||
}
|
||||
|
||||
// minTokenHits is how many of the label's words a row must carry to be worth
|
||||
// looking at. Every word was once required, which meant a single word the
|
||||
// catalogue does not use — "Parle G biscuit pack", "Milk Bikis pack" — kept
|
||||
// the right product out of the result entirely, leaving the vector search to
|
||||
// answer alone and confidently wrong. Most of them is enough; scoring sorts
|
||||
// out the rest.
|
||||
func minTokenHits(n int) int {
|
||||
if n <= 2 {
|
||||
return n
|
||||
}
|
||||
return (n*2 + 2) / 3 // two thirds, rounded up; never below 2 for n >= 3
|
||||
}
|
||||
|
||||
// TextSearch is the fallback when there is no embedder, and the tie-breaker
|
||||
// beside it when there is: rows whose name or title contains the label, or
|
||||
// contains every word of it.
|
||||
// carry most of its words.
|
||||
func (r *scanRepository) TextSearch(ctx context.Context, label string, limit int) ([]CatalogueHit, error) {
|
||||
tables, err := r.brandTables(ctx)
|
||||
if err != nil {
|
||||
@@ -495,18 +508,32 @@ func (r *scanRepository) TextSearch(ctx context.Context, label string, limit int
|
||||
hay = "LOWER(COALESCE(product_name, '') || ' ' || COALESCE(title, '') || ' ' || COALESCE(search_query, ''))"
|
||||
}
|
||||
|
||||
conds := []string{hay + " LIKE ?"}
|
||||
args = append(args, "%"+label+"%")
|
||||
all := make([]string, 0, len(tokens))
|
||||
for _, tok := range tokens {
|
||||
all = append(all, hay+" LIKE ?")
|
||||
args = append(args, "%"+tok+"%")
|
||||
// How well a row matches, as a number: the whole label as a substring
|
||||
// outweighs any number of loose words, then one point per word found.
|
||||
hits := make([]string, 0, len(tokens)+1)
|
||||
hits = append(hits, "(CASE WHEN "+hay+" LIKE ? THEN 100 ELSE 0 END)")
|
||||
for range tokens {
|
||||
hits = append(hits, "(CASE WHEN "+hay+" LIKE ? THEN 1 ELSE 0 END)")
|
||||
}
|
||||
conds = append(conds, "("+strings.Join(all, " AND ")+")")
|
||||
rank := strings.Join(hits, " + ")
|
||||
|
||||
// The expression appears twice in the SQL — once to filter, once to
|
||||
// order — so its arguments are bound twice, in that order.
|
||||
bind := func() {
|
||||
args = append(args, "%"+label+"%")
|
||||
for _, tok := range tokens {
|
||||
args = append(args, "%"+tok+"%")
|
||||
}
|
||||
}
|
||||
bind()
|
||||
bind()
|
||||
|
||||
// Ordering matters as much as the threshold: a looser WHERE lets more
|
||||
// rows qualify, and an unordered LIMIT would then be free to return
|
||||
// the wrong ones. Best match per brand first, id to keep it stable.
|
||||
branches = append(branches, fmt.Sprintf(
|
||||
`(SELECT %s, -1::float8 AS distance FROM %s WHERE %s LIMIT %d)`,
|
||||
hitColumns(brand, cols), table, strings.Join(conds, " OR "), limit))
|
||||
`(SELECT %s, -1::float8 AS distance FROM %s WHERE (%s) >= %d ORDER BY (%s) DESC, id LIMIT %d)`,
|
||||
hitColumns(brand, cols), table, rank, minTokenHits(len(tokens)), rank, limit))
|
||||
}
|
||||
if len(branches) == 0 {
|
||||
return nil, nil
|
||||
|
||||
27
utils/geo.go
27
utils/geo.go
@@ -83,6 +83,7 @@ func SearchTokens(label string) []string {
|
||||
var tokens []string
|
||||
seen := make(map[string]bool)
|
||||
kept := 0 // multi-character tokens so far: a lone letter needs one
|
||||
var filler []string // packaging words, kept only if nothing else survives
|
||||
for _, raw := range strings.FieldsFunc(strings.ToLower(label), func(r rune) bool {
|
||||
return !(r >= 'a' && r <= 'z' || r >= '0' && r <= '9')
|
||||
}) {
|
||||
@@ -92,15 +93,39 @@ func SearchTokens(label string) []string {
|
||||
if len(raw) < 2 && (kept == 0 || isMultiplier(raw)) {
|
||||
continue
|
||||
}
|
||||
seen[raw] = true
|
||||
if isPackaging(raw) {
|
||||
filler = append(filler, raw)
|
||||
continue
|
||||
}
|
||||
if len(raw) > 1 {
|
||||
kept++
|
||||
}
|
||||
seen[raw] = true
|
||||
tokens = append(tokens, raw)
|
||||
}
|
||||
// "Dettol bottle pack" is a scan of Dettol. Only when the label is nothing
|
||||
// but packaging does that packaging become the search.
|
||||
if len(tokens) == 0 {
|
||||
return filler
|
||||
}
|
||||
return tokens
|
||||
}
|
||||
|
||||
// isPackaging is what the label says about the wrapper rather than the
|
||||
// product: "Dettol bottle pack", "Parle G biscuit pack". Lens reads these off
|
||||
// the packet and no catalogue name carries them, so every one of them used to
|
||||
// be a word the row had to contain — and "Dettol bottle pack" found no Dettol
|
||||
// at all. Treated like pack sizes: real words, just not the product's name.
|
||||
func isPackaging(tok string) bool {
|
||||
switch tok {
|
||||
case "pack", "packs", "packet", "packets", "bottle", "bottles",
|
||||
"box", "boxes", "jar", "jars", "tin", "tins", "pouch", "pouches",
|
||||
"carton", "cartons", "sachet", "sachets", "combo", "refill":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// isMultiplier is the "x" of "2 x 50gm" and the "n" of a multipack — a single
|
||||
// character that joins sizes rather than naming a product.
|
||||
func isMultiplier(tok string) bool {
|
||||
|
||||
@@ -102,3 +102,29 @@ func TestFoldAndTightenSeparators(t *testing.T) {
|
||||
t.Fatalf("tighten: got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Lens reads the wrapper as well as the product. No catalogue name carries
|
||||
// "bottle" or "pack", so requiring them found no Dettol at all.
|
||||
func TestSearchTokensDropsPackagingWords(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
label string
|
||||
want []string
|
||||
}{
|
||||
{"Dettol bottle pack", []string{"dettol"}},
|
||||
{"Parle G biscuit pack", []string{"parle", "g", "biscuit"}},
|
||||
{"Milk Bikis pack", []string{"milk", "bikis"}},
|
||||
{"Nescafe jar 50g", []string{"nescafe"}},
|
||||
{"pack", []string{"pack"}}, // nothing else: the wrapper is the search
|
||||
{"combo pack", []string{"combo", "pack"}}, // ditto, both kept
|
||||
} {
|
||||
got := SearchTokens(tc.label)
|
||||
if len(got) != len(tc.want) {
|
||||
t.Fatalf("%q: got %v want %v", tc.label, got, tc.want)
|
||||
}
|
||||
for i := range tc.want {
|
||||
if got[i] != tc.want[i] {
|
||||
t.Fatalf("%q: got %v want %v", tc.label, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user