text search
This commit is contained in:
@@ -158,6 +158,18 @@ Same `ScanStore` shape as inside `stores[]` above, without options.
|
|||||||
`EMBEDDING_PROVIDER/MODEL/API_KEY` and **must** be the one that indexed
|
`EMBEDDING_PROVIDER/MODEL/API_KEY` and **must** be the one that indexed
|
||||||
the catalogue — the first search checks the vector width and refuses a
|
the catalogue — the first search checks the vector width and refuses a
|
||||||
mismatch by name.
|
mismatch by name.
|
||||||
|
- **The word match asks for most of the label, not all of it**
|
||||||
|
(`minTokenHits`: two thirds, rounded up, and both of a two-word label).
|
||||||
|
Requiring every word meant one word the catalogue does not use took the
|
||||||
|
right product out of the running entirely — "Dettol bottle pack" retrieved
|
||||||
|
no Dettol, "Parle G biscuit pack" retrieved no Parle-G — and the vector
|
||||||
|
search then answered alone, confidently and wrongly, at a score the floor
|
||||||
|
could not catch. Each brand's rows are ordered by how much of the label
|
||||||
|
they carry (the whole label as a substring outranks any number of loose
|
||||||
|
words) so that the per-brand `LIMIT` keeps the best rows and not merely the
|
||||||
|
first ones the planner reached. Packaging words — "pack", "bottle", "jar",
|
||||||
|
"sachet" and friends, see `utils.isPackaging` — are dropped before any of
|
||||||
|
this, like pack sizes, unless the label is nothing else.
|
||||||
- **The catalogue's model** (verified 2026-09-15 by cosine against a stored
|
- **The catalogue's model** (verified 2026-09-15 by cosine against a stored
|
||||||
row: 1.0000): `all-MiniLM-L6-v2`, 384-d, unit-normalised, embedding the
|
row: 1.0000): `all-MiniLM-L6-v2`, 384-d, unit-normalised, embedding the
|
||||||
`search_query` column (brand + name + category + blurb + price range).
|
`search_query` column (brand + name + category + blurb + price range).
|
||||||
|
|||||||
@@ -208,3 +208,35 @@ func TestNormaliseBrandKeyRefusesToInventAKey(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Every word of the label was once required, so one word the catalogue does
|
||||||
|
// not use ("Parle G biscuit pack") kept the right product out of the result
|
||||||
|
// altogether and left the vector search to answer alone.
|
||||||
|
func TestMinTokenHitsAsksForMostWordsNotAllOfThem(t *testing.T) {
|
||||||
|
for _, tc := range []struct{ tokens, want int }{
|
||||||
|
{1, 1}, // one word: it has to be there
|
||||||
|
{2, 2}, // "Parle G" — both, and both are in Parle-G
|
||||||
|
{3, 2}, // "Milk Bikis pack" — the pack is allowed to be missing
|
||||||
|
{4, 3}, // "Parle G biscuit pack"
|
||||||
|
{5, 4},
|
||||||
|
{6, 4},
|
||||||
|
} {
|
||||||
|
if got := minTokenHits(tc.tokens); got != tc.want {
|
||||||
|
t.Errorf("%d tokens: need %d, want %d", tc.tokens, got, tc.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A threshold that could fall to 1 would let any single common word drag in
|
||||||
|
// whole brand tables; one that stayed at n would be the bug all over again.
|
||||||
|
func TestMinTokenHitsStaysBetweenTwoAndAll(t *testing.T) {
|
||||||
|
for n := 3; n <= 30; n++ {
|
||||||
|
got := minTokenHits(n)
|
||||||
|
if got < 2 {
|
||||||
|
t.Fatalf("%d tokens: %d is too loose", n, got)
|
||||||
|
}
|
||||||
|
if got >= n {
|
||||||
|
t.Fatalf("%d tokens: %d still demands every word", n, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -467,9 +467,22 @@ func (r *scanRepository) VectorSearch(ctx context.Context, vector []float32, lim
|
|||||||
return hits, nil
|
return hits, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// minTokenHits is how many of the label's words a row must carry to be worth
|
||||||
|
// looking at. Every word was once required, which meant a single word the
|
||||||
|
// catalogue does not use — "Parle G biscuit pack", "Milk Bikis pack" — kept
|
||||||
|
// the right product out of the result entirely, leaving the vector search to
|
||||||
|
// answer alone and confidently wrong. Most of them is enough; scoring sorts
|
||||||
|
// out the rest.
|
||||||
|
func minTokenHits(n int) int {
|
||||||
|
if n <= 2 {
|
||||||
|
return n
|
||||||
|
}
|
||||||
|
return (n*2 + 2) / 3 // two thirds, rounded up; never below 2 for n >= 3
|
||||||
|
}
|
||||||
|
|
||||||
// TextSearch is the fallback when there is no embedder, and the tie-breaker
|
// TextSearch is the fallback when there is no embedder, and the tie-breaker
|
||||||
// beside it when there is: rows whose name or title contains the label, or
|
// beside it when there is: rows whose name or title contains the label, or
|
||||||
// contains every word of it.
|
// carry most of its words.
|
||||||
func (r *scanRepository) TextSearch(ctx context.Context, label string, limit int) ([]CatalogueHit, error) {
|
func (r *scanRepository) TextSearch(ctx context.Context, label string, limit int) ([]CatalogueHit, error) {
|
||||||
tables, err := r.brandTables(ctx)
|
tables, err := r.brandTables(ctx)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -495,18 +508,32 @@ func (r *scanRepository) TextSearch(ctx context.Context, label string, limit int
|
|||||||
hay = "LOWER(COALESCE(product_name, '') || ' ' || COALESCE(title, '') || ' ' || COALESCE(search_query, ''))"
|
hay = "LOWER(COALESCE(product_name, '') || ' ' || COALESCE(title, '') || ' ' || COALESCE(search_query, ''))"
|
||||||
}
|
}
|
||||||
|
|
||||||
conds := []string{hay + " LIKE ?"}
|
// How well a row matches, as a number: the whole label as a substring
|
||||||
args = append(args, "%"+label+"%")
|
// outweighs any number of loose words, then one point per word found.
|
||||||
all := make([]string, 0, len(tokens))
|
hits := make([]string, 0, len(tokens)+1)
|
||||||
for _, tok := range tokens {
|
hits = append(hits, "(CASE WHEN "+hay+" LIKE ? THEN 100 ELSE 0 END)")
|
||||||
all = append(all, hay+" LIKE ?")
|
for range tokens {
|
||||||
args = append(args, "%"+tok+"%")
|
hits = append(hits, "(CASE WHEN "+hay+" LIKE ? THEN 1 ELSE 0 END)")
|
||||||
}
|
}
|
||||||
conds = append(conds, "("+strings.Join(all, " AND ")+")")
|
rank := strings.Join(hits, " + ")
|
||||||
|
|
||||||
|
// The expression appears twice in the SQL — once to filter, once to
|
||||||
|
// order — so its arguments are bound twice, in that order.
|
||||||
|
bind := func() {
|
||||||
|
args = append(args, "%"+label+"%")
|
||||||
|
for _, tok := range tokens {
|
||||||
|
args = append(args, "%"+tok+"%")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
bind()
|
||||||
|
bind()
|
||||||
|
|
||||||
|
// Ordering matters as much as the threshold: a looser WHERE lets more
|
||||||
|
// rows qualify, and an unordered LIMIT would then be free to return
|
||||||
|
// the wrong ones. Best match per brand first, id to keep it stable.
|
||||||
branches = append(branches, fmt.Sprintf(
|
branches = append(branches, fmt.Sprintf(
|
||||||
`(SELECT %s, -1::float8 AS distance FROM %s WHERE %s LIMIT %d)`,
|
`(SELECT %s, -1::float8 AS distance FROM %s WHERE (%s) >= %d ORDER BY (%s) DESC, id LIMIT %d)`,
|
||||||
hitColumns(brand, cols), table, strings.Join(conds, " OR "), limit))
|
hitColumns(brand, cols), table, rank, minTokenHits(len(tokens)), rank, limit))
|
||||||
}
|
}
|
||||||
if len(branches) == 0 {
|
if len(branches) == 0 {
|
||||||
return nil, nil
|
return nil, nil
|
||||||
|
|||||||
27
utils/geo.go
27
utils/geo.go
@@ -83,6 +83,7 @@ func SearchTokens(label string) []string {
|
|||||||
var tokens []string
|
var tokens []string
|
||||||
seen := make(map[string]bool)
|
seen := make(map[string]bool)
|
||||||
kept := 0 // multi-character tokens so far: a lone letter needs one
|
kept := 0 // multi-character tokens so far: a lone letter needs one
|
||||||
|
var filler []string // packaging words, kept only if nothing else survives
|
||||||
for _, raw := range strings.FieldsFunc(strings.ToLower(label), func(r rune) bool {
|
for _, raw := range strings.FieldsFunc(strings.ToLower(label), func(r rune) bool {
|
||||||
return !(r >= 'a' && r <= 'z' || r >= '0' && r <= '9')
|
return !(r >= 'a' && r <= 'z' || r >= '0' && r <= '9')
|
||||||
}) {
|
}) {
|
||||||
@@ -92,15 +93,39 @@ func SearchTokens(label string) []string {
|
|||||||
if len(raw) < 2 && (kept == 0 || isMultiplier(raw)) {
|
if len(raw) < 2 && (kept == 0 || isMultiplier(raw)) {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
seen[raw] = true
|
||||||
|
if isPackaging(raw) {
|
||||||
|
filler = append(filler, raw)
|
||||||
|
continue
|
||||||
|
}
|
||||||
if len(raw) > 1 {
|
if len(raw) > 1 {
|
||||||
kept++
|
kept++
|
||||||
}
|
}
|
||||||
seen[raw] = true
|
|
||||||
tokens = append(tokens, raw)
|
tokens = append(tokens, raw)
|
||||||
}
|
}
|
||||||
|
// "Dettol bottle pack" is a scan of Dettol. Only when the label is nothing
|
||||||
|
// but packaging does that packaging become the search.
|
||||||
|
if len(tokens) == 0 {
|
||||||
|
return filler
|
||||||
|
}
|
||||||
return tokens
|
return tokens
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// isPackaging is what the label says about the wrapper rather than the
|
||||||
|
// product: "Dettol bottle pack", "Parle G biscuit pack". Lens reads these off
|
||||||
|
// the packet and no catalogue name carries them, so every one of them used to
|
||||||
|
// be a word the row had to contain — and "Dettol bottle pack" found no Dettol
|
||||||
|
// at all. Treated like pack sizes: real words, just not the product's name.
|
||||||
|
func isPackaging(tok string) bool {
|
||||||
|
switch tok {
|
||||||
|
case "pack", "packs", "packet", "packets", "bottle", "bottles",
|
||||||
|
"box", "boxes", "jar", "jars", "tin", "tins", "pouch", "pouches",
|
||||||
|
"carton", "cartons", "sachet", "sachets", "combo", "refill":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
// isMultiplier is the "x" of "2 x 50gm" and the "n" of a multipack — a single
|
// isMultiplier is the "x" of "2 x 50gm" and the "n" of a multipack — a single
|
||||||
// character that joins sizes rather than naming a product.
|
// character that joins sizes rather than naming a product.
|
||||||
func isMultiplier(tok string) bool {
|
func isMultiplier(tok string) bool {
|
||||||
|
|||||||
@@ -102,3 +102,29 @@ func TestFoldAndTightenSeparators(t *testing.T) {
|
|||||||
t.Fatalf("tighten: got %q", got)
|
t.Fatalf("tighten: got %q", got)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Lens reads the wrapper as well as the product. No catalogue name carries
|
||||||
|
// "bottle" or "pack", so requiring them found no Dettol at all.
|
||||||
|
func TestSearchTokensDropsPackagingWords(t *testing.T) {
|
||||||
|
for _, tc := range []struct {
|
||||||
|
label string
|
||||||
|
want []string
|
||||||
|
}{
|
||||||
|
{"Dettol bottle pack", []string{"dettol"}},
|
||||||
|
{"Parle G biscuit pack", []string{"parle", "g", "biscuit"}},
|
||||||
|
{"Milk Bikis pack", []string{"milk", "bikis"}},
|
||||||
|
{"Nescafe jar 50g", []string{"nescafe"}},
|
||||||
|
{"pack", []string{"pack"}}, // nothing else: the wrapper is the search
|
||||||
|
{"combo pack", []string{"combo", "pack"}}, // ditto, both kept
|
||||||
|
} {
|
||||||
|
got := SearchTokens(tc.label)
|
||||||
|
if len(got) != len(tc.want) {
|
||||||
|
t.Fatalf("%q: got %v want %v", tc.label, got, tc.want)
|
||||||
|
}
|
||||||
|
for i := range tc.want {
|
||||||
|
if got[i] != tc.want[i] {
|
||||||
|
t.Fatalf("%q: got %v want %v", tc.label, got, tc.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user