Merge origin/main: keep the substring rule, read its tie

main had moved on with retrieval work validated against real queries —
minTokenHits (the word match needs two thirds of the label, not all of
it), separator folding so "Parle G"/"Parle-G"/"ParleG" all reach Parle-G,
the floor at 0.50 after "Paracetamol" came back as "Paneer Makhni 500ml"
at 0.304, and ties broken on cosine distance instead of name. All of that
is kept exactly as it was.

The conflict was in textScore: this branch replaced the substring rule
with a coverage formula to stop a bare brand name resolving to one
arbitrary product. That is the wrong half to change. The substring rule
scores every product of a brand 0.95 IDENTICALLY, and that tie is not the
bug — it is the signal. isAmbiguous reads it, so the branch's coverage
rewrite is dropped and the ambiguity layer alone does the work:

  "britannia" → all 258 rows tie at 0.95 → ambiguous: true + candidates
  "Parle G"   → folding and the single-character token still land it
  a real name → runner-up far behind → match, unchanged

Dropped with it: scanSpecificEnough, the per-hit text score, and the
proportional confirmation bonus — the flat +0.10 is back. Simpler, and it
leaves main's tuning untouched.

TestTextScoreRewardsSpecificityNotJustOverlap tested the removed formula
and is replaced by TestABrandNameScoresItsProductsIdentically, which
guards the tie itself: a formula that broke it on name length or word
count would bring the bug back.

Docs carry both rationales, and now say plainly that confidence stays
high on the ambiguous path — gate on `ambiguous`, never on `confidence`.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-09-23 11:05:03 +05:30
25 changed files with 1661 additions and 228 deletions

View File

@@ -4,9 +4,11 @@ import (
"encoding/json"
"fmt"
"log"
"strings"
"time"
"nearle/models"
"nearle/repositories"
"time"
)
type ProductService interface {
@@ -22,7 +24,7 @@ type ProductService interface {
RemoveProductVariant(tenantid, variantid int) error
VariantChildIDs(tenantid int) (map[int]bool, error)
CreateProductStock(stocks []models.Productstock) error
CreateProduct(product models.Products) error
CreateProduct(product models.Products) (models.Products, error)
UpdateProduct(product models.Products) error
DeleteProduct(productID int) error
GetStockStatement(tenantID, locationID, subcategoryID, pageno, pagesize int, keyword string) ([]models.Productstockstatement, error)
@@ -169,8 +171,27 @@ func (s *productService) UpdateProductStatus(productIDs []int, status string) er
return s.repo.UpdateProductStatus(productIDs, status)
}
func (s *productService) CreateProduct(product models.Products) error {
return s.repo.CreateProduct(product)
// CreateProduct stores one product and hands it back with its id filled in.
//
// It used to return only an error, and the id was lost on the way out: the
// repository took the struct by value, GORM wrote the generated productid onto
// that copy, and the copy was discarded — so the endpoint answered
// `productid: 0` for a row that certainly had one.
//
// The caller needs it. A product is not sellable until it has been priced at an
// outlet and stocked there, and both of those calls are keyed on productid, so
// every importer had to create, re-read the tenant's whole catalogue, and match
// its own rows back by SKU to carry on — which is also why creating two
// products with the same SKU quietly attached the second one's stock to the
// first.
func (s *productService) CreateProduct(product models.Products) (models.Products, error) {
id, err := s.repo.CreateProductReturningID(product)
if err != nil {
return models.Products{}, err
}
product.Productid = id
return product, nil
}
func (s *productService) UpdateProduct(product models.Products) error {
@@ -308,6 +329,54 @@ func (s *productService) DeleteProductLocation(tenantid, locationid, productid i
return s.repo.DeleteProductLocation(tenantid, locationid, productid)
}
// catalogueFactsOf collects the catalogue fields the product table has no
// column for, so an import keeps them instead of leaving them behind.
//
// Only what the catalogue actually stated: an empty field is omitted rather
// than written as `""` or `[]`, so a reader can tell "the catalogue did not say"
// from "the catalogue said none". The drawer prints a row per fact and an empty
// string would print an empty row.
//
// The keys are the catalogue's own wire names. They are what the console
// already reads off a live catalogue row, so the same rendering works against
// either source without a translation layer in between.
func catalogueFactsOf(p *models.CatalogueProduct) map[string]any {
facts := map[string]any{}
if p == nil {
return facts
}
put := func(key, value string) {
if v := strings.TrimSpace(value); v != "" {
facts[key] = v
}
}
putList := func(key string, values []string) {
kept := make([]string, 0, len(values))
for _, v := range values {
if t := strings.TrimSpace(v); t != "" {
kept = append(kept, t)
}
}
if len(kept) > 0 {
facts[key] = kept
}
}
put("title", p.Title)
put("category", p.Category)
put("variant_key", p.VariantKey)
put("sku_source", p.SKUSource)
put("price_range", p.PriceRange)
put("fssai_license", p.FSSAILicense)
put("search_query", p.SearchQuery)
putList("providers", p.Providers)
putList("highlights", p.Highlights)
putList("nutrients", p.Nutrients)
return facts
}
// ImportCatalogueProduct bridges a global catalogue product (CatalogueDB) into
// a tenant's own store catalogue: it snapshots the catalogue product into the
// tenant's `products` table on first import (keyed on brand+catalogueid so
@@ -417,6 +486,27 @@ func (s *productService) ImportCatalogueProduct(reqs []models.ImportCataloguePro
Taxpercent: req.Taxpercent,
Approve: 1,
}
// Everything the snapshot has no column for, kept as the catalogue
// stated it.
//
// Ten of the catalogue's eighteen fields used to stop here. Two of
// them SHOULD — `category` is remapped to the platform's own
// categoryid, and `price_range` is replaced by the price the shop
// sets — but they are kept anyway, because what other retailers
// charge is the most useful thing on the drawer when somebody is
// deciding what to charge, and the catalogue's own category is how
// a mis-filed product gets noticed.
//
// Encoding failure is swallowed, like the images below: the product
// is worth creating without its facts, and refusing an import over
// a nutrition line would be the wrong trade.
if encoded, err := json.Marshal(catalogueFactsOf(catalogueProduct)); err == nil {
snapshot.Cataloguefacts = string(encoded)
} else {
log.Printf("import: could not encode catalogue facts for %s/%d: %v",
req.Brand, req.Catalogueid, err)
}
if len(catalogueProduct.Images) > 0 {
// The first stays where every reader already looks for it.
snapshot.Productimage = catalogueProduct.Images[0]

View File

@@ -1,6 +1,7 @@
package services
import (
"errors"
"testing"
"nearle/models"
@@ -47,6 +48,10 @@ type fakeProductRepo struct {
publishedRefs []models.ProductLocationRef
created []models.Products
categorySet map[int][2]int // productid -> {categoryid, subcategoryid}
// Set to make the insert fail, for the tests that check a failed create
// does not hand back a half-made product.
createErr error
}
func newFakeRepo() *fakeProductRepo {
@@ -119,6 +124,9 @@ func (f *fakeProductRepo) UpdateProductCategory(productid, categoryid, subcatego
// re-import branch, so this only became reachable when publishing did.
func (f *fakeProductRepo) CreateProductReturningID(product models.Products) (int, error) {
f.calls = append(f.calls, "CreateProductReturningID")
if f.createErr != nil {
return 0, f.createErr
}
f.created = append(f.created, product)
return 9001, nil
}
@@ -767,3 +775,72 @@ func TestPricingFilterDoesNotDisturbTheCallersSlice(t *testing.T) {
t.Error("the caller's slice was modified")
}
}
/* ── Creating a product hands back its id ───────────────────────────────────
*
* `POST /products/create` answered `productid: 0` for every product it created:
* the repository took the struct by value, GORM wrote the generated id onto
* that copy, and the copy went out of scope. The endpoint is the only way to
* create a product, and a product cannot be priced or stocked without its id,
* so every caller had to re-read the tenant's whole catalogue and find its own
* row again by SKU — a column nothing enforces, in an importer that creates
* duplicates by design.
*/
func TestCreateProductReturnsTheIdTheDatabaseAssigned(t *testing.T) {
repo := &fakeProductRepo{}
svc := NewProductService(repo, &fakeCatalogueService{})
created, err := svc.CreateProduct(models.Products{
Tenantid: 9001,
Productname: "Test Rice 5kg",
Productsku: "TM-RICE-5K",
})
if err != nil {
t.Fatalf("CreateProduct: %v", err)
}
// 9001 is what the fake's CreateProductReturningID returns. The point is
// that it reaches the caller at all — it used to be dropped.
if created.Productid != 9001 {
t.Errorf("productid = %d, want 9001 — the id was lost on the way out", created.Productid)
}
// The rest of the product survives the round trip, because the response is
// what the console shows and what it prices and stocks against.
if created.Productsku != "TM-RICE-5K" || created.Productname != "Test Rice 5kg" {
t.Errorf("the product came back altered: %+v", created)
}
}
func TestCreateProductGoesThroughTheOneCreatePath(t *testing.T) {
// There were two repository methods doing this same INSERT, one of which
// discarded the id. Only one remains, and this is what pins that: a second
// path would have to be added here to be used at all.
repo := &fakeProductRepo{}
svc := NewProductService(repo, &fakeCatalogueService{})
if _, err := svc.CreateProduct(models.Products{Tenantid: 9001}); err != nil {
t.Fatalf("CreateProduct: %v", err)
}
if len(repo.calls) != 1 || repo.calls[0] != "CreateProductReturningID" {
t.Errorf("want exactly one call to CreateProductReturningID, got %v", repo.calls)
}
}
func TestAFailedCreateReturnsNoProduct(t *testing.T) {
// The caller prices and stocks against what comes back, so a half-made
// product with a zero id would be worse than an error — it would send a
// price and a stock movement to product 0.
repo := &fakeProductRepo{createErr: errors.New("duplicate key")}
svc := NewProductService(repo, &fakeCatalogueService{})
created, err := svc.CreateProduct(models.Products{Tenantid: 9001, Productsku: "DUP"})
if err == nil {
t.Fatal("a failed insert was reported as a success")
}
if created.Productid != 0 || created.Productsku != "" {
t.Errorf("a product was returned for a failed create: %+v", created)
}
}

View File

@@ -48,18 +48,19 @@ const (
scanLookupTimeout = 5 * time.Second
scanMaxLabelLen = 200
scanCatalogueTopK = 15
// Below this the best hit is not shown as a match at all.
scanMinScore = 0.30
// How close the runner-up has to be before the leader stops being an
// answer and the two become a question. See isAmbiguous.
// Below this the best hit is not shown as a match at all. A correct label
// scores ~0.92 against its own product's vector and ~0.23 against an
// unrelated one, so the floor sits in the empty middle of that split
// rather than just above the unrelated band: at 0.30, "Paracetamol"
// came back as "Paneer Makhni 500ml" (0.304) — a near-miss on an
// unrelated row clears a floor set that close to the noise.
scanMinScore = 0.50
// How close the runner-up may be before the leader stops being an answer
// and the two become a question. See isAmbiguous.
scanAmbiguityMargin = 0.06
// A "did you mean?" list longer than this is not a choice, it is a
// catalogue — the customer is standing in a shop holding a packet.
scanMaxCandidates = 10
// How much of the winning product's name the label has to account for
// before it counts as having identified it. See isAmbiguous and
// textScore.
scanSpecificEnough = 0.55
)
// ScanErrors the controller maps to statuses. Everything else is a 500.
@@ -401,6 +402,14 @@ func (s *scanService) Confirm(ctx context.Context, req models.ScanConfirmRequest
}
resp.Store = chosen
// Distance on the store they tapped, for every outcome and not just the
// out-of-stock one below: the app renders this store from the reply it
// gets. The phone's fix is free to parse; the saved address costs a
// query, so it is only reached for on the path that also ranks other
// outlets. Without either, distanceKm leaves the -1 the repository set.
lat, lng, hasPos := utils.ParseLatLng(string(req.Latitude), string(req.Longitude))
chosen.DistanceKm = distanceKm(*chosen, lat, lng, hasPos)
row, err := s.repo.ProductAt(ctx, req.Tenantid, req.Locationid, req.Productid)
if err != nil {
return nil, err
@@ -428,10 +437,10 @@ func (s *scanService) Confirm(ctx context.Context, req models.ScanConfirmRequest
}
// The same product elsewhere, nearest first, with enough of it.
lat, lng, hasPos := utils.ParseLatLng(string(req.Latitude), string(req.Longitude))
if !hasPos {
if hl, hg, ok, err := s.repo.CustomerHome(ctx, req.Customerid); err == nil && ok {
lat, lng, hasPos = hl, hg, true
chosen.DistanceKm = distanceKm(*chosen, lat, lng, hasPos)
}
}
others := make([]models.ScanStore, 0, len(stores))
@@ -516,12 +525,7 @@ func (s *scanService) Stores(ctx context.Context, customerid int, latStr, lngStr
type scoredHit struct {
repositories.CatalogueHit
// score ranks; text says how specifically the label names THIS product.
// Kept apart because they answer different questions: a vector neighbour
// can rank first while the label ("britannia") names no one product, and
// only the second number knows that.
score float64
text float64
}
func (h scoredHit) toMatch(method string) models.ScanCatalogueMatch {
@@ -554,12 +558,11 @@ func (s *scanService) searchCatalogue(ctx context.Context, label string) ([]scor
method = "vector+text"
}
tokens := utils.SearchTokens(label)
if cached, ok := s.repo.CachedHits(ctx, method+":"+s.modelName(), label); ok {
return scoreCachedHits(cached, label, tokens), method, nil
return scoreCachedHits(cached), method, nil
}
tokens := utils.SearchTokens(label)
byKey := make(map[string]*scoredHit)
keyOf := func(h repositories.CatalogueHit) string { return h.Brand + "#" + fmt.Sprint(h.ID) }
@@ -599,15 +602,10 @@ func (s *scanService) searchCatalogue(ctx context.Context, label string) ([]scor
for _, h := range thits {
ts := textScore(h, label, tokens)
if existing, ok := byKey[keyOf(h)]; ok {
// The bonus is proportional: only a text match that actually
// names the product confirms a vector hit. A flat +0.10 let a
// bare brand name — which matches every one of that brand's
// products weakly — inflate all of them equally.
existing.score = math.Min(1, math.Max(existing.score, ts)+0.10*ts)
existing.text = ts
existing.score = math.Min(1, math.Max(existing.score, ts)+0.10)
continue
}
byKey[keyOf(h)] = &scoredHit{CatalogueHit: h, score: ts, text: ts}
byKey[keyOf(h)] = &scoredHit{CatalogueHit: h, score: ts}
}
hits := make([]scoredHit, 0, len(byKey))
@@ -629,32 +627,42 @@ func (s *scanService) searchCatalogue(ctx context.Context, label string) ([]scor
return hits, method, nil
}
// scoreCachedHits restores the ranking the cache holds, and recomputes the
// text score from the row itself — the cache carries one number per row, and
// recomputing costs nothing while leaving out the specificity signal would
// make every cached lookup read as ambiguous.
func scoreCachedHits(cached []repositories.CatalogueHit, label string, tokens []string) []scoredHit {
func scoreCachedHits(cached []repositories.CatalogueHit) []scoredHit {
hits := make([]scoredHit, 0, len(cached))
for _, c := range cached {
hits = append(hits, scoredHit{
CatalogueHit: c,
score: 1 - c.Distance,
text: textScore(c, label, tokens),
})
hits = append(hits, scoredHit{CatalogueHit: c, score: 1 - c.Distance})
}
sortHits(hits)
return hits
}
// sortHits ranks by blended score, then by the model's own similarity, and
// only then by name. Name alone used to break every tie, which quietly made
// punctuation decide relevance: "Parle Monaco Classic" sorts above "Parle-G
// Original …" because a space precedes a hyphen in ASCII, so equal-scoring
// crackers beat the biscuit that was actually scanned.
func sortHits(hits []scoredHit) {
sort.SliceStable(hits, func(i, j int) bool {
if hits[i].score != hits[j].score {
return hits[i].score > hits[j].score
}
di, dj := vectorRank(hits[i].Distance), vectorRank(hits[j].Distance)
if di != dj {
return di < dj
}
return hits[i].ProductName < hits[j].ProductName
})
}
// vectorRank orders by cosine distance, nearest first, with a row the model
// never saw (-1, text-only) sorting behind every row it did.
func vectorRank(d float64) float64 {
if d < 0 {
return math.MaxFloat64
}
return d
}
func (s *scanService) modelName() string {
if s.embedder == nil {
return "none"
@@ -674,79 +682,44 @@ func (s *scanService) embed(ctx context.Context, label string) ([]float32, error
return v, nil
}
// textScore is how well a catalogue row matches the words Lens read.
// textScore is how well a catalogue row's name matches the words Lens read.
// The whole label as a substring of the name is near-certain; otherwise the
// share of label words found in name+title, scaled so that "all of them"
// stops short of the substring case.
//
// Both directions count, and that is the whole point:
//
// - labelCoverage — how much of what the customer said this product
// accounts for. "Milk Bikis" against "Milk Bikis 100g" is all of it.
// - nameCoverage — how much of the product the label accounts for, which
// is what makes the match SPECIFIC. "britannia" explains one word of
// "Britannia Good Day Cashew Cookies", so it does not identify it.
//
// The score is their harmonic mean, so a high score needs both.
//
// This replaces `strings.Contains(name, label) → 0.95`, which asked only the
// first question. A bare brand name is a substring of every one of that
// brand's products, so all 258 Britannia rows scored 0.95, the tie broke
// alphabetically, and the customer was shown one arbitrary biscuit with
// "confidence": 0.95. Lens returns a bare wordmark often — it is usually the
// most legible thing on a packet — so that was not an edge case.
//
// Now those rows score ~0.33 and, crucially, score it EQUALLY, which is what
// isAmbiguous reads to answer "did you mean?" instead of guessing.
// A label that is a substring of MANY names — a bare brand, "britannia" —
// therefore scores them all 0.95, identically. That tie is not a flaw to
// score around: it is the signal, and isAmbiguous reads it to answer "did
// you mean?" rather than letting the sort order pick a winner.
func textScore(h repositories.CatalogueHit, label string, tokens []string) float64 {
name := strings.ToLower(h.ProductName)
hay := name + " " + strings.ToLower(h.Title)
label = strings.ToLower(strings.TrimSpace(label))
// The substring test compares separator-folded forms, so the brand's own
// punctuation does not decide the match: "Parle G", "Parle-G" and
// "ParleG" all have to reach "Parle-G Original Glucose Biscuits".
if label != "" {
foldedName, foldedLabel := utils.FoldSeparators(name), utils.FoldSeparators(label)
if foldedLabel != "" && strings.Contains(foldedName, foldedLabel) {
return 0.95
}
// Separators dropped rather than folded. Only for a label long enough
// that a run of letters means something — "lay" inside "malayalam" is
// not a match anyone wants.
if tight := utils.TightenLabel(label); len(tight) >= 4 && strings.Contains(utils.TightenLabel(name), tight) {
return 0.95
}
}
if len(tokens) == 0 {
return 0
}
hay := strings.ToLower(h.ProductName + " " + h.Title)
found := 0
for _, t := range tokens {
if strings.Contains(hay, t) {
found++
}
}
if found == 0 {
return 0
}
labelCoverage := float64(found) / float64(len(tokens))
// Pack sizes are dropped from both sides (SearchTokens), so "100g" never
// counts as a word the label failed to explain.
nameTokens := utils.SearchTokens(h.ProductName)
if len(nameTokens) == 0 {
return 0.5 * labelCoverage
}
explained := 0
for _, n := range nameTokens {
for _, t := range tokens {
if tokenMatch(n, t) {
explained++
break
}
}
}
if explained == 0 {
// Matched the title but not the name. Weak, not zero.
return 0.4 * labelCoverage
}
nameCoverage := float64(explained) / float64(len(nameTokens))
return 2 * labelCoverage * nameCoverage / (labelCoverage + nameCoverage)
}
// tokenMatch is equality, plus containment for words long enough that a
// shared prefix means something ("cookie"/"cookies", "chocolate"/"choco").
// Short tokens must match exactly, or "day" would match "daybreak".
func tokenMatch(a, b string) bool {
if a == b {
return true
}
if len(a) >= 5 && strings.Contains(b, a) {
return true
}
return len(b) >= 5 && strings.Contains(a, b)
return 0.8 * float64(found) / float64(len(tokens))
}
// productKey identifies a product across its pack sizes: the catalogue's own
@@ -781,28 +754,19 @@ func distinctProducts(hits []scoredHit) []scoredHit {
}
// isAmbiguous reports that naming the leader as THE match would be a guess
// dressed up as an answer. Two ways that happens:
// dressed up as an answer, because something else is level with it.
//
// 1. Something else is level with it. A margin rather than an absolute
// threshold, because what matters is not how high the best score is but
// whether anything is tied with it.
// 2. Nothing is level, but the label does not actually name a product —
// a bare brand, a generic word, or a spelling the catalogue does not
// carry. The leader may still rank first on vector similarity, and
// ranking first among vague matches is not identification.
// A margin rather than an absolute threshold: what matters is not how high
// the best score is but whether anything is tied with it. A bare brand name
// is a substring of every one of that brand's names, so textScore gives them
// all 0.95 — a perfect tie at a HIGH score, which no floor would catch.
//
// Erring towards asking is deliberate. Asking costs the customer one tap on
// a picture; guessing wrong costs them the wrong biscuit and costs us the
// belief that the scanner works. An exact product name still scores ~1.0 on
// specificity, so the common case is unaffected.
// belief that the scanner works. A label that names one product leaves the
// runner-up far behind, so the common case is unaffected.
func isAmbiguous(distinct []scoredHit) bool {
if len(distinct) < 2 {
return false
}
if distinct[1].score >= distinct[0].score-scanAmbiguityMargin {
return true
}
return distinct[0].text < scanSpecificEnough
return len(distinct) >= 2 && distinct[1].score >= distinct[0].score-scanAmbiguityMargin
}
// catalogueFamily is `of` and its other pack sizes, drawn from hits.

View File

@@ -453,11 +453,11 @@ Ambiguity.
Lens hands back whatever was most legible on the packet, and on a packet that
is very often the brand wordmark alone. "britannia" fits 258 catalogue rows
equally well, so there is no best one — and the old scoring said otherwise:
every product whose name contained the label scored 0.95, the tie broke
alphabetically, and the customer was shown one arbitrary biscuit with
"confidence": 0.95 and a price. These tests are the contract that it asks
instead.
equally well, so there is no best one. textScore gives all of them 0.95 —
correctly, the label IS in every one of those names — and with nothing to
read that tie, the sort order picked a winner and the customer was shown one
arbitrary biscuit with "confidence": 0.95 and a price. These tests are the
contract that it asks instead.
*/
// Three different Britannia products, of which the customer's stores stock
@@ -515,8 +515,12 @@ func TestABareBrandNameAsksInsteadOfGuessing(t *testing.T) {
t.Errorf("only Marie Gold is stocked, but %s reports available", c.ProductName)
}
}
if resp.Confidence >= 0.55 {
t.Errorf("confidence should record how weak the identification was, got %v", resp.Confidence)
// Note what confidence does NOT say here. The label appears verbatim in
// all three names, so relevance is high — and the answer is still a
// question. An app that gated on `confidence` instead of `ambiguous`
// would show a price for the wrong biscuit, which is the whole bug.
if resp.Confidence < 0.9 {
t.Errorf("a verbatim brand match scores high; %v suggests the scoring changed", resp.Confidence)
}
if !strings.Contains(resp.Message, "Which one") {
t.Errorf("the message should ask, got %q", resp.Message)
@@ -565,27 +569,31 @@ func TestASpecificLabelStillWinsOutright(t *testing.T) {
}
}
func TestTextScoreRewardsSpecificityNotJustOverlap(t *testing.T) {
// The property the ambiguity check rests on: a label that is a substring of
// several names scores them EQUALLY. Nothing downstream can tell "did you
// mean?" from "found it" if a formula breaks that tie on name length, word
// count or anything else incidental — which is how one arbitrary Britannia
// biscuit used to come back with a price on it.
func TestABrandNameScoresItsProductsIdentically(t *testing.T) {
cashew := repositories.CatalogueHit{ProductName: "Britannia Good Day Cashew Cookies 200g"}
butter := repositories.CatalogueHit{ProductName: "Britannia Good Day Butter Cookies 100g"}
// Deliberately a much shorter name: length must not become a tie-breaker.
marie := repositories.CatalogueHit{ProductName: "Britannia Marie Gold"}
brandOnly := textScore(cashew, "britannia", utils.SearchTokens("britannia"))
if brandOnly > 0.45 {
t.Errorf("a brand name explains one word of five and must not score as an identification, got %.3f", brandOnly)
tokens := utils.SearchTokens("britannia")
a, b, c := textScore(cashew, "britannia", tokens), textScore(butter, "britannia", tokens), textScore(marie, "britannia", tokens)
if a != b || b != c {
t.Fatalf("a brand must score its products equally, got %.3f / %.3f / %.3f", a, b, c)
}
if brandOnly != textScore(butter, "britannia", utils.SearchTokens("britannia")) {
t.Error("a brand must score its products equally — that tie is what makes the label read as ambiguous")
if a == 0 {
t.Fatal("the brand name is in every one of those names; scoring it 0 would hide them all")
}
full := textScore(cashew, "Britannia Good Day Cashew Cookies", utils.SearchTokens("Britannia Good Day Cashew Cookies"))
if full < 0.95 {
t.Errorf("the product's own name should be near-certain, got %.3f", full)
}
// A pack size on either side is not a word the label failed to explain.
sized := textScore(repositories.CatalogueHit{ProductName: "Milk Bikis 100g"}, "Milk Bikis", utils.SearchTokens("Milk Bikis"))
if sized < 0.95 {
t.Errorf("pack sizes must not count against the match, got %.3f", sized)
// And a label that does name a product must NOT tie with its siblings,
// or everything would be a question.
specific := utils.SearchTokens("good day cashew")
if textScore(cashew, "good day cashew", specific) <= textScore(butter, "good day cashew", specific) {
t.Error("a label naming one product must outscore its siblings")
}
if none := textScore(cashew, "dabur honey", utils.SearchTokens("dabur honey")); none != 0 {
@@ -657,3 +665,120 @@ func TestAMissingCatalogueRefIsNotAMatch(t *testing.T) {
t.Errorf("the message should say the product is gone, got %q", resp.Message)
}
}
// A vector neighbour that is merely not-quite-unrelated used to clear the old
// 0.30 floor: in production "Paracetamol" came back as "Paneer Makhni 500ml"
// on a 0.304 similarity. Correct labels land near 0.92, so nothing this weak
// is a match.
func TestLookupRefusesANearMissAboveTheOldFloor(t *testing.T) {
repo := newLookupFixture()
repo.vector = []repositories.CatalogueHit{{Brand: "amul", ID: 4, ProductName: "Paneer Makhni 500ml", Distance: 0.696}} // score 0.304
repo.text = nil
svc := NewScanService(repo, fakeEmbedder{vec: []float32{0.1}})
resp, err := svc.Lookup(context.Background(), models.ScanLookupRequest{Customerid: 5, Label: "Paracetamol"})
if err != nil {
t.Fatal(err)
}
if resp.Match != nil {
t.Fatalf("0.304 is a near-miss, not a match; got %+v", resp.Match)
}
if resp.Available || len(resp.Stores) != 0 {
t.Fatalf("nothing should be offered without a match; got %+v", resp)
}
}
// Confirm answers about the store the customer tapped, so that store carries a
// distance on every outcome — not only on the out-of-stock path that ranks
// alternatives. Absent any position it stays -1, the documented "unknown".
func TestConfirmReportsDistanceToTheChosenStore(t *testing.T) {
repo := newLookupFixture()
repo.at = map[int]*repositories.StoreOptionRow{200: &repo.options[1]}
svc := NewScanService(repo, nil)
req := models.ScanConfirmRequest{Customerid: 5, Tenantid: 2, Locationid: 20, Productid: 200, Quantity: 4}
withPos := req
withPos.Latitude, withPos.Longitude = "11.035", "77.035"
resp, err := svc.Confirm(context.Background(), withPos)
if err != nil {
t.Fatal(err)
}
if !resp.Ok || resp.Store == nil {
t.Fatalf("expected the in-stock answer, got %+v", resp)
}
if resp.Store.DistanceKm <= 0 {
t.Fatalf("the phone sent a fix, so the tapped store has a distance; got %v", resp.Store.DistanceKm)
}
// No fix from the phone, but a saved address on file.
repo.homeLat, repo.homeLng, repo.homeOK = 11.035, 77.035, true
resp, err = svc.Confirm(context.Background(), req)
if err != nil {
t.Fatal(err)
}
if resp.Store == nil || resp.Store.DistanceKm != -1 {
t.Fatalf("in stock is answered without reaching for the saved address; got %v", resp.Store)
}
// Neither: unknown, and the app sorts it last.
repo.homeOK = false
resp, err = svc.Confirm(context.Background(), req)
if err != nil {
t.Fatal(err)
}
if resp.Store == nil || resp.Store.DistanceKm != -1 {
t.Fatalf("no position at all is -1; got %v", resp.Store)
}
}
var parleG = repositories.CatalogueHit{Brand: "parle", ID: 1, ProductName: "Parle-G Original Glucose Biscuits 250g", Title: "Parle-G", VariantKey: "parle_g", ImageID: "parle_parle_g_250g", Distance: 0.20}
var monaco = repositories.CatalogueHit{Brand: "parle", ID: 2, ProductName: "Parle Monaco Classic Regular 200g", Title: "Monaco", VariantKey: "monaco", ImageID: "parle_monaco_200g", Distance: 0.20}
// Lens reads "Parle-G" off the packet and the customer types "Parle G". Both
// spellings, and the run-together one, have to reach the biscuit — not the
// salted cracker that merely shares a brand. In production "Parle G" returned
// "Parle Monaco Classic Regular 200g" at a confident 0.9.
func TestLookupMatchesAHyphenatedNameHoweverItIsWritten(t *testing.T) {
for _, label := range []string{"Parle G", "Parle-G", "ParleG", "parle g"} {
repo := newLookupFixture()
repo.vector = []repositories.CatalogueHit{monaco, parleG} // model puts the cracker first
repo.text = []repositories.CatalogueHit{monaco, parleG}
svc := NewScanService(repo, fakeEmbedder{vec: []float32{0.1}})
resp, err := svc.Lookup(context.Background(), models.ScanLookupRequest{Customerid: 5, Label: label})
if err != nil {
t.Fatal(err)
}
if resp.Match == nil {
t.Fatalf("%q: a stocked product went unrecognised", label)
}
if resp.Match.Catalogueid != parleG.ID {
t.Fatalf("%q: matched %q (%.3f), want Parle-G", label, resp.Match.ProductName, resp.Match.Score)
}
}
}
// Equal blended scores used to be settled by product name, which let ASCII
// decide relevance: a space sorts before a hyphen, so "Parle Monaco …" beat
// "Parle-G …". The model's own similarity settles it instead.
func TestSortHitsBreaksTiesOnSimilarityNotPunctuation(t *testing.T) {
near := parleG
near.Distance = 0.10 // the model is surer about this one
far := monaco
far.Distance = 0.40
hits := []scoredHit{{CatalogueHit: far, score: 0.9}, {CatalogueHit: near, score: 0.9}}
sortHits(hits)
if hits[0].ID != near.ID {
t.Fatalf("the nearer vector should win a tie, got %q", hits[0].ProductName)
}
// A row the model never scored (-1, text-only) ranks behind one it did.
textOnly := parleG
textOnly.Distance = -1
hits = []scoredHit{{CatalogueHit: textOnly, score: 0.9}, {CatalogueHit: far, score: 0.9}}
sortHits(hits)
if hits[0].ID != far.ID {
t.Fatalf("a scored row outranks an unscored one, got %q", hits[0].ProductName)
}
}

View File

@@ -1,6 +1,8 @@
package services
import (
"errors"
"nearle/models"
"nearle/repositories"
"time"
@@ -22,6 +24,21 @@ func NewStockRequestService(repo repositories.StockRequestRepository, productSer
}
func (s *stockRequestService) CreateStockRequest(req *models.StockRequest) error {
// A request for nothing is not a request.
//
// Nothing downstream rejected it, so a branch could raise a request for
// zero units and it sat in the admin's queue looking exactly like a real
// one — and approving it moved no stock, which reads as the ledger being
// broken rather than the request being empty. A negative would move stock
// the wrong way on receipt, since UpdateStockRequest writes Qty straight
// into the ledger as an 'in'.
//
// Returned as an ordinary error: the controller already reports per-item
// reasons, so one bad row in a batch is named and the rest still land.
if req.Qty <= 0 {
return errors.New("quantity must be more than zero")
}
return s.repo.CreateStockRequest(req)
}