Compare commits
7 Commits
01bc89ab77
...
feat/env-l
| Author | SHA1 | Date | |
|---|---|---|---|
| 24339a8b51 | |||
| 692c10e553 | |||
| 2f1501883b | |||
| c06b029cb2 | |||
| 28af3e05f2 | |||
| 42ea007fe7 | |||
| 76bff883ec |
@@ -120,8 +120,12 @@ current example of all seven.
|
||||
- **One MQTT client id per replica.** A second connection with the same id
|
||||
evicts the first. Never run a local process with the production
|
||||
`MQTT_URL`.
|
||||
- **`scratch/` tools read production** when run with `.env.production`.
|
||||
They are read-only by construction; keep them that way.
|
||||
- **`scratch/` tools read production** when run with `.env.production`, and
|
||||
most are read-only. Two are not — `termbackfill` and
|
||||
`cataloguefactsbackfill` repair rows that no endpoint can reach. Both
|
||||
default to a dry run that prints every change and write only when passed
|
||||
`apply`, and both print the SQL to undo themselves afterwards. A new tool
|
||||
that writes follows that shape or it does not write.
|
||||
|
||||
## Operations cheat-sheet (Kubernetes, namespace `nearle`)
|
||||
|
||||
|
||||
@@ -191,7 +191,14 @@ func (ctl *ProductController) CreateProduct(c *fiber.Ctx) error {
|
||||
})
|
||||
}
|
||||
|
||||
if err := ctl.productService.CreateProduct(product); err != nil {
|
||||
// The created row, not the parsed body.
|
||||
//
|
||||
// This returned the struct it had just parsed off the request, which by
|
||||
// definition carried `productid: 0` — the id is assigned by the database a
|
||||
// moment later and was never read back. Every caller that needed the id
|
||||
// went and looked the product up again by SKU.
|
||||
created, err := ctl.productService.CreateProduct(product)
|
||||
if err != nil {
|
||||
return c.JSON(fiber.Map{
|
||||
"code": http.StatusInternalServerError,
|
||||
"message": "Failed to create product",
|
||||
@@ -203,7 +210,7 @@ func (ctl *ProductController) CreateProduct(c *fiber.Ctx) error {
|
||||
"code": http.StatusCreated,
|
||||
"message": "Product created successfully",
|
||||
"status": true,
|
||||
"data": product,
|
||||
"data": created,
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
@@ -44,6 +44,25 @@ func (ctl *TenantController) SearchTenant(c *fiber.Ctx) error {
|
||||
func (ctl *TenantController) GetAllTenants(c *fiber.Ctx) error {
|
||||
pageno, _ := strconv.Atoi(c.Query("pageno"))
|
||||
pagesize, _ := strconv.Atoi(c.Query("pagesize"))
|
||||
|
||||
// Paging is defaulted, not required.
|
||||
//
|
||||
// The repository builds LIMIT/OFFSET from these directly, so a caller that
|
||||
// omitted either — or sent pageno=0 — got an empty result reported as
|
||||
// `code 200, status true, message "Success"`. "There are no tenants on the
|
||||
// platform" and "you forgot a query parameter" are very different answers
|
||||
// and this endpoint gave the first for the second.
|
||||
//
|
||||
// Defaulted rather than rejected with a 400: every existing caller that
|
||||
// works today keeps working, and a platform list with no paging asked for
|
||||
// has an obvious right answer — the first page.
|
||||
if pageno < 1 {
|
||||
pageno = 1
|
||||
}
|
||||
if pagesize < 1 {
|
||||
pagesize = 50
|
||||
}
|
||||
|
||||
status := c.Query("status")
|
||||
aid, _ := strconv.Atoi(c.Query("applocationid"))
|
||||
tenanttype := c.Query("tenanttype")
|
||||
|
||||
@@ -125,13 +125,13 @@ registered stores are used.
|
||||
"ambiguous": true,
|
||||
"candidates": [
|
||||
{ "brand": "britannia", "catalogueid": 23, "product_name": "Britannia Marie Gold",
|
||||
"size": "250 g", "image": "https://…", "score": 0.5, "method": "text", "available": true },
|
||||
"size": "250 g", "image": "https://…", "score": 0.95, "method": "text", "available": true },
|
||||
{ "brand": "britannia", "catalogueid": 22, "product_name": "Britannia Good Day Butter Cookies",
|
||||
"image": "https://…", "score": 0.333, "method": "text" },
|
||||
"image": "https://…", "score": 0.95, "method": "text" },
|
||||
{ "brand": "britannia", "catalogueid": 21, "product_name": "Britannia Good Day Cashew Cookies",
|
||||
"image": "https://…", "score": 0.333, "method": "text" }
|
||||
"image": "https://…", "score": 0.95, "method": "text" }
|
||||
],
|
||||
"confidence": 0.5,
|
||||
"confidence": 0.95,
|
||||
"available": false,
|
||||
"stores": [],
|
||||
"catalogue_variants": [],
|
||||
@@ -139,6 +139,11 @@ registered stores are used.
|
||||
}
|
||||
```
|
||||
|
||||
- **`confidence` is not low here, and that is not a bug.** "britannia" really
|
||||
does appear in all three names, so relevance is high — what is missing is
|
||||
*identification*. Gate on `ambiguous`, never on `confidence`: an app that
|
||||
reads 0.95 as "sure enough to show a price" reintroduces the exact bug this
|
||||
path exists to prevent.
|
||||
- **`available` on a candidate** means at least one of the customer's
|
||||
registered stores has it in stock right now. Candidates are ordered
|
||||
available-first, so the list can show what is buyable before what is not
|
||||
@@ -171,6 +176,10 @@ registered stores are used.
|
||||
cart/order calls exactly as you would from the catalogue screen.
|
||||
- `distance_km: -1` means the distance is unknown (no fix from the phone and
|
||||
no saved address, or the store has no coordinates). Do not render it as 0.
|
||||
Send `latitude`/`longitude` on `/confirm` too if you display distance from
|
||||
its reply: the saved address is only consulted there when the shelf is
|
||||
empty and alternatives have to be ranked, so without a fix the store you
|
||||
tapped comes back `-1`.
|
||||
|
||||
## `POST /confirm`
|
||||
|
||||
@@ -186,7 +195,7 @@ nothing is cached on this path.
|
||||
{
|
||||
"ok": false,
|
||||
"reason": "out_of_stock", // in_stock | insufficient_stock | out_of_stock | not_sold_here | store_not_registered
|
||||
"store": { "…the store they tapped…" },
|
||||
"store": { "…the store they tapped…" }, // distance_km filled from the fix you send
|
||||
"option": { "productid": 100, "stock": 0, "…": "…" },
|
||||
"requested": 2,
|
||||
"alternative": { // absent when nobody has enough
|
||||
@@ -225,6 +234,18 @@ Same `ScanStore` shape as inside `stores[]` above, without options.
|
||||
`EMBEDDING_PROVIDER/MODEL/API_KEY` and **must** be the one that indexed
|
||||
the catalogue — the first search checks the vector width and refuses a
|
||||
mismatch by name.
|
||||
- **The word match asks for most of the label, not all of it**
|
||||
(`minTokenHits`: two thirds, rounded up, and both of a two-word label).
|
||||
Requiring every word meant one word the catalogue does not use took the
|
||||
right product out of the running entirely — "Dettol bottle pack" retrieved
|
||||
no Dettol, "Parle G biscuit pack" retrieved no Parle-G — and the vector
|
||||
search then answered alone, confidently and wrongly, at a score the floor
|
||||
could not catch. Each brand's rows are ordered by how much of the label
|
||||
they carry (the whole label as a substring outranks any number of loose
|
||||
words) so that the per-brand `LIMIT` keeps the best rows and not merely the
|
||||
first ones the planner reached. Packaging words — "pack", "bottle", "jar",
|
||||
"sachet" and friends, see `utils.isPackaging` — are dropped before any of
|
||||
this, like pack sizes, unless the label is nothing else.
|
||||
- **The catalogue's model** (verified 2026-09-15 by cosine against a stored
|
||||
row: 1.0000): `all-MiniLM-L6-v2`, 384-d, unit-normalised, embedding the
|
||||
`search_query` column (brand + name + category + blurb + price range).
|
||||
@@ -238,10 +259,12 @@ Same `ScanStore` shape as inside `stores[]` above, without options.
|
||||
EMBEDDING_DIMENSIONS=384
|
||||
```
|
||||
A bare label ("Milk Bikis") scores ~0.92 against its product's stored
|
||||
vector and ~0.23 against an unrelated one, which is what the 0.30 floor in
|
||||
`scanService.go` is set against. If the catalogue team ever re-embeds
|
||||
with another model, change `EMBEDDING_MODEL`/`DIMENSIONS` here and
|
||||
nothing else.
|
||||
vector and ~0.23 against an unrelated one, which is what the 0.50 floor in
|
||||
`scanService.go` is set against — the middle of that split, not the edge of
|
||||
the noise. It was 0.30 until a near-miss got through in production
|
||||
("Paracetamol" → "Paneer Makhni 500ml", 0.304). If the catalogue team ever
|
||||
re-embeds with another model, change `EMBEDDING_MODEL`/`DIMENSIONS` here
|
||||
and nothing else.
|
||||
- **Speed**: the label's vector (7 days) and the ranked catalogue hits
|
||||
(30 min) are cached in Redis and in-process, so a popular product costs
|
||||
one model call platform-wide. Customer, stores and catalogue are read
|
||||
@@ -358,34 +381,44 @@ brand-name bug described under Scoring.
|
||||
|---|---|---|
|
||||
| `scanLookupTimeout` | 5 s | whole lookup, including the model call |
|
||||
| `scanCatalogueTopK` | 15 | rows taken from each brand table and from the merge |
|
||||
| `scanMinScore` | 0.30 | below this the best hit is not shown as a match |
|
||||
| `scanMinScore` | 0.50 | below this the best hit is not shown as a match |
|
||||
| `scanAmbiguityMargin` | 0.06 | how close the runner-up may be before the answer becomes a question |
|
||||
| `scanSpecificEnough` | 0.55 | text coverage the leader needs before it counts as identified |
|
||||
| `scanMaxCandidates` | 10 | longest "did you mean?" list |
|
||||
| `embedTimeout` (`utils/embedding.go`) | 4 s | one model call |
|
||||
| `scanVectorTTL` / `scanHitsTTL` (`scanRepository.go`) | 7 d / 30 min | cache lifetimes |
|
||||
|
||||
**Scoring.** Each hit carries two numbers, and they answer different
|
||||
questions:
|
||||
Scores: vector = `1 − cosine distance`; text = 0.95 for the whole label
|
||||
inside the name, else `0.8 × (label words found / label words)`; combined =
|
||||
`max(vector, text) + 0.10` when both hit, capped at 1. Ties are broken by
|
||||
cosine distance — nearest first, a text-only row last — and only then by
|
||||
name.
|
||||
|
||||
- `score` ranks. Vector = `1 − cosine distance`. Combined =
|
||||
`max(vector, text) + 0.10 × text`, capped at 1 — the confirmation bonus is
|
||||
proportional, so only a text match that actually names the product
|
||||
strengthens a vector hit.
|
||||
- `text` says how *specifically* the label names this product, and is the
|
||||
harmonic mean of two coverages: how much of the label the product accounts
|
||||
for, and how much of the product's name the label accounts for. Pack sizes
|
||||
are dropped from both sides.
|
||||
The label and the product name are both separator-folded before that
|
||||
substring test (`utils.FoldSeparators`), and compared again with separators
|
||||
removed (`utils.TightenLabel`, labels of 4+ characters), so the brand's own
|
||||
punctuation does not decide the match: "Parle G", "Parle-G" and "ParleG" all
|
||||
reach *Parle-G Original Glucose Biscuits*. A single-character token survives
|
||||
tokenising when it follows a word, because it is often the whole name — the
|
||||
"G" of Parle-G, the "K" of Special K. It is still dropped when it stands
|
||||
alone or is a pack multiplier.
|
||||
|
||||
Why both: `"britannia"` is a substring of all 258 Britannia product names.
|
||||
Judging on overlap alone scored every one of them 0.95, the tie broke
|
||||
alphabetically, and one arbitrary biscuit came back with a price. Now they
|
||||
score ~0.33 *equally*, which `isAmbiguous` reads as "ask, don't guess" — via
|
||||
the margin test (something is level with the leader) or the specificity test
|
||||
(the leader may rank first on vector similarity while the label names no one
|
||||
product). Erring towards asking is deliberate: asking costs one tap on a
|
||||
picture, guessing wrong costs the customer's belief that the scanner works.
|
||||
An exact product name still scores ~1.0, so the common case is untouched.
|
||||
All three mattered at once: before this, "Parle G" tied with *Parle Monaco
|
||||
Classic* at 0.9 (the "G" was dropped, so only "parle" matched either row),
|
||||
and the name tie-break handed it to Monaco because a space precedes a hyphen
|
||||
in ASCII. A confident, wrong answer — the kind no score floor can catch.
|
||||
|
||||
**When the substring rule ties, that tie is the answer.** A bare brand name
|
||||
is a substring of every one of that brand's names, so all of them score 0.95
|
||||
— identically, at a high score no floor would ever catch. Rather than
|
||||
scoring around it, `isAmbiguous` reads it: if the runner-up is within
|
||||
`scanAmbiguityMargin` of the leader, the reply becomes `ambiguous: true`
|
||||
with `candidates` instead of a match (see Response B). Erring towards asking
|
||||
is deliberate — one tap on a picture against the wrong biscuit. A label that
|
||||
names one product leaves the runner-up far behind, so the common case is
|
||||
untouched, and `services/scan_test.go`'s
|
||||
`TestABrandNameScoresItsProductsIdentically` guards the tie itself: a
|
||||
formula that broke it on name length or word count would bring the bug
|
||||
back.
|
||||
|
||||
### Changing the embedding model
|
||||
|
||||
|
||||
@@ -39,11 +39,41 @@ inside a git repository. `.gitignore` excludes `*.sql` here for that reason.
|
||||
|
||||
## Getting something to test against
|
||||
|
||||
An empty schema boots but has no tenants, so there is nothing to sign in as.
|
||||
Two options:
|
||||
`nearledb/02-seed.sql` is committed and applied automatically, so a fresh
|
||||
volume already has a merchant to sign into. It invents one rather than copying
|
||||
one, which is why it can live here at all.
|
||||
|
||||
- **Onboard a tenant through the console** once it is pointed at localhost.
|
||||
That exercises the real path and is usually what you want.
|
||||
- **Copy a few rows** you actually need — a tenant, its locations, its
|
||||
app_users — with `pg_dump --data-only --table=...`. Check what you are
|
||||
copying: `app_users.password` is stored in clear.
|
||||
| Account | Password | Opens |
|
||||
|---|---|---|
|
||||
| `super@nearle.invalid` | `localdev` | Nearle Admin — the platform workspace |
|
||||
| `admin@testmart.invalid` | `localdev` | Store Admin — all of Testmart's branches |
|
||||
| `main@testmart.invalid` | `localdev` | Store user — Testmart Main only |
|
||||
|
||||
It also seeds the role ladder, three aisles under category 2, and a second
|
||||
merchant (`Halfmart`) deliberately left in the broken `categoryid = 0` shape as
|
||||
a permanent regression fixture. The sequences are moved past the seeded ids at
|
||||
the end, so the first row you create locally does not come back as id 1.
|
||||
|
||||
If you need something it does not cover:
|
||||
|
||||
- **Onboard a tenant through the console.** That exercises the real path and is
|
||||
usually what you want.
|
||||
- **Copy a few rows** you actually need with `pg_dump --data-only --table=...`.
|
||||
Check what you are copying: `app_users.password` is stored in clear.
|
||||
|
||||
## The catalogue database
|
||||
|
||||
`cataloguedb/02-seed.sql` is committed too, and also entirely invented. The
|
||||
real catalogue is another team's scrape of real retailers and a dump of it does
|
||||
not belong on a laptop.
|
||||
|
||||
Without it the catalogue database exists but holds no catalogue: every
|
||||
`brand_*` table is missing, `getbrands` answers 500, and the global catalogue
|
||||
screen, the import flow and `importcatalogueproduct` cannot be exercised at
|
||||
all. The seed gives you two brands:
|
||||
|
||||
- `brand_testbrand` — every column the reader knows about, four products, one
|
||||
of them deliberately with no images.
|
||||
- `brand_sparsebrand` — only `id`, `product_name` and a price, to keep the
|
||||
degraded-but-still-listed path covered. Brands are discovered by table name,
|
||||
so adding another is just another `brand_*` table.
|
||||
|
||||
109
init/cataloguedb/02-seed.sql
Normal file
109
init/cataloguedb/02-seed.sql
Normal file
@@ -0,0 +1,109 @@
|
||||
-- A synthetic global catalogue to develop against.
|
||||
--
|
||||
-- INVENTED DATA, exactly like `nearledb/02-seed.sql` and for the same reason:
|
||||
-- the real catalogue is somebody else's scrape of real retailers, and a dump of
|
||||
-- it does not belong on a laptop inside a git repository.
|
||||
--
|
||||
-- ── Why this file has to exist ──────────────────────────────────────────────
|
||||
--
|
||||
-- `init/cataloguedb/` was empty, so a local stack had a catalogue DATABASE with
|
||||
-- no catalogue in it. Every `brand_*` table was missing, `getbrands` answered
|
||||
-- 500, and the whole catalogue-import path — the global catalogue screen, the
|
||||
-- import flow, `importcatalogueproduct` — could not be exercised locally at
|
||||
-- all. It is a documented feature with its own integration doc and it had no
|
||||
-- local coverage whatsoever.
|
||||
--
|
||||
-- ── The shape ───────────────────────────────────────────────────────────────
|
||||
--
|
||||
-- Brands are discovered from `information_schema` by table name, so a table
|
||||
-- called `brand_<something>` IS a brand; there is no registry to add it to.
|
||||
-- `catalogueCoreColumns` requires only `id` and `product_name` — everything
|
||||
-- else is selected when present and replaced with NULL when absent, so a
|
||||
-- partial table degrades rather than disappearing. These two are written full
|
||||
-- so that the degraded path is a deliberate test, not the only thing available:
|
||||
-- `brand_testbrand` has every column, and `brand_sparsebrand` deliberately has
|
||||
-- only the core two plus a price, to exercise `columnsFor`.
|
||||
--
|
||||
-- `image_id` is the durable key across re-scrapes — catalogue ids are not
|
||||
-- stable and `models.Products.Imageid` is what the import stores — so every
|
||||
-- product here has one and they are distinct.
|
||||
|
||||
CREATE EXTENSION IF NOT EXISTS vector;
|
||||
|
||||
-- ── A brand with the full column set ────────────────────────────────────────
|
||||
CREATE TABLE IF NOT EXISTS brand_testbrand (
|
||||
id BIGSERIAL PRIMARY KEY,
|
||||
product_name TEXT NOT NULL,
|
||||
title TEXT,
|
||||
description TEXT,
|
||||
category TEXT,
|
||||
image_id TEXT,
|
||||
size TEXT,
|
||||
variant_key TEXT,
|
||||
product_sku TEXT,
|
||||
sku_source TEXT,
|
||||
-- A RANGE, not a price. The global catalogue carries what retailers were
|
||||
-- seen charging; the shop sets its own price at import time, which is why
|
||||
-- the console collects one before an import can be enabled.
|
||||
price_range TEXT,
|
||||
providers TEXT[],
|
||||
fssai_license TEXT,
|
||||
highlights TEXT[],
|
||||
nutrients TEXT[],
|
||||
search_query TEXT,
|
||||
image_url TEXT,
|
||||
image_urls TEXT[],
|
||||
created_at TIMESTAMPTZ DEFAULT NOW(),
|
||||
updated_at TIMESTAMPTZ DEFAULT NOW()
|
||||
);
|
||||
|
||||
INSERT INTO brand_testbrand
|
||||
(product_name, title, description, category, image_id, size, variant_key,
|
||||
product_sku, sku_source, price_range, providers, fssai_license,
|
||||
highlights, nutrients, search_query, image_url, image_urls)
|
||||
VALUES
|
||||
('Testbrand Basmati Rice 5kg', 'Testbrand Basmati Rice', 'Long grain basmati, aged twelve months.',
|
||||
'Rice & Grains', 'IMG-TB-RICE-5K', '5 kg', 'rice-5kg', 'TB-RICE-5K', 'scrape',
|
||||
'380-420', ARRAY['bigbasket','amazon'], '12345678901234',
|
||||
ARRAY['Aged 12 months','Extra long grain'], ARRAY['Energy 350kcal','Protein 7g'],
|
||||
'basmati rice 5kg', 'https://placehold.co/300x300?text=Rice5kg',
|
||||
ARRAY['https://placehold.co/300x300?text=Rice5kg','https://placehold.co/300x300?text=Rice5kg-back']),
|
||||
|
||||
('Testbrand Basmati Rice 1kg', 'Testbrand Basmati Rice', 'Long grain basmati, aged twelve months.',
|
||||
'Rice & Grains', 'IMG-TB-RICE-1K', '1 kg', 'rice-1kg', 'TB-RICE-1K', 'scrape',
|
||||
'85-99', ARRAY['bigbasket'], '12345678901234',
|
||||
ARRAY['Aged 12 months'], ARRAY['Energy 350kcal','Protein 7g'],
|
||||
'basmati rice 1kg', 'https://placehold.co/300x300?text=Rice1kg',
|
||||
ARRAY['https://placehold.co/300x300?text=Rice1kg']),
|
||||
|
||||
('Testbrand Sunflower Oil 1L', 'Testbrand Sunflower Oil', 'Refined sunflower oil, light and neutral.',
|
||||
'Oils & Ghee', 'IMG-TB-OIL-1L', '1 L', 'oil-1l', 'TB-OIL-1L', 'scrape',
|
||||
'150-185', ARRAY['bigbasket','jiomart'], '99999999999999',
|
||||
ARRAY['Vitamin E','Light frying'], ARRAY['Energy 900kcal','Fat 100g'],
|
||||
'sunflower oil 1 litre', 'https://placehold.co/300x300?text=Oil1L',
|
||||
ARRAY['https://placehold.co/300x300?text=Oil1L']),
|
||||
|
||||
-- No images at all. `ImportCatalogueProduct` only sets `productimages` when
|
||||
-- the product has photos, so this row is the one that proves an import still
|
||||
-- works when it does not — the case that used to hit the jsonb empty-string
|
||||
-- failure in `products`.
|
||||
('Testbrand Salt 1kg', 'Testbrand Iodised Salt', 'Free-flowing iodised salt.',
|
||||
'Everyday', 'IMG-TB-SALT-1K', '1 kg', 'salt-1kg', 'TB-SALT-1K', 'scrape',
|
||||
'20-28', ARRAY['jiomart'], NULL,
|
||||
NULL, NULL, 'iodised salt 1kg', NULL, NULL);
|
||||
|
||||
-- ── A brand with only the core columns ──────────────────────────────────────
|
||||
--
|
||||
-- Discovery used to demand all eighteen columns, which made a table like this
|
||||
-- INVISIBLE rather than merely thin — 16 of 35 live brands were unreachable
|
||||
-- from this side for exactly that reason. Keeping one here means the
|
||||
-- degraded-but-listed path is covered by the seed and stays covered.
|
||||
CREATE TABLE IF NOT EXISTS brand_sparsebrand (
|
||||
id BIGSERIAL PRIMARY KEY,
|
||||
product_name TEXT NOT NULL,
|
||||
price_range TEXT
|
||||
);
|
||||
|
||||
INSERT INTO brand_sparsebrand (product_name, price_range) VALUES
|
||||
('Sparsebrand Biscuits 100g', '20-30'),
|
||||
('Sparsebrand Tea 250g', '110-140');
|
||||
@@ -161,4 +161,72 @@ INSERT INTO productstocks (
|
||||
(9504, 9001, 9102, 9301, NOW(), 'in', 12, 'Active')
|
||||
ON CONFLICT (productstockid) DO NOTHING;
|
||||
|
||||
-- ── The platform operator ───────────────────────────────────────────────────
|
||||
--
|
||||
-- Without this there is nobody who can open the Nearle Admin workspace, which
|
||||
-- is the one this console was built for first. `resolveRole` checks
|
||||
-- `issuperadmin` BEFORE roleid — deliberately, because the flag is derived by
|
||||
-- the server and a roleid is just a number in a row — so no amount of role 1
|
||||
-- gets you in without it, and every local session landed in Store Admin
|
||||
-- instead. The accounts above are one per role and this was the role they were
|
||||
-- missing.
|
||||
--
|
||||
-- Not attached to either merchant in spirit, only in columns: a platform
|
||||
-- operator has to carry a tenantid because the column is not nullable, and
|
||||
-- nothing in the admin workspace reads it.
|
||||
INSERT INTO app_users (
|
||||
userid, authname, firstname, lastname, email, dialcode, contactno,
|
||||
configid, roleid, password, tenantid, locationid, applocationid,
|
||||
status, issuperadmin
|
||||
) VALUES
|
||||
(9299, 'super@nearle.invalid', 'Nearle', 'Operator', 'super@nearle.invalid',
|
||||
'+91', '9000009999', 1, 1, 'localdev', 9001, 9101, 9001, 'Active', true)
|
||||
ON CONFLICT (userid) DO NOTHING;
|
||||
|
||||
-- ── The role ladder ─────────────────────────────────────────────────────────
|
||||
--
|
||||
-- `getstaffs` LEFT JOINs app_roles for `rolename`, so an empty table is not an
|
||||
-- error — every person on Users & access simply reads "—" where their role
|
||||
-- should be. The ids are the ones the rest of the system already assumes:
|
||||
-- 1 and 3 reach Store Admin, 4 is a branch manager, 7 and 8 are till accounts
|
||||
-- and are excluded from every back-office query by the backend itself.
|
||||
INSERT INTO app_roles (roleid, rolename, configid) VALUES
|
||||
(1, 'Super admin', 1),
|
||||
(3, 'Admin', 1),
|
||||
(4, 'Manager', 1),
|
||||
(7, 'Supervisor', 1),
|
||||
(8, 'Cashier', 1)
|
||||
ON CONFLICT (roleid) DO NOTHING;
|
||||
|
||||
-- ── Aisles under the category the customer app browses ──────────────────────
|
||||
--
|
||||
-- categoryid 2 is the only category the app lists, and the aisle a shopper
|
||||
-- reads is the SUBCATEGORY. With none of these the sheet importer has nothing
|
||||
-- to resolve a row's category against, so every imported product falls back to
|
||||
-- subcategoryid 0 and lands under "Uncategorized".
|
||||
INSERT INTO productsubcategories (subcatid, categoryid, tenantid, subcatname, status, sortorder)
|
||||
VALUES
|
||||
(9601, 2, 9001, 'Rice & Grains', 'Active', 1),
|
||||
(9602, 2, 9001, 'Oils & Ghee', 'Active', 2),
|
||||
(9603, 2, 9001, 'Snacks', 'Active', 3)
|
||||
ON CONFLICT (subcatid) DO NOTHING;
|
||||
|
||||
-- ── Move the sequences past the seeded ids ──────────────────────────────────
|
||||
--
|
||||
-- Everything above inserts an explicit id, which does NOT advance the sequence
|
||||
-- behind that column. So the first tenant, outlet or product created against a
|
||||
-- fresh local database came back as id 1 — harmless here, but it means local
|
||||
-- ids look nothing like the ones the same code produces in production, and a
|
||||
-- seed that ever collides with a sequence value fails on a duplicate key.
|
||||
--
|
||||
-- `GREATEST(..., 1)` because setval refuses a value below the sequence minimum,
|
||||
-- and a table the seed does not touch is legitimately empty.
|
||||
SELECT setval('tenants_tenantid_seq', GREATEST((SELECT COALESCE(MAX(tenantid),0) FROM tenants), 1));
|
||||
SELECT setval('tenantlocations_locationid_seq', GREATEST((SELECT COALESCE(MAX(locationid),0) FROM tenantlocations), 1));
|
||||
SELECT setval('app_users_userid_seq', GREATEST((SELECT COALESCE(MAX(userid),0) FROM app_users), 1));
|
||||
SELECT setval('products_productid_seq', GREATEST((SELECT COALESCE(MAX(productid),0) FROM products), 1));
|
||||
SELECT setval('productlocations_productlocationid_seq', GREATEST((SELECT COALESCE(MAX(productlocationid),0) FROM productlocations), 1));
|
||||
SELECT setval('productstocks_productstockid_seq', GREATEST((SELECT COALESCE(MAX(productstockid),0) FROM productstocks), 1));
|
||||
SELECT setval('customers_customerid_seq', GREATEST((SELECT COALESCE(MAX(customerid),0) FROM customers), 1));
|
||||
|
||||
COMMIT;
|
||||
|
||||
86
main.go
86
main.go
@@ -79,6 +79,38 @@ func main() {
|
||||
log.Println("⚠️ could not add products.productimages, extra photos will not be stored:", err)
|
||||
}
|
||||
|
||||
// What the global catalogue knew about this product, kept.
|
||||
//
|
||||
// The import copies eight of the catalogue's eighteen fields onto the
|
||||
// tenant's product and left the other ten behind — among them the FSSAI
|
||||
// licence, the nutrition lines, the highlights, the provider list, the
|
||||
// price range and the variant key. The console needs exactly those to
|
||||
// decide what to charge, so `ProductDrawer` went back to the catalogue for
|
||||
// them on every open.
|
||||
//
|
||||
// That lookup is not a substitute for storing them. A tenant's product is a
|
||||
// SNAPSHOT and outlives its source row: the catalogue is re-scraped, a
|
||||
// variant is retired, and the licence number and the nutrition panel for a
|
||||
// product the shop is still selling are gone with no way back. Measured
|
||||
// locally by retiring one row — the product survived, everything the drawer
|
||||
// shows about it did not.
|
||||
//
|
||||
// One jsonb column rather than six typed ones, and rather than the
|
||||
// `productspecs` table that has sat unused since the schema was written.
|
||||
// The value is a snapshot of somebody else's record, read as a whole and
|
||||
// displayed as a whole — it is never joined, aggregated or filtered — and
|
||||
// the catalogue grows fields faster than this side can add migrations.
|
||||
// Postgres can still reach inside it (`cataloguefacts->>'fssai_license'`)
|
||||
// on the day somebody needs to. `productimages` beside it made the same
|
||||
// call for the same reason.
|
||||
//
|
||||
// Not fatal on failure, exactly like the column above: a product without
|
||||
// its catalogue facts is the product we have today.
|
||||
if err := db.DB.Exec(
|
||||
`ALTER TABLE products ADD COLUMN IF NOT EXISTS cataloguefacts jsonb`).Error; err != nil {
|
||||
log.Println("⚠️ could not add products.cataloguefacts, catalogue detail will not survive a re-scrape:", err)
|
||||
}
|
||||
|
||||
// When a product became visible to a store, and the only thing that decides
|
||||
// whether it is.
|
||||
//
|
||||
@@ -173,6 +205,60 @@ func main() {
|
||||
log.Println("productvariants.variantid given a key generator (one time)")
|
||||
}
|
||||
|
||||
// Key generators for the two partner tables, for exactly the reason above.
|
||||
//
|
||||
// `partnerinfo.partnerid` and `partnerlocations.partnerlocationid` are both
|
||||
// NOT NULL with no default and no identity, so GORM — which sends nothing
|
||||
// for a key it expects the database to mint — had every insert refused with
|
||||
// a not-null violation. `createpartner` therefore could not write a partner
|
||||
// OR its regions: the endpoint exists, the form exists, and the row could
|
||||
// never land. The five partners on the platform were all inserted by hand,
|
||||
// which is the symptom rather than a choice.
|
||||
//
|
||||
// This matters more than one broken button. `GetPartners` now separates the
|
||||
// partners registered through this console from the ones another product
|
||||
// left in the shared `partnerinfo` by joining `partnerlocations` — and only
|
||||
// a successful create writes that table. Without a key generator no partner
|
||||
// can ever be registered, so nothing would ever have a link row and the
|
||||
// Rider partners page would be empty forever.
|
||||
//
|
||||
// Both sequences start above the ids already there, so the hand-inserted
|
||||
// rows keep theirs.
|
||||
for _, key := range []struct{ table, column string }{
|
||||
{"partnerinfo", "partnerid"},
|
||||
{"partnerlocations", "partnerlocationid"},
|
||||
} {
|
||||
var keyed int64
|
||||
if err := db.DB.Raw(`
|
||||
SELECT COUNT(1) FROM information_schema.columns
|
||||
WHERE table_name = ? AND column_name = ?
|
||||
AND (column_default IS NOT NULL OR is_identity = 'YES')`,
|
||||
key.table, key.column).Scan(&keyed).Error; err != nil {
|
||||
log.Fatalf("could not check %s.%s: %v", key.table, key.column, err)
|
||||
}
|
||||
if keyed > 0 {
|
||||
continue
|
||||
}
|
||||
|
||||
seq := key.table + "_" + key.column + "_seq"
|
||||
if err := db.DB.Exec(fmt.Sprintf(
|
||||
`CREATE SEQUENCE IF NOT EXISTS %s START WITH 1 OWNED BY %s.%s`,
|
||||
seq, key.table, key.column)).Error; err != nil {
|
||||
log.Fatalf("could not create %s: %v", seq, err)
|
||||
}
|
||||
if err := db.DB.Exec(fmt.Sprintf(
|
||||
`SELECT setval('%s', COALESCE((SELECT MAX(%s) FROM %s), 0) + 1, false)`,
|
||||
seq, key.column, key.table)).Error; err != nil {
|
||||
log.Fatalf("could not position %s: %v", seq, err)
|
||||
}
|
||||
if err := db.DB.Exec(fmt.Sprintf(
|
||||
`ALTER TABLE %s ALTER COLUMN %s SET DEFAULT nextval('%s')`,
|
||||
key.table, key.column, seq)).Error; err != nil {
|
||||
log.Fatalf("could not default %s.%s: %v", key.table, key.column, err)
|
||||
}
|
||||
log.Printf("%s.%s given a key generator (one time)", key.table, key.column)
|
||||
}
|
||||
|
||||
// The catalogue's own stable key for an imported product.
|
||||
//
|
||||
// `catalogueid` was never able to be this. The catalogue is rebuilt by
|
||||
|
||||
@@ -296,10 +296,20 @@ type NewPartner struct {
|
||||
Where they work — ONE district, not a set.
|
||||
|
||||
`Applocationid` is the home region and goes on the partner row itself,
|
||||
because `GetPartners` filters on it and the rider app reads it.
|
||||
`Applocationids` is every region they cover and goes to
|
||||
`partnerlocations` — one partner routinely serves several cities, and
|
||||
that is the whole reason the link table exists.
|
||||
because the rider app reads it. The same region is also written to
|
||||
`partnerlocations`, which is the table that may hold SEVERAL — a partner
|
||||
routinely serves more than one city, and that is why the link table
|
||||
exists, and partners with two are live — partner 44 covers regions 1 and
|
||||
2. Nothing on THIS path creates one: `regionsOf` returns this single
|
||||
field and the console's form offers one district, never a set. So a
|
||||
multi-region partner can be read and must be handled, but cannot yet be
|
||||
made here.
|
||||
|
||||
`GetPartners` reads the link table rather than this field, for two
|
||||
reasons. It is the column allowed to grow, so a partner who covers a
|
||||
second city will be found there without another change. And
|
||||
`partnerinfo` is shared with another product that writes no link rows,
|
||||
so having one is what marks a partner as ours.
|
||||
*/
|
||||
Applocationid int `json:"applocationid"`
|
||||
/*
|
||||
|
||||
@@ -155,6 +155,23 @@ type Products struct {
|
||||
// `catalogueProductColumns` casts its text[] columns to text.
|
||||
Productimages string `json:"productimages,omitempty" gorm:"column:productimages;type:jsonb"`
|
||||
|
||||
// The catalogue's own record of this product, as it stood at import.
|
||||
//
|
||||
// Holds the fields the snapshot does not have columns for — fssai_license,
|
||||
// highlights, nutrients, providers, price_range, variant_key, title,
|
||||
// sku_source, search_query — so the console can show them without asking
|
||||
// the catalogue again. It asked on every drawer open, and got nothing back
|
||||
// the moment a re-scrape retired the source row, taking a licence number
|
||||
// and a nutrition panel off a product the shop was still selling.
|
||||
//
|
||||
// Empty for anything that did not come from the catalogue: a sheet-imported
|
||||
// product has no such record, and the drawer falls back to the live lookup
|
||||
// for those exactly as before.
|
||||
//
|
||||
// A string for the same reason `Productimages` is one — GORM's raw
|
||||
// scan-into-struct silently drops slice- and map-kind destination fields.
|
||||
Cataloguefacts string `json:"cataloguefacts,omitempty" gorm:"column:cataloguefacts;type:jsonb"`
|
||||
|
||||
Productdesc string `json:"productdesc,omitempty"`
|
||||
Productsku string `json:"productsku,omitempty"`
|
||||
Brandid int `json:"brandid,omitempty"`
|
||||
@@ -225,6 +242,28 @@ type Locationproducts struct {
|
||||
Productimage string `json:"productimage,omitempty"`
|
||||
Productdesc string `json:"productdesc,omitempty"`
|
||||
Productsku string `json:"productsku,omitempty"`
|
||||
|
||||
// Three columns this read used to leave in the table.
|
||||
//
|
||||
// All three are stored on `products` and none of them reached the store
|
||||
// catalogue screen, because this struct had no field to scan them into —
|
||||
// so the console could not use what the import had gone to the trouble of
|
||||
// saving:
|
||||
//
|
||||
// Imageid the catalogue's durable key, and what HealthScorePanel
|
||||
// joins on. Absent, the panel reads it as "this product
|
||||
// never came from the catalogue" and renders nothing — for
|
||||
// EVERY product, including ones that plainly did.
|
||||
// Productimages the rest of a product's photos. `imagesOf()` parses this
|
||||
// and always got undefined, so the gallery fell back to
|
||||
// the single `productimage` and the extra images — 90 of
|
||||
// nestle's 123 products have them — were never shown.
|
||||
// Cataloguefacts the licence, nutrition, highlights, providers and price
|
||||
// range kept at import so they survive a re-scrape.
|
||||
Imageid string `json:"imageid,omitempty"`
|
||||
Productimages string `json:"productimages,omitempty"`
|
||||
Cataloguefacts string `json:"cataloguefacts,omitempty"`
|
||||
|
||||
Brandid int `json:"brandid,omitempty"`
|
||||
Productbrand string `json:"productbrand,omitempty"`
|
||||
Productunit string `json:"productunit"`
|
||||
|
||||
@@ -71,6 +71,14 @@ type Tenantinfo struct {
|
||||
Allocationid int `json:"allocationid"`
|
||||
Allocationtype string `json:"allocationtype"`
|
||||
Allocationmode int `json:"allocationmode"`
|
||||
|
||||
// How many outlets this merchant has.
|
||||
//
|
||||
// Only `GetAllTenants` fills this; it is 0 everywhere else, which is why it
|
||||
// is last and optional rather than part of the record proper. The console's
|
||||
// store list previously derived it by counting duplicate rows, and this
|
||||
// endpoint has never returned duplicates — see the note on the query.
|
||||
Branchcount int `json:"branchcount"`
|
||||
}
|
||||
|
||||
type Tenantlocations struct {
|
||||
@@ -190,6 +198,10 @@ type StaffInfo struct {
|
||||
Tenantid int `json:"tenantid"`
|
||||
Locationid int `json:"locationid"`
|
||||
Locationname string `json:"locationname"`
|
||||
// Whether this login still works, straight off `app_users.status`.
|
||||
// Without it every row on the console's Users & access screen read
|
||||
// "Unknown", because the field was never selected or sent.
|
||||
Status string `json:"status"`
|
||||
}
|
||||
|
||||
type Tenantuser struct {
|
||||
|
||||
@@ -208,3 +208,35 @@ func TestNormaliseBrandKeyRefusesToInventAKey(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Every word of the label was once required, so one word the catalogue does
|
||||
// not use ("Parle G biscuit pack") kept the right product out of the result
|
||||
// altogether and left the vector search to answer alone.
|
||||
func TestMinTokenHitsAsksForMostWordsNotAllOfThem(t *testing.T) {
|
||||
for _, tc := range []struct{ tokens, want int }{
|
||||
{1, 1}, // one word: it has to be there
|
||||
{2, 2}, // "Parle G" — both, and both are in Parle-G
|
||||
{3, 2}, // "Milk Bikis pack" — the pack is allowed to be missing
|
||||
{4, 3}, // "Parle G biscuit pack"
|
||||
{5, 4},
|
||||
{6, 4},
|
||||
} {
|
||||
if got := minTokenHits(tc.tokens); got != tc.want {
|
||||
t.Errorf("%d tokens: need %d, want %d", tc.tokens, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A threshold that could fall to 1 would let any single common word drag in
|
||||
// whole brand tables; one that stayed at n would be the bug all over again.
|
||||
func TestMinTokenHitsStaysBetweenTwoAndAll(t *testing.T) {
|
||||
for n := 3; n <= 30; n++ {
|
||||
got := minTokenHits(n)
|
||||
if got < 2 {
|
||||
t.Fatalf("%d tokens: %d is too loose", n, got)
|
||||
}
|
||||
if got >= n {
|
||||
t.Fatalf("%d tokens: %d still demands every word", n, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -87,30 +87,67 @@ func (r *partnerRepository) GetPartners(aid, pid, uid int) ([]models.Partnerinfo
|
||||
var q1 string
|
||||
var args []interface{}
|
||||
|
||||
// Every variant joins partnerlocations, and that join is the whole point.
|
||||
//
|
||||
// ── It is what separates our partners from somebody else's ──────────────
|
||||
//
|
||||
// `partnerinfo` is shared. It has no column saying which product a row
|
||||
// belongs to — no configid, no appid — so a partner created by another app
|
||||
// on this database is indistinguishable from ours by its own fields, and
|
||||
// this read used to return every Active row on the platform. The console
|
||||
// made that worse rather than better: it asks `getapplocations` for EVERY
|
||||
// region and then fetches partners region by region, so the applocationid
|
||||
// filter below never narrowed anything.
|
||||
//
|
||||
// `partnerlocations` is the difference. Only `CreatePartner` writes it —
|
||||
// one row per region, in the same transaction as the partner — so a row in
|
||||
// that table means "registered through this console". The partners that
|
||||
// predate it were inserted by hand and have none, which is why two of them
|
||||
// are called "Test".
|
||||
//
|
||||
// ── The region filter reads the link table, not the home region ─────────
|
||||
//
|
||||
// `partnerinfo.applocationid` is the HOME region — CreatePartner writes
|
||||
// `regions[0]` there — while partnerlocations holds every region covered.
|
||||
// Those are not the same thing, and not only in theory: partner 44,
|
||||
// Xpress-Cbe-Main, has a home region of 1 and link rows for 1 AND 2, so
|
||||
// filtering on the partner row hid them from every Madurai query. That is
|
||||
// the case the link table exists for.
|
||||
//
|
||||
// DISTINCT because such a partner has one row per region in the join and is
|
||||
// still one partner. Only partnerinfo columns are selected, so there is
|
||||
// nothing per-region for it to fail to collapse.
|
||||
//
|
||||
// A caller fanning out over regions and concatenating the answers still has
|
||||
// to dedupe — the same partner is legitimately in two of them. The console's
|
||||
// `useAllPartners` does; it listed Xpress-Cbe-Main twice until it did.
|
||||
const columns = `select distinct p.partnerid,p.applocationid,p.partnertypeid,p.partnername,
|
||||
p.primarycontact,p.primaryemail,p.contactno,p.address,p.suburb,p.state,p.city,p.partnerimage
|
||||
from partnerinfo p
|
||||
inner join partnerlocations l on l.partnerid = p.partnerid
|
||||
where p.status='Active'`
|
||||
|
||||
if pid != 0 {
|
||||
q1 = `select partnerid,applocationid,partnertypeid,partnername,primarycontact,primaryemail,
|
||||
contactno,address,suburb,state,city,partnerimage
|
||||
from partnerinfo where status='Active' and partnerid=?`
|
||||
// Scoped the same way on purpose: asking for a partner by id must not
|
||||
// be a way round the separation above.
|
||||
q1 = columns + ` and p.partnerid=?`
|
||||
args = append(args, pid)
|
||||
|
||||
} else if aid != 0 {
|
||||
q1 = `select partnerid,applocationid,partnertypeid,partnername,primarycontact,primaryemail,
|
||||
contactno,address,suburb,state,city,partnerimage
|
||||
from partnerinfo where status='Active' and applocationid=?`
|
||||
q1 = columns + ` and l.applocationid=?`
|
||||
args = append(args, aid)
|
||||
|
||||
} else {
|
||||
q1 = `select partnerid,applocationid,partnertypeid,partnername,primarycontact,primaryemail,
|
||||
contactno,address,suburb,state,city,partnerimage
|
||||
from partnerinfo where status='Active'`
|
||||
q1 = columns
|
||||
}
|
||||
|
||||
q1 += ` order by p.partnername, p.partnerid`
|
||||
|
||||
err := r.db.Raw(q1, args...).Find(&data).Error
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
print(q1)
|
||||
return data, nil
|
||||
}
|
||||
|
||||
@@ -615,13 +652,17 @@ them are named "Test".
|
||||
|
||||
Where a partner works is recorded twice, on purpose and not by accident:
|
||||
|
||||
partnerinfo.applocationid their home region — `GetPartners` filters on it
|
||||
and the rider app reads it
|
||||
partnerinfo.applocationid their home region — the rider app reads it
|
||||
partnerlocations every region they cover
|
||||
|
||||
Both are kept in step here. Writing only the first would confine a partner to
|
||||
one city, and writing only the second would hide them from every existing
|
||||
query. */
|
||||
one city, and writing only the second would hide them from the rider app.
|
||||
|
||||
`GetPartners` reads the SECOND: it joins partnerlocations, which both scopes a
|
||||
region query to every city a partner actually covers and — because only this
|
||||
function writes that table — separates partners registered here from the ones
|
||||
another product put in the shared `partnerinfo`. So the link rows are not
|
||||
bookkeeping; they are what makes a partner ours. */
|
||||
|
||||
// CreatePartner onboards a delivery partner and records the regions they cover.
|
||||
func (r *partnerRepository) CreatePartner(input models.NewPartner) (int, error) {
|
||||
|
||||
@@ -26,7 +26,6 @@ type ProductRepository interface {
|
||||
UpdateProductStatus(productIDs []int, status string) error
|
||||
SyncProductLocationStatus(refs []models.ProductLocationRef) error
|
||||
EnsureProductLocation(refs []models.ProductLocationRef) error
|
||||
CreateProduct(product models.Products) error
|
||||
UpdateProduct(product models.Products) error
|
||||
DeleteProduct(productID int) error
|
||||
GetStockStatement(tenantID, locationID, subcategoryID, pageno, pagesize int, keyword string) ([]models.Productstockstatement, error)
|
||||
@@ -393,19 +392,32 @@ func (r *productRepository) UpdateProductStatus(productIDs []int, status string)
|
||||
Update("productstatus", status).Error
|
||||
}
|
||||
|
||||
func (r *productRepository) CreateProduct(product models.Products) error {
|
||||
tx := r.db.Begin()
|
||||
|
||||
if err := tx.Create(&product).Error; err != nil {
|
||||
tx.Rollback()
|
||||
return err
|
||||
// normaliseProductJSON makes a product safe to INSERT.
|
||||
//
|
||||
// `products.productimages` is jsonb and `models.Products.Productimages` is a
|
||||
// plain string, so a caller that never set it hands GORM the zero value — and
|
||||
// GORM puts that empty string in the INSERT rather than omitting the column.
|
||||
// Postgres answers "invalid input syntax for type json (SQLSTATE 22P02)" and
|
||||
// the whole row is rejected, over a field nobody asked for.
|
||||
//
|
||||
// That was not a corner case: the console's sheet importer sends no
|
||||
// productimages at all, so EVERY product it created failed with a 500, and
|
||||
// ImportCatalogueProduct leaves the field empty for any catalogue product that
|
||||
// has no photos. An empty ARRAY is the honest value — there are no extra
|
||||
// images — and it is what `catalogueUploadService` already does for its own
|
||||
// jsonb column, for the same reason.
|
||||
//
|
||||
// Applied at the one create path, which is the last point before the SQL, and
|
||||
// the constraint being satisfied is the database's.
|
||||
func normaliseProductJSON(product *models.Products) {
|
||||
if strings.TrimSpace(product.Productimages) == "" {
|
||||
product.Productimages = "[]"
|
||||
}
|
||||
|
||||
if err := tx.Commit().Error; err != nil {
|
||||
return err
|
||||
// An OBJECT, not an array: this one holds named catalogue fields, and `{}`
|
||||
// is what a reader parsing it expects to find when there are none.
|
||||
if strings.TrimSpace(product.Cataloguefacts) == "" {
|
||||
product.Cataloguefacts = "{}"
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func (r *productRepository) UpdateProduct(product models.Products) error {
|
||||
@@ -1316,10 +1328,22 @@ func (r *productRepository) FindTenantProductByCatalogueRef(tenantid int, brand
|
||||
return &product, nil
|
||||
}
|
||||
|
||||
// CreateProductReturningID inserts a new product snapshot and returns its
|
||||
// generated productid. Kept separate from CreateProduct so existing callers
|
||||
// of CreateProduct are unaffected.
|
||||
// CreateProductReturningID inserts a product and returns its generated
|
||||
// productid.
|
||||
//
|
||||
// This is now the only way to create one. There used to be a second method,
|
||||
// `CreateProduct`, that did the same INSERT and threw the id away — it took
|
||||
// the struct by value, so GORM wrote the generated id onto a copy that went
|
||||
// out of scope, and `POST /products/create` answered `productid: 0` for every
|
||||
// product it had just created. The console worked around it by creating, then
|
||||
// re-reading the whole tenant catalogue, then matching back by SKU.
|
||||
//
|
||||
// The two were kept apart so that "existing callers are unaffected", but the
|
||||
// only caller of the id-less one was the endpoint that needed the id most.
|
||||
// One create path also means the jsonb guard above has one place to live.
|
||||
func (r *productRepository) CreateProductReturningID(product models.Products) (int, error) {
|
||||
normaliseProductJSON(&product)
|
||||
|
||||
if err := r.db.Create(&product).Error; err != nil {
|
||||
return 0, err
|
||||
}
|
||||
|
||||
@@ -470,9 +470,22 @@ func (r *scanRepository) VectorSearch(ctx context.Context, vector []float32, lim
|
||||
return hits, nil
|
||||
}
|
||||
|
||||
// minTokenHits is how many of the label's words a row must carry to be worth
|
||||
// looking at. Every word was once required, which meant a single word the
|
||||
// catalogue does not use — "Parle G biscuit pack", "Milk Bikis pack" — kept
|
||||
// the right product out of the result entirely, leaving the vector search to
|
||||
// answer alone and confidently wrong. Most of them is enough; scoring sorts
|
||||
// out the rest.
|
||||
func minTokenHits(n int) int {
|
||||
if n <= 2 {
|
||||
return n
|
||||
}
|
||||
return (n*2 + 2) / 3 // two thirds, rounded up; never below 2 for n >= 3
|
||||
}
|
||||
|
||||
// TextSearch is the fallback when there is no embedder, and the tie-breaker
|
||||
// beside it when there is: rows whose name or title contains the label, or
|
||||
// contains every word of it.
|
||||
// carry most of its words.
|
||||
func (r *scanRepository) TextSearch(ctx context.Context, label string, limit int) ([]CatalogueHit, error) {
|
||||
tables, err := r.brandTables(ctx)
|
||||
if err != nil {
|
||||
@@ -498,18 +511,32 @@ func (r *scanRepository) TextSearch(ctx context.Context, label string, limit int
|
||||
hay = "LOWER(COALESCE(product_name, '') || ' ' || COALESCE(title, '') || ' ' || COALESCE(search_query, ''))"
|
||||
}
|
||||
|
||||
conds := []string{hay + " LIKE ?"}
|
||||
// How well a row matches, as a number: the whole label as a substring
|
||||
// outweighs any number of loose words, then one point per word found.
|
||||
hits := make([]string, 0, len(tokens)+1)
|
||||
hits = append(hits, "(CASE WHEN "+hay+" LIKE ? THEN 100 ELSE 0 END)")
|
||||
for range tokens {
|
||||
hits = append(hits, "(CASE WHEN "+hay+" LIKE ? THEN 1 ELSE 0 END)")
|
||||
}
|
||||
rank := strings.Join(hits, " + ")
|
||||
|
||||
// The expression appears twice in the SQL — once to filter, once to
|
||||
// order — so its arguments are bound twice, in that order.
|
||||
bind := func() {
|
||||
args = append(args, "%"+label+"%")
|
||||
all := make([]string, 0, len(tokens))
|
||||
for _, tok := range tokens {
|
||||
all = append(all, hay+" LIKE ?")
|
||||
args = append(args, "%"+tok+"%")
|
||||
}
|
||||
conds = append(conds, "("+strings.Join(all, " AND ")+")")
|
||||
}
|
||||
bind()
|
||||
bind()
|
||||
|
||||
// Ordering matters as much as the threshold: a looser WHERE lets more
|
||||
// rows qualify, and an unordered LIMIT would then be free to return
|
||||
// the wrong ones. Best match per brand first, id to keep it stable.
|
||||
branches = append(branches, fmt.Sprintf(
|
||||
`(SELECT %s, -1::float8 AS distance FROM %s WHERE %s LIMIT %d)`,
|
||||
hitColumns(brand, cols), table, strings.Join(conds, " OR "), limit))
|
||||
`(SELECT %s, -1::float8 AS distance FROM %s WHERE (%s) >= %d ORDER BY (%s) DESC, id LIMIT %d)`,
|
||||
hitColumns(brand, cols), table, rank, minTokenHits(len(tokens)), rank, limit))
|
||||
}
|
||||
if len(branches) == 0 {
|
||||
return nil, nil
|
||||
|
||||
@@ -85,7 +85,22 @@ func (r *tenantRepository) GetAllTenants(pageno, pagesize, aid int, status, tena
|
||||
|
||||
var data []models.Tenantinfo
|
||||
|
||||
base := `SELECT * FROM tenants a WHERE 1 = 1`
|
||||
// `branchcount` is selected here because there is nowhere else to get it.
|
||||
//
|
||||
// This returns one row per TENANT — there is no join to tenantlocations at
|
||||
// all — but the console's store list read it as one row per
|
||||
// tenant-location pair and counted the duplicates, so every merchant on the
|
||||
// platform showed exactly one branch, and the "Branches" and "Avg branches"
|
||||
// tiles above the list were the tenant count wearing another name. The
|
||||
// tenant's own detail page, which reads gettenantlocations, disagreed with
|
||||
// the list it was opened from.
|
||||
//
|
||||
// A correlated subquery rather than a LEFT JOIN + GROUP BY: the row shape
|
||||
// stays exactly as it was, so nothing else that reads this endpoint has to
|
||||
// change, and every filter below still applies to `a` alone.
|
||||
base := `SELECT a.*,
|
||||
(SELECT COUNT(*) FROM tenantlocations tl WHERE tl.tenantid = a.tenantid) AS branchcount
|
||||
FROM tenants a WHERE 1 = 1`
|
||||
|
||||
var (
|
||||
conds []string
|
||||
@@ -337,7 +352,12 @@ func (r *tenantRepository) GetStaffs(tid int) ([]models.StaffInfo, error) {
|
||||
a.state,a.postcode,a.userfcmtoken,a.pin,a.applocationid,
|
||||
a.roleid,a.partnerid,a.tenantid,a.locationid,
|
||||
b.locationname,
|
||||
COALESCE(c.rolename,'') AS rolename
|
||||
COALESCE(c.rolename,'') AS rolename,
|
||||
-- Whether the account still works. Absent from this SELECT
|
||||
-- until now, so Users & access had nothing to read and showed
|
||||
-- every person on the platform as "Unknown" — an admin could not
|
||||
-- tell a working login from one that had been switched off.
|
||||
COALESCE(a.status,'') AS status
|
||||
FROM app_users a
|
||||
LEFT JOIN tenantlocations b ON a.locationid = b.locationid
|
||||
LEFT JOIN app_roles c ON c.roleid = a.roleid
|
||||
@@ -625,6 +645,51 @@ func (r *tenantRepository) CreateTenantUser(data models.Tenants) (bool, error) {
|
||||
var custloc models.Customerlocations
|
||||
var tcust models.Tenantcustomers
|
||||
|
||||
// A tenant with configid 0 is unreachable, and it takes its customer row
|
||||
// with it.
|
||||
//
|
||||
// Step 3 below already forces `user.Configid = 1`, with a comment
|
||||
// explaining that AppLogin only ever queries configid 1 and a zero makes
|
||||
// the account permanently unfindable. The same zero was left to flow into
|
||||
// `tenants` itself and into the `customers` row copied from it at step 4,
|
||||
// where nothing corrected it — so a caller that omits configid (the console
|
||||
// sends it; the mobile route and anything else need not) created a business
|
||||
// and a customer that no scoped read can see.
|
||||
//
|
||||
// Defaulted rather than rejected: 1 is the only value any caller has ever
|
||||
// meant here, and refusing the create would break callers that work today.
|
||||
if data.Configid == 0 {
|
||||
data.Configid = 1
|
||||
}
|
||||
|
||||
// Give the primary outlet the scaffolding the tenant already has.
|
||||
//
|
||||
// The outlet itself is created by GORM, as the `Tenantlocations`
|
||||
// association on the struct below — the console nests a full object in the
|
||||
// request and step 1 saves it with the tenant. What it does NOT do is fill
|
||||
// anything the caller left out, and two of those columns matter:
|
||||
//
|
||||
// applocationid — `orderRepository.go` calls it "authoritative" and has
|
||||
// no fallback anywhere for a 0.
|
||||
// moduleid — same file: "tenantlocations carries 0 for
|
||||
// moduleid/partnerid at outlets whose live orders
|
||||
// nonetheless use non-zero values", worked around there
|
||||
// by copying scaffolding off the most recent real order.
|
||||
// A shop commissioned a minute ago has no such order.
|
||||
//
|
||||
// Neither column has a database default, and no onboarding form asks for
|
||||
// them — they describe the platform, not the shop. The tenant's own values
|
||||
// are the right answer and are already right here.
|
||||
//
|
||||
// Filled before the insert rather than corrected after it, so there is one
|
||||
// write and no window where the row exists with a zero in it.
|
||||
if data.Tenantlocations.Applocationid == 0 {
|
||||
data.Tenantlocations.Applocationid = data.Applocationid
|
||||
}
|
||||
if data.Tenantlocations.Moduleid == 0 {
|
||||
data.Tenantlocations.Moduleid = data.Moduleid
|
||||
}
|
||||
|
||||
tx := r.db.Begin()
|
||||
|
||||
// Step 1: Insert into tenants
|
||||
|
||||
@@ -255,6 +255,30 @@ func (r *userRepository) GetTenantUserById(userid int) models.TenantUserInfo {
|
||||
}
|
||||
|
||||
func (r *userRepository) CreateUser(user models.User) (int, error) {
|
||||
// Inherit the delivery region from the tenant when the caller did not name
|
||||
// one.
|
||||
//
|
||||
// `app_users.applocationid` has no column default, and no console form
|
||||
// collects it — it is a platform region, not something a merchant picks
|
||||
// per person. So every back-office account created through this path landed
|
||||
// with 0, which is not a region: `orderRepository.go` calls the equivalent
|
||||
// column on tenantlocations "authoritative" and has no fallback for a zero,
|
||||
// and 43 of 75 live branches are already in that state.
|
||||
//
|
||||
// A lookup rather than a default value, because the right answer is
|
||||
// whichever region the business trades in. Failure is not fatal: the
|
||||
// account is still worth creating, and a 0 here is exactly what would have
|
||||
// been written anyway.
|
||||
if user.Applocationid == 0 && user.Tenantid > 0 {
|
||||
var inherited int
|
||||
if err := r.db.Raw(
|
||||
`SELECT COALESCE(applocationid, 0) FROM tenants WHERE tenantid = ?`,
|
||||
user.Tenantid,
|
||||
).Scan(&inherited).Error; err == nil && inherited > 0 {
|
||||
user.Applocationid = inherited
|
||||
}
|
||||
}
|
||||
|
||||
tx := r.db.Begin()
|
||||
|
||||
if err := tx.Table("app_users").Create(&user).Error; err != nil {
|
||||
|
||||
402
scratch/cataloguefactsbackfill/main.go
Normal file
402
scratch/cataloguefactsbackfill/main.go
Normal file
@@ -0,0 +1,402 @@
|
||||
// Backfills products.cataloguefacts for products imported before the column existed.
|
||||
//
|
||||
// The catalogue import copied eight of the catalogue's eighteen fields onto a
|
||||
// tenant's product and left the other ten behind — the FSSAI licence, nutrients,
|
||||
// highlights, providers, the typical price range, the variant key. The console
|
||||
// covered for it by asking the catalogue again on every drawer open, and that
|
||||
// stops working the moment a re-scrape retires the source row: a tenant's
|
||||
// product is a SNAPSHOT and outlives it, so a licence number came off a product
|
||||
// the shop was still selling with no way back.
|
||||
//
|
||||
// The import keeps them now. Every product imported BEFORE that does not have
|
||||
// them, and no amount of new code fixes a row that was written last month — so
|
||||
// this reads each one's catalogue entry while it is still there and stores it.
|
||||
//
|
||||
// go run ./scratch/cataloguefactsbackfill # dry run — shows every change
|
||||
// go run ./scratch/cataloguefactsbackfill apply # writes, then prints the undo
|
||||
//
|
||||
// ── What it will and will not touch ─────────────────────────────────────────
|
||||
//
|
||||
// Only products with an `imageid` and a NULL `cataloguefacts`. That is the
|
||||
// whole safety story:
|
||||
//
|
||||
// - NULL means nothing was ever written. A product whose facts are already
|
||||
// stored — including one stored as `{}` because the catalogue genuinely had
|
||||
// nothing to say — is never overwritten, so re-running this is a no-op
|
||||
// rather than a second opinion.
|
||||
// - No `imageid` means it never came from the catalogue. Sheet-imported
|
||||
// products have no entry to read and are left alone.
|
||||
// - A catalogue row that has already been retired cannot be recovered by
|
||||
// anything, here or later. Those are counted and named rather than written
|
||||
// as empty, because `{}` would claim the catalogue said nothing when the
|
||||
// truth is that nobody asked in time.
|
||||
//
|
||||
// Brand tables are discovered rather than assumed, and their columns are
|
||||
// checked one by one before being selected: the catalogue is another team's
|
||||
// scrape, brands appear between runs, and a table missing `nutrients` is a
|
||||
// perfectly good catalogue of products. Demanding the full column set is the
|
||||
// exact mistake that once made 16 of 35 live brands invisible to this side.
|
||||
package main
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"log"
|
||||
"os"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"github.com/joho/godotenv"
|
||||
"gorm.io/driver/postgres"
|
||||
"gorm.io/gorm"
|
||||
"gorm.io/gorm/logger"
|
||||
|
||||
"nearle/models"
|
||||
)
|
||||
|
||||
// The columns worth keeping, in the order the drawer reads them. Scalars and
|
||||
// arrays are separated because an array comes back as a Postgres text[] literal
|
||||
// and has to be parsed before it can be re-encoded as JSON.
|
||||
var scalarFacts = []string{
|
||||
"title", "category", "variant_key", "sku_source",
|
||||
"price_range", "fssai_license", "search_query",
|
||||
}
|
||||
|
||||
var arrayFacts = []string{"providers", "highlights", "nutrients"}
|
||||
|
||||
type product struct {
|
||||
Productid int
|
||||
Productbrand string
|
||||
Imageid string
|
||||
Productname string
|
||||
Tenantid int
|
||||
}
|
||||
|
||||
func main() {
|
||||
apply := len(os.Args) > 1 && os.Args[1] == "apply"
|
||||
|
||||
_ = godotenv.Load()
|
||||
|
||||
main, err := open("DB_HOST", "DB_PORT", "DB_USER", "DB_PASSWORD", "DB_NAME")
|
||||
if err != nil {
|
||||
log.Fatal("nearledb: ", err)
|
||||
}
|
||||
cat, err := open("CATALOGUE_DB_HOST", "CATALOGUE_DB_PORT", "CATALOGUE_DB_USER",
|
||||
"CATALOGUE_DB_PASSWORD", "CATALOGUE_DB_NAME")
|
||||
if err != nil {
|
||||
log.Fatal("cataloguedb: ", err)
|
||||
}
|
||||
|
||||
// The column has to exist before there is anything to fill. Checked rather
|
||||
// than assumed so this says so plainly instead of failing inside a query.
|
||||
var hasColumn int
|
||||
main.Raw(`SELECT COUNT(*) FROM information_schema.columns
|
||||
WHERE table_name = 'products' AND column_name = 'cataloguefacts'`).Scan(&hasColumn)
|
||||
if hasColumn == 0 {
|
||||
log.Fatal("products.cataloguefacts does not exist — start the API once to run the migration, then re-run this")
|
||||
}
|
||||
|
||||
var candidates []product
|
||||
main.Raw(`SELECT productid, tenantid, COALESCE(productbrand,'') AS productbrand,
|
||||
COALESCE(imageid,'') AS imageid, COALESCE(productname,'') AS productname
|
||||
FROM products
|
||||
WHERE COALESCE(imageid,'') <> '' AND cataloguefacts IS NULL
|
||||
ORDER BY productbrand, productid`).Scan(&candidates)
|
||||
|
||||
var (
|
||||
total int
|
||||
alreadyDone int
|
||||
noImageid int
|
||||
)
|
||||
main.Raw(`SELECT COUNT(*) FROM products`).Scan(&total)
|
||||
main.Raw(`SELECT COUNT(*) FROM products WHERE cataloguefacts IS NOT NULL`).Scan(&alreadyDone)
|
||||
main.Raw(`SELECT COUNT(*) FROM products WHERE COALESCE(imageid,'') = ''`).Scan(&noImageid)
|
||||
|
||||
fmt.Printf("products on the platform : %d\n", total)
|
||||
fmt.Printf(" never came from the catalogue : %d (no imageid — left alone)\n", noImageid)
|
||||
fmt.Printf(" facts already stored : %d (never overwritten)\n", alreadyDone)
|
||||
fmt.Printf(" to backfill : %d\n\n", len(candidates))
|
||||
|
||||
if len(candidates) == 0 {
|
||||
fmt.Println("nothing to do.")
|
||||
return
|
||||
}
|
||||
|
||||
// One column check per brand table, not per product: the shape is a
|
||||
// property of the table and a per-row check would be thousands of
|
||||
// information_schema reads to learn the same thing.
|
||||
columnsByTable := map[string][]string{}
|
||||
missingTable := map[string]bool{}
|
||||
|
||||
type update struct {
|
||||
product product
|
||||
facts string
|
||||
}
|
||||
var (
|
||||
updates []update
|
||||
retired []product
|
||||
unknown []product
|
||||
emptyOnly []product
|
||||
)
|
||||
|
||||
for _, p := range candidates {
|
||||
table := brandTable(p.Productbrand)
|
||||
if table == "" {
|
||||
unknown = append(unknown, p)
|
||||
continue
|
||||
}
|
||||
if missingTable[table] {
|
||||
retired = append(retired, p)
|
||||
continue
|
||||
}
|
||||
|
||||
cols, known := columnsByTable[table]
|
||||
if !known {
|
||||
cols = factColumnsOf(cat, table)
|
||||
if cols == nil {
|
||||
missingTable[table] = true
|
||||
retired = append(retired, p)
|
||||
continue
|
||||
}
|
||||
columnsByTable[table] = cols
|
||||
}
|
||||
|
||||
facts, found := factsFor(cat, table, cols, p.Imageid)
|
||||
if !found {
|
||||
retired = append(retired, p)
|
||||
continue
|
||||
}
|
||||
if len(facts) == 0 {
|
||||
// The row is there and had nothing in these columns. Worth writing
|
||||
// `{}` — it is the true answer and it stops the console asking the
|
||||
// catalogue again on every open.
|
||||
emptyOnly = append(emptyOnly, p)
|
||||
}
|
||||
|
||||
encoded, err := json.Marshal(facts)
|
||||
if err != nil {
|
||||
log.Printf("could not encode facts for product %d: %v", p.Productid, err)
|
||||
continue
|
||||
}
|
||||
updates = append(updates, update{product: p, facts: string(encoded)})
|
||||
}
|
||||
|
||||
fmt.Printf("%-9s %-14s %-22s %-34s %s\n", "product", "brand", "imageid", "name", "facts recovered")
|
||||
for _, u := range updates {
|
||||
var keys []string
|
||||
var got map[string]any
|
||||
_ = json.Unmarshal([]byte(u.facts), &got)
|
||||
for k := range got {
|
||||
keys = append(keys, k)
|
||||
}
|
||||
sort.Strings(keys)
|
||||
summary := strings.Join(keys, ",")
|
||||
if summary == "" {
|
||||
summary = "(catalogue row has none)"
|
||||
}
|
||||
fmt.Printf("%-9d %-14s %-22s %-34s %s\n",
|
||||
u.product.Productid, trim(u.product.Productbrand, 14), trim(u.product.Imageid, 22),
|
||||
trim(u.product.Productname, 34), summary)
|
||||
}
|
||||
|
||||
if len(retired) > 0 {
|
||||
fmt.Printf("\n!! %d product(s) cannot be recovered — their catalogue row is gone:\n", len(retired))
|
||||
for _, p := range retired {
|
||||
fmt.Printf(" %-9d %-14s %-22s %s\n", p.Productid, trim(p.Productbrand, 14),
|
||||
trim(p.Imageid, 22), trim(p.Productname, 40))
|
||||
}
|
||||
fmt.Println(" These are left NULL. The console falls back to the live lookup for them,")
|
||||
fmt.Println(" which will also find nothing — the detail was lost before this ran.")
|
||||
}
|
||||
|
||||
if len(unknown) > 0 {
|
||||
fmt.Printf("\n!! %d product(s) carry a brand with no table in the catalogue:\n", len(unknown))
|
||||
for _, p := range unknown {
|
||||
fmt.Printf(" %-9d %-14s %s\n", p.Productid, trim(p.Productbrand, 14), trim(p.Productname, 40))
|
||||
}
|
||||
}
|
||||
|
||||
fmt.Printf("\nwill write %d product(s)", len(updates))
|
||||
if len(emptyOnly) > 0 {
|
||||
fmt.Printf(", %d of them as `{}` because the catalogue row carries none of these fields", len(emptyOnly))
|
||||
}
|
||||
fmt.Printf("; leaving %d NULL\n", len(retired)+len(unknown))
|
||||
|
||||
if len(updates) == 0 {
|
||||
return
|
||||
}
|
||||
if !apply {
|
||||
fmt.Println("\ndry run — nothing written. re-run with `apply` to write.")
|
||||
return
|
||||
}
|
||||
|
||||
// One row at a time, each guarded by `cataloguefacts IS NULL` again.
|
||||
// Between the read above and this write another import could have stored
|
||||
// the real thing, and this must never be the one that overwrites it.
|
||||
written := 0
|
||||
ids := make([]int, 0, len(updates))
|
||||
for _, u := range updates {
|
||||
res := main.Exec(`UPDATE products SET cataloguefacts = ?::jsonb
|
||||
WHERE productid = ? AND cataloguefacts IS NULL`,
|
||||
u.facts, u.product.Productid)
|
||||
if res.Error != nil {
|
||||
log.Printf("product %d: %v", u.product.Productid, res.Error)
|
||||
continue
|
||||
}
|
||||
if res.RowsAffected > 0 {
|
||||
written++
|
||||
ids = append(ids, u.product.Productid)
|
||||
}
|
||||
}
|
||||
|
||||
fmt.Printf("\nwrote %d product(s)\n", written)
|
||||
|
||||
var stillNull int
|
||||
main.Raw(`SELECT COUNT(*) FROM products
|
||||
WHERE COALESCE(imageid,'') <> '' AND cataloguefacts IS NULL`).Scan(&stillNull)
|
||||
fmt.Printf("catalogue-linked products still without facts: %d\n", stillNull)
|
||||
|
||||
if len(ids) > 0 {
|
||||
fmt.Printf("\nundo:\n UPDATE products SET cataloguefacts = NULL WHERE productid IN (%s);\n",
|
||||
joinInts(ids))
|
||||
}
|
||||
}
|
||||
|
||||
func open(hostKey, portKey, userKey, passKey, nameKey string) (*gorm.DB, error) {
|
||||
dsn := fmt.Sprintf("host=%s port=%s user=%s password=%s dbname=%s sslmode=disable",
|
||||
os.Getenv(hostKey), os.Getenv(portKey), os.Getenv(userKey),
|
||||
os.Getenv(passKey), os.Getenv(nameKey))
|
||||
return gorm.Open(postgres.Open(dsn), &gorm.Config{Logger: logger.Default.LogMode(logger.Silent)})
|
||||
}
|
||||
|
||||
// brandTable mirrors the repository's rule: a brand IS a `brand_<name>` table.
|
||||
//
|
||||
// Lowercased and stripped of anything that is not a letter, digit or
|
||||
// underscore. The table name cannot be parameterized in SQL, so this is the
|
||||
// one place it is built and it refuses to build anything else.
|
||||
func brandTable(brand string) string {
|
||||
cleaned := strings.Map(func(r rune) rune {
|
||||
switch {
|
||||
case r >= 'a' && r <= 'z', r >= '0' && r <= '9', r == '_':
|
||||
return r
|
||||
case r >= 'A' && r <= 'Z':
|
||||
return r + 32
|
||||
}
|
||||
return -1
|
||||
}, strings.TrimSpace(brand))
|
||||
|
||||
if cleaned == "" {
|
||||
return ""
|
||||
}
|
||||
return "brand_" + cleaned
|
||||
}
|
||||
|
||||
// factColumnsOf returns which of the fact columns this brand table actually
|
||||
// has, or nil when the table is not there at all.
|
||||
func factColumnsOf(db *gorm.DB, table string) []string {
|
||||
var have []string
|
||||
db.Raw(`SELECT column_name FROM information_schema.columns
|
||||
WHERE table_schema = 'public' AND table_name = ?`, table).Scan(&have)
|
||||
if len(have) == 0 {
|
||||
return nil
|
||||
}
|
||||
|
||||
present := map[string]bool{}
|
||||
for _, c := range have {
|
||||
present[c] = true
|
||||
}
|
||||
// image_id is how a product is found at all. Without it the table cannot
|
||||
// answer the question, whatever else it holds.
|
||||
if !present["image_id"] {
|
||||
return nil
|
||||
}
|
||||
|
||||
var keep []string
|
||||
for _, c := range append(append([]string{}, scalarFacts...), arrayFacts...) {
|
||||
if present[c] {
|
||||
keep = append(keep, c)
|
||||
}
|
||||
}
|
||||
return keep
|
||||
}
|
||||
|
||||
// factsFor reads one catalogue row and returns only what it actually stated.
|
||||
//
|
||||
// An empty field is omitted rather than stored as "" or [], so a reader can
|
||||
// tell "the catalogue did not say" from "the catalogue said none" — the drawer
|
||||
// prints a row per fact and an empty string would print an empty row.
|
||||
func factsFor(db *gorm.DB, table string, cols []string, imageID string) (map[string]any, bool) {
|
||||
if len(cols) == 0 {
|
||||
return map[string]any{}, true
|
||||
}
|
||||
|
||||
selects := make([]string, 0, len(cols))
|
||||
for _, c := range cols {
|
||||
if isArrayFact(c) {
|
||||
selects = append(selects, c+"::text AS "+c)
|
||||
continue
|
||||
}
|
||||
selects = append(selects, c)
|
||||
}
|
||||
|
||||
row := map[string]any{}
|
||||
res := db.Raw(`SELECT `+strings.Join(selects, ", ")+` FROM `+table+
|
||||
` WHERE image_id = ? LIMIT 1`, imageID).Scan(&row)
|
||||
if res.Error != nil || res.RowsAffected == 0 {
|
||||
return nil, false
|
||||
}
|
||||
|
||||
facts := map[string]any{}
|
||||
for _, c := range cols {
|
||||
raw, ok := row[c]
|
||||
if !ok || raw == nil {
|
||||
continue
|
||||
}
|
||||
text := strings.TrimSpace(fmt.Sprintf("%v", raw))
|
||||
if text == "" {
|
||||
continue
|
||||
}
|
||||
if isArrayFact(c) {
|
||||
values := models.ParsePGArray(text)
|
||||
kept := make([]string, 0, len(values))
|
||||
for _, v := range values {
|
||||
if t := strings.TrimSpace(v); t != "" {
|
||||
kept = append(kept, t)
|
||||
}
|
||||
}
|
||||
if len(kept) > 0 {
|
||||
facts[c] = kept
|
||||
}
|
||||
continue
|
||||
}
|
||||
facts[c] = text
|
||||
}
|
||||
return facts, true
|
||||
}
|
||||
|
||||
func isArrayFact(name string) bool {
|
||||
for _, c := range arrayFacts {
|
||||
if c == name {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func trim(s string, n int) string {
|
||||
if len(s) <= n {
|
||||
return s
|
||||
}
|
||||
if n <= 1 {
|
||||
return s[:n]
|
||||
}
|
||||
return s[:n-1] + "…"
|
||||
}
|
||||
|
||||
func joinInts(ids []int) string {
|
||||
parts := make([]string, len(ids))
|
||||
for i, id := range ids {
|
||||
parts[i] = fmt.Sprint(id)
|
||||
}
|
||||
return strings.Join(parts, ",")
|
||||
}
|
||||
@@ -4,9 +4,11 @@ import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"log"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"nearle/models"
|
||||
"nearle/repositories"
|
||||
"time"
|
||||
)
|
||||
|
||||
type ProductService interface {
|
||||
@@ -22,7 +24,7 @@ type ProductService interface {
|
||||
RemoveProductVariant(tenantid, variantid int) error
|
||||
VariantChildIDs(tenantid int) (map[int]bool, error)
|
||||
CreateProductStock(stocks []models.Productstock) error
|
||||
CreateProduct(product models.Products) error
|
||||
CreateProduct(product models.Products) (models.Products, error)
|
||||
UpdateProduct(product models.Products) error
|
||||
DeleteProduct(productID int) error
|
||||
GetStockStatement(tenantID, locationID, subcategoryID, pageno, pagesize int, keyword string) ([]models.Productstockstatement, error)
|
||||
@@ -169,8 +171,27 @@ func (s *productService) UpdateProductStatus(productIDs []int, status string) er
|
||||
return s.repo.UpdateProductStatus(productIDs, status)
|
||||
}
|
||||
|
||||
func (s *productService) CreateProduct(product models.Products) error {
|
||||
return s.repo.CreateProduct(product)
|
||||
// CreateProduct stores one product and hands it back with its id filled in.
|
||||
//
|
||||
// It used to return only an error, and the id was lost on the way out: the
|
||||
// repository took the struct by value, GORM wrote the generated productid onto
|
||||
// that copy, and the copy was discarded — so the endpoint answered
|
||||
// `productid: 0` for a row that certainly had one.
|
||||
//
|
||||
// The caller needs it. A product is not sellable until it has been priced at an
|
||||
// outlet and stocked there, and both of those calls are keyed on productid, so
|
||||
// every importer had to create, re-read the tenant's whole catalogue, and match
|
||||
// its own rows back by SKU to carry on — which is also why creating two
|
||||
// products with the same SKU quietly attached the second one's stock to the
|
||||
// first.
|
||||
func (s *productService) CreateProduct(product models.Products) (models.Products, error) {
|
||||
id, err := s.repo.CreateProductReturningID(product)
|
||||
if err != nil {
|
||||
return models.Products{}, err
|
||||
}
|
||||
|
||||
product.Productid = id
|
||||
return product, nil
|
||||
}
|
||||
|
||||
func (s *productService) UpdateProduct(product models.Products) error {
|
||||
@@ -308,6 +329,54 @@ func (s *productService) DeleteProductLocation(tenantid, locationid, productid i
|
||||
return s.repo.DeleteProductLocation(tenantid, locationid, productid)
|
||||
}
|
||||
|
||||
// catalogueFactsOf collects the catalogue fields the product table has no
|
||||
// column for, so an import keeps them instead of leaving them behind.
|
||||
//
|
||||
// Only what the catalogue actually stated: an empty field is omitted rather
|
||||
// than written as `""` or `[]`, so a reader can tell "the catalogue did not say"
|
||||
// from "the catalogue said none". The drawer prints a row per fact and an empty
|
||||
// string would print an empty row.
|
||||
//
|
||||
// The keys are the catalogue's own wire names. They are what the console
|
||||
// already reads off a live catalogue row, so the same rendering works against
|
||||
// either source without a translation layer in between.
|
||||
func catalogueFactsOf(p *models.CatalogueProduct) map[string]any {
|
||||
facts := map[string]any{}
|
||||
if p == nil {
|
||||
return facts
|
||||
}
|
||||
|
||||
put := func(key, value string) {
|
||||
if v := strings.TrimSpace(value); v != "" {
|
||||
facts[key] = v
|
||||
}
|
||||
}
|
||||
putList := func(key string, values []string) {
|
||||
kept := make([]string, 0, len(values))
|
||||
for _, v := range values {
|
||||
if t := strings.TrimSpace(v); t != "" {
|
||||
kept = append(kept, t)
|
||||
}
|
||||
}
|
||||
if len(kept) > 0 {
|
||||
facts[key] = kept
|
||||
}
|
||||
}
|
||||
|
||||
put("title", p.Title)
|
||||
put("category", p.Category)
|
||||
put("variant_key", p.VariantKey)
|
||||
put("sku_source", p.SKUSource)
|
||||
put("price_range", p.PriceRange)
|
||||
put("fssai_license", p.FSSAILicense)
|
||||
put("search_query", p.SearchQuery)
|
||||
putList("providers", p.Providers)
|
||||
putList("highlights", p.Highlights)
|
||||
putList("nutrients", p.Nutrients)
|
||||
|
||||
return facts
|
||||
}
|
||||
|
||||
// ImportCatalogueProduct bridges a global catalogue product (CatalogueDB) into
|
||||
// a tenant's own store catalogue: it snapshots the catalogue product into the
|
||||
// tenant's `products` table on first import (keyed on brand+catalogueid so
|
||||
@@ -417,6 +486,27 @@ func (s *productService) ImportCatalogueProduct(reqs []models.ImportCataloguePro
|
||||
Taxpercent: req.Taxpercent,
|
||||
Approve: 1,
|
||||
}
|
||||
// Everything the snapshot has no column for, kept as the catalogue
|
||||
// stated it.
|
||||
//
|
||||
// Ten of the catalogue's eighteen fields used to stop here. Two of
|
||||
// them SHOULD — `category` is remapped to the platform's own
|
||||
// categoryid, and `price_range` is replaced by the price the shop
|
||||
// sets — but they are kept anyway, because what other retailers
|
||||
// charge is the most useful thing on the drawer when somebody is
|
||||
// deciding what to charge, and the catalogue's own category is how
|
||||
// a mis-filed product gets noticed.
|
||||
//
|
||||
// Encoding failure is swallowed, like the images below: the product
|
||||
// is worth creating without its facts, and refusing an import over
|
||||
// a nutrition line would be the wrong trade.
|
||||
if encoded, err := json.Marshal(catalogueFactsOf(catalogueProduct)); err == nil {
|
||||
snapshot.Cataloguefacts = string(encoded)
|
||||
} else {
|
||||
log.Printf("import: could not encode catalogue facts for %s/%d: %v",
|
||||
req.Brand, req.Catalogueid, err)
|
||||
}
|
||||
|
||||
if len(catalogueProduct.Images) > 0 {
|
||||
// The first stays where every reader already looks for it.
|
||||
snapshot.Productimage = catalogueProduct.Images[0]
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
package services
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"testing"
|
||||
|
||||
"nearle/models"
|
||||
@@ -47,6 +48,10 @@ type fakeProductRepo struct {
|
||||
publishedRefs []models.ProductLocationRef
|
||||
created []models.Products
|
||||
categorySet map[int][2]int // productid -> {categoryid, subcategoryid}
|
||||
|
||||
// Set to make the insert fail, for the tests that check a failed create
|
||||
// does not hand back a half-made product.
|
||||
createErr error
|
||||
}
|
||||
|
||||
func newFakeRepo() *fakeProductRepo {
|
||||
@@ -119,6 +124,9 @@ func (f *fakeProductRepo) UpdateProductCategory(productid, categoryid, subcatego
|
||||
// re-import branch, so this only became reachable when publishing did.
|
||||
func (f *fakeProductRepo) CreateProductReturningID(product models.Products) (int, error) {
|
||||
f.calls = append(f.calls, "CreateProductReturningID")
|
||||
if f.createErr != nil {
|
||||
return 0, f.createErr
|
||||
}
|
||||
f.created = append(f.created, product)
|
||||
return 9001, nil
|
||||
}
|
||||
@@ -767,3 +775,72 @@ func TestPricingFilterDoesNotDisturbTheCallersSlice(t *testing.T) {
|
||||
t.Error("the caller's slice was modified")
|
||||
}
|
||||
}
|
||||
|
||||
/* ── Creating a product hands back its id ───────────────────────────────────
|
||||
*
|
||||
* `POST /products/create` answered `productid: 0` for every product it created:
|
||||
* the repository took the struct by value, GORM wrote the generated id onto
|
||||
* that copy, and the copy went out of scope. The endpoint is the only way to
|
||||
* create a product, and a product cannot be priced or stocked without its id,
|
||||
* so every caller had to re-read the tenant's whole catalogue and find its own
|
||||
* row again by SKU — a column nothing enforces, in an importer that creates
|
||||
* duplicates by design.
|
||||
*/
|
||||
|
||||
func TestCreateProductReturnsTheIdTheDatabaseAssigned(t *testing.T) {
|
||||
repo := &fakeProductRepo{}
|
||||
svc := NewProductService(repo, &fakeCatalogueService{})
|
||||
|
||||
created, err := svc.CreateProduct(models.Products{
|
||||
Tenantid: 9001,
|
||||
Productname: "Test Rice 5kg",
|
||||
Productsku: "TM-RICE-5K",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("CreateProduct: %v", err)
|
||||
}
|
||||
|
||||
// 9001 is what the fake's CreateProductReturningID returns. The point is
|
||||
// that it reaches the caller at all — it used to be dropped.
|
||||
if created.Productid != 9001 {
|
||||
t.Errorf("productid = %d, want 9001 — the id was lost on the way out", created.Productid)
|
||||
}
|
||||
|
||||
// The rest of the product survives the round trip, because the response is
|
||||
// what the console shows and what it prices and stocks against.
|
||||
if created.Productsku != "TM-RICE-5K" || created.Productname != "Test Rice 5kg" {
|
||||
t.Errorf("the product came back altered: %+v", created)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCreateProductGoesThroughTheOneCreatePath(t *testing.T) {
|
||||
// There were two repository methods doing this same INSERT, one of which
|
||||
// discarded the id. Only one remains, and this is what pins that: a second
|
||||
// path would have to be added here to be used at all.
|
||||
repo := &fakeProductRepo{}
|
||||
svc := NewProductService(repo, &fakeCatalogueService{})
|
||||
|
||||
if _, err := svc.CreateProduct(models.Products{Tenantid: 9001}); err != nil {
|
||||
t.Fatalf("CreateProduct: %v", err)
|
||||
}
|
||||
|
||||
if len(repo.calls) != 1 || repo.calls[0] != "CreateProductReturningID" {
|
||||
t.Errorf("want exactly one call to CreateProductReturningID, got %v", repo.calls)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAFailedCreateReturnsNoProduct(t *testing.T) {
|
||||
// The caller prices and stocks against what comes back, so a half-made
|
||||
// product with a zero id would be worse than an error — it would send a
|
||||
// price and a stock movement to product 0.
|
||||
repo := &fakeProductRepo{createErr: errors.New("duplicate key")}
|
||||
svc := NewProductService(repo, &fakeCatalogueService{})
|
||||
|
||||
created, err := svc.CreateProduct(models.Products{Tenantid: 9001, Productsku: "DUP"})
|
||||
if err == nil {
|
||||
t.Fatal("a failed insert was reported as a success")
|
||||
}
|
||||
if created.Productid != 0 || created.Productsku != "" {
|
||||
t.Errorf("a product was returned for a failed create: %+v", created)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -48,18 +48,19 @@ const (
|
||||
scanLookupTimeout = 5 * time.Second
|
||||
scanMaxLabelLen = 200
|
||||
scanCatalogueTopK = 15
|
||||
// Below this the best hit is not shown as a match at all.
|
||||
scanMinScore = 0.30
|
||||
// How close the runner-up has to be before the leader stops being an
|
||||
// answer and the two become a question. See isAmbiguous.
|
||||
// Below this the best hit is not shown as a match at all. A correct label
|
||||
// scores ~0.92 against its own product's vector and ~0.23 against an
|
||||
// unrelated one, so the floor sits in the empty middle of that split
|
||||
// rather than just above the unrelated band: at 0.30, "Paracetamol"
|
||||
// came back as "Paneer Makhni 500ml" (0.304) — a near-miss on an
|
||||
// unrelated row clears a floor set that close to the noise.
|
||||
scanMinScore = 0.50
|
||||
// How close the runner-up may be before the leader stops being an answer
|
||||
// and the two become a question. See isAmbiguous.
|
||||
scanAmbiguityMargin = 0.06
|
||||
// A "did you mean?" list longer than this is not a choice, it is a
|
||||
// catalogue — the customer is standing in a shop holding a packet.
|
||||
scanMaxCandidates = 10
|
||||
// How much of the winning product's name the label has to account for
|
||||
// before it counts as having identified it. See isAmbiguous and
|
||||
// textScore.
|
||||
scanSpecificEnough = 0.55
|
||||
)
|
||||
|
||||
// ScanErrors the controller maps to statuses. Everything else is a 500.
|
||||
@@ -401,6 +402,14 @@ func (s *scanService) Confirm(ctx context.Context, req models.ScanConfirmRequest
|
||||
}
|
||||
resp.Store = chosen
|
||||
|
||||
// Distance on the store they tapped, for every outcome and not just the
|
||||
// out-of-stock one below: the app renders this store from the reply it
|
||||
// gets. The phone's fix is free to parse; the saved address costs a
|
||||
// query, so it is only reached for on the path that also ranks other
|
||||
// outlets. Without either, distanceKm leaves the -1 the repository set.
|
||||
lat, lng, hasPos := utils.ParseLatLng(string(req.Latitude), string(req.Longitude))
|
||||
chosen.DistanceKm = distanceKm(*chosen, lat, lng, hasPos)
|
||||
|
||||
row, err := s.repo.ProductAt(ctx, req.Tenantid, req.Locationid, req.Productid)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -428,10 +437,10 @@ func (s *scanService) Confirm(ctx context.Context, req models.ScanConfirmRequest
|
||||
}
|
||||
|
||||
// The same product elsewhere, nearest first, with enough of it.
|
||||
lat, lng, hasPos := utils.ParseLatLng(string(req.Latitude), string(req.Longitude))
|
||||
if !hasPos {
|
||||
if hl, hg, ok, err := s.repo.CustomerHome(ctx, req.Customerid); err == nil && ok {
|
||||
lat, lng, hasPos = hl, hg, true
|
||||
chosen.DistanceKm = distanceKm(*chosen, lat, lng, hasPos)
|
||||
}
|
||||
}
|
||||
others := make([]models.ScanStore, 0, len(stores))
|
||||
@@ -516,12 +525,7 @@ func (s *scanService) Stores(ctx context.Context, customerid int, latStr, lngStr
|
||||
|
||||
type scoredHit struct {
|
||||
repositories.CatalogueHit
|
||||
// score ranks; text says how specifically the label names THIS product.
|
||||
// Kept apart because they answer different questions: a vector neighbour
|
||||
// can rank first while the label ("britannia") names no one product, and
|
||||
// only the second number knows that.
|
||||
score float64
|
||||
text float64
|
||||
}
|
||||
|
||||
func (h scoredHit) toMatch(method string) models.ScanCatalogueMatch {
|
||||
@@ -554,12 +558,11 @@ func (s *scanService) searchCatalogue(ctx context.Context, label string) ([]scor
|
||||
method = "vector+text"
|
||||
}
|
||||
|
||||
tokens := utils.SearchTokens(label)
|
||||
|
||||
if cached, ok := s.repo.CachedHits(ctx, method+":"+s.modelName(), label); ok {
|
||||
return scoreCachedHits(cached, label, tokens), method, nil
|
||||
return scoreCachedHits(cached), method, nil
|
||||
}
|
||||
|
||||
tokens := utils.SearchTokens(label)
|
||||
byKey := make(map[string]*scoredHit)
|
||||
keyOf := func(h repositories.CatalogueHit) string { return h.Brand + "#" + fmt.Sprint(h.ID) }
|
||||
|
||||
@@ -599,15 +602,10 @@ func (s *scanService) searchCatalogue(ctx context.Context, label string) ([]scor
|
||||
for _, h := range thits {
|
||||
ts := textScore(h, label, tokens)
|
||||
if existing, ok := byKey[keyOf(h)]; ok {
|
||||
// The bonus is proportional: only a text match that actually
|
||||
// names the product confirms a vector hit. A flat +0.10 let a
|
||||
// bare brand name — which matches every one of that brand's
|
||||
// products weakly — inflate all of them equally.
|
||||
existing.score = math.Min(1, math.Max(existing.score, ts)+0.10*ts)
|
||||
existing.text = ts
|
||||
existing.score = math.Min(1, math.Max(existing.score, ts)+0.10)
|
||||
continue
|
||||
}
|
||||
byKey[keyOf(h)] = &scoredHit{CatalogueHit: h, score: ts, text: ts}
|
||||
byKey[keyOf(h)] = &scoredHit{CatalogueHit: h, score: ts}
|
||||
}
|
||||
|
||||
hits := make([]scoredHit, 0, len(byKey))
|
||||
@@ -629,32 +627,42 @@ func (s *scanService) searchCatalogue(ctx context.Context, label string) ([]scor
|
||||
return hits, method, nil
|
||||
}
|
||||
|
||||
// scoreCachedHits restores the ranking the cache holds, and recomputes the
|
||||
// text score from the row itself — the cache carries one number per row, and
|
||||
// recomputing costs nothing while leaving out the specificity signal would
|
||||
// make every cached lookup read as ambiguous.
|
||||
func scoreCachedHits(cached []repositories.CatalogueHit, label string, tokens []string) []scoredHit {
|
||||
func scoreCachedHits(cached []repositories.CatalogueHit) []scoredHit {
|
||||
hits := make([]scoredHit, 0, len(cached))
|
||||
for _, c := range cached {
|
||||
hits = append(hits, scoredHit{
|
||||
CatalogueHit: c,
|
||||
score: 1 - c.Distance,
|
||||
text: textScore(c, label, tokens),
|
||||
})
|
||||
hits = append(hits, scoredHit{CatalogueHit: c, score: 1 - c.Distance})
|
||||
}
|
||||
sortHits(hits)
|
||||
return hits
|
||||
}
|
||||
|
||||
// sortHits ranks by blended score, then by the model's own similarity, and
|
||||
// only then by name. Name alone used to break every tie, which quietly made
|
||||
// punctuation decide relevance: "Parle Monaco Classic" sorts above "Parle-G
|
||||
// Original …" because a space precedes a hyphen in ASCII, so equal-scoring
|
||||
// crackers beat the biscuit that was actually scanned.
|
||||
func sortHits(hits []scoredHit) {
|
||||
sort.SliceStable(hits, func(i, j int) bool {
|
||||
if hits[i].score != hits[j].score {
|
||||
return hits[i].score > hits[j].score
|
||||
}
|
||||
di, dj := vectorRank(hits[i].Distance), vectorRank(hits[j].Distance)
|
||||
if di != dj {
|
||||
return di < dj
|
||||
}
|
||||
return hits[i].ProductName < hits[j].ProductName
|
||||
})
|
||||
}
|
||||
|
||||
// vectorRank orders by cosine distance, nearest first, with a row the model
|
||||
// never saw (-1, text-only) sorting behind every row it did.
|
||||
func vectorRank(d float64) float64 {
|
||||
if d < 0 {
|
||||
return math.MaxFloat64
|
||||
}
|
||||
return d
|
||||
}
|
||||
|
||||
func (s *scanService) modelName() string {
|
||||
if s.embedder == nil {
|
||||
return "none"
|
||||
@@ -674,79 +682,44 @@ func (s *scanService) embed(ctx context.Context, label string) ([]float32, error
|
||||
return v, nil
|
||||
}
|
||||
|
||||
// textScore is how well a catalogue row matches the words Lens read.
|
||||
// textScore is how well a catalogue row's name matches the words Lens read.
|
||||
// The whole label as a substring of the name is near-certain; otherwise the
|
||||
// share of label words found in name+title, scaled so that "all of them"
|
||||
// stops short of the substring case.
|
||||
//
|
||||
// Both directions count, and that is the whole point:
|
||||
//
|
||||
// - labelCoverage — how much of what the customer said this product
|
||||
// accounts for. "Milk Bikis" against "Milk Bikis 100g" is all of it.
|
||||
// - nameCoverage — how much of the product the label accounts for, which
|
||||
// is what makes the match SPECIFIC. "britannia" explains one word of
|
||||
// "Britannia Good Day Cashew Cookies", so it does not identify it.
|
||||
//
|
||||
// The score is their harmonic mean, so a high score needs both.
|
||||
//
|
||||
// This replaces `strings.Contains(name, label) → 0.95`, which asked only the
|
||||
// first question. A bare brand name is a substring of every one of that
|
||||
// brand's products, so all 258 Britannia rows scored 0.95, the tie broke
|
||||
// alphabetically, and the customer was shown one arbitrary biscuit with
|
||||
// "confidence": 0.95. Lens returns a bare wordmark often — it is usually the
|
||||
// most legible thing on a packet — so that was not an edge case.
|
||||
//
|
||||
// Now those rows score ~0.33 and, crucially, score it EQUALLY, which is what
|
||||
// isAmbiguous reads to answer "did you mean?" instead of guessing.
|
||||
// A label that is a substring of MANY names — a bare brand, "britannia" —
|
||||
// therefore scores them all 0.95, identically. That tie is not a flaw to
|
||||
// score around: it is the signal, and isAmbiguous reads it to answer "did
|
||||
// you mean?" rather than letting the sort order pick a winner.
|
||||
func textScore(h repositories.CatalogueHit, label string, tokens []string) float64 {
|
||||
name := strings.ToLower(h.ProductName)
|
||||
hay := name + " " + strings.ToLower(h.Title)
|
||||
label = strings.ToLower(strings.TrimSpace(label))
|
||||
// The substring test compares separator-folded forms, so the brand's own
|
||||
// punctuation does not decide the match: "Parle G", "Parle-G" and
|
||||
// "ParleG" all have to reach "Parle-G Original Glucose Biscuits".
|
||||
if label != "" {
|
||||
foldedName, foldedLabel := utils.FoldSeparators(name), utils.FoldSeparators(label)
|
||||
if foldedLabel != "" && strings.Contains(foldedName, foldedLabel) {
|
||||
return 0.95
|
||||
}
|
||||
// Separators dropped rather than folded. Only for a label long enough
|
||||
// that a run of letters means something — "lay" inside "malayalam" is
|
||||
// not a match anyone wants.
|
||||
if tight := utils.TightenLabel(label); len(tight) >= 4 && strings.Contains(utils.TightenLabel(name), tight) {
|
||||
return 0.95
|
||||
}
|
||||
}
|
||||
if len(tokens) == 0 {
|
||||
return 0
|
||||
}
|
||||
hay := strings.ToLower(h.ProductName + " " + h.Title)
|
||||
|
||||
found := 0
|
||||
for _, t := range tokens {
|
||||
if strings.Contains(hay, t) {
|
||||
found++
|
||||
}
|
||||
}
|
||||
if found == 0 {
|
||||
return 0
|
||||
}
|
||||
labelCoverage := float64(found) / float64(len(tokens))
|
||||
|
||||
// Pack sizes are dropped from both sides (SearchTokens), so "100g" never
|
||||
// counts as a word the label failed to explain.
|
||||
nameTokens := utils.SearchTokens(h.ProductName)
|
||||
if len(nameTokens) == 0 {
|
||||
return 0.5 * labelCoverage
|
||||
}
|
||||
explained := 0
|
||||
for _, n := range nameTokens {
|
||||
for _, t := range tokens {
|
||||
if tokenMatch(n, t) {
|
||||
explained++
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
if explained == 0 {
|
||||
// Matched the title but not the name. Weak, not zero.
|
||||
return 0.4 * labelCoverage
|
||||
}
|
||||
nameCoverage := float64(explained) / float64(len(nameTokens))
|
||||
|
||||
return 2 * labelCoverage * nameCoverage / (labelCoverage + nameCoverage)
|
||||
}
|
||||
|
||||
// tokenMatch is equality, plus containment for words long enough that a
|
||||
// shared prefix means something ("cookie"/"cookies", "chocolate"/"choco").
|
||||
// Short tokens must match exactly, or "day" would match "daybreak".
|
||||
func tokenMatch(a, b string) bool {
|
||||
if a == b {
|
||||
return true
|
||||
}
|
||||
if len(a) >= 5 && strings.Contains(b, a) {
|
||||
return true
|
||||
}
|
||||
return len(b) >= 5 && strings.Contains(a, b)
|
||||
return 0.8 * float64(found) / float64(len(tokens))
|
||||
}
|
||||
|
||||
// productKey identifies a product across its pack sizes: the catalogue's own
|
||||
@@ -781,28 +754,19 @@ func distinctProducts(hits []scoredHit) []scoredHit {
|
||||
}
|
||||
|
||||
// isAmbiguous reports that naming the leader as THE match would be a guess
|
||||
// dressed up as an answer. Two ways that happens:
|
||||
// dressed up as an answer, because something else is level with it.
|
||||
//
|
||||
// 1. Something else is level with it. A margin rather than an absolute
|
||||
// threshold, because what matters is not how high the best score is but
|
||||
// whether anything is tied with it.
|
||||
// 2. Nothing is level, but the label does not actually name a product —
|
||||
// a bare brand, a generic word, or a spelling the catalogue does not
|
||||
// carry. The leader may still rank first on vector similarity, and
|
||||
// ranking first among vague matches is not identification.
|
||||
// A margin rather than an absolute threshold: what matters is not how high
|
||||
// the best score is but whether anything is tied with it. A bare brand name
|
||||
// is a substring of every one of that brand's names, so textScore gives them
|
||||
// all 0.95 — a perfect tie at a HIGH score, which no floor would catch.
|
||||
//
|
||||
// Erring towards asking is deliberate. Asking costs the customer one tap on
|
||||
// a picture; guessing wrong costs them the wrong biscuit and costs us the
|
||||
// belief that the scanner works. An exact product name still scores ~1.0 on
|
||||
// specificity, so the common case is unaffected.
|
||||
// belief that the scanner works. A label that names one product leaves the
|
||||
// runner-up far behind, so the common case is unaffected.
|
||||
func isAmbiguous(distinct []scoredHit) bool {
|
||||
if len(distinct) < 2 {
|
||||
return false
|
||||
}
|
||||
if distinct[1].score >= distinct[0].score-scanAmbiguityMargin {
|
||||
return true
|
||||
}
|
||||
return distinct[0].text < scanSpecificEnough
|
||||
return len(distinct) >= 2 && distinct[1].score >= distinct[0].score-scanAmbiguityMargin
|
||||
}
|
||||
|
||||
// catalogueFamily is `of` and its other pack sizes, drawn from hits.
|
||||
|
||||
@@ -453,11 +453,11 @@ Ambiguity.
|
||||
|
||||
Lens hands back whatever was most legible on the packet, and on a packet that
|
||||
is very often the brand wordmark alone. "britannia" fits 258 catalogue rows
|
||||
equally well, so there is no best one — and the old scoring said otherwise:
|
||||
every product whose name contained the label scored 0.95, the tie broke
|
||||
alphabetically, and the customer was shown one arbitrary biscuit with
|
||||
"confidence": 0.95 and a price. These tests are the contract that it asks
|
||||
instead.
|
||||
equally well, so there is no best one. textScore gives all of them 0.95 —
|
||||
correctly, the label IS in every one of those names — and with nothing to
|
||||
read that tie, the sort order picked a winner and the customer was shown one
|
||||
arbitrary biscuit with "confidence": 0.95 and a price. These tests are the
|
||||
contract that it asks instead.
|
||||
*/
|
||||
|
||||
// Three different Britannia products, of which the customer's stores stock
|
||||
@@ -515,8 +515,12 @@ func TestABareBrandNameAsksInsteadOfGuessing(t *testing.T) {
|
||||
t.Errorf("only Marie Gold is stocked, but %s reports available", c.ProductName)
|
||||
}
|
||||
}
|
||||
if resp.Confidence >= 0.55 {
|
||||
t.Errorf("confidence should record how weak the identification was, got %v", resp.Confidence)
|
||||
// Note what confidence does NOT say here. The label appears verbatim in
|
||||
// all three names, so relevance is high — and the answer is still a
|
||||
// question. An app that gated on `confidence` instead of `ambiguous`
|
||||
// would show a price for the wrong biscuit, which is the whole bug.
|
||||
if resp.Confidence < 0.9 {
|
||||
t.Errorf("a verbatim brand match scores high; %v suggests the scoring changed", resp.Confidence)
|
||||
}
|
||||
if !strings.Contains(resp.Message, "Which one") {
|
||||
t.Errorf("the message should ask, got %q", resp.Message)
|
||||
@@ -565,27 +569,31 @@ func TestASpecificLabelStillWinsOutright(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestTextScoreRewardsSpecificityNotJustOverlap(t *testing.T) {
|
||||
// The property the ambiguity check rests on: a label that is a substring of
|
||||
// several names scores them EQUALLY. Nothing downstream can tell "did you
|
||||
// mean?" from "found it" if a formula breaks that tie on name length, word
|
||||
// count or anything else incidental — which is how one arbitrary Britannia
|
||||
// biscuit used to come back with a price on it.
|
||||
func TestABrandNameScoresItsProductsIdentically(t *testing.T) {
|
||||
cashew := repositories.CatalogueHit{ProductName: "Britannia Good Day Cashew Cookies 200g"}
|
||||
butter := repositories.CatalogueHit{ProductName: "Britannia Good Day Butter Cookies 100g"}
|
||||
// Deliberately a much shorter name: length must not become a tie-breaker.
|
||||
marie := repositories.CatalogueHit{ProductName: "Britannia Marie Gold"}
|
||||
|
||||
brandOnly := textScore(cashew, "britannia", utils.SearchTokens("britannia"))
|
||||
if brandOnly > 0.45 {
|
||||
t.Errorf("a brand name explains one word of five and must not score as an identification, got %.3f", brandOnly)
|
||||
tokens := utils.SearchTokens("britannia")
|
||||
a, b, c := textScore(cashew, "britannia", tokens), textScore(butter, "britannia", tokens), textScore(marie, "britannia", tokens)
|
||||
if a != b || b != c {
|
||||
t.Fatalf("a brand must score its products equally, got %.3f / %.3f / %.3f", a, b, c)
|
||||
}
|
||||
if brandOnly != textScore(butter, "britannia", utils.SearchTokens("britannia")) {
|
||||
t.Error("a brand must score its products equally — that tie is what makes the label read as ambiguous")
|
||||
if a == 0 {
|
||||
t.Fatal("the brand name is in every one of those names; scoring it 0 would hide them all")
|
||||
}
|
||||
|
||||
full := textScore(cashew, "Britannia Good Day Cashew Cookies", utils.SearchTokens("Britannia Good Day Cashew Cookies"))
|
||||
if full < 0.95 {
|
||||
t.Errorf("the product's own name should be near-certain, got %.3f", full)
|
||||
}
|
||||
|
||||
// A pack size on either side is not a word the label failed to explain.
|
||||
sized := textScore(repositories.CatalogueHit{ProductName: "Milk Bikis 100g"}, "Milk Bikis", utils.SearchTokens("Milk Bikis"))
|
||||
if sized < 0.95 {
|
||||
t.Errorf("pack sizes must not count against the match, got %.3f", sized)
|
||||
// And a label that does name a product must NOT tie with its siblings,
|
||||
// or everything would be a question.
|
||||
specific := utils.SearchTokens("good day cashew")
|
||||
if textScore(cashew, "good day cashew", specific) <= textScore(butter, "good day cashew", specific) {
|
||||
t.Error("a label naming one product must outscore its siblings")
|
||||
}
|
||||
|
||||
if none := textScore(cashew, "dabur honey", utils.SearchTokens("dabur honey")); none != 0 {
|
||||
@@ -657,3 +665,120 @@ func TestAMissingCatalogueRefIsNotAMatch(t *testing.T) {
|
||||
t.Errorf("the message should say the product is gone, got %q", resp.Message)
|
||||
}
|
||||
}
|
||||
|
||||
// A vector neighbour that is merely not-quite-unrelated used to clear the old
|
||||
// 0.30 floor: in production "Paracetamol" came back as "Paneer Makhni 500ml"
|
||||
// on a 0.304 similarity. Correct labels land near 0.92, so nothing this weak
|
||||
// is a match.
|
||||
func TestLookupRefusesANearMissAboveTheOldFloor(t *testing.T) {
|
||||
repo := newLookupFixture()
|
||||
repo.vector = []repositories.CatalogueHit{{Brand: "amul", ID: 4, ProductName: "Paneer Makhni 500ml", Distance: 0.696}} // score 0.304
|
||||
repo.text = nil
|
||||
svc := NewScanService(repo, fakeEmbedder{vec: []float32{0.1}})
|
||||
|
||||
resp, err := svc.Lookup(context.Background(), models.ScanLookupRequest{Customerid: 5, Label: "Paracetamol"})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if resp.Match != nil {
|
||||
t.Fatalf("0.304 is a near-miss, not a match; got %+v", resp.Match)
|
||||
}
|
||||
if resp.Available || len(resp.Stores) != 0 {
|
||||
t.Fatalf("nothing should be offered without a match; got %+v", resp)
|
||||
}
|
||||
}
|
||||
|
||||
// Confirm answers about the store the customer tapped, so that store carries a
|
||||
// distance on every outcome — not only on the out-of-stock path that ranks
|
||||
// alternatives. Absent any position it stays -1, the documented "unknown".
|
||||
func TestConfirmReportsDistanceToTheChosenStore(t *testing.T) {
|
||||
repo := newLookupFixture()
|
||||
repo.at = map[int]*repositories.StoreOptionRow{200: &repo.options[1]}
|
||||
svc := NewScanService(repo, nil)
|
||||
req := models.ScanConfirmRequest{Customerid: 5, Tenantid: 2, Locationid: 20, Productid: 200, Quantity: 4}
|
||||
|
||||
withPos := req
|
||||
withPos.Latitude, withPos.Longitude = "11.035", "77.035"
|
||||
resp, err := svc.Confirm(context.Background(), withPos)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !resp.Ok || resp.Store == nil {
|
||||
t.Fatalf("expected the in-stock answer, got %+v", resp)
|
||||
}
|
||||
if resp.Store.DistanceKm <= 0 {
|
||||
t.Fatalf("the phone sent a fix, so the tapped store has a distance; got %v", resp.Store.DistanceKm)
|
||||
}
|
||||
|
||||
// No fix from the phone, but a saved address on file.
|
||||
repo.homeLat, repo.homeLng, repo.homeOK = 11.035, 77.035, true
|
||||
resp, err = svc.Confirm(context.Background(), req)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if resp.Store == nil || resp.Store.DistanceKm != -1 {
|
||||
t.Fatalf("in stock is answered without reaching for the saved address; got %v", resp.Store)
|
||||
}
|
||||
|
||||
// Neither: unknown, and the app sorts it last.
|
||||
repo.homeOK = false
|
||||
resp, err = svc.Confirm(context.Background(), req)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if resp.Store == nil || resp.Store.DistanceKm != -1 {
|
||||
t.Fatalf("no position at all is -1; got %v", resp.Store)
|
||||
}
|
||||
}
|
||||
|
||||
var parleG = repositories.CatalogueHit{Brand: "parle", ID: 1, ProductName: "Parle-G Original Glucose Biscuits 250g", Title: "Parle-G", VariantKey: "parle_g", ImageID: "parle_parle_g_250g", Distance: 0.20}
|
||||
var monaco = repositories.CatalogueHit{Brand: "parle", ID: 2, ProductName: "Parle Monaco Classic Regular 200g", Title: "Monaco", VariantKey: "monaco", ImageID: "parle_monaco_200g", Distance: 0.20}
|
||||
|
||||
// Lens reads "Parle-G" off the packet and the customer types "Parle G". Both
|
||||
// spellings, and the run-together one, have to reach the biscuit — not the
|
||||
// salted cracker that merely shares a brand. In production "Parle G" returned
|
||||
// "Parle Monaco Classic Regular 200g" at a confident 0.9.
|
||||
func TestLookupMatchesAHyphenatedNameHoweverItIsWritten(t *testing.T) {
|
||||
for _, label := range []string{"Parle G", "Parle-G", "ParleG", "parle g"} {
|
||||
repo := newLookupFixture()
|
||||
repo.vector = []repositories.CatalogueHit{monaco, parleG} // model puts the cracker first
|
||||
repo.text = []repositories.CatalogueHit{monaco, parleG}
|
||||
svc := NewScanService(repo, fakeEmbedder{vec: []float32{0.1}})
|
||||
|
||||
resp, err := svc.Lookup(context.Background(), models.ScanLookupRequest{Customerid: 5, Label: label})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if resp.Match == nil {
|
||||
t.Fatalf("%q: a stocked product went unrecognised", label)
|
||||
}
|
||||
if resp.Match.Catalogueid != parleG.ID {
|
||||
t.Fatalf("%q: matched %q (%.3f), want Parle-G", label, resp.Match.ProductName, resp.Match.Score)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Equal blended scores used to be settled by product name, which let ASCII
|
||||
// decide relevance: a space sorts before a hyphen, so "Parle Monaco …" beat
|
||||
// "Parle-G …". The model's own similarity settles it instead.
|
||||
func TestSortHitsBreaksTiesOnSimilarityNotPunctuation(t *testing.T) {
|
||||
near := parleG
|
||||
near.Distance = 0.10 // the model is surer about this one
|
||||
far := monaco
|
||||
far.Distance = 0.40
|
||||
|
||||
hits := []scoredHit{{CatalogueHit: far, score: 0.9}, {CatalogueHit: near, score: 0.9}}
|
||||
sortHits(hits)
|
||||
if hits[0].ID != near.ID {
|
||||
t.Fatalf("the nearer vector should win a tie, got %q", hits[0].ProductName)
|
||||
}
|
||||
|
||||
// A row the model never scored (-1, text-only) ranks behind one it did.
|
||||
textOnly := parleG
|
||||
textOnly.Distance = -1
|
||||
hits = []scoredHit{{CatalogueHit: textOnly, score: 0.9}, {CatalogueHit: far, score: 0.9}}
|
||||
sortHits(hits)
|
||||
if hits[0].ID != far.ID {
|
||||
t.Fatalf("a scored row outranks an unscored one, got %q", hits[0].ProductName)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
package services
|
||||
|
||||
import (
|
||||
"errors"
|
||||
|
||||
"nearle/models"
|
||||
"nearle/repositories"
|
||||
"time"
|
||||
@@ -22,6 +24,21 @@ func NewStockRequestService(repo repositories.StockRequestRepository, productSer
|
||||
}
|
||||
|
||||
func (s *stockRequestService) CreateStockRequest(req *models.StockRequest) error {
|
||||
// A request for nothing is not a request.
|
||||
//
|
||||
// Nothing downstream rejected it, so a branch could raise a request for
|
||||
// zero units and it sat in the admin's queue looking exactly like a real
|
||||
// one — and approving it moved no stock, which reads as the ledger being
|
||||
// broken rather than the request being empty. A negative would move stock
|
||||
// the wrong way on receipt, since UpdateStockRequest writes Qty straight
|
||||
// into the ledger as an 'in'.
|
||||
//
|
||||
// Returned as an ordinary error: the controller already reports per-item
|
||||
// reasons, so one bad row in a batch is named and the rest still land.
|
||||
if req.Qty <= 0 {
|
||||
return errors.New("quantity must be more than zero")
|
||||
}
|
||||
|
||||
return s.repo.CreateStockRequest(req)
|
||||
}
|
||||
|
||||
|
||||
69
utils/geo.go
69
utils/geo.go
@@ -69,23 +69,86 @@ func parseClock(s string) (int, bool) {
|
||||
}
|
||||
|
||||
// SearchTokens splits a label into the words worth matching on: lowercased,
|
||||
// punctuation stripped, single characters and pack-size noise dropped. "Milk
|
||||
// Bikis 100g" → ["milk", "bikis"]; the size is matched separately, if at all.
|
||||
// punctuation stripped, pack-size noise dropped. "Milk Bikis 100g" →
|
||||
// ["milk", "bikis"]; the size is matched separately, if at all.
|
||||
//
|
||||
// A single character is kept when it follows a word, because in this
|
||||
// catalogue that character is often the whole product: the "G" in "Parle G",
|
||||
// the "K" in "Special K". Dropping it made "Parle G" score the same against
|
||||
// "Parle-G Original Glucose Biscuits" as against "Parle Monaco Classic", and
|
||||
// the tie went to Monaco. It is still dropped when it stands alone — a
|
||||
// one-letter label is not a search — and bare multipliers ("2 x 50gm") are
|
||||
// never words.
|
||||
func SearchTokens(label string) []string {
|
||||
var tokens []string
|
||||
seen := make(map[string]bool)
|
||||
kept := 0 // multi-character tokens so far: a lone letter needs one
|
||||
var filler []string // packaging words, kept only if nothing else survives
|
||||
for _, raw := range strings.FieldsFunc(strings.ToLower(label), func(r rune) bool {
|
||||
return !(r >= 'a' && r <= 'z' || r >= '0' && r <= '9')
|
||||
}) {
|
||||
if len(raw) < 2 || isPackSize(raw) || seen[raw] {
|
||||
if isPackSize(raw) || seen[raw] {
|
||||
continue
|
||||
}
|
||||
if len(raw) < 2 && (kept == 0 || isMultiplier(raw)) {
|
||||
continue
|
||||
}
|
||||
seen[raw] = true
|
||||
if isPackaging(raw) {
|
||||
filler = append(filler, raw)
|
||||
continue
|
||||
}
|
||||
if len(raw) > 1 {
|
||||
kept++
|
||||
}
|
||||
tokens = append(tokens, raw)
|
||||
}
|
||||
// "Dettol bottle pack" is a scan of Dettol. Only when the label is nothing
|
||||
// but packaging does that packaging become the search.
|
||||
if len(tokens) == 0 {
|
||||
return filler
|
||||
}
|
||||
return tokens
|
||||
}
|
||||
|
||||
// isPackaging is what the label says about the wrapper rather than the
|
||||
// product: "Dettol bottle pack", "Parle G biscuit pack". Lens reads these off
|
||||
// the packet and no catalogue name carries them, so every one of them used to
|
||||
// be a word the row had to contain — and "Dettol bottle pack" found no Dettol
|
||||
// at all. Treated like pack sizes: real words, just not the product's name.
|
||||
func isPackaging(tok string) bool {
|
||||
switch tok {
|
||||
case "pack", "packs", "packet", "packets", "bottle", "bottles",
|
||||
"box", "boxes", "jar", "jars", "tin", "tins", "pouch", "pouches",
|
||||
"carton", "cartons", "sachet", "sachets", "combo", "refill":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// isMultiplier is the "x" of "2 x 50gm" and the "n" of a multipack — a single
|
||||
// character that joins sizes rather than naming a product.
|
||||
func isMultiplier(tok string) bool {
|
||||
return tok == "x" || tok == "n"
|
||||
}
|
||||
|
||||
// FoldSeparators turns every run of punctuation into one space, so a label
|
||||
// typed without the brand's own punctuation still matches it: "Parle G" and
|
||||
// "Parle-G" both fold to "parle g". Lens reads letterforms off a packet, and
|
||||
// people type what they see, so the hyphen is not reliably either present or
|
||||
// absent on the way in.
|
||||
func FoldSeparators(s string) string {
|
||||
return strings.Join(strings.FieldsFunc(strings.ToLower(s), func(r rune) bool {
|
||||
return !(r >= 'a' && r <= 'z' || r >= '0' && r <= '9')
|
||||
}), " ")
|
||||
}
|
||||
|
||||
// TightenLabel removes separators outright rather than folding them, catching
|
||||
// the other way people write a hyphenated name: "ParleG" against "Parle-G".
|
||||
func TightenLabel(s string) string {
|
||||
return strings.ReplaceAll(FoldSeparators(s), " ", "")
|
||||
}
|
||||
|
||||
// isPackSize is "100g", "1kg", "500ml", "2l", "250gm" — a number with a unit
|
||||
// glued on, or a bare number.
|
||||
func isPackSize(tok string) bool {
|
||||
|
||||
@@ -63,3 +63,68 @@ func TestSearchTokens(t *testing.T) {
|
||||
t.Error("pack sizes alone are not searchable")
|
||||
}
|
||||
}
|
||||
|
||||
// A single letter is often the whole product name in this catalogue, so it
|
||||
// survives when it follows a word — but not when it stands alone, and not
|
||||
// when it is the multiplier in a pack size.
|
||||
func TestSearchTokensKeepsALetterThatFollowsAWord(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
label string
|
||||
want []string
|
||||
}{
|
||||
{"Parle G", []string{"parle", "g"}},
|
||||
{"Parle-G", []string{"parle", "g"}},
|
||||
{"Special K Original", []string{"special", "k", "original"}},
|
||||
{"G", nil}, // a letter alone is not a search
|
||||
{"2 x 50gm", nil}, // multiplier and pack size, no words
|
||||
{"Milk Bikis 100g, Britannia (2 x 50gm)", []string{"milk", "bikis", "britannia"}},
|
||||
} {
|
||||
got := SearchTokens(tc.label)
|
||||
if len(got) != len(tc.want) {
|
||||
t.Fatalf("%q: got %v want %v", tc.label, got, tc.want)
|
||||
}
|
||||
for i := range tc.want {
|
||||
if got[i] != tc.want[i] {
|
||||
t.Fatalf("%q: got %v want %v", tc.label, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestFoldAndTightenSeparators(t *testing.T) {
|
||||
if got := FoldSeparators("Parle-G Original"); got != "parle g original" {
|
||||
t.Fatalf("fold: got %q", got)
|
||||
}
|
||||
if FoldSeparators("Parle-G") != FoldSeparators("Parle G") {
|
||||
t.Error("a hyphen and a space are the same separator to us")
|
||||
}
|
||||
if got := TightenLabel("Parle-G"); got != "parleg" {
|
||||
t.Fatalf("tighten: got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Lens reads the wrapper as well as the product. No catalogue name carries
|
||||
// "bottle" or "pack", so requiring them found no Dettol at all.
|
||||
func TestSearchTokensDropsPackagingWords(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
label string
|
||||
want []string
|
||||
}{
|
||||
{"Dettol bottle pack", []string{"dettol"}},
|
||||
{"Parle G biscuit pack", []string{"parle", "g", "biscuit"}},
|
||||
{"Milk Bikis pack", []string{"milk", "bikis"}},
|
||||
{"Nescafe jar 50g", []string{"nescafe"}},
|
||||
{"pack", []string{"pack"}}, // nothing else: the wrapper is the search
|
||||
{"combo pack", []string{"combo", "pack"}}, // ditto, both kept
|
||||
} {
|
||||
got := SearchTokens(tc.label)
|
||||
if len(got) != len(tc.want) {
|
||||
t.Fatalf("%q: got %v want %v", tc.label, got, tc.want)
|
||||
}
|
||||
for i := range tc.want {
|
||||
if got[i] != tc.want[i] {
|
||||
t.Fatalf("%q: got %v want %v", tc.label, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user