168 lines
6.1 KiB
Go
168 lines
6.1 KiB
Go
package services
|
|
|
|
import (
|
|
"fmt"
|
|
)
|
|
|
|
// Repairing the pointers a shop's products keep back into the global catalogue.
|
|
//
|
|
// Every product imported from the catalogue stores the id of the row it came
|
|
// from. That id is not stable: the catalogue is rebuilt by scrape and renumbered
|
|
// each time, so pepsico's live ids run 3, 6, 9 … 27, 30 and a product imported
|
|
// when it was id 26 now points at nothing. Measured 2026-08-31: eleven of the
|
|
// nineteen links on the platform were dangling.
|
|
//
|
|
// Three things break, all of them quietly:
|
|
//
|
|
// 1. The catalogue browser's "already imported" ticks land on ids that no
|
|
// longer exist, so the rows a shop really holds show as not imported —
|
|
// and importing them again makes a duplicate.
|
|
// 2. Re-importing fails outright: the import reads the catalogue row by id
|
|
// first, and answers "catalogue product not found".
|
|
// 3. The next scrape creates duplicates, because the dedupe keys on the id
|
|
// that just changed.
|
|
//
|
|
// `image_id` fixes all three going forward. This pass fixes what is already
|
|
// stored, and it can only do one of two things to a product — neither of which
|
|
// invents anything:
|
|
//
|
|
// - The id still resolves: adopt that row's `image_id`. The link now survives
|
|
// the next scrape.
|
|
// - It does not: clear the id. NOT a repair by name — the re-scrape that
|
|
// renumbered these also changed their pack sizes (Cheetos Chips 100g became
|
|
// 250g, Kurkure Menthol 10g became 50g), so the product the link named no
|
|
// longer exists in any form and matching by name would attach a shop's
|
|
// product to a DIFFERENT one. Clearing is the honest outcome: the pointer
|
|
// already pointed at nothing, and the product goes on selling from its own
|
|
// row exactly as before.
|
|
|
|
// RelinkOutcome is what happened to one product.
|
|
type RelinkOutcome struct {
|
|
Productid int `json:"productid"`
|
|
Productname string `json:"productname"`
|
|
Brand string `json:"brand"`
|
|
Catalogueid int `json:"catalogueid"`
|
|
// "linked" — the id resolved and the stable key was adopted.
|
|
// "cleared" — the id resolved to nothing and was removed.
|
|
// "already" — the product already carried the right stable key.
|
|
Action string `json:"action"`
|
|
Imageid string `json:"imageid,omitempty"`
|
|
Reason string `json:"reason,omitempty"`
|
|
}
|
|
|
|
// RelinkReport is the whole pass over one tenant.
|
|
type RelinkReport struct {
|
|
Tenantid int `json:"tenantid"`
|
|
Checked int `json:"checked"`
|
|
Linked int `json:"linked"`
|
|
Cleared int `json:"cleared"`
|
|
Already int `json:"already"`
|
|
DryRun bool `json:"dryrun"`
|
|
Outcomes []RelinkOutcome `json:"outcomes"`
|
|
}
|
|
|
|
// RelinkCatalogue checks every catalogue-linked product of one tenant.
|
|
//
|
|
// `dryRun` is the default at the caller, deliberately: this rewrites a column
|
|
// that decides what a browse screen claims a shop already has, and it should be
|
|
// possible to read the whole plan before any of it happens.
|
|
func (s *productService) RelinkCatalogue(tenantid int, dryRun bool) (*RelinkReport, error) {
|
|
products, err := s.repo.ListCatalogueLinkedProducts(tenantid)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
|
|
report := &RelinkReport{Tenantid: tenantid, DryRun: dryRun, Outcomes: []RelinkOutcome{}}
|
|
|
|
for _, product := range products {
|
|
report.Checked++
|
|
outcome := RelinkOutcome{
|
|
Productid: product.Productid,
|
|
Productname: product.Productname,
|
|
Brand: product.Productbrand,
|
|
Catalogueid: product.Catalogueid,
|
|
}
|
|
|
|
row, err := s.catalogueService.GetProductByID(product.Productbrand, int64(product.Catalogueid))
|
|
// A lookup that ERRORS is not the same as one that finds nothing, and
|
|
// the difference decides whether a link is cleared. A catalogue that is
|
|
// down would otherwise read as "every row is gone" and wipe every link
|
|
// on the platform in one pass.
|
|
if err != nil {
|
|
return nil, fmt.Errorf("catalogue lookup failed for %s/%d: %w",
|
|
product.Productbrand, product.Catalogueid, err)
|
|
}
|
|
|
|
if row == nil {
|
|
outcome.Action = "cleared"
|
|
outcome.Reason = "no catalogue row with this id — the scrape that renumbered it also removed this pack size"
|
|
if !dryRun {
|
|
if err := s.repo.SetCatalogueLink(product.Productid, "", 0); err != nil {
|
|
return nil, err
|
|
}
|
|
}
|
|
report.Cleared++
|
|
report.Outcomes = append(report.Outcomes, outcome)
|
|
continue
|
|
}
|
|
|
|
outcome.Imageid = row.ImageID
|
|
if product.Imageid == row.ImageID && row.ImageID != "" {
|
|
outcome.Action = "already"
|
|
report.Already++
|
|
report.Outcomes = append(report.Outcomes, outcome)
|
|
continue
|
|
}
|
|
|
|
// The id resolves, but to a DIFFERENT product than the one stored.
|
|
// Cleared rather than adopted: a renumber can hand an id to an unrelated
|
|
// product, and silently repointing a shop's row at it would attach the
|
|
// wrong catalogue entry — worse than no link at all.
|
|
if !sameProduct(row.ProductName, product.Productname) {
|
|
outcome.Action = "cleared"
|
|
outcome.Reason = fmt.Sprintf("id %d now names %q, not %q",
|
|
product.Catalogueid, row.ProductName, product.Productname)
|
|
outcome.Imageid = ""
|
|
if !dryRun {
|
|
if err := s.repo.SetCatalogueLink(product.Productid, "", 0); err != nil {
|
|
return nil, err
|
|
}
|
|
}
|
|
report.Cleared++
|
|
report.Outcomes = append(report.Outcomes, outcome)
|
|
continue
|
|
}
|
|
|
|
outcome.Action = "linked"
|
|
if !dryRun {
|
|
if err := s.repo.SetCatalogueLink(product.Productid, row.ImageID, int(row.ID)); err != nil {
|
|
return nil, err
|
|
}
|
|
}
|
|
report.Linked++
|
|
report.Outcomes = append(report.Outcomes, outcome)
|
|
}
|
|
|
|
return report, nil
|
|
}
|
|
|
|
// sameProduct is deliberately strict: a name differing by one character is a
|
|
// different image_id and therefore a different product in the catalogue's own
|
|
// terms, so anything looser here would adopt the wrong row.
|
|
func sameProduct(a, b string) bool {
|
|
return normaliseName(a) == normaliseName(b)
|
|
}
|
|
|
|
func normaliseName(value string) string {
|
|
out := make([]rune, 0, len(value))
|
|
for _, r := range value {
|
|
switch {
|
|
case r >= 'a' && r <= 'z', r >= '0' && r <= '9':
|
|
out = append(out, r)
|
|
case r >= 'A' && r <= 'Z':
|
|
out = append(out, r+('a'-'A'))
|
|
}
|
|
}
|
|
return string(out)
|
|
}
|