package services import ( "fmt" ) // Repairing the pointers a shop's products keep back into the global catalogue. // // Every product imported from the catalogue stores the id of the row it came // from. That id is not stable: the catalogue is rebuilt by scrape and renumbered // each time, so pepsico's live ids run 3, 6, 9 … 27, 30 and a product imported // when it was id 26 now points at nothing. Measured 2026-08-31: eleven of the // nineteen links on the platform were dangling. // // Three things break, all of them quietly: // // 1. The catalogue browser's "already imported" ticks land on ids that no // longer exist, so the rows a shop really holds show as not imported — // and importing them again makes a duplicate. // 2. Re-importing fails outright: the import reads the catalogue row by id // first, and answers "catalogue product not found". // 3. The next scrape creates duplicates, because the dedupe keys on the id // that just changed. // // `image_id` fixes all three going forward. This pass fixes what is already // stored, and it can only do one of two things to a product — neither of which // invents anything: // // - The id still resolves: adopt that row's `image_id`. The link now survives // the next scrape. // - It does not: clear the id. NOT a repair by name — the re-scrape that // renumbered these also changed their pack sizes (Cheetos Chips 100g became // 250g, Kurkure Menthol 10g became 50g), so the product the link named no // longer exists in any form and matching by name would attach a shop's // product to a DIFFERENT one. Clearing is the honest outcome: the pointer // already pointed at nothing, and the product goes on selling from its own // row exactly as before. // RelinkOutcome is what happened to one product. type RelinkOutcome struct { Productid int `json:"productid"` Productname string `json:"productname"` Brand string `json:"brand"` Catalogueid int `json:"catalogueid"` // "linked" — the id resolved and the stable key was adopted. // "cleared" — the id resolved to nothing and was removed. // "already" — the product already carried the right stable key. Action string `json:"action"` Imageid string `json:"imageid,omitempty"` Reason string `json:"reason,omitempty"` } // RelinkReport is the whole pass over one tenant. type RelinkReport struct { Tenantid int `json:"tenantid"` Checked int `json:"checked"` Linked int `json:"linked"` Cleared int `json:"cleared"` Already int `json:"already"` DryRun bool `json:"dryrun"` Outcomes []RelinkOutcome `json:"outcomes"` } // RelinkCatalogue checks every catalogue-linked product of one tenant. // // `dryRun` is the default at the caller, deliberately: this rewrites a column // that decides what a browse screen claims a shop already has, and it should be // possible to read the whole plan before any of it happens. func (s *productService) RelinkCatalogue(tenantid int, dryRun bool) (*RelinkReport, error) { products, err := s.repo.ListCatalogueLinkedProducts(tenantid) if err != nil { return nil, err } report := &RelinkReport{Tenantid: tenantid, DryRun: dryRun, Outcomes: []RelinkOutcome{}} for _, product := range products { report.Checked++ outcome := RelinkOutcome{ Productid: product.Productid, Productname: product.Productname, Brand: product.Productbrand, Catalogueid: product.Catalogueid, } row, err := s.catalogueService.GetProductByID(product.Productbrand, int64(product.Catalogueid)) // A lookup that ERRORS is not the same as one that finds nothing, and // the difference decides whether a link is cleared. A catalogue that is // down would otherwise read as "every row is gone" and wipe every link // on the platform in one pass. if err != nil { return nil, fmt.Errorf("catalogue lookup failed for %s/%d: %w", product.Productbrand, product.Catalogueid, err) } if row == nil { outcome.Action = "cleared" outcome.Reason = "no catalogue row with this id — the scrape that renumbered it also removed this pack size" if !dryRun { if err := s.repo.SetCatalogueLink(product.Productid, "", 0); err != nil { return nil, err } } report.Cleared++ report.Outcomes = append(report.Outcomes, outcome) continue } outcome.Imageid = row.ImageID if product.Imageid == row.ImageID && row.ImageID != "" { outcome.Action = "already" report.Already++ report.Outcomes = append(report.Outcomes, outcome) continue } // The id resolves, but to a DIFFERENT product than the one stored. // Cleared rather than adopted: a renumber can hand an id to an unrelated // product, and silently repointing a shop's row at it would attach the // wrong catalogue entry — worse than no link at all. if !sameProduct(row.ProductName, product.Productname) { outcome.Action = "cleared" outcome.Reason = fmt.Sprintf("id %d now names %q, not %q", product.Catalogueid, row.ProductName, product.Productname) outcome.Imageid = "" if !dryRun { if err := s.repo.SetCatalogueLink(product.Productid, "", 0); err != nil { return nil, err } } report.Cleared++ report.Outcomes = append(report.Outcomes, outcome) continue } outcome.Action = "linked" if !dryRun { if err := s.repo.SetCatalogueLink(product.Productid, row.ImageID, int(row.ID)); err != nil { return nil, err } } report.Linked++ report.Outcomes = append(report.Outcomes, outcome) } return report, nil } // sameProduct is deliberately strict: a name differing by one character is a // different image_id and therefore a different product in the catalogue's own // terms, so anything looser here would adopt the wrong row. func sameProduct(a, b string) bool { return normaliseName(a) == normaliseName(b) } func normaliseName(value string) string { out := make([]rune, 0, len(value)) for _, r := range value { switch { case r >= 'a' && r <= 'z', r >= '0' && r <= '9': out = append(out, r) case r >= 'A' && r <= 'Z': out = append(out, r+('a'-'A')) } } return string(out) }