Catalog feature updates on column fields
This commit is contained in:
248
scripts/backfill_barcode_identity.py
Normal file
248
scripts/backfill_barcode_identity.py
Normal file
@@ -0,0 +1,248 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Fill gtin / ean13 / upc / barcode_type from barcodes the catalog already holds.
|
||||
|
||||
WHY
|
||||
Measured against production on 2026-09-08:
|
||||
|
||||
barcode 300 rows (18.4%)
|
||||
gtin 30 rows (8.7% of the tables that had the column)
|
||||
ean13 22 rows (6.4%)
|
||||
upc 0 rows (0.0%)
|
||||
|
||||
Every one of the missing values is arithmetic on digits already sitting in
|
||||
the same row. Nothing needs to be looked up, matched or fetched. They were
|
||||
empty because the only code that computed them lived inside the network
|
||||
cascade (`ENABLE_BARCODE_LOOKUP`, false in production) and because
|
||||
`upsert_brand_products` dropped the fields before they reached Postgres.
|
||||
|
||||
`BarcodeIdentityStage` now does this for every NEW upload. This script does
|
||||
it once for the rows already stored.
|
||||
|
||||
WHAT IT TOUCHES
|
||||
brand_<slug>.gtin, .ean13, .upc, .barcode_type - and ONLY where they are
|
||||
currently empty. Every UPDATE pins the row's own current barcode in its
|
||||
WHERE clause, so a concurrent write is never lost.
|
||||
|
||||
It also records provenance in `field_sources` under the derived keys, so
|
||||
the coverage report can tell a derived value from a looked-up one.
|
||||
|
||||
WHAT IT WILL NOT DO
|
||||
* It will not change, reformat or delete `barcode`. A barcode that fails
|
||||
checksum validation is REPORTED and skipped - the 38 rows holding the
|
||||
placeholder `8900000000000.0` are found this way, not repaired. Repairing
|
||||
them needs a real source, which is a different job.
|
||||
* It will not overwrite a gtin/ean13/upc that already has a value, even if
|
||||
it disagrees with the barcode. A disagreement is reported instead: it
|
||||
means one of the two is wrong and a script should not pick.
|
||||
* It makes no network request of any kind.
|
||||
|
||||
USAGE
|
||||
python -m scripts.backfill_barcode_identity # dry run
|
||||
python -m scripts.backfill_barcode_identity --brand amul # repeatable
|
||||
python -m scripts.backfill_barcode_identity --apply
|
||||
python -m scripts.backfill_barcode_identity --json
|
||||
|
||||
`--dry-run` is the default and `--apply` must be explicit: backend/.env points
|
||||
at the PRODUCTION database. The target host is printed on startup.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import logging
|
||||
import sys
|
||||
from collections import Counter
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from psycopg.types.json import Json
|
||||
|
||||
from app.infrastructure.settings import DB_HOST, DB_NAME
|
||||
from app.services.enrichment.barcode.models import BarcodeType
|
||||
from app.services.enrichment.barcode.validators import (
|
||||
classify_barcode_type,
|
||||
to_ean13,
|
||||
validate_barcode,
|
||||
)
|
||||
from app.services.vector_store import _connect
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
||||
logger = logging.getLogger("backfill_barcode_identity")
|
||||
|
||||
DERIVED = ("gtin", "ean13", "upc", "barcode_type")
|
||||
|
||||
|
||||
def brand_tables(cur, only: Optional[List[str]] = None) -> List[str]:
|
||||
cur.execute(
|
||||
"SELECT table_name FROM information_schema.tables "
|
||||
"WHERE table_schema = 'public' AND table_name LIKE 'brand\\_%' "
|
||||
"ORDER BY table_name"
|
||||
)
|
||||
tables = [r[0] for r in cur.fetchall()]
|
||||
if only:
|
||||
wanted = {f"brand_{s.strip().lower().replace(' ', '_').replace('-', '_')}"
|
||||
for s in only}
|
||||
tables = [t for t in tables if t in wanted]
|
||||
return tables
|
||||
|
||||
|
||||
def has_columns(cur, table: str) -> bool:
|
||||
"""Every script here probes information_schema before selecting, because
|
||||
the column set genuinely differed per table until very recently."""
|
||||
cur.execute(
|
||||
"SELECT column_name FROM information_schema.columns "
|
||||
"WHERE table_schema = 'public' AND table_name = %s",
|
||||
(table,),
|
||||
)
|
||||
present = {r[0] for r in cur.fetchall()}
|
||||
return {"barcode", *DERIVED, "field_sources"} <= present
|
||||
|
||||
|
||||
def derive(barcode: str) -> Optional[Dict[str, Any]]:
|
||||
code = validate_barcode(barcode)
|
||||
if not code:
|
||||
return None
|
||||
kind = classify_barcode_type(code)
|
||||
return {
|
||||
"gtin": code,
|
||||
"ean13": to_ean13(code),
|
||||
"upc": code if kind is BarcodeType.UPC_A else None,
|
||||
"barcode_type": kind.value,
|
||||
}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
ap = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument("--brand", action="append", dest="brands")
|
||||
ap.add_argument("--apply", action="store_true")
|
||||
ap.add_argument("--dry-run", action="store_true", default=False)
|
||||
ap.add_argument("--json", action="store_true")
|
||||
args = ap.parse_args()
|
||||
|
||||
apply = args.apply and not args.dry_run
|
||||
|
||||
conn = _connect()
|
||||
if conn is None:
|
||||
logger.error("Database unreachable - nothing to do.")
|
||||
return 2
|
||||
|
||||
logger.info("database : %s / %s", DB_HOST, DB_NAME)
|
||||
logger.info("mode : %s", "APPLY (writing)" if apply else "dry run (no writes)")
|
||||
logger.info("")
|
||||
|
||||
tally = Counter()
|
||||
invalid: List[Dict[str, str]] = []
|
||||
conflicts: List[Dict[str, Any]] = []
|
||||
per_brand: Dict[str, int] = {}
|
||||
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
for table in brand_tables(cur, args.brands):
|
||||
if not has_columns(cur, table):
|
||||
tally["tables_skipped_missing_columns"] += 1
|
||||
continue
|
||||
|
||||
cur.execute(
|
||||
f'SELECT id, product_name, barcode, gtin, ean13, upc, barcode_type, '
|
||||
f'field_sources FROM "{table}" '
|
||||
f"WHERE barcode IS NOT NULL AND btrim(barcode) <> ''"
|
||||
)
|
||||
rows = cur.fetchall()
|
||||
written = 0
|
||||
|
||||
for rid, name, barcode, gtin, ean13, upc, btype, sources in rows:
|
||||
tally["barcoded_rows"] += 1
|
||||
derived = derive(barcode)
|
||||
if derived is None:
|
||||
tally["invalid_barcode"] += 1
|
||||
invalid.append({"table": table, "product": name, "barcode": barcode})
|
||||
continue
|
||||
|
||||
current = {"gtin": gtin, "ean13": ean13, "upc": upc, "barcode_type": btype}
|
||||
# Only fill blanks; report a populated value that disagrees.
|
||||
updates = {}
|
||||
for col, want in derived.items():
|
||||
have = current.get(col)
|
||||
if have is None or str(have).strip() == "":
|
||||
if want is not None:
|
||||
updates[col] = want
|
||||
elif str(have).strip() != str(want or "").strip():
|
||||
conflicts.append({"table": table, "product": name, "column": col,
|
||||
"stored": have, "derived": want})
|
||||
if not updates:
|
||||
tally["already_complete"] += 1
|
||||
continue
|
||||
|
||||
merged = dict(sources or {})
|
||||
for col in updates:
|
||||
merged[col] = {"method": "derived",
|
||||
"source": "validators.validate_barcode"}
|
||||
|
||||
tally["rows_to_update"] += 1
|
||||
for col in updates:
|
||||
tally[f"fill_{col}"] += 1
|
||||
written += 1
|
||||
|
||||
if apply:
|
||||
assignments = ", ".join(f"{c} = %s" for c in updates)
|
||||
cur.execute(
|
||||
f'UPDATE "{table}" SET {assignments}, field_sources = %s, '
|
||||
f"updated_at = CURRENT_TIMESTAMP "
|
||||
f"WHERE id = %s AND barcode = %s",
|
||||
(*updates.values(), Json(merged), rid, barcode),
|
||||
)
|
||||
|
||||
if written:
|
||||
per_brand[table] = written
|
||||
|
||||
if apply:
|
||||
conn.commit()
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
if args.json:
|
||||
print(json.dumps({"applied": apply, "tally": dict(tally),
|
||||
"per_brand": per_brand, "invalid": invalid,
|
||||
"conflicts": conflicts}, indent=2))
|
||||
return 0
|
||||
|
||||
for table, n in sorted(per_brand.items(), key=lambda kv: -kv[1]):
|
||||
logger.info(" %-32s %d row(s)", table, n)
|
||||
logger.info("")
|
||||
logger.info("barcoded rows %d", tally["barcoded_rows"])
|
||||
logger.info(" not a valid GTIN %d", tally["invalid_barcode"])
|
||||
logger.info(" already complete %d", tally["already_complete"])
|
||||
logger.info(" %s %d",
|
||||
"updated" if apply else "to update", tally["rows_to_update"])
|
||||
for col in DERIVED:
|
||||
logger.info(" %-14s %d", col, tally[f"fill_{col}"])
|
||||
|
||||
if invalid:
|
||||
logger.info("")
|
||||
logger.info("%d row(s) hold something that is not a barcode (left untouched):",
|
||||
len(invalid))
|
||||
for item in invalid[:10]:
|
||||
logger.info(" %-28s %s", item["product"][:28], item["barcode"])
|
||||
if len(invalid) > 10:
|
||||
logger.info(" ... and %d more", len(invalid) - 10)
|
||||
|
||||
if conflicts:
|
||||
logger.info("")
|
||||
logger.info("%d stored value(s) DISAGREE with the barcode (left untouched, "
|
||||
"one of the two is wrong):", len(conflicts))
|
||||
for c in conflicts[:10]:
|
||||
logger.info(" %-28s %s stored=%s derived=%s",
|
||||
c["product"][:28], c["column"], c["stored"], c["derived"])
|
||||
|
||||
if not apply and tally["rows_to_update"]:
|
||||
logger.info("")
|
||||
logger.info("Dry run - nothing written. Re-run with --apply to commit.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -218,8 +218,16 @@ def main() -> int:
|
||||
candidate_brand=product.get("brands") or "",
|
||||
candidate_size=product.get("quantity") or "",
|
||||
)
|
||||
# barcode_is_identity mirrors what fetch_verified_nutrition_by_barcode
|
||||
# passes, and it has to: this gate runs FIRST, so without it the row is
|
||||
# rejected here and the service's relaxed check is never reached. That
|
||||
# is exactly what happened - a run on 2026-09-08 reported 149 of 300
|
||||
# rows as "found, wrong product" where the barcode had resolved
|
||||
# perfectly and OFF simply stores the short name ("Munch" for our
|
||||
# "Nestle Munch 8.9g"). See matching.name_is_contained.
|
||||
matched, _sim = is_match(candidate, row["brand"], row["product_name"],
|
||||
row["size"], min_name_similarity=args.min_similarity)
|
||||
row["size"], min_name_similarity=args.min_similarity,
|
||||
barcode_is_identity=True)
|
||||
if not matched:
|
||||
rejected.append(line)
|
||||
time.sleep(PAUSE_SECONDS)
|
||||
|
||||
212
scripts/backfill_offline_fields.py
Normal file
212
scripts/backfill_offline_fields.py
Normal file
@@ -0,0 +1,212 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Fill the fields that need no network: FSSAI licences, and not-applicable marks.
|
||||
|
||||
WHY
|
||||
Two separate gaps, both closable from data already on this machine.
|
||||
|
||||
1. FSSAI licences on brands that HAVE one.
|
||||
Measured 2026-09-08: 264 rows have no `fssai_license`. Of those, 185
|
||||
belong to brands with a curated licence in
|
||||
`brand_registry.FSSAI_LICENSES` - Amul 10, HUL 82, MTR 39, Dabur 17,
|
||||
CavinKare 17, Nestle 16, Kaleesuwari 4. The map knows the answer; the
|
||||
rows were written before stage 1 filled it, or by a path that skipped it.
|
||||
|
||||
Consensus over a brand's own rows was the other candidate mechanism and
|
||||
it fills ZERO of these - every brand with blanks either already has a
|
||||
mapping or has no populated row to learn from. It is left in place for
|
||||
future uploads into an established brand, but it is not what closes this.
|
||||
|
||||
2. Marking what cannot apply.
|
||||
Roughly 30% of the catalog is shampoo, soap, detergent and toothpaste.
|
||||
Those rows will never have nutrients, a health score or an FSSAI FOOD
|
||||
licence, and a coverage report that counts them as "missing" shows a
|
||||
permanent red number - which is exactly the pressure that eventually
|
||||
gets it "fixed" by inventing values. This writes an explicit
|
||||
`not_applicable` into `field_sources` so the report can exclude them
|
||||
honestly.
|
||||
|
||||
WHAT IT TOUCHES
|
||||
* brand_<slug>.fssai_license - ONLY where blank, and ONLY from
|
||||
`FSSAI_LICENSES`. Never from another brand, never a constant.
|
||||
* brand_<slug>.field_sources - the provenance record for both operations.
|
||||
|
||||
WHAT IT WILL NOT DO
|
||||
* It will not invent an FSSAI number. A brand absent from the curated map
|
||||
gets nothing. `10012042000244` is Lion Dates' real licence and the reason
|
||||
this rule is written down - see tests/test_no_fabricated_identifiers.py.
|
||||
* It will not overwrite an existing licence, even one that disagrees with
|
||||
the map. A disagreement is reported instead.
|
||||
* It makes no network request.
|
||||
|
||||
USAGE
|
||||
python -m scripts.backfill_offline_fields # dry run
|
||||
python -m scripts.backfill_offline_fields --apply
|
||||
python -m scripts.backfill_offline_fields --json
|
||||
|
||||
`--dry-run` is the default; backend/.env points at PRODUCTION.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import logging
|
||||
import sys
|
||||
from collections import Counter
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from psycopg.types.json import Json
|
||||
|
||||
from app.infrastructure.settings import DB_HOST, DB_NAME
|
||||
from app.services.brand_registry import get_fssai_license
|
||||
from app.services.consumability import is_non_consumable
|
||||
from app.services.vector_store import _connect
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
||||
logger = logging.getLogger("backfill_offline_fields")
|
||||
|
||||
# Columns a non-consumable row can never have a value for.
|
||||
NON_FOOD_NA = ("nutrients", "nutrients_per_100g", "nutrition_score",
|
||||
"health_score", "fssai_license")
|
||||
|
||||
|
||||
def brand_tables(cur, only: Optional[List[str]] = None) -> List[str]:
|
||||
cur.execute(
|
||||
"SELECT table_name FROM information_schema.tables "
|
||||
"WHERE table_schema = 'public' AND table_name LIKE 'brand\\_%' "
|
||||
"AND table_name <> 'brand_zzsmoketest' ORDER BY table_name"
|
||||
)
|
||||
tables = [r[0] for r in cur.fetchall()]
|
||||
if only:
|
||||
wanted = {f"brand_{s.strip().lower().replace(' ', '_').replace('-', '_')}"
|
||||
for s in only}
|
||||
tables = [t for t in tables if t in wanted]
|
||||
return tables
|
||||
|
||||
|
||||
def main() -> int:
|
||||
ap = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument("--brand", action="append", dest="brands")
|
||||
ap.add_argument("--apply", action="store_true")
|
||||
ap.add_argument("--dry-run", action="store_true", default=False)
|
||||
ap.add_argument("--json", action="store_true")
|
||||
args = ap.parse_args()
|
||||
|
||||
apply = args.apply and not args.dry_run
|
||||
|
||||
conn = _connect()
|
||||
if conn is None:
|
||||
logger.error("Database unreachable.")
|
||||
return 2
|
||||
|
||||
logger.info("database : %s / %s", DB_HOST, DB_NAME)
|
||||
logger.info("mode : %s", "APPLY (writing)" if apply else "dry run (no writes)")
|
||||
logger.info("")
|
||||
|
||||
tally = Counter()
|
||||
per_brand: Dict[str, Dict[str, int]] = {}
|
||||
disagreements: List[Dict[str, str]] = []
|
||||
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
for table in brand_tables(cur, args.brands):
|
||||
display = table[len("brand_"):].replace("_", " ")
|
||||
mapped = get_fssai_license(display)
|
||||
|
||||
cur.execute(
|
||||
f'SELECT id, product_name, category, fssai_license, field_sources '
|
||||
f'FROM "{table}"'
|
||||
)
|
||||
rows = cur.fetchall()
|
||||
counts = Counter()
|
||||
|
||||
for rid, name, category, licence, sources in rows:
|
||||
merged = dict(sources or {})
|
||||
updates: Dict[str, Any] = {}
|
||||
|
||||
non_food = is_non_consumable(category or "", name or "")
|
||||
|
||||
if non_food:
|
||||
for column in NON_FOOD_NA:
|
||||
if merged.get(column, {}).get("method") != "not_applicable":
|
||||
merged[column] = {"method": "not_applicable",
|
||||
"source": "non_consumable_product"}
|
||||
counts["marked_not_applicable"] += 1
|
||||
elif not (licence or "").strip():
|
||||
if mapped:
|
||||
updates["fssai_license"] = mapped
|
||||
merged["fssai_license"] = {"method": "sourced",
|
||||
"source": "brand_registry"}
|
||||
counts["fssai_filled"] += 1
|
||||
else:
|
||||
merged["fssai_license"] = {"method": "unknown",
|
||||
"source": "no_registry_entry"}
|
||||
counts["fssai_unknown"] += 1
|
||||
elif mapped and licence.strip() != mapped:
|
||||
disagreements.append({"table": table, "product": name,
|
||||
"stored": licence, "registry": mapped})
|
||||
counts["fssai_disagrees"] += 1
|
||||
|
||||
if merged == (sources or {}) and not updates:
|
||||
continue
|
||||
|
||||
if apply:
|
||||
if updates:
|
||||
cur.execute(
|
||||
f'UPDATE "{table}" SET fssai_license = %s, '
|
||||
f"field_sources = %s, updated_at = CURRENT_TIMESTAMP "
|
||||
f"WHERE id = %s AND (fssai_license IS NULL "
|
||||
f" OR btrim(fssai_license) = '')",
|
||||
(updates["fssai_license"], Json(merged), rid),
|
||||
)
|
||||
else:
|
||||
cur.execute(
|
||||
f'UPDATE "{table}" SET field_sources = %s, '
|
||||
f"updated_at = CURRENT_TIMESTAMP WHERE id = %s",
|
||||
(Json(merged), rid),
|
||||
)
|
||||
|
||||
if counts:
|
||||
per_brand[table] = dict(counts)
|
||||
tally.update(counts)
|
||||
|
||||
if apply:
|
||||
conn.commit()
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
if args.json:
|
||||
print(json.dumps({"applied": apply, "tally": dict(tally),
|
||||
"per_brand": per_brand,
|
||||
"disagreements": disagreements}, indent=2))
|
||||
return 0
|
||||
|
||||
for table, counts in sorted(per_brand.items(),
|
||||
key=lambda kv: -sum(kv[1].values())):
|
||||
parts = ", ".join(f"{k.replace('_', ' ')} {v}" for k, v in sorted(counts.items()))
|
||||
logger.info(" %-30s %s", table, parts)
|
||||
|
||||
logger.info("")
|
||||
logger.info("fssai filled from the brand registry %d", tally["fssai_filled"])
|
||||
logger.info("fssai left blank, no registry entry %d", tally["fssai_unknown"])
|
||||
logger.info("rows marked not-applicable (non-food) %d", tally["marked_not_applicable"])
|
||||
if disagreements:
|
||||
logger.info("")
|
||||
logger.info("%d row(s) hold a licence that DISAGREES with the registry "
|
||||
"(left untouched):", len(disagreements))
|
||||
for d in disagreements[:10]:
|
||||
logger.info(" %-28s stored=%s registry=%s",
|
||||
d["product"][:28], d["stored"], d["registry"])
|
||||
|
||||
if not apply and sum(tally.values()):
|
||||
logger.info("")
|
||||
logger.info("Dry run - nothing written. Re-run with --apply to commit.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
279
scripts/catalog_coverage.py
Normal file
279
scripts/catalog_coverage.py
Normal file
@@ -0,0 +1,279 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
How completely is the catalog filled, and where did each value come from?
|
||||
|
||||
WHY
|
||||
"Fill every column" is not actually the goal, and a report that treats it
|
||||
as one produces a permanent, unfixable red number that somebody eventually
|
||||
"fixes" by inventing data. Three of these columns can never be filled for
|
||||
large parts of the catalog, and that is correct:
|
||||
|
||||
* `upc` - every barcode here is GS1 India (prefix 890), which issues
|
||||
EAN-13 and GTIN-8. UPC-A is a North American symbology. Measured: 0 of
|
||||
300 barcodes are UPC-A, and none ever will be.
|
||||
* `nutrients` / `health_score` - roughly 30% of rows are shampoo, soap,
|
||||
detergent and toothpaste. Soap has no protein content.
|
||||
* `fssai_license` - an FSSAI licence covers a FOOD business. P&G,
|
||||
Colgate-Palmolive and Reckitt Benckiser should not carry one.
|
||||
|
||||
So this report counts four states, not two:
|
||||
|
||||
sourced a real value from a real source
|
||||
derived computed from another field we hold (gtin from barcode)
|
||||
estimated a category-level or consensus guess, flagged as such
|
||||
not applicable cannot exist for this row, and should not
|
||||
MISSING we have not got it yet - the only number worth chasing
|
||||
|
||||
Coverage percentages are taken against the APPLICABLE denominator, so the
|
||||
numbers describe work remaining rather than work impossible.
|
||||
|
||||
WHAT IT TOUCHES
|
||||
Nothing. Every statement is a SELECT. There is no --apply because there is
|
||||
nothing to apply.
|
||||
|
||||
USAGE
|
||||
python -m scripts.catalog_coverage
|
||||
python -m scripts.catalog_coverage --brand amul --brand cadbury
|
||||
python -m scripts.catalog_coverage --column barcode --column nutrients
|
||||
python -m scripts.catalog_coverage --json > coverage.json
|
||||
python -m scripts.catalog_coverage --by-provenance
|
||||
|
||||
Run it before and after any enrichment change: the diff of two --json runs is
|
||||
the evidence that the change did what it claimed.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import logging
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Set
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from app.infrastructure.settings import DB_HOST, DB_NAME
|
||||
from app.services.vector_store import _connect
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
||||
logger = logging.getLogger("catalog_coverage")
|
||||
|
||||
|
||||
# Columns worth reporting on, in the order a reader wants them.
|
||||
TRACKED = [
|
||||
"product_name", "title", "description", "category", "image_url",
|
||||
"price_range", "size_variants", "providers", "highlights",
|
||||
"fssai_license", "product_sku", "hsn_code", "gst_percent", "tax_amount",
|
||||
"selling_price", "final_selling_price",
|
||||
"barcode", "barcode_type", "gtin", "ean13", "upc",
|
||||
"nutrients", "nutrients_per_100g", "nutrition_score", "health_score",
|
||||
]
|
||||
|
||||
# Columns that simply cannot apply to some rows, and the rule for which.
|
||||
#
|
||||
# food_only - meaningless for a non-consumable product
|
||||
# india_only - UPC-A does not occur in a GS1 India catalog
|
||||
# barcoded - derived from a barcode, so absent when the barcode is
|
||||
NOT_APPLICABLE_RULES = {
|
||||
"nutrients": "food_only",
|
||||
"nutrients_per_100g": "food_only",
|
||||
"nutrition_score": "food_only",
|
||||
"health_score": "food_only",
|
||||
"fssai_license": "food_only",
|
||||
"upc": "india_only",
|
||||
"gtin": "barcoded",
|
||||
"ean13": "barcoded",
|
||||
"barcode_type": "barcoded",
|
||||
}
|
||||
|
||||
# A value that is present but means "nothing here".
|
||||
EMPTY_LITERALS = ("", "[]", "{}", "null", "0", "Uncategorized")
|
||||
|
||||
|
||||
def brand_tables(cur, only: Optional[List[str]] = None) -> List[str]:
|
||||
cur.execute(
|
||||
"SELECT table_name FROM information_schema.tables "
|
||||
"WHERE table_schema = 'public' AND table_name LIKE 'brand\\_%' "
|
||||
"AND table_name <> 'brand_zzsmoketest' ORDER BY table_name"
|
||||
)
|
||||
tables = [r[0] for r in cur.fetchall()]
|
||||
if only:
|
||||
wanted = {f"brand_{s.strip().lower().replace(' ', '_').replace('-', '_')}"
|
||||
for s in only}
|
||||
tables = [t for t in tables if t in wanted]
|
||||
return tables
|
||||
|
||||
|
||||
def columns_of(cur, table: str) -> Set[str]:
|
||||
"""Probed per table rather than assumed. The column set genuinely differed
|
||||
per table until the schema migration, and a script that assumes otherwise
|
||||
dies on the first old table it meets."""
|
||||
cur.execute(
|
||||
"SELECT column_name FROM information_schema.columns "
|
||||
"WHERE table_schema = 'public' AND table_name = %s",
|
||||
(table,),
|
||||
)
|
||||
return {r[0] for r in cur.fetchall()}
|
||||
|
||||
|
||||
def _is_non_food(category: str) -> bool:
|
||||
from app.services.consumability import is_non_consumable
|
||||
try:
|
||||
return bool(is_non_consumable(category or "", ""))
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def scan(cur, table: str, wanted: List[str]) -> Dict[str, Dict[str, int]]:
|
||||
present = columns_of(cur, table)
|
||||
cols = [c for c in wanted if c in present]
|
||||
if not cols:
|
||||
return {}
|
||||
|
||||
select = ", ".join(f'"{c}"' for c in cols)
|
||||
extra = ', "category"' if "category" in present else ""
|
||||
barcode_idx = cols.index("barcode") if "barcode" in cols else None
|
||||
|
||||
cur.execute(f'SELECT {select}{extra} FROM "{table}"')
|
||||
rows = cur.fetchall()
|
||||
|
||||
stats: Dict[str, Dict[str, int]] = {
|
||||
c: {"rows": 0, "filled": 0, "not_applicable": 0} for c in cols
|
||||
}
|
||||
|
||||
for row in rows:
|
||||
category = row[len(cols)] if extra else ""
|
||||
non_food = _is_non_food(category)
|
||||
has_barcode = bool(barcode_idx is not None and row[barcode_idx])
|
||||
|
||||
for i, col in enumerate(cols):
|
||||
s = stats[col]
|
||||
s["rows"] += 1
|
||||
|
||||
rule = NOT_APPLICABLE_RULES.get(col)
|
||||
if ((rule == "food_only" and non_food)
|
||||
or (rule == "india_only")
|
||||
or (rule == "barcoded" and not has_barcode)):
|
||||
s["not_applicable"] += 1
|
||||
continue
|
||||
|
||||
value = row[i]
|
||||
if value is None:
|
||||
continue
|
||||
if isinstance(value, (list, tuple, dict)) and not value:
|
||||
continue
|
||||
if isinstance(value, str) and value.strip() in EMPTY_LITERALS:
|
||||
continue
|
||||
s["filled"] += 1
|
||||
|
||||
return stats
|
||||
|
||||
|
||||
def provenance(cur, table: str) -> Dict[str, Dict[str, int]]:
|
||||
"""How each filled value was arrived at, read from `field_sources`."""
|
||||
if "field_sources" not in columns_of(cur, table):
|
||||
return {}
|
||||
cur.execute(f'SELECT field_sources FROM "{table}" '
|
||||
f"WHERE field_sources IS NOT NULL AND field_sources <> '{{}}'::jsonb")
|
||||
out: Dict[str, Dict[str, int]] = {}
|
||||
for (blob,) in cur.fetchall():
|
||||
for column, record in (blob or {}).items():
|
||||
method = (record or {}).get("method", "unspecified")
|
||||
out.setdefault(column, {}).setdefault(method, 0)
|
||||
out[column][method] += 1
|
||||
return out
|
||||
|
||||
|
||||
def main() -> int:
|
||||
ap = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument("--brand", action="append", dest="brands")
|
||||
ap.add_argument("--column", action="append", dest="columns")
|
||||
ap.add_argument("--json", action="store_true")
|
||||
ap.add_argument("--by-provenance", action="store_true",
|
||||
help="also break filled values down by how they were obtained")
|
||||
args = ap.parse_args()
|
||||
|
||||
wanted = args.columns or TRACKED
|
||||
|
||||
conn = _connect()
|
||||
if conn is None:
|
||||
logger.error("Database unreachable.")
|
||||
return 2
|
||||
|
||||
totals: Dict[str, Dict[str, int]] = {c: {"rows": 0, "filled": 0, "not_applicable": 0}
|
||||
for c in wanted}
|
||||
per_brand: Dict[str, Any] = {}
|
||||
prov_totals: Dict[str, Dict[str, int]] = {}
|
||||
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
tables = brand_tables(cur, args.brands)
|
||||
for table in tables:
|
||||
stats = scan(cur, table, wanted)
|
||||
if not stats:
|
||||
continue
|
||||
per_brand[table] = stats
|
||||
for col, s in stats.items():
|
||||
for k in ("rows", "filled", "not_applicable"):
|
||||
totals[col][k] += s[k]
|
||||
|
||||
if args.by_provenance:
|
||||
for col, methods in provenance(cur, table).items():
|
||||
for method, n in methods.items():
|
||||
prov_totals.setdefault(col, {}).setdefault(method, 0)
|
||||
prov_totals[col][method] += n
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
def applicable(s):
|
||||
return s["rows"] - s["not_applicable"]
|
||||
|
||||
if args.json:
|
||||
print(json.dumps({
|
||||
"database": f"{DB_HOST}/{DB_NAME}",
|
||||
"totals": {c: {**s, "applicable": applicable(s),
|
||||
"pct": round(100 * s["filled"] / applicable(s), 1)
|
||||
if applicable(s) else None}
|
||||
for c, s in totals.items() if s["rows"]},
|
||||
"provenance": prov_totals,
|
||||
"per_brand": per_brand,
|
||||
}, indent=2))
|
||||
return 0
|
||||
|
||||
row_count = max((s["rows"] for s in totals.values()), default=0)
|
||||
logger.info("database : %s / %s", DB_HOST, DB_NAME)
|
||||
logger.info("%d product rows across %d brand tables", row_count, len(per_brand))
|
||||
logger.info("")
|
||||
logger.info("%-22s %8s %8s %7s %s", "column", "filled", "of", "pct", "not applicable")
|
||||
logger.info("%s", "-" * 72)
|
||||
|
||||
for col in wanted:
|
||||
s = totals.get(col)
|
||||
if not s or not s["rows"]:
|
||||
continue
|
||||
app_n = applicable(s)
|
||||
pct = f'{100 * s["filled"] / app_n:5.1f}%' if app_n else " -"
|
||||
na = f'{s["not_applicable"]:d}' if s["not_applicable"] else ""
|
||||
flag = ""
|
||||
if app_n and s["filled"] < app_n:
|
||||
flag = f' <- {app_n - s["filled"]} missing'
|
||||
logger.info("%-22s %8d %8d %7s %-6s%s", col, s["filled"], app_n, pct, na, flag)
|
||||
|
||||
if args.by_provenance and prov_totals:
|
||||
logger.info("")
|
||||
logger.info("provenance of filled values (from field_sources)")
|
||||
logger.info("%s", "-" * 72)
|
||||
for col in sorted(prov_totals):
|
||||
methods = ", ".join(f"{m} {n}" for m, n in
|
||||
sorted(prov_totals[col].items(), key=lambda kv: -kv[1]))
|
||||
logger.info("%-22s %s", col, methods)
|
||||
elif args.by_provenance:
|
||||
logger.info("")
|
||||
logger.info("No field_sources recorded yet - run an enrichment pass first.")
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
257
scripts/migrate_brand_schema.py
Normal file
257
scripts/migrate_brand_schema.py
Normal file
@@ -0,0 +1,257 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Bring every brand table up to the current schema.
|
||||
|
||||
WHY
|
||||
`vector_store._ensure_columns` is the only migration mechanism in this
|
||||
repo - there is no Alembic and no migrations directory - and it runs only
|
||||
as a side effect of `ensure_brand_schema`, i.e. on the next WRITE to a
|
||||
table. A brand nobody uploads to therefore never gets a new column.
|
||||
|
||||
Measured on 2026-09-08, before the enrichment columns landed: seven of
|
||||
fifty-six brand tables carried `gtin` / `ean13` / `upc` /
|
||||
`barcode_source` / `barcode_verified` / `barcode_lookup_status` /
|
||||
`barcode_last_updated`, all added out-of-band by hand. The other
|
||||
forty-nine did not, and no code path would ever have added them. This
|
||||
script closes that gap deliberately instead of waiting for a write that
|
||||
may never come.
|
||||
|
||||
WHAT IT TOUCHES
|
||||
* `ALTER TABLE brand_<slug> ADD COLUMN IF NOT EXISTS ...` for every column
|
||||
in `_ensure_columns`' `col_defs` that the table does not already have.
|
||||
* The retroactive UNIQUE index on `image_id`, and the DROP NOT NULL sweep
|
||||
over legacy columns - both are part of `_ensure_columns` and cannot be
|
||||
run separately.
|
||||
|
||||
Nothing else. No row is read, updated or deleted by this script.
|
||||
|
||||
WHAT IT WILL NOT DO
|
||||
* It will not change the type of a column that already exists.
|
||||
`ADD COLUMN IF NOT EXISTS` skips a column that is present, whatever its
|
||||
type. This is deliberate: the seven hand-migrated tables define the
|
||||
types the rest must match, which is why `col_defs` says TIMESTAMP for
|
||||
`barcode_last_updated` and REAL for the tax figures rather than the
|
||||
types those values look like they want. Type drift is REPORTED here,
|
||||
never silently "fixed".
|
||||
* It will not create a brand table that does not exist.
|
||||
* It will not touch `nutrition_facts` or any non-brand table.
|
||||
|
||||
WHY IT IS SAFE ON A LIVE DATABASE
|
||||
`ADD COLUMN` with no DEFAULT and no NOT NULL is a catalogue-only change in
|
||||
PostgreSQL 11+: no table rewrite, no full-table lock, no time proportional
|
||||
to row count. On 1 630 rows across 56 tables this is milliseconds. The one
|
||||
exception is `field_sources`, which does carry a DEFAULT - and since
|
||||
PostgreSQL 11 a non-volatile default is also metadata-only.
|
||||
|
||||
USAGE
|
||||
python -m scripts.migrate_brand_schema # dry run, all brands
|
||||
python -m scripts.migrate_brand_schema --brand amul # repeatable
|
||||
python -m scripts.migrate_brand_schema --apply
|
||||
python -m scripts.migrate_brand_schema --json
|
||||
|
||||
`--dry-run` is the default and `--apply` must be explicit: backend/.env points
|
||||
at the PRODUCTION database, so an accidental run must not be able to write.
|
||||
The target host is printed on startup.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import logging
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from app.infrastructure.settings import DB_HOST, DB_NAME
|
||||
|
||||
from app.services.vector_store import _connect, _ensure_columns
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
||||
logger = logging.getLogger("migrate_brand_schema")
|
||||
|
||||
|
||||
def brand_tables(cur, only: Optional[List[str]] = None) -> List[str]:
|
||||
"""Every brand table actually present, in name order."""
|
||||
cur.execute(
|
||||
"SELECT table_name FROM information_schema.tables "
|
||||
"WHERE table_schema = 'public' AND table_name LIKE 'brand\\_%' "
|
||||
"ORDER BY table_name"
|
||||
)
|
||||
tables = [r[0] for r in cur.fetchall()]
|
||||
if only:
|
||||
wanted = {f"brand_{s.strip().lower().replace(' ', '_').replace('-', '_')}"
|
||||
for s in only}
|
||||
tables = [t for t in tables if t in wanted]
|
||||
return tables
|
||||
|
||||
|
||||
def existing_columns(cur, table: str) -> Dict[str, str]:
|
||||
cur.execute(
|
||||
"SELECT column_name, data_type FROM information_schema.columns "
|
||||
"WHERE table_schema = 'public' AND table_name = %s",
|
||||
(table,),
|
||||
)
|
||||
return {r[0]: r[1] for r in cur.fetchall()}
|
||||
|
||||
|
||||
# What `information_schema.data_type` reports for each col_defs type, so a
|
||||
# type-drift check does not raise false alarms on spelling differences.
|
||||
_TYPE_ALIASES = {
|
||||
"TEXT": {"text"},
|
||||
"TEXT[]": {"ARRAY"},
|
||||
"NUMERIC": {"numeric"},
|
||||
"REAL": {"real"},
|
||||
"BOOLEAN": {"boolean"},
|
||||
"TIMESTAMP": {"timestamp without time zone"},
|
||||
"JSONB": {"jsonb"},
|
||||
"DOUBLE PRECISION": {"double precision"},
|
||||
"vector(384)": {"USER-DEFINED"},
|
||||
}
|
||||
|
||||
|
||||
class RecordingCursor:
|
||||
"""Wraps a real cursor so a dry run can see the statements without
|
||||
executing them. Reads are passed through - the whole point is to compute
|
||||
the diff against what is really on the table."""
|
||||
|
||||
def __init__(self, inner):
|
||||
self._inner = inner
|
||||
self.statements: List[str] = []
|
||||
|
||||
def execute(self, sql, params=None):
|
||||
text = " ".join(str(sql).split())
|
||||
upper = text.upper()
|
||||
if upper.startswith("SELECT"):
|
||||
return self._inner.execute(sql, params)
|
||||
self.statements.append(text)
|
||||
return None
|
||||
|
||||
def fetchall(self):
|
||||
return self._inner.fetchall()
|
||||
|
||||
def fetchone(self):
|
||||
return self._inner.fetchone()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
ap = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument("--brand", action="append", dest="brands",
|
||||
help="brand table suffix; repeatable. Default: every brand table.")
|
||||
ap.add_argument("--apply", action="store_true", help="actually run the ALTERs")
|
||||
ap.add_argument("--dry-run", action="store_true", default=False,
|
||||
help="report only (the default)")
|
||||
ap.add_argument("--json", action="store_true", help="machine-readable output")
|
||||
args = ap.parse_args()
|
||||
|
||||
apply = args.apply and not args.dry_run
|
||||
|
||||
conn = _connect()
|
||||
if conn is None:
|
||||
logger.error("Database unreachable - nothing to do.")
|
||||
return 2
|
||||
|
||||
logger.info("database : %s / %s", DB_HOST, DB_NAME)
|
||||
logger.info("mode : %s", "APPLY (writing)" if apply else "dry run (no writes)")
|
||||
logger.info("")
|
||||
|
||||
declared_types = _declared_types()
|
||||
if not declared_types:
|
||||
logger.error("Could not read col_defs out of _ensure_columns - refusing to "
|
||||
"guess at the schema. Has that function been restructured?")
|
||||
return 2
|
||||
|
||||
report: List[Dict[str, Any]] = []
|
||||
total_missing = 0
|
||||
total_drift = 0
|
||||
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
tables = brand_tables(cur, args.brands)
|
||||
if not tables:
|
||||
logger.error("No brand tables matched.")
|
||||
return 1
|
||||
|
||||
for table in tables:
|
||||
before = existing_columns(cur, table)
|
||||
|
||||
recorder = RecordingCursor(cur)
|
||||
_ensure_columns(recorder, table)
|
||||
adds = [s for s in recorder.statements if "ADD COLUMN" in s]
|
||||
|
||||
# Type drift: a column that exists but whose type is not what
|
||||
# col_defs would have created. Reported, never altered.
|
||||
drift = []
|
||||
for col, declared in declared_types.items():
|
||||
actual = before.get(col)
|
||||
if actual is None:
|
||||
continue
|
||||
allowed = _TYPE_ALIASES.get(declared.upper(), set())
|
||||
if allowed and actual not in allowed:
|
||||
drift.append({"column": col, "declared": declared, "actual": actual})
|
||||
|
||||
entry = {
|
||||
"table": table,
|
||||
"missing_columns": [s.split("ADD COLUMN IF NOT EXISTS ")[1] for s in adds],
|
||||
"type_drift": drift,
|
||||
}
|
||||
report.append(entry)
|
||||
total_missing += len(adds)
|
||||
total_drift += len(drift)
|
||||
|
||||
if apply and adds:
|
||||
_ensure_columns(cur, table)
|
||||
|
||||
if apply:
|
||||
conn.commit()
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
if args.json:
|
||||
print(json.dumps({"applied": apply, "tables": report}, indent=2))
|
||||
return 0
|
||||
|
||||
width = max(len(e["table"]) for e in report)
|
||||
changed = [e for e in report if e["missing_columns"] or e["type_drift"]]
|
||||
for entry in sorted(changed, key=lambda e: -len(e["missing_columns"])):
|
||||
logger.info("%-*s %d column(s) missing", width, entry["table"],
|
||||
len(entry["missing_columns"]))
|
||||
for col in entry["missing_columns"]:
|
||||
logger.info("%-*s + %s", width, "", col)
|
||||
for d in entry["type_drift"]:
|
||||
logger.info("%-*s ! %s is %s, col_defs declares %s (NOT changed)",
|
||||
width, "", d["column"], d["actual"], d["declared"])
|
||||
|
||||
logger.info("")
|
||||
logger.info("%d table(s) scanned, %d already current",
|
||||
len(report), len(report) - len(changed))
|
||||
logger.info("%d column(s) %s, %d type mismatch(es) reported",
|
||||
total_missing, "added" if apply else "would be added", total_drift)
|
||||
if not apply and total_missing:
|
||||
logger.info("")
|
||||
logger.info("Re-run with --apply to write these changes.")
|
||||
return 0
|
||||
|
||||
|
||||
def _declared_types() -> Dict[str, str]:
|
||||
"""The col_defs dict, read back out of the function that owns it.
|
||||
|
||||
Parsed from source rather than duplicated here, so this script cannot
|
||||
drift from the single migration mechanism it exists to drive.
|
||||
"""
|
||||
import ast
|
||||
import inspect
|
||||
import textwrap
|
||||
|
||||
tree = ast.parse(textwrap.dedent(inspect.getsource(_ensure_columns)))
|
||||
for node in ast.walk(tree):
|
||||
if isinstance(node, ast.Assign) and getattr(node.targets[0], "id", "") == "col_defs":
|
||||
return {ast.literal_eval(k): ast.literal_eval(v)
|
||||
for k, v in zip(node.value.keys, node.value.values)}
|
||||
return {}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user