Catalog feature updates on column fields

This commit is contained in:
sriram
2026-09-08 15:18:29 +05:30
parent 2749bee1a3
commit 10b24c6348
60 changed files with 9224 additions and 31 deletions

View File

@@ -0,0 +1,248 @@
#!/usr/bin/env python3
"""
Fill gtin / ean13 / upc / barcode_type from barcodes the catalog already holds.
WHY
Measured against production on 2026-09-08:
barcode 300 rows (18.4%)
gtin 30 rows (8.7% of the tables that had the column)
ean13 22 rows (6.4%)
upc 0 rows (0.0%)
Every one of the missing values is arithmetic on digits already sitting in
the same row. Nothing needs to be looked up, matched or fetched. They were
empty because the only code that computed them lived inside the network
cascade (`ENABLE_BARCODE_LOOKUP`, false in production) and because
`upsert_brand_products` dropped the fields before they reached Postgres.
`BarcodeIdentityStage` now does this for every NEW upload. This script does
it once for the rows already stored.
WHAT IT TOUCHES
brand_<slug>.gtin, .ean13, .upc, .barcode_type - and ONLY where they are
currently empty. Every UPDATE pins the row's own current barcode in its
WHERE clause, so a concurrent write is never lost.
It also records provenance in `field_sources` under the derived keys, so
the coverage report can tell a derived value from a looked-up one.
WHAT IT WILL NOT DO
* It will not change, reformat or delete `barcode`. A barcode that fails
checksum validation is REPORTED and skipped - the 38 rows holding the
placeholder `8900000000000.0` are found this way, not repaired. Repairing
them needs a real source, which is a different job.
* It will not overwrite a gtin/ean13/upc that already has a value, even if
it disagrees with the barcode. A disagreement is reported instead: it
means one of the two is wrong and a script should not pick.
* It makes no network request of any kind.
USAGE
python -m scripts.backfill_barcode_identity # dry run
python -m scripts.backfill_barcode_identity --brand amul # repeatable
python -m scripts.backfill_barcode_identity --apply
python -m scripts.backfill_barcode_identity --json
`--dry-run` is the default and `--apply` must be explicit: backend/.env points
at the PRODUCTION database. The target host is printed on startup.
"""
from __future__ import annotations
import argparse
import json
import logging
import sys
from collections import Counter
from pathlib import Path
from typing import Any, Dict, List, Optional
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from psycopg.types.json import Json
from app.infrastructure.settings import DB_HOST, DB_NAME
from app.services.enrichment.barcode.models import BarcodeType
from app.services.enrichment.barcode.validators import (
classify_barcode_type,
to_ean13,
validate_barcode,
)
from app.services.vector_store import _connect
logging.basicConfig(level=logging.INFO, format="%(message)s")
logger = logging.getLogger("backfill_barcode_identity")
DERIVED = ("gtin", "ean13", "upc", "barcode_type")
def brand_tables(cur, only: Optional[List[str]] = None) -> List[str]:
cur.execute(
"SELECT table_name FROM information_schema.tables "
"WHERE table_schema = 'public' AND table_name LIKE 'brand\\_%' "
"ORDER BY table_name"
)
tables = [r[0] for r in cur.fetchall()]
if only:
wanted = {f"brand_{s.strip().lower().replace(' ', '_').replace('-', '_')}"
for s in only}
tables = [t for t in tables if t in wanted]
return tables
def has_columns(cur, table: str) -> bool:
"""Every script here probes information_schema before selecting, because
the column set genuinely differed per table until very recently."""
cur.execute(
"SELECT column_name FROM information_schema.columns "
"WHERE table_schema = 'public' AND table_name = %s",
(table,),
)
present = {r[0] for r in cur.fetchall()}
return {"barcode", *DERIVED, "field_sources"} <= present
def derive(barcode: str) -> Optional[Dict[str, Any]]:
code = validate_barcode(barcode)
if not code:
return None
kind = classify_barcode_type(code)
return {
"gtin": code,
"ean13": to_ean13(code),
"upc": code if kind is BarcodeType.UPC_A else None,
"barcode_type": kind.value,
}
def main() -> int:
ap = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--brand", action="append", dest="brands")
ap.add_argument("--apply", action="store_true")
ap.add_argument("--dry-run", action="store_true", default=False)
ap.add_argument("--json", action="store_true")
args = ap.parse_args()
apply = args.apply and not args.dry_run
conn = _connect()
if conn is None:
logger.error("Database unreachable - nothing to do.")
return 2
logger.info("database : %s / %s", DB_HOST, DB_NAME)
logger.info("mode : %s", "APPLY (writing)" if apply else "dry run (no writes)")
logger.info("")
tally = Counter()
invalid: List[Dict[str, str]] = []
conflicts: List[Dict[str, Any]] = []
per_brand: Dict[str, int] = {}
try:
with conn.cursor() as cur:
for table in brand_tables(cur, args.brands):
if not has_columns(cur, table):
tally["tables_skipped_missing_columns"] += 1
continue
cur.execute(
f'SELECT id, product_name, barcode, gtin, ean13, upc, barcode_type, '
f'field_sources FROM "{table}" '
f"WHERE barcode IS NOT NULL AND btrim(barcode) <> ''"
)
rows = cur.fetchall()
written = 0
for rid, name, barcode, gtin, ean13, upc, btype, sources in rows:
tally["barcoded_rows"] += 1
derived = derive(barcode)
if derived is None:
tally["invalid_barcode"] += 1
invalid.append({"table": table, "product": name, "barcode": barcode})
continue
current = {"gtin": gtin, "ean13": ean13, "upc": upc, "barcode_type": btype}
# Only fill blanks; report a populated value that disagrees.
updates = {}
for col, want in derived.items():
have = current.get(col)
if have is None or str(have).strip() == "":
if want is not None:
updates[col] = want
elif str(have).strip() != str(want or "").strip():
conflicts.append({"table": table, "product": name, "column": col,
"stored": have, "derived": want})
if not updates:
tally["already_complete"] += 1
continue
merged = dict(sources or {})
for col in updates:
merged[col] = {"method": "derived",
"source": "validators.validate_barcode"}
tally["rows_to_update"] += 1
for col in updates:
tally[f"fill_{col}"] += 1
written += 1
if apply:
assignments = ", ".join(f"{c} = %s" for c in updates)
cur.execute(
f'UPDATE "{table}" SET {assignments}, field_sources = %s, '
f"updated_at = CURRENT_TIMESTAMP "
f"WHERE id = %s AND barcode = %s",
(*updates.values(), Json(merged), rid, barcode),
)
if written:
per_brand[table] = written
if apply:
conn.commit()
finally:
conn.close()
if args.json:
print(json.dumps({"applied": apply, "tally": dict(tally),
"per_brand": per_brand, "invalid": invalid,
"conflicts": conflicts}, indent=2))
return 0
for table, n in sorted(per_brand.items(), key=lambda kv: -kv[1]):
logger.info(" %-32s %d row(s)", table, n)
logger.info("")
logger.info("barcoded rows %d", tally["barcoded_rows"])
logger.info(" not a valid GTIN %d", tally["invalid_barcode"])
logger.info(" already complete %d", tally["already_complete"])
logger.info(" %s %d",
"updated" if apply else "to update", tally["rows_to_update"])
for col in DERIVED:
logger.info(" %-14s %d", col, tally[f"fill_{col}"])
if invalid:
logger.info("")
logger.info("%d row(s) hold something that is not a barcode (left untouched):",
len(invalid))
for item in invalid[:10]:
logger.info(" %-28s %s", item["product"][:28], item["barcode"])
if len(invalid) > 10:
logger.info(" ... and %d more", len(invalid) - 10)
if conflicts:
logger.info("")
logger.info("%d stored value(s) DISAGREE with the barcode (left untouched, "
"one of the two is wrong):", len(conflicts))
for c in conflicts[:10]:
logger.info(" %-28s %s stored=%s derived=%s",
c["product"][:28], c["column"], c["stored"], c["derived"])
if not apply and tally["rows_to_update"]:
logger.info("")
logger.info("Dry run - nothing written. Re-run with --apply to commit.")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -218,8 +218,16 @@ def main() -> int:
candidate_brand=product.get("brands") or "",
candidate_size=product.get("quantity") or "",
)
# barcode_is_identity mirrors what fetch_verified_nutrition_by_barcode
# passes, and it has to: this gate runs FIRST, so without it the row is
# rejected here and the service's relaxed check is never reached. That
# is exactly what happened - a run on 2026-09-08 reported 149 of 300
# rows as "found, wrong product" where the barcode had resolved
# perfectly and OFF simply stores the short name ("Munch" for our
# "Nestle Munch 8.9g"). See matching.name_is_contained.
matched, _sim = is_match(candidate, row["brand"], row["product_name"],
row["size"], min_name_similarity=args.min_similarity)
row["size"], min_name_similarity=args.min_similarity,
barcode_is_identity=True)
if not matched:
rejected.append(line)
time.sleep(PAUSE_SECONDS)

View File

@@ -0,0 +1,212 @@
#!/usr/bin/env python3
"""
Fill the fields that need no network: FSSAI licences, and not-applicable marks.
WHY
Two separate gaps, both closable from data already on this machine.
1. FSSAI licences on brands that HAVE one.
Measured 2026-09-08: 264 rows have no `fssai_license`. Of those, 185
belong to brands with a curated licence in
`brand_registry.FSSAI_LICENSES` - Amul 10, HUL 82, MTR 39, Dabur 17,
CavinKare 17, Nestle 16, Kaleesuwari 4. The map knows the answer; the
rows were written before stage 1 filled it, or by a path that skipped it.
Consensus over a brand's own rows was the other candidate mechanism and
it fills ZERO of these - every brand with blanks either already has a
mapping or has no populated row to learn from. It is left in place for
future uploads into an established brand, but it is not what closes this.
2. Marking what cannot apply.
Roughly 30% of the catalog is shampoo, soap, detergent and toothpaste.
Those rows will never have nutrients, a health score or an FSSAI FOOD
licence, and a coverage report that counts them as "missing" shows a
permanent red number - which is exactly the pressure that eventually
gets it "fixed" by inventing values. This writes an explicit
`not_applicable` into `field_sources` so the report can exclude them
honestly.
WHAT IT TOUCHES
* brand_<slug>.fssai_license - ONLY where blank, and ONLY from
`FSSAI_LICENSES`. Never from another brand, never a constant.
* brand_<slug>.field_sources - the provenance record for both operations.
WHAT IT WILL NOT DO
* It will not invent an FSSAI number. A brand absent from the curated map
gets nothing. `10012042000244` is Lion Dates' real licence and the reason
this rule is written down - see tests/test_no_fabricated_identifiers.py.
* It will not overwrite an existing licence, even one that disagrees with
the map. A disagreement is reported instead.
* It makes no network request.
USAGE
python -m scripts.backfill_offline_fields # dry run
python -m scripts.backfill_offline_fields --apply
python -m scripts.backfill_offline_fields --json
`--dry-run` is the default; backend/.env points at PRODUCTION.
"""
from __future__ import annotations
import argparse
import json
import logging
import sys
from collections import Counter
from pathlib import Path
from typing import Any, Dict, List, Optional
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from psycopg.types.json import Json
from app.infrastructure.settings import DB_HOST, DB_NAME
from app.services.brand_registry import get_fssai_license
from app.services.consumability import is_non_consumable
from app.services.vector_store import _connect
logging.basicConfig(level=logging.INFO, format="%(message)s")
logger = logging.getLogger("backfill_offline_fields")
# Columns a non-consumable row can never have a value for.
NON_FOOD_NA = ("nutrients", "nutrients_per_100g", "nutrition_score",
"health_score", "fssai_license")
def brand_tables(cur, only: Optional[List[str]] = None) -> List[str]:
cur.execute(
"SELECT table_name FROM information_schema.tables "
"WHERE table_schema = 'public' AND table_name LIKE 'brand\\_%' "
"AND table_name <> 'brand_zzsmoketest' ORDER BY table_name"
)
tables = [r[0] for r in cur.fetchall()]
if only:
wanted = {f"brand_{s.strip().lower().replace(' ', '_').replace('-', '_')}"
for s in only}
tables = [t for t in tables if t in wanted]
return tables
def main() -> int:
ap = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--brand", action="append", dest="brands")
ap.add_argument("--apply", action="store_true")
ap.add_argument("--dry-run", action="store_true", default=False)
ap.add_argument("--json", action="store_true")
args = ap.parse_args()
apply = args.apply and not args.dry_run
conn = _connect()
if conn is None:
logger.error("Database unreachable.")
return 2
logger.info("database : %s / %s", DB_HOST, DB_NAME)
logger.info("mode : %s", "APPLY (writing)" if apply else "dry run (no writes)")
logger.info("")
tally = Counter()
per_brand: Dict[str, Dict[str, int]] = {}
disagreements: List[Dict[str, str]] = []
try:
with conn.cursor() as cur:
for table in brand_tables(cur, args.brands):
display = table[len("brand_"):].replace("_", " ")
mapped = get_fssai_license(display)
cur.execute(
f'SELECT id, product_name, category, fssai_license, field_sources '
f'FROM "{table}"'
)
rows = cur.fetchall()
counts = Counter()
for rid, name, category, licence, sources in rows:
merged = dict(sources or {})
updates: Dict[str, Any] = {}
non_food = is_non_consumable(category or "", name or "")
if non_food:
for column in NON_FOOD_NA:
if merged.get(column, {}).get("method") != "not_applicable":
merged[column] = {"method": "not_applicable",
"source": "non_consumable_product"}
counts["marked_not_applicable"] += 1
elif not (licence or "").strip():
if mapped:
updates["fssai_license"] = mapped
merged["fssai_license"] = {"method": "sourced",
"source": "brand_registry"}
counts["fssai_filled"] += 1
else:
merged["fssai_license"] = {"method": "unknown",
"source": "no_registry_entry"}
counts["fssai_unknown"] += 1
elif mapped and licence.strip() != mapped:
disagreements.append({"table": table, "product": name,
"stored": licence, "registry": mapped})
counts["fssai_disagrees"] += 1
if merged == (sources or {}) and not updates:
continue
if apply:
if updates:
cur.execute(
f'UPDATE "{table}" SET fssai_license = %s, '
f"field_sources = %s, updated_at = CURRENT_TIMESTAMP "
f"WHERE id = %s AND (fssai_license IS NULL "
f" OR btrim(fssai_license) = '')",
(updates["fssai_license"], Json(merged), rid),
)
else:
cur.execute(
f'UPDATE "{table}" SET field_sources = %s, '
f"updated_at = CURRENT_TIMESTAMP WHERE id = %s",
(Json(merged), rid),
)
if counts:
per_brand[table] = dict(counts)
tally.update(counts)
if apply:
conn.commit()
finally:
conn.close()
if args.json:
print(json.dumps({"applied": apply, "tally": dict(tally),
"per_brand": per_brand,
"disagreements": disagreements}, indent=2))
return 0
for table, counts in sorted(per_brand.items(),
key=lambda kv: -sum(kv[1].values())):
parts = ", ".join(f"{k.replace('_', ' ')} {v}" for k, v in sorted(counts.items()))
logger.info(" %-30s %s", table, parts)
logger.info("")
logger.info("fssai filled from the brand registry %d", tally["fssai_filled"])
logger.info("fssai left blank, no registry entry %d", tally["fssai_unknown"])
logger.info("rows marked not-applicable (non-food) %d", tally["marked_not_applicable"])
if disagreements:
logger.info("")
logger.info("%d row(s) hold a licence that DISAGREES with the registry "
"(left untouched):", len(disagreements))
for d in disagreements[:10]:
logger.info(" %-28s stored=%s registry=%s",
d["product"][:28], d["stored"], d["registry"])
if not apply and sum(tally.values()):
logger.info("")
logger.info("Dry run - nothing written. Re-run with --apply to commit.")
return 0
if __name__ == "__main__":
raise SystemExit(main())

279
scripts/catalog_coverage.py Normal file
View File

@@ -0,0 +1,279 @@
#!/usr/bin/env python3
"""
How completely is the catalog filled, and where did each value come from?
WHY
"Fill every column" is not actually the goal, and a report that treats it
as one produces a permanent, unfixable red number that somebody eventually
"fixes" by inventing data. Three of these columns can never be filled for
large parts of the catalog, and that is correct:
* `upc` - every barcode here is GS1 India (prefix 890), which issues
EAN-13 and GTIN-8. UPC-A is a North American symbology. Measured: 0 of
300 barcodes are UPC-A, and none ever will be.
* `nutrients` / `health_score` - roughly 30% of rows are shampoo, soap,
detergent and toothpaste. Soap has no protein content.
* `fssai_license` - an FSSAI licence covers a FOOD business. P&G,
Colgate-Palmolive and Reckitt Benckiser should not carry one.
So this report counts four states, not two:
sourced a real value from a real source
derived computed from another field we hold (gtin from barcode)
estimated a category-level or consensus guess, flagged as such
not applicable cannot exist for this row, and should not
MISSING we have not got it yet - the only number worth chasing
Coverage percentages are taken against the APPLICABLE denominator, so the
numbers describe work remaining rather than work impossible.
WHAT IT TOUCHES
Nothing. Every statement is a SELECT. There is no --apply because there is
nothing to apply.
USAGE
python -m scripts.catalog_coverage
python -m scripts.catalog_coverage --brand amul --brand cadbury
python -m scripts.catalog_coverage --column barcode --column nutrients
python -m scripts.catalog_coverage --json > coverage.json
python -m scripts.catalog_coverage --by-provenance
Run it before and after any enrichment change: the diff of two --json runs is
the evidence that the change did what it claimed.
"""
from __future__ import annotations
import argparse
import json
import logging
import sys
from pathlib import Path
from typing import Any, Dict, List, Optional, Set
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.infrastructure.settings import DB_HOST, DB_NAME
from app.services.vector_store import _connect
logging.basicConfig(level=logging.INFO, format="%(message)s")
logger = logging.getLogger("catalog_coverage")
# Columns worth reporting on, in the order a reader wants them.
TRACKED = [
"product_name", "title", "description", "category", "image_url",
"price_range", "size_variants", "providers", "highlights",
"fssai_license", "product_sku", "hsn_code", "gst_percent", "tax_amount",
"selling_price", "final_selling_price",
"barcode", "barcode_type", "gtin", "ean13", "upc",
"nutrients", "nutrients_per_100g", "nutrition_score", "health_score",
]
# Columns that simply cannot apply to some rows, and the rule for which.
#
# food_only - meaningless for a non-consumable product
# india_only - UPC-A does not occur in a GS1 India catalog
# barcoded - derived from a barcode, so absent when the barcode is
NOT_APPLICABLE_RULES = {
"nutrients": "food_only",
"nutrients_per_100g": "food_only",
"nutrition_score": "food_only",
"health_score": "food_only",
"fssai_license": "food_only",
"upc": "india_only",
"gtin": "barcoded",
"ean13": "barcoded",
"barcode_type": "barcoded",
}
# A value that is present but means "nothing here".
EMPTY_LITERALS = ("", "[]", "{}", "null", "0", "Uncategorized")
def brand_tables(cur, only: Optional[List[str]] = None) -> List[str]:
cur.execute(
"SELECT table_name FROM information_schema.tables "
"WHERE table_schema = 'public' AND table_name LIKE 'brand\\_%' "
"AND table_name <> 'brand_zzsmoketest' ORDER BY table_name"
)
tables = [r[0] for r in cur.fetchall()]
if only:
wanted = {f"brand_{s.strip().lower().replace(' ', '_').replace('-', '_')}"
for s in only}
tables = [t for t in tables if t in wanted]
return tables
def columns_of(cur, table: str) -> Set[str]:
"""Probed per table rather than assumed. The column set genuinely differed
per table until the schema migration, and a script that assumes otherwise
dies on the first old table it meets."""
cur.execute(
"SELECT column_name FROM information_schema.columns "
"WHERE table_schema = 'public' AND table_name = %s",
(table,),
)
return {r[0] for r in cur.fetchall()}
def _is_non_food(category: str) -> bool:
from app.services.consumability import is_non_consumable
try:
return bool(is_non_consumable(category or "", ""))
except Exception:
return False
def scan(cur, table: str, wanted: List[str]) -> Dict[str, Dict[str, int]]:
present = columns_of(cur, table)
cols = [c for c in wanted if c in present]
if not cols:
return {}
select = ", ".join(f'"{c}"' for c in cols)
extra = ', "category"' if "category" in present else ""
barcode_idx = cols.index("barcode") if "barcode" in cols else None
cur.execute(f'SELECT {select}{extra} FROM "{table}"')
rows = cur.fetchall()
stats: Dict[str, Dict[str, int]] = {
c: {"rows": 0, "filled": 0, "not_applicable": 0} for c in cols
}
for row in rows:
category = row[len(cols)] if extra else ""
non_food = _is_non_food(category)
has_barcode = bool(barcode_idx is not None and row[barcode_idx])
for i, col in enumerate(cols):
s = stats[col]
s["rows"] += 1
rule = NOT_APPLICABLE_RULES.get(col)
if ((rule == "food_only" and non_food)
or (rule == "india_only")
or (rule == "barcoded" and not has_barcode)):
s["not_applicable"] += 1
continue
value = row[i]
if value is None:
continue
if isinstance(value, (list, tuple, dict)) and not value:
continue
if isinstance(value, str) and value.strip() in EMPTY_LITERALS:
continue
s["filled"] += 1
return stats
def provenance(cur, table: str) -> Dict[str, Dict[str, int]]:
"""How each filled value was arrived at, read from `field_sources`."""
if "field_sources" not in columns_of(cur, table):
return {}
cur.execute(f'SELECT field_sources FROM "{table}" '
f"WHERE field_sources IS NOT NULL AND field_sources <> '{{}}'::jsonb")
out: Dict[str, Dict[str, int]] = {}
for (blob,) in cur.fetchall():
for column, record in (blob or {}).items():
method = (record or {}).get("method", "unspecified")
out.setdefault(column, {}).setdefault(method, 0)
out[column][method] += 1
return out
def main() -> int:
ap = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--brand", action="append", dest="brands")
ap.add_argument("--column", action="append", dest="columns")
ap.add_argument("--json", action="store_true")
ap.add_argument("--by-provenance", action="store_true",
help="also break filled values down by how they were obtained")
args = ap.parse_args()
wanted = args.columns or TRACKED
conn = _connect()
if conn is None:
logger.error("Database unreachable.")
return 2
totals: Dict[str, Dict[str, int]] = {c: {"rows": 0, "filled": 0, "not_applicable": 0}
for c in wanted}
per_brand: Dict[str, Any] = {}
prov_totals: Dict[str, Dict[str, int]] = {}
try:
with conn.cursor() as cur:
tables = brand_tables(cur, args.brands)
for table in tables:
stats = scan(cur, table, wanted)
if not stats:
continue
per_brand[table] = stats
for col, s in stats.items():
for k in ("rows", "filled", "not_applicable"):
totals[col][k] += s[k]
if args.by_provenance:
for col, methods in provenance(cur, table).items():
for method, n in methods.items():
prov_totals.setdefault(col, {}).setdefault(method, 0)
prov_totals[col][method] += n
finally:
conn.close()
def applicable(s):
return s["rows"] - s["not_applicable"]
if args.json:
print(json.dumps({
"database": f"{DB_HOST}/{DB_NAME}",
"totals": {c: {**s, "applicable": applicable(s),
"pct": round(100 * s["filled"] / applicable(s), 1)
if applicable(s) else None}
for c, s in totals.items() if s["rows"]},
"provenance": prov_totals,
"per_brand": per_brand,
}, indent=2))
return 0
row_count = max((s["rows"] for s in totals.values()), default=0)
logger.info("database : %s / %s", DB_HOST, DB_NAME)
logger.info("%d product rows across %d brand tables", row_count, len(per_brand))
logger.info("")
logger.info("%-22s %8s %8s %7s %s", "column", "filled", "of", "pct", "not applicable")
logger.info("%s", "-" * 72)
for col in wanted:
s = totals.get(col)
if not s or not s["rows"]:
continue
app_n = applicable(s)
pct = f'{100 * s["filled"] / app_n:5.1f}%' if app_n else " -"
na = f'{s["not_applicable"]:d}' if s["not_applicable"] else ""
flag = ""
if app_n and s["filled"] < app_n:
flag = f' <- {app_n - s["filled"]} missing'
logger.info("%-22s %8d %8d %7s %-6s%s", col, s["filled"], app_n, pct, na, flag)
if args.by_provenance and prov_totals:
logger.info("")
logger.info("provenance of filled values (from field_sources)")
logger.info("%s", "-" * 72)
for col in sorted(prov_totals):
methods = ", ".join(f"{m} {n}" for m, n in
sorted(prov_totals[col].items(), key=lambda kv: -kv[1]))
logger.info("%-22s %s", col, methods)
elif args.by_provenance:
logger.info("")
logger.info("No field_sources recorded yet - run an enrichment pass first.")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,257 @@
#!/usr/bin/env python3
"""
Bring every brand table up to the current schema.
WHY
`vector_store._ensure_columns` is the only migration mechanism in this
repo - there is no Alembic and no migrations directory - and it runs only
as a side effect of `ensure_brand_schema`, i.e. on the next WRITE to a
table. A brand nobody uploads to therefore never gets a new column.
Measured on 2026-09-08, before the enrichment columns landed: seven of
fifty-six brand tables carried `gtin` / `ean13` / `upc` /
`barcode_source` / `barcode_verified` / `barcode_lookup_status` /
`barcode_last_updated`, all added out-of-band by hand. The other
forty-nine did not, and no code path would ever have added them. This
script closes that gap deliberately instead of waiting for a write that
may never come.
WHAT IT TOUCHES
* `ALTER TABLE brand_<slug> ADD COLUMN IF NOT EXISTS ...` for every column
in `_ensure_columns`' `col_defs` that the table does not already have.
* The retroactive UNIQUE index on `image_id`, and the DROP NOT NULL sweep
over legacy columns - both are part of `_ensure_columns` and cannot be
run separately.
Nothing else. No row is read, updated or deleted by this script.
WHAT IT WILL NOT DO
* It will not change the type of a column that already exists.
`ADD COLUMN IF NOT EXISTS` skips a column that is present, whatever its
type. This is deliberate: the seven hand-migrated tables define the
types the rest must match, which is why `col_defs` says TIMESTAMP for
`barcode_last_updated` and REAL for the tax figures rather than the
types those values look like they want. Type drift is REPORTED here,
never silently "fixed".
* It will not create a brand table that does not exist.
* It will not touch `nutrition_facts` or any non-brand table.
WHY IT IS SAFE ON A LIVE DATABASE
`ADD COLUMN` with no DEFAULT and no NOT NULL is a catalogue-only change in
PostgreSQL 11+: no table rewrite, no full-table lock, no time proportional
to row count. On 1 630 rows across 56 tables this is milliseconds. The one
exception is `field_sources`, which does carry a DEFAULT - and since
PostgreSQL 11 a non-volatile default is also metadata-only.
USAGE
python -m scripts.migrate_brand_schema # dry run, all brands
python -m scripts.migrate_brand_schema --brand amul # repeatable
python -m scripts.migrate_brand_schema --apply
python -m scripts.migrate_brand_schema --json
`--dry-run` is the default and `--apply` must be explicit: backend/.env points
at the PRODUCTION database, so an accidental run must not be able to write.
The target host is printed on startup.
"""
from __future__ import annotations
import argparse
import json
import logging
import sys
from pathlib import Path
from typing import Any, Dict, List, Optional
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.infrastructure.settings import DB_HOST, DB_NAME
from app.services.vector_store import _connect, _ensure_columns
logging.basicConfig(level=logging.INFO, format="%(message)s")
logger = logging.getLogger("migrate_brand_schema")
def brand_tables(cur, only: Optional[List[str]] = None) -> List[str]:
"""Every brand table actually present, in name order."""
cur.execute(
"SELECT table_name FROM information_schema.tables "
"WHERE table_schema = 'public' AND table_name LIKE 'brand\\_%' "
"ORDER BY table_name"
)
tables = [r[0] for r in cur.fetchall()]
if only:
wanted = {f"brand_{s.strip().lower().replace(' ', '_').replace('-', '_')}"
for s in only}
tables = [t for t in tables if t in wanted]
return tables
def existing_columns(cur, table: str) -> Dict[str, str]:
cur.execute(
"SELECT column_name, data_type FROM information_schema.columns "
"WHERE table_schema = 'public' AND table_name = %s",
(table,),
)
return {r[0]: r[1] for r in cur.fetchall()}
# What `information_schema.data_type` reports for each col_defs type, so a
# type-drift check does not raise false alarms on spelling differences.
_TYPE_ALIASES = {
"TEXT": {"text"},
"TEXT[]": {"ARRAY"},
"NUMERIC": {"numeric"},
"REAL": {"real"},
"BOOLEAN": {"boolean"},
"TIMESTAMP": {"timestamp without time zone"},
"JSONB": {"jsonb"},
"DOUBLE PRECISION": {"double precision"},
"vector(384)": {"USER-DEFINED"},
}
class RecordingCursor:
"""Wraps a real cursor so a dry run can see the statements without
executing them. Reads are passed through - the whole point is to compute
the diff against what is really on the table."""
def __init__(self, inner):
self._inner = inner
self.statements: List[str] = []
def execute(self, sql, params=None):
text = " ".join(str(sql).split())
upper = text.upper()
if upper.startswith("SELECT"):
return self._inner.execute(sql, params)
self.statements.append(text)
return None
def fetchall(self):
return self._inner.fetchall()
def fetchone(self):
return self._inner.fetchone()
def main() -> int:
ap = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--brand", action="append", dest="brands",
help="brand table suffix; repeatable. Default: every brand table.")
ap.add_argument("--apply", action="store_true", help="actually run the ALTERs")
ap.add_argument("--dry-run", action="store_true", default=False,
help="report only (the default)")
ap.add_argument("--json", action="store_true", help="machine-readable output")
args = ap.parse_args()
apply = args.apply and not args.dry_run
conn = _connect()
if conn is None:
logger.error("Database unreachable - nothing to do.")
return 2
logger.info("database : %s / %s", DB_HOST, DB_NAME)
logger.info("mode : %s", "APPLY (writing)" if apply else "dry run (no writes)")
logger.info("")
declared_types = _declared_types()
if not declared_types:
logger.error("Could not read col_defs out of _ensure_columns - refusing to "
"guess at the schema. Has that function been restructured?")
return 2
report: List[Dict[str, Any]] = []
total_missing = 0
total_drift = 0
try:
with conn.cursor() as cur:
tables = brand_tables(cur, args.brands)
if not tables:
logger.error("No brand tables matched.")
return 1
for table in tables:
before = existing_columns(cur, table)
recorder = RecordingCursor(cur)
_ensure_columns(recorder, table)
adds = [s for s in recorder.statements if "ADD COLUMN" in s]
# Type drift: a column that exists but whose type is not what
# col_defs would have created. Reported, never altered.
drift = []
for col, declared in declared_types.items():
actual = before.get(col)
if actual is None:
continue
allowed = _TYPE_ALIASES.get(declared.upper(), set())
if allowed and actual not in allowed:
drift.append({"column": col, "declared": declared, "actual": actual})
entry = {
"table": table,
"missing_columns": [s.split("ADD COLUMN IF NOT EXISTS ")[1] for s in adds],
"type_drift": drift,
}
report.append(entry)
total_missing += len(adds)
total_drift += len(drift)
if apply and adds:
_ensure_columns(cur, table)
if apply:
conn.commit()
finally:
conn.close()
if args.json:
print(json.dumps({"applied": apply, "tables": report}, indent=2))
return 0
width = max(len(e["table"]) for e in report)
changed = [e for e in report if e["missing_columns"] or e["type_drift"]]
for entry in sorted(changed, key=lambda e: -len(e["missing_columns"])):
logger.info("%-*s %d column(s) missing", width, entry["table"],
len(entry["missing_columns"]))
for col in entry["missing_columns"]:
logger.info("%-*s + %s", width, "", col)
for d in entry["type_drift"]:
logger.info("%-*s ! %s is %s, col_defs declares %s (NOT changed)",
width, "", d["column"], d["actual"], d["declared"])
logger.info("")
logger.info("%d table(s) scanned, %d already current",
len(report), len(report) - len(changed))
logger.info("%d column(s) %s, %d type mismatch(es) reported",
total_missing, "added" if apply else "would be added", total_drift)
if not apply and total_missing:
logger.info("")
logger.info("Re-run with --apply to write these changes.")
return 0
def _declared_types() -> Dict[str, str]:
"""The col_defs dict, read back out of the function that owns it.
Parsed from source rather than duplicated here, so this script cannot
drift from the single migration mechanism it exists to drive.
"""
import ast
import inspect
import textwrap
tree = ast.parse(textwrap.dedent(inspect.getsource(_ensure_columns)))
for node in ast.walk(tree):
if isinstance(node, ast.Assign) and getattr(node.targets[0], "id", "") == "col_defs":
return {ast.literal_eval(k): ast.literal_eval(v)
for k, v in zip(node.value.keys, node.value.values)}
return {}
if __name__ == "__main__":
raise SystemExit(main())