Files
catalogue_backend/scripts/catalog_coverage.py
2026-09-08 15:18:29 +05:30

280 lines
10 KiB
Python

#!/usr/bin/env python3
"""
How completely is the catalog filled, and where did each value come from?
WHY
"Fill every column" is not actually the goal, and a report that treats it
as one produces a permanent, unfixable red number that somebody eventually
"fixes" by inventing data. Three of these columns can never be filled for
large parts of the catalog, and that is correct:
* `upc` - every barcode here is GS1 India (prefix 890), which issues
EAN-13 and GTIN-8. UPC-A is a North American symbology. Measured: 0 of
300 barcodes are UPC-A, and none ever will be.
* `nutrients` / `health_score` - roughly 30% of rows are shampoo, soap,
detergent and toothpaste. Soap has no protein content.
* `fssai_license` - an FSSAI licence covers a FOOD business. P&G,
Colgate-Palmolive and Reckitt Benckiser should not carry one.
So this report counts four states, not two:
sourced a real value from a real source
derived computed from another field we hold (gtin from barcode)
estimated a category-level or consensus guess, flagged as such
not applicable cannot exist for this row, and should not
MISSING we have not got it yet - the only number worth chasing
Coverage percentages are taken against the APPLICABLE denominator, so the
numbers describe work remaining rather than work impossible.
WHAT IT TOUCHES
Nothing. Every statement is a SELECT. There is no --apply because there is
nothing to apply.
USAGE
python -m scripts.catalog_coverage
python -m scripts.catalog_coverage --brand amul --brand cadbury
python -m scripts.catalog_coverage --column barcode --column nutrients
python -m scripts.catalog_coverage --json > coverage.json
python -m scripts.catalog_coverage --by-provenance
Run it before and after any enrichment change: the diff of two --json runs is
the evidence that the change did what it claimed.
"""
from __future__ import annotations
import argparse
import json
import logging
import sys
from pathlib import Path
from typing import Any, Dict, List, Optional, Set
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.infrastructure.settings import DB_HOST, DB_NAME
from app.services.vector_store import _connect
logging.basicConfig(level=logging.INFO, format="%(message)s")
logger = logging.getLogger("catalog_coverage")
# Columns worth reporting on, in the order a reader wants them.
TRACKED = [
"product_name", "title", "description", "category", "image_url",
"price_range", "size_variants", "providers", "highlights",
"fssai_license", "product_sku", "hsn_code", "gst_percent", "tax_amount",
"selling_price", "final_selling_price",
"barcode", "barcode_type", "gtin", "ean13", "upc",
"nutrients", "nutrients_per_100g", "nutrition_score", "health_score",
]
# Columns that simply cannot apply to some rows, and the rule for which.
#
# food_only - meaningless for a non-consumable product
# india_only - UPC-A does not occur in a GS1 India catalog
# barcoded - derived from a barcode, so absent when the barcode is
NOT_APPLICABLE_RULES = {
"nutrients": "food_only",
"nutrients_per_100g": "food_only",
"nutrition_score": "food_only",
"health_score": "food_only",
"fssai_license": "food_only",
"upc": "india_only",
"gtin": "barcoded",
"ean13": "barcoded",
"barcode_type": "barcoded",
}
# A value that is present but means "nothing here".
EMPTY_LITERALS = ("", "[]", "{}", "null", "0", "Uncategorized")
def brand_tables(cur, only: Optional[List[str]] = None) -> List[str]:
cur.execute(
"SELECT table_name FROM information_schema.tables "
"WHERE table_schema = 'public' AND table_name LIKE 'brand\\_%' "
"AND table_name <> 'brand_zzsmoketest' ORDER BY table_name"
)
tables = [r[0] for r in cur.fetchall()]
if only:
wanted = {f"brand_{s.strip().lower().replace(' ', '_').replace('-', '_')}"
for s in only}
tables = [t for t in tables if t in wanted]
return tables
def columns_of(cur, table: str) -> Set[str]:
"""Probed per table rather than assumed. The column set genuinely differed
per table until the schema migration, and a script that assumes otherwise
dies on the first old table it meets."""
cur.execute(
"SELECT column_name FROM information_schema.columns "
"WHERE table_schema = 'public' AND table_name = %s",
(table,),
)
return {r[0] for r in cur.fetchall()}
def _is_non_food(category: str) -> bool:
from app.services.consumability import is_non_consumable
try:
return bool(is_non_consumable(category or "", ""))
except Exception:
return False
def scan(cur, table: str, wanted: List[str]) -> Dict[str, Dict[str, int]]:
present = columns_of(cur, table)
cols = [c for c in wanted if c in present]
if not cols:
return {}
select = ", ".join(f'"{c}"' for c in cols)
extra = ', "category"' if "category" in present else ""
barcode_idx = cols.index("barcode") if "barcode" in cols else None
cur.execute(f'SELECT {select}{extra} FROM "{table}"')
rows = cur.fetchall()
stats: Dict[str, Dict[str, int]] = {
c: {"rows": 0, "filled": 0, "not_applicable": 0} for c in cols
}
for row in rows:
category = row[len(cols)] if extra else ""
non_food = _is_non_food(category)
has_barcode = bool(barcode_idx is not None and row[barcode_idx])
for i, col in enumerate(cols):
s = stats[col]
s["rows"] += 1
rule = NOT_APPLICABLE_RULES.get(col)
if ((rule == "food_only" and non_food)
or (rule == "india_only")
or (rule == "barcoded" and not has_barcode)):
s["not_applicable"] += 1
continue
value = row[i]
if value is None:
continue
if isinstance(value, (list, tuple, dict)) and not value:
continue
if isinstance(value, str) and value.strip() in EMPTY_LITERALS:
continue
s["filled"] += 1
return stats
def provenance(cur, table: str) -> Dict[str, Dict[str, int]]:
"""How each filled value was arrived at, read from `field_sources`."""
if "field_sources" not in columns_of(cur, table):
return {}
cur.execute(f'SELECT field_sources FROM "{table}" '
f"WHERE field_sources IS NOT NULL AND field_sources <> '{{}}'::jsonb")
out: Dict[str, Dict[str, int]] = {}
for (blob,) in cur.fetchall():
for column, record in (blob or {}).items():
method = (record or {}).get("method", "unspecified")
out.setdefault(column, {}).setdefault(method, 0)
out[column][method] += 1
return out
def main() -> int:
ap = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--brand", action="append", dest="brands")
ap.add_argument("--column", action="append", dest="columns")
ap.add_argument("--json", action="store_true")
ap.add_argument("--by-provenance", action="store_true",
help="also break filled values down by how they were obtained")
args = ap.parse_args()
wanted = args.columns or TRACKED
conn = _connect()
if conn is None:
logger.error("Database unreachable.")
return 2
totals: Dict[str, Dict[str, int]] = {c: {"rows": 0, "filled": 0, "not_applicable": 0}
for c in wanted}
per_brand: Dict[str, Any] = {}
prov_totals: Dict[str, Dict[str, int]] = {}
try:
with conn.cursor() as cur:
tables = brand_tables(cur, args.brands)
for table in tables:
stats = scan(cur, table, wanted)
if not stats:
continue
per_brand[table] = stats
for col, s in stats.items():
for k in ("rows", "filled", "not_applicable"):
totals[col][k] += s[k]
if args.by_provenance:
for col, methods in provenance(cur, table).items():
for method, n in methods.items():
prov_totals.setdefault(col, {}).setdefault(method, 0)
prov_totals[col][method] += n
finally:
conn.close()
def applicable(s):
return s["rows"] - s["not_applicable"]
if args.json:
print(json.dumps({
"database": f"{DB_HOST}/{DB_NAME}",
"totals": {c: {**s, "applicable": applicable(s),
"pct": round(100 * s["filled"] / applicable(s), 1)
if applicable(s) else None}
for c, s in totals.items() if s["rows"]},
"provenance": prov_totals,
"per_brand": per_brand,
}, indent=2))
return 0
row_count = max((s["rows"] for s in totals.values()), default=0)
logger.info("database : %s / %s", DB_HOST, DB_NAME)
logger.info("%d product rows across %d brand tables", row_count, len(per_brand))
logger.info("")
logger.info("%-22s %8s %8s %7s %s", "column", "filled", "of", "pct", "not applicable")
logger.info("%s", "-" * 72)
for col in wanted:
s = totals.get(col)
if not s or not s["rows"]:
continue
app_n = applicable(s)
pct = f'{100 * s["filled"] / app_n:5.1f}%' if app_n else " -"
na = f'{s["not_applicable"]:d}' if s["not_applicable"] else ""
flag = ""
if app_n and s["filled"] < app_n:
flag = f' <- {app_n - s["filled"]} missing'
logger.info("%-22s %8d %8d %7s %-6s%s", col, s["filled"], app_n, pct, na, flag)
if args.by_provenance and prov_totals:
logger.info("")
logger.info("provenance of filled values (from field_sources)")
logger.info("%s", "-" * 72)
for col in sorted(prov_totals):
methods = ", ".join(f"{m} {n}" for m, n in
sorted(prov_totals[col].items(), key=lambda kv: -kv[1]))
logger.info("%-22s %s", col, methods)
elif args.by_provenance:
logger.info("")
logger.info("No field_sources recorded yet - run an enrichment pass first.")
return 0
if __name__ == "__main__":
raise SystemExit(main())