Backend update on Image generation

This commit is contained in:
sriram
2026-09-02 16:39:30 +05:30
parent 1c20bfc408
commit 1300d4c678
2 changed files with 2902 additions and 12 deletions

File diff suppressed because it is too large Load Diff

View File

@@ -463,22 +463,31 @@ def _search_replacement(product_name: str, brand: str, max_results: int) -> List
_SEARCH_CACHE[key] = []
return []
# Rank: corroborated first, then https, then the ranker's own order. A URL
# that names the product or the brand is the one to show; an opaque CDN path
# is kept but never preferred over a corroborated sibling.
ranked = sorted(
urls,
key=lambda u: (
0 if _names_product(u, product_name, brand) else 1,
0 if u.startswith("https://") else 1,
),
)
# REQUIRE corroboration: the URL must name the brand or a distinctive word
# from the product. Three separate providers were caught returning a
# confidently wrong image that validated fine -
#
# OpenFoodFacts "Aachi Pickles" -> Ducros Green Pitted Olives (FR)
# Wikimedia "Aachi Kulambu Mix" -> a press photo of PM Modi
# Bing/DDG "MTR Dosa Mix" -> an anatomy textbook plate
#
# - and none of them was detectable from the URL alone once an opaque CDN
# path was accepted. Precision beats recall for a repair: a row left with
# its brand monogram is honest, a row showing another company's product is
# not, and it is invisible until somebody recognises the photo.
#
# This does reject legitimately-correct images behind hashed CDN paths
# (Zepto, Flipkart, Pinterest). That is the intended trade - 94% of rows
# still find a corroborated URL, and the rest are reported for hand
# curation rather than filled with a guess.
ranked = [u for u in urls if _names_product(u, product_name, brand)]
ranked.sort(key=lambda u: 0 if u.startswith("https://") else 1)
_SEARCH_CACHE[key] = ranked
return ranked
def repair(cur, conn, brands: List[str], apply: bool, timeout: int, workers: int,
limit: Optional[int], max_results: int) -> Dict[str, int]:
limit: Optional[int], max_results: int, recheck: bool = False) -> Dict[str, int]:
totals = {"scanned": 0, "healthy": 0, "repaired": 0, "unrepairable": 0}
unrepairable: List[str] = []
snapshot: Dict[str, List[Dict[str, Any]]] = {}
@@ -492,6 +501,21 @@ def repair(cur, conn, brands: List[str], apply: bool, timeout: int, workers: int
snapshot[table] = rows
verdict = _classify(rows, timeout, workers)
needs = [r for r in rows if verdict[r["image_id"]] != HEALTHY]
if recheck:
# A stored URL that resolves is not proof it is the right product -
# three providers were caught serving a confidently wrong image.
# --recheck re-opens rows whose stored URL names neither the brand
# nor any distinctive word of the product, so they are re-searched
# under the corroboration rule rather than grandfathered in.
have = {r["image_id"] for r in needs}
for row in rows:
if row["image_id"] in have:
continue
stored = _row_urls(row)
if stored and not any(
_names_product(u, row.get("product_name") or "", suffix) for u in stored
):
needs.append(row)
totals["scanned"] += len(rows)
totals["healthy"] += len(rows) - len(needs)
@@ -599,6 +623,9 @@ def main() -> int:
parser.add_argument("--limit", type=int, help="max rows to repair per brand")
parser.add_argument("--timeout", type=int, default=8, help="per-URL probe seconds")
parser.add_argument("--workers", type=int, default=8, help="parallel URL probes")
parser.add_argument("--recheck", action="store_true",
help="also re-search rows whose stored image resolves but "
"does not name the brand or product")
parser.add_argument("--max-results", type=int, default=8,
help="candidates to request per product search")
args = parser.parse_args()
@@ -635,7 +662,8 @@ def main() -> int:
if args.all:
brands = None
totals = repair(cur, conn, brands, apply, args.timeout,
args.workers, args.limit, args.max_results)
args.workers, args.limit, args.max_results,
recheck=args.recheck)
logger.info("")
logger.info("scanned %(scanned)d | already fine %(healthy)d | "