Backend update on Image generation
This commit is contained in:
2862
data/brand_image_repair_backup_20260902_161336.json
Normal file
2862
data/brand_image_repair_backup_20260902_161336.json
Normal file
File diff suppressed because it is too large
Load Diff
@@ -463,22 +463,31 @@ def _search_replacement(product_name: str, brand: str, max_results: int) -> List
|
||||
_SEARCH_CACHE[key] = []
|
||||
return []
|
||||
|
||||
# Rank: corroborated first, then https, then the ranker's own order. A URL
|
||||
# that names the product or the brand is the one to show; an opaque CDN path
|
||||
# is kept but never preferred over a corroborated sibling.
|
||||
ranked = sorted(
|
||||
urls,
|
||||
key=lambda u: (
|
||||
0 if _names_product(u, product_name, brand) else 1,
|
||||
0 if u.startswith("https://") else 1,
|
||||
),
|
||||
)
|
||||
# REQUIRE corroboration: the URL must name the brand or a distinctive word
|
||||
# from the product. Three separate providers were caught returning a
|
||||
# confidently wrong image that validated fine -
|
||||
#
|
||||
# OpenFoodFacts "Aachi Pickles" -> Ducros Green Pitted Olives (FR)
|
||||
# Wikimedia "Aachi Kulambu Mix" -> a press photo of PM Modi
|
||||
# Bing/DDG "MTR Dosa Mix" -> an anatomy textbook plate
|
||||
#
|
||||
# - and none of them was detectable from the URL alone once an opaque CDN
|
||||
# path was accepted. Precision beats recall for a repair: a row left with
|
||||
# its brand monogram is honest, a row showing another company's product is
|
||||
# not, and it is invisible until somebody recognises the photo.
|
||||
#
|
||||
# This does reject legitimately-correct images behind hashed CDN paths
|
||||
# (Zepto, Flipkart, Pinterest). That is the intended trade - 94% of rows
|
||||
# still find a corroborated URL, and the rest are reported for hand
|
||||
# curation rather than filled with a guess.
|
||||
ranked = [u for u in urls if _names_product(u, product_name, brand)]
|
||||
ranked.sort(key=lambda u: 0 if u.startswith("https://") else 1)
|
||||
_SEARCH_CACHE[key] = ranked
|
||||
return ranked
|
||||
|
||||
|
||||
def repair(cur, conn, brands: List[str], apply: bool, timeout: int, workers: int,
|
||||
limit: Optional[int], max_results: int) -> Dict[str, int]:
|
||||
limit: Optional[int], max_results: int, recheck: bool = False) -> Dict[str, int]:
|
||||
totals = {"scanned": 0, "healthy": 0, "repaired": 0, "unrepairable": 0}
|
||||
unrepairable: List[str] = []
|
||||
snapshot: Dict[str, List[Dict[str, Any]]] = {}
|
||||
@@ -492,6 +501,21 @@ def repair(cur, conn, brands: List[str], apply: bool, timeout: int, workers: int
|
||||
snapshot[table] = rows
|
||||
verdict = _classify(rows, timeout, workers)
|
||||
needs = [r for r in rows if verdict[r["image_id"]] != HEALTHY]
|
||||
if recheck:
|
||||
# A stored URL that resolves is not proof it is the right product -
|
||||
# three providers were caught serving a confidently wrong image.
|
||||
# --recheck re-opens rows whose stored URL names neither the brand
|
||||
# nor any distinctive word of the product, so they are re-searched
|
||||
# under the corroboration rule rather than grandfathered in.
|
||||
have = {r["image_id"] for r in needs}
|
||||
for row in rows:
|
||||
if row["image_id"] in have:
|
||||
continue
|
||||
stored = _row_urls(row)
|
||||
if stored and not any(
|
||||
_names_product(u, row.get("product_name") or "", suffix) for u in stored
|
||||
):
|
||||
needs.append(row)
|
||||
totals["scanned"] += len(rows)
|
||||
totals["healthy"] += len(rows) - len(needs)
|
||||
|
||||
@@ -599,6 +623,9 @@ def main() -> int:
|
||||
parser.add_argument("--limit", type=int, help="max rows to repair per brand")
|
||||
parser.add_argument("--timeout", type=int, default=8, help="per-URL probe seconds")
|
||||
parser.add_argument("--workers", type=int, default=8, help="parallel URL probes")
|
||||
parser.add_argument("--recheck", action="store_true",
|
||||
help="also re-search rows whose stored image resolves but "
|
||||
"does not name the brand or product")
|
||||
parser.add_argument("--max-results", type=int, default=8,
|
||||
help="candidates to request per product search")
|
||||
args = parser.parse_args()
|
||||
@@ -635,7 +662,8 @@ def main() -> int:
|
||||
if args.all:
|
||||
brands = None
|
||||
totals = repair(cur, conn, brands, apply, args.timeout,
|
||||
args.workers, args.limit, args.max_results)
|
||||
args.workers, args.limit, args.max_results,
|
||||
recheck=args.recheck)
|
||||
|
||||
logger.info("")
|
||||
logger.info("scanned %(scanned)d | already fine %(healthy)d | "
|
||||
|
||||
Reference in New Issue
Block a user