imag vector generation with dimentionality reduction
This commit is contained in:
184
scripts/export_mobilenet_embedder.py
Normal file
184
scripts/export_mobilenet_embedder.py
Normal file
@@ -0,0 +1,184 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Build `mobilenet_v3_small_embedder.tflite` - the image model behind img_vector.
|
||||
|
||||
WHY THIS EXISTS
|
||||
---------------
|
||||
The Nearle Flutter app embeds product photos with a MobileNetV3-Small TFLite
|
||||
model whose output is the 1024-wide penultimate layer (input [1,224,224,3]
|
||||
float32 in 0..1, output [1,1024], L2-normalised afterwards). That file is a
|
||||
hand-made export, not a package: nothing installs it. When the app's own copy
|
||||
is not to hand, this script produces an equivalent one from Keras' pretrained
|
||||
ImageNet weights, so the catalog can be vectorised at all.
|
||||
|
||||
WHAT IT BUILDS
|
||||
--------------
|
||||
Input(224, 224, 3) float32, 0..1 - the app divides by 255 and
|
||||
-> Rescaling(scale=2, offset=-1) nothing else; MobileNetV3 wants -1..1, so
|
||||
-> MobileNetV3Small backbone the rescale lives INSIDE the graph
|
||||
(imagenet weights, include_top=True,
|
||||
include_preprocessing=False)
|
||||
-> "Conv_2" 1x1 conv (576 -> 1024) + hard-swish <- the embedding
|
||||
-> Flatten [1, 1024]
|
||||
|
||||
`Conv_2` is the only 1024-wide layer in MobileNetV3-Small, so it is the layer
|
||||
any "1024-d MobileNetV3-Small embedder" taps. Dropout and the 1000-way Logits
|
||||
layer after it are discarded. No normalisation in the graph - the app
|
||||
normalises after inference, and so does app/services/image_embedder.py.
|
||||
|
||||
IS IT THE SAME AS THE APP'S FILE?
|
||||
---------------------------------
|
||||
Same architecture, same pretrained weights, same tap, same input convention.
|
||||
Whether the numbers agree to the last decimal depends on how the app's copy
|
||||
was exported (TF version, converter flags), and that can only be checked
|
||||
against the app itself: take one cropped photo from the app, note the first
|
||||
eight values of its `[VECTOR][IMAGE]` log line, run
|
||||
`image_embedder.embedding_for_bytes()` on the same file, and compare. Agreement
|
||||
to 3-4 decimals means the two systems' vectors are interchangeable. If they
|
||||
differ, catalog<->catalog search still works (the vectors are self-consistent);
|
||||
drop the app's real file over this one and run
|
||||
`python -m scripts.backfill_image_vectors --all --force --apply`.
|
||||
|
||||
HOW TO RUN IT
|
||||
-------------
|
||||
TensorFlow is NOT a dependency of this project and must not become one (the
|
||||
container is memory-capped and already carries torch). Run the export in a
|
||||
throwaway container, from backend/:
|
||||
|
||||
docker run --rm -v "${PWD}:/work" -w /work python:3.11-slim sh -c \\
|
||||
"pip install -q tensorflow-cpu && python scripts/export_mobilenet_embedder.py \\
|
||||
--out app/services/models/mobilenet/mobilenet_v3_small_embedder.tflite"
|
||||
|
||||
Or, without Docker, in a throwaway venv at a SHORT path (TensorFlow's wheel
|
||||
trips Windows' path-length limit inside a deep temp directory; TF has no
|
||||
wheel for Python 3.14, so use 3.13 or 3.11):
|
||||
|
||||
py -3.13 -m venv C:/tfexport_venv
|
||||
C:/tfexport_venv/Scripts/pip install "tensorflow==2.21.*" "ai-edge-litert>=2.2.0"
|
||||
C:/tfexport_venv/Scripts/python scripts/export_mobilenet_embedder.py --out ...
|
||||
|
||||
The committed file was produced this way on 2026-09-17 with TensorFlow 2.21.0
|
||||
/ Keras 3.15.1. The pretrained weights are fixed, so re-running produces the
|
||||
same model. The script self-checks the result with the same runtime the
|
||||
backend uses (ai-edge-litert) when it is importable, else with tf.lite.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
INPUT_SIZE = 224
|
||||
EMBED_DIM = 1024
|
||||
TAP_LAYER = "Conv_2" # Keras' name for the 1x1 conv that widens 576 -> 1024
|
||||
|
||||
|
||||
def build_embedder():
|
||||
"""The Keras model described in the module docstring."""
|
||||
import tensorflow as tf
|
||||
from tensorflow import keras
|
||||
|
||||
base = keras.applications.MobileNetV3Small(
|
||||
input_shape=(INPUT_SIZE, INPUT_SIZE, 3),
|
||||
weights="imagenet",
|
||||
include_top=True, # we need the head's Conv_2, which include_top=False drops
|
||||
include_preprocessing=False, # the 0..1 -> -1..1 rescale is added explicitly below
|
||||
)
|
||||
# Keras 2 named it "Conv_2", Keras 3 "conv_2"; match case-insensitively.
|
||||
names = [layer.name for layer in base.layers]
|
||||
try:
|
||||
idx = [n.lower() for n in names].index(TAP_LAYER.lower())
|
||||
except ValueError:
|
||||
raise SystemExit(f"no layer named {TAP_LAYER} in MobileNetV3Small; layers: {names}")
|
||||
conv = base.layers[idx]
|
||||
# The layer right after it is the hard-swish activation whose output is
|
||||
# the embedding (then come dropout and the 1000-way logits, discarded).
|
||||
act = base.layers[idx + 1]
|
||||
if int(conv.output.shape[-1]) != EMBED_DIM or int(act.output.shape[-1]) != EMBED_DIM:
|
||||
raise SystemExit(f"{conv.name}/{act.name} are not {EMBED_DIM} wide: "
|
||||
f"{conv.output.shape[-1]}/{act.output.shape[-1]}")
|
||||
if "activation" not in act.name.lower() and "swish" not in act.name.lower():
|
||||
raise SystemExit(f"layer after {conv.name} is {act.name}, expected the hard-swish activation")
|
||||
|
||||
inputs = keras.Input(shape=(INPUT_SIZE, INPUT_SIZE, 3), dtype="float32", name="image_0_1")
|
||||
x = keras.layers.Rescaling(scale=2.0, offset=-1.0, name="rescale_0_1_to_pm1")(inputs)
|
||||
features = keras.Model(base.input, act.output, name="mobilenet_v3_small_features")(x)
|
||||
outputs = keras.layers.Flatten(name="embedding")(features)
|
||||
model = keras.Model(inputs, outputs, name="mobilenet_v3_small_embedder")
|
||||
if tuple(model.output.shape) != (None, EMBED_DIM):
|
||||
raise SystemExit(f"unexpected output shape {model.output.shape}")
|
||||
return model
|
||||
|
||||
|
||||
def convert_to_tflite(model) -> bytes:
|
||||
import tensorflow as tf
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
saved = Path(tmp) / "saved_model"
|
||||
# Keras 3 exports a SavedModel via .export(); Keras 2 via tf.saved_model.save.
|
||||
if hasattr(model, "export"):
|
||||
model.export(str(saved))
|
||||
else:
|
||||
tf.saved_model.save(model, str(saved))
|
||||
converter = tf.lite.TFLiteConverter.from_saved_model(str(saved))
|
||||
# Float32, builtin ops only, no quantisation: the backend runtime is
|
||||
# plain ai-edge-litert and must not need SELECT_TF_OPS.
|
||||
converter.target_spec.supported_ops = [tf.lite.OpsSet.TFLITE_BUILTINS]
|
||||
return converter.convert()
|
||||
|
||||
|
||||
def self_check(path: Path) -> None:
|
||||
import numpy as np
|
||||
|
||||
try:
|
||||
from ai_edge_litert.interpreter import Interpreter
|
||||
runtime = "ai-edge-litert"
|
||||
except ImportError:
|
||||
import tensorflow as tf
|
||||
Interpreter = tf.lite.Interpreter
|
||||
runtime = "tf.lite"
|
||||
|
||||
it = Interpreter(model_path=str(path))
|
||||
it.allocate_tensors()
|
||||
inp, out = it.get_input_details(), it.get_output_details()
|
||||
assert len(inp) == 1 and list(inp[0]["shape"]) == [1, INPUT_SIZE, INPUT_SIZE, 3], inp
|
||||
assert np.dtype(inp[0]["dtype"]) == np.float32, inp[0]["dtype"]
|
||||
assert len(out) == 1 and list(out[0]["shape"]) == [1, EMBED_DIM], out
|
||||
|
||||
rng = np.random.default_rng(0)
|
||||
for label, x in (
|
||||
("zeros", np.zeros((1, INPUT_SIZE, INPUT_SIZE, 3), np.float32)),
|
||||
("random", rng.random((1, INPUT_SIZE, INPUT_SIZE, 3), dtype=np.float32)),
|
||||
):
|
||||
it.set_tensor(inp[0]["index"], x)
|
||||
it.invoke()
|
||||
v = it.get_tensor(out[0]["index"])[0]
|
||||
norm = float(np.linalg.norm(v))
|
||||
assert np.all(np.isfinite(v)) and norm > 0, label
|
||||
print(f" {label:6s} first 8: {np.round(v[:8], 5).tolist()} norm={norm:.4f} "
|
||||
f"zeros={int((v == 0).sum())}/{EMBED_DIM}")
|
||||
print(f" self-check passed with {runtime}: input [1,{INPUT_SIZE},{INPUT_SIZE},3] float32, output [1,{EMBED_DIM}]")
|
||||
|
||||
|
||||
def main() -> int:
|
||||
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument("--out", required=True, help="where to write the .tflite")
|
||||
args = ap.parse_args()
|
||||
out = Path(args.out)
|
||||
out.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
import tensorflow as tf
|
||||
print(f"tensorflow {tf.__version__}, keras {tf.keras.__version__ if hasattr(tf.keras, '__version__') else '?'}")
|
||||
|
||||
model = build_embedder()
|
||||
print(f"model: {model.name}, params={model.count_params():,}, tap={TAP_LAYER}+hard-swish")
|
||||
data = convert_to_tflite(model)
|
||||
out.write_bytes(data)
|
||||
print(f"wrote {out} ({len(data) / 1e6:.1f} MB)")
|
||||
self_check(out)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user