From 726b5512e0c3f17a63641176a9dcf556289b1b82 Mon Sep 17 00:00:00 2001 From: Andrii Ryzhkov Date: Sun, 6 Sep 2026 09:40:37 +0200 Subject: [PATCH 1/2] Declare embedding_dim in the OpenCLIP RN101 manifest --- models/embedding-openclip-rn101-yfcc15m/model.yaml | 3 +++ 1 file changed, 3 insertions(+) diff --git a/models/embedding-openclip-rn101-yfcc15m/model.yaml b/models/embedding-openclip-rn101-yfcc15m/model.yaml index 4c1448f..4951ed1 100644 --- a/models/embedding-openclip-rn101-yfcc15m/model.yaml +++ b/models/embedding-openclip-rn101-yfcc15m/model.yaml @@ -17,6 +17,9 @@ attributes: norm_mean: [0.0, 0.0, 0.0] norm_std: [1.0, 1.0, 1.0] output_l2_normalized: true + # output width. darktable sizes its buffers and the vector index from + # this, and refuses a model that does not declare it rather than guessing + embedding_dim: 512 model_card: long_description: "OpenCLIP RN101. Powers tag suggestions and similar-image search. Trained on 15 million Flickr photos shared under Creative Commons licences – the only widely-available CLIP dataset built from author-consented images" From caba25af09a3ab3a0750ca08a259db01a4968446 Mon Sep 17 00:00:00 2001 From: Andrii Ryzhkov Date: Sun, 6 Sep 2026 09:49:26 +0200 Subject: [PATCH 2/2] Drop the default tag vocabulary from the embedding model --- darktable_ai/demo.py | 4 +- .../README.md | 22 +- .../convert.py | 516 +----------------- .../embedding-openclip-rn101-yfcc15m/demo.py | 88 +-- .../model.yaml | 3 +- .../embedding-openclip-rn101-yfcc15m/tags.md | 152 ------ samples/{embed => embedding}/example_01.jpg | Bin samples/{embed => embedding}/example_02.jpg | Bin samples/{embed => embedding}/example_03.jpg | Bin 9 files changed, 37 insertions(+), 748 deletions(-) delete mode 100644 models/embedding-openclip-rn101-yfcc15m/tags.md rename samples/{embed => embedding}/example_01.jpg (100%) rename samples/{embed => embedding}/example_02.jpg (100%) rename samples/{embed => embedding}/example_03.jpg (100%) diff --git a/darktable_ai/demo.py b/darktable_ai/demo.py index c31aa18..3a7f5c5 100644 --- a/darktable_ai/demo.py +++ b/darktable_ai/demo.py @@ -24,9 +24,11 @@ _SAMPLE_EXTS = _PROCESSED_IMAGE_EXTS | _RAW_IMAGE_EXTS # Task → output file extension. Raw-domain tasks can't round-trip through PNG -# because they produce linear HDR or >8-bit data. +# because they produce linear HDR or >8-bit data; embedding models produce no +# image at all. _OUTPUT_EXT_BY_TASK = { "rawdenoise": ".tif", + "embedding": ".json", } diff --git a/models/embedding-openclip-rn101-yfcc15m/README.md b/models/embedding-openclip-rn101-yfcc15m/README.md index 79cfab9..1043aba 100644 --- a/models/embedding-openclip-rn101-yfcc15m/README.md +++ b/models/embedding-openclip-rn101-yfcc15m/README.md @@ -29,19 +29,17 @@ rather than the everything-on-the-web mix that boosts LAION's ImageNet score. ## How it's used in darktable -Two cooperating workflows: +Two workflows, both image-to-image: - **Per-user tag taxonomy.** Average image embeddings per user-applied tag → centroid in 512-dim space. Suggest the same tag on new images whose embedding is close to that centroid. -- **Cold-start defaults.** For users without enough examples of their own, - fall back to 86 precomputed centroids covering common photographic - concepts (see [`tags.md`](tags.md)). Hierarchical names use darktable's - `|` separator: `genre|landscape`, `subject|animal|dog`, - `lighting|golden hour`, etc. +- **Image similarity.** The same embeddings answer "find more like this" + with no text involved. -The text encoder is run *only at convert time* to bake those centroids -into `tags.json`. At runtime, darktable runs only the image encoder. +The package ships the image encoder only. There is no default tag +vocabulary: a centroid built from a user's own photos describes what that +user means by a tag far better than a generic label ever did. ## Architecture @@ -57,7 +55,6 @@ into `tags.json`. At runtime, darktable runs only the image encoder. | File | Purpose | Size | |--------------|--------------------------------------------------------------------------|---------------| | `model.onnx` | image → 512-dim L2-normalised embedding (mean/std subtraction baked in) | ~60 MB (FP16) | -| `tags.json` | precomputed 512-dim centroids for the 86-tag default taxonomy | ~270 KB | The convert wrapper bakes mean/std subtraction and L2 normalisation into the ONNX graph so the caller only needs to feed `[0, 1]` RGB pixels. @@ -69,10 +66,9 @@ Callers do not need to add any cast logic. ## Notes -- **No text encoder ONNX in the runtime package.** The text encoder runs only - at darktable-ai build time to produce `tags.json`. Free-text search would - require shipping the text encoder separately – deferred until that feature - lands in darktable. +- **No text encoder in the package.** Only the image encoder is exported, and + none is run at convert time. Free-text search would require shipping the + text encoder separately – deferred until that feature lands in darktable. - **Multilingual upgrade path is clean.** A future multilingual text encoder distilled to match this YFCC15M-aligned space could be added as an optional sidecar package without invalidating users' indexed image embeddings. diff --git a/models/embedding-openclip-rn101-yfcc15m/convert.py b/models/embedding-openclip-rn101-yfcc15m/convert.py index 5664003..94ca23c 100644 --- a/models/embedding-openclip-rn101-yfcc15m/convert.py +++ b/models/embedding-openclip-rn101-yfcc15m/convert.py @@ -1,10 +1,7 @@ -"""Export OpenCLIP RN101 (YFCC15M) image encoder to ONNX and pre-compute tag embeddings. +"""Export the OpenCLIP RN101 (YFCC15M) image encoder to ONNX. Produces: model.onnx – image encoder with baked-in CLIP normalization + L2 norm - tags.json – pre-computed text embeddings for 86 hierarchical photo tags - -Tag vocabulary is defined in tags.md (human-readable reference). Why RN101 + YFCC15M specifically: YFCC15M is the only widely-available CLIP training dataset where every image was opt-in licensed by its author @@ -15,7 +12,6 @@ """ import argparse -import json import os import numpy as np @@ -26,474 +22,6 @@ import open_clip -# --------------------------------------------------------------------------- -# Tag vocabulary for zero-shot tag suggestion. -# Hierarchical names using "|" separator (darktable native). -# Each entry is (tag, CLIP prompt). See tags.md for the human-readable list. -# --------------------------------------------------------------------------- - -# Each tag carries a *list* of prompts. Encoding ensembles them: encode -# each, average, re-normalize. This is the standard CLIP zero-shot recipe -# (Radford et al. 2021) and substantially de-biases each centroid from -# the quirks of any one phrasing. -# -# Prompts try to combine: -# - one direct canonical form ("a photo of X") -# - one or two descriptive forms with visual cues (helps high-frequency -# classes like cat/dog stop matching every animal photo) -# - context cues for setting/lighting/technique tags - -TAG_VOCAB = [ - # --- genre (12) --- - ("genre|landscape", [ - "a landscape photograph", - "a scenic landscape photo", - "a wide outdoor landscape view", - "a nature landscape image", - ]), - ("genre|portrait", [ - "a portrait photograph", - "a portrait of a person", - "a close-up portrait photo", - "a head-and-shoulders portrait", - ]), - ("genre|street", [ - "a street photography image", - "a candid street photo", - "a photo of city street life", - "an unposed urban scene", - ]), - ("genre|wildlife", [ - "a wildlife photograph", - "a photo of a wild animal in nature", - "a nature wildlife image", - ]), - ("genre|macro", [ - "a macro photograph", - "an extreme close-up photo", - "a photo of a tiny subject magnified", - ]), - ("genre|architecture", [ - "an architecture photograph", - "a photo of a building's architecture", - "an architectural photograph of a structure", - ]), - ("genre|food", [ - "a food photograph", - "a photo of a prepared meal or dish", - "a culinary food photo", - ]), - ("genre|sports", [ - "a sports photograph", - "a photo of athletes in action", - "an action shot from a sporting event", - ]), - ("genre|event", [ - "an event photograph", - "a photo of a wedding, party, or ceremony", - "a photo from a public event", - ]), - ("genre|abstract", [ - "an abstract photograph", - "an abstract image with no clear subject", - "an abstract pattern photo", - ]), - ("genre|still life", [ - "a still life photograph", - "an arranged still life photo of objects", - "a tabletop still life image", - ]), - ("genre|aerial", [ - "an aerial photograph", - "a top-down view from above", - "a drone or aerial photo of the ground", - ]), - # --- subject|people (6) --- - ("subject|people|person", [ - "a photograph of a person", - "a single individual in a photo", - "a person captured in an image", - ]), - ("subject|people|couple", [ - "a photograph of a couple", - "two people together in a photo", - "a romantic couple captured in a picture", - ]), - ("subject|people|group", [ - "a photograph of a group of people", - "several people together in a photo", - "a crowd or group photo", - ]), - ("subject|people|child", [ - "a photograph of a child", - "a photo of a young kid", - "a picture of a school-aged child", - ]), - ("subject|people|baby", [ - "a photograph of a baby", - "a photo of an infant", - "a picture of a newborn or toddler", - ]), - ("subject|people|elderly person", [ - "a photograph of an elderly person", - "a photo of an older adult", - "a picture of a senior citizen", - ]), - # --- subject|animal (8) --- - ("subject|animal|dog", [ - "a photograph of a dog", - "a close-up photo of a dog with fur and a snout", - "a domestic dog or puppy", - "a picture of a canine pet", - ]), - ("subject|animal|cat", [ - "a photograph of a cat", - "a close-up photo of a cat with whiskers and pointed ears", - "a domestic cat or kitten", - "a picture of a feline pet", - ]), - ("subject|animal|bird", [ - "a photograph of a bird", - "a close-up of a bird with feathers and a beak", - "a wild or perched bird", - ]), - ("subject|animal|horse", [ - "a photograph of a horse", - "a photo of a horse with mane and hooves", - "an equine animal in a photo", - ]), - ("subject|animal|insect", [ - "a photograph of an insect", - "a close-up of a bug with six legs and antennae", - "a macro photo of an insect", - ]), - ("subject|animal|fish", [ - "a photograph of a fish", - "a photo of a fish underwater with fins", - "a picture of a marine or freshwater fish", - ]), - ("subject|animal|reptile", [ - "a photograph of a reptile", - "a photo of a lizard, snake, or turtle", - "a cold-blooded reptile with scales", - ]), - ("subject|animal|wild animal", [ - "a photograph of a wild animal", - "a photo of an animal in its natural habitat", - "a wildlife photo of an undomesticated creature", - ]), - # --- subject|nature (10) --- - ("subject|nature|flower", [ - "a photograph of a flower", - "a close-up of a flower with petals", - "a bloom or blossom in a photo", - ]), - ("subject|nature|tree", [ - "a photograph of a tree", - "a photo of a tree with leaves and branches", - "a single prominent tree in a photo", - ]), - ("subject|nature|mountain", [ - "a photograph of a mountain", - "a photo of a mountain peak or range", - "a mountainous landscape", - ]), - ("subject|nature|waterfall", [ - "a photograph of a waterfall", - "a photo of falling water down rocks", - "a cascading waterfall in nature", - ]), - ("subject|nature|river", [ - "a photograph of a river", - "a photo of a flowing river or stream", - "a river winding through landscape", - ]), - ("subject|nature|lake", [ - "a photograph of a lake", - "a photo of a calm lake or pond", - "a still body of water in nature", - ]), - ("subject|nature|ocean", [ - "a photograph of the ocean", - "a photo of the sea with waves", - "an ocean or seascape image", - ]), - ("subject|nature|cloud", [ - "a photograph of clouds", - "a sky filled with clouds", - "a cloudscape photo", - ]), - ("subject|nature|rock", [ - "a photograph of rocks", - "a photo of rocky terrain or boulders", - "a stone or rock formation", - ]), - ("subject|nature|field", [ - "a photograph of a field", - "a photo of an open grassy or agricultural field", - "a meadow or pasture", - ]), - # --- subject|vehicle (5) --- - ("subject|vehicle|car", [ - "a photograph of a car", - "a photo of an automobile with wheels", - "a picture of a parked or driving car", - ]), - ("subject|vehicle|bicycle", [ - "a photograph of a bicycle", - "a photo of a bike with two wheels", - "a cyclist or bicycle in a picture", - ]), - ("subject|vehicle|boat", [ - "a photograph of a boat", - "a photo of a boat or ship on water", - "a watercraft in a picture", - ]), - ("subject|vehicle|train", [ - "a photograph of a train", - "a photo of a train on tracks", - "a locomotive or rail vehicle", - ]), - ("subject|vehicle|airplane", [ - "a photograph of an airplane", - "a photo of an aircraft in flight or on the ground", - "a plane with wings and engines", - ]), - # --- subject|structure (5) --- - ("subject|structure|building", [ - "a photograph of a building", - "a photo of an architectural structure or house", - "an exterior shot of a building", - ]), - ("subject|structure|bridge", [ - "a photograph of a bridge", - "a photo of a bridge spanning water or a gap", - "an architectural bridge structure", - ]), - ("subject|structure|tower", [ - "a photograph of a tower", - "a photo of a tall vertical structure", - "a tower rising into the sky", - ]), - ("subject|structure|statue", [ - "a photograph of a statue", - "a photo of a sculpted figure or monument", - "a sculpture in a public space", - ]), - ("subject|structure|ruin", [ - "a photograph of a ruin", - "a photo of ancient or abandoned ruins", - "a derelict historical structure", - ]), - # --- setting (8) --- - ("setting|indoor", [ - "an indoor photograph", - "a photo taken inside a building", - "an interior scene image", - ]), - ("setting|outdoor", [ - "an outdoor photograph", - "a photo taken outdoors", - "an open-air outdoor scene", - ]), - ("setting|urban", [ - "an urban photograph", - "a photo taken in a city or town", - "an image of urban streets and buildings", - ]), - ("setting|rural", [ - "a rural photograph", - "a photo taken in the countryside", - "an image of rural farmland or villages", - ]), - ("setting|beach", [ - "a photograph at a beach", - "a beach scene with sand and water", - "a coastal photograph by the sea", - ]), - ("setting|forest", [ - "a photograph in a forest", - "a photo of trees in a woodland", - "a forest scene with dense trees", - ]), - ("setting|desert", [ - "a photograph in a desert", - "a photo of a dry sandy desert landscape", - "a barren desert scene", - ]), - ("setting|studio", [ - "a studio photograph", - "a photo with controlled studio lighting and backdrop", - "a posed studio image", - ]), - # --- lighting (8) --- - ("lighting|sunrise", [ - "a photograph taken at sunrise", - "an early morning photo with the sun rising", - "a sunrise scene with warm sky colors", - ]), - ("lighting|sunset", [ - "a photograph taken at sunset", - "an evening photo with the sun setting", - "a sunset sky with warm orange colors", - ]), - ("lighting|golden hour", [ - "a photograph during golden hour", - "a photo with warm low-angle golden sunlight", - "an image bathed in golden afternoon light", - ]), - ("lighting|blue hour", [ - "a photograph during blue hour", - "a twilight photo with a deep blue sky", - "a dusk or pre-dawn image with cool blue tones", - ]), - ("lighting|night", [ - "a photograph taken at night", - "a dark nighttime image", - "a photo in low light after dark", - ]), - ("lighting|backlit", [ - "a backlit photograph", - "a photo with the light source behind the subject", - "a subject lit from behind with rim light", - ]), - ("lighting|silhouette", [ - "a silhouette photograph", - "a dark subject outlined against a bright background", - "a backlit silhouette of a figure or object", - ]), - ("lighting|low light", [ - "a low light photograph", - "a dim photo with little available light", - "an image taken in poor lighting conditions", - ]), - # --- technique (8) --- - ("technique|black and white", [ - "a black and white photograph", - "a monochrome grayscale image", - "a desaturated B&W photo", - ]), - ("technique|long exposure", [ - "a long exposure photograph", - "a photo with motion blurred by a slow shutter", - "a smooth water or sky long exposure image", - ]), - ("technique|bokeh", [ - "a photograph with bokeh", - "a photo with a blurred out-of-focus background", - "an image showing creamy bokeh circles", - ]), - ("technique|panorama", [ - "a panoramic photograph", - "a very wide aspect ratio panorama image", - "a stitched panoramic scene", - ]), - ("technique|close-up", [ - "a close-up photograph", - "a tight close-up of a subject", - "a near-distance detailed photo", - ]), - ("technique|wide angle", [ - "a wide angle photograph", - "a photo taken with a wide field of view", - "an expansive wide-lens image", - ]), - ("technique|motion blur", [ - "a photograph with motion blur", - "a photo where moving subjects are blurred", - "an image with intentional motion streaks", - ]), - ("technique|reflection", [ - "a photograph with reflections", - "a photo showing reflective surfaces like water or glass", - "an image with mirrored reflections", - ]), - # --- mood (6) --- - ("mood|dramatic", [ - "a dramatic photograph", - "an image with intense contrast and powerful mood", - "a striking dramatic scene", - ]), - ("mood|peaceful", [ - "a peaceful photograph", - "a calm tranquil scene", - "a serene quiet image", - ]), - ("mood|moody", [ - "a moody photograph", - "a dark atmospheric image", - "a brooding or melancholy photo", - ]), - ("mood|vibrant", [ - "a vibrant photograph", - "a colorful saturated image", - "an energetic high-color photo", - ]), - ("mood|minimal", [ - "a minimalist photograph", - "an image with very few elements and negative space", - "a clean simple minimalist composition", - ]), - ("mood|chaotic", [ - "a chaotic photograph", - "a busy crowded disorderly scene", - "an image with many overlapping elements", - ]), - # --- weather (6) --- - ("weather|sunny", [ - "a photograph in sunny weather", - "a bright clear-sky photo with sunshine", - "an image taken on a sunny day", - ]), - ("weather|cloudy", [ - "a photograph in cloudy weather", - "a photo under an overcast cloudy sky", - "an image with a gray cloudy sky", - ]), - ("weather|rainy", [ - "a photograph in rainy weather", - "a photo with visible rain or wet surfaces", - "a rainy day image", - ]), - ("weather|snowy", [ - "a photograph in snowy weather", - "a photo with snow covering the ground", - "a snowy winter scene", - ]), - ("weather|foggy", [ - "a photograph in foggy weather", - "a misty foggy scene with reduced visibility", - "an image obscured by fog or mist", - ]), - ("weather|stormy", [ - "a photograph in stormy weather", - "a photo of a storm with dramatic clouds", - "an image during severe weather", - ]), - # --- season (4) --- - ("season|spring", [ - "a photograph taken in spring", - "a springtime scene with fresh green growth and flowers", - "an image with spring bloom", - ]), - ("season|summer", [ - "a photograph taken in summer", - "a summer scene with lush greenery and warm light", - "an image from the summer season", - ]), - ("season|autumn", [ - "a photograph taken in autumn", - "an autumn scene with red, orange, and yellow leaves", - "a fall foliage image", - ]), - ("season|winter", [ - "a photograph taken in winter", - "a winter scene with bare trees or snow", - "a cold-weather image from the winter season", - ]), -] - - # --------------------------------------------------------------------------- # ONNX wrapper # --------------------------------------------------------------------------- @@ -605,45 +133,11 @@ def export_image_encoder(model, output_path, opset, fp16=False): print(f"Output shape: {ort_out.shape}, norm: {np.linalg.norm(ort_out, axis=-1)}") -def generate_tags(model, tokenizer, output_path): - """Pre-compute text embeddings for photo tags using prompt ensembling. - - For each tag: - 1. encode all of its prompt variants - 2. average the L2-normalized embeddings - 3. re-normalize the average to unit length - - No centering — centering shifts the cosine similarity range so much - that the threshold becomes meaningless, and the top-K filter on the - consumer side is a better way to limit over-eager tags. - """ - tags = [tag for tag, _ in TAG_VOCAB] - - centroids = [] - with torch.no_grad(): - for _tag, prompts in TAG_VOCAB: - tokens = tokenizer(prompts) - feats = model.encode_text(tokens) - feats = F.normalize(feats, dim=-1) # normalize each prompt - centroid = feats.mean(dim=0) # average - centroid = F.normalize(centroid, dim=-1) # renormalize - centroids.append(centroid.cpu().numpy()) - - embeddings = [c.tolist() for c in centroids] - - data = {"tags": tags, "embeddings": embeddings} - with open(output_path, "w") as f: - json.dump(data, f, indent=2) - - print(f"Generated {len(tags)} tag centroids → {output_path}") - print(f" (ensembled from {sum(len(p) for _, p in TAG_VOCAB)} prompts)") - - # --------------------------------------------------------------------------- # Main # --------------------------------------------------------------------------- -def convert(output, tags_output, opset=20, fp16=False): +def convert(output, opset=20, fp16=False): """Entry point for programmatic conversion.""" os.makedirs(os.path.dirname(output) or ".", exist_ok=True) @@ -658,10 +152,7 @@ def convert(output, tags_output, opset=20, fp16=False): for param in model.parameters(): param.requires_grad = False - tokenizer = open_clip.get_tokenizer("RN101") - export_image_encoder(model, output, opset, fp16=fp16) - generate_tags(model, tokenizer, tags_output) print("Done!") @@ -671,13 +162,12 @@ def main(): description="Export OpenCLIP RN101 (YFCC15M) image encoder to ONNX" ) parser.add_argument("--output", required=True, help="Output ONNX path") - parser.add_argument("--tags-output", required=True, help="Output tags.json path") parser.add_argument("--opset", type=int, default=20, help="ONNX opset version") parser.add_argument("--fp16", action="store_true", help="convert weights to FP16 after export (default: FP32)") args = parser.parse_args() - convert(args.output, args.tags_output, args.opset, fp16=args.fp16) + convert(args.output, args.opset, fp16=args.fp16) if __name__ == "__main__": diff --git a/models/embedding-openclip-rn101-yfcc15m/demo.py b/models/embedding-openclip-rn101-yfcc15m/demo.py index 75083fd..308c09e 100644 --- a/models/embedding-openclip-rn101-yfcc15m/demo.py +++ b/models/embedding-openclip-rn101-yfcc15m/demo.py @@ -1,6 +1,8 @@ -"""Demo: compute image embedding and show top-5 zero-shot tags. +"""Demo: compute the image embedding and write it out as JSON. -Saves a PNG with the original image and top-5 tag predictions overlaid. +There is nothing to draw on an image for an embedding model, so the demo +writes the vector itself. The reported norm is the useful check: the export +bakes L2 normalisation into the graph, so it must come back as 1.0. """ import argparse @@ -10,7 +12,7 @@ import numpy as np import onnxruntime as ort -from PIL import Image, ImageDraw, ImageFont, ImageOps +from PIL import Image, ImageOps IMAGE_SIZE = 224 @@ -31,57 +33,14 @@ def preprocess(image): return arr -def draw_tags(image, tags_scores): - """Draw top tags on the image and return the annotated image.""" - draw = ImageDraw.Draw(image) - w, h = image.size - font_size = max(16, min(w, h) // 25) - pad = max(4, font_size // 4) - try: - font = ImageFont.truetype("/System/Library/Fonts/Helvetica.ttc", font_size) - except (OSError, IOError): - try: - font = ImageFont.truetype("/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf", font_size) - except (OSError, IOError): - font = ImageFont.load_default() - - y = pad * 2 - for tag, score in tags_scores: - text = f"{tag}: {score:.2f}" - bbox = draw.textbbox((0, 0), text, font=font) - tw, th = bbox[2] - bbox[0], bbox[3] - bbox[1] - draw.rectangle( - [pad, y - pad // 2, pad * 2 + tw, y + th + pad // 2], - fill=(0, 0, 0, 180), - ) - draw.text((pad + pad // 2, y), text, fill=(255, 255, 255), font=font) - y += th + pad * 2 - - return image - - def run_inference(model_path, image_path, output_path): os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True) - # Load tags - model_dir = os.path.dirname(model_path) - tags_path = os.path.join(model_dir, "tags.json") - if not os.path.isfile(tags_path): - print(f"Warning: tags.json not found at {tags_path}") - tags, tag_embeddings = [], None - else: - with open(tags_path) as f: - data = json.load(f) - tags = data["tags"] - tag_embeddings = np.array(data["embeddings"], dtype=np.float32) - t0 = time.perf_counter() - # Load model session = ort.InferenceSession(model_path, providers=["CPUExecutionProvider"]) t_load = time.perf_counter() - # Load and preprocess image image = Image.open(image_path) image = ImageOps.exif_transpose(image) image = image.convert("RGB") @@ -89,32 +48,27 @@ def run_inference(model_path, image_path, output_path): input_tensor = preprocess(image) t_pre = time.perf_counter() - # Run inference (embedding,) = session.run(None, {"image": input_tensor}) t_inf = time.perf_counter() + vector = embedding[0].astype(np.float32) + norm = float(np.linalg.norm(vector)) + print(f" Image: {orig_w}x{orig_h}") - print(f" Embedding: shape={embedding.shape}, norm={np.linalg.norm(embedding):.4f}") + print(f" Embedding: dim={vector.shape[0]}, norm={norm:.4f}") print(f" Load: {t_load - t0:.3f}s Preprocess: {t_pre - t_load:.3f}s Inference: {t_inf - t_pre:.3f}s") - # Zero-shot classification - if tag_embeddings is not None: - scores = (embedding @ tag_embeddings.T)[0] # dot product = cosine sim - top_idx = np.argsort(scores)[::-1][:5] - tags_scores = [(tags[i], float(scores[i])) for i in top_idx] - - print(" Top-5 tags:") - for tag, score in tags_scores: - print(f" {score:.3f} {tag}") - - # Save annotated image - annotated = image.copy().convert("RGBA") - overlay = Image.new("RGBA", annotated.size, (0, 0, 0, 0)) - annotated = draw_tags(overlay, tags_scores) - result = Image.alpha_composite(image.convert("RGBA"), annotated) - result.convert("RGB").save(output_path) - else: - image.save(output_path) + with open(output_path, "w") as f: + json.dump( + { + "image": os.path.basename(image_path), + "dim": int(vector.shape[0]), + "norm": norm, + "embedding": [float(v) for v in vector], + }, + f, + indent=2, + ) print(f" Saved: {output_path}") print(f" Total: {time.perf_counter() - t0:.3f}s") @@ -129,7 +83,7 @@ def main(): parser = argparse.ArgumentParser(description="OpenCLIP embedding demo") parser.add_argument("--model", required=True, help="Path to model.onnx") parser.add_argument("--image", required=True, help="Input image path") - parser.add_argument("--output", required=True, help="Output PNG path") + parser.add_argument("--output", required=True, help="Output JSON path") args = parser.parse_args() demo(args.model, args.image, args.output) diff --git a/models/embedding-openclip-rn101-yfcc15m/model.yaml b/models/embedding-openclip-rn101-yfcc15m/model.yaml index 4951ed1..7e3b1d4 100644 --- a/models/embedding-openclip-rn101-yfcc15m/model.yaml +++ b/models/embedding-openclip-rn101-yfcc15m/model.yaml @@ -1,6 +1,6 @@ id: embedding-openclip-rn101-yfcc15m name: "embedding openclip rn101 yfcc15m" -description: "OpenCLIP RN101 for tag suggestions and image-similarity search; trained on Flickr Creative Commons photos (YFCC15M)" +description: "OpenCLIP RN101 image encoder for auto-tagging and image-similarity search; trained on Flickr Creative Commons photos (YFCC15M)" task: embedding version: "0.1" arch: openclip-rn101 @@ -36,6 +36,5 @@ convert: - script: convert.py args: output: "{output}/model.onnx" - tags_output: "{output}/tags.json" opset: 20 fp16: true diff --git a/models/embedding-openclip-rn101-yfcc15m/tags.md b/models/embedding-openclip-rn101-yfcc15m/tags.md deleted file mode 100644 index df08e6c..0000000 --- a/models/embedding-openclip-rn101-yfcc15m/tags.md +++ /dev/null @@ -1,152 +0,0 @@ -# Default tag vocabulary - -86 hierarchical tags for zero-shot image classification. -Tags use `|` as hierarchy separator, matching darktable's tag system. - -## genre (12) - -| Tag | CLIP prompt | -|-----|-------------| -| `genre\|landscape` | landscape photography | -| `genre\|portrait` | portrait photography | -| `genre\|street` | street photography | -| `genre\|wildlife` | wildlife photography | -| `genre\|macro` | macro photography | -| `genre\|architecture` | architecture photography | -| `genre\|food` | food photography | -| `genre\|sports` | sports photography | -| `genre\|event` | event photography | -| `genre\|abstract` | abstract photography | -| `genre\|still life` | still life photography | -| `genre\|aerial` | aerial photography | - -## subject (34) - -### subject|people (6) - -| Tag | CLIP prompt | -|-----|-------------| -| `subject\|people\|person` | a photo of a person | -| `subject\|people\|couple` | a photo of a couple | -| `subject\|people\|group` | a photo of a group of people | -| `subject\|people\|child` | a photo of a child | -| `subject\|people\|baby` | a photo of a baby | -| `subject\|people\|elderly person` | a photo of an elderly person | - -### subject|animal (8) - -| Tag | CLIP prompt | -|-----|-------------| -| `subject\|animal\|dog` | a photo of a dog | -| `subject\|animal\|cat` | a photo of a cat | -| `subject\|animal\|bird` | a photo of a bird | -| `subject\|animal\|horse` | a photo of a horse | -| `subject\|animal\|insect` | a photo of an insect | -| `subject\|animal\|fish` | a photo of a fish | -| `subject\|animal\|reptile` | a photo of a reptile | -| `subject\|animal\|wild animal` | a photo of a wild animal | - -### subject|nature (10) - -| Tag | CLIP prompt | -|-----|-------------| -| `subject\|nature\|flower` | a photo of a flower | -| `subject\|nature\|tree` | a photo of a tree | -| `subject\|nature\|mountain` | a photo of a mountain | -| `subject\|nature\|waterfall` | a photo of a waterfall | -| `subject\|nature\|river` | a photo of a river | -| `subject\|nature\|lake` | a photo of a lake | -| `subject\|nature\|ocean` | a photo of the ocean | -| `subject\|nature\|cloud` | a photo of clouds | -| `subject\|nature\|rock` | a photo of rocks | -| `subject\|nature\|field` | a photo of a field | - -### subject|vehicle (5) - -| Tag | CLIP prompt | -|-----|-------------| -| `subject\|vehicle\|car` | a photo of a car | -| `subject\|vehicle\|bicycle` | a photo of a bicycle | -| `subject\|vehicle\|boat` | a photo of a boat | -| `subject\|vehicle\|train` | a photo of a train | -| `subject\|vehicle\|airplane` | a photo of an airplane | - -### subject|structure (5) - -| Tag | CLIP prompt | -|-----|-------------| -| `subject\|structure\|building` | a photo of a building | -| `subject\|structure\|bridge` | a photo of a bridge | -| `subject\|structure\|tower` | a photo of a tower | -| `subject\|structure\|statue` | a photo of a statue | -| `subject\|structure\|ruin` | a photo of a ruin | - -## setting (8) - -| Tag | CLIP prompt | -|-----|-------------| -| `setting\|indoor` | an indoor photograph | -| `setting\|outdoor` | an outdoor photograph | -| `setting\|urban` | a photo taken in an urban setting | -| `setting\|rural` | a photo taken in a rural setting | -| `setting\|beach` | a photo taken at a beach | -| `setting\|forest` | a photo taken in a forest | -| `setting\|desert` | a photo taken in a desert | -| `setting\|studio` | a studio photograph | - -## lighting (8) - -| Tag | CLIP prompt | -|-----|-------------| -| `lighting\|sunrise` | a photo taken at sunrise | -| `lighting\|sunset` | a photo taken at sunset | -| `lighting\|golden hour` | a photo taken during golden hour | -| `lighting\|blue hour` | a photo taken during blue hour | -| `lighting\|night` | a photo taken at night | -| `lighting\|backlit` | a backlit photograph | -| `lighting\|silhouette` | a silhouette photograph | -| `lighting\|low light` | a low light photograph | - -## technique (8) - -| Tag | CLIP prompt | -|-----|-------------| -| `technique\|black and white` | a black and white photograph | -| `technique\|long exposure` | a long exposure photograph | -| `technique\|bokeh` | a photograph with bokeh | -| `technique\|panorama` | a panoramic photograph | -| `technique\|close-up` | a close-up photograph | -| `technique\|wide angle` | a wide angle photograph | -| `technique\|motion blur` | a photograph with motion blur | -| `technique\|reflection` | a photograph with reflections | - -## mood (6) - -| Tag | CLIP prompt | -|-----|-------------| -| `mood\|dramatic` | a dramatic photograph | -| `mood\|peaceful` | a peaceful photograph | -| `mood\|moody` | a moody photograph | -| `mood\|vibrant` | a vibrant photograph | -| `mood\|minimal` | a minimalist photograph | -| `mood\|chaotic` | a chaotic photograph | - -## weather (6) - -| Tag | CLIP prompt | -|-----|-------------| -| `weather\|sunny` | a photo taken in sunny weather | -| `weather\|cloudy` | a photo taken in cloudy weather | -| `weather\|rainy` | a photo taken in rainy weather | -| `weather\|snowy` | a photo taken in snowy weather | -| `weather\|foggy` | a photo taken in foggy weather | -| `weather\|stormy` | a photo taken in stormy weather | - -## season (4) - -| Tag | CLIP prompt | -|-----|-------------| -| `season\|spring` | a photo taken in spring | -| `season\|summer` | a photo taken in summer | -| `season\|autumn` | a photo taken in autumn | -| `season\|winter` | a photo taken in winter | diff --git a/samples/embed/example_01.jpg b/samples/embedding/example_01.jpg similarity index 100% rename from samples/embed/example_01.jpg rename to samples/embedding/example_01.jpg diff --git a/samples/embed/example_02.jpg b/samples/embedding/example_02.jpg similarity index 100% rename from samples/embed/example_02.jpg rename to samples/embedding/example_02.jpg diff --git a/samples/embed/example_03.jpg b/samples/embedding/example_03.jpg similarity index 100% rename from samples/embed/example_03.jpg rename to samples/embedding/example_03.jpg