""" VLM-based change description via Ollama (Llama 3.2 Vision). Llama 3.2 Vision (mllama) only accepts ONE image per request, so we concatenate the before and after images into a single side-by-side composite with labels and send that. Requires: - Ollama running locally on port 11434 - llama3.2-vision model pulled (`ollama pull llama3.2-vision`) """ import base64 import io import json import os import urllib.request import urllib.error import cv2 import numpy as np OLLAMA_URL = "http://localhost:11434/api/generate" OLLAMA_MODEL = "llama3.2-vision:latest" GROQ_URL = "https://api.groq.com/openai/v1/chat/completions" GROQ_MODEL = "meta-llama/llama-4-scout-17b-16e-instruct" TIMEOUT_SECONDS = 180 def _make_side_by_side(img1_path, img2_path, max_height=384): """Build a single composite image with BEFORE | AFTER labels. Returns base64-encoded PNG, or None on failure.""" a = cv2.imread(img1_path) b = cv2.imread(img2_path) if a is None or b is None: return None # Normalize heights def resize_to_height(img, h): scale = h / img.shape[0] return cv2.resize(img, (int(round(img.shape[1] * scale)), h), interpolation=cv2.INTER_AREA) a = resize_to_height(a, max_height) b = resize_to_height(b, max_height) # Header strip with labels header_h = 50 total_w = a.shape[1] + b.shape[1] + 20 # 20px gap between header = np.full((header_h, total_w, 3), 30, dtype=np.uint8) cv2.putText(header, "BEFORE", (a.shape[1] // 2 - 70, 35), cv2.FONT_HERSHEY_SIMPLEX, 1.0, (255, 255, 255), 2) cv2.putText(header, "AFTER", (a.shape[1] + 20 + b.shape[1] // 2 - 60, 35), cv2.FONT_HERSHEY_SIMPLEX, 1.0, (255, 255, 255), 2) # Glue panels: a, gap, b gap = np.full((max_height, 20, 3), 30, dtype=np.uint8) body = np.hstack([a, gap, b]) composite = np.vstack([header, body]) ok, buf = cv2.imencode(".png", composite) if not ok: return None return base64.b64encode(buf.tobytes()).decode("ascii") def describe_change(img1_path, img2_path, metrics): """Returns a plain-English description string, or None on failure. Uses Groq API if GROQ_API_KEY is in env, else falls back to local Ollama.""" groq_api_key = os.environ.get("GROQ_API_KEY") try: composite_b64 = _make_side_by_side(img1_path, img2_path) if composite_b64 is None: return None prompt_lines = [ "This single image contains TWO photos placed side by side.", "The LEFT photo is labeled BEFORE. The RIGHT photo is labeled AFTER.", "Describe ONLY what physically changed from BEFORE to AFTER in 1-3 short sentences.", "Be specific about objects (color, shape, position).", "If nothing meaningful changed, say 'No visible changes.'", "Do not list things that are the same in both photos.", ] if metrics: added = metrics.get("added") removed = metrics.get("removed") if added is not None or removed is not None: prompt_lines.append( f"\nDetector hint: added={added}, removed={removed}." ) prompt = "\n".join(prompt_lines) if groq_api_key: # --- Use Groq API --- payload = { "model": GROQ_MODEL, "messages": [ { "role": "user", "content": [ {"type": "text", "text": prompt}, { "type": "image_url", "image_url": { "url": f"data:image/png;base64,{composite_b64}" } } ] } ], "temperature": 0.2, "max_tokens": 120 } req = urllib.request.Request( GROQ_URL, data=json.dumps(payload).encode("utf-8"), headers={ "Content-Type": "application/json", "Authorization": f"Bearer {groq_api_key}", "User-Agent": "Mozilla/5.0" }, ) with urllib.request.urlopen(req, timeout=TIMEOUT_SECONDS) as resp: body = json.loads(resp.read().decode("utf-8")) text = (body["choices"][0]["message"]["content"] or "").strip() print("[describe] generated description using Groq API") return text or None else: # --- Fallback to local Ollama --- payload = { "model": OLLAMA_MODEL, "prompt": prompt, "images": [composite_b64], "stream": False, "options": { "temperature": 0.2, "num_predict": 120, "num_ctx": 2048, }, "keep_alive": "10m", } req = urllib.request.Request( OLLAMA_URL, data=json.dumps(payload).encode("utf-8"), headers={"Content-Type": "application/json"}, ) with urllib.request.urlopen(req, timeout=TIMEOUT_SECONDS) as resp: body = json.loads(resp.read().decode("utf-8")) text = (body.get("response") or "").strip() print("[describe] generated description using local Ollama") return text or None except urllib.error.HTTPError as e: try: err_body = e.read().decode("utf-8") print(f"[describe] Groq API HTTPError {e.code}: {err_body}") except Exception: print(f"[describe] Groq API HTTPError {e}: (could not read body)") return None except urllib.error.URLError as e: if groq_api_key: print(f"[describe] Groq API error/unreachable: {e}") else: print(f"[describe] Ollama unreachable: {e}") return None except Exception as e: print(f"[describe] error: {e}") return None