| """ |
| Object Intelligence service — combines object detection (YOLO) + QR codes |
| + barcodes into a unified object intelligence report. |
| |
| For each detected object, produces: |
| - class label + confidence + bounding box |
| - cropped image (base64) |
| - classification flags (is_vehicle, is_screen, is_document, etc.) |
| - searchable metadata (decoded text for QR/barcodes, etc.) |
| |
| Categorizes objects into: |
| - vehicles (car, truck, bus, motorcycle, etc.) |
| - license_plates (detected via OCR on vehicle regions — heuristic) |
| - qr_codes (from QR provider) |
| - barcodes (from barcode provider) |
| - documents (book, document-like objects) |
| - screens (tv, laptop, monitor) |
| - phones (cell phone) |
| - laptops (laptop) |
| - watches (clock, watch) |
| - logos (TODO — would need a logo detector) |
| - text_regions (from OCR provider) |
| |
| Pure computation — relies on existing providers via the orchestrator. |
| """ |
|
|
| from __future__ import annotations |
|
|
| import base64 |
| import time |
| from typing import List |
|
|
| import cv2 |
| import numpy as np |
|
|
| from cores.vision import BBox, crop_region, numpy_to_base64 |
| from models.jobs import JobRequest |
| from models.reports import ( |
| DetectedObjectIntelligence, |
| ObjectIntelligenceResult, |
| ) |
| from models.providers import ProviderCapability |
| from orchestrator.runner import Orchestrator |
| from pipeline import InputValidator, ImagePreprocessor, ImageHasher, FeatureExtractor |
| from utils.logging import execution_context, new_execution_id |
|
|
|
|
| |
| VEHICLE_CLASSES = {"car", "truck", "bus", "motorcycle", "bicycle", "boat", "airplane", "train"} |
| SCREEN_CLASSES = {"tv", "laptop", "mouse", "remote", "keyboard", "cell phone"} |
| DOCUMENT_CLASSES = {"book"} |
| PHONE_CLASSES = {"cell phone"} |
| LAPTOP_CLASSES = {"laptop"} |
| WATCH_CLASSES = {"clock"} |
| |
|
|
| |
|
|
|
|
| class ObjectIntelligenceService: |
| """Combines object detection + QR + barcode into unified intelligence.""" |
|
|
| def __init__( |
| self, |
| orchestrator: Orchestrator, |
| validator: InputValidator, |
| preprocessor: ImagePreprocessor, |
| hasher: ImageHasher, |
| feature_extractor: FeatureExtractor, |
| ) -> None: |
| self._orchestrator = orchestrator |
| self._validator = validator |
| self._preprocessor = preprocessor |
| self._hasher = hasher |
| self._feature_extractor = feature_extractor |
|
|
| async def analyze(self, request: JobRequest) -> dict: |
| """Run object intelligence: YOLO + QR + barcode + classification.""" |
| eid = new_execution_id() |
| with execution_context(execution_id=eid, provider_id="object_intelligence_service"): |
| t0 = time.perf_counter() |
|
|
| vr = self._validator.validate( |
| image_url=request.image_url, |
| image_base64=request.image_base64, |
| ) |
| if not vr.valid: |
| return {"success": False, "error": vr.error, "error_type": "ValidationError"} |
|
|
| if vr.source == "url": |
| pre = self._preprocessor.from_url(request.image_url) |
| else: |
| pre = self._preprocessor.from_bytes(vr.image_bytes, vr.source) |
|
|
| img_hash = self._hasher.hash(pre.image) |
| pipeline_output = self._feature_extractor.extract( |
| pre.image, img_hash, pre.width, pre.height, pre.source, |
| original_bytes=pre.original_bytes, |
| original_format=pre.original_format, |
| ) |
|
|
| |
| results = await self._orchestrator.run( |
| pipeline_output=pipeline_output, |
| capabilities=[ProviderCapability.OBJECT_DETECTION], |
| provider_whitelist=request.providers or None, |
| execution_id=eid, |
| ) |
|
|
| |
| all_objects: list[dict] = [] |
| for name, result in results.items(): |
| if not result.success: |
| continue |
| for obj in result.normalized.get("objects", []): |
| obj["_provider"] = name |
| all_objects.append(obj) |
|
|
| |
| img = pre.image |
| intelligent_objects: list[DetectedObjectIntelligence] = [] |
| for obj in all_objects: |
| label = obj.get("label", "unknown") |
| confidence = float(obj.get("confidence", 0.0)) |
| box_dict = obj.get("box", {"x": 0, "y": 0, "w": 0, "h": 0}) |
|
|
| |
| bbox = BBox(box_dict["x"], box_dict["y"], box_dict["w"], box_dict["h"]) |
| crop = crop_region(img, bbox, margin=0.05) |
| crop_b64 = None |
| if crop.size > 0: |
| try: |
| crop_b64 = numpy_to_base64(crop, ".jpg", quality=85) |
| except Exception: |
| crop_b64 = None |
|
|
| |
| is_vehicle = label in VEHICLE_CLASSES |
| is_screen = label in SCREEN_CLASSES |
| is_document = label in DOCUMENT_CLASSES |
| is_phone = label in PHONE_CLASSES |
| is_laptop = label in LAPTOP_CLASSES |
| is_watch = label in WATCH_CLASSES |
| is_qr = obj.get("is_qr_code", False) |
| is_barcode = obj.get("is_barcode", False) |
| |
| is_license_plate = False |
| is_logo = False |
| is_text_region = False |
|
|
| |
| searchable: dict = {"provider": obj.get("_provider", "")} |
| if is_qr or is_barcode: |
| decoded = obj.get("decoded_text", "") |
| if decoded: |
| searchable["decoded_text"] = decoded |
| searchable["search_url"] = f"https://www.google.com/search?q={decoded}" |
| if is_barcode: |
| btype = obj.get("barcode_type", "") |
| if btype: |
| searchable["barcode_type"] = btype |
|
|
| intelligent_objects.append(DetectedObjectIntelligence( |
| class_label=label, |
| confidence=round(confidence, 4), |
| box=box_dict, |
| crop_base64=crop_b64, |
| is_vehicle=is_vehicle, |
| is_screen=is_screen, |
| is_document=is_document, |
| is_phone=is_phone, |
| is_laptop=is_laptop, |
| is_watch=is_watch, |
| is_logo=is_logo, |
| is_text_region=is_text_region, |
| is_license_plate=is_license_plate, |
| is_qr_code=is_qr, |
| is_barcode=is_barcode, |
| searchable_metadata=searchable, |
| )) |
|
|
| |
| vehicles = [o for o in intelligent_objects if o.is_vehicle] |
| license_plates = [o for o in intelligent_objects if o.is_license_plate] |
| qr_codes = [o for o in intelligent_objects if o.is_qr_code] |
| barcodes = [o for o in intelligent_objects if o.is_barcode] |
| documents = [o for o in intelligent_objects if o.is_document] |
| screens = [o for o in intelligent_objects if o.is_screen] |
| phones = [o for o in intelligent_objects if o.is_phone] |
| laptops = [o for o in intelligent_objects if o.is_laptop] |
| watches = [o for o in intelligent_objects if o.is_watch] |
| logos = [o for o in intelligent_objects if o.is_logo] |
| text_regions = [o for o in intelligent_objects if o.is_text_region] |
|
|
| elapsed = (time.perf_counter() - t0) * 1000.0 |
| result = ObjectIntelligenceResult( |
| total_objects=len(intelligent_objects), |
| objects=intelligent_objects, |
| vehicles=vehicles, |
| license_plates=license_plates, |
| qr_codes=qr_codes, |
| barcodes=barcodes, |
| documents=documents, |
| screens=screens, |
| phones=phones, |
| laptops=laptops, |
| watches=watches, |
| logos=logos, |
| text_regions=text_regions, |
| elapsed_ms=round(elapsed, 3), |
| ) |
|
|
| return { |
| "success": True, |
| "object_intelligence": result.model_dump(), |
| "elapsed_ms": round(elapsed, 3), |
| } |
|
|