import cv2 import numpy as np from typing import List, Dict, Any, Optional, Tuple from backend.ml.orchestrator import ObjectIntelligenceOrchestrator from backend.database.storage import db class ObjectDetector: """ Object Intelligence SDK Client. Provides unified interface for Known Object Detection (RT-DETR), Open-Vocabulary Discovery (YOLO-World), and Specific Object Recognition (Visual Embeddings). """ def __init__(self): self.orchestrator = ObjectIntelligenceOrchestrator() self.mode = "combined" self.is_locked = False self.lock_target_name = "" self.vocabulary = ["cup", "laptop", "screwdriver", "backpack", "bottle"] self.confidence_thresh = 0.45 self.similarity_thresh = 0.65 def set_mode(self, mode: str): """Set detection mode: 'combined', 'known', 'open_vocabulary', or 'specific'.""" valid_modes = ["combined", "known", "open_vocabulary", "specific"] if mode.lower() in valid_modes: self.mode = mode.lower() else: raise ValueError(f"Invalid mode '{mode}'. Supported modes: {valid_modes}") def set_vocabulary(self, prompts: List[str]): """Set open vocabulary text prompts for YOLO-World.""" self.vocabulary = prompts self.orchestrator.yolo_world.set_vocabulary(prompts) def lock(self, target: str): """Enable Lock Mode to return ONLY objects matching the specified target concept/identity.""" self.is_locked = True self.lock_target_name = target def unlock(self): """Disable Lock Mode and resume displaying all detected objects.""" self.is_locked = False self.lock_target_name = "" def add_object(self, name: str, images: List[str], category: str = "custom", description: str = "") -> str: """Teach a new specific object to the platform using reference image file paths.""" obj = db.get_object_by_name(name) if obj is None: obj = db.create_object(name=name, category=category, description=description) for img_path in images: emb = self.orchestrator.embedding_recognizer.extract_embedding_from_image_path(img_path) if emb is not None: db.add_embedding(obj.id, emb.tolist()) db.add_image_record(obj.id, img_path) return obj.id def detect(self, image: np.ndarray) -> List[Dict[str, Any]]: """Run detection pipeline on an image numpy matrix (BGR/RGB).""" annotated_frame, detections, metadata = self.orchestrator.process_frame( image=image, mode=self.mode, lock_mode=self.is_locked, lock_target=self.lock_target_name, open_vocab_prompts=self.vocabulary, confidence_thresh=self.confidence_thresh, similarity_thresh=self.similarity_thresh ) return detections def detect_and_draw(self, image: np.ndarray) -> Tuple[np.ndarray, List[Dict[str, Any]]]: """Run detection pipeline and return (annotated_image, detections).""" annotated_frame, detections, metadata = self.orchestrator.process_frame( image=image, mode=self.mode, lock_mode=self.is_locked, lock_target=self.lock_target_name, open_vocab_prompts=self.vocabulary, confidence_thresh=self.confidence_thresh, similarity_thresh=self.similarity_thresh ) return annotated_frame, detections