muhammadpriv001's picture
Frontend 1.0.0
4346a4c
Raw History Blame Contribute Delete
3.54 kB
import cv2
import numpy as np
from typing import List, Dict, Any, Optional, Tuple
from backend.ml.orchestrator import ObjectIntelligenceOrchestrator
from backend.database.storage import db
class ObjectDetector:
"""
Object Intelligence SDK Client.
Provides unified interface for Known Object Detection (RT-DETR),
Open-Vocabulary Discovery (YOLO-World), and Specific Object Recognition (Visual Embeddings).
"""
def __init__(self):
self.orchestrator = ObjectIntelligenceOrchestrator()
self.mode = "combined"
self.is_locked = False
self.lock_target_name = ""
self.vocabulary = ["cup", "laptop", "screwdriver", "backpack", "bottle"]
self.confidence_thresh = 0.45
self.similarity_thresh = 0.65
def set_mode(self, mode: str):
"""Set detection mode: 'combined', 'known', 'open_vocabulary', or 'specific'."""
valid_modes = ["combined", "known", "open_vocabulary", "specific"]
if mode.lower() in valid_modes:
self.mode = mode.lower()
else:
raise ValueError(f"Invalid mode '{mode}'. Supported modes: {valid_modes}")
def set_vocabulary(self, prompts: List[str]):
"""Set open vocabulary text prompts for YOLO-World."""
self.vocabulary = prompts
self.orchestrator.yolo_world.set_vocabulary(prompts)
def lock(self, target: str):
"""Enable Lock Mode to return ONLY objects matching the specified target concept/identity."""
self.is_locked = True
self.lock_target_name = target
def unlock(self):
"""Disable Lock Mode and resume displaying all detected objects."""
self.is_locked = False
self.lock_target_name = ""
def add_object(self, name: str, images: List[str], category: str = "custom", description: str = "") -> str:
"""Teach a new specific object to the platform using reference image file paths."""
obj = db.get_object_by_name(name)
if obj is None:
obj = db.create_object(name=name, category=category, description=description)
for img_path in images:
emb = self.orchestrator.embedding_recognizer.extract_embedding_from_image_path(img_path)
if emb is not None:
db.add_embedding(obj.id, emb.tolist())
db.add_image_record(obj.id, img_path)
return obj.id
def detect(self, image: np.ndarray) -> List[Dict[str, Any]]:
"""Run detection pipeline on an image numpy matrix (BGR/RGB)."""
annotated_frame, detections, metadata = self.orchestrator.process_frame(
image=image,
mode=self.mode,
lock_mode=self.is_locked,
lock_target=self.lock_target_name,
open_vocab_prompts=self.vocabulary,
confidence_thresh=self.confidence_thresh,
similarity_thresh=self.similarity_thresh
)
return detections
def detect_and_draw(self, image: np.ndarray) -> Tuple[np.ndarray, List[Dict[str, Any]]]:
"""Run detection pipeline and return (annotated_image, detections)."""
annotated_frame, detections, metadata = self.orchestrator.process_frame(
image=image,
mode=self.mode,
lock_mode=self.is_locked,
lock_target=self.lock_target_name,
open_vocab_prompts=self.vocabulary,
confidence_thresh=self.confidence_thresh,
similarity_thresh=self.similarity_thresh
)
return annotated_frame, detections