Spaces:
Running on Zero
Running on Zero
File size: 3,535 Bytes
4346a4c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 | import cv2
import numpy as np
from typing import List, Dict, Any, Optional, Tuple
from backend.ml.orchestrator import ObjectIntelligenceOrchestrator
from backend.database.storage import db
class ObjectDetector:
"""
Object Intelligence SDK Client.
Provides unified interface for Known Object Detection (RT-DETR),
Open-Vocabulary Discovery (YOLO-World), and Specific Object Recognition (Visual Embeddings).
"""
def __init__(self):
self.orchestrator = ObjectIntelligenceOrchestrator()
self.mode = "combined"
self.is_locked = False
self.lock_target_name = ""
self.vocabulary = ["cup", "laptop", "screwdriver", "backpack", "bottle"]
self.confidence_thresh = 0.45
self.similarity_thresh = 0.65
def set_mode(self, mode: str):
"""Set detection mode: 'combined', 'known', 'open_vocabulary', or 'specific'."""
valid_modes = ["combined", "known", "open_vocabulary", "specific"]
if mode.lower() in valid_modes:
self.mode = mode.lower()
else:
raise ValueError(f"Invalid mode '{mode}'. Supported modes: {valid_modes}")
def set_vocabulary(self, prompts: List[str]):
"""Set open vocabulary text prompts for YOLO-World."""
self.vocabulary = prompts
self.orchestrator.yolo_world.set_vocabulary(prompts)
def lock(self, target: str):
"""Enable Lock Mode to return ONLY objects matching the specified target concept/identity."""
self.is_locked = True
self.lock_target_name = target
def unlock(self):
"""Disable Lock Mode and resume displaying all detected objects."""
self.is_locked = False
self.lock_target_name = ""
def add_object(self, name: str, images: List[str], category: str = "custom", description: str = "") -> str:
"""Teach a new specific object to the platform using reference image file paths."""
obj = db.get_object_by_name(name)
if obj is None:
obj = db.create_object(name=name, category=category, description=description)
for img_path in images:
emb = self.orchestrator.embedding_recognizer.extract_embedding_from_image_path(img_path)
if emb is not None:
db.add_embedding(obj.id, emb.tolist())
db.add_image_record(obj.id, img_path)
return obj.id
def detect(self, image: np.ndarray) -> List[Dict[str, Any]]:
"""Run detection pipeline on an image numpy matrix (BGR/RGB)."""
annotated_frame, detections, metadata = self.orchestrator.process_frame(
image=image,
mode=self.mode,
lock_mode=self.is_locked,
lock_target=self.lock_target_name,
open_vocab_prompts=self.vocabulary,
confidence_thresh=self.confidence_thresh,
similarity_thresh=self.similarity_thresh
)
return detections
def detect_and_draw(self, image: np.ndarray) -> Tuple[np.ndarray, List[Dict[str, Any]]]:
"""Run detection pipeline and return (annotated_image, detections)."""
annotated_frame, detections, metadata = self.orchestrator.process_frame(
image=image,
mode=self.mode,
lock_mode=self.is_locked,
lock_target=self.lock_target_name,
open_vocab_prompts=self.vocabulary,
confidence_thresh=self.confidence_thresh,
similarity_thresh=self.similarity_thresh
)
return annotated_frame, detections
|