File size: 3,535 Bytes
4346a4c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
import cv2
import numpy as np
from typing import List, Dict, Any, Optional, Tuple
from backend.ml.orchestrator import ObjectIntelligenceOrchestrator
from backend.database.storage import db

class ObjectDetector:
    """
    Object Intelligence SDK Client.
    Provides unified interface for Known Object Detection (RT-DETR),
    Open-Vocabulary Discovery (YOLO-World), and Specific Object Recognition (Visual Embeddings).
    """

    def __init__(self):
        self.orchestrator = ObjectIntelligenceOrchestrator()
        self.mode = "combined"
        self.is_locked = False
        self.lock_target_name = ""
        self.vocabulary = ["cup", "laptop", "screwdriver", "backpack", "bottle"]
        self.confidence_thresh = 0.45
        self.similarity_thresh = 0.65

    def set_mode(self, mode: str):
        """Set detection mode: 'combined', 'known', 'open_vocabulary', or 'specific'."""
        valid_modes = ["combined", "known", "open_vocabulary", "specific"]
        if mode.lower() in valid_modes:
            self.mode = mode.lower()
        else:
            raise ValueError(f"Invalid mode '{mode}'. Supported modes: {valid_modes}")

    def set_vocabulary(self, prompts: List[str]):
        """Set open vocabulary text prompts for YOLO-World."""
        self.vocabulary = prompts
        self.orchestrator.yolo_world.set_vocabulary(prompts)

    def lock(self, target: str):
        """Enable Lock Mode to return ONLY objects matching the specified target concept/identity."""
        self.is_locked = True
        self.lock_target_name = target

    def unlock(self):
        """Disable Lock Mode and resume displaying all detected objects."""
        self.is_locked = False
        self.lock_target_name = ""

    def add_object(self, name: str, images: List[str], category: str = "custom", description: str = "") -> str:
        """Teach a new specific object to the platform using reference image file paths."""
        obj = db.get_object_by_name(name)
        if obj is None:
            obj = db.create_object(name=name, category=category, description=description)

        for img_path in images:
            emb = self.orchestrator.embedding_recognizer.extract_embedding_from_image_path(img_path)
            if emb is not None:
                db.add_embedding(obj.id, emb.tolist())
                db.add_image_record(obj.id, img_path)

        return obj.id

    def detect(self, image: np.ndarray) -> List[Dict[str, Any]]:
        """Run detection pipeline on an image numpy matrix (BGR/RGB)."""
        annotated_frame, detections, metadata = self.orchestrator.process_frame(
            image=image,
            mode=self.mode,
            lock_mode=self.is_locked,
            lock_target=self.lock_target_name,
            open_vocab_prompts=self.vocabulary,
            confidence_thresh=self.confidence_thresh,
            similarity_thresh=self.similarity_thresh
        )
        return detections

    def detect_and_draw(self, image: np.ndarray) -> Tuple[np.ndarray, List[Dict[str, Any]]]:
        """Run detection pipeline and return (annotated_image, detections)."""
        annotated_frame, detections, metadata = self.orchestrator.process_frame(
            image=image,
            mode=self.mode,
            lock_mode=self.is_locked,
            lock_target=self.lock_target_name,
            open_vocab_prompts=self.vocabulary,
            confidence_thresh=self.confidence_thresh,
            similarity_thresh=self.similarity_thresh
        )
        return annotated_frame, detections