File size: 2,287 Bytes
41cbbf0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
import os
import sys

import torch
from PIL import Image, ImageDraw, ImageFont
from transformers import AutoImageProcessor, AutoModelForObjectDetection


# Hugging Face repository
repo_id = "ConservationDrones/DroneMegaDetector"

# Load the demo image, or an image supplied on the command line.
image_path = sys.argv[1] if len(sys.argv) > 1 else "example.png"
image = Image.open(image_path).convert("RGB")


# Load the processor and model from Hugging Face.
processor = AutoImageProcessor.from_pretrained(
    repo_id,
)

model = AutoModelForObjectDetection.from_pretrained(
    repo_id,
).eval()


# Use 4 CPU threads, as in the original local demo.
torch.set_num_threads(4)


# Preprocess the image and run inference.
encoded = processor(
    images=image,
    return_tensors="pt",
)

with torch.inference_mode():
    outputs = model(**encoded)


# Keep scores above 0.3 and map boxes back to the original image dimensions.
results = processor.post_process_object_detection(
    outputs,
    threshold=0.3,
    target_sizes=[(image.height, image.width)],
)[0]


# Print the class, confidence, and [xmin, ymin, xmax, ymax] in source pixels.
print(
    f"{image_path}: {len(results['scores'])} detections (score >= 0.3)"
)

for score, label, box in zip(
    results["scores"],
    results["labels"],
    results["boxes"],
):
    print(
        model.config.id2label[label.item()],
        f"{score.item():.3f}",
        [round(v, 1) for v in box.tolist()],
    )


# Draw the detections.
draw = ImageDraw.Draw(image)
font = ImageFont.load_default(size=18)

for score, label, box in zip(
    results["scores"],
    results["labels"],
    results["boxes"],
):
    bounds = box.tolist()

    text = (
        f"{model.config.id2label[label.item()]} "
        f"{score.item():.2f}"
    )

    draw.rectangle(
        bounds,
        outline="red",
        width=3,
    )

    text_x = min(
        max(2, bounds[0]),
        image.width - draw.textlength(text, font=font) - 2,
    )

    draw.text(
        (text_x, max(0, bounds[1] - 20)),
        text,
        font=font,
        fill="white",
        stroke_width=2,
        stroke_fill="black",
    )


# Save the annotated image.
output_path = "example_annotated.jpg"
image.save(output_path, quality=95)

print(f"Saved {output_path}")