diff --git a/video/pwtf-dvd/model_code/inference/test_tools/__init__.py b/video/pwtf-dvd/model_code/inference/test_tools/__init__.py
deleted file mode 100644
index e69de29bb2d1d6434b8b29ae775ad8c2e48c5391..0000000000000000000000000000000000000000
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/common.py b/video/pwtf-dvd/model_code/inference/test_tools/common.py
deleted file mode 100644
index e76ac698b8fe9a32dfbdf4e294d5afa521250e71..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/test_tools/common.py
+++ /dev/null
@@ -1,122 +0,0 @@
-import os
-
-os.environ["KMP_DUPLICATE_LIB_OK"] = "TRUE"
-
-from .ct.detection.utils import grab_all_frames, get_valid_faces, sample_chunks
-from .ct.operations import multiple_tracking
-import numpy as np
-from .ct.face_alignment import LandmarkPredictor
-from .ct.detection import FaceDetector
-import cv2
-from .utils import flatten,partition
-
-
-detector = FaceDetector(0)
-predictor = LandmarkPredictor(0)
-
-
-def get_five(ldm68):
- groups = [range(36, 42), range(42, 48), [30], [48], [54]]
- points = []
- for group in groups:
- points.append(ldm68[group].mean(0))
- return np.array(points)
-
-
-def get_bbox(mask):
- try:
- y, x = np.nonzero(mask[..., 0])
- return x.min() - 1, y.min() - 1, x.max() + 1, y.max() + 1
- except:
- return None
-
-
-def get_bigger_box(image, box, scale=0.5):
- height, width = image.shape[:2]
- box = np.rint(box).astype(np.int)
- new_box = box.reshape(2, 2)
- size = new_box[1] - new_box[0]
- diff = scale * size
- diff = diff[None, :] * np.array([-1, 1])[:, None]
- new_box = new_box + diff
- new_box[:, 0] = np.clip(new_box[:, 0], 0, width - 1)
- new_box[:, 1] = np.clip(new_box[:, 1], 0, height - 1)
- new_box = np.rint(new_box).astype(np.int)
- return new_box.reshape(-1)
-
-
-def process_bigger_clips(clips, dete_res, clip_size, step, scale=0.5):
- assert len(clips) % clip_size == 0
- detect_results = sample_chunks(dete_res, clip_size, step)
- clips = sample_chunks(clips, clip_size, step)
- new_clips = []
- for i, (frame_clip, record_clip) in enumerate(zip(clips, detect_results)):
- tracks = multiple_tracking(record_clip)
- for j, track in enumerate(tracks):
- new_images = []
- for (box, ldm, _), frame in zip(track, frame_clip):
- big_box = get_bigger_box(frame, box, scale)
- x1, y1, x2, y2 = big_box
- top_left = big_box[:2][None, :]
- new_ldm5 = ldm - top_left
- box = np.rint(box).astype(np.int)
- new_box = (box.reshape(2, 2) - top_left).reshape(-1)
- feed = LandmarkPredictor.prepare_feed(frame, box)
- ldm68 = predictor(feed) - top_left
- new_images.append(
- (frame[y1:y2, x1:x2], big_box, new_box, new_ldm5, ldm68)
- )
- new_clips.append(new_images)
- return new_clips
-
-
-def post(detected_faces):
- return [[face[:4], None, face[-1]] for face in detected_faces]
-
-
-def check(detect_res):
- return min([len(faces) for faces in detect_res]) != 0
-
-
-def detect_all(file, sfd_only=False, return_frames=False, max_size=None):
- frames = grab_all_frames(file, max_size=max_size, cvt=True)
- if not sfd_only:
- detect_res = flatten(
- [detector.detect(item) for item in partition(frames, 50)]
- )
- detect_res = get_valid_faces(detect_res, thres=0.5)
- else:
- raise NotImplementedError
-
- all_68 = get_lm68(frames, detect_res)
- if not return_frames:
- return detect_res, all_68
- else:
- return detect_res, all_68, frames
-
-
-def get_lm68(frames, detect_res):
- assert len(frames) == len(detect_res)
- frame_count = len(frames)
- all_68 = []
- for i in range(frame_count):
- frame = frames[i]
- faces = detect_res[i]
- if len(faces) == 0:
- res_68 = []
- else:
- feeds = []
- for face in faces:
- assert len(face) == 3
- box = face[0]
- feed = LandmarkPredictor.prepare_feed(frame, box)
- feeds.append(feed)
- res_68 = predictor(feeds)
- assert len(res_68) == len(faces)
- for face, l_68 in zip(faces, res_68):
- if face[1] is None:
- face[1] = get_five(l_68)
- all_68.append(res_68)
-
- assert len(all_68) == len(detect_res)
- return all_68
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__init__.py b/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__init__.py
deleted file mode 100644
index 23447e84f4f5f23a4c4818df2f26cc19a9200363..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__init__.py
+++ /dev/null
@@ -1,56 +0,0 @@
-import cv2
-from .detector import RetinaFace
-from .utils import *
-
-
-def assert_bounded(val, low, up):
- return val >= low and val < up
-
-
-def check_valid(face, w, h):
- box = face[0]
- if box[0] > box[2]:
- return False
- if box[1] > box[3]:
- return False
- for idx, bound in zip([0, 1, 2, 3], [w, h, w, h]):
- if not assert_bounded(box[idx], 0, bound):
- return False
- pts = face[1]
- for p in pts:
- for idx, bound in zip([0, 1], [w, h]):
- if not assert_bounded(p[idx], 0, bound):
- return False
- return True
-
-
-def post_detect(detect_results, scale, w, h):
- new_results = []
- for frame_faces in detect_results:
- new_frame_faces = []
- for box, ldm, score in frame_faces:
- box = box * scale
- ldm = ldm * scale
- face = (box, ldm, score)
- if check_valid(face, w=w, h=h):
- new_frame_faces.append(face)
- new_results.append(new_frame_faces)
- return new_results
-
-
-class FaceDetector(RetinaFace):
- def scale_detect(self, images):
- max_res = 1920
- h, w = images[0].shape[:2]
- if max(h, w) > max_res:
- init_scale = max(h, w) / max_res
- else:
- init_scale = 1
- resize_scale = 2 * init_scale
- resize_w = int(w / resize_scale)
- resize_h = int(h / resize_scale)
- detect_input = [cv2.resize(frame, (resize_w, resize_h)) for frame in images]
- detect_results = post_detect(
- self.detect(detect_input), scale=resize_scale, w=w, h=h,
- )
- return detect_results
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/__init__.cpython-310.pyc b/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/__init__.cpython-310.pyc
deleted file mode 100644
index b7867d4f0406ae13a65803d34f5e0277b85c4e6b..0000000000000000000000000000000000000000
Binary files a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/__init__.cpython-310.pyc and /dev/null differ
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/__init__.cpython-39.pyc b/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/__init__.cpython-39.pyc
deleted file mode 100644
index 3d02deb055e831664e5d69d8e8b30d7bddb3fbe8..0000000000000000000000000000000000000000
Binary files a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/__init__.cpython-39.pyc and /dev/null differ
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/alignment.cpython-310.pyc b/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/alignment.cpython-310.pyc
deleted file mode 100644
index 3c45ded26892d794cc9c0a4743827810fd4744fa..0000000000000000000000000000000000000000
Binary files a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/alignment.cpython-310.pyc and /dev/null differ
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/alignment.cpython-39.pyc b/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/alignment.cpython-39.pyc
deleted file mode 100644
index e9eff5fdad91bb55fd6cb03c73d7202c0e7affbf..0000000000000000000000000000000000000000
Binary files a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/alignment.cpython-39.pyc and /dev/null differ
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/detector.cpython-310.pyc b/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/detector.cpython-310.pyc
deleted file mode 100644
index 9d96f7fba8bc57d4bb24f30852796cf89f0e9edc..0000000000000000000000000000000000000000
Binary files a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/detector.cpython-310.pyc and /dev/null differ
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/detector.cpython-39.pyc b/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/detector.cpython-39.pyc
deleted file mode 100644
index f35f421664abdeb1f40948ac44ca390a0f2bcb59..0000000000000000000000000000000000000000
Binary files a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/detector.cpython-39.pyc and /dev/null differ
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/utils.cpython-310.pyc b/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/utils.cpython-310.pyc
deleted file mode 100644
index 64df2dc550efd8abd74625af40aca7eb0414e2b7..0000000000000000000000000000000000000000
Binary files a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/utils.cpython-310.pyc and /dev/null differ
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/utils.cpython-39.pyc b/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/utils.cpython-39.pyc
deleted file mode 100644
index 2bcf8357e779762c95fb5a961d083c93c1b6990c..0000000000000000000000000000000000000000
Binary files a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/__pycache__/utils.cpython-39.pyc and /dev/null differ
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/alignment.py b/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/alignment.py
deleted file mode 100644
index 64692a3490c7e8e29e399bd174c5906716b3e6fc..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/alignment.py
+++ /dev/null
@@ -1,608 +0,0 @@
-from itertools import product as product
-from math import ceil
-
-import numpy as np
-import torch
-import torch.backends.cudnn as cudnn
-import torch.nn as nn
-import torch.nn.functional as F
-import torchvision.models._utils as _utils
-
-
-def conv_bn(inp, oup, stride=1, leaky=0):
- return nn.Sequential(
- nn.Conv2d(inp, oup, 3, stride, 1, bias=False),
- nn.BatchNorm2d(oup),
- nn.LeakyReLU(negative_slope=leaky, inplace=True),
- )
-
-
-def conv_bn_no_relu(inp, oup, stride):
- return nn.Sequential(
- nn.Conv2d(inp, oup, 3, stride, 1, bias=False), nn.BatchNorm2d(oup),
- )
-
-
-def conv_bn1X1(inp, oup, stride, leaky=0):
- return nn.Sequential(
- nn.Conv2d(inp, oup, 1, stride, padding=0, bias=False),
- nn.BatchNorm2d(oup),
- nn.LeakyReLU(negative_slope=leaky, inplace=True),
- )
-
-
-def conv_dw(inp, oup, stride, leaky=0.1):
- return nn.Sequential(
- nn.Conv2d(inp, inp, 3, stride, 1, groups=inp, bias=False),
- nn.BatchNorm2d(inp),
- nn.LeakyReLU(negative_slope=leaky, inplace=True),
- nn.Conv2d(inp, oup, 1, 1, 0, bias=False),
- nn.BatchNorm2d(oup),
- nn.LeakyReLU(negative_slope=leaky, inplace=True),
- )
-
-
-class SSH(nn.Module):
- def __init__(self, in_channel, out_channel):
- super(SSH, self).__init__()
- assert out_channel % 4 == 0
- leaky = 0
- if out_channel <= 64:
- leaky = 0.1
- self.conv3X3 = conv_bn_no_relu(in_channel, out_channel // 2, stride=1)
-
- self.conv5X5_1 = conv_bn(in_channel, out_channel // 4, stride=1, leaky=leaky)
- self.conv5X5_2 = conv_bn_no_relu(out_channel // 4, out_channel // 4, stride=1)
-
- self.conv7X7_2 = conv_bn(
- out_channel // 4, out_channel // 4, stride=1, leaky=leaky
- )
- self.conv7x7_3 = conv_bn_no_relu(out_channel // 4, out_channel // 4, stride=1)
-
- def forward(self, input):
- conv3X3 = self.conv3X3(input)
-
- conv5X5_1 = self.conv5X5_1(input)
- conv5X5 = self.conv5X5_2(conv5X5_1)
-
- conv7X7_2 = self.conv7X7_2(conv5X5_1)
- conv7X7 = self.conv7x7_3(conv7X7_2)
-
- out = torch.cat([conv3X3, conv5X5, conv7X7], dim=1)
- out = F.relu(out)
- return out
-
-
-class FPN(nn.Module):
- def __init__(self, in_channels_list, out_channels):
- super(FPN, self).__init__()
- leaky = 0
- if out_channels <= 64:
- leaky = 0.1
- self.output1 = conv_bn1X1(
- in_channels_list[0], out_channels, stride=1, leaky=leaky
- )
- self.output2 = conv_bn1X1(
- in_channels_list[1], out_channels, stride=1, leaky=leaky
- )
- self.output3 = conv_bn1X1(
- in_channels_list[2], out_channels, stride=1, leaky=leaky
- )
-
- self.merge1 = conv_bn(out_channels, out_channels, leaky=leaky)
- self.merge2 = conv_bn(out_channels, out_channels, leaky=leaky)
-
- def forward(self, input):
- # names = list(input.keys())
- input = list(input.values())
-
- output1 = self.output1(input[0])
- output2 = self.output2(input[1])
- output3 = self.output3(input[2])
-
- up3 = F.interpolate(
- output3, size=[output2.size(2), output2.size(3)], mode="nearest"
- )
- output2 = output2 + up3
- output2 = self.merge2(output2)
-
- up2 = F.interpolate(
- output2, size=[output1.size(2), output1.size(3)], mode="nearest"
- )
- output1 = output1 + up2
- output1 = self.merge1(output1)
-
- out = [output1, output2, output3]
- return out
-
-
-class MobileNetV1(nn.Module):
- def __init__(self):
- super(MobileNetV1, self).__init__()
- self.stage1 = nn.Sequential(
- conv_bn(3, 8, 2, leaky=0.1), # 3
- conv_dw(8, 16, 1), # 7
- conv_dw(16, 32, 2), # 11
- conv_dw(32, 32, 1), # 19
- conv_dw(32, 64, 2), # 27
- conv_dw(64, 64, 1), # 43
- )
- self.stage2 = nn.Sequential(
- conv_dw(64, 128, 2), # 43 + 16 = 59
- conv_dw(128, 128, 1), # 59 + 32 = 91
- conv_dw(128, 128, 1), # 91 + 32 = 123
- conv_dw(128, 128, 1), # 123 + 32 = 155
- conv_dw(128, 128, 1), # 155 + 32 = 187
- conv_dw(128, 128, 1), # 187 + 32 = 219
- )
- self.stage3 = nn.Sequential(
- conv_dw(128, 256, 2), # 219 +3 2 = 241
- conv_dw(256, 256, 1), # 241 + 64 = 301
- )
- self.avg = nn.AdaptiveAvgPool2d((1, 1))
- self.fc = nn.Linear(256, 1000)
-
- def forward(self, x):
- x = self.stage1(x)
- x = self.stage2(x)
- x = self.stage3(x)
- x = self.avg(x)
- # x = self.model(x)
- x = x.view(-1, 256)
- x = self.fc(x)
- return x
-
-
-class ClassHead(nn.Module):
- def __init__(self, inchannels=512, num_anchors=3):
- super(ClassHead, self).__init__()
- self.num_anchors = num_anchors
- self.conv1x1 = nn.Conv2d(
- inchannels, self.num_anchors * 2, kernel_size=(1, 1), stride=1, padding=0
- )
-
- def forward(self, x):
- out = self.conv1x1(x)
- out = out.permute(0, 2, 3, 1).contiguous()
-
- return out.view(out.shape[0], -1, 2)
-
-
-class BboxHead(nn.Module):
- def __init__(self, inchannels=512, num_anchors=3):
- super(BboxHead, self).__init__()
- self.conv1x1 = nn.Conv2d(
- inchannels, num_anchors * 4, kernel_size=(1, 1), stride=1, padding=0
- )
-
- def forward(self, x):
- out = self.conv1x1(x)
- out = out.permute(0, 2, 3, 1).contiguous()
-
- return out.view(out.shape[0], -1, 4)
-
-
-class LandmarkHead(nn.Module):
- def __init__(self, inchannels=512, num_anchors=3):
- super(LandmarkHead, self).__init__()
- self.conv1x1 = nn.Conv2d(
- inchannels, num_anchors * 10, kernel_size=(1, 1), stride=1, padding=0
- )
-
- def forward(self, x):
- out = self.conv1x1(x)
- out = out.permute(0, 2, 3, 1).contiguous()
-
- return out.view(out.shape[0], -1, 10)
-
-
-class RetinaFace(nn.Module):
- def __init__(self, cfg=None, phase="train"):
- """
- :param cfg: Network related settings.
- :param phase: train or test.
- """
- super(RetinaFace, self).__init__()
- self.phase = phase
- backbone = None
- if cfg["name"] == "mobilenet0.25":
- backbone = MobileNetV1()
- elif cfg["name"] == "Resnet50":
- import torchvision.models as models
-
- backbone = models.resnet50(pretrained=cfg["pretrain"])
-
- self.body = _utils.IntermediateLayerGetter(backbone, cfg["return_layers"])
- in_channels_stage2 = cfg["in_channel"]
- in_channels_list = [
- in_channels_stage2 * 2,
- in_channels_stage2 * 4,
- in_channels_stage2 * 8,
- ]
- out_channels = cfg["out_channel"]
- self.fpn = FPN(in_channels_list, out_channels)
- self.ssh1 = SSH(out_channels, out_channels)
- self.ssh2 = SSH(out_channels, out_channels)
- self.ssh3 = SSH(out_channels, out_channels)
-
- self.ClassHead = self._make_class_head(fpn_num=3, inchannels=cfg["out_channel"])
- self.BboxHead = self._make_bbox_head(fpn_num=3, inchannels=cfg["out_channel"])
- self.LandmarkHead = self._make_landmark_head(
- fpn_num=3, inchannels=cfg["out_channel"]
- )
-
- def _make_class_head(self, fpn_num=3, inchannels=64, anchor_num=2):
- classhead = nn.ModuleList()
- for i in range(fpn_num):
- classhead.append(ClassHead(inchannels, anchor_num))
- return classhead
-
- def _make_bbox_head(self, fpn_num=3, inchannels=64, anchor_num=2):
- bboxhead = nn.ModuleList()
- for i in range(fpn_num):
- bboxhead.append(BboxHead(inchannels, anchor_num))
- return bboxhead
-
- def _make_landmark_head(self, fpn_num=3, inchannels=64, anchor_num=2):
- landmarkhead = nn.ModuleList()
- for i in range(fpn_num):
- landmarkhead.append(LandmarkHead(inchannels, anchor_num))
- return landmarkhead
-
- def forward(self, inputs):
- out = self.body(inputs)
-
- # FPN
- fpn = self.fpn(out)
-
- # SSH
- feature1 = self.ssh1(fpn[0])
- feature2 = self.ssh2(fpn[1])
- feature3 = self.ssh3(fpn[2])
- features = [feature1, feature2, feature3]
-
- bbox_regressions = torch.cat(
- [self.BboxHead[i](feature) for i, feature in enumerate(features)], dim=1
- )
- classifications = torch.cat(
- [self.ClassHead[i](feature) for i, feature in enumerate(features)], dim=1
- )
- ldm_regressions = torch.cat(
- [self.LandmarkHead[i](feature) for i, feature in enumerate(features)], dim=1
- )
-
- if self.phase == "train":
- output = (bbox_regressions, classifications, ldm_regressions)
- else:
- output = (
- bbox_regressions,
- F.softmax(classifications, dim=-1),
- ldm_regressions,
- )
- return output
-
-
-# Adapted from https://github.com/Hakuyume/chainer-ssd
-def decode(loc, priors, variances):
- boxes = torch.cat(
- (
- priors[:, :2] + loc[:, :2] * variances[0] * priors[:, 2:],
- priors[:, 2:] * torch.exp(loc[:, 2:] * variances[1]),
- ),
- 1,
- )
- boxes[:, :2] -= boxes[:, 2:] / 2
- boxes[:, 2:] += boxes[:, :2]
- return boxes
-
-
-def decode_landm(pre, priors, variances):
- landms = torch.cat(
- (
- priors[:, :2] + pre[:, :2] * variances[0] * priors[:, 2:],
- priors[:, :2] + pre[:, 2:4] * variances[0] * priors[:, 2:],
- priors[:, :2] + pre[:, 4:6] * variances[0] * priors[:, 2:],
- priors[:, :2] + pre[:, 6:8] * variances[0] * priors[:, 2:],
- priors[:, :2] + pre[:, 8:10] * variances[0] * priors[:, 2:],
- ),
- dim=1,
- )
- return landms
-
-
-def py_cpu_nms(dets, thresh):
- """Pure Python NMS baseline."""
- x1 = dets[:, 0]
- y1 = dets[:, 1]
- x2 = dets[:, 2]
- y2 = dets[:, 3]
- scores = dets[:, 4]
-
- areas = (x2 - x1 + 1) * (y2 - y1 + 1)
- order = scores.argsort()[::-1]
-
- keep = []
- while order.size > 0:
- i = order[0]
- keep.append(i)
- xx1 = np.maximum(x1[i], x1[order[1:]])
- yy1 = np.maximum(y1[i], y1[order[1:]])
- xx2 = np.minimum(x2[i], x2[order[1:]])
- yy2 = np.minimum(y2[i], y2[order[1:]])
-
- w = np.maximum(0.0, xx2 - xx1 + 1)
- h = np.maximum(0.0, yy2 - yy1 + 1)
- inter = w * h
- ovr = inter / (areas[i] + areas[order[1:]] - inter)
-
- inds = np.where(ovr <= thresh)[0]
- order = order[inds + 1]
-
- return keep
-
-
-class PriorBox(object):
- def __init__(self, cfg, image_size=None, phase="train"):
- super(PriorBox, self).__init__()
- self.min_sizes = cfg["min_sizes"]
- self.steps = cfg["steps"]
- self.clip = cfg["clip"]
- self.image_size = image_size
- self.feature_maps = [
- [ceil(self.image_size[0] / step), ceil(self.image_size[1] / step)]
- for step in self.steps
- ]
- self.name = "s"
-
- def forward(self):
- anchors = []
- for k, f in enumerate(self.feature_maps):
- min_sizes = self.min_sizes[k]
- for i, j in product(range(f[0]), range(f[1])):
- for min_size in min_sizes:
- s_kx = min_size / self.image_size[1]
- s_ky = min_size / self.image_size[0]
- dense_cx = [
- x * self.steps[k] / self.image_size[1] for x in [j + 0.5]
- ]
- dense_cy = [
- y * self.steps[k] / self.image_size[0] for y in [i + 0.5]
- ]
- for cy, cx in product(dense_cy, dense_cx):
- anchors += [cx, cy, s_kx, s_ky]
-
- # back to torch land
- output = torch.Tensor(anchors).view(-1, 4)
- if self.clip:
- output.clamp_(max=1, min=0)
- return output
-
-
-cfg_mnet = {
- "name": "mobilenet0.25",
- "min_sizes": [[16, 32], [64, 128], [256, 512]],
- "steps": [8, 16, 32],
- "variance": [0.1, 0.2],
- "clip": False,
- "loc_weight": 2.0,
- "gpu_train": True,
- "batch_size": 32,
- "ngpu": 1,
- "epoch": 250,
- "decay1": 190,
- "decay2": 220,
- "image_size": 640,
- "pretrain": True,
- "return_layers": {"stage1": 1, "stage2": 2, "stage3": 3},
- "in_channel": 32,
- "out_channel": 64,
-}
-
-cfg_re50 = {
- "name": "Resnet50",
- "min_sizes": [[16, 32], [64, 128], [256, 512]],
- "steps": [8, 16, 32],
- "variance": [0.1, 0.2],
- "clip": False,
- "loc_weight": 2.0,
- "gpu_train": True,
- "batch_size": 24,
- "ngpu": 4,
- "epoch": 100,
- "decay1": 70,
- "decay2": 90,
- "image_size": 840,
- "pretrain": False,
- "return_layers": {"layer2": 1, "layer3": 2, "layer4": 3},
- "in_channel": 256,
- "out_channel": 256,
-}
-
-
-def check_keys(model, pretrained_state_dict):
- ckpt_keys = set(pretrained_state_dict.keys())
- model_keys = set(model.state_dict().keys())
- used_pretrained_keys = model_keys & ckpt_keys
- assert len(used_pretrained_keys) > 0, "load NONE from pretrained checkpoint"
- return True
-
-
-def remove_prefix(state_dict, prefix):
- """ Old style model is stored with all names of parameters sharing common prefix 'module.' """
- f = lambda x: x.split(prefix, 1)[-1] if x.startswith(prefix) else x
- return {f(key): value for key, value in state_dict.items()}
-
-
-def load_model(model, pretrained_path, load_to_cpu):
- if load_to_cpu:
- if pretrained_path is None:
- url = "https://github.com/yinglinzheng/face_weights/releases/download/v1/mobilenet0.25_Final.pth"
- pretrained_dict = torch.utils.model_zoo.load_url(url)
- else:
- pretrained_dict = torch.load(
- pretrained_path, map_location=lambda storage, loc: storage
- )
- else:
- device = torch.cuda.current_device()
- pretrained_dict = torch.load(
- pretrained_path, map_location=lambda storage, loc: storage.cuda(device)
- )
- if "state_dict" in pretrained_dict.keys():
- pretrained_dict = remove_prefix(pretrained_dict["state_dict"], "module.")
- else:
- pretrained_dict = remove_prefix(pretrained_dict, "module.")
- check_keys(model, pretrained_dict)
- model.load_state_dict(pretrained_dict, strict=False)
- return model
-
-
-def load_net(model_path, device, network="mobilenet"):
- if network == "mobilenet":
- cfg = cfg_mnet
- elif network == "resnet50":
- cfg = cfg_re50
- # net and model
- net = RetinaFace(cfg=cfg, phase="test")
- net = load_model(net, model_path, True)
- net.eval()
- cudnn.benchmark = True
- net = net.to(device)
- return net
-
-
-def parse_det(det):
- landmarks = det[5:].reshape(5, 2)
- box = det[:4]
- score = det[4]
- return box, landmarks, score
-
-
-def post_process(
- loc,
- conf,
- landms,
- prior_data,
- cfg,
- scale,
- scale1,
- resize,
- confidence_threshold,
- top_k,
- nms_threshold,
- keep_top_k,
-):
- boxes = decode(loc, prior_data, cfg["variance"])
- boxes = boxes * scale / resize
- boxes = boxes.cpu().numpy()
- scores = conf.cpu().numpy()[:, 1]
- landms_copy = decode_landm(landms, prior_data, cfg["variance"])
-
- landms_copy = landms_copy * scale1 / resize
- landms_copy = landms_copy.cpu().numpy()
-
- # ignore low scores
- inds = np.where(scores > confidence_threshold)[0]
- boxes = boxes[inds]
- landms_copy = landms_copy[inds]
- scores = scores[inds]
-
- # keep top-K before NMS
- order = scores.argsort()[::-1][:top_k]
- boxes = boxes[order]
- landms_copy = landms_copy[order]
- scores = scores[order]
-
- # do NMS
- dets = np.hstack((boxes, scores[:, np.newaxis])).astype(np.float32, copy=False)
- keep = py_cpu_nms(dets, nms_threshold)
- # keep = nms(dets, args.nms_threshold,force_cpu=args.cpu)
- dets = dets[keep, :]
- landms_copy = landms_copy[keep]
-
- # keep top-K faster NMS
- dets = dets[:keep_top_k, :]
- landms_copy = landms_copy[:keep_top_k, :]
-
- dets = np.concatenate((dets, landms_copy), axis=1)
- # show image
- dets = sorted(dets, key=lambda x: x[4], reverse=True)
- dets = [parse_det(x) for x in dets]
-
- return dets
-
-
-def batch_detect(net, images, device, is_tensor=False, normalized=False):
- with torch.no_grad():
- confidence_threshold = 0.02
- cfg = cfg_mnet
- top_k = 5000
- nms_threshold = 0.4
- keep_top_k = 750
- resize = 1
- if not is_tensor:
- try:
- img = np.float32(images)
- except ValueError:
- raise NotImplementedError("Input images must of same size")
- img = torch.from_numpy(img)
- else:
- img = images.float()
- img = img.to(device)
- mean = (
- torch.as_tensor([104, 117, 123], dtype=img.dtype, device=img.device)
- .unsqueeze(0)
- .unsqueeze(0)
- .unsqueeze(0)
- )
- img -= mean
- img = img.permute(0, 3, 1, 2)
- (batch_size, _, im_height, im_width,) = img.shape
- scale = torch.as_tensor(
- [im_width, im_height, im_width, im_height],
- dtype=img.dtype,
- device=img.device,
- )
- scale = scale.to(device)
-
- loc, conf, landms = net(img) # forward pass
-
- priorbox = PriorBox(cfg, image_size=(im_height, im_width))
- priors = priorbox.forward()
- prior_data = priors.to(device)
- scale1 = torch.as_tensor(
- [
- img.shape[3],
- img.shape[2],
- img.shape[3],
- img.shape[2],
- img.shape[3],
- img.shape[2],
- img.shape[3],
- img.shape[2],
- img.shape[3],
- img.shape[2],
- ],
- dtype=img.dtype,
- device=img.device,
- )
- scale1 = scale1.to(device)
-
- all_dets = [
- post_process(
- loc_i,
- conf_i,
- landms_i,
- prior_data,
- cfg,
- scale,
- scale1,
- resize,
- confidence_threshold,
- top_k,
- nms_threshold,
- keep_top_k,
- )
- for loc_i, conf_i, landms_i in zip(loc, conf, landms)
- ]
-
- return all_dets
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/detector.py b/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/detector.py
deleted file mode 100644
index f38050e7a050eeb05323e7d710a63c2bf1a12455..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/detector.py
+++ /dev/null
@@ -1,46 +0,0 @@
-import os
-
-import numpy as np
-import torch
-
-from .alignment import load_net, batch_detect
-
-
-def get_project_dir():
- current_path = os.path.abspath(os.path.join(__file__, "../"))
- return current_path
-
-
-def relative(path):
- path = os.path.join(get_project_dir(), path)
- return os.path.abspath(path)
-
-
-class RetinaFace:
- def __init__(
- self, gpu_id=-1, model_path=None, network="mobilenet",
- ):
- self.gpu_id = gpu_id
- self.device = (
- torch.device("cpu") if gpu_id == -1 else torch.device("cuda", gpu_id)
- )
- self.model = load_net(model_path, self.device, network)
-
- def detect(self, images):
- if isinstance(images, np.ndarray):
- if len(images.shape) == 3:
- return batch_detect(self.model, [images], self.device)[0]
- elif len(images.shape) == 4:
- return batch_detect(self.model, images, self.device)
- elif isinstance(images, list):
- return batch_detect(self.model, np.array(images), self.device)
- elif isinstance(images, torch.Tensor):
- if len(images.shape) == 3:
- return batch_detect(self.model, images.unsqueeze(0), self.device)[0]
- elif len(images.shape) == 4:
- return batch_detect(self.model, images, self.device)
- else:
- raise NotImplementedError()
-
- def __call__(self, images):
- return self.detect(images)
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/utils.py b/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/utils.py
deleted file mode 100644
index 512d824f02645ae9473331a553b18850e642b08f..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/test_tools/ct/detection/utils.py
+++ /dev/null
@@ -1,146 +0,0 @@
-import cv2
-from test_tools.utils import flatten
-import numpy as np
-
-
-def chunks(l, n, step=None):
- if step is None:
- step = n
- return [l[i : i + n] for i in range(0, len(l), step)]
-
-
-def sample_chunks(l, n, step=None):
- return [l[i : i + n] for i in range(0, len(l), step) if i + n <= len(l)]
-
-
-def grab_all_frames(path, max_size, cvt=False):
- capture = cv2.VideoCapture(path)
- ret = True
- frames = []
- while ret:
- ret, frame = capture.read()
- if ret:
- if cvt:
- frame = frame[..., ::-1]
- frames.append(frame)
- if len(frames) == max_size:
- break
- capture.release()
- return frames
-
-
-def get_clips_uniform(path, count, clip_size):
- capture = cv2.VideoCapture(path)
- n_frames = int(capture.get(cv2.CAP_PROP_FRAME_COUNT))
- max_clip_available = n_frames + 1 - clip_size
- if count > max_clip_available:
- count = max_clip_available
- final_start = max_clip_available - 1
- start_indices = np.linspace(0, final_start, count, endpoint=True, dtype=np.int)
- all_clip_idx = [list(range(start, start + clip_size)) for start in start_indices]
- valid = set(flatten(all_clip_idx))
- max_idx = max(valid)
-
- frames = {}
- for idx in range(max_idx + 1):
- # Get the next frame, but don't decode if we're not using it.
- ret = capture.grab()
- if not ret:
- continue
-
- if idx in valid:
- ret, frame = capture.retrieve()
- if not ret or frame is None:
- continue
- else:
- # frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
- frames[idx] = frame
-
- capture.release()
- clips = []
- for clip_idx in all_clip_idx:
- clip = []
- flag = True
- for idx in clip_idx:
- if idx not in frames:
- flag = False
- break
- clip.append(frames[idx])
- if flag:
- clips.append(clip)
- return clips
-
-
-def get_valid_faces(detect_results, max_count=10, thres=0.5, at_least=False):
- new_results = []
- for i, faces in enumerate(detect_results):
- if len(faces) > max_count:
- faces = faces[:max_count]
- l = []
- for j, face in enumerate(faces):
- if face[-1] < thres and not (j == 0 and at_least):
- continue
- box, lm, score = face
- box = box.astype(np.float32)
- lm = lm.astype(np.float32)
- l.append((box, lm, score))
- new_results.append(l)
- return new_results
-
-
-def scale_box(box, scale_h, scale_w, h, w):
- x1, y1, x2, y2 = box.astype(np.int32)
- center_x = (x1 + x2) // 2
- center_y = (y1 + y2) // 2
- box_h = int((y2 - y1) * scale_h)
- box_w = int((x2 - x1) * scale_w)
- new_x1 = center_x - box_w // 2
- new_x2 = new_x1 + box_w
- new_y1 = center_y - box_h // 2
- new_y2 = new_y1 + box_h
- new_x1 = max(new_x1, 0)
- new_y1 = max(new_y1, 0)
- new_y2 = min(new_y2, h)
- new_x2 = min(new_x2, w)
- return new_x1, new_y1, new_x2, new_y2
-
-
-def get_bbox(detect_res):
- tmp_detect_res = get_valid_faces(detect_res, max_count=4, thres=0.5)
- all_face_bboxs = []
- for faces in tmp_detect_res:
- all_face_bboxs.extend([face[0] for face in faces])
- all_face_bboxs = np.array(all_face_bboxs).astype(np.int)
- x1 = all_face_bboxs[:, 0].min()
- x2 = all_face_bboxs[:, 2].max()
- y1 = all_face_bboxs[:, 1].min()
- y2 = all_face_bboxs[:, 3].max()
-
- return x1, y1, x2, y2
-
-
-def delta_detect_res(detect_res, x1, y1):
- diff = np.array([[x1, y1]])
- new_detect_res = []
- for faces in detect_res:
- f = []
- for face in faces:
- box, lm, score = face
- box = box.astype(np.float)
- box[[0, 2]] -= x1
- box[[1, 3]] -= y1
- lm = lm.astype(np.float) - diff
- f.append((box, lm, score))
- new_detect_res.append(f)
- return new_detect_res
-
-
-def pre_crop(clips, detect_res):
- box = np.array(get_bbox(detect_res))
- w = box[2] - box[0]
- h = box[3] - box[1]
- x1, y1, x2, y2 = scale_box(
- box, 1.5, 1.2 if w > 2 * h else 1.5, clips[0].shape[0], clips[0].shape[1]
- )
- clips = np.array(clips)
- return clips[:, y1:y2, x1:x2], delta_detect_res(detect_res, x1, y1)
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/face_alignment/__init__.py b/video/pwtf-dvd/model_code/inference/test_tools/ct/face_alignment/__init__.py
deleted file mode 100644
index b0545f2f8c2aa872a127bf4e815c1e614a971723..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/test_tools/ct/face_alignment/__init__.py
+++ /dev/null
@@ -1 +0,0 @@
-from .predictor import LandmarkPredictor
\ No newline at end of file
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/face_alignment/basenet.py b/video/pwtf-dvd/model_code/inference/test_tools/ct/face_alignment/basenet.py
deleted file mode 100644
index 699b163e0584d56cc6d8f916b171f7ba86326efa..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/test_tools/ct/face_alignment/basenet.py
+++ /dev/null
@@ -1,107 +0,0 @@
-# Backbone networks used for face landmark detection
-# Cunjian Chen (cunjian@msu.edu)
-
-import torch.nn as nn
-import torchvision.models as models
-
-
-class ConvBlock(nn.Module):
- def __init__(self, inp, oup, k, s, p, dw=False, linear=False):
- super(ConvBlock, self).__init__()
- self.linear = linear
- if dw:
- self.conv = nn.Conv2d(inp, oup, k, s, p, groups=inp, bias=False)
- else:
- self.conv = nn.Conv2d(inp, oup, k, s, p, bias=False)
- self.bn = nn.BatchNorm2d(oup)
- if not linear:
- self.prelu = nn.PReLU(oup)
-
- def forward(self, x):
- x = self.conv(x)
- x = self.bn(x)
- if self.linear:
- return x
- else:
- return self.prelu(x)
-
-
-# SE module
-# https://github.com/wujiyang/Face_Pytorch/blob/master/backbone/cbam.py
-class SEModule(nn.Module):
- """Squeeze and Excitation Module"""
-
- def __init__(self, channels, reduction):
- super(SEModule, self).__init__()
- self.avg_pool = nn.AdaptiveAvgPool2d(1)
- self.fc1 = nn.Conv2d(
- channels, channels // reduction, kernel_size=1, padding=0, bias=False
- )
- self.relu = nn.ReLU(inplace=True)
- self.fc2 = nn.Conv2d(
- channels // reduction, channels, kernel_size=1, padding=0, bias=False
- )
- self.sigmoid = nn.Sigmoid()
-
- def forward(self, x):
- input = x
- x = self.avg_pool(x)
- x = self.fc1(x)
- x = self.relu(x)
- x = self.fc2(x)
- x = self.sigmoid(x)
-
- return input * x
-
-
-# USE global depthwise convolution layer. Compatible with MobileNetV2 (224×224), MobileNetV2_ExternalData (224×224)
-class MobileNet_GDConv(nn.Module):
- def __init__(self, num_classes):
- super(MobileNet_GDConv, self).__init__()
- self.pretrain_net = models.mobilenet_v2(pretrained=False)
- self.base_net = nn.Sequential(*list(self.pretrain_net.children())[:-1])
- self.linear7 = ConvBlock(1280, 1280, (7, 7), 1, 0, dw=True, linear=True)
- self.linear1 = ConvBlock(1280, num_classes, 1, 1, 0, linear=True)
-
- def forward(self, x):
- x = self.base_net(x)
- x = self.linear7(x)
- x = self.linear1(x)
- x = x.view(x.size(0), -1)
- return x
-
-
-# USE global depthwise convolution layer. Compatible with MobileNetV2 (56×56)
-class MobileNet_GDConv_56(nn.Module):
- def __init__(self, num_classes):
- super(MobileNet_GDConv_56, self).__init__()
- self.pretrain_net = models.mobilenet_v2(pretrained=False)
- self.base_net = nn.Sequential(*list(self.pretrain_net.children())[:-1])
- self.linear7 = ConvBlock(1280, 1280, (2, 2), 1, 0, dw=True, linear=True)
- self.linear1 = ConvBlock(1280, num_classes, 1, 1, 0, linear=True)
-
- def forward(self, x):
- x = self.base_net(x)
- x = self.linear7(x)
- x = self.linear1(x)
- x = x.view(x.size(0), -1)
- return x
-
-
-# MobileNetV2 with SE; Compatible with MobileNetV2_SE (224×224) and MobileNetV2_SE_RE (224×224)
-class MobileNet_GDConv_SE(nn.Module):
- def __init__(self, num_classes):
- super(MobileNet_GDConv_SE, self).__init__()
- self.pretrain_net = models.mobilenet_v2(pretrained=True)
- self.base_net = nn.Sequential(*list(self.pretrain_net.children())[:-1])
- self.linear7 = ConvBlock(1280, 1280, (7, 7), 1, 0, dw=True, linear=True)
- self.linear1 = ConvBlock(1280, num_classes, 1, 1, 0, linear=True)
- self.attention = SEModule(1280, 8)
-
- def forward(self, x):
- x = self.base_net(x)
- x = self.attention(x)
- x = self.linear7(x)
- x = self.linear1(x)
- x = x.view(x.size(0), -1)
- return x
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/face_alignment/predictor.py b/video/pwtf-dvd/model_code/inference/test_tools/ct/face_alignment/predictor.py
deleted file mode 100644
index b210b673ab3af90d805e7fff9b0b66b00826fd1f..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/test_tools/ct/face_alignment/predictor.py
+++ /dev/null
@@ -1,143 +0,0 @@
-# Face alignment demo
-# Uses MTCNN as face detector
-# Cunjian Chen (ccunjian@gmail.com)
-import torch
-import cv2
-import numpy as np
-from torch.utils.data import DataLoader
-from .basenet import MobileNet_GDConv
-
-
-def get_device(gpu_id):
- if gpu_id > -1:
- return torch.device(f"cuda:{str(gpu_id)}")
- else:
- return torch.device("cpu")
-
-
-def load_model(file):
- model = MobileNet_GDConv(136)
- if file is not None:
- model.load_state_dict(torch.load(file, map_location="cpu"))
- else:
- url = "https://github.com/yinglinzheng/face_weights/releases/download/v1/mobilenet_224_model_best_gdconv_external.pth"
- model.load_state_dict(torch.utils.model_zoo.load_url(url))
- return model
-
-
-# landmark of (5L, 2L) from [0,1] to real range
-def reproject(bbox, landmark):
- landmark_ = landmark.clone()
- x1, y1, x2, y2 = bbox
- w = x2 - x1
- h = y2 - y1
- landmark_[:, 0] *= w
- landmark_[:, 0] += x1
- landmark_[:, 1] *= h
- landmark_[:, 1] += y1
- return landmark_
-
-
-def prepare_feed(img, face):
- height, width, _ = img.shape
- mean = np.asarray([0.485, 0.456, 0.406])
- std = np.asarray([0.229, 0.224, 0.225])
- out_size = 224
- x1, y1, x2, y2 = face[:4]
-
- w = x2 - x1 + 1
- h = y2 - y1 + 1
- size = int(min([w, h]) * 1.2)
- cx = x1 + w // 2
- cy = y1 + h // 2
- x1 = cx - size // 2
- x2 = x1 + size
- y1 = cy - size // 2
- y2 = y1 + size
-
- dx = max(0, -x1)
- dy = max(0, -y1)
- x1 = max(0, x1)
- y1 = max(0, y1)
-
- edx = max(0, x2 - width)
- edy = max(0, y2 - height)
- x2 = min(width, x2)
- y2 = min(height, y2)
- new_bbox = torch.Tensor([x1, y1, x2, y2]).int()
- x1, y1, x2, y2 = new_bbox
- cropped = img[y1:y2, x1:x2]
- if dx > 0 or dy > 0 or edx > 0 or edy > 0:
- cropped = cv2.copyMakeBorder(
- cropped, int(dy), int(edy), int(dx), int(edx), cv2.BORDER_CONSTANT, 0
- )
- cropped_face = cv2.resize(cropped, (out_size, out_size))
-
- if cropped_face.shape[0] <= 0 or cropped_face.shape[1] <= 0:
- return None
- test_face = cropped_face.copy()
- test_face = test_face / 255.0
- test_face = (test_face - mean) / std
- test_face = test_face.transpose((2, 0, 1))
- data = torch.from_numpy(test_face).float()
- return dict(data=data, bbox=new_bbox)
-
-
-@torch.no_grad()
-def single_predict(model, feed, device):
- landmark = model(feed["data"].unsqueeze(0).to(device)).cpu()
- landmark = landmark.reshape(-1, 2)
- landmark = reproject(feed["bbox"], landmark)
- return landmark.numpy()
-
-
-@torch.no_grad()
-def batch_predict(model, feeds, device):
- if not isinstance(feeds, list):
- feeds = [feeds]
- # loader = DataLoader(FeedDataset(feeds), batch_size=50, shuffle=False)
- data = []
- for feed in feeds:
- data.append(feed["data"].unsqueeze(0))
- data = torch.cat(data, 0).to(device)
- results = []
-
- landmarks = model(data).cpu()
- for landmark, feed in zip(landmarks, feeds):
- landmark = landmark.reshape(-1, 2)
- landmark = reproject(feed["bbox"], landmark)
- results.append(landmark.numpy())
- return results
-
-
-@torch.no_grad()
-def batch_predict2(model, feeds, device, batch_size=None):
- if not isinstance(feeds, list):
- feeds = [feeds]
- if batch_size is None:
- batch_size = len(feeds)
- loader = DataLoader(feeds, batch_size=len(feeds), shuffle=False)
- results = []
- for feed in loader:
- landmarks = model(feed["data"].to(device)).cpu()
- for landmark, bbox in zip(landmarks, feed["bbox"]):
- landmark = landmark.reshape(-1, 2)
- landmark = reproject(bbox, landmark)
- results.append(landmark.numpy())
- return results
-
-
-class LandmarkPredictor:
- def __init__(self, gpu_id=0, file=None):
- self.device = get_device(gpu_id)
- self.model = load_model(file).to(self.device).eval()
-
- def __call__(self, feeds):
- results = batch_predict2(self.model, feeds, self.device)
- if not isinstance(feeds, list):
- results = results[0]
- return results
-
- @staticmethod
- def prepare_feed(img, face):
- return prepare_feed(img, face)
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/face_alignment/utils.py b/video/pwtf-dvd/model_code/inference/test_tools/ct/face_alignment/utils.py
deleted file mode 100644
index e871b155963b0d1ce6891589d32615c5fc97bb66..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/test_tools/ct/face_alignment/utils.py
+++ /dev/null
@@ -1,17 +0,0 @@
-import cv2
-
-
-def drawLandmark_multiple(img, bbox, landmark):
- """
- Input:
- - img: gray or RGB
- - bbox: type of BBox
- - landmark: reproject landmark of (5L, 2L)
- Output:
- - img marked with landmark and bbox
- """
- x1, y1, x2, y2 = bbox
- cv2.rectangle(img, (x1, y1), (x2, y2), (0, 0, 255), 2)
- for x, y in landmark:
- cv2.circle(img, (int(x), int(y)), 2, (0, 255, 0), -1)
- return img
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/operations.py b/video/pwtf-dvd/model_code/inference/test_tools/ct/operations.py
deleted file mode 100644
index 68375fcc4efd768abcfddf39521af21d574c4aac..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/test_tools/ct/operations.py
+++ /dev/null
@@ -1,79 +0,0 @@
-import os
-
-import os
-import cv2
-import numpy as np
-from .tracking.sort import iou
-
-
-def face_iou(f1, f2):
- return iou(f1[0], f2[0])
-
-
-def simple_tracking(batch_landmarks, index=0, thres=0.5):
- track = []
-
- for i, faces in enumerate(batch_landmarks):
- if i == 0:
- if len(faces) <= index or faces[index][-1] < 0.8:
- return None
- if index != 0:
- for idx in range(index):
- if face_iou(faces[idx], faces[index]) > thres:
- return None
- track.append(faces[index])
- else:
- last = track[i - 1]
- if len(faces) == 0:
- return None
- sorted_faces = sorted(faces, key=lambda x: face_iou(x, last), reverse=True)
- if face_iou(sorted_faces[0], last) < thres:
- return None
- track.append(sorted_faces[0])
- return track
-
-
-def multiple_tracking(batch_landmarks):
- tracks = []
- for i in range(len(batch_landmarks[0])):
- track = simple_tracking(batch_landmarks, index=i)
- if track is None:
- continue
- tracks.append(track)
- return tracks
-
-def find_longest(detect_res):
- fc = len(detect_res)
- tuples = []
- start = 0
- end = 0
- previous_count = -1
- all_tracks = []
- # start 取得到,end 取不到
- while start < (fc - 1):
- for end in range(start + 2, fc + 1):
- tracks = multiple_tracking(detect_res[start:end])
- if (len(tracks) != previous_count and previous_count != -1) or len(
- tracks
- ) == 0:
- break
- previous_count = len(tracks)
- if end - start > 2:
- if end != fc:
- un_reach_end = end - 1
- else:
- un_reach_end = end
- sub_tracks = multiple_tracking(detect_res[start:un_reach_end])
- if end == fc and len(sub_tracks) == 0:
- un_reach_end = end - 1
- sub_tracks = multiple_tracking(detect_res[start:un_reach_end])
- if len(sub_tracks) > 0:
- tpl = (start, un_reach_end)
- tuples.append(tpl)
- all_tracks.append(sub_tracks[0])
- else:
- raise NotImplementedError
- previous_count = -1
- end = un_reach_end
- start = end
- return tuples, all_tracks
\ No newline at end of file
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/tracking/__init__.py b/video/pwtf-dvd/model_code/inference/test_tools/ct/tracking/__init__.py
deleted file mode 100644
index e69de29bb2d1d6434b8b29ae775ad8c2e48c5391..0000000000000000000000000000000000000000
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/tracking/sort.py b/video/pwtf-dvd/model_code/inference/test_tools/ct/tracking/sort.py
deleted file mode 100644
index dc7b0838e7110a2d3521c6d4cfefbb40cc266c23..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/test_tools/ct/tracking/sort.py
+++ /dev/null
@@ -1,285 +0,0 @@
-"""
- SORT: A Simple, Online and Realtime Tracker
- Copyright (C) 2016 Alex Bewley alex@dynamicdetection.com
-
- This program is free software: you can redistribute it and/or modify
- it under the terms of the GNU General Public License as published by
- the Free Software Foundation, either version 3 of the License, or
- (at your option) any later version.
-
- This program is distributed in the hope that it will be useful,
- but WITHOUT ANY WARRANTY; without even the implied warranty of
- MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
- GNU General Public License for more details.
-
- You should have received a copy of the GNU General Public License
- along with this program. If not, see .
-"""
-from __future__ import print_function
-import os.path
-import numpy as np
-import matplotlib.pyplot as plt
-import matplotlib.patches as patches
-from scipy.optimize import linear_sum_assignment
-import glob
-import time
-import argparse
-from filterpy.kalman import KalmanFilter
-
-
-def iou(bb_test, bb_gt):
- """
- Computes IUO between two bboxes in the form [x1,y1,x2,y2]
- """
- xx1 = np.maximum(bb_test[0], bb_gt[0])
- yy1 = np.maximum(bb_test[1], bb_gt[1])
- xx2 = np.minimum(bb_test[2], bb_gt[2])
- yy2 = np.minimum(bb_test[3], bb_gt[3])
- w = np.maximum(0.0, xx2 - xx1)
- h = np.maximum(0.0, yy2 - yy1)
- wh = w * h
- o = wh / (
- (bb_test[2] - bb_test[0]) * (bb_test[3] - bb_test[1])
- + (bb_gt[2] - bb_gt[0]) * (bb_gt[3] - bb_gt[1])
- - wh
- )
- return o
-
-
-def convert_bbox_to_z(bbox):
- """
- Takes a bounding box in the form [x1,y1,x2,y2] and returns z in the form
- [x,y,s,r] where x,y is the centre of the box and s is the scale/area and r is
- the aspect ratio
- """
- w = bbox[2] - bbox[0]
- h = bbox[3] - bbox[1]
- x = bbox[0] + w / 2.0
- y = bbox[1] + h / 2.0
- s = w * h # scale is just area
- r = w / float(h)
- return np.array([x, y, s, r]).reshape((4, 1))
-
-
-def convert_x_to_bbox(x, score=None):
- """
- Takes a bounding box in the centre form [x,y,s,r] and returns it in the form
- [x1,y1,x2,y2] where x1,y1 is the top left and x2,y2 is the bottom right
- """
- w = np.sqrt(x[2] * x[3])
- h = x[2] / w
- if score == None:
- return np.array(
- [x[0] - w / 2.0, x[1] - h / 2.0, x[0] + w / 2.0, x[1] + h / 2.0]
- ).reshape((1, 4))
- else:
- return np.array(
- [x[0] - w / 2.0, x[1] - h / 2.0, x[0] + w / 2.0, x[1] + h / 2.0, score]
- ).reshape((1, 5))
-
-
-class KalmanBoxTracker(object):
- """
- This class represents the internel state of individual tracked objects observed as bbox.
- """
-
- count = 0
-
- def __init__(self, bbox):
- """
- Initialises a tracker using initial bounding box.
- """
- # define constant velocity model
- self.kf = KalmanFilter(dim_x=7, dim_z=4)
- self.kf.F = np.array(
- [
- [1, 0, 0, 0, 1, 0, 0],
- [0, 1, 0, 0, 0, 1, 0],
- [0, 0, 1, 0, 0, 0, 1],
- [0, 0, 0, 1, 0, 0, 0],
- [0, 0, 0, 0, 1, 0, 0],
- [0, 0, 0, 0, 0, 1, 0],
- [0, 0, 0, 0, 0, 0, 1],
- ]
- )
- self.kf.H = np.array(
- [
- [1, 0, 0, 0, 0, 0, 0],
- [0, 1, 0, 0, 0, 0, 0],
- [0, 0, 1, 0, 0, 0, 0],
- [0, 0, 0, 1, 0, 0, 0],
- ]
- )
-
- self.kf.R[2:, 2:] *= 10.0
- self.kf.P[
- 4:, 4:
- ] *= 1000.0 # give high uncertainty to the unobservable initial velocities
- self.kf.P *= 10.0
- self.kf.Q[-1, -1] *= 0.01
- self.kf.Q[4:, 4:] *= 0.01
-
- self.kf.x[:4] = convert_bbox_to_z(bbox)
- self.time_since_update = 0
- self.id = KalmanBoxTracker.count
- KalmanBoxTracker.count += 1
- self.history = []
- self.hits = 0
- self.hit_streak = 0
- self.age = 0
-
- def update(self, bbox):
- """
- Updates the state vector with observed bbox.
- """
- self.time_since_update = 0
- self.history = []
- self.hits += 1
- self.hit_streak += 1
- self.kf.update(convert_bbox_to_z(bbox))
-
- def predict(self):
- """
- Advances the state vector and returns the predicted bounding box estimate.
- """
- if (self.kf.x[6] + self.kf.x[2]) <= 0:
- self.kf.x[6] *= 0.0
- self.kf.predict()
- self.age += 1
- if self.time_since_update > 0:
- self.hit_streak = 0
- self.time_since_update += 1
- self.history.append(convert_x_to_bbox(self.kf.x))
- return self.history[-1]
-
- def get_state(self):
- """
- Returns the current bounding box estimate.
- """
- return convert_x_to_bbox(self.kf.x)
-
-
-def associate_detections_to_trackers(detections, trackers, iou_threshold=0.3):
- """
- Assigns detections to tracked object (both represented as bounding boxes)
-
- Returns 3 lists of matches, unmatched_detections and unmatched_trackers
- """
- if len(trackers) == 0:
- return (
- np.empty((0, 2), dtype=int),
- np.arange(len(detections)),
- np.empty((0, 5), dtype=int),
- )
- iou_matrix = np.zeros((len(detections), len(trackers)), dtype=np.float32)
-
- for d, det in enumerate(detections):
- for t, trk in enumerate(trackers):
- iou_matrix[d, t] = iou(det, trk)
-
- matched_indices = linear_sum_assignment(-iou_matrix)
- matched_indices = np.array(list(zip(*matched_indices)), dtype=np.int)
- matched_indices.shape = (-1, 2)
- # print(matched_indices)
- # print(type(matched_indices))
-
- unmatched_detections = []
- for d, det in enumerate(detections):
- if d not in matched_indices[:, 0]:
- unmatched_detections.append(d)
- unmatched_trackers = []
- for t, trk in enumerate(trackers):
- if t not in matched_indices[:, 1]:
- unmatched_trackers.append(t)
-
- # filter out matched with low IOU
- matches = []
- for m in matched_indices:
- if iou_matrix[m[0], m[1]] < iou_threshold:
- unmatched_detections.append(m[0])
- unmatched_trackers.append(m[1])
- else:
- matches.append(m.reshape(1, 2))
- if len(matches) == 0:
- matches = np.empty((0, 2), dtype=int)
- else:
- matches = np.concatenate(matches, axis=0)
-
- return matches, np.array(unmatched_detections), np.array(unmatched_trackers)
-
-
-class Sort(object):
- def __init__(self, max_age=1, min_hits=3):
- """
- Sets key parameters for SORT
- """
- self.max_age = max_age
- self.min_hits = min_hits
- self.trackers = []
- self.frame_count = 0
-
- def update(self, dets):
- """
- Params:
- dets - a numpy array of detections in the format [[x1,y1,x2,y2,score],[x1,y1,x2,y2,score],...]
- Requires: this method must be called once for each frame even with empty detections.
- Returns the a similar array, where the last column is the object ID.
-
- NOTE: The number of objects returned may differ from the number of detections provided.
- """
- self.frame_count += 1
- # get predicted locations from existing trackers.
- trks = np.zeros((len(self.trackers), 5))
- to_del = []
- ret = []
- for t, trk in enumerate(trks):
- pos = self.trackers[t].predict()[0]
- trk[:] = [pos[0], pos[1], pos[2], pos[3], 0]
- if np.any(np.isnan(pos)):
- to_del.append(t)
- trks = np.ma.compress_rows(np.ma.masked_invalid(trks))
- for t in reversed(to_del):
- self.trackers.pop(t)
- matched, unmatched_dets, unmatched_trks = associate_detections_to_trackers(
- dets, trks
- )
-
- # update matched trackers with assigned detections
- for t, trk in enumerate(self.trackers):
- if t not in unmatched_trks:
- d = matched[np.where(matched[:, 1] == t)[0], 0]
- trk.update(dets[d, :][0])
-
- # create and initialise new trackers for unmatched detections
- for i in unmatched_dets:
- trk = KalmanBoxTracker(dets[i, :])
- self.trackers.append(trk)
- i = len(self.trackers)
- for trk in reversed(self.trackers):
- d = trk.get_state()[0]
- if (trk.time_since_update < 1) and (
- trk.hit_streak >= self.min_hits or self.frame_count <= self.min_hits
- ):
- ret.append(
- np.concatenate((d, [trk.id + 1])).reshape(1, -1)
- ) # +1 as MOT benchmark requires positive
- i -= 1
- # remove dead tracklet
- if trk.time_since_update > self.max_age:
- self.trackers.pop(i)
- if len(ret) > 0:
- return np.concatenate(ret)
- return np.empty((0, 5))
-
-
-def parse_args():
- """Parse input arguments."""
- parser = argparse.ArgumentParser(description="SORT demo")
- parser.add_argument(
- "--display",
- dest="display",
- help="Display online tracker output (slow) [False]",
- action="store_true",
- )
- args = parser.parse_args()
- return args
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/tracking/tracker.py b/video/pwtf-dvd/model_code/inference/test_tools/ct/tracking/tracker.py
deleted file mode 100644
index 20dd79f41a56bcb4873e83bb9c76db9ffbf0f627..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/test_tools/ct/tracking/tracker.py
+++ /dev/null
@@ -1,27 +0,0 @@
-from .sort import Sort
-import numpy as np
-
-
-def get_detections(faces):
- detections = []
- for face in faces:
- x1, y1, x2, y2 = face[0]
- detections.append((x1, y1, x2, y2, face[-1]))
- return np.array(detections)
-
-
-def get_tracks(detect_results):
- tracks = {}
- mot_tracker = Sort()
- for faces in detect_results:
- detections = get_detections(faces)
- track_bbs_ids = mot_tracker.update(detections)
- for track in track_bbs_ids: # 单独框出每一张人脸
- id = int(track[-1])
- box = track[:4]
- if id in tracks:
- tracks[id].append(box)
- else:
- tracks[id] = [box]
-
- return [track for id, track in tracks.items() if len(track) == len(detect_results)]
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/ct/utils.py b/video/pwtf-dvd/model_code/inference/test_tools/ct/utils.py
deleted file mode 100644
index 5ff180fb2a24b9e8bd673d3428e3b101da16a6b0..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/test_tools/ct/utils.py
+++ /dev/null
@@ -1,5 +0,0 @@
-import cv2
-
-
-def write_img(file, img):
- cv2.imwrite(file, img, [cv2.IMWRITE_PNG_COMPRESSION, 0])
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/faster_crop_align_xray.py b/video/pwtf-dvd/model_code/inference/test_tools/faster_crop_align_xray.py
deleted file mode 100644
index 3e99f2f0930e26539793e3d76d74e9df7afc186b..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/test_tools/faster_crop_align_xray.py
+++ /dev/null
@@ -1,73 +0,0 @@
-import numpy as np
-import cv2
-from .warp_for_xray import (
- estimiate_batch_transform,
- transform_landmarks,
- std_points_256,
-)
-import numpy as np
-
-
-class FasterCropAlignXRay:
- """
- 修正到统一坐标系,统一图像大小到标准尺寸
- """
-
- def __init__(self, size=256):
- self.image_size = size
- self.std_points = std_points_256 * size / 256.0
-
- def __call__(self, landmarks, images=None, jitter=False):
- landmarks = [landmark[:4] for landmark in landmarks]
- ori_boxes = np.array([ori_box for _, _, _, ori_box in landmarks])
- five_landmarks = np.array([ldm5 for _, ldm5, _, _ in landmarks])
- landmarks68 = np.array([ldm68 for _, _, ldm68, _ in landmarks])
- # assert landmarks68.min() > 0
-
- left_top = ori_boxes[:, :2].min(0)
-
- right_bottom = ori_boxes[:, 2:].max(0)
-
- size = right_bottom - left_top
-
- w, h = size
-
- diff = ori_boxes[:, :2] - left_top[None, ...]
-
- new_five_landmarks = five_landmarks + diff[:, None, :]
- new_landmarks68 = landmarks68 + diff[:, None, :]
-
- landmark_for_estimiate = new_five_landmarks.copy()
- if jitter:
- landmark_for_estimiate += np.random.uniform(
- -4, 4, landmark_for_estimiate.shape
- )
-
- tfm, trans = estimiate_batch_transform(
- landmark_for_estimiate, tgt_pts=self.std_points
- )
-
- transformed_landmarks68 = np.array(
- [transform_landmarks(ldm68, trans) for ldm68 in new_landmarks68]
- )
-
- if images is not None:
- transformed_images = [
- self.process_sinlge(tfm, image, d, h, w)
- for image, d in zip(images, diff)
- ] # 拼接 func 的参数
- transformed_images = np.stack(transformed_images)
- return transformed_landmarks68, transformed_images
- else:
- return transformed_landmarks68
-
- def process_sinlge(self, tfm, image, d, h, w):
- assert isinstance(image, np.ndarray)
- new_image = np.zeros((h, w, 3), dtype=np.uint8)
- x, y = d
- ih, iw, _ = image.shape
- new_image[y : y + ih, x : x + iw] = image
- transformed_image = cv2.warpAffine(
- new_image, tfm, (self.image_size, self.image_size)
- )
- return transformed_image
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/supply_writer.py b/video/pwtf-dvd/model_code/inference/test_tools/supply_writer.py
deleted file mode 100644
index 0394dfbb8fc50d47984eaa937537de144e51ee81..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/test_tools/supply_writer.py
+++ /dev/null
@@ -1,49 +0,0 @@
-import cv2
-
-class SupplyWriter:
- def __init__(self, intput_video, output_video, opt_thres, rgb_input=True):
- reader = cv2.VideoCapture(intput_video)
- fourcc = cv2.VideoWriter_fourcc(*"XVID")
- fps = reader.get(cv2.CAP_PROP_FPS)
- width = int(reader.get(3))
- height = int(reader.get(4))
- reader.release()
- self.padding = 40
-
- self.writer = cv2.VideoWriter(output_video, fourcc, fps, (height, width)[::-1])
- self.rgb_input = rgb_input
- self.opt_thres = opt_thres
-
- def run(self, images, scores, boxes):
- # Text variables
- font_face = cv2.FONT_HERSHEY_SIMPLEX
- thickness = 5
- font_scale = 3
-
- for image, score, box in zip(images, scores, boxes):
- if self.rgb_input:
- image = cv2.cvtColor(image, cv2.COLOR_RGB2BGR)
- if box is not None:
- label = "fake" if score > self.opt_thres else "real"
- x1, y1, x2, y2 = box
- x = int(x1)
- y = int(y1)
- w = int(x2 - x1)
- h = int(y2 - y1)
- color = (
- (255, 255, 0) if label == "real" else (0, 255, 255)
- ) # BGR 255 0
- cv2.putText(
- image,
- label,
- (x, y + h + 68),
- font_face,
- font_scale,
- color,
- thickness,
- 2,
- )
- # draw box over face
- cv2.rectangle(image, (x, y), (x + w, y + h), color, 10)
- self.writer.write(image)
- self.writer.release()
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/utils.py b/video/pwtf-dvd/model_code/inference/test_tools/utils.py
deleted file mode 100644
index 8069d72b1db1bab8155f88782b5fb6a15fb9563c..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/test_tools/utils.py
+++ /dev/null
@@ -1,115 +0,0 @@
-import numpy as np
-import cv2
-import os
-import platform
-import json
-import errno
-
-
-
-def weak_check(detect_res):
- return sum([len(faces) for faces in detect_res]) > len(detect_res) * 0.75
-
-
-def get_crop_box(shape, box, scale=0.5):
- height, width = shape
- box = np.rint(box).astype(np.int32)
- new_box = box.reshape(2, 2)
- size = new_box[1] - new_box[0]
- diff = scale * size
- diff = diff[None, :] * np.array([-1, 1])[:, None]
- new_box = new_box + diff
- new_box[:, 0] = np.clip(new_box[:, 0], 0, width - 1)
- new_box[:, 1] = np.clip(new_box[:, 1], 0, height - 1)
- new_box = np.rint(new_box).astype(np.int32)
- return new_box.reshape(-1)
-
-
-def get_fps(input_file):
- reader = cv2.VideoCapture(input_file)
- fps = reader.get(cv2.CAP_PROP_FPS)
- reader.release()
- return fps
-
-
-
-def mkdir_p(dirname):
- """Like "mkdir -p", make a dir recursively, but do nothing if the dir exists
- 这个是线程安全的, from Lingzhi Li
- Args:
- dirname(str):
- """
- assert dirname is not None
- if dirname == "" or os.path.isdir(dirname):
- return
- try:
- os.makedirs(dirname)
- except OSError as e:
- if e.errno != errno.EEXIST:
- raise e
-
-
-def mkdir(*args):
- for folder in args:
- if not os.path.isdir(folder):
- mkdir_p(folder)
-
-
-def make_join(*args):
- folder = os.path.join(*args)
- mkdir(folder)
- return folder
-
-
-def list_dir(folder, condition=None, key=lambda x: x, reverse=False, co_join=[]):
- files = os.listdir(folder)
- if condition is not None:
- files = filter(condition, files)
- co_join = [folder] + co_join
- if key is not None:
- files = sorted(files, key=key, reverse=reverse)
- files = [(file, *[os.path.join(fold, file) for fold in co_join]) for file in files]
- return files
-
-def get_jointer(file):
- def jointer(folder):
- return os.path.join(folder, file)
-
- return jointer
-
-def flatten(l):
- return [item for sublist in l for item in sublist]
-
-
-def is_win():
- return platform.system() == "Windows"
-
-
-def get_postfix(post_fix):
- return lambda x: x.endswith(post_fix)
-
-
-def partition(images, size):
- """
- Returns a new list with elements
- of which is a list of certain size.
-
- >>> partition([1, 2, 3, 4], 3)
- [[1, 2, 3], [4]]
- """
- return [
- images[i : i + size] if i + size <= len(images) else images[i:]
- for i in range(0, len(images), size)
- ]
-
-
-def load_json(file):
- with open(file, "r") as f:
- res = json.load(f)
- return res
-
-
-def save_json(file, obj):
- with open(file, "w", encoding="utf-8") as f:
- json.dump(obj, f, indent=4, ensure_ascii=False)
-
diff --git a/video/pwtf-dvd/model_code/inference/test_tools/warp_for_xray.py b/video/pwtf-dvd/model_code/inference/test_tools/warp_for_xray.py
deleted file mode 100644
index ea68def0bcc6b384f1198f21fb0d72253f8085d6..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/test_tools/warp_for_xray.py
+++ /dev/null
@@ -1,574 +0,0 @@
-import numpy as np
-import cv2
-
-# -*- coding: utf-8 -*-
-"""
-Created on Tue Jul 11 06:54:28 2017
-
-@author: zhaoyafei
-"""
-
-import numpy as np
-from numpy.linalg import inv, norm, lstsq
-from numpy.linalg import matrix_rank as rank
-
-"""
-Introduction:
-----------
-numpy implemetation form matlab function CP2TFORM(...)
-with 'transformtype':
- 1) 'nonreflective similarity'
- 2) 'similarity'
-
-
-MATLAB code:
-----------
-%--------------------------------------
-% Function findNonreflectiveSimilarity
-%
-function [trans, output] = findNonreflectiveSimilarity(uv,xy,options)
-%
-% For a nonreflective similarity:
-%
-% let sc = s*cos(theta)
-% let ss = s*sin(theta)
-%
-% [ sc -ss
-% [u v] = [x y 1] * ss sc
-% tx ty]
-%
-% There are 4 unknowns: sc,ss,tx,ty.
-%
-% Another way to write this is:
-%
-% u = [x y 1 0] * [sc
-% ss
-% tx
-% ty]
-%
-% v = [y -x 0 1] * [sc
-% ss
-% tx
-% ty]
-%
-% With 2 or more correspondence points we can combine the u equations and
-% the v equations for one linear system to solve for sc,ss,tx,ty.
-%
-% [ u1 ] = [ x1 y1 1 0 ] * [sc]
-% [ u2 ] [ x2 y2 1 0 ] [ss]
-% [ ... ] [ ... ] [tx]
-% [ un ] [ xn yn 1 0 ] [ty]
-% [ v1 ] [ y1 -x1 0 1 ]
-% [ v2 ] [ y2 -x2 0 1 ]
-% [ ... ] [ ... ]
-% [ vn ] [ yn -xn 0 1 ]
-%
-% Or rewriting the above matrix equation:
-% U = X * r, where r = [sc ss tx ty]'
-% so r = X\ U.
-%
-
-K = options.K;
-M = size(xy,1);
-x = xy(:,1);
-y = xy(:,2);
-X = [x y ones(M,1) zeros(M,1);
- y -x zeros(M,1) ones(M,1) ];
-
-u = uv(:,1);
-v = uv(:,2);
-U = [u; v];
-
-% We know that X * r = U
-if rank(X) >= 2*K
- r = X \ U;
-else
- error(message('images:cp2tform:twoUniquePointsReq'))
-end
-
-sc = r(1);
-ss = r(2);
-tx = r(3);
-ty = r(4);
-
-Tinv = [sc -ss 0;
- ss sc 0;
- tx ty 1];
-
-T = inv(Tinv);
-T(:,3) = [0 0 1]';
-
-trans = maketform('affine', T);
-output = [];
-
-%-------------------------
-% Function findSimilarity
-%
-function [trans, output] = findSimilarity(uv,xy,options)
-%
-% The similarities are a superset of the nonreflective similarities as they may
-% also include reflection.
-%
-% let sc = s*cos(theta)
-% let ss = s*sin(theta)
-%
-% [ sc -ss
-% [u v] = [x y 1] * ss sc
-% tx ty]
-%
-% OR
-%
-% [ sc ss
-% [u v] = [x y 1] * ss -sc
-% tx ty]
-%
-% Algorithm:
-% 1) Solve for trans1, a nonreflective similarity.
-% 2) Reflect the xy data across the Y-axis,
-% and solve for trans2r, also a nonreflective similarity.
-% 3) Transform trans2r to trans2, undoing the reflection done in step 2.
-% 4) Use TFORMFWD to transform uv using both trans1 and trans2,
-% and compare the results, Returnsing the transformation corresponding
-% to the smaller L2 norm.
-
-% Need to reset options.K to prepare for calls to findNonreflectiveSimilarity.
-% This is safe because we already checked that there are enough point pairs.
-options.K = 2;
-
-% Solve for trans1
-[trans1, output] = findNonreflectiveSimilarity(uv,xy,options);
-
-
-% Solve for trans2
-
-% manually reflect the xy data across the Y-axis
-xyR = xy;
-xyR(:,1) = -1*xyR(:,1);
-
-trans2r = findNonreflectiveSimilarity(uv,xyR,options);
-
-% manually reflect the tform to undo the reflection done on xyR
-TreflectY = [-1 0 0;
- 0 1 0;
- 0 0 1];
-trans2 = maketform('affine', trans2r.tdata.T * TreflectY);
-
-
-% Figure out if trans1 or trans2 is better
-xy1 = tformfwd(trans1,uv);
-norm1 = norm(xy1-xy);
-
-xy2 = tformfwd(trans2,uv);
-norm2 = norm(xy2-xy);
-
-if norm1 <= norm2
- trans = trans1;
-else
- trans = trans2;
-end
-"""
-
-
-class MatlabCp2tormException(Exception):
- def __str__(self):
- return "In File {}:{}".format(__file__, super.__str__(self))
-
-
-def tformfwd(trans, uv):
- """
- Function:
- ----------
- apply affine transform 'trans' to uv
-
- Parameters:
- ----------
- @trans: 3x3 np.array
- transform matrix
- @uv: Kx2 np.array
- each row is a pair of coordinates (x, y)
-
- Returns:
- ----------
- @xy: Kx2 np.array
- each row is a pair of transformed coordinates (x, y)
- """
- uv = np.hstack((uv, np.ones((uv.shape[0], 1))))
- xy = np.dot(uv, trans)
- xy = xy[:, 0:-1]
- return xy
-
-
-def tforminv(trans, uv):
- """
- Function:
- ----------
- apply the inverse of affine transform 'trans' to uv
-
- Parameters:
- ----------
- @trans: 3x3 np.array
- transform matrix
- @uv: Kx2 np.array
- each row is a pair of coordinates (x, y)
-
- Returns:
- ----------
- @xy: Kx2 np.array
- each row is a pair of inverse-transformed coordinates (x, y)
- """
- Tinv = inv(trans)
- xy = tformfwd(Tinv, uv)
- return xy
-
-
-def findNonreflectiveSimilarity(uv, xy, options=None):
- """
- Function:
- ----------
- Find Non-reflective Similarity Transform Matrix 'trans':
- u = uv[:, 0]
- v = uv[:, 1]
- x = xy[:, 0]
- y = xy[:, 1]
- [x, y, 1] = [u, v, 1] * trans
-
- Parameters:
- ----------
- @uv: Kx2 np.array
- source points each row is a pair of coordinates (x, y)
- @xy: Kx2 np.array
- each row is a pair of inverse-transformed
- @option: not used, keep it as None
-
- Returns:
- @trans: 3x3 np.array
- transform matrix from uv to xy
- @trans_inv: 3x3 np.array
- inverse of trans, transform matrix from xy to uv
-
- Matlab:
- ----------
- % For a nonreflective similarity:
- %
- % let sc = s*cos(theta)
- % let ss = s*sin(theta)
- %
- % [ sc -ss
- % [u v] = [x y 1] * ss sc
- % tx ty]
- %
- % There are 4 unknowns: sc,ss,tx,ty.
- %
- % Another way to write this is:
- %
- % u = [x y 1 0] * [sc
- % ss
- % tx
- % ty]
- %
- % v = [y -x 0 1] * [sc
- % ss
- % tx
- % ty]
- %
- % With 2 or more correspondence points we can combine the u equations and
- % the v equations for one linear system to solve for sc,ss,tx,ty.
- %
- % [ u1 ] = [ x1 y1 1 0 ] * [sc]
- % [ u2 ] [ x2 y2 1 0 ] [ss]
- % [ ... ] [ ... ] [tx]
- % [ un ] [ xn yn 1 0 ] [ty]
- % [ v1 ] [ y1 -x1 0 1 ]
- % [ v2 ] [ y2 -x2 0 1 ]
- % [ ... ] [ ... ]
- % [ vn ] [ yn -xn 0 1 ]
- %
- % Or rewriting the above matrix equation:
- % U = X * r, where r = [sc ss tx ty]'
- % so r = X\ U.
- %
- """
- options = {"K": 2}
-
- K = options["K"]
- M = xy.shape[0]
- x = xy[:, 0].reshape((-1, 1)) # use reshape to keep a column vector
- y = xy[:, 1].reshape((-1, 1)) # use reshape to keep a column vector
- # print '--->x, y:\n', x, y
-
- tmp1 = np.hstack((x, y, np.ones((M, 1)), np.zeros((M, 1))))
- tmp2 = np.hstack((y, -x, np.zeros((M, 1)), np.ones((M, 1))))
- X = np.vstack((tmp1, tmp2))
- # print '--->X.shape: ', X.shape
- # print 'X:\n', X
-
- u = uv[:, 0].reshape((-1, 1)) # use reshape to keep a column vector
- v = uv[:, 1].reshape((-1, 1)) # use reshape to keep a column vector
- U = np.vstack((u, v))
- # print '--->U.shape: ', U.shape
- # print 'U:\n', U
-
- # We know that X * r = U
- if rank(X) >= 2 * K:
- r, _, _, _ = lstsq(X, U, rcond=-1)
- r = np.squeeze(r)
- else:
- raise Exception("cp2tform:twoUniquePointsReq")
-
- # print '--->r:\n', r
-
- sc = r[0]
- ss = r[1]
- tx = r[2]
- ty = r[3]
-
- Tinv = np.array([[sc, -ss, 0], [ss, sc, 0], [tx, ty, 1]])
-
- # print '--->Tinv:\n', Tinv
-
- T = inv(Tinv)
- # print '--->T:\n', T
-
- T[:, 2] = np.array([0, 0, 1])
-
- return T, Tinv
-
-
-def findSimilarity(uv, xy, options=None):
- """
- Function:
- ----------
- Find Reflective Similarity Transform Matrix 'trans':
- u = uv[:, 0]
- v = uv[:, 1]
- x = xy[:, 0]
- y = xy[:, 1]
- [x, y, 1] = [u, v, 1] * trans
-
- Parameters:
- ----------
- @uv: Kx2 np.array
- source points each row is a pair of coordinates (x, y)
- @xy: Kx2 np.array
- each row is a pair of inverse-transformed
- @option: not used, keep it as None
-
- Returns:
- ----------
- @trans: 3x3 np.array
- transform matrix from uv to xy
- @trans_inv: 3x3 np.array
- inverse of trans, transform matrix from xy to uv
-
- Matlab:
- ----------
- % The similarities are a superset of the nonreflective similarities as they may
- % also include reflection.
- %
- % let sc = s*cos(theta)
- % let ss = s*sin(theta)
- %
- % [ sc -ss
- % [u v] = [x y 1] * ss sc
- % tx ty]
- %
- % OR
- %
- % [ sc ss
- % [u v] = [x y 1] * ss -sc
- % tx ty]
- %
- % Algorithm:
- % 1) Solve for trans1, a nonreflective similarity.
- % 2) Reflect the xy data across the Y-axis,
- % and solve for trans2r, also a nonreflective similarity.
- % 3) Transform trans2r to trans2, undoing the reflection done in step 2.
- % 4) Use TFORMFWD to transform uv using both trans1 and trans2,
- % and compare the results, Returnsing the transformation corresponding
- % to the smaller L2 norm.
-
- % Need to reset options.K to prepare for calls to findNonreflectiveSimilarity.
- % This is safe because we already checked that there are enough point pairs.
- """
- options = {"K": 2}
-
- # uv = np.array(uv)
- # xy = np.array(xy)
-
- # Solve for trans1
- trans1, trans1_inv = findNonreflectiveSimilarity(uv, xy, options)
-
- # Solve for trans2
-
- # manually reflect the xy data across the Y-axis
- xyR = xy
- xyR[:, 0] = -1 * xyR[:, 0]
-
- trans2r, trans2r_inv = findNonreflectiveSimilarity(uv, xyR, options)
-
- # manually reflect the tform to undo the reflection done on xyR
- TreflectY = np.array([[-1, 0, 0], [0, 1, 0], [0, 0, 1]])
-
- trans2 = np.dot(trans2r, TreflectY)
-
- # Figure out if trans1 or trans2 is better
- xy1 = tformfwd(trans1, uv)
- norm1 = norm(xy1 - xy)
-
- xy2 = tformfwd(trans2, uv)
- norm2 = norm(xy2 - xy)
-
- if norm1 <= norm2:
- return trans1, trans1_inv
- else:
- trans2_inv = inv(trans2)
- return trans2, trans2_inv
-
-
-def get_similarity_transform(src_pts, dst_pts, reflective=True):
- """
- Function:
- ----------
- Find Similarity Transform Matrix 'trans':
- u = src_pts[:, 0]
- v = src_pts[:, 1]
- x = dst_pts[:, 0]
- y = dst_pts[:, 1]
- [x, y, 1] = [u, v, 1] * trans
-
- Parameters:
- ----------
- @src_pts: Kx2 np.array
- source points, each row is a pair of coordinates (x, y)
- @dst_pts: Kx2 np.array
- destination points, each row is a pair of transformed
- coordinates (x, y)
- @reflective: True or False
- if True:
- use reflective similarity transform
- else:
- use non-reflective similarity transform
-
- Returns:
- ----------
- @trans: 3x3 np.array
- transform matrix from uv to xy
- trans_inv: 3x3 np.array
- inverse of trans, transform matrix from xy to uv
- """
-
- if reflective:
- trans, trans_inv = findSimilarity(src_pts, dst_pts)
- else:
- trans, trans_inv = findNonreflectiveSimilarity(src_pts, dst_pts)
-
- return trans, trans_inv
-
-
-def cvt_tform_mat_for_cv2(trans):
- """
- Function:
- ----------
- Convert Transform Matrix 'trans' into 'cv2_trans' which could be
- directly used by cv2.warpAffine():
- u = src_pts[:, 0]
- v = src_pts[:, 1]
- x = dst_pts[:, 0]
- y = dst_pts[:, 1]
- [x, y].T = cv_trans * [u, v, 1].T
-
- Parameters:
- ----------
- @trans: 3x3 np.array
- transform matrix from uv to xy
-
- Returns:
- ----------
- @cv2_trans: 2x3 np.array
- transform matrix from src_pts to dst_pts, could be directly used
- for cv2.warpAffine()
- """
- cv2_trans = trans[:, 0:2].T
-
- return cv2_trans
-
-
-def get_similarity_transform_for_cv2(src_pts, dst_pts, reflective=True):
- """
- Function:
- ----------
- Find Similarity Transform Matrix 'cv2_trans' which could be
- directly used by cv2.warpAffine():
- u = src_pts[:, 0]
- v = src_pts[:, 1]
- x = dst_pts[:, 0]
- y = dst_pts[:, 1]
- [x, y].T = cv_trans * [u, v, 1].T
-
- Parameters:
- ----------
- @src_pts: Kx2 np.array
- source points, each row is a pair of coordinates (x, y)
- @dst_pts: Kx2 np.array
- destination points, each row is a pair of transformed
- coordinates (x, y)
- reflective: True or False
- if True:
- use reflective similarity transform
- else:
- use non-reflective similarity transform
-
- Returns:
- ----------
- @cv2_trans: 2x3 np.array
- transform matrix from src_pts to dst_pts, could be directly used
- for cv2.warpAffine()
- """
- trans, trans_inv = get_similarity_transform(src_pts, dst_pts, reflective)
- cv2_trans = cvt_tform_mat_for_cv2(trans)
- return cv2_trans, trans
-
-
-std_points_317 = np.array(
- [
- [85.82991, 115.7792],
- [169.0532, 114.3381],
- [127.574, 167.0006],
- [90.6964, 204.7014],
- [167.3069, 203.3733],
- ]
-)
-
-
-padding = 30
-
-std_points_317 = std_points_317 + padding
-
-std_points_256=std_points_317.copy()
-std_points_256[..., 0] -= 30
-std_points_256[..., 1] -= 60
-
-def warp_as_face_x_ray(img, src_pts, tgt_pts=std_points_317):
- tfm, trans = get_similarity_transform_for_cv2(src_pts.copy(), tgt_pts.copy())
- return cv2.warpAffine(img, tfm, (317, 317)), trans
-
-
-def estimiate_batch_transform(all_src_pts, tgt_pts=std_points_317):
- tgt_pts = np.repeat(tgt_pts[None, ...], len(all_src_pts), 0).reshape(-1, 2)
- src_pts = np.array(all_src_pts).reshape(-1, 2)
- tfm, trans = get_similarity_transform_for_cv2(src_pts, tgt_pts)
- return tfm, trans
-
-
-def batch_warp_as_face_x_ray(images, all_src_pts, tgt_pts=std_points_317):
- tfm, trans = estimiate_batch_transform(all_src_pts, tgt_pts)
- return [cv2.warpAffine(img, tfm, (317, 317)) for img in images], trans
-
-
-def transform_landmarks(landmarks, trans):
- transformed = np.hstack((landmarks, np.ones((landmarks.shape[0], 1))))
- transformed = np.dot(transformed, trans)
- return transformed[:, :2]
-
-def compute_reverse_trans(trans):
- return np.linalg.inv(trans)
\ No newline at end of file
diff --git a/video/pwtf-dvd/model_code/inference/utils/__init__.py b/video/pwtf-dvd/model_code/inference/utils/__init__.py
deleted file mode 100644
index 7437239283d4494444c4b0993d34c90f6891e0c8..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/utils/__init__.py
+++ /dev/null
@@ -1,7 +0,0 @@
-
-__all__ = [] # do not use ' from utils import *'
-
-from .common import *
-from .plugin_loader import PluginLoader
-#from .plugin_loaderv2 import PluginLoader as PluginLoaderV2
-from .model_loader import add_loader
\ No newline at end of file
diff --git a/video/pwtf-dvd/model_code/inference/utils/common.py b/video/pwtf-dvd/model_code/inference/utils/common.py
deleted file mode 100644
index 7c975bf6a08077296351bd10c03ea719538e48a3..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/utils/common.py
+++ /dev/null
@@ -1,80 +0,0 @@
-#!/usr/bin/python
-# -*- coding: UTF-8 -*-
-
-
-import os
-import torch
-from torch.autograd import Variable
-import errno
-import torch.distributed as dist
-import math
-from functools import reduce
-def make_folder(path, version):
- if not os.path.exists(os.path.join(path, version)):
- print(os.path.join(path, version))
- os.makedirs(os.path.join(path, version))
-
-
-def tensor2var(x, grad=False):
- if torch.cuda.is_available():
- x = x.cuda()
- return Variable(x, requires_grad=grad)
-
-def var2tensor(x):
- return x.data.cpu()
-
-def var2numpy(x):
- return x.data.cpu().numpy()
-
-def denorm(x):
- out = (x + 1) / 2
- return out.clamp_(0, 1)
-
-def mkdir_p(dirname):
- """ Like "mkdir -p", make a dir recursively, but do nothing if the dir exists
- Args:
- dirname(str):
- """
- assert dirname is not None
- if dirname == '' or os.path.isdir(dirname):
- return
- try:
- os.makedirs(dirname)
- except OSError as e:
- if e.errno != errno.EEXIST:
- raise e
-
-
-def skipShardSplit(aList, drop_last=False, num_replicas=None, rank=None):
- if not isinstance(aList, list) and not isinstance(aList, tuple):
- aList = List
-
- if num_replicas is None:
- num_replicas = dist.get_world_size() if dist.is_initialized() else 1
- if rank is None:
- rank = dist.get_rank() if dist.is_initialized() else 0
-
- num_replicas = num_replicas
- rank = rank
- drop_last = drop_last
-
- if drop_last:
- aList = aList[0: (len(aList) // num_replicas) * num_replicas]
-
- # subsample
- aList = aList[rank::num_replicas]
-
- return aList
-
-def mixb2a(a,b):
- if len(b) > len(a):
- a,b = b,a
- if len(b) == 0:
- return a
- chunk_num = (len(b))
- a_chunk = splitIntoChunk(a, chunk_num)
- b_chunk = list(map(lambda x:[x],b))
- return reduce(lambda x, y: x+y, [_a+_b for _a,_b in zip(a_chunk, b_chunk)])
-
-def splitIntoChunk(aList, chunk_num):
- return [aList[math.ceil(k * (len(aList) / chunk_num)):math.ceil((k + 1) * (len(aList) / chunk_num)):] for k in range(chunk_num)]
\ No newline at end of file
diff --git a/video/pwtf-dvd/model_code/inference/utils/logger.py b/video/pwtf-dvd/model_code/inference/utils/logger.py
deleted file mode 100644
index b5e3c52c3ae53a698ba67581a43bf20d832eca0d..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/utils/logger.py
+++ /dev/null
@@ -1,182 +0,0 @@
-#!/usr/bin/python
-# -*- coding: UTF-8 -*-
-# Modified by: algohunt
-# Microsoft Research & Peking University
-# lilingzhi@pku.edu.cn
-# Copyright (c) 2019
-
-
-# -*- coding: utf-8 -*-
-
-"""
-Borrow from tensorpack credit goes to yuxin wu
-The logger module itself has the common logging functions of Python's
-:class:`logging.Logger`. For example:
-
-.. code-block:: python
-
- from utils import logger
- logger.set_logger_dir('train_log/test')
- logger.info("Test")
- logger.error("Error happened!")
-"""
-
-
-import logging
-import os
-import os.path
-import shutil
-import sys
-from datetime import datetime, timedelta
-from six.moves import input
-from termcolor import colored
-import time
-
-__all__ = ['set_logger_dir', 'auto_set_dir', 'get_logger_dir']
-
-
-class _MyFormatter(logging.Formatter):
- def format(self, record):
- date = colored('[%(asctime)s @%(filename)s:%(lineno)d]', 'green')
- msg = '%(message)s'
- if record.levelno == logging.WARNING:
- fmt = date + ' ' + colored('WRN', 'red', attrs=['blink']) + ' ' + msg
- elif record.levelno == logging.ERROR or record.levelno == logging.CRITICAL:
- fmt = date + ' ' + colored('ERR', 'red', attrs=['blink', 'underline']) + ' ' + msg
- elif record.levelno == logging.DEBUG:
- fmt = date + ' ' + colored('DBG', 'yellow', attrs=['blink']) + ' ' + msg
- else:
- fmt = date + ' ' + msg
- if hasattr(self, '_style'):
- # Python3 compatibility
- self._style._fmt = fmt
- self._fmt = fmt
- return super(_MyFormatter, self).format(record)
-
-
-def _getlogger():
- logger = logging.getLogger('tensorpack')
- logger.propagate = False
- logger.setLevel(logging.INFO)
- handler = logging.StreamHandler(sys.stdout)
- handler.setFormatter(_MyFormatter(datefmt='%m%d %H:%M:%S'))
- logger.addHandler(handler)
- return logger
-
-
-_logger = _getlogger()
-_LOGGING_METHOD = ['info', 'warning', 'error', 'critical', 'exception', 'debug', 'setLevel']
-# export logger functions
-for func in _LOGGING_METHOD:
- locals()[func] = getattr(_logger, func)
- __all__.append(func)
-# 'warn' is deprecated in logging module
-warn = _logger.warning
-__all__.append('warn')
-
-
-def _get_time_str():
- utc_time = datetime.utcfromtimestamp(time.time())
- beijing_time = utc_time- timedelta(hours=8)
- return beijing_time.strftime('%m%d-%H%M%S')
-
-
-# globals: logger file and directory:
-LOG_DIR = None
-_FILE_HANDLER = None
-
-
-def _set_file(path):
- global _FILE_HANDLER
- if os.path.isfile(path):
- backup_name = path + '.' + _get_time_str()
- shutil.move(path, backup_name)
- _logger.info("Existing log file '{}' backuped to '{}'".format(path, backup_name)) # noqa: F821
- hdl = logging.FileHandler(
- filename=path, encoding='utf-8', mode='w')
- hdl.setFormatter(_MyFormatter(datefmt='%m%d %H:%M:%S'))
-
- _FILE_HANDLER = hdl
- _logger.addHandler(hdl)
- _logger.info("Argv: " + ' '.join(sys.argv))
-
-
-def set_logger_dir(dirname, action=None):
- """
- Set the directory for global logging.
-
- Args:
- dirname(str): log directory
- action(str): an action of ["k","d","q"] to be performed
- when the directory exists. Will ask user by default.
-
- "d": delete the directory. Note that the deletion may fail when
- the directory is used by tensorboard.
-
- "k": keep the directory. This is useful when you resume from a
- previous training and want the directory to look as if the
- training was not interrupted.
- Note that this option does not load old models or any other
- old states for you. It simply does nothing.
-
- """
- global LOG_DIR, _FILE_HANDLER
- if _FILE_HANDLER:
- # unload and close the old file handler, so that we may safely delete the logger directory
- _logger.removeHandler(_FILE_HANDLER)
- del _FILE_HANDLER
-
- def dir_nonempty(dirname):
- # If directory exists and nonempty (ignore hidden files), prompt for action
- return os.path.isdir(dirname) and len([x for x in os.listdir(dirname) if x[0] != '.'])
-
- if dir_nonempty(dirname):
- if not action:
- _logger.warn("""\
-Log directory {} exists! Use 'd' to delete it. """.format(dirname))
- _logger.warn("""\
-If you're resuming from a previous run, you can choose to keep it.
-Press any other key to exit. """)
- while not action:
- action = input("Select Action: k (keep) / d (delete) / q (quit):").lower().strip()
- act = action
- if act == 'b':
- backup_name = dirname + _get_time_str()
- shutil.move(dirname, backup_name)
- info("Directory '{}' backuped to '{}'".format(dirname, backup_name)) # noqa: F821
- elif act == 'd':
- shutil.rmtree(dirname, ignore_errors=True)
- if dir_nonempty(dirname):
- shutil.rmtree(dirname, ignore_errors=False)
- elif act == 'n':
- dirname = dirname + _get_time_str()
- info("Use a new log directory {}".format(dirname)) # noqa: F821
- elif act == 'k':
- pass
- else:
- raise OSError("Directory {} exits!".format(dirname))
- LOG_DIR = dirname
- from . import mkdir_p
- mkdir_p(dirname)
- _set_file(os.path.join(dirname, 'log.log'))
-
-
-def auto_set_dir(action=None, name=None):
- """
- Use :func:`logger.set_logger_dir` to set log directory to
- "./train_log/{scriptname}:{name}". "scriptname" is the name of the main python file currently running"""
- mod = sys.modules['__main__']
- basename = os.path.basename(mod.__file__)
- auto_dirname = os.path.join('train_log', basename[:basename.rfind('.')])
- if name:
- auto_dirname += '_%s' % name if os.name == 'nt' else ':%s' % name
- set_logger_dir(auto_dirname, action=action)
-
-
-def get_logger_dir():
- """
- Returns:
- The logger directory, or None if not set.
- The directory is used for general logging, tensorboard events, checkpoints, etc.
- """
- return LOG_DIR
diff --git a/video/pwtf-dvd/model_code/inference/utils/model_loader.py b/video/pwtf-dvd/model_code/inference/utils/model_loader.py
deleted file mode 100644
index 9be6e723022cd03a23018a72181230caf9525282..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/utils/model_loader.py
+++ /dev/null
@@ -1,117 +0,0 @@
-#!/usr/bin/python
-# -*- coding: UTF-8 -*-
-
-
-import types
-from utils import logger
-from config import config as cfg
-import os
-import sys
-import glob
-import torch
-import traceback
-import types
-import torch.distributed as dist
-import copy
-from .torch_save import torch_save
-
-def add_loader(target, name,max_to_keep=2):
-
- def get_rank(self):
- return dist.get_rank() if dist.is_initialized() else 0
-
- def save_models(self, epoch):
- """ Backup and save the models """
- if self.get_rank() == 0:
- logger.debug("Backing up and saving models")
- if not os.path.exists(self.model_dir):
- os.mkdir(self.model_dir)
-
- torch_save(self.state_dict(), self.get_checkpoint_path(epoch))
- if os.path.exists(self.get_checkpoint_path(epoch - self.max_to_keep)):
- os.remove(self.get_checkpoint_path(epoch - self.max_to_keep))
- logger.info("{} models saved".format(self.name))
-
- def load(self, fullpath=None, epoch=-1):
- """ Force Loading a model, or load the latest model"""
- if fullpath is None:
- fullpath, loaded_epoch = self.find_last(epoch)
- else:
- loaded_epoch = epoch
-
- if fullpath is None:
- logger.info("No existing {} model found".format(self.name))
- return False, -1
- logger.debug("Loading model: '%s'", fullpath)
- try:
- saved_state_dict = torch.load(fullpath, map_location='cpu')
- self.load_state_dict(saved_state_dict)
- logger.info(" consume training from {}".format(fullpath))
- except ValueError as err:
- logger.warning("Failed loading existing training data for {}. Generating new models".format(self.name))
- logger.debug("Exception: %s", str(err))
- return False, -1
- except OSError as err:
- logger.warning("Failed loading existing training data for {}. Generating new models".format(self.name))
- logger.debug("Exception: %s", str(err))
- return False, -1
- except RuntimeError as err:
- logger.warning("{} model has corrupted, try to load earlier one".format(self.name))
- logger.debug("Exception: %s", str(err))
- return False, -1
- except:
- logger.error(traceback.format_exc())
- raise
-
- return True, loaded_epoch
-
- def get_checkpoint_path(self, epoch):
- """" returning the checkpoint path w.r.t epoch which should be {name}_{epoch}.pth"""
- return os.path.join(self.model_dir, self.name + '_' +str(epoch) + '.pth')
-
-
- def find_last(self, epoch=-1, model_dir=None):
- """Finds the last checkpoint file of the last trained model in the
- model directory.
- Returns:
- checkpoint :The path of the last checkpoint file
-
- """
- if model_dir is None:
- model_dir = self.model_dir
- if not os.path.exists(model_dir):
- logger.info("model dir not exists {} ".format(model_dir))
- return None, -1
- #assert os.path.exists(self.model_dir), "model dir not exists {}".format(self.model_dir)
- checkpoints = glob.glob(os.path.join(model_dir, '*.pth'))
-
-
- checkpoints = list(filter(lambda x: os.path.basename(x).startswith(self.name), checkpoints))
- if len(checkpoints) == 0:
- return None, -1
- checkpoints = {int(os.path.basename(x).split('.')[0].split('_')[-1]):x for x in checkpoints}
-
- start = min(checkpoints.keys())
- end = max(checkpoints.keys())
-
- if epoch == -1:
- return checkpoints[end], end
- elif epoch < start :
- raise RuntimeError(
- "model for epoch {} has been deleted as we only keep {} models".format(epoch,self.max_to_keep))
- elif epoch > end:
- raise RuntimeError(
- "epoch {} is bigger than all exist checkpoints".format(epoch))
- else:
- return checkpoints[epoch], epoch
-
- target.find_last = types.MethodType(find_last, target)
- target.get_checkpoint_path = types.MethodType(get_checkpoint_path, target)
- target.load = types.MethodType(load, target)
- target.save_models = types.MethodType(save_models, target)
- target.get_rank = types.MethodType(get_rank, target)
-
- target.max_to_keep = max_to_keep
- target.name = name
- target.model_dir = os.path.join(cfg.path.model_dir, cfg.setting_name)
- return target
\ No newline at end of file
diff --git a/video/pwtf-dvd/model_code/inference/utils/plugin_loader.py b/video/pwtf-dvd/model_code/inference/utils/plugin_loader.py
deleted file mode 100644
index bcd5ababfeb1bb358ef15fe56f35f58d164b7c62..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/utils/plugin_loader.py
+++ /dev/null
@@ -1,69 +0,0 @@
-#!/usr/bin/python
-# -*- coding: UTF-8 -*-
-
-
-
-""" Plugin loader for extract, training and model tasks """
-
-from utils import logger
-import os
-from importlib import import_module
-from typing import Type
-from trainer._base import TrainerBase
-from torch.utils.data import Dataset
-from model._base import ModelBase
-
-class PluginLoader():
- """
- Plugin loader for extract, training and model tasks
- function: get_{model_type}
- args: {model_name}
- will return the a class named model_type under model_type.model_name.py
-
- as it return a class you should also annotate the returning classtype to make
- code linting avaliable in some IDE
- """
- @staticmethod
- def get_classifier(name) -> Type[ModelBase]:
- """ Return requested attribute encoder plugin """
- return PluginLoader._import("model.classifier", name)
-
- @staticmethod
- def get_trainer(name) -> Type[TrainerBase]:
- """ Return requested trainer plugin """
- return PluginLoader._import("trainer", name)
-
- @staticmethod
- def get_dataset(name) -> Type[Dataset]:
- """ Return requested trainer plugin """
- return PluginLoader._import("dataset", name)
-
- @staticmethod
- def _import(attr, name):
- """ Import the plugin's module """
- name = name.replace("-", "_")
- ttl = attr.split(".")[-1].title()
- logger.info("Loading %s from %s plugin...", ttl, name.title())
- attr = "model" if attr == "Trainer" else attr.lower()
- mod = ".".join((attr, name))
- module = import_module(mod)
- logger.info(str(module) + str(ttl))
- return getattr(module, ttl)
-
- @staticmethod
- def get_available_trainer():
- """ Return a list of available models """
- modelpath = os.path.join(os.path.dirname(__file__), "trainer")
- models = sorted(item.name.replace(".py", "").replace("_", "-")
- for item in os.scandir(modelpath)
- if not item.name.startswith("_")
- and item.name.endswith(".py"))
- return models
-
- @staticmethod
- def get_default_model():
- """ Return the default model """
- models = PluginLoader.get_available_models()
- return 'original' if 'original' in models else models[0]
-
-
diff --git a/video/pwtf-dvd/model_code/inference/utils/torch_save.py b/video/pwtf-dvd/model_code/inference/utils/torch_save.py
deleted file mode 100644
index 88fce6ba32ac66b3607a8496a03ae6e8f6321866..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/inference/utils/torch_save.py
+++ /dev/null
@@ -1,9 +0,0 @@
-import torch
-
-def torch_save(arr,file):
- if torch.__version__>="1.6.0":
- torch.save(arr, file, _use_new_zipfile_serialization=False)
- else:
- torch.save(arr, file)
-
-
diff --git a/video/pwtf-dvd/model_code/preprocessing/preprocess.py b/video/pwtf-dvd/model_code/preprocessing/preprocess.py
deleted file mode 100644
index f0ffc0860ecc92f717b459d3e299ec6313848db6..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/preprocessing/preprocess.py
+++ /dev/null
@@ -1,258 +0,0 @@
-import os
-from os.path import join
-import argparse
-import glob
-import subprocess
-import cv2
-from tqdm import tqdm
-import numpy as np
-import logging
-import torch
-from test_tools.common import detect_all, grab_all_frames
-from test_tools.faster_crop_align_xray import FasterCropAlignXRay
-from test_tools.warp_for_xray import (
- estimiate_batch_transform,
- transform_landmarks,
- std_points_256,
-)
-from test_tools.ct.operations import find_longest, multiple_tracking
-from test_tools.utils import get_crop_box
-import datetime
-# from FaceForensics.face_detection_save import get_boundingbox
-
-os.environ['CUDA_LAUNCH_BLOCKING'] = "1"
-os.environ["CUDA_VISIBLE_DEVICES"] = "0"
-
-device=torch.device('cuda')
-#Date
-now = datetime.datetime.now()
-logger = logging.getLogger("main") #Logger 선언
-stream_handler = logging.StreamHandler() # Logger output 방법 선언
-formatter = logging.Formatter('[%(asctime)s][%(levelname)s|%(filename)s:%(lineno)s] >> %(message)s')
-stream_handler.setFormatter(formatter)
-logger.addHandler(stream_handler)
-logger.setLevel(logging.DEBUG)
-
-crop_align_func = FasterCropAlignXRay(256)
-max_frame= 10000
-
-### video path를 받으면 crop된 face를 저장하는 함수 ###
-def crop_face_from_video(video_path,cache_path,crop_path,clip_size):
- # mp4 파일이 아니면 return
- if 'mp4' not in video_path : return
- # video name
- video_name = video_path.split('/')[-1].replace('.mp4','')
- # 만약 crop image path에 crop된 이미지가 110개 이상이면 return
-
- if os.path.exists(crop_path):
- if len(os.listdir(crop_path))>clip_limit:
- logger.info(f'{video_name} already exists')
- return
-
- ##########################################
- # detect_res : list, 전체 frame, whole frame
- # detect_res [] : list, len = 사람 수로 예상 the number of detected face in a frame
- # detect_res [] [] : tuple, length = 3
- # detect_res [] [] 의 각 요소는 각각 box, lm5 : landmark (5,2) , score
- ##########################################
- # all_lm68 : list, 전체 frame, whole frame
- # all_lm68 : list, len = 사람 수로 예상, the number of detected face in a frame
- # all_lm68 : np.array : landmark 68개 (68,2)
- ##########################################
- # frames : each frame's np.array
-
-
- # cache_file : cache file path
- # landmark와 box를 저장하는 cache file
- cache_file = f"{cache_path}.pth"
-
- if os.path.exists(cache_file):
- # cache file이 존재하면 load하고 frame만 불러옴
- detect_res, all_lm68 = torch.load(cache_file)
- frames = grab_all_frames(video_path, max_size=max_frame, cvt=True)
- logger.info("detection result loaded from cache")
- else:
- # cache file이 존재하지 않으면 detect_all 함수를 통해 detect_res, all_lm68, frames를 불러옴
- # detection_all 함수는 retina_face를 이용해서 box와 landmark를 찾는 함수
- detect_res, all_lm68, frames = detect_all(
- video_path, return_frames=True, max_size=10000
- )
- torch.save((detect_res, all_lm68), cache_file)
- try:
- shape = frames[0].shape[:2]
- except IndexError: # if there is no frame in the video, error list에 저장
- f = open("./indexerror.txt", 'a')
- f.write("{}\n".format(video_path))
- f.close()
- return
-
- # 모든 detect_res
- all_detect_res = []
-
- assert len(all_lm68) == len(detect_res)
- # in each frame, save the detected face's bounding box, landmark(5, 68), score as a tuple and save it in a list
- for faces, faces_lm68 in zip(detect_res, all_lm68):
- new_faces = []
- for (box, lm5, score), face_lm68 in zip(faces, faces_lm68):
- new_face = (box, lm5, face_lm68, score)
- new_faces.append(new_face)
- all_detect_res.append(new_faces)
- detect_res = all_detect_res
- # SORT tracking
- # tracks : list, len = 사람 수로 예상, the number of detected face in a frame
- # tracks [] : list, len = 프레임 수, the number of frames
- # tracks [] [] : tuple, length = 4, 각각 box, lm5 : landmark (5,2) , lm68 : landmark (68,2), score
- tracks = multiple_tracking(detect_res)
- # tuples : list, len = 사람 수로 예상, the number of detected face in a frame
- # tuples [] : tuple, length = 2, 각각 0, 프레임 수 the number of frames
- tuples = [(0, len(detect_res))] * len(tracks)
- # if there is no face detected, find the longest face in the video
- if len(tracks) == 0:
- tuples, tracks = find_longest(detect_res)
- data_storage = {}
- frame_boxes = {}
- super_clips = []
- frame_res = {}
- super_clips_start_end = []
- # super_clips : tracking된 face들을 의미하는 것으로 보임
- for track_i, ((start, end), track) in enumerate(zip(tuples, tracks)): # each track(=face)
-
- # if detect_res's length is not equal to track's length, raise error
- assert len(detect_res[start:end]) == len(track)
-
- super_clips.append(len(track))
- super_clips_start_end.append((start, end))
- for face, frame_idx, j in zip(track, range(start, end), range(len(track))): # frame에서 각각의 face
- box,lm5,lm68 = face[:3] # box, lm5, lm68
- big_box = get_crop_box(shape, box, scale=0.5) # get crop box
-
- top_left = big_box[:2][None, :] # top left point
-
- new_lm5 = lm5 - top_left
- new_lm68 = lm68 - top_left
-
- new_box = (box.reshape(2, 2) - top_left).reshape(-1)
-
- info = (new_box, new_lm5, new_lm68, big_box) # face info
-
-
- x1, y1, x2, y2 = big_box
- cropped = frames[frame_idx][y1:y2, x1:x2]
- # cropped = cv2.resize(cropped, (512, 512))
- # face들을 tracking한 박스들로 crop함
- # landmark들도 box에 맞게 변환
- # data_storage에 저장 i는 face id, j는 frame을 의미
- base_key = f"{track_i}_{j}_" # i : face, j : frame
- data_storage[base_key + "img"] = cropped
- data_storage[base_key + "ldm"] = info
- data_storage[base_key + "idx"] = frame_idx
- frame_boxes[frame_idx] = np.rint(box).astype(np.int64)
- # 총 crop된 face들과 그 face들의 frame 수를 알려줌
- logger.info(f"{crop_path} : sampling clips from super clips {super_clips}")
- clips_for_video = []
- clip_size = clip_size
- pad_length = clip_size - 1
-
- # 각 face id 별로 clip을 만듦
- # 아래의 영어 표기로는 8clip을 의미하지만 정확하겐 clip size 만큼 함
- for super_clip_idx, super_clip_size in enumerate(super_clips): # cut the super clip into clips, overlap 7frames, 8frames per clip
- inner_index = list(range(super_clip_size))
-
- if super_clip_size < clip_size: # if there is not enough frames to make a clip, pad the frames
- # to do : how to operate the padding
- # 정확하게 이 코드가 어떻게 동작하는지 모르겠지만
- # 대략적으로 frame들을 clipsize로 나눌때 부족하면
- # clip size만큼의 frame이 되도록 padding을 함
- if super_clip_size < clip_size//2 : continue
- post_module = inner_index[1:-1][::-1] + inner_index
-
- l_post = len(post_module)
- post_module = post_module * (pad_length // l_post + 1)
- post_module = post_module[:pad_length]
- assert len(post_module) == pad_length
-
- pre_module = inner_index + inner_index[1:-1][::-1]
- l_pre = len(post_module)
- pre_module = pre_module * (pad_length // l_pre + 1)
- pre_module = pre_module[-pad_length:]
- assert len(pre_module) == pad_length
-
- inner_index = pre_module + inner_index + post_module
-
- super_clip_size = len(inner_index)
-
- frame_range = [
- inner_index[i : i + clip_size] for i in range(super_clip_size) if i + clip_size <= super_clip_size
- ]
- for indices in frame_range:
- clip = [(super_clip_idx, t) for t in indices]
- clips_for_video.append(clip)
-
- # landmarks, images = crop_align_func(landmarks, images) # i : face, j : frame
- processed_clips = 0 # Track number of processed clips
- for clip in clips_for_video:
- # Check if we've reached the clip limit
- if processed_clips >= clip_limit:
- logger.info(f"Reached clip limit of {clip_limit}, stopping processing")
- break
-
- # 각 자른 clip에 대해서 진행
- images = [data_storage[f"{i}_{j}_img"] for i, j in clip] # call cropped face images from data_storage, i : face, j : frame
- landmarks = [data_storage[f"{i}_{j}_ldm"] for i, j in clip] # call landmarks from data_storage, i : face, j : frame
- # landmark를 기준으로 crop align func을 진행
- # 해당 함수가 clip에 있는 얼굴들의 landmark 평균을 기준으로 박스를 설정하고
- # 박스를 기준으로 crop align을 진행
- # 다르게 말하면, landmark 평균을 기준으로 박스의 geometry를 설정하고
- # 박스의 geometry는 고정한체로 얼굴이 움직이는 걸 찍었다 생각하면 됨
- # 다시 또 말하면, 카메라를 고정하고 사람이 움직이는 것을 찍은것처럼
- # PPT 참조
- landmarks, images = crop_align_func(landmarks, images) # align the face images by landmarks in the clip
- i, j = clip[-1]
- k = super_clips[i]%clip_size
-
- ##########################################################################
- # 코드 변경시 이 함수에서는 이부분만 변경할 것을 권고 !!!!!!!!!!!!!!!!!!!!!!!!
- # 특히, cv2.imwrite함수만 변경할 것을 추천
- ##########################################################################
- if (j+1)%clip_size==0: # if last frame number of the clip is multiple of clip_size, save all images in the clip
- # it means save face alignments in 8 frames in the video so that they don't overlap
- for f, (i,j) in enumerate(clip) :
- cv2.imwrite(join(crop_path, f'{i:02}_{j:04}.png'), cv2.cvtColor(images[f], cv2.COLOR_BGR2RGB))
- if j == super_clips[i]-1: # if the clip have last frame image, save all images in the clip
- if k!=0 : # if the clip is not multiple of clip_size, save the last k images in the clip
- # k is the number of frames that are not overlapped
- for l in range(clip_size-k,clip_size):
- ci,cj = clip[l]
- cv2.imwrite(join(crop_path, f'{ci:02}_{cj:04}.png'), cv2.cvtColor(images[l], cv2.COLOR_BGR2RGB)) # clip means face alignment images in 8 frames, non-overlap
-
- processed_clips += 1 # Increment processed clip counter
- ##########################################################################
-
-
-if __name__ == '__main__':
- p = argparse.ArgumentParser(
- formatter_class=argparse.ArgumentDefaultsHelpFormatter
- )
- p.add_argument('--video_path','-i', type=str, default='/videos.mp4', help='path to input video')
- p.add_argument('--save_path','-s', type=str, default='/data/crop_face', help='path to save cropped faces')
- p.add_argument('--cachepath', '-c', type=str, default='/data/cache', help='path to cache detection results')
- p.add_argument('--clipsize','-l',type=int,default=32, help='number of frames in a clip')
- args = p.parse_args()
- video_path = args.video_path
- save_path = args.save_path
- cache_path = args.cachepath
- clip_size = args.clipsize
-
- crop_face_from_video(video_path, cache_path, crop_path, clip_size)
-
-
-
-
-###################### reference ######################
-
-# this code reference from FTCN Official Code in git hub
-# link is https://github.com/yinglinzheng/FTCN
-
-# - Zheng, Y., Bao, J., Chen, D., Zeng, M., & Wen, F. (2021). Exploring Temporal Coherence for More General Video Face Forgery Detection. In Proceedings of the IEEE/CVF International Conference on Computer Vision (pp. 15044–15054).
-
-
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/__init__.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/__init__.py
deleted file mode 100644
index e69de29bb2d1d6434b8b29ae775ad8c2e48c5391..0000000000000000000000000000000000000000
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/common.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/common.py
deleted file mode 100644
index e76ac698b8fe9a32dfbdf4e294d5afa521250e71..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/preprocessing/test_tools/common.py
+++ /dev/null
@@ -1,122 +0,0 @@
-import os
-
-os.environ["KMP_DUPLICATE_LIB_OK"] = "TRUE"
-
-from .ct.detection.utils import grab_all_frames, get_valid_faces, sample_chunks
-from .ct.operations import multiple_tracking
-import numpy as np
-from .ct.face_alignment import LandmarkPredictor
-from .ct.detection import FaceDetector
-import cv2
-from .utils import flatten,partition
-
-
-detector = FaceDetector(0)
-predictor = LandmarkPredictor(0)
-
-
-def get_five(ldm68):
- groups = [range(36, 42), range(42, 48), [30], [48], [54]]
- points = []
- for group in groups:
- points.append(ldm68[group].mean(0))
- return np.array(points)
-
-
-def get_bbox(mask):
- try:
- y, x = np.nonzero(mask[..., 0])
- return x.min() - 1, y.min() - 1, x.max() + 1, y.max() + 1
- except:
- return None
-
-
-def get_bigger_box(image, box, scale=0.5):
- height, width = image.shape[:2]
- box = np.rint(box).astype(np.int)
- new_box = box.reshape(2, 2)
- size = new_box[1] - new_box[0]
- diff = scale * size
- diff = diff[None, :] * np.array([-1, 1])[:, None]
- new_box = new_box + diff
- new_box[:, 0] = np.clip(new_box[:, 0], 0, width - 1)
- new_box[:, 1] = np.clip(new_box[:, 1], 0, height - 1)
- new_box = np.rint(new_box).astype(np.int)
- return new_box.reshape(-1)
-
-
-def process_bigger_clips(clips, dete_res, clip_size, step, scale=0.5):
- assert len(clips) % clip_size == 0
- detect_results = sample_chunks(dete_res, clip_size, step)
- clips = sample_chunks(clips, clip_size, step)
- new_clips = []
- for i, (frame_clip, record_clip) in enumerate(zip(clips, detect_results)):
- tracks = multiple_tracking(record_clip)
- for j, track in enumerate(tracks):
- new_images = []
- for (box, ldm, _), frame in zip(track, frame_clip):
- big_box = get_bigger_box(frame, box, scale)
- x1, y1, x2, y2 = big_box
- top_left = big_box[:2][None, :]
- new_ldm5 = ldm - top_left
- box = np.rint(box).astype(np.int)
- new_box = (box.reshape(2, 2) - top_left).reshape(-1)
- feed = LandmarkPredictor.prepare_feed(frame, box)
- ldm68 = predictor(feed) - top_left
- new_images.append(
- (frame[y1:y2, x1:x2], big_box, new_box, new_ldm5, ldm68)
- )
- new_clips.append(new_images)
- return new_clips
-
-
-def post(detected_faces):
- return [[face[:4], None, face[-1]] for face in detected_faces]
-
-
-def check(detect_res):
- return min([len(faces) for faces in detect_res]) != 0
-
-
-def detect_all(file, sfd_only=False, return_frames=False, max_size=None):
- frames = grab_all_frames(file, max_size=max_size, cvt=True)
- if not sfd_only:
- detect_res = flatten(
- [detector.detect(item) for item in partition(frames, 50)]
- )
- detect_res = get_valid_faces(detect_res, thres=0.5)
- else:
- raise NotImplementedError
-
- all_68 = get_lm68(frames, detect_res)
- if not return_frames:
- return detect_res, all_68
- else:
- return detect_res, all_68, frames
-
-
-def get_lm68(frames, detect_res):
- assert len(frames) == len(detect_res)
- frame_count = len(frames)
- all_68 = []
- for i in range(frame_count):
- frame = frames[i]
- faces = detect_res[i]
- if len(faces) == 0:
- res_68 = []
- else:
- feeds = []
- for face in faces:
- assert len(face) == 3
- box = face[0]
- feed = LandmarkPredictor.prepare_feed(frame, box)
- feeds.append(feed)
- res_68 = predictor(feeds)
- assert len(res_68) == len(faces)
- for face, l_68 in zip(faces, res_68):
- if face[1] is None:
- face[1] = get_five(l_68)
- all_68.append(res_68)
-
- assert len(all_68) == len(detect_res)
- return all_68
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/detection/__init__.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/detection/__init__.py
deleted file mode 100644
index 23447e84f4f5f23a4c4818df2f26cc19a9200363..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/detection/__init__.py
+++ /dev/null
@@ -1,56 +0,0 @@
-import cv2
-from .detector import RetinaFace
-from .utils import *
-
-
-def assert_bounded(val, low, up):
- return val >= low and val < up
-
-
-def check_valid(face, w, h):
- box = face[0]
- if box[0] > box[2]:
- return False
- if box[1] > box[3]:
- return False
- for idx, bound in zip([0, 1, 2, 3], [w, h, w, h]):
- if not assert_bounded(box[idx], 0, bound):
- return False
- pts = face[1]
- for p in pts:
- for idx, bound in zip([0, 1], [w, h]):
- if not assert_bounded(p[idx], 0, bound):
- return False
- return True
-
-
-def post_detect(detect_results, scale, w, h):
- new_results = []
- for frame_faces in detect_results:
- new_frame_faces = []
- for box, ldm, score in frame_faces:
- box = box * scale
- ldm = ldm * scale
- face = (box, ldm, score)
- if check_valid(face, w=w, h=h):
- new_frame_faces.append(face)
- new_results.append(new_frame_faces)
- return new_results
-
-
-class FaceDetector(RetinaFace):
- def scale_detect(self, images):
- max_res = 1920
- h, w = images[0].shape[:2]
- if max(h, w) > max_res:
- init_scale = max(h, w) / max_res
- else:
- init_scale = 1
- resize_scale = 2 * init_scale
- resize_w = int(w / resize_scale)
- resize_h = int(h / resize_scale)
- detect_input = [cv2.resize(frame, (resize_w, resize_h)) for frame in images]
- detect_results = post_detect(
- self.detect(detect_input), scale=resize_scale, w=w, h=h,
- )
- return detect_results
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/detection/alignment.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/detection/alignment.py
deleted file mode 100644
index 64692a3490c7e8e29e399bd174c5906716b3e6fc..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/detection/alignment.py
+++ /dev/null
@@ -1,608 +0,0 @@
-from itertools import product as product
-from math import ceil
-
-import numpy as np
-import torch
-import torch.backends.cudnn as cudnn
-import torch.nn as nn
-import torch.nn.functional as F
-import torchvision.models._utils as _utils
-
-
-def conv_bn(inp, oup, stride=1, leaky=0):
- return nn.Sequential(
- nn.Conv2d(inp, oup, 3, stride, 1, bias=False),
- nn.BatchNorm2d(oup),
- nn.LeakyReLU(negative_slope=leaky, inplace=True),
- )
-
-
-def conv_bn_no_relu(inp, oup, stride):
- return nn.Sequential(
- nn.Conv2d(inp, oup, 3, stride, 1, bias=False), nn.BatchNorm2d(oup),
- )
-
-
-def conv_bn1X1(inp, oup, stride, leaky=0):
- return nn.Sequential(
- nn.Conv2d(inp, oup, 1, stride, padding=0, bias=False),
- nn.BatchNorm2d(oup),
- nn.LeakyReLU(negative_slope=leaky, inplace=True),
- )
-
-
-def conv_dw(inp, oup, stride, leaky=0.1):
- return nn.Sequential(
- nn.Conv2d(inp, inp, 3, stride, 1, groups=inp, bias=False),
- nn.BatchNorm2d(inp),
- nn.LeakyReLU(negative_slope=leaky, inplace=True),
- nn.Conv2d(inp, oup, 1, 1, 0, bias=False),
- nn.BatchNorm2d(oup),
- nn.LeakyReLU(negative_slope=leaky, inplace=True),
- )
-
-
-class SSH(nn.Module):
- def __init__(self, in_channel, out_channel):
- super(SSH, self).__init__()
- assert out_channel % 4 == 0
- leaky = 0
- if out_channel <= 64:
- leaky = 0.1
- self.conv3X3 = conv_bn_no_relu(in_channel, out_channel // 2, stride=1)
-
- self.conv5X5_1 = conv_bn(in_channel, out_channel // 4, stride=1, leaky=leaky)
- self.conv5X5_2 = conv_bn_no_relu(out_channel // 4, out_channel // 4, stride=1)
-
- self.conv7X7_2 = conv_bn(
- out_channel // 4, out_channel // 4, stride=1, leaky=leaky
- )
- self.conv7x7_3 = conv_bn_no_relu(out_channel // 4, out_channel // 4, stride=1)
-
- def forward(self, input):
- conv3X3 = self.conv3X3(input)
-
- conv5X5_1 = self.conv5X5_1(input)
- conv5X5 = self.conv5X5_2(conv5X5_1)
-
- conv7X7_2 = self.conv7X7_2(conv5X5_1)
- conv7X7 = self.conv7x7_3(conv7X7_2)
-
- out = torch.cat([conv3X3, conv5X5, conv7X7], dim=1)
- out = F.relu(out)
- return out
-
-
-class FPN(nn.Module):
- def __init__(self, in_channels_list, out_channels):
- super(FPN, self).__init__()
- leaky = 0
- if out_channels <= 64:
- leaky = 0.1
- self.output1 = conv_bn1X1(
- in_channels_list[0], out_channels, stride=1, leaky=leaky
- )
- self.output2 = conv_bn1X1(
- in_channels_list[1], out_channels, stride=1, leaky=leaky
- )
- self.output3 = conv_bn1X1(
- in_channels_list[2], out_channels, stride=1, leaky=leaky
- )
-
- self.merge1 = conv_bn(out_channels, out_channels, leaky=leaky)
- self.merge2 = conv_bn(out_channels, out_channels, leaky=leaky)
-
- def forward(self, input):
- # names = list(input.keys())
- input = list(input.values())
-
- output1 = self.output1(input[0])
- output2 = self.output2(input[1])
- output3 = self.output3(input[2])
-
- up3 = F.interpolate(
- output3, size=[output2.size(2), output2.size(3)], mode="nearest"
- )
- output2 = output2 + up3
- output2 = self.merge2(output2)
-
- up2 = F.interpolate(
- output2, size=[output1.size(2), output1.size(3)], mode="nearest"
- )
- output1 = output1 + up2
- output1 = self.merge1(output1)
-
- out = [output1, output2, output3]
- return out
-
-
-class MobileNetV1(nn.Module):
- def __init__(self):
- super(MobileNetV1, self).__init__()
- self.stage1 = nn.Sequential(
- conv_bn(3, 8, 2, leaky=0.1), # 3
- conv_dw(8, 16, 1), # 7
- conv_dw(16, 32, 2), # 11
- conv_dw(32, 32, 1), # 19
- conv_dw(32, 64, 2), # 27
- conv_dw(64, 64, 1), # 43
- )
- self.stage2 = nn.Sequential(
- conv_dw(64, 128, 2), # 43 + 16 = 59
- conv_dw(128, 128, 1), # 59 + 32 = 91
- conv_dw(128, 128, 1), # 91 + 32 = 123
- conv_dw(128, 128, 1), # 123 + 32 = 155
- conv_dw(128, 128, 1), # 155 + 32 = 187
- conv_dw(128, 128, 1), # 187 + 32 = 219
- )
- self.stage3 = nn.Sequential(
- conv_dw(128, 256, 2), # 219 +3 2 = 241
- conv_dw(256, 256, 1), # 241 + 64 = 301
- )
- self.avg = nn.AdaptiveAvgPool2d((1, 1))
- self.fc = nn.Linear(256, 1000)
-
- def forward(self, x):
- x = self.stage1(x)
- x = self.stage2(x)
- x = self.stage3(x)
- x = self.avg(x)
- # x = self.model(x)
- x = x.view(-1, 256)
- x = self.fc(x)
- return x
-
-
-class ClassHead(nn.Module):
- def __init__(self, inchannels=512, num_anchors=3):
- super(ClassHead, self).__init__()
- self.num_anchors = num_anchors
- self.conv1x1 = nn.Conv2d(
- inchannels, self.num_anchors * 2, kernel_size=(1, 1), stride=1, padding=0
- )
-
- def forward(self, x):
- out = self.conv1x1(x)
- out = out.permute(0, 2, 3, 1).contiguous()
-
- return out.view(out.shape[0], -1, 2)
-
-
-class BboxHead(nn.Module):
- def __init__(self, inchannels=512, num_anchors=3):
- super(BboxHead, self).__init__()
- self.conv1x1 = nn.Conv2d(
- inchannels, num_anchors * 4, kernel_size=(1, 1), stride=1, padding=0
- )
-
- def forward(self, x):
- out = self.conv1x1(x)
- out = out.permute(0, 2, 3, 1).contiguous()
-
- return out.view(out.shape[0], -1, 4)
-
-
-class LandmarkHead(nn.Module):
- def __init__(self, inchannels=512, num_anchors=3):
- super(LandmarkHead, self).__init__()
- self.conv1x1 = nn.Conv2d(
- inchannels, num_anchors * 10, kernel_size=(1, 1), stride=1, padding=0
- )
-
- def forward(self, x):
- out = self.conv1x1(x)
- out = out.permute(0, 2, 3, 1).contiguous()
-
- return out.view(out.shape[0], -1, 10)
-
-
-class RetinaFace(nn.Module):
- def __init__(self, cfg=None, phase="train"):
- """
- :param cfg: Network related settings.
- :param phase: train or test.
- """
- super(RetinaFace, self).__init__()
- self.phase = phase
- backbone = None
- if cfg["name"] == "mobilenet0.25":
- backbone = MobileNetV1()
- elif cfg["name"] == "Resnet50":
- import torchvision.models as models
-
- backbone = models.resnet50(pretrained=cfg["pretrain"])
-
- self.body = _utils.IntermediateLayerGetter(backbone, cfg["return_layers"])
- in_channels_stage2 = cfg["in_channel"]
- in_channels_list = [
- in_channels_stage2 * 2,
- in_channels_stage2 * 4,
- in_channels_stage2 * 8,
- ]
- out_channels = cfg["out_channel"]
- self.fpn = FPN(in_channels_list, out_channels)
- self.ssh1 = SSH(out_channels, out_channels)
- self.ssh2 = SSH(out_channels, out_channels)
- self.ssh3 = SSH(out_channels, out_channels)
-
- self.ClassHead = self._make_class_head(fpn_num=3, inchannels=cfg["out_channel"])
- self.BboxHead = self._make_bbox_head(fpn_num=3, inchannels=cfg["out_channel"])
- self.LandmarkHead = self._make_landmark_head(
- fpn_num=3, inchannels=cfg["out_channel"]
- )
-
- def _make_class_head(self, fpn_num=3, inchannels=64, anchor_num=2):
- classhead = nn.ModuleList()
- for i in range(fpn_num):
- classhead.append(ClassHead(inchannels, anchor_num))
- return classhead
-
- def _make_bbox_head(self, fpn_num=3, inchannels=64, anchor_num=2):
- bboxhead = nn.ModuleList()
- for i in range(fpn_num):
- bboxhead.append(BboxHead(inchannels, anchor_num))
- return bboxhead
-
- def _make_landmark_head(self, fpn_num=3, inchannels=64, anchor_num=2):
- landmarkhead = nn.ModuleList()
- for i in range(fpn_num):
- landmarkhead.append(LandmarkHead(inchannels, anchor_num))
- return landmarkhead
-
- def forward(self, inputs):
- out = self.body(inputs)
-
- # FPN
- fpn = self.fpn(out)
-
- # SSH
- feature1 = self.ssh1(fpn[0])
- feature2 = self.ssh2(fpn[1])
- feature3 = self.ssh3(fpn[2])
- features = [feature1, feature2, feature3]
-
- bbox_regressions = torch.cat(
- [self.BboxHead[i](feature) for i, feature in enumerate(features)], dim=1
- )
- classifications = torch.cat(
- [self.ClassHead[i](feature) for i, feature in enumerate(features)], dim=1
- )
- ldm_regressions = torch.cat(
- [self.LandmarkHead[i](feature) for i, feature in enumerate(features)], dim=1
- )
-
- if self.phase == "train":
- output = (bbox_regressions, classifications, ldm_regressions)
- else:
- output = (
- bbox_regressions,
- F.softmax(classifications, dim=-1),
- ldm_regressions,
- )
- return output
-
-
-# Adapted from https://github.com/Hakuyume/chainer-ssd
-def decode(loc, priors, variances):
- boxes = torch.cat(
- (
- priors[:, :2] + loc[:, :2] * variances[0] * priors[:, 2:],
- priors[:, 2:] * torch.exp(loc[:, 2:] * variances[1]),
- ),
- 1,
- )
- boxes[:, :2] -= boxes[:, 2:] / 2
- boxes[:, 2:] += boxes[:, :2]
- return boxes
-
-
-def decode_landm(pre, priors, variances):
- landms = torch.cat(
- (
- priors[:, :2] + pre[:, :2] * variances[0] * priors[:, 2:],
- priors[:, :2] + pre[:, 2:4] * variances[0] * priors[:, 2:],
- priors[:, :2] + pre[:, 4:6] * variances[0] * priors[:, 2:],
- priors[:, :2] + pre[:, 6:8] * variances[0] * priors[:, 2:],
- priors[:, :2] + pre[:, 8:10] * variances[0] * priors[:, 2:],
- ),
- dim=1,
- )
- return landms
-
-
-def py_cpu_nms(dets, thresh):
- """Pure Python NMS baseline."""
- x1 = dets[:, 0]
- y1 = dets[:, 1]
- x2 = dets[:, 2]
- y2 = dets[:, 3]
- scores = dets[:, 4]
-
- areas = (x2 - x1 + 1) * (y2 - y1 + 1)
- order = scores.argsort()[::-1]
-
- keep = []
- while order.size > 0:
- i = order[0]
- keep.append(i)
- xx1 = np.maximum(x1[i], x1[order[1:]])
- yy1 = np.maximum(y1[i], y1[order[1:]])
- xx2 = np.minimum(x2[i], x2[order[1:]])
- yy2 = np.minimum(y2[i], y2[order[1:]])
-
- w = np.maximum(0.0, xx2 - xx1 + 1)
- h = np.maximum(0.0, yy2 - yy1 + 1)
- inter = w * h
- ovr = inter / (areas[i] + areas[order[1:]] - inter)
-
- inds = np.where(ovr <= thresh)[0]
- order = order[inds + 1]
-
- return keep
-
-
-class PriorBox(object):
- def __init__(self, cfg, image_size=None, phase="train"):
- super(PriorBox, self).__init__()
- self.min_sizes = cfg["min_sizes"]
- self.steps = cfg["steps"]
- self.clip = cfg["clip"]
- self.image_size = image_size
- self.feature_maps = [
- [ceil(self.image_size[0] / step), ceil(self.image_size[1] / step)]
- for step in self.steps
- ]
- self.name = "s"
-
- def forward(self):
- anchors = []
- for k, f in enumerate(self.feature_maps):
- min_sizes = self.min_sizes[k]
- for i, j in product(range(f[0]), range(f[1])):
- for min_size in min_sizes:
- s_kx = min_size / self.image_size[1]
- s_ky = min_size / self.image_size[0]
- dense_cx = [
- x * self.steps[k] / self.image_size[1] for x in [j + 0.5]
- ]
- dense_cy = [
- y * self.steps[k] / self.image_size[0] for y in [i + 0.5]
- ]
- for cy, cx in product(dense_cy, dense_cx):
- anchors += [cx, cy, s_kx, s_ky]
-
- # back to torch land
- output = torch.Tensor(anchors).view(-1, 4)
- if self.clip:
- output.clamp_(max=1, min=0)
- return output
-
-
-cfg_mnet = {
- "name": "mobilenet0.25",
- "min_sizes": [[16, 32], [64, 128], [256, 512]],
- "steps": [8, 16, 32],
- "variance": [0.1, 0.2],
- "clip": False,
- "loc_weight": 2.0,
- "gpu_train": True,
- "batch_size": 32,
- "ngpu": 1,
- "epoch": 250,
- "decay1": 190,
- "decay2": 220,
- "image_size": 640,
- "pretrain": True,
- "return_layers": {"stage1": 1, "stage2": 2, "stage3": 3},
- "in_channel": 32,
- "out_channel": 64,
-}
-
-cfg_re50 = {
- "name": "Resnet50",
- "min_sizes": [[16, 32], [64, 128], [256, 512]],
- "steps": [8, 16, 32],
- "variance": [0.1, 0.2],
- "clip": False,
- "loc_weight": 2.0,
- "gpu_train": True,
- "batch_size": 24,
- "ngpu": 4,
- "epoch": 100,
- "decay1": 70,
- "decay2": 90,
- "image_size": 840,
- "pretrain": False,
- "return_layers": {"layer2": 1, "layer3": 2, "layer4": 3},
- "in_channel": 256,
- "out_channel": 256,
-}
-
-
-def check_keys(model, pretrained_state_dict):
- ckpt_keys = set(pretrained_state_dict.keys())
- model_keys = set(model.state_dict().keys())
- used_pretrained_keys = model_keys & ckpt_keys
- assert len(used_pretrained_keys) > 0, "load NONE from pretrained checkpoint"
- return True
-
-
-def remove_prefix(state_dict, prefix):
- """ Old style model is stored with all names of parameters sharing common prefix 'module.' """
- f = lambda x: x.split(prefix, 1)[-1] if x.startswith(prefix) else x
- return {f(key): value for key, value in state_dict.items()}
-
-
-def load_model(model, pretrained_path, load_to_cpu):
- if load_to_cpu:
- if pretrained_path is None:
- url = "https://github.com/yinglinzheng/face_weights/releases/download/v1/mobilenet0.25_Final.pth"
- pretrained_dict = torch.utils.model_zoo.load_url(url)
- else:
- pretrained_dict = torch.load(
- pretrained_path, map_location=lambda storage, loc: storage
- )
- else:
- device = torch.cuda.current_device()
- pretrained_dict = torch.load(
- pretrained_path, map_location=lambda storage, loc: storage.cuda(device)
- )
- if "state_dict" in pretrained_dict.keys():
- pretrained_dict = remove_prefix(pretrained_dict["state_dict"], "module.")
- else:
- pretrained_dict = remove_prefix(pretrained_dict, "module.")
- check_keys(model, pretrained_dict)
- model.load_state_dict(pretrained_dict, strict=False)
- return model
-
-
-def load_net(model_path, device, network="mobilenet"):
- if network == "mobilenet":
- cfg = cfg_mnet
- elif network == "resnet50":
- cfg = cfg_re50
- # net and model
- net = RetinaFace(cfg=cfg, phase="test")
- net = load_model(net, model_path, True)
- net.eval()
- cudnn.benchmark = True
- net = net.to(device)
- return net
-
-
-def parse_det(det):
- landmarks = det[5:].reshape(5, 2)
- box = det[:4]
- score = det[4]
- return box, landmarks, score
-
-
-def post_process(
- loc,
- conf,
- landms,
- prior_data,
- cfg,
- scale,
- scale1,
- resize,
- confidence_threshold,
- top_k,
- nms_threshold,
- keep_top_k,
-):
- boxes = decode(loc, prior_data, cfg["variance"])
- boxes = boxes * scale / resize
- boxes = boxes.cpu().numpy()
- scores = conf.cpu().numpy()[:, 1]
- landms_copy = decode_landm(landms, prior_data, cfg["variance"])
-
- landms_copy = landms_copy * scale1 / resize
- landms_copy = landms_copy.cpu().numpy()
-
- # ignore low scores
- inds = np.where(scores > confidence_threshold)[0]
- boxes = boxes[inds]
- landms_copy = landms_copy[inds]
- scores = scores[inds]
-
- # keep top-K before NMS
- order = scores.argsort()[::-1][:top_k]
- boxes = boxes[order]
- landms_copy = landms_copy[order]
- scores = scores[order]
-
- # do NMS
- dets = np.hstack((boxes, scores[:, np.newaxis])).astype(np.float32, copy=False)
- keep = py_cpu_nms(dets, nms_threshold)
- # keep = nms(dets, args.nms_threshold,force_cpu=args.cpu)
- dets = dets[keep, :]
- landms_copy = landms_copy[keep]
-
- # keep top-K faster NMS
- dets = dets[:keep_top_k, :]
- landms_copy = landms_copy[:keep_top_k, :]
-
- dets = np.concatenate((dets, landms_copy), axis=1)
- # show image
- dets = sorted(dets, key=lambda x: x[4], reverse=True)
- dets = [parse_det(x) for x in dets]
-
- return dets
-
-
-def batch_detect(net, images, device, is_tensor=False, normalized=False):
- with torch.no_grad():
- confidence_threshold = 0.02
- cfg = cfg_mnet
- top_k = 5000
- nms_threshold = 0.4
- keep_top_k = 750
- resize = 1
- if not is_tensor:
- try:
- img = np.float32(images)
- except ValueError:
- raise NotImplementedError("Input images must of same size")
- img = torch.from_numpy(img)
- else:
- img = images.float()
- img = img.to(device)
- mean = (
- torch.as_tensor([104, 117, 123], dtype=img.dtype, device=img.device)
- .unsqueeze(0)
- .unsqueeze(0)
- .unsqueeze(0)
- )
- img -= mean
- img = img.permute(0, 3, 1, 2)
- (batch_size, _, im_height, im_width,) = img.shape
- scale = torch.as_tensor(
- [im_width, im_height, im_width, im_height],
- dtype=img.dtype,
- device=img.device,
- )
- scale = scale.to(device)
-
- loc, conf, landms = net(img) # forward pass
-
- priorbox = PriorBox(cfg, image_size=(im_height, im_width))
- priors = priorbox.forward()
- prior_data = priors.to(device)
- scale1 = torch.as_tensor(
- [
- img.shape[3],
- img.shape[2],
- img.shape[3],
- img.shape[2],
- img.shape[3],
- img.shape[2],
- img.shape[3],
- img.shape[2],
- img.shape[3],
- img.shape[2],
- ],
- dtype=img.dtype,
- device=img.device,
- )
- scale1 = scale1.to(device)
-
- all_dets = [
- post_process(
- loc_i,
- conf_i,
- landms_i,
- prior_data,
- cfg,
- scale,
- scale1,
- resize,
- confidence_threshold,
- top_k,
- nms_threshold,
- keep_top_k,
- )
- for loc_i, conf_i, landms_i in zip(loc, conf, landms)
- ]
-
- return all_dets
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/detection/detector.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/detection/detector.py
deleted file mode 100644
index f38050e7a050eeb05323e7d710a63c2bf1a12455..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/detection/detector.py
+++ /dev/null
@@ -1,46 +0,0 @@
-import os
-
-import numpy as np
-import torch
-
-from .alignment import load_net, batch_detect
-
-
-def get_project_dir():
- current_path = os.path.abspath(os.path.join(__file__, "../"))
- return current_path
-
-
-def relative(path):
- path = os.path.join(get_project_dir(), path)
- return os.path.abspath(path)
-
-
-class RetinaFace:
- def __init__(
- self, gpu_id=-1, model_path=None, network="mobilenet",
- ):
- self.gpu_id = gpu_id
- self.device = (
- torch.device("cpu") if gpu_id == -1 else torch.device("cuda", gpu_id)
- )
- self.model = load_net(model_path, self.device, network)
-
- def detect(self, images):
- if isinstance(images, np.ndarray):
- if len(images.shape) == 3:
- return batch_detect(self.model, [images], self.device)[0]
- elif len(images.shape) == 4:
- return batch_detect(self.model, images, self.device)
- elif isinstance(images, list):
- return batch_detect(self.model, np.array(images), self.device)
- elif isinstance(images, torch.Tensor):
- if len(images.shape) == 3:
- return batch_detect(self.model, images.unsqueeze(0), self.device)[0]
- elif len(images.shape) == 4:
- return batch_detect(self.model, images, self.device)
- else:
- raise NotImplementedError()
-
- def __call__(self, images):
- return self.detect(images)
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/detection/utils.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/detection/utils.py
deleted file mode 100644
index 381fe3c717daeced48806cd273aecd971a16e84c..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/detection/utils.py
+++ /dev/null
@@ -1,147 +0,0 @@
-import cv2
-# from test_tools.utils import flatten
-import numpy as np
-def flatten(l):
- return [item for sublist in l for item in sublist]
-
-def chunks(l, n, step=None):
- if step is None:
- step = n
- return [l[i : i + n] for i in range(0, len(l), step)]
-
-
-def sample_chunks(l, n, step=None):
- return [l[i : i + n] for i in range(0, len(l), step) if i + n <= len(l)]
-
-
-def grab_all_frames(path, max_size, cvt=False):
- capture = cv2.VideoCapture(path)
- ret = True
- frames = []
- while ret:
- ret, frame = capture.read()
- if ret:
- if cvt:
- frame = frame[..., ::-1]
- frames.append(frame)
- if len(frames) == max_size:
- break
- capture.release()
- return frames
-
-
-def get_clips_uniform(path, count, clip_size):
- capture = cv2.VideoCapture(path)
- n_frames = int(capture.get(cv2.CAP_PROP_FRAME_COUNT))
- max_clip_available = n_frames + 1 - clip_size
- if count > max_clip_available:
- count = max_clip_available
- final_start = max_clip_available - 1
- start_indices = np.linspace(0, final_start, count, endpoint=True, dtype=np.int)
- all_clip_idx = [list(range(start, start + clip_size)) for start in start_indices]
- valid = set(flatten(all_clip_idx))
- max_idx = max(valid)
-
- frames = {}
- for idx in range(max_idx + 1):
- # Get the next frame, but don't decode if we're not using it.
- ret = capture.grab()
- if not ret:
- continue
-
- if idx in valid:
- ret, frame = capture.retrieve()
- if not ret or frame is None:
- continue
- else:
- # frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
- frames[idx] = frame
-
- capture.release()
- clips = []
- for clip_idx in all_clip_idx:
- clip = []
- flag = True
- for idx in clip_idx:
- if idx not in frames:
- flag = False
- break
- clip.append(frames[idx])
- if flag:
- clips.append(clip)
- return clips
-
-
-def get_valid_faces(detect_results, max_count=10, thres=0.5, at_least=False):
- new_results = []
- for i, faces in enumerate(detect_results):
- if len(faces) > max_count:
- faces = faces[:max_count]
- l = []
- for j, face in enumerate(faces):
- if face[-1] < thres and not (j == 0 and at_least):
- continue
- box, lm, score = face
- box = box.astype(np.float64)
- lm = lm.astype(np.float64)
- l.append((box, lm, score))
- new_results.append(l)
- return new_results
-
-
-def scale_box(box, scale_h, scale_w, h, w):
- x1, y1, x2, y2 = box.astype(np.int32)
- center_x = (x1 + x2) // 2
- center_y = (y1 + y2) // 2
- box_h = int((y2 - y1) * scale_h)
- box_w = int((x2 - x1) * scale_w)
- new_x1 = center_x - box_w // 2
- new_x2 = new_x1 + box_w
- new_y1 = center_y - box_h // 2
- new_y2 = new_y1 + box_h
- new_x1 = max(new_x1, 0)
- new_y1 = max(new_y1, 0)
- new_y2 = min(new_y2, h)
- new_x2 = min(new_x2, w)
- return new_x1, new_y1, new_x2, new_y2
-
-
-def get_bbox(detect_res):
- tmp_detect_res = get_valid_faces(detect_res, max_count=4, thres=0.5)
- all_face_bboxs = []
- for faces in tmp_detect_res:
- all_face_bboxs.extend([face[0] for face in faces])
- all_face_bboxs = np.array(all_face_bboxs).astype(np.int)
- x1 = all_face_bboxs[:, 0].min()
- x2 = all_face_bboxs[:, 2].max()
- y1 = all_face_bboxs[:, 1].min()
- y2 = all_face_bboxs[:, 3].max()
-
- return x1, y1, x2, y2
-
-
-def delta_detect_res(detect_res, x1, y1):
- diff = np.array([[x1, y1]])
- new_detect_res = []
- for faces in detect_res:
- f = []
- for face in faces:
- box, lm, score = face
- box = box.astype(np.float64)
- box[[0, 2]] -= x1
- box[[1, 3]] -= y1
- lm = lm.astype(np.float64) - diff
- f.append((box, lm, score))
- new_detect_res.append(f)
- return new_detect_res
-
-
-def pre_crop(clips, detect_res):
- box = np.array(get_bbox(detect_res))
- w = box[2] - box[0]
- h = box[3] - box[1]
- x1, y1, x2, y2 = scale_box(
- box, 1.5, 1.2 if w > 2 * h else 1.5, clips[0].shape[0], clips[0].shape[1]
- )
- clips = np.array(clips)
- return clips[:, y1:y2, x1:x2], delta_detect_res(detect_res, x1, y1)
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/face_alignment/__init__.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/face_alignment/__init__.py
deleted file mode 100644
index b0545f2f8c2aa872a127bf4e815c1e614a971723..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/face_alignment/__init__.py
+++ /dev/null
@@ -1 +0,0 @@
-from .predictor import LandmarkPredictor
\ No newline at end of file
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/face_alignment/basenet.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/face_alignment/basenet.py
deleted file mode 100644
index 699b163e0584d56cc6d8f916b171f7ba86326efa..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/face_alignment/basenet.py
+++ /dev/null
@@ -1,107 +0,0 @@
-# Backbone networks used for face landmark detection
-# Cunjian Chen (cunjian@msu.edu)
-
-import torch.nn as nn
-import torchvision.models as models
-
-
-class ConvBlock(nn.Module):
- def __init__(self, inp, oup, k, s, p, dw=False, linear=False):
- super(ConvBlock, self).__init__()
- self.linear = linear
- if dw:
- self.conv = nn.Conv2d(inp, oup, k, s, p, groups=inp, bias=False)
- else:
- self.conv = nn.Conv2d(inp, oup, k, s, p, bias=False)
- self.bn = nn.BatchNorm2d(oup)
- if not linear:
- self.prelu = nn.PReLU(oup)
-
- def forward(self, x):
- x = self.conv(x)
- x = self.bn(x)
- if self.linear:
- return x
- else:
- return self.prelu(x)
-
-
-# SE module
-# https://github.com/wujiyang/Face_Pytorch/blob/master/backbone/cbam.py
-class SEModule(nn.Module):
- """Squeeze and Excitation Module"""
-
- def __init__(self, channels, reduction):
- super(SEModule, self).__init__()
- self.avg_pool = nn.AdaptiveAvgPool2d(1)
- self.fc1 = nn.Conv2d(
- channels, channels // reduction, kernel_size=1, padding=0, bias=False
- )
- self.relu = nn.ReLU(inplace=True)
- self.fc2 = nn.Conv2d(
- channels // reduction, channels, kernel_size=1, padding=0, bias=False
- )
- self.sigmoid = nn.Sigmoid()
-
- def forward(self, x):
- input = x
- x = self.avg_pool(x)
- x = self.fc1(x)
- x = self.relu(x)
- x = self.fc2(x)
- x = self.sigmoid(x)
-
- return input * x
-
-
-# USE global depthwise convolution layer. Compatible with MobileNetV2 (224×224), MobileNetV2_ExternalData (224×224)
-class MobileNet_GDConv(nn.Module):
- def __init__(self, num_classes):
- super(MobileNet_GDConv, self).__init__()
- self.pretrain_net = models.mobilenet_v2(pretrained=False)
- self.base_net = nn.Sequential(*list(self.pretrain_net.children())[:-1])
- self.linear7 = ConvBlock(1280, 1280, (7, 7), 1, 0, dw=True, linear=True)
- self.linear1 = ConvBlock(1280, num_classes, 1, 1, 0, linear=True)
-
- def forward(self, x):
- x = self.base_net(x)
- x = self.linear7(x)
- x = self.linear1(x)
- x = x.view(x.size(0), -1)
- return x
-
-
-# USE global depthwise convolution layer. Compatible with MobileNetV2 (56×56)
-class MobileNet_GDConv_56(nn.Module):
- def __init__(self, num_classes):
- super(MobileNet_GDConv_56, self).__init__()
- self.pretrain_net = models.mobilenet_v2(pretrained=False)
- self.base_net = nn.Sequential(*list(self.pretrain_net.children())[:-1])
- self.linear7 = ConvBlock(1280, 1280, (2, 2), 1, 0, dw=True, linear=True)
- self.linear1 = ConvBlock(1280, num_classes, 1, 1, 0, linear=True)
-
- def forward(self, x):
- x = self.base_net(x)
- x = self.linear7(x)
- x = self.linear1(x)
- x = x.view(x.size(0), -1)
- return x
-
-
-# MobileNetV2 with SE; Compatible with MobileNetV2_SE (224×224) and MobileNetV2_SE_RE (224×224)
-class MobileNet_GDConv_SE(nn.Module):
- def __init__(self, num_classes):
- super(MobileNet_GDConv_SE, self).__init__()
- self.pretrain_net = models.mobilenet_v2(pretrained=True)
- self.base_net = nn.Sequential(*list(self.pretrain_net.children())[:-1])
- self.linear7 = ConvBlock(1280, 1280, (7, 7), 1, 0, dw=True, linear=True)
- self.linear1 = ConvBlock(1280, num_classes, 1, 1, 0, linear=True)
- self.attention = SEModule(1280, 8)
-
- def forward(self, x):
- x = self.base_net(x)
- x = self.attention(x)
- x = self.linear7(x)
- x = self.linear1(x)
- x = x.view(x.size(0), -1)
- return x
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/face_alignment/predictor.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/face_alignment/predictor.py
deleted file mode 100644
index b210b673ab3af90d805e7fff9b0b66b00826fd1f..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/face_alignment/predictor.py
+++ /dev/null
@@ -1,143 +0,0 @@
-# Face alignment demo
-# Uses MTCNN as face detector
-# Cunjian Chen (ccunjian@gmail.com)
-import torch
-import cv2
-import numpy as np
-from torch.utils.data import DataLoader
-from .basenet import MobileNet_GDConv
-
-
-def get_device(gpu_id):
- if gpu_id > -1:
- return torch.device(f"cuda:{str(gpu_id)}")
- else:
- return torch.device("cpu")
-
-
-def load_model(file):
- model = MobileNet_GDConv(136)
- if file is not None:
- model.load_state_dict(torch.load(file, map_location="cpu"))
- else:
- url = "https://github.com/yinglinzheng/face_weights/releases/download/v1/mobilenet_224_model_best_gdconv_external.pth"
- model.load_state_dict(torch.utils.model_zoo.load_url(url))
- return model
-
-
-# landmark of (5L, 2L) from [0,1] to real range
-def reproject(bbox, landmark):
- landmark_ = landmark.clone()
- x1, y1, x2, y2 = bbox
- w = x2 - x1
- h = y2 - y1
- landmark_[:, 0] *= w
- landmark_[:, 0] += x1
- landmark_[:, 1] *= h
- landmark_[:, 1] += y1
- return landmark_
-
-
-def prepare_feed(img, face):
- height, width, _ = img.shape
- mean = np.asarray([0.485, 0.456, 0.406])
- std = np.asarray([0.229, 0.224, 0.225])
- out_size = 224
- x1, y1, x2, y2 = face[:4]
-
- w = x2 - x1 + 1
- h = y2 - y1 + 1
- size = int(min([w, h]) * 1.2)
- cx = x1 + w // 2
- cy = y1 + h // 2
- x1 = cx - size // 2
- x2 = x1 + size
- y1 = cy - size // 2
- y2 = y1 + size
-
- dx = max(0, -x1)
- dy = max(0, -y1)
- x1 = max(0, x1)
- y1 = max(0, y1)
-
- edx = max(0, x2 - width)
- edy = max(0, y2 - height)
- x2 = min(width, x2)
- y2 = min(height, y2)
- new_bbox = torch.Tensor([x1, y1, x2, y2]).int()
- x1, y1, x2, y2 = new_bbox
- cropped = img[y1:y2, x1:x2]
- if dx > 0 or dy > 0 or edx > 0 or edy > 0:
- cropped = cv2.copyMakeBorder(
- cropped, int(dy), int(edy), int(dx), int(edx), cv2.BORDER_CONSTANT, 0
- )
- cropped_face = cv2.resize(cropped, (out_size, out_size))
-
- if cropped_face.shape[0] <= 0 or cropped_face.shape[1] <= 0:
- return None
- test_face = cropped_face.copy()
- test_face = test_face / 255.0
- test_face = (test_face - mean) / std
- test_face = test_face.transpose((2, 0, 1))
- data = torch.from_numpy(test_face).float()
- return dict(data=data, bbox=new_bbox)
-
-
-@torch.no_grad()
-def single_predict(model, feed, device):
- landmark = model(feed["data"].unsqueeze(0).to(device)).cpu()
- landmark = landmark.reshape(-1, 2)
- landmark = reproject(feed["bbox"], landmark)
- return landmark.numpy()
-
-
-@torch.no_grad()
-def batch_predict(model, feeds, device):
- if not isinstance(feeds, list):
- feeds = [feeds]
- # loader = DataLoader(FeedDataset(feeds), batch_size=50, shuffle=False)
- data = []
- for feed in feeds:
- data.append(feed["data"].unsqueeze(0))
- data = torch.cat(data, 0).to(device)
- results = []
-
- landmarks = model(data).cpu()
- for landmark, feed in zip(landmarks, feeds):
- landmark = landmark.reshape(-1, 2)
- landmark = reproject(feed["bbox"], landmark)
- results.append(landmark.numpy())
- return results
-
-
-@torch.no_grad()
-def batch_predict2(model, feeds, device, batch_size=None):
- if not isinstance(feeds, list):
- feeds = [feeds]
- if batch_size is None:
- batch_size = len(feeds)
- loader = DataLoader(feeds, batch_size=len(feeds), shuffle=False)
- results = []
- for feed in loader:
- landmarks = model(feed["data"].to(device)).cpu()
- for landmark, bbox in zip(landmarks, feed["bbox"]):
- landmark = landmark.reshape(-1, 2)
- landmark = reproject(bbox, landmark)
- results.append(landmark.numpy())
- return results
-
-
-class LandmarkPredictor:
- def __init__(self, gpu_id=0, file=None):
- self.device = get_device(gpu_id)
- self.model = load_model(file).to(self.device).eval()
-
- def __call__(self, feeds):
- results = batch_predict2(self.model, feeds, self.device)
- if not isinstance(feeds, list):
- results = results[0]
- return results
-
- @staticmethod
- def prepare_feed(img, face):
- return prepare_feed(img, face)
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/face_alignment/utils.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/face_alignment/utils.py
deleted file mode 100644
index e871b155963b0d1ce6891589d32615c5fc97bb66..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/face_alignment/utils.py
+++ /dev/null
@@ -1,17 +0,0 @@
-import cv2
-
-
-def drawLandmark_multiple(img, bbox, landmark):
- """
- Input:
- - img: gray or RGB
- - bbox: type of BBox
- - landmark: reproject landmark of (5L, 2L)
- Output:
- - img marked with landmark and bbox
- """
- x1, y1, x2, y2 = bbox
- cv2.rectangle(img, (x1, y1), (x2, y2), (0, 0, 255), 2)
- for x, y in landmark:
- cv2.circle(img, (int(x), int(y)), 2, (0, 255, 0), -1)
- return img
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/operations.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/operations.py
deleted file mode 100644
index 68375fcc4efd768abcfddf39521af21d574c4aac..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/operations.py
+++ /dev/null
@@ -1,79 +0,0 @@
-import os
-
-import os
-import cv2
-import numpy as np
-from .tracking.sort import iou
-
-
-def face_iou(f1, f2):
- return iou(f1[0], f2[0])
-
-
-def simple_tracking(batch_landmarks, index=0, thres=0.5):
- track = []
-
- for i, faces in enumerate(batch_landmarks):
- if i == 0:
- if len(faces) <= index or faces[index][-1] < 0.8:
- return None
- if index != 0:
- for idx in range(index):
- if face_iou(faces[idx], faces[index]) > thres:
- return None
- track.append(faces[index])
- else:
- last = track[i - 1]
- if len(faces) == 0:
- return None
- sorted_faces = sorted(faces, key=lambda x: face_iou(x, last), reverse=True)
- if face_iou(sorted_faces[0], last) < thres:
- return None
- track.append(sorted_faces[0])
- return track
-
-
-def multiple_tracking(batch_landmarks):
- tracks = []
- for i in range(len(batch_landmarks[0])):
- track = simple_tracking(batch_landmarks, index=i)
- if track is None:
- continue
- tracks.append(track)
- return tracks
-
-def find_longest(detect_res):
- fc = len(detect_res)
- tuples = []
- start = 0
- end = 0
- previous_count = -1
- all_tracks = []
- # start 取得到,end 取不到
- while start < (fc - 1):
- for end in range(start + 2, fc + 1):
- tracks = multiple_tracking(detect_res[start:end])
- if (len(tracks) != previous_count and previous_count != -1) or len(
- tracks
- ) == 0:
- break
- previous_count = len(tracks)
- if end - start > 2:
- if end != fc:
- un_reach_end = end - 1
- else:
- un_reach_end = end
- sub_tracks = multiple_tracking(detect_res[start:un_reach_end])
- if end == fc and len(sub_tracks) == 0:
- un_reach_end = end - 1
- sub_tracks = multiple_tracking(detect_res[start:un_reach_end])
- if len(sub_tracks) > 0:
- tpl = (start, un_reach_end)
- tuples.append(tpl)
- all_tracks.append(sub_tracks[0])
- else:
- raise NotImplementedError
- previous_count = -1
- end = un_reach_end
- start = end
- return tuples, all_tracks
\ No newline at end of file
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/tracking/__init__.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/tracking/__init__.py
deleted file mode 100644
index e69de29bb2d1d6434b8b29ae775ad8c2e48c5391..0000000000000000000000000000000000000000
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/tracking/sort.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/tracking/sort.py
deleted file mode 100644
index dc7b0838e7110a2d3521c6d4cfefbb40cc266c23..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/tracking/sort.py
+++ /dev/null
@@ -1,285 +0,0 @@
-"""
- SORT: A Simple, Online and Realtime Tracker
- Copyright (C) 2016 Alex Bewley alex@dynamicdetection.com
-
- This program is free software: you can redistribute it and/or modify
- it under the terms of the GNU General Public License as published by
- the Free Software Foundation, either version 3 of the License, or
- (at your option) any later version.
-
- This program is distributed in the hope that it will be useful,
- but WITHOUT ANY WARRANTY; without even the implied warranty of
- MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
- GNU General Public License for more details.
-
- You should have received a copy of the GNU General Public License
- along with this program. If not, see .
-"""
-from __future__ import print_function
-import os.path
-import numpy as np
-import matplotlib.pyplot as plt
-import matplotlib.patches as patches
-from scipy.optimize import linear_sum_assignment
-import glob
-import time
-import argparse
-from filterpy.kalman import KalmanFilter
-
-
-def iou(bb_test, bb_gt):
- """
- Computes IUO between two bboxes in the form [x1,y1,x2,y2]
- """
- xx1 = np.maximum(bb_test[0], bb_gt[0])
- yy1 = np.maximum(bb_test[1], bb_gt[1])
- xx2 = np.minimum(bb_test[2], bb_gt[2])
- yy2 = np.minimum(bb_test[3], bb_gt[3])
- w = np.maximum(0.0, xx2 - xx1)
- h = np.maximum(0.0, yy2 - yy1)
- wh = w * h
- o = wh / (
- (bb_test[2] - bb_test[0]) * (bb_test[3] - bb_test[1])
- + (bb_gt[2] - bb_gt[0]) * (bb_gt[3] - bb_gt[1])
- - wh
- )
- return o
-
-
-def convert_bbox_to_z(bbox):
- """
- Takes a bounding box in the form [x1,y1,x2,y2] and returns z in the form
- [x,y,s,r] where x,y is the centre of the box and s is the scale/area and r is
- the aspect ratio
- """
- w = bbox[2] - bbox[0]
- h = bbox[3] - bbox[1]
- x = bbox[0] + w / 2.0
- y = bbox[1] + h / 2.0
- s = w * h # scale is just area
- r = w / float(h)
- return np.array([x, y, s, r]).reshape((4, 1))
-
-
-def convert_x_to_bbox(x, score=None):
- """
- Takes a bounding box in the centre form [x,y,s,r] and returns it in the form
- [x1,y1,x2,y2] where x1,y1 is the top left and x2,y2 is the bottom right
- """
- w = np.sqrt(x[2] * x[3])
- h = x[2] / w
- if score == None:
- return np.array(
- [x[0] - w / 2.0, x[1] - h / 2.0, x[0] + w / 2.0, x[1] + h / 2.0]
- ).reshape((1, 4))
- else:
- return np.array(
- [x[0] - w / 2.0, x[1] - h / 2.0, x[0] + w / 2.0, x[1] + h / 2.0, score]
- ).reshape((1, 5))
-
-
-class KalmanBoxTracker(object):
- """
- This class represents the internel state of individual tracked objects observed as bbox.
- """
-
- count = 0
-
- def __init__(self, bbox):
- """
- Initialises a tracker using initial bounding box.
- """
- # define constant velocity model
- self.kf = KalmanFilter(dim_x=7, dim_z=4)
- self.kf.F = np.array(
- [
- [1, 0, 0, 0, 1, 0, 0],
- [0, 1, 0, 0, 0, 1, 0],
- [0, 0, 1, 0, 0, 0, 1],
- [0, 0, 0, 1, 0, 0, 0],
- [0, 0, 0, 0, 1, 0, 0],
- [0, 0, 0, 0, 0, 1, 0],
- [0, 0, 0, 0, 0, 0, 1],
- ]
- )
- self.kf.H = np.array(
- [
- [1, 0, 0, 0, 0, 0, 0],
- [0, 1, 0, 0, 0, 0, 0],
- [0, 0, 1, 0, 0, 0, 0],
- [0, 0, 0, 1, 0, 0, 0],
- ]
- )
-
- self.kf.R[2:, 2:] *= 10.0
- self.kf.P[
- 4:, 4:
- ] *= 1000.0 # give high uncertainty to the unobservable initial velocities
- self.kf.P *= 10.0
- self.kf.Q[-1, -1] *= 0.01
- self.kf.Q[4:, 4:] *= 0.01
-
- self.kf.x[:4] = convert_bbox_to_z(bbox)
- self.time_since_update = 0
- self.id = KalmanBoxTracker.count
- KalmanBoxTracker.count += 1
- self.history = []
- self.hits = 0
- self.hit_streak = 0
- self.age = 0
-
- def update(self, bbox):
- """
- Updates the state vector with observed bbox.
- """
- self.time_since_update = 0
- self.history = []
- self.hits += 1
- self.hit_streak += 1
- self.kf.update(convert_bbox_to_z(bbox))
-
- def predict(self):
- """
- Advances the state vector and returns the predicted bounding box estimate.
- """
- if (self.kf.x[6] + self.kf.x[2]) <= 0:
- self.kf.x[6] *= 0.0
- self.kf.predict()
- self.age += 1
- if self.time_since_update > 0:
- self.hit_streak = 0
- self.time_since_update += 1
- self.history.append(convert_x_to_bbox(self.kf.x))
- return self.history[-1]
-
- def get_state(self):
- """
- Returns the current bounding box estimate.
- """
- return convert_x_to_bbox(self.kf.x)
-
-
-def associate_detections_to_trackers(detections, trackers, iou_threshold=0.3):
- """
- Assigns detections to tracked object (both represented as bounding boxes)
-
- Returns 3 lists of matches, unmatched_detections and unmatched_trackers
- """
- if len(trackers) == 0:
- return (
- np.empty((0, 2), dtype=int),
- np.arange(len(detections)),
- np.empty((0, 5), dtype=int),
- )
- iou_matrix = np.zeros((len(detections), len(trackers)), dtype=np.float32)
-
- for d, det in enumerate(detections):
- for t, trk in enumerate(trackers):
- iou_matrix[d, t] = iou(det, trk)
-
- matched_indices = linear_sum_assignment(-iou_matrix)
- matched_indices = np.array(list(zip(*matched_indices)), dtype=np.int)
- matched_indices.shape = (-1, 2)
- # print(matched_indices)
- # print(type(matched_indices))
-
- unmatched_detections = []
- for d, det in enumerate(detections):
- if d not in matched_indices[:, 0]:
- unmatched_detections.append(d)
- unmatched_trackers = []
- for t, trk in enumerate(trackers):
- if t not in matched_indices[:, 1]:
- unmatched_trackers.append(t)
-
- # filter out matched with low IOU
- matches = []
- for m in matched_indices:
- if iou_matrix[m[0], m[1]] < iou_threshold:
- unmatched_detections.append(m[0])
- unmatched_trackers.append(m[1])
- else:
- matches.append(m.reshape(1, 2))
- if len(matches) == 0:
- matches = np.empty((0, 2), dtype=int)
- else:
- matches = np.concatenate(matches, axis=0)
-
- return matches, np.array(unmatched_detections), np.array(unmatched_trackers)
-
-
-class Sort(object):
- def __init__(self, max_age=1, min_hits=3):
- """
- Sets key parameters for SORT
- """
- self.max_age = max_age
- self.min_hits = min_hits
- self.trackers = []
- self.frame_count = 0
-
- def update(self, dets):
- """
- Params:
- dets - a numpy array of detections in the format [[x1,y1,x2,y2,score],[x1,y1,x2,y2,score],...]
- Requires: this method must be called once for each frame even with empty detections.
- Returns the a similar array, where the last column is the object ID.
-
- NOTE: The number of objects returned may differ from the number of detections provided.
- """
- self.frame_count += 1
- # get predicted locations from existing trackers.
- trks = np.zeros((len(self.trackers), 5))
- to_del = []
- ret = []
- for t, trk in enumerate(trks):
- pos = self.trackers[t].predict()[0]
- trk[:] = [pos[0], pos[1], pos[2], pos[3], 0]
- if np.any(np.isnan(pos)):
- to_del.append(t)
- trks = np.ma.compress_rows(np.ma.masked_invalid(trks))
- for t in reversed(to_del):
- self.trackers.pop(t)
- matched, unmatched_dets, unmatched_trks = associate_detections_to_trackers(
- dets, trks
- )
-
- # update matched trackers with assigned detections
- for t, trk in enumerate(self.trackers):
- if t not in unmatched_trks:
- d = matched[np.where(matched[:, 1] == t)[0], 0]
- trk.update(dets[d, :][0])
-
- # create and initialise new trackers for unmatched detections
- for i in unmatched_dets:
- trk = KalmanBoxTracker(dets[i, :])
- self.trackers.append(trk)
- i = len(self.trackers)
- for trk in reversed(self.trackers):
- d = trk.get_state()[0]
- if (trk.time_since_update < 1) and (
- trk.hit_streak >= self.min_hits or self.frame_count <= self.min_hits
- ):
- ret.append(
- np.concatenate((d, [trk.id + 1])).reshape(1, -1)
- ) # +1 as MOT benchmark requires positive
- i -= 1
- # remove dead tracklet
- if trk.time_since_update > self.max_age:
- self.trackers.pop(i)
- if len(ret) > 0:
- return np.concatenate(ret)
- return np.empty((0, 5))
-
-
-def parse_args():
- """Parse input arguments."""
- parser = argparse.ArgumentParser(description="SORT demo")
- parser.add_argument(
- "--display",
- dest="display",
- help="Display online tracker output (slow) [False]",
- action="store_true",
- )
- args = parser.parse_args()
- return args
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/tracking/tracker.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/tracking/tracker.py
deleted file mode 100644
index 20dd79f41a56bcb4873e83bb9c76db9ffbf0f627..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/tracking/tracker.py
+++ /dev/null
@@ -1,27 +0,0 @@
-from .sort import Sort
-import numpy as np
-
-
-def get_detections(faces):
- detections = []
- for face in faces:
- x1, y1, x2, y2 = face[0]
- detections.append((x1, y1, x2, y2, face[-1]))
- return np.array(detections)
-
-
-def get_tracks(detect_results):
- tracks = {}
- mot_tracker = Sort()
- for faces in detect_results:
- detections = get_detections(faces)
- track_bbs_ids = mot_tracker.update(detections)
- for track in track_bbs_ids: # 单独框出每一张人脸
- id = int(track[-1])
- box = track[:4]
- if id in tracks:
- tracks[id].append(box)
- else:
- tracks[id] = [box]
-
- return [track for id, track in tracks.items() if len(track) == len(detect_results)]
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/utils.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/utils.py
deleted file mode 100644
index 5ff180fb2a24b9e8bd673d3428e3b101da16a6b0..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/preprocessing/test_tools/ct/utils.py
+++ /dev/null
@@ -1,5 +0,0 @@
-import cv2
-
-
-def write_img(file, img):
- cv2.imwrite(file, img, [cv2.IMWRITE_PNG_COMPRESSION, 0])
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/faster_crop_align_xray.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/faster_crop_align_xray.py
deleted file mode 100644
index 3e99f2f0930e26539793e3d76d74e9df7afc186b..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/preprocessing/test_tools/faster_crop_align_xray.py
+++ /dev/null
@@ -1,73 +0,0 @@
-import numpy as np
-import cv2
-from .warp_for_xray import (
- estimiate_batch_transform,
- transform_landmarks,
- std_points_256,
-)
-import numpy as np
-
-
-class FasterCropAlignXRay:
- """
- 修正到统一坐标系,统一图像大小到标准尺寸
- """
-
- def __init__(self, size=256):
- self.image_size = size
- self.std_points = std_points_256 * size / 256.0
-
- def __call__(self, landmarks, images=None, jitter=False):
- landmarks = [landmark[:4] for landmark in landmarks]
- ori_boxes = np.array([ori_box for _, _, _, ori_box in landmarks])
- five_landmarks = np.array([ldm5 for _, ldm5, _, _ in landmarks])
- landmarks68 = np.array([ldm68 for _, _, ldm68, _ in landmarks])
- # assert landmarks68.min() > 0
-
- left_top = ori_boxes[:, :2].min(0)
-
- right_bottom = ori_boxes[:, 2:].max(0)
-
- size = right_bottom - left_top
-
- w, h = size
-
- diff = ori_boxes[:, :2] - left_top[None, ...]
-
- new_five_landmarks = five_landmarks + diff[:, None, :]
- new_landmarks68 = landmarks68 + diff[:, None, :]
-
- landmark_for_estimiate = new_five_landmarks.copy()
- if jitter:
- landmark_for_estimiate += np.random.uniform(
- -4, 4, landmark_for_estimiate.shape
- )
-
- tfm, trans = estimiate_batch_transform(
- landmark_for_estimiate, tgt_pts=self.std_points
- )
-
- transformed_landmarks68 = np.array(
- [transform_landmarks(ldm68, trans) for ldm68 in new_landmarks68]
- )
-
- if images is not None:
- transformed_images = [
- self.process_sinlge(tfm, image, d, h, w)
- for image, d in zip(images, diff)
- ] # 拼接 func 的参数
- transformed_images = np.stack(transformed_images)
- return transformed_landmarks68, transformed_images
- else:
- return transformed_landmarks68
-
- def process_sinlge(self, tfm, image, d, h, w):
- assert isinstance(image, np.ndarray)
- new_image = np.zeros((h, w, 3), dtype=np.uint8)
- x, y = d
- ih, iw, _ = image.shape
- new_image[y : y + ih, x : x + iw] = image
- transformed_image = cv2.warpAffine(
- new_image, tfm, (self.image_size, self.image_size)
- )
- return transformed_image
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/supply_writer.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/supply_writer.py
deleted file mode 100644
index 0394dfbb8fc50d47984eaa937537de144e51ee81..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/preprocessing/test_tools/supply_writer.py
+++ /dev/null
@@ -1,49 +0,0 @@
-import cv2
-
-class SupplyWriter:
- def __init__(self, intput_video, output_video, opt_thres, rgb_input=True):
- reader = cv2.VideoCapture(intput_video)
- fourcc = cv2.VideoWriter_fourcc(*"XVID")
- fps = reader.get(cv2.CAP_PROP_FPS)
- width = int(reader.get(3))
- height = int(reader.get(4))
- reader.release()
- self.padding = 40
-
- self.writer = cv2.VideoWriter(output_video, fourcc, fps, (height, width)[::-1])
- self.rgb_input = rgb_input
- self.opt_thres = opt_thres
-
- def run(self, images, scores, boxes):
- # Text variables
- font_face = cv2.FONT_HERSHEY_SIMPLEX
- thickness = 5
- font_scale = 3
-
- for image, score, box in zip(images, scores, boxes):
- if self.rgb_input:
- image = cv2.cvtColor(image, cv2.COLOR_RGB2BGR)
- if box is not None:
- label = "fake" if score > self.opt_thres else "real"
- x1, y1, x2, y2 = box
- x = int(x1)
- y = int(y1)
- w = int(x2 - x1)
- h = int(y2 - y1)
- color = (
- (255, 255, 0) if label == "real" else (0, 255, 255)
- ) # BGR 255 0
- cv2.putText(
- image,
- label,
- (x, y + h + 68),
- font_face,
- font_scale,
- color,
- thickness,
- 2,
- )
- # draw box over face
- cv2.rectangle(image, (x, y), (x + w, y + h), color, 10)
- self.writer.write(image)
- self.writer.release()
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/utils.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/utils.py
deleted file mode 100644
index 051f29c7c5808fd08aae14b6a44c54490986ce3a..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/preprocessing/test_tools/utils.py
+++ /dev/null
@@ -1,115 +0,0 @@
-import numpy as np
-import cv2
-import os
-import platform
-import json
-import errno
-
-
-
-def weak_check(detect_res):
- return sum([len(faces) for faces in detect_res]) > len(detect_res) * 0.75
-
-
-def get_crop_box(shape, box, scale=0.5):
- height, width = shape
- box = np.rint(box).astype(np.int64)
- new_box = box.reshape(2, 2)
- size = new_box[1] - new_box[0]
- diff = scale * size
- diff = diff[None, :] * np.array([-1, 1])[:, None]
- new_box = new_box + diff
- new_box[:, 0] = np.clip(new_box[:, 0], 0, width - 1)
- new_box[:, 1] = np.clip(new_box[:, 1], 0, height - 1)
- new_box = np.rint(new_box).astype(np.int64)
- return new_box.reshape(-1)
-
-
-def get_fps(input_file):
- reader = cv2.VideoCapture(input_file)
- fps = reader.get(cv2.CAP_PROP_FPS)
- reader.release()
- return fps
-
-
-
-def mkdir_p(dirname):
- """Like "mkdir -p", make a dir recursively, but do nothing if the dir exists
- 这个是线程安全的, from Lingzhi Li
- Args:
- dirname(str):
- """
- assert dirname is not None
- if dirname == "" or os.path.isdir(dirname):
- return
- try:
- os.makedirs(dirname)
- except OSError as e:
- if e.errno != errno.EEXIST:
- raise e
-
-
-def mkdir(*args):
- for folder in args:
- if not os.path.isdir(folder):
- mkdir_p(folder)
-
-
-def make_join(*args):
- folder = os.path.join(*args)
- mkdir(folder)
- return folder
-
-
-def list_dir(folder, condition=None, key=lambda x: x, reverse=False, co_join=[]):
- files = os.listdir(folder)
- if condition is not None:
- files = filter(condition, files)
- co_join = [folder] + co_join
- if key is not None:
- files = sorted(files, key=key, reverse=reverse)
- files = [(file, *[os.path.join(fold, file) for fold in co_join]) for file in files]
- return files
-
-def get_jointer(file):
- def jointer(folder):
- return os.path.join(folder, file)
-
- return jointer
-
-def flatten(l):
- return [item for sublist in l for item in sublist]
-
-
-def is_win():
- return platform.system() == "Windows"
-
-
-def get_postfix(post_fix):
- return lambda x: x.endswith(post_fix)
-
-
-def partition(images, size):
- """
- Returns a new list with elements
- of which is a list of certain size.
-
- >>> partition([1, 2, 3, 4], 3)
- [[1, 2, 3], [4]]
- """
- return [
- images[i : i + size] if i + size <= len(images) else images[i:]
- for i in range(0, len(images), size)
- ]
-
-
-def load_json(file):
- with open(file, "r") as f:
- res = json.load(f)
- return res
-
-
-def save_json(file, obj):
- with open(file, "w", encoding="utf-8") as f:
- json.dump(obj, f, indent=4, ensure_ascii=False)
-
diff --git a/video/pwtf-dvd/model_code/preprocessing/test_tools/warp_for_xray.py b/video/pwtf-dvd/model_code/preprocessing/test_tools/warp_for_xray.py
deleted file mode 100644
index ea68def0bcc6b384f1198f21fb0d72253f8085d6..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/model_code/preprocessing/test_tools/warp_for_xray.py
+++ /dev/null
@@ -1,574 +0,0 @@
-import numpy as np
-import cv2
-
-# -*- coding: utf-8 -*-
-"""
-Created on Tue Jul 11 06:54:28 2017
-
-@author: zhaoyafei
-"""
-
-import numpy as np
-from numpy.linalg import inv, norm, lstsq
-from numpy.linalg import matrix_rank as rank
-
-"""
-Introduction:
-----------
-numpy implemetation form matlab function CP2TFORM(...)
-with 'transformtype':
- 1) 'nonreflective similarity'
- 2) 'similarity'
-
-
-MATLAB code:
-----------
-%--------------------------------------
-% Function findNonreflectiveSimilarity
-%
-function [trans, output] = findNonreflectiveSimilarity(uv,xy,options)
-%
-% For a nonreflective similarity:
-%
-% let sc = s*cos(theta)
-% let ss = s*sin(theta)
-%
-% [ sc -ss
-% [u v] = [x y 1] * ss sc
-% tx ty]
-%
-% There are 4 unknowns: sc,ss,tx,ty.
-%
-% Another way to write this is:
-%
-% u = [x y 1 0] * [sc
-% ss
-% tx
-% ty]
-%
-% v = [y -x 0 1] * [sc
-% ss
-% tx
-% ty]
-%
-% With 2 or more correspondence points we can combine the u equations and
-% the v equations for one linear system to solve for sc,ss,tx,ty.
-%
-% [ u1 ] = [ x1 y1 1 0 ] * [sc]
-% [ u2 ] [ x2 y2 1 0 ] [ss]
-% [ ... ] [ ... ] [tx]
-% [ un ] [ xn yn 1 0 ] [ty]
-% [ v1 ] [ y1 -x1 0 1 ]
-% [ v2 ] [ y2 -x2 0 1 ]
-% [ ... ] [ ... ]
-% [ vn ] [ yn -xn 0 1 ]
-%
-% Or rewriting the above matrix equation:
-% U = X * r, where r = [sc ss tx ty]'
-% so r = X\ U.
-%
-
-K = options.K;
-M = size(xy,1);
-x = xy(:,1);
-y = xy(:,2);
-X = [x y ones(M,1) zeros(M,1);
- y -x zeros(M,1) ones(M,1) ];
-
-u = uv(:,1);
-v = uv(:,2);
-U = [u; v];
-
-% We know that X * r = U
-if rank(X) >= 2*K
- r = X \ U;
-else
- error(message('images:cp2tform:twoUniquePointsReq'))
-end
-
-sc = r(1);
-ss = r(2);
-tx = r(3);
-ty = r(4);
-
-Tinv = [sc -ss 0;
- ss sc 0;
- tx ty 1];
-
-T = inv(Tinv);
-T(:,3) = [0 0 1]';
-
-trans = maketform('affine', T);
-output = [];
-
-%-------------------------
-% Function findSimilarity
-%
-function [trans, output] = findSimilarity(uv,xy,options)
-%
-% The similarities are a superset of the nonreflective similarities as they may
-% also include reflection.
-%
-% let sc = s*cos(theta)
-% let ss = s*sin(theta)
-%
-% [ sc -ss
-% [u v] = [x y 1] * ss sc
-% tx ty]
-%
-% OR
-%
-% [ sc ss
-% [u v] = [x y 1] * ss -sc
-% tx ty]
-%
-% Algorithm:
-% 1) Solve for trans1, a nonreflective similarity.
-% 2) Reflect the xy data across the Y-axis,
-% and solve for trans2r, also a nonreflective similarity.
-% 3) Transform trans2r to trans2, undoing the reflection done in step 2.
-% 4) Use TFORMFWD to transform uv using both trans1 and trans2,
-% and compare the results, Returnsing the transformation corresponding
-% to the smaller L2 norm.
-
-% Need to reset options.K to prepare for calls to findNonreflectiveSimilarity.
-% This is safe because we already checked that there are enough point pairs.
-options.K = 2;
-
-% Solve for trans1
-[trans1, output] = findNonreflectiveSimilarity(uv,xy,options);
-
-
-% Solve for trans2
-
-% manually reflect the xy data across the Y-axis
-xyR = xy;
-xyR(:,1) = -1*xyR(:,1);
-
-trans2r = findNonreflectiveSimilarity(uv,xyR,options);
-
-% manually reflect the tform to undo the reflection done on xyR
-TreflectY = [-1 0 0;
- 0 1 0;
- 0 0 1];
-trans2 = maketform('affine', trans2r.tdata.T * TreflectY);
-
-
-% Figure out if trans1 or trans2 is better
-xy1 = tformfwd(trans1,uv);
-norm1 = norm(xy1-xy);
-
-xy2 = tformfwd(trans2,uv);
-norm2 = norm(xy2-xy);
-
-if norm1 <= norm2
- trans = trans1;
-else
- trans = trans2;
-end
-"""
-
-
-class MatlabCp2tormException(Exception):
- def __str__(self):
- return "In File {}:{}".format(__file__, super.__str__(self))
-
-
-def tformfwd(trans, uv):
- """
- Function:
- ----------
- apply affine transform 'trans' to uv
-
- Parameters:
- ----------
- @trans: 3x3 np.array
- transform matrix
- @uv: Kx2 np.array
- each row is a pair of coordinates (x, y)
-
- Returns:
- ----------
- @xy: Kx2 np.array
- each row is a pair of transformed coordinates (x, y)
- """
- uv = np.hstack((uv, np.ones((uv.shape[0], 1))))
- xy = np.dot(uv, trans)
- xy = xy[:, 0:-1]
- return xy
-
-
-def tforminv(trans, uv):
- """
- Function:
- ----------
- apply the inverse of affine transform 'trans' to uv
-
- Parameters:
- ----------
- @trans: 3x3 np.array
- transform matrix
- @uv: Kx2 np.array
- each row is a pair of coordinates (x, y)
-
- Returns:
- ----------
- @xy: Kx2 np.array
- each row is a pair of inverse-transformed coordinates (x, y)
- """
- Tinv = inv(trans)
- xy = tformfwd(Tinv, uv)
- return xy
-
-
-def findNonreflectiveSimilarity(uv, xy, options=None):
- """
- Function:
- ----------
- Find Non-reflective Similarity Transform Matrix 'trans':
- u = uv[:, 0]
- v = uv[:, 1]
- x = xy[:, 0]
- y = xy[:, 1]
- [x, y, 1] = [u, v, 1] * trans
-
- Parameters:
- ----------
- @uv: Kx2 np.array
- source points each row is a pair of coordinates (x, y)
- @xy: Kx2 np.array
- each row is a pair of inverse-transformed
- @option: not used, keep it as None
-
- Returns:
- @trans: 3x3 np.array
- transform matrix from uv to xy
- @trans_inv: 3x3 np.array
- inverse of trans, transform matrix from xy to uv
-
- Matlab:
- ----------
- % For a nonreflective similarity:
- %
- % let sc = s*cos(theta)
- % let ss = s*sin(theta)
- %
- % [ sc -ss
- % [u v] = [x y 1] * ss sc
- % tx ty]
- %
- % There are 4 unknowns: sc,ss,tx,ty.
- %
- % Another way to write this is:
- %
- % u = [x y 1 0] * [sc
- % ss
- % tx
- % ty]
- %
- % v = [y -x 0 1] * [sc
- % ss
- % tx
- % ty]
- %
- % With 2 or more correspondence points we can combine the u equations and
- % the v equations for one linear system to solve for sc,ss,tx,ty.
- %
- % [ u1 ] = [ x1 y1 1 0 ] * [sc]
- % [ u2 ] [ x2 y2 1 0 ] [ss]
- % [ ... ] [ ... ] [tx]
- % [ un ] [ xn yn 1 0 ] [ty]
- % [ v1 ] [ y1 -x1 0 1 ]
- % [ v2 ] [ y2 -x2 0 1 ]
- % [ ... ] [ ... ]
- % [ vn ] [ yn -xn 0 1 ]
- %
- % Or rewriting the above matrix equation:
- % U = X * r, where r = [sc ss tx ty]'
- % so r = X\ U.
- %
- """
- options = {"K": 2}
-
- K = options["K"]
- M = xy.shape[0]
- x = xy[:, 0].reshape((-1, 1)) # use reshape to keep a column vector
- y = xy[:, 1].reshape((-1, 1)) # use reshape to keep a column vector
- # print '--->x, y:\n', x, y
-
- tmp1 = np.hstack((x, y, np.ones((M, 1)), np.zeros((M, 1))))
- tmp2 = np.hstack((y, -x, np.zeros((M, 1)), np.ones((M, 1))))
- X = np.vstack((tmp1, tmp2))
- # print '--->X.shape: ', X.shape
- # print 'X:\n', X
-
- u = uv[:, 0].reshape((-1, 1)) # use reshape to keep a column vector
- v = uv[:, 1].reshape((-1, 1)) # use reshape to keep a column vector
- U = np.vstack((u, v))
- # print '--->U.shape: ', U.shape
- # print 'U:\n', U
-
- # We know that X * r = U
- if rank(X) >= 2 * K:
- r, _, _, _ = lstsq(X, U, rcond=-1)
- r = np.squeeze(r)
- else:
- raise Exception("cp2tform:twoUniquePointsReq")
-
- # print '--->r:\n', r
-
- sc = r[0]
- ss = r[1]
- tx = r[2]
- ty = r[3]
-
- Tinv = np.array([[sc, -ss, 0], [ss, sc, 0], [tx, ty, 1]])
-
- # print '--->Tinv:\n', Tinv
-
- T = inv(Tinv)
- # print '--->T:\n', T
-
- T[:, 2] = np.array([0, 0, 1])
-
- return T, Tinv
-
-
-def findSimilarity(uv, xy, options=None):
- """
- Function:
- ----------
- Find Reflective Similarity Transform Matrix 'trans':
- u = uv[:, 0]
- v = uv[:, 1]
- x = xy[:, 0]
- y = xy[:, 1]
- [x, y, 1] = [u, v, 1] * trans
-
- Parameters:
- ----------
- @uv: Kx2 np.array
- source points each row is a pair of coordinates (x, y)
- @xy: Kx2 np.array
- each row is a pair of inverse-transformed
- @option: not used, keep it as None
-
- Returns:
- ----------
- @trans: 3x3 np.array
- transform matrix from uv to xy
- @trans_inv: 3x3 np.array
- inverse of trans, transform matrix from xy to uv
-
- Matlab:
- ----------
- % The similarities are a superset of the nonreflective similarities as they may
- % also include reflection.
- %
- % let sc = s*cos(theta)
- % let ss = s*sin(theta)
- %
- % [ sc -ss
- % [u v] = [x y 1] * ss sc
- % tx ty]
- %
- % OR
- %
- % [ sc ss
- % [u v] = [x y 1] * ss -sc
- % tx ty]
- %
- % Algorithm:
- % 1) Solve for trans1, a nonreflective similarity.
- % 2) Reflect the xy data across the Y-axis,
- % and solve for trans2r, also a nonreflective similarity.
- % 3) Transform trans2r to trans2, undoing the reflection done in step 2.
- % 4) Use TFORMFWD to transform uv using both trans1 and trans2,
- % and compare the results, Returnsing the transformation corresponding
- % to the smaller L2 norm.
-
- % Need to reset options.K to prepare for calls to findNonreflectiveSimilarity.
- % This is safe because we already checked that there are enough point pairs.
- """
- options = {"K": 2}
-
- # uv = np.array(uv)
- # xy = np.array(xy)
-
- # Solve for trans1
- trans1, trans1_inv = findNonreflectiveSimilarity(uv, xy, options)
-
- # Solve for trans2
-
- # manually reflect the xy data across the Y-axis
- xyR = xy
- xyR[:, 0] = -1 * xyR[:, 0]
-
- trans2r, trans2r_inv = findNonreflectiveSimilarity(uv, xyR, options)
-
- # manually reflect the tform to undo the reflection done on xyR
- TreflectY = np.array([[-1, 0, 0], [0, 1, 0], [0, 0, 1]])
-
- trans2 = np.dot(trans2r, TreflectY)
-
- # Figure out if trans1 or trans2 is better
- xy1 = tformfwd(trans1, uv)
- norm1 = norm(xy1 - xy)
-
- xy2 = tformfwd(trans2, uv)
- norm2 = norm(xy2 - xy)
-
- if norm1 <= norm2:
- return trans1, trans1_inv
- else:
- trans2_inv = inv(trans2)
- return trans2, trans2_inv
-
-
-def get_similarity_transform(src_pts, dst_pts, reflective=True):
- """
- Function:
- ----------
- Find Similarity Transform Matrix 'trans':
- u = src_pts[:, 0]
- v = src_pts[:, 1]
- x = dst_pts[:, 0]
- y = dst_pts[:, 1]
- [x, y, 1] = [u, v, 1] * trans
-
- Parameters:
- ----------
- @src_pts: Kx2 np.array
- source points, each row is a pair of coordinates (x, y)
- @dst_pts: Kx2 np.array
- destination points, each row is a pair of transformed
- coordinates (x, y)
- @reflective: True or False
- if True:
- use reflective similarity transform
- else:
- use non-reflective similarity transform
-
- Returns:
- ----------
- @trans: 3x3 np.array
- transform matrix from uv to xy
- trans_inv: 3x3 np.array
- inverse of trans, transform matrix from xy to uv
- """
-
- if reflective:
- trans, trans_inv = findSimilarity(src_pts, dst_pts)
- else:
- trans, trans_inv = findNonreflectiveSimilarity(src_pts, dst_pts)
-
- return trans, trans_inv
-
-
-def cvt_tform_mat_for_cv2(trans):
- """
- Function:
- ----------
- Convert Transform Matrix 'trans' into 'cv2_trans' which could be
- directly used by cv2.warpAffine():
- u = src_pts[:, 0]
- v = src_pts[:, 1]
- x = dst_pts[:, 0]
- y = dst_pts[:, 1]
- [x, y].T = cv_trans * [u, v, 1].T
-
- Parameters:
- ----------
- @trans: 3x3 np.array
- transform matrix from uv to xy
-
- Returns:
- ----------
- @cv2_trans: 2x3 np.array
- transform matrix from src_pts to dst_pts, could be directly used
- for cv2.warpAffine()
- """
- cv2_trans = trans[:, 0:2].T
-
- return cv2_trans
-
-
-def get_similarity_transform_for_cv2(src_pts, dst_pts, reflective=True):
- """
- Function:
- ----------
- Find Similarity Transform Matrix 'cv2_trans' which could be
- directly used by cv2.warpAffine():
- u = src_pts[:, 0]
- v = src_pts[:, 1]
- x = dst_pts[:, 0]
- y = dst_pts[:, 1]
- [x, y].T = cv_trans * [u, v, 1].T
-
- Parameters:
- ----------
- @src_pts: Kx2 np.array
- source points, each row is a pair of coordinates (x, y)
- @dst_pts: Kx2 np.array
- destination points, each row is a pair of transformed
- coordinates (x, y)
- reflective: True or False
- if True:
- use reflective similarity transform
- else:
- use non-reflective similarity transform
-
- Returns:
- ----------
- @cv2_trans: 2x3 np.array
- transform matrix from src_pts to dst_pts, could be directly used
- for cv2.warpAffine()
- """
- trans, trans_inv = get_similarity_transform(src_pts, dst_pts, reflective)
- cv2_trans = cvt_tform_mat_for_cv2(trans)
- return cv2_trans, trans
-
-
-std_points_317 = np.array(
- [
- [85.82991, 115.7792],
- [169.0532, 114.3381],
- [127.574, 167.0006],
- [90.6964, 204.7014],
- [167.3069, 203.3733],
- ]
-)
-
-
-padding = 30
-
-std_points_317 = std_points_317 + padding
-
-std_points_256=std_points_317.copy()
-std_points_256[..., 0] -= 30
-std_points_256[..., 1] -= 60
-
-def warp_as_face_x_ray(img, src_pts, tgt_pts=std_points_317):
- tfm, trans = get_similarity_transform_for_cv2(src_pts.copy(), tgt_pts.copy())
- return cv2.warpAffine(img, tfm, (317, 317)), trans
-
-
-def estimiate_batch_transform(all_src_pts, tgt_pts=std_points_317):
- tgt_pts = np.repeat(tgt_pts[None, ...], len(all_src_pts), 0).reshape(-1, 2)
- src_pts = np.array(all_src_pts).reshape(-1, 2)
- tfm, trans = get_similarity_transform_for_cv2(src_pts, tgt_pts)
- return tfm, trans
-
-
-def batch_warp_as_face_x_ray(images, all_src_pts, tgt_pts=std_points_317):
- tfm, trans = estimiate_batch_transform(all_src_pts, tgt_pts)
- return [cv2.warpAffine(img, tfm, (317, 317)) for img in images], trans
-
-
-def transform_landmarks(landmarks, trans):
- transformed = np.hstack((landmarks, np.ones((landmarks.shape[0], 1))))
- transformed = np.dot(transformed, trans)
- return transformed[:, :2]
-
-def compute_reverse_trans(trans):
- return np.linalg.inv(trans)
\ No newline at end of file
diff --git a/video/pwtf-dvd/requirements.txt b/video/pwtf-dvd/requirements.txt
deleted file mode 100644
index 6aff79022bb8a120c5d0980ca47676a180b1a9fd..0000000000000000000000000000000000000000
--- a/video/pwtf-dvd/requirements.txt
+++ /dev/null
@@ -1,16 +0,0 @@
-fastapi
-uvicorn
-pydantic
-python-multipart
-torch>=2.0.0
-torchvision>=0.15.0
-opencv-python-headless
-numpy<2.0.0
-Pillow
-einops
-fvcore
-pyyaml
-scipy
-filterpy
-matplotlib
-simplejson
diff --git a/video/recce/Dockerfile b/video/recce/Dockerfile
deleted file mode 100644
index 65fbdfa1a5e7bbd9375142c70e0ed4074ed45b44..0000000000000000000000000000000000000000
--- a/video/recce/Dockerfile
+++ /dev/null
@@ -1,57 +0,0 @@
-FROM nvidia/cuda:12.1.1-cudnn8-runtime-ubuntu22.04
-
-ENV DEBIAN_FRONTEND=noninteractive
-ENV PYTHONUNBUFFERED=1
-
-WORKDIR /app
-
-# Install Python 3.10 and system dependencies for OpenCV
-RUN apt-get update && apt-get install -y --no-install-recommends \
- python3 python3-pip \
- libgl1 \
- libglib2.0-0 \
- libsm6 \
- libxext6 \
- libxrender-dev \
- && rm -rf /var/lib/apt/lists/*
-
-RUN ln -sf /usr/bin/python3 /usr/bin/python
-
-# Install PyTorch with CUDA 12.1
-RUN pip install --no-cache-dir \
- torch==2.5.1 torchvision==0.20.1 \
- --index-url https://download.pytorch.org/whl/cu121
-
-# Copy requirements and install (facenet-pytorch needs --no-deps due to torch<2.3 pin)
-COPY requirements.txt .
-RUN pip install --no-cache-dir --no-deps facenet-pytorch && \
- pip install --no-cache-dir -r requirements.txt
-
-# Create logs and weights directories
-RUN mkdir -p logs weights
-
-# Copy model code (vendored RECCE architecture)
-COPY model_code/ /app/model_code/
-
-# Copy weights
-COPY weights/ /app/weights/
-
-# Copy application code
-COPY app.py .
-
-# Environment variables
-ENV MODEL_PORT=7007
-ENV PRELOAD_MODEL=false
-ENV MODEL_TIMEOUT=1800
-ENV WEIGHTS_PATH=/app/weights/recce_checkpoint.pth
-
-# Expose port
-EXPOSE 7007
-
-# Drop root privileges
-RUN adduser --disabled-password --gecos '' appuser && \
- chown -R appuser:appuser /app/logs /app/weights
-USER appuser
-
-# Run the service
-CMD ["python", "app.py"]
diff --git a/video/recce/__pycache__/app.cpython-313.pyc b/video/recce/__pycache__/app.cpython-313.pyc
deleted file mode 100644
index 5ba9764dcf9204dc7f986c2c2c4997d2e167bfe8..0000000000000000000000000000000000000000
Binary files a/video/recce/__pycache__/app.cpython-313.pyc and /dev/null differ
diff --git a/video/recce/app.py b/video/recce/app.py
deleted file mode 100644
index 8acc5ee4eb52e6f66d582ea157762d364afdea72..0000000000000000000000000000000000000000
--- a/video/recce/app.py
+++ /dev/null
@@ -1,490 +0,0 @@
-"""RECCE deepfake video detection service.
-
-Wraps the RECCE (CVPR 2022) reconstruction-classification face forgery
-detection model with a FastAPI endpoint. Uses an Xception encoder with
-guided attention and graph reasoning, trained via DeepfakeBench on
-FaceForensics++ (c40).
-
-The checkpoint originates from DeepfakeBench v1.0.1 which wraps the
-original RECCE architecture under a ``model.`` prefix and trains with
-2-class (real/fake) output instead of the original repo's 1-class
-sigmoid. Weights are loaded by stripping the ``model.`` prefix and
-instantiating ``Recce(num_classes=2)``.
-
-Reference: Cao et al., "End-to-End Reconstruction-Classification
-Learning for Face Forgery Detection", CVPR 2022.
-"""
-
-import base64
-import gc
-import logging
-import os
-import platform
-import sys
-import tempfile
-import threading
-import time
-from typing import Any, Dict, List, Optional, Tuple
-
-import cv2
-import numpy as np
-import torch
-import torch.nn.functional as F
-import uvicorn
-from facenet_pytorch import MTCNN
-from fastapi import FastAPI, HTTPException
-from PIL import Image
-from pydantic import BaseModel, ConfigDict, Field
-
-# ── Model code import ──────────────────────────────────────────────────────
-# The original RECCE model code is vendored under model_code/. We add it
-# to sys.path so that ``from model.network import Recce`` resolves.
-_MODEL_CODE_DIR = os.path.join(os.path.dirname(__file__), "model_code")
-if _MODEL_CODE_DIR not in sys.path:
- sys.path.insert(0, _MODEL_CODE_DIR)
-
-# Patch timm's xception to skip pretrained-weight download. We load our
-# own checkpoint, so downloading ImageNet weights wastes bandwidth and
-# fails in air-gapped Docker containers.
-import timm.models # noqa: E402
-_original_xception = timm.models.xception
-
-
-def _xception_no_pretrained(**kwargs):
- """Force pretrained=False to avoid downloading default weights."""
- kwargs["pretrained"] = False
- return _original_xception(**kwargs)
-
-
-timm.models.xception = _xception_no_pretrained
-
-from model.network import Recce # noqa: E402
-
-logging.basicConfig(level=logging.INFO)
-logger = logging.getLogger(__name__)
-
-MODEL_PORT = int(os.environ.get("MODEL_PORT", 7007))
-PRELOAD_MODEL = os.environ.get("PRELOAD_MODEL", "false").lower() == "true"
-MODEL_TIMEOUT = int(os.environ.get("MODEL_TIMEOUT", 1800))
-WEIGHTS_PATH = os.environ.get(
- "WEIGHTS_PATH", "/app/weights/recce_checkpoint.pth"
-)
-
-# RECCE uses 299x299 face crops (from config/Recce.yml and inference.py)
-IMAGE_SIZE = (299, 299)
-# Number of frames to sample from each video
-NUM_FRAMES = 32
-# Face crop margin factor
-MARGIN_FACTOR = 0.5
-
-
-def _get_device() -> torch.device:
- """Select optimal device: CUDA > MPS (Apple) > CPU."""
- override = os.environ.get("DEEPSAFE_DEVICE", "").strip().lower()
- if override == "cpu":
- return torch.device("cpu")
- if override == "cuda" and torch.cuda.is_available():
- return torch.device("cuda")
- if (
- override == "mps"
- and hasattr(torch.backends, "mps")
- and torch.backends.mps.is_available()
- ):
- return torch.device("mps")
- if torch.cuda.is_available():
- return torch.device("cuda")
- if (
- platform.system() == "Darwin"
- and hasattr(torch.backends, "mps")
- and torch.backends.mps.is_available()
- ):
- return torch.device("mps")
- return torch.device("cpu")
-
-
-# ── Global state ───────────────────────────────────────────────────────────
-
-_model: Optional[Recce] = None
-_face_detector: Optional[MTCNN] = None
-_device: Optional[torch.device] = None
-_load_lock = threading.Lock()
-
-
-def _load_models() -> None:
- """Load RECCE detector and MTCNN face detector (thread-safe)."""
- global _model, _face_detector, _device
-
- if _model is not None:
- return
-
- with _load_lock:
- # Double-check after acquiring lock
- if _model is not None:
- return
-
- _device = _get_device()
- if _device.type == "cuda":
- torch.backends.cudnn.benchmark = True
- torch.set_float32_matmul_precision("high")
- if _device.type == "cuda":
- logger.info(
- "Device: cuda (%s, %.1f GB VRAM)",
- torch.cuda.get_device_name(0),
- torch.cuda.get_device_properties(0).total_memory / 1024**3,
- )
- else:
- logger.warning(
- "Device: %s (no CUDA available"
- " -- check nvidia-container-toolkit)",
- _device,
- )
- logger.info("Loading RECCE model on %s ...", _device)
-
- # ── Face detector (MTCNN) ──────────────────────────────────────
- _face_detector = MTCNN(
- keep_all=True,
- device=_device,
- post_process=False,
- )
-
- # ── RECCE classifier ───────────────────────────────────────────
- # DeepfakeBench trains RECCE with 2-class output (real, fake)
- net = Recce(num_classes=2)
-
- if not os.path.exists(WEIGHTS_PATH):
- raise FileNotFoundError(
- f"RECCE weights not found at {WEIGHTS_PATH}"
- )
-
- checkpoint = torch.load(
- WEIGHTS_PATH, map_location="cpu", weights_only=False
- )
-
- # DeepfakeBench wraps the RECCE model under a 'model.' prefix.
- # Strip it so the keys match the original Recce class.
- if any(k.startswith("model.") for k in checkpoint.keys()):
- state_dict = {
- k[len("model."):]: v
- for k, v in checkpoint.items()
- if k.startswith("model.")
- }
- else:
- state_dict = checkpoint
-
- net.load_state_dict(state_dict)
- net = net.to(_device)
- net.eval()
-
- _model = net
- logger.info("RECCE model loaded successfully.")
-
-
-def _is_model_loaded() -> bool:
- """Return True if both the classifier and face detector are loaded."""
- return _model is not None and _face_detector is not None
-
-
-# ── FastAPI app ────────────────────────────────────────────────────────────
-
-app = FastAPI(
- title="RECCE Detection Service",
- description=(
- "End-to-End Reconstruction-Classification Learning "
- "for Face Forgery Detection (Xception backbone, CVPR 2022)"
- ),
- version="1.0.0",
-)
-
-
-class PredictRequest(BaseModel):
- """Incoming prediction request."""
-
- video_data: str # Base64-encoded video bytes
- threshold: float = 0.5
-
-
-class PredictResponse(BaseModel):
- """Outgoing prediction result."""
-
- model_config = ConfigDict(populate_by_name=True)
-
- model: str = "recce_detection"
- probability: float
- prediction: int
- class_name: str = Field(..., alias="class")
- inference_time: float
- metadata: Dict[str, Any]
-
-
-@app.on_event("startup")
-async def startup_event():
- """Optionally preload model at startup."""
- if PRELOAD_MODEL:
- _load_models()
-
-
-@app.get("/")
-def root():
- """Service info endpoint."""
- return {
- "service": "recce_detection",
- "port": MODEL_PORT,
- "model_loaded": _is_model_loaded(),
- "device": str(_device) if _device else "unknown",
- }
-
-
-def _gpu_health_info() -> dict:
- """Return GPU metrics for the health endpoint."""
- if (
- torch.cuda.is_available()
- and _device is not None
- and _device.type == "cuda"
- ):
- return {
- "gpu_name": torch.cuda.get_device_name(0),
- "vram_used_mb": round(
- torch.cuda.memory_allocated(0) / 1024**2
- ),
- "vram_total_mb": round(
- torch.cuda.get_device_properties(0).total_memory / 1024**2
- ),
- }
- return {}
-
-
-@app.get("/health")
-def health():
- """Health check endpoint."""
- return {
- "status": "healthy",
- "model": "recce_detection",
- "device": str(_device) if _device else "cpu",
- "model_loaded": _is_model_loaded(),
- "weights_exist": os.path.exists(WEIGHTS_PATH),
- **_gpu_health_info(),
- }
-
-
-# ── Video / face utilities ─────────────────────────────────────────────────
-
-
-def _extract_frames(
- video_path: str, num_frames: int = NUM_FRAMES
-) -> List[np.ndarray]:
- """Uniformly sample *num_frames* RGB frames from a video file.
-
- Args:
- video_path: Path to the video on disk.
- num_frames: Number of frames to extract.
-
- Returns:
- List of RGB uint8 numpy arrays (H, W, 3).
- """
- cap = cv2.VideoCapture(video_path)
- total = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))
-
- if total <= 0:
- cap.release()
- return []
-
- indices = np.linspace(
- 0, total - 1, num_frames, endpoint=True, dtype=int
- )
- frames: List[np.ndarray] = []
-
- for idx in indices:
- cap.set(cv2.CAP_PROP_POS_FRAMES, int(idx))
- ret, frame = cap.read()
- if ret:
- frames.append(cv2.cvtColor(frame, cv2.COLOR_BGR2RGB))
-
- cap.release()
- return frames
-
-
-def _crop_face(
- img: np.ndarray,
- bbox: Tuple[float, float, float, float],
- margin: float = MARGIN_FACTOR,
-) -> np.ndarray:
- """Crop a face region from an image with a relative margin.
-
- Args:
- img: RGB image array (H, W, 3).
- bbox: (x0, y0, x1, y1) face bounding box.
- margin: Fraction of bbox dimension to add as padding.
-
- Returns:
- Cropped face region as a numpy array.
- """
- h_img, w_img = img.shape[:2]
- x0, y0, x1, y1 = bbox
- w = x1 - x0
- h = y1 - y0
-
- x0_new = max(0, int(x0 - w * margin / 2))
- x1_new = min(w_img, int(x1 + w * margin / 2) + 1)
- y0_new = max(0, int(y0 - h * margin / 2))
- y1_new = min(h_img, int(y1 + h * margin / 2) + 1)
-
- return img[y0_new:y1_new, x0_new:x1_new]
-
-
-def _detect_and_crop_faces(frame: np.ndarray) -> List[np.ndarray]:
- """Detect faces in a single frame and return cropped+resized chips.
-
- Uses MTCNN for face detection, then crops with margin and resizes
- to IMAGE_SIZE (299x299).
-
- Args:
- frame: RGB image array (H, W, 3).
-
- Returns:
- List of face crops resized to IMAGE_SIZE, as uint8 arrays.
- """
- assert _face_detector is not None
-
- pil_img = Image.fromarray(frame)
- boxes, _ = _face_detector.detect(pil_img)
-
- if boxes is None or len(boxes) == 0:
- return []
-
- crops: List[np.ndarray] = []
- for box in boxes:
- x0, y0, x1, y1 = box.tolist()
- face = _crop_face(frame, (x0, y0, x1, y1))
- if face.size == 0:
- continue
- resized = cv2.resize(face, IMAGE_SIZE)
- crops.append(resized)
-
- return crops
-
-
-def _preprocess_face(face_crop: np.ndarray) -> torch.Tensor:
- """Apply RECCE-specific preprocessing to a face crop.
-
- RECCE uses Normalize(mean=[0.5]*3, std=[0.5]*3) which maps
- [0, 255] uint8 to [-1, 1] float32. This matches the
- albumentations pipeline in the original inference.py.
-
- Args:
- face_crop: RGB uint8 array of shape (299, 299, 3).
-
- Returns:
- Tensor of shape (3, 299, 299) in range [-1, 1].
- """
- # uint8 [0, 255] -> float32 [0, 1] -> normalized [-1, 1]
- tensor = torch.tensor(face_crop).permute(2, 0, 1).float().div(255.0)
- tensor = (tensor - 0.5) / 0.5
- return tensor
-
-
-# ── Prediction endpoint ────────────────────────────────────────────────────
-
-
-@app.post("/predict", response_model=PredictResponse)
-async def predict(request: PredictRequest):
- """Run RECCE face-forgery detection on a base64-encoded video.
-
- Pipeline:
- 1. Decode video and write to temp file.
- 2. Extract uniformly-sampled frames.
- 3. Detect and crop faces per frame (MTCNN).
- 4. Preprocess each face crop (resize 299x299, normalize to [-1,1]).
- 5. Classify each face crop with RECCE (2-class softmax).
- 6. For each frame, take the max fake probability across faces.
- 7. Average the per-frame max probabilities.
-
- If no faces are detected in any frame the service returns
- probability=0.5 (undetermined) rather than raising an error.
- """
- if not _is_model_loaded():
- _load_models()
-
- start_time = time.time()
-
- # ── Decode video ───────────────────────────────────────────────
- with tempfile.NamedTemporaryFile(suffix=".mp4", delete=False) as tmp:
- try:
- video_bytes = base64.b64decode(request.video_data)
- tmp.write(video_bytes)
- tmp_path = tmp.name
- except Exception as e:
- raise HTTPException(
- status_code=400, detail=f"Failed to decode video: {e}"
- )
-
- try:
- # ── Extract frames ─────────────────────────────────────────
- frames = _extract_frames(tmp_path, NUM_FRAMES)
- if not frames:
- raise HTTPException(
- status_code=400,
- detail="Could not extract frames from video.",
- )
-
- # ── Detect faces and classify ──────────────────────────────
- per_frame_max: List[float] = []
- total_faces = 0
-
- for frame in frames:
- crops = _detect_and_crop_faces(frame)
- if not crops:
- continue
-
- # Preprocess and build batch tensor
- tensors = [_preprocess_face(c) for c in crops]
- batch_tensor = torch.stack(tensors).to(_device)
-
- with torch.no_grad():
- logits = _model(batch_tensor)
- # RECCE forward() calls squeeze() which drops the
- # batch dim when batch=1, producing shape (2,)
- # instead of (1, 2). Always ensure 2D.
- if logits.dim() == 1:
- logits = logits.unsqueeze(0)
- probs = F.softmax(logits, dim=1)[:, 1] # fake prob
-
- frame_max = probs.max().cpu().item()
- per_frame_max.append(frame_max)
- total_faces += len(crops)
-
- # ── Aggregate ──────────────────────────────────────────────
- if per_frame_max:
- probability = float(np.mean(per_frame_max))
- else:
- # No faces detected in any frame -- undetermined
- probability = 0.5
-
- prediction = 1 if probability >= request.threshold else 0
- class_name = "fake" if prediction == 1 else "real"
-
- return PredictResponse(
- probability=probability,
- prediction=prediction,
- class_name=class_name,
- inference_time=time.time() - start_time,
- metadata={
- "frames_sampled": len(frames),
- "frames_with_faces": len(per_frame_max),
- "total_faces_detected": total_faces,
- "device": str(_device),
- },
- )
-
- except HTTPException:
- raise
- except Exception as e:
- logger.exception("Error during RECCE prediction")
- raise HTTPException(status_code=500, detail=str(e))
- finally:
- if os.path.exists(tmp_path):
- os.remove(tmp_path)
- gc.collect()
-
-
-if __name__ == "__main__":
- uvicorn.run(app, host="0.0.0.0", port=MODEL_PORT)
diff --git a/video/recce/model_code/.gitignore b/video/recce/model_code/.gitignore
deleted file mode 100644
index 4c59502f6d5d0ac94d4206f836bb4bb32f975057..0000000000000000000000000000000000000000
--- a/video/recce/model_code/.gitignore
+++ /dev/null
@@ -1,134 +0,0 @@
-# ignore directory
-runs/
-.idea/
-
-
-# Byte-compiled / optimized / DLL files
-__pycache__/
-*.py[cod]
-*$py.class
-
-# C extensions
-*.so
-
-# Distribution / packaging
-.Python
-build/
-develop-eggs/
-dist/
-downloads/
-eggs/
-.eggs/
-lib/
-lib64/
-parts/
-sdist/
-var/
-wheels/
-pip-wheel-metadata/
-share/python-wheels/
-*.egg-info/
-.installed.cfg
-*.egg
-MANIFEST
-
-# PyInstaller
-# Usually these files are written by a python script from a template
-# before PyInstaller builds the exe, so as to inject date/other infos into it.
-*.manifest
-*.spec
-
-# Installer logs
-pip-log.txt
-pip-delete-this-directory.txt
-
-# Unit test / coverage reports
-htmlcov/
-.tox/
-.nox/
-.coverage
-.coverage.*
-.cache
-nosetests.xml
-coverage.xml
-*.cover
-*.py,cover
-.hypothesis/
-.pytest_cache/
-
-# Translations
-*.mo
-*.pot
-
-# Django stuff:
-*.log
-local_settings.py
-db.sqlite3
-db.sqlite3-journal
-
-# Flask stuff:
-instance/
-.webassets-cache
-
-# Scrapy stuff:
-.scrapy
-
-# Sphinx documentation
-docs/_build/
-
-# PyBuilder
-target/
-
-# Jupyter Notebook
-.ipynb_checkpoints
-
-# IPython
-profile_default/
-ipython_config.py
-
-# pyenv
-.python-version
-
-# pipenv
-# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
-# However, in case of collaboration, if having platform-specific dependencies or dependencies
-# having no cross-platform support, pipenv may install dependencies that don't work, or not
-# install all needed dependencies.
-#Pipfile.lock
-
-# PEP 582; used by e.g. github.com/David-OConnor/pyflow
-__pypackages__/
-
-# Celery stuff
-celerybeat-schedule
-celerybeat.pid
-
-# SageMath parsed files
-*.sage.py
-
-# Environments
-.env
-.venv
-env/
-venv/
-ENV/
-env.bak/
-venv.bak/
-
-# Spyder project settings
-.spyderproject
-.spyproject
-
-# Rope project settings
-.ropeproject
-
-# mkdocs documentation
-/site
-
-# mypy
-.mypy_cache/
-.dmypy.json
-dmypy.json
-
-# Pyre type checker
-.pyre/
diff --git a/video/recce/model_code/LICENSE b/video/recce/model_code/LICENSE
deleted file mode 100644
index 0db8340366af7e7f31596692ad9de4e746e1e3da..0000000000000000000000000000000000000000
--- a/video/recce/model_code/LICENSE
+++ /dev/null
@@ -1,21 +0,0 @@
-MIT License
-
-Copyright (c) 2022 SJTU Vision and Learning Group
-
-Permission is hereby granted, free of charge, to any person obtaining a copy
-of this software and associated documentation files (the "Software"), to deal
-in the Software without restriction, including without limitation the rights
-to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
-copies of the Software, and to permit persons to whom the Software is
-furnished to do so, subject to the following conditions:
-
-The above copyright notice and this permission notice shall be included in all
-copies or substantial portions of the Software.
-
-THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
-IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
-FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
-AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
-LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
-OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
-SOFTWARE.
diff --git a/video/recce/model_code/README.md b/video/recce/model_code/README.md
deleted file mode 100644
index 1547c5c58097d5a9fc718458e3a8c4fcf9ddf5a8..0000000000000000000000000000000000000000
--- a/video/recce/model_code/README.md
+++ /dev/null
@@ -1,82 +0,0 @@
-# RECCE CVPR 2022
-
-:page_facing_up: End-to-End Reconstruction-Classification Learning for Face Forgery Detection
-
-:boy: Junyi Cao, Chao Ma, Taiping Yao, Shen Chen, Shouhong Ding, Xiaokang Yang
-
-**Please consider citing our paper if you find it interesting or helpful to your research.**
-```
-@InProceedings{Cao_2022_CVPR,
- author = {Cao, Junyi and Ma, Chao and Yao, Taiping and Chen, Shen and Ding, Shouhong and Yang, Xiaokang},
- title = {End-to-End Reconstruction-Classification Learning for Face Forgery Detection},
- booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)},
- month = {June},
- year = {2022},
- pages = {4113-4122}
-}
-```
-
-----
-
-### Introduction
-
-This repository is an implementation for *End-to-End Reconstruction-Classification Learning for Face Forgery Detection* presented in CVPR 2022. In the paper, we propose a novel **REC**onstruction-**C**lassification l**E**arning framework called **RECCE** to detect face forgeries. The code is based on Pytorch. Please follow the instructions below to get started.
-
-
-### Motivation
-
-Briefly, we train a reconstruction network over genuine images only and use the output of the latent feature by the encoder to perform binary classification. Due to the discrepancy in the data distribution between genuine and forged faces, the reconstruction differences of forged faces are obvious and also indicate the probably forged regions.
-
-
-### Basic Requirements
-Please ensure that you have already installed the following packages.
-- [Pytorch](https://pytorch.org/get-started/previous-versions/) 1.7.1
-- [Torchvision](https://pytorch.org/get-started/previous-versions/) 0.8.2
-- [Albumentations](https://github.com/albumentations-team/albumentations#spatial-level-transforms) 1.0.3
-- [Timm](https://github.com/rwightman/pytorch-image-models) 0.3.4
-- [TensorboardX](https://pypi.org/project/tensorboardX/#history) 2.1
-- [Scipy](https://pypi.org/project/scipy/#history) 1.5.2
-- [PyYaml](https://pypi.org/project/PyYAML/#history) 5.3.1
-
-### Dataset Preparation
-- We include the dataset loaders for several commonly-used face forgery datasets, *i.e.,* [FaceForensics++](https://github.com/ondyari/FaceForensics), [Celeb-DF](https://www.cs.albany.edu/~lsw/celeb-deepfakeforensics.html), [WildDeepfake](https://github.com/deepfakeinthewild/deepfake-in-the-wild), and [DFDC](https://ai.facebook.com/datasets/dfdc). You can enter the dataset website to download the original data.
-- For FaceForensics++, Celeb-DF, and DFDC, since the original data are in video format, you should first extract the facial images from the sequences and store them. We use [RetinaFace](https://github.com/biubug6/Pytorch_Retinaface) to do this.
-
-### Config Files
-- We have already provided the config templates in `config/`. You can adjust the parameters in the yaml files to specify a training process. More information is presented in [config/README.md](./config/README.md).
-
-### Training
-- We use `torch.distributed` package to train the models, for more information, please refer to [PyTorch Distributed Overview](https://pytorch.org/tutorials/beginner/dist_overview.html).
-- To train a model, run the following script in your console.
-```{bash}
-CUDA_VISIBLE_DEVICES=0 python -m torch.distributed.launch --nproc_per_node=1 --master_port 12345 train.py --config path/to/config.yaml
-```
-- `--config`: Specify the path of the config file.
-
-### Testing
-- To test a model, run the following script in your console.
-```{bash}
-python test.py --config path/to/config.yaml
-```
-- `--config`: Specify the path of the config file.
-
-### Inference
-- We provide the script in `inference.py` to help you do inference using custom data.
-- To do inference, run the following script in your console.
-```{bash}
-python inference.py --bin path/to/model.bin --image_folder path/to/image_folder --device $DEVICE --image_size $IMAGE_SIZE
-```
-- `--bin`: Specify the path of the model bin generated by the training script of this project.
-- `--image_folder`: Specify the directory of custom facial images. The script accepts images end with `.jpg` or `.png`.
-- `--device`: Specify the device to run the experiment, e.g., `cpu`, `cuda:0`.
-- `--image_size`: Specify the spatial size of input images.
-- The program will output the fake probability for each input image like this:
- ```
- path: path/to/image1.jpg | fake probability: 0.1296 | prediction: real
- path: path/to/image2.jpg | fake probability: 0.9146 | prediction: fake
- ```
-- Type `python inference.py -h` in your console for more information about available arguments.
-
-
-### Acknowledgement
-- We thank Qiqi Gu for helping plot the schematic diagram of the proposed method in the manuscript.
diff --git a/video/recce/model_code/config/README.md b/video/recce/model_code/config/README.md
deleted file mode 100644
index dfb996189a40279bef13deb836f4b1b87ef0e57f..0000000000000000000000000000000000000000
--- a/video/recce/model_code/config/README.md
+++ /dev/null
@@ -1,50 +0,0 @@
-## Configuration Files
-
-#### Model Configuration
-- We use a yaml file to specify the hyperparameters of a model. All the training logs will be placed in `${project_root}/runs/${model_name}/${experiment_id}`. An example are shown below.
-
-```yaml
-model:
- name: Recce # Model Name
- num_classes: 1
-config:
- lambda_1: 0.1 # balancing weight for L_r
- lambda_2: 0.1 # balancing weight for L_m
- distribute:
- backend: nccl
- optimizer:
- name: adam
- lr: 0.0002
- weight_decay: 0.00001
- scheduler:
- name: StepLR
- step_size: 22500
- gamma: 0.5
- resume: False
- resume_best: False
- id: FF++c40 # Specify a unique experiment id.
- loss: binary_ce # Loss type, either 'binary_ce' or 'cross_entropy'.
- metric: Acc # Main metric, either 'Acc', 'AUC', or 'LogLoss'.
- debug: False
- device: "cuda:1" # NOTE: Used only when testing, annotation this line when training.
- ckpt: best_model_1000 # NOTE: Used only when testing to specify a checkpoint id, annotating this line when training.
-data:
- train_batch_size: 32
- val_batch_size: 64
- test_batch_size: 64
- name: FaceForensics
- file: "./config/dataset/faceforensics.yml" # config file for a dataset
- train_branch: "train_cfg"
- val_branch: "test_cfg"
- test_branch: "test_cfg"
-```
-
-- We set different hyper-parameters for the learning rate scheduler according to the used dataset as follows:
- - FaceForensics++: The learning rate is decayed by 0.5 every 10 epochs.
- - Celeb-DF: The learning rate is decayed by 0.5 every 10 epochs.
- - WildDeepfake: The learning rate is decayed by 0.9 every 3000 iterations.
- - DFDC: The learning rate is decayed by 0.5 every 3 epochs.
-
-#### Dataset Configuration
-- We also use a yaml file to specify the dataset to load for the experiment. These files are placed under `config/dataset/` subfold.
-- Briefly, you should change the `root` parameter according to your storage path.
\ No newline at end of file
diff --git a/video/recce/model_code/config/Recce.yml b/video/recce/model_code/config/Recce.yml
deleted file mode 100644
index beb2f60001712ef0ceab1c03bfd0d2b46aa832b9..0000000000000000000000000000000000000000
--- a/video/recce/model_code/config/Recce.yml
+++ /dev/null
@@ -1,33 +0,0 @@
-model:
- name: Recce
- num_classes: 1
-config:
- lambda_1: 0.1
- lambda_2: 0.1
- distribute:
- backend: nccl
- optimizer:
- name: adam
- lr: 0.0002
- weight_decay: 0.00001
- scheduler:
- name: StepLR
- step_size: 22500
- gamma: 0.5
- resume: False
- resume_best: False
- id: FF++c40
- loss: binary_ce
- metric: Acc
- debug: False
-# device: "cuda:1"
-# ckpt: best_model_1000
-data:
- train_batch_size: 32
- val_batch_size: 64
- test_batch_size: 64
- name: FaceForensics
- file: "./config/dataset/faceforensics.yml"
- train_branch: "train_cfg"
- val_branch: "test_cfg"
- test_branch: "test_cfg"
diff --git a/video/recce/model_code/config/dataset/celeb_df.yml b/video/recce/model_code/config/dataset/celeb_df.yml
deleted file mode 100644
index a70cbe9fdf008e8f0653a131767f90296259c7e2..0000000000000000000000000000000000000000
--- a/video/recce/model_code/config/dataset/celeb_df.yml
+++ /dev/null
@@ -1,32 +0,0 @@
-train_cfg:
- root: "path/to/data"
- split: "train"
- balance: True
- log_steps: 1000
- val_steps: 1000
- num_steps: 92000
- transforms:
- - name: "Resize"
- params:
- height: 299
- width: 299
- - name: "HorizontalFlip"
- params:
- p: 0.5
- - name: "Normalize"
- params:
- mean: [0.5, 0.5, 0.5]
- std: [0.5, 0.5, 0.5]
-test_cfg:
- root: "path/to/data"
- split: "test"
- balance: False
- transforms:
- - name: "Resize"
- params:
- height: 299
- width: 299
- - name: "Normalize"
- params:
- mean: [0.5, 0.5, 0.5]
- std: [0.5, 0.5, 0.5]
\ No newline at end of file
diff --git a/video/recce/model_code/config/dataset/dfdc.yml b/video/recce/model_code/config/dataset/dfdc.yml
deleted file mode 100644
index 097a71a2c50537c7d6e6a5a3fb3b1035ec8ba84a..0000000000000000000000000000000000000000
--- a/video/recce/model_code/config/dataset/dfdc.yml
+++ /dev/null
@@ -1,30 +0,0 @@
-train_cfg:
- root: "path/to/data"
- split: "train"
- log_steps: 1000
- val_steps: 1000
- num_steps: 100000
- transforms:
- - name: "Resize"
- params:
- height: 299
- width: 299
- - name: "HorizontalFlip"
- params:
- p: 0.5
- - name: "Normalize"
- params:
- mean: [0.5, 0.5, 0.5]
- std: [0.5, 0.5, 0.5]
-test_cfg:
- root: "path/to/data"
- split: "test"
- transforms:
- - name: "Resize"
- params:
- height: 299
- width: 299
- - name: "Normalize"
- params:
- mean: [0.5, 0.5, 0.5]
- std: [0.5, 0.5, 0.5]
\ No newline at end of file
diff --git a/video/recce/model_code/config/dataset/faceforensics.yml b/video/recce/model_code/config/dataset/faceforensics.yml
deleted file mode 100644
index e67c9625fd801fb726745b266c89245f4ef856d2..0000000000000000000000000000000000000000
--- a/video/recce/model_code/config/dataset/faceforensics.yml
+++ /dev/null
@@ -1,34 +0,0 @@
-train_cfg:
- root: "path/to/data"
- split: "train"
- method: "all"
- compression: "c40"
- log_steps: 1000
- val_steps: 1000
- num_steps: 90000
- transforms:
- - name: "Resize"
- params:
- height: 299
- width: 299
- - name: "HorizontalFlip"
- params:
- p: 0.5
- - name: "Normalize"
- params:
- mean: [0.5, 0.5, 0.5]
- std: [0.5, 0.5, 0.5]
-test_cfg:
- root: "path/to/data"
- split: "test"
- method: "all"
- compression: "c40"
- transforms:
- - name: "Resize"
- params:
- height: 299
- width: 299
- - name: "Normalize"
- params:
- mean: [0.5, 0.5, 0.5]
- std: [0.5, 0.5, 0.5]
\ No newline at end of file
diff --git a/video/recce/model_code/config/dataset/wilddeepfake.yml b/video/recce/model_code/config/dataset/wilddeepfake.yml
deleted file mode 100644
index 630f31b9e0e20c405d5e5c51b4b48ae9f0a7ac04..0000000000000000000000000000000000000000
--- a/video/recce/model_code/config/dataset/wilddeepfake.yml
+++ /dev/null
@@ -1,32 +0,0 @@
-train_cfg:
- root: "path/to/data"
- split: "train"
- num_image_train:
- log_steps: 200
- val_steps: 200
- num_steps: 40000
- transforms:
- - name: "Resize"
- params:
- height: 224
- width: 224
- - name: "HorizontalFlip"
- params:
- p: 0.5
- - name: "Normalize"
- params:
- mean: [0.5, 0.5, 0.5]
- std: [0.5, 0.5, 0.5]
-test_cfg:
- root: "path/to/data"
- split: "test"
- num_image_test:
- transforms:
- - name: "Resize"
- params:
- height: 224
- width: 224
- - name: "Normalize"
- params:
- mean: [0.5, 0.5, 0.5]
- std: [0.5, 0.5, 0.5]
\ No newline at end of file
diff --git a/video/recce/model_code/dataset/__init__.py b/video/recce/model_code/dataset/__init__.py
deleted file mode 100644
index 27584c89c9379621bffb513aafafe2cc1bd41b8c..0000000000000000000000000000000000000000
--- a/video/recce/model_code/dataset/__init__.py
+++ /dev/null
@@ -1,17 +0,0 @@
-from .abstract_dataset import AbstractDataset
-from .faceforensics import FaceForensics
-from .wild_deepfake import WildDeepfake
-from .celeb_df import CelebDF
-from .dfdc import DFDC
-
-LOADERS = {
- "FaceForensics": FaceForensics,
- "WildDeepfake": WildDeepfake,
- "CelebDF": CelebDF,
- "DFDC": DFDC,
-}
-
-
-def load_dataset(name="FaceForensics"):
- print(f"Loading dataset: '{name}'...")
- return LOADERS[name]
diff --git a/video/recce/model_code/dataset/abstract_dataset.py b/video/recce/model_code/dataset/abstract_dataset.py
deleted file mode 100644
index 5b029f8c2b3d0a41c383abfd235ceb41e028e8fc..0000000000000000000000000000000000000000
--- a/video/recce/model_code/dataset/abstract_dataset.py
+++ /dev/null
@@ -1,41 +0,0 @@
-import cv2
-import torch
-import numpy as np
-from torchvision.datasets import VisionDataset
-import albumentations
-from albumentations import Compose
-from albumentations.pytorch.transforms import ToTensorV2
-
-
-class AbstractDataset(VisionDataset):
- def __init__(self, cfg, seed=2022, transforms=None, transform=None, target_transform=None):
- super(AbstractDataset, self).__init__(cfg['root'], transforms=transforms,
- transform=transform, target_transform=target_transform)
- # fix for re-production
- np.random.seed(seed)
-
- self.images = list()
- self.targets = list()
- self.split = cfg['split']
- if self.transforms is None:
- self.transforms = Compose(
- [getattr(albumentations, _['name'])(**_['params']) for _ in cfg['transforms']] +
- [ToTensorV2()]
- )
-
- def __len__(self):
- return len(self.images)
-
- def __getitem__(self, index):
- path = self.images[index]
- tgt = self.targets[index]
- return path, tgt
-
- def load_item(self, items):
- images = list()
- for item in items:
- img = cv2.imread(item)
- img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)
- image = self.transforms(image=img)['image']
- images.append(image)
- return torch.stack(images, dim=0)
diff --git a/video/recce/model_code/dataset/celeb_df.py b/video/recce/model_code/dataset/celeb_df.py
deleted file mode 100644
index f5bd36653918e0da454c7bf5988561a2b9d885a8..0000000000000000000000000000000000000000
--- a/video/recce/model_code/dataset/celeb_df.py
+++ /dev/null
@@ -1,126 +0,0 @@
-import numpy as np
-from glob import glob
-from os import listdir
-from os.path import join
-from dataset import AbstractDataset
-
-SPLITS = ["train", "test"]
-
-
-class CelebDF(AbstractDataset):
- """
- Celeb-DF v2 Dataset proposed in "Celeb-DF: A Large-scale Challenging Dataset for DeepFake Forensics".
- """
-
- def __init__(self, cfg, seed=2022, transforms=None, transform=None, target_transform=None):
- # pre-check
- if cfg['split'] not in SPLITS:
- raise ValueError(f"split should be one of {SPLITS}, but found {cfg['split']}.")
- super(CelebDF, self).__init__(cfg, seed, transforms, transform, target_transform)
- print(f"Loading data from 'Celeb-DF' of split '{cfg['split']}'"
- f"\nPlease wait patiently...")
- self.categories = ['original', 'fake']
- self.root = cfg['root']
- images_ids = self.__get_images_ids()
- test_ids = self.__get_test_ids()
- train_ids = [images_ids[0] - test_ids[0],
- images_ids[1] - test_ids[1],
- images_ids[2] - test_ids[2]]
- self.images, self.targets = self.__get_images(
- test_ids if cfg['split'] == "test" else train_ids, cfg['balance'])
- assert len(self.images) == len(self.targets), "The number of images and targets not consistent."
- print("Data from 'Celeb-DF' loaded.\n")
- print(f"Dataset contains {len(self.images)} images.\n")
-
- def __get_images_ids(self):
- youtube_real = listdir(join(self.root, 'YouTube-real', 'images'))
- celeb_real = listdir(join(self.root, 'Celeb-real', 'images'))
- celeb_fake = listdir(join(self.root, 'Celeb-synthesis', 'images'))
- return set(youtube_real), set(celeb_real), set(celeb_fake)
-
- def __get_test_ids(self):
- youtube_real = set()
- celeb_real = set()
- celeb_fake = set()
- with open(join(self.root, "List_of_testing_videos.txt"), "r", encoding="utf-8") as f:
- contents = f.readlines()
- for line in contents:
- name = line.split(" ")[-1]
- number = name.split("/")[-1].split(".")[0]
- if "YouTube-real" in name:
- youtube_real.add(number)
- elif "Celeb-real" in name:
- celeb_real.add(number)
- elif "Celeb-synthesis" in name:
- celeb_fake.add(number)
- else:
- raise ValueError("'List_of_testing_videos.txt' file corrupted.")
- return youtube_real, celeb_real, celeb_fake
-
- def __get_images(self, ids, balance=False):
- real = list()
- fake = list()
- # YouTube-real
- for _ in ids[0]:
- real.extend(glob(join(self.root, 'YouTube-real', 'images', _, '*.png')))
- # Celeb-real
- for _ in ids[1]:
- real.extend(glob(join(self.root, 'Celeb-real', 'images', _, '*.png')))
- # Celeb-synthesis
- for _ in ids[2]:
- fake.extend(glob(join(self.root, 'Celeb-synthesis', 'images', _, '*.png')))
- print(f"Real: {len(real)}, Fake: {len(fake)}")
- if balance:
- fake = np.random.choice(fake, size=len(real), replace=False)
- print(f"After Balance | Real: {len(real)}, Fake: {len(fake)}")
- real_tgt = [0] * len(real)
- fake_tgt = [1] * len(fake)
- return [*real, *fake], [*real_tgt, *fake_tgt]
-
-
-if __name__ == '__main__':
- import yaml
-
- config_path = "../config/dataset/celeb_df.yml"
- with open(config_path) as config_file:
- config = yaml.load(config_file, Loader=yaml.FullLoader)
- config = config["train_cfg"]
- # config = config["test_cfg"]
-
- def run_dataset():
- dataset = CelebDF(config)
- print(f"dataset: {len(dataset)}")
- for i, _ in enumerate(dataset):
- path, target = _
- print(f"path: {path}, target: {target}")
- if i >= 9:
- break
-
-
- def run_dataloader(display_samples=False):
- from torch.utils import data
- import matplotlib.pyplot as plt
-
- dataset = CelebDF(config)
- dataloader = data.DataLoader(dataset, batch_size=8, shuffle=True)
- print(f"dataset: {len(dataset)}")
- for i, _ in enumerate(dataloader):
- path, targets = _
- image = dataloader.dataset.load_item(path)
- print(f"image: {image.shape}, target: {targets}")
- if display_samples:
- plt.figure()
- img = image[0].permute([1, 2, 0]).numpy()
- plt.imshow(img)
- # plt.savefig("./img_" + str(i) + ".png")
- plt.show()
- if i >= 9:
- break
-
-
- ###########################
- # run the functions below #
- ###########################
-
- # run_dataset()
- run_dataloader(False)
diff --git a/video/recce/model_code/dataset/dfdc.py b/video/recce/model_code/dataset/dfdc.py
deleted file mode 100644
index 098ede98fbe30ffb9afb66daed0416a4458dbd55..0000000000000000000000000000000000000000
--- a/video/recce/model_code/dataset/dfdc.py
+++ /dev/null
@@ -1,124 +0,0 @@
-import json
-from glob import glob
-from os.path import join
-from dataset import AbstractDataset
-
-SPLIT = ["train", "val", "test"]
-LABEL_MAP = {"REAL": 0, "FAKE": 1}
-
-
-class DFDC(AbstractDataset):
- """
- Deepfake Detection Challenge organized by Facebook
- """
-
- def __init__(self, cfg, seed=2022, transforms=None, transform=None, target_transform=None):
- # pre-check
- if cfg['split'] not in SPLIT:
- raise ValueError(f"split should be one of {SPLIT}, but found {cfg['split']}.")
- super(DFDC, self).__init__(cfg, seed, transforms, transform, target_transform)
- print(f"Loading data from 'DFDC' of split '{cfg['split']}'"
- f"\nPlease wait patiently...")
- self.categories = ['original', 'fake']
- self.root = cfg['root']
- self.num_real = 0
- self.num_fake = 0
- if self.split == "test":
- self.__load_test_data()
- elif self.split == "train":
- self.__load_train_data()
- assert len(self.images) == len(self.targets), "Length of images and targets not the same!"
- print(f"Data from 'DFDC' loaded.")
- print(f"Real: {self.num_real}, Fake: {self.num_fake}.")
- print(f"Dataset contains {len(self.images)} images\n")
-
- def __load_test_data(self):
- label_path = join(self.root, "test", "labels.csv")
- with open(label_path, encoding="utf-8") as file:
- content = file.readlines()
- for _ in content:
- if ".mp4" in _:
- key = _.split(".")[0]
- label = _.split(",")[1].strip()
- label = int(label)
- imgs = glob(join(self.root, "test", "images", key, "*.png"))
- num = len(imgs)
- self.images.extend(imgs)
- self.targets.extend([label] * num)
- if label == 0:
- self.num_real += num
- elif label == 1:
- self.num_fake += num
-
- def __load_train_data(self):
- train_folds = glob(join(self.root, "dfdc_train_part_*"))
- for fold in train_folds:
- fold_imgs = list()
- fold_tgts = list()
- metadata_path = join(fold, "metadata.json")
- try:
- with open(metadata_path, "r", encoding="utf-8") as file:
- metadata = json.loads(file.readline())
- for k, v in metadata.items():
- index = k.split(".")[0]
- label = LABEL_MAP[v["label"]]
- imgs = glob(join(fold, "images", index, "*.png"))
- fold_imgs.extend(imgs)
- fold_tgts.extend([label] * len(imgs))
- if label == 0:
- self.num_real += len(imgs)
- elif label == 1:
- self.num_fake += len(imgs)
- self.images.extend(fold_imgs)
- self.targets.extend(fold_tgts)
- except FileNotFoundError:
- continue
-
-
-if __name__ == '__main__':
- import yaml
-
- config_path = "../config/dataset/dfdc.yml"
- with open(config_path) as config_file:
- config = yaml.load(config_file, Loader=yaml.FullLoader)
- config = config["train_cfg"]
- # config = config["test_cfg"]
-
-
- def run_dataset():
- dataset = DFDC(config)
- print(f"dataset: {len(dataset)}")
- for i, _ in enumerate(dataset):
- path, target = _
- print(f"path: {path}, target: {target}")
- if i >= 9:
- break
-
-
- def run_dataloader(display_samples=False):
- from torch.utils import data
- import matplotlib.pyplot as plt
-
- dataset = DFDC(config)
- dataloader = data.DataLoader(dataset, batch_size=8, shuffle=True)
- print(f"dataset: {len(dataset)}")
- for i, _ in enumerate(dataloader):
- path, targets = _
- image = dataloader.dataset.load_item(path)
- print(f"image: {image.shape}, target: {targets}")
- if display_samples:
- plt.figure()
- img = image[0].permute([1, 2, 0]).numpy()
- plt.imshow(img)
- # plt.savefig("./img_" + str(i) + ".png")
- plt.show()
- if i >= 9:
- break
-
-
- ###########################
- # run the functions below #
- ###########################
-
- # run_dataset()
- run_dataloader(False)
diff --git a/video/recce/model_code/dataset/faceforensics.py b/video/recce/model_code/dataset/faceforensics.py
deleted file mode 100644
index baf9fa43f250e6e585a6f1f771e17250373a5f0a..0000000000000000000000000000000000000000
--- a/video/recce/model_code/dataset/faceforensics.py
+++ /dev/null
@@ -1,107 +0,0 @@
-import torch
-import numpy as np
-from os.path import join
-from dataset import AbstractDataset
-
-METHOD = ['all', 'Deepfakes', 'Face2Face', 'FaceSwap', 'NeuralTextures']
-SPLIT = ['train', 'val', 'test']
-COMP2NAME = {'c0': 'raw', 'c23': 'c23', 'c40': 'c40'}
-SOURCE_MAP = {'youtube': 2, 'Deepfakes': 3, 'Face2Face': 4, 'FaceSwap': 5, 'NeuralTextures': 6}
-
-
-class FaceForensics(AbstractDataset):
- """
- FaceForensics++ Dataset proposed in "FaceForensics++: Learning to Detect Manipulated Facial Images"
- """
-
- def __init__(self, cfg, seed=2022, transforms=None, transform=None, target_transform=None):
- # pre-check
- if cfg['split'] not in SPLIT:
- raise ValueError(f"split should be one of {SPLIT}, "
- f"but found {cfg['split']}.")
- if cfg['method'] not in METHOD:
- raise ValueError(f"method should be one of {METHOD}, "
- f"but found {cfg['method']}.")
- if cfg['compression'] not in COMP2NAME.keys():
- raise ValueError(f"compression should be one of {COMP2NAME.keys()}, "
- f"but found {cfg['compression']}.")
- super(FaceForensics, self).__init__(
- cfg, seed, transforms, transform, target_transform)
- print(f"Loading data from 'FF++ {cfg['method']}' of split '{cfg['split']}' "
- f"and compression '{cfg['compression']}'\nPlease wait patiently...")
-
- self.categories = ['original', 'fake']
- # load the path of dataset images
- indices = join(self.root, cfg['split'] + "_" + cfg['compression'] + ".pickle")
- indices = torch.load(indices)
- if cfg['method'] == "all":
- # full dataset
- self.images = [join(cfg['root'], _[0]) for _ in indices]
- self.targets = [_[1] for _ in indices]
- else:
- # specific manipulated method
- self.images = list()
- self.targets = list()
- nums = 0
- for _ in indices:
- if cfg['method'] in _[0]:
- self.images.append(join(cfg['root'], _[0]))
- self.targets.append(_[1])
- nums = len(self.targets)
- ori = list()
- for _ in indices:
- if "original_sequences" in _[0]:
- ori.append(join(cfg['root'], _[0]))
- choices = np.random.choice(ori, size=nums, replace=False)
- self.images.extend(choices)
- self.targets.extend([0] * nums)
- print("Data from 'FF++' loaded.\n")
- print(f"Dataset contains {len(self.images)} images.\n")
-
-
-if __name__ == '__main__':
- import yaml
-
- config_path = "../config/dataset/faceforensics.yml"
- with open(config_path) as config_file:
- config = yaml.load(config_file, Loader=yaml.FullLoader)
- config = config["train_cfg"]
- # config = config["test_cfg"]
-
- def run_dataset():
- dataset = FaceForensics(config)
- print(f"dataset: {len(dataset)}")
- for i, _ in enumerate(dataset):
- path, target = _
- print(f"path: {path}, target: {target}")
- if i >= 9:
- break
-
-
- def run_dataloader(display_samples=False):
- from torch.utils import data
- import matplotlib.pyplot as plt
-
- dataset = FaceForensics(config)
- dataloader = data.DataLoader(dataset, batch_size=8, shuffle=True)
- print(f"dataset: {len(dataset)}")
- for i, _ in enumerate(dataloader):
- path, targets = _
- image = dataloader.dataset.load_item(path)
- print(f"image: {image.shape}, target: {targets}")
- if display_samples:
- plt.figure()
- img = image[0].permute([1, 2, 0]).numpy()
- plt.imshow(img)
- # plt.savefig("./img_" + str(i) + ".png")
- plt.show()
- if i >= 9:
- break
-
-
- ###########################
- # run the functions below #
- ###########################
-
- # run_dataset()
- run_dataloader(False)
diff --git a/video/recce/model_code/dataset/wild_deepfake.py b/video/recce/model_code/dataset/wild_deepfake.py
deleted file mode 100644
index 5f39c1a81f04fadae39dd46d0aae9cefc023b6ab..0000000000000000000000000000000000000000
--- a/video/recce/model_code/dataset/wild_deepfake.py
+++ /dev/null
@@ -1,100 +0,0 @@
-import torch
-import numpy as np
-from os.path import join
-from dataset import AbstractDataset
-
-SPLITS = ["train", "test"]
-
-
-class WildDeepfake(AbstractDataset):
- """
- Wild Deepfake Dataset proposed in "WildDeepfake: A Challenging Real-World Dataset for Deepfake Detection"
- """
-
- def __init__(self, cfg, seed=2022, transforms=None, transform=None, target_transform=None):
- # pre-check
- if cfg['split'] not in SPLITS:
- raise ValueError(f"split should be one of {SPLITS}, but found {cfg['split']}.")
- super(WildDeepfake, self).__init__(cfg, seed, transforms, transform, target_transform)
- print(f"Loading data from 'WildDeepfake' of split '{cfg['split']}'"
- f"\nPlease wait patiently...")
- self.categories = ['original', 'fake']
- self.root = cfg['root']
- self.num_train = cfg.get('num_image_train', None)
- self.num_test = cfg.get('num_image_test', None)
- self.images, self.targets = self.__get_images()
- print(f"Data from 'WildDeepfake' loaded.")
- print(f"Dataset contains {len(self.images)} images.\n")
-
- def __get_images(self):
- if self.split == 'train':
- num = self.num_train
- elif self.split == 'test':
- num = self.num_test
- else:
- num = None
- real_images = torch.load(join(self.root, self.split, "real.pickle"))
- if num is not None:
- real_images = np.random.choice(real_images, num // 3, replace=False)
- real_tgts = [torch.tensor(0)] * len(real_images)
- print(f"real: {len(real_tgts)}")
- fake_images = torch.load(join(self.root, self.split, "fake.pickle"))
- if num is not None:
- fake_images = np.random.choice(fake_images, num - num // 3, replace=False)
- fake_tgts = [torch.tensor(1)] * len(fake_images)
- print(f"fake: {len(fake_tgts)}")
- return real_images + fake_images, real_tgts + fake_tgts
-
- def __getitem__(self, index):
- path = join(self.root, self.split, self.images[index])
- tgt = self.targets[index]
- return path, tgt
-
-
-if __name__ == '__main__':
- import yaml
-
- config_path = "../config/dataset/wilddeepfake.yml"
- with open(config_path) as config_file:
- config = yaml.load(config_file, Loader=yaml.FullLoader)
- config = config["train_cfg"]
- # config = config["test_cfg"]
-
-
- def run_dataset():
- dataset = WildDeepfake(config)
- print(f"dataset: {len(dataset)}")
- for i, _ in enumerate(dataset):
- path, target = _
- print(f"path: {path}, target: {target}")
- if i >= 9:
- break
-
-
- def run_dataloader(display_samples=False):
- from torch.utils import data
- import matplotlib.pyplot as plt
-
- dataset = WildDeepfake(config)
- dataloader = data.DataLoader(dataset, batch_size=8, shuffle=True)
- print(f"dataset: {len(dataset)}")
- for i, _ in enumerate(dataloader):
- path, targets = _
- image = dataloader.dataset.load_item(path)
- print(f"image: {image.shape}, target: {targets}")
- if display_samples:
- plt.figure()
- img = image[0].permute([1, 2, 0]).numpy()
- plt.imshow(img)
- # plt.savefig("./img_" + str(i) + ".png")
- plt.show()
- if i >= 9:
- break
-
-
- ###########################
- # run the functions below #
- ###########################
-
- # run_dataset()
- run_dataloader(False)
diff --git a/video/recce/model_code/inference.py b/video/recce/model_code/inference.py
deleted file mode 100644
index 02c6d70219b2cf02dad6551d1fda5d9946014a4f..0000000000000000000000000000000000000000
--- a/video/recce/model_code/inference.py
+++ /dev/null
@@ -1,129 +0,0 @@
-import cv2
-import torch
-import random
-import argparse
-from glob import glob
-from os.path import join
-from model.network import Recce
-from model.common import freeze_weights
-from albumentations import Compose, Normalize, Resize
-from albumentations.pytorch.transforms import ToTensorV2
-
-# fix random seed
-seed = 0
-random.seed(seed)
-torch.manual_seed(seed)
-torch.cuda.manual_seed(seed)
-torch.cuda.manual_seed_all(seed)
-
-parser = argparse.ArgumentParser(description="This code helps you use a trained model to "
- "do inference.")
-parser.add_argument("--weight", "-w",
- type=str,
- default=None,
- help="Specify the path to the model weight (the state dict file). "
- "Do not use this argument when '--bin' is set.")
-parser.add_argument("--bin", "-b",
- type=str,
- default=None,
- help="Specify the path to the model bin which ends up with '.bin' "
- "(which is generated by the trainer of this project). "
- "Do not use this argument when '--weight' is set.")
-parser.add_argument("--image", "-i",
- type=str,
- default=None,
- help="Specify the path to the input image. "
- "Do not use this argument when '--image_folder' is set.")
-parser.add_argument("--image_folder", "-f",
- type=str,
- default=None,
- help="Specify the directory to evaluate all the images. "
- "Do not use this argument when '--image' is set.")
-parser.add_argument('--device', '-d', type=str,
- default="cpu",
- help="Specify the device to load the model. Default: 'cpu'.")
-parser.add_argument('--image_size', '-s', type=int,
- default=299,
- help="Specify the spatial size of the input image(s). Default: 299.")
-parser.add_argument('--visualize', '-v', action="store_true",
- default=False, help='Visualize images.')
-
-
-def preprocess(file_path):
- img = cv2.imread(file_path)
- img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)
- compose = Compose([Resize(height=args.image_size, width=args.image_size),
- Normalize(mean=[0.5] * 3, std=[0.5] * 3),
- ToTensorV2()])
- img = compose(image=img)['image'].unsqueeze(0)
- return img
-
-
-def prepare_data():
- paths = list()
- images = list()
- # check the console arguments
- if args.image and args.image_folder:
- raise ValueError("Only one of '--image' or '--image_folder' can be set.")
- elif args.image:
- images.append(preprocess(args.image))
- paths.append(args.image)
- elif args.image_folder:
- image_paths = glob(join(args.image_folder, "*.jpg"))
- image_paths.extend(glob(join(args.image_folder, "*.png")))
- for _ in image_paths:
- images.append(preprocess(_))
- paths.append(_)
- else:
- raise ValueError("Neither of '--image' nor '--image_folder' is set. Please specify either "
- "one of these two arguments to load input image(s) properly.")
- return paths, images
-
-
-def inference(model, images, paths, device):
- for img, pt in zip(images, paths):
- img = img.to(device)
- prediction = model(img)
- prediction = torch.sigmoid(prediction).cpu()
- fake = True if prediction >= 0.5 else False
- print(f"path: {pt} \t\t| fake probability: {prediction.item():.4f} \t| "
- f"prediction: {'fake' if fake else 'real'}")
- if args.visualize:
- cvimg = cv2.imread(pt)
- cvimg = cv2.putText(cvimg, f'p: {prediction.item():.2f}, ' + f"{'fake' if fake else 'real'}",
- (5, 50), cv2.FONT_HERSHEY_SIMPLEX, 0.5,
- (0, 0, 255) if fake else (255, 0, 0), 2)
- cv2.imshow("image", cvimg)
- cv2.waitKey(0)
- cv2.destroyWindow("image")
-
-
-def main():
- print("Arguments:\n", args, end="\n\n")
- # set device
- device = torch.device(args.device)
- # load model
- model = eval("Recce")(num_classes=1)
- # check the console arguments
- if args.weight and args.bin:
- raise ValueError("Only one of '--weight' or '--bin' can be set.")
- elif args.weight:
- weights = torch.load(args.weight, map_location="cpu")
- elif args.bin:
- weights = torch.load(args.bin, map_location="cpu")["model"]
- else:
- raise ValueError("Neither of '--weight' nor '--bin' is set. Please specify either "
- "one of these two arguments to load model's weight properly.")
- model.load_state_dict(weights)
- model = model.to(device)
- freeze_weights(model)
- model.eval()
-
- paths, images = prepare_data()
- print("Inference:")
- inference(model, images=images, paths=paths, device=device)
-
-
-if __name__ == '__main__':
- args = parser.parse_args()
- main()
diff --git a/video/recce/model_code/loss/__init__.py b/video/recce/model_code/loss/__init__.py
deleted file mode 100644
index fb8d2007cb4a3f852c7bde93fe84b0e162408d71..0000000000000000000000000000000000000000
--- a/video/recce/model_code/loss/__init__.py
+++ /dev/null
@@ -1,12 +0,0 @@
-import torch.nn as nn
-
-
-def get_loss(name="cross_entropy", device="cuda:0"):
- print(f"Using loss: '{LOSSES[name]}'")
- return LOSSES[name].to(device)
-
-
-LOSSES = {
- "binary_ce": nn.BCEWithLogitsLoss(),
- "cross_entropy": nn.CrossEntropyLoss()
-}
diff --git a/video/recce/model_code/model/__init__.py b/video/recce/model_code/model/__init__.py
deleted file mode 100644
index c8028d946b3ef30ba6b9ade62b503c3455c49ed6..0000000000000000000000000000000000000000
--- a/video/recce/model_code/model/__init__.py
+++ /dev/null
@@ -1,12 +0,0 @@
-from .network import *
-from .common import *
-
-MODELS = {
- "Recce": Recce
-}
-
-
-def load_model(name="Recce"):
- assert name in MODELS.keys(), f"Model name can only be one of {MODELS.keys()}."
- print(f"Using model: '{name}'")
- return MODELS[name]
diff --git a/video/recce/model_code/model/common.py b/video/recce/model_code/model/common.py
deleted file mode 100644
index 1c8a1b962ac277a2164d4387d60083632ce9096b..0000000000000000000000000000000000000000
--- a/video/recce/model_code/model/common.py
+++ /dev/null
@@ -1,200 +0,0 @@
-import torch
-import torch.nn as nn
-import torch.nn.functional as F
-
-
-def freeze_weights(module):
- for param in module.parameters():
- param.requires_grad = False
-
-
-def l1_regularize(module):
- reg_loss = 0.
- for key, param in module.reg_params.items():
- if "weight" in key and param.requires_grad:
- reg_loss += torch.sum(torch.abs(param))
- return reg_loss
-
-
-class SeparableConv2d(nn.Module):
- def __init__(self, in_channels, out_channels, kernel_size=1, stride=1, padding=0, dilation=1, bias=False):
- super(SeparableConv2d, self).__init__()
-
- self.conv1 = nn.Conv2d(in_channels, in_channels, kernel_size, stride, padding, dilation,
- groups=in_channels, bias=bias)
- self.pointwise = nn.Conv2d(in_channels, out_channels, 1, 1, 0, 1, 1, bias=bias)
-
- def forward(self, x):
- x = self.conv1(x)
- x = self.pointwise(x)
- return x
-
-
-class Block(nn.Module):
- def __init__(self, in_channels, out_channels, reps, strides=1,
- start_with_relu=True, grow_first=True, with_bn=True):
- super(Block, self).__init__()
-
- self.with_bn = with_bn
-
- if out_channels != in_channels or strides != 1:
- self.skip = nn.Conv2d(in_channels, out_channels, 1, stride=strides, bias=False)
- if with_bn:
- self.skipbn = nn.BatchNorm2d(out_channels)
- else:
- self.skip = None
-
- rep = []
- for i in range(reps):
- if grow_first:
- inc = in_channels if i == 0 else out_channels
- outc = out_channels
- else:
- inc = in_channels
- outc = in_channels if i < (reps - 1) else out_channels
- rep.append(nn.ReLU(inplace=True))
- rep.append(SeparableConv2d(inc, outc, 3, stride=1, padding=1))
- if with_bn:
- rep.append(nn.BatchNorm2d(outc))
-
- if not start_with_relu:
- rep = rep[1:]
- else:
- rep[0] = nn.ReLU(inplace=False)
-
- if strides != 1:
- rep.append(nn.MaxPool2d(3, strides, 1))
- self.rep = nn.Sequential(*rep)
-
- def forward(self, inp):
- x = self.rep(inp)
-
- if self.skip is not None:
- skip = self.skip(inp)
- if self.with_bn:
- skip = self.skipbn(skip)
- else:
- skip = inp
-
- x += skip
- return x
-
-
-class GraphReasoning(nn.Module):
- """ Graph Reasoning Module for information aggregation. """
-
- def __init__(self, va_in, va_out, vb_in, vb_out, vc_in, vc_out, spatial_ratio, drop_rate):
- super(GraphReasoning, self).__init__()
- self.ratio = spatial_ratio
- self.va_embedding = nn.Sequential(
- nn.Conv2d(va_in, va_out, 1, bias=False),
- nn.ReLU(True),
- nn.Conv2d(va_out, va_out, 1, bias=False),
- )
- self.va_gated_b = nn.Sequential(
- nn.Conv2d(va_in, va_out, 1, bias=False),
- nn.Sigmoid()
- )
- self.va_gated_c = nn.Sequential(
- nn.Conv2d(va_in, va_out, 1, bias=False),
- nn.Sigmoid()
- )
- self.vb_embedding = nn.Sequential(
- nn.Linear(vb_in, vb_out, bias=False),
- nn.ReLU(True),
- nn.Linear(vb_out, vb_out, bias=False),
- )
- self.vc_embedding = nn.Sequential(
- nn.Linear(vc_in, vc_out, bias=False),
- nn.ReLU(True),
- nn.Linear(vc_out, vc_out, bias=False),
- )
- self.unfold_b = nn.Unfold(kernel_size=spatial_ratio[0], stride=spatial_ratio[0])
- self.unfold_c = nn.Unfold(kernel_size=spatial_ratio[1], stride=spatial_ratio[1])
- self.reweight_ab = nn.Sequential(
- nn.Linear(va_out + vb_out, 1, bias=False),
- nn.ReLU(True),
- nn.Softmax(dim=1)
- )
- self.reweight_ac = nn.Sequential(
- nn.Linear(va_out + vc_out, 1, bias=False),
- nn.ReLU(True),
- nn.Softmax(dim=1)
- )
- self.reproject = nn.Sequential(
- nn.Conv2d(va_out + vb_out + vc_out, va_in, kernel_size=1, bias=False),
- nn.ReLU(True),
- nn.Conv2d(va_in, va_in, kernel_size=1, bias=False),
- nn.Dropout(drop_rate) if drop_rate is not None else nn.Identity(),
- )
-
- def forward(self, vert_a, vert_b, vert_c):
- emb_vert_a = self.va_embedding(vert_a)
- emb_vert_a = emb_vert_a.reshape([emb_vert_a.shape[0], emb_vert_a.shape[1], -1])
-
- gate_vert_b = 1 - self.va_gated_b(vert_a)
- gate_vert_b = gate_vert_b.reshape(*emb_vert_a.shape)
- gate_vert_c = 1 - self.va_gated_c(vert_a)
- gate_vert_c = gate_vert_c.reshape(*emb_vert_a.shape)
-
- vert_b = self.unfold_b(vert_b).reshape(
- [vert_b.shape[0], vert_b.shape[1], self.ratio[0] * self.ratio[0], -1])
- vert_b = vert_b.permute([0, 2, 3, 1])
- emb_vert_b = self.vb_embedding(vert_b)
-
- vert_c = self.unfold_c(vert_c).reshape(
- [vert_c.shape[0], vert_c.shape[1], self.ratio[1] * self.ratio[1], -1])
- vert_c = vert_c.permute([0, 2, 3, 1])
- emb_vert_c = self.vc_embedding(vert_c)
-
- agg_vb = list()
- agg_vc = list()
- for j in range(emb_vert_a.shape[-1]):
- # ab propagating
- emb_v_a = torch.stack([emb_vert_a[:, :, j]] * (self.ratio[0] ** 2), dim=1)
- emb_v_b = emb_vert_b[:, :, j, :]
- emb_v_ab = torch.cat([emb_v_a, emb_v_b], dim=-1)
- w = self.reweight_ab(emb_v_ab)
- agg_vb.append(torch.bmm(emb_v_b.transpose(1, 2), w).squeeze() * gate_vert_b[:, :, j])
-
- # ac propagating
- emb_v_a = torch.stack([emb_vert_a[:, :, j]] * (self.ratio[1] ** 2), dim=1)
- emb_v_c = emb_vert_c[:, :, j, :]
- emb_v_ac = torch.cat([emb_v_a, emb_v_c], dim=-1)
- w = self.reweight_ac(emb_v_ac)
- agg_vc.append(torch.bmm(emb_v_c.transpose(1, 2), w).squeeze() * gate_vert_c[:, :, j])
-
- agg_vert_b = torch.stack(agg_vb, dim=-1)
- agg_vert_c = torch.stack(agg_vc, dim=-1)
- agg_vert_bc = torch.cat([agg_vert_b, agg_vert_c], dim=1)
- agg_vert_abc = torch.cat([agg_vert_bc, emb_vert_a], dim=1)
- agg_vert_abc = torch.sigmoid(agg_vert_abc)
- agg_vert_abc = agg_vert_abc.reshape(vert_a.shape[0], -1, vert_a.shape[2], vert_a.shape[3])
- return self.reproject(agg_vert_abc)
-
-
-class GuidedAttention(nn.Module):
- """ Reconstruction Guided Attention. """
-
- def __init__(self, depth=728, drop_rate=0.2):
- super(GuidedAttention, self).__init__()
- self.depth = depth
- self.gated = nn.Sequential(
- nn.Conv2d(3, 3, kernel_size=3, stride=1, padding=1, bias=False),
- nn.ReLU(True),
- nn.Conv2d(3, 1, 1, bias=False),
- nn.Sigmoid()
- )
- self.h = nn.Sequential(
- nn.Conv2d(depth, depth, 1, 1, bias=False),
- nn.BatchNorm2d(depth),
- nn.ReLU(True),
- )
- self.dropout = nn.Dropout(drop_rate)
-
- def forward(self, x, pred_x, embedding):
- residual_full = torch.abs(x - pred_x)
- residual_x = F.interpolate(residual_full, size=embedding.shape[-2:],
- mode='bilinear', align_corners=True)
- res_map = self.gated(residual_x)
- return res_map * self.h(embedding) + self.dropout(embedding)
diff --git a/video/recce/model_code/model/network/Recce.py b/video/recce/model_code/model/network/Recce.py
deleted file mode 100644
index eb647cfc245542108bf4de3450ee3db05c55fb6d..0000000000000000000000000000000000000000
--- a/video/recce/model_code/model/network/Recce.py
+++ /dev/null
@@ -1,133 +0,0 @@
-from functools import partial
-from timm.models import xception
-from model.common import SeparableConv2d, Block
-from model.common import GuidedAttention, GraphReasoning
-
-import torch
-import torch.nn as nn
-import torch.nn.functional as F
-
-encoder_params = {
- "xception": {
- "features": 2048,
- "init_op": partial(xception, pretrained=True)
- }
-}
-
-
-class Recce(nn.Module):
- """ End-to-End Reconstruction-Classification Learning for Face Forgery Detection """
-
- def __init__(self, num_classes, drop_rate=0.2):
- super(Recce, self).__init__()
- self.name = "xception"
- self.loss_inputs = dict()
- self.encoder = encoder_params[self.name]["init_op"]()
- self.global_pool = nn.AdaptiveAvgPool2d((1, 1))
- self.dropout = nn.Dropout(drop_rate)
- self.fc = nn.Linear(encoder_params[self.name]["features"], num_classes)
-
- self.attention = GuidedAttention(depth=728, drop_rate=drop_rate)
- self.reasoning = GraphReasoning(728, 256, 256, 256, 128, 256, [2, 4], drop_rate)
-
- self.decoder1 = nn.Sequential(
- nn.UpsamplingNearest2d(scale_factor=2),
- SeparableConv2d(728, 256, 3, 1, 1, bias=False),
- nn.BatchNorm2d(256),
- nn.ReLU(inplace=True)
- )
- self.decoder2 = Block(256, 256, 3, 1)
- self.decoder3 = nn.Sequential(
- nn.UpsamplingNearest2d(scale_factor=2),
- SeparableConv2d(256, 128, 3, 1, 1, bias=False),
- nn.BatchNorm2d(128),
- nn.ReLU(inplace=True)
- )
- self.decoder4 = Block(128, 128, 3, 1)
- self.decoder5 = nn.Sequential(
- nn.UpsamplingNearest2d(scale_factor=2),
- SeparableConv2d(128, 64, 3, 1, 1, bias=False),
- nn.BatchNorm2d(64),
- nn.ReLU(inplace=True)
- )
- self.decoder6 = nn.Sequential(
- nn.Conv2d(64, 3, 1, 1, bias=False),
- nn.Tanh()
- )
-
- def norm_n_corr(self, x):
- norm_embed = F.normalize(self.global_pool(x), p=2, dim=1)
- corr = (torch.matmul(norm_embed.squeeze(), norm_embed.squeeze().T) + 1.) / 2.
- return norm_embed, corr
-
- @staticmethod
- def add_white_noise(tensor, mean=0., std=1e-6):
- rand = torch.rand([tensor.shape[0], 1, 1, 1])
- rand = torch.where(rand > 0.5, 1., 0.).to(tensor.device)
- white_noise = torch.normal(mean, std, size=tensor.shape, device=tensor.device)
- noise_t = tensor + white_noise * rand
- noise_t = torch.clip(noise_t, -1., 1.)
- return noise_t
-
- def forward(self, x):
- # clear the loss inputs
- self.loss_inputs = dict(recons=[], contra=[])
- noise_x = self.add_white_noise(x) if self.training else x
- out = self.encoder.conv1(noise_x)
- out = self.encoder.bn1(out)
- out = self.encoder.act1(out)
- out = self.encoder.conv2(out)
- out = self.encoder.bn2(out)
- out = self.encoder.act2(out)
- out = self.encoder.block1(out)
- out = self.encoder.block2(out)
- out = self.encoder.block3(out)
- embedding = self.encoder.block4(out)
-
- norm_embed, corr = self.norm_n_corr(embedding)
- self.loss_inputs['contra'].append(corr)
-
- out = self.dropout(embedding)
- out = self.decoder1(out)
- out_d2 = self.decoder2(out)
-
- norm_embed, corr = self.norm_n_corr(out_d2)
- self.loss_inputs['contra'].append(corr)
-
- out = self.decoder3(out_d2)
- out_d4 = self.decoder4(out)
-
- norm_embed, corr = self.norm_n_corr(out_d4)
- self.loss_inputs['contra'].append(corr)
-
- out = self.decoder5(out_d4)
- pred = self.decoder6(out)
-
- recons_x = F.interpolate(pred, size=x.shape[-2:], mode='bilinear', align_corners=True)
- self.loss_inputs['recons'].append(recons_x)
-
- embedding = self.encoder.block5(embedding)
- embedding = self.encoder.block6(embedding)
- embedding = self.encoder.block7(embedding)
-
- fusion = self.reasoning(embedding, out_d2, out_d4) + embedding
-
- embedding = self.encoder.block8(fusion)
- img_att = self.attention(x, recons_x, embedding)
-
- embedding = self.encoder.block9(img_att)
- embedding = self.encoder.block10(embedding)
- embedding = self.encoder.block11(embedding)
- embedding = self.encoder.block12(embedding)
-
- embedding = self.encoder.conv3(embedding)
- embedding = self.encoder.bn3(embedding)
- embedding = self.encoder.act3(embedding)
- embedding = self.encoder.conv4(embedding)
- embedding = self.encoder.bn4(embedding)
- embedding = self.encoder.act4(embedding)
-
- embedding = self.global_pool(embedding).squeeze()
-
- out = self.dropout(embedding)
- return self.fc(out)
diff --git a/video/recce/model_code/model/network/__init__.py b/video/recce/model_code/model/network/__init__.py
deleted file mode 100644
index e552eb84c758d2b1ac6542725d7faed4d1fca3bd..0000000000000000000000000000000000000000
--- a/video/recce/model_code/model/network/__init__.py
+++ /dev/null
@@ -1 +0,0 @@
-from .Recce import Recce
diff --git a/video/recce/model_code/optimizer/__init__.py b/video/recce/model_code/optimizer/__init__.py
deleted file mode 100644
index 74db9744a1f35e9f082e3a61997025ad4c22a8e7..0000000000000000000000000000000000000000
--- a/video/recce/model_code/optimizer/__init__.py
+++ /dev/null
@@ -1,30 +0,0 @@
-from torch.optim import SGD
-from torch.optim import Adam
-from torch.optim import ASGD
-from torch.optim import Adamax
-from torch.optim import Adadelta
-from torch.optim import Adagrad
-from torch.optim import RMSprop
-
-key2opt = {
- 'sgd': SGD,
- 'adam': Adam,
- 'asgd': ASGD,
- 'adamax': Adamax,
- 'adadelta': Adadelta,
- 'adagrad': Adagrad,
- 'rmsprop': RMSprop,
-}
-
-
-def get_optimizer(optimizer_name=None):
- if optimizer_name is None:
- print("Using default 'SGD' optimizer")
- return SGD
-
- else:
- if optimizer_name not in key2opt:
- raise NotImplementedError(f"Optimizer '{optimizer_name}' not implemented")
-
- print(f"Using optimizer: '{optimizer_name}'")
- return key2opt[optimizer_name]
diff --git a/video/recce/model_code/scheduler/__init__.py b/video/recce/model_code/scheduler/__init__.py
deleted file mode 100644
index d02f3edf959d8e10b66c08e276ced619ecfd15dd..0000000000000000000000000000000000000000
--- a/video/recce/model_code/scheduler/__init__.py
+++ /dev/null
@@ -1,36 +0,0 @@
-from torch.optim.lr_scheduler import _LRScheduler
-from torch.optim.lr_scheduler import StepLR
-from torch.optim.lr_scheduler import MultiStepLR
-from torch.optim.lr_scheduler import ExponentialLR
-from torch.optim.lr_scheduler import CosineAnnealingLR
-from torch.optim.lr_scheduler import CosineAnnealingWarmRestarts
-from torch.optim.lr_scheduler import ReduceLROnPlateau
-
-
-class ConstantLR(_LRScheduler):
- def __init__(self, optimizer, last_epoch=-1):
- super(ConstantLR, self).__init__(optimizer, last_epoch)
-
- def get_lr(self):
- return [base_lr for base_lr in self.base_lrs]
-
-
-SCHEDULERS = {
- 'ConstantLR': ConstantLR,
- "StepLR": StepLR,
- "MultiStepLR": MultiStepLR,
- "CosineAnnealingLR": CosineAnnealingLR,
- "CosineAnnealingWarmRestarts": CosineAnnealingWarmRestarts,
- "ExponentialLR": ExponentialLR,
- "ReduceLROnPlateau": ReduceLROnPlateau
-}
-
-
-def get_scheduler(optimizer, kwargs):
- if kwargs is None:
- print("No lr scheduler is used.")
- return ConstantLR(optimizer)
- name = kwargs["name"]
- kwargs.pop("name")
- print("Using scheduler: '%s' with params: %s" % (name, kwargs))
- return SCHEDULERS[name](optimizer, **kwargs)
diff --git a/video/recce/model_code/test.py b/video/recce/model_code/test.py
deleted file mode 100644
index 7467842f9affda299fbd5a06bb44cc6d691600f9..0000000000000000000000000000000000000000
--- a/video/recce/model_code/test.py
+++ /dev/null
@@ -1,31 +0,0 @@
-import yaml
-import argparse
-
-from trainer import ExpTester
-
-
-def arg_parser():
- parser = argparse.ArgumentParser(description="config")
- parser.add_argument("--config",
- type=str,
- default="config/Recce.yml",
- help="Specify the path of configuration file to be used.")
- parser.add_argument('--display', '-d', action="store_true",
- default=False, help='Display some images.')
- return parser.parse_args()
-
-
-if __name__ == '__main__':
- import torch
-
- torch.backends.cudnn.benchmark = True
- torch.backends.cudnn.enabled = True
-
- arg = arg_parser()
- config = arg.config
-
- with open(config) as config_file:
- config = yaml.load(config_file, Loader=yaml.FullLoader)
-
- trainer = ExpTester(config, stage="Test")
- trainer.test(display_images=arg.display)
diff --git a/video/recce/model_code/train.py b/video/recce/model_code/train.py
deleted file mode 100644
index cae7ad48eddf9d945142cc77c818df3dd8b30a71..0000000000000000000000000000000000000000
--- a/video/recce/model_code/train.py
+++ /dev/null
@@ -1,33 +0,0 @@
-import yaml
-import argparse
-
-from trainer import ExpMultiGpuTrainer
-
-
-def arg_parser():
- parser = argparse.ArgumentParser(description="config")
- parser.add_argument("--config",
- type=str,
- default="config/Recce.yml",
- help="Specified the path of configuration file to be used.")
- parser.add_argument("--local_rank", default=0,
- type=int,
- help="Specified the node rank for distributed training.")
- return parser.parse_args()
-
-
-if __name__ == '__main__':
- import torch
-
- torch.backends.cudnn.benchmark = True
- torch.backends.cudnn.enabled = True
-
- arg = arg_parser()
- config = arg.config
-
- with open(config) as config_file:
- config = yaml.load(config_file, Loader=yaml.FullLoader)
- config["config"]["local_rank"] = arg.local_rank
-
- trainer = ExpMultiGpuTrainer(config, stage="Train")
- trainer.train()
diff --git a/video/recce/model_code/trainer/__init__.py b/video/recce/model_code/trainer/__init__.py
deleted file mode 100644
index d7f4da6b97a20cf6c6dae6761dffea145a336cea..0000000000000000000000000000000000000000
--- a/video/recce/model_code/trainer/__init__.py
+++ /dev/null
@@ -1,5 +0,0 @@
-from .abstract_trainer import AbstractTrainer, LEGAL_METRIC
-from .exp_mgpu_trainer import ExpMultiGpuTrainer
-from .exp_tester import ExpTester
-from .utils import center_print, reduce_tensor
-from .utils import exp_recons_loss
diff --git a/video/recce/model_code/trainer/abstract_trainer.py b/video/recce/model_code/trainer/abstract_trainer.py
deleted file mode 100644
index 4e1354a94ec55db2fcad0d951ecaca2f7b804fbe..0000000000000000000000000000000000000000
--- a/video/recce/model_code/trainer/abstract_trainer.py
+++ /dev/null
@@ -1,100 +0,0 @@
-import os
-import torch
-import random
-from collections import OrderedDict
-from torchvision.utils import make_grid
-
-LEGAL_METRIC = ['Acc', 'AUC', 'LogLoss']
-
-
-class AbstractTrainer(object):
- def __init__(self, config, stage="Train"):
- feasible_stage = ["Train", "Test"]
- if stage not in feasible_stage:
- raise ValueError(f"stage should be in {feasible_stage}, but found '{stage}'")
-
- self.config = config
- model_cfg = config.get("model", None)
- data_cfg = config.get("data", None)
- config_cfg = config.get("config", None)
-
- self.model_name = model_cfg.pop("name")
-
- self.gpu = None
- self.dir = None
- self.debug = None
- self.device = None
- self.resume = None
- self.local_rank = None
- self.num_classes = None
-
- self.best_metric = 0.0
- self.best_step = 1
- self.start_step = 1
-
- self._initiated_settings(model_cfg, data_cfg, config_cfg)
-
- if stage == 'Train':
- self._train_settings(model_cfg, data_cfg, config_cfg)
- if stage == 'Test':
- self._test_settings(model_cfg, data_cfg, config_cfg)
-
- def _initiated_settings(self, model_cfg, data_cfg, config_cfg):
- raise NotImplementedError("Not implemented in abstract class.")
-
- def _train_settings(self, model_cfg, data_cfg, config_cfg):
- raise NotImplementedError("Not implemented in abstract class.")
-
- def _test_settings(self, model_cfg, data_cfg, config_cfg):
- raise NotImplementedError("Not implemented in abstract class.")
-
- def _save_ckpt(self, step, best=False):
- raise NotImplementedError("Not implemented in abstract class.")
-
- def _load_ckpt(self, best=False, train=False):
- raise NotImplementedError("Not implemented in abstract class.")
-
- def to_device(self, items):
- return [obj.to(self.device) for obj in items]
-
- @staticmethod
- def fixed_randomness():
- random.seed(0)
- torch.manual_seed(0)
- torch.cuda.manual_seed(0)
- torch.cuda.manual_seed_all(0)
-
- def train(self):
- raise NotImplementedError("Not implemented in abstract class.")
-
- def validate(self, epoch, step, timer, writer):
- raise NotImplementedError("Not implemented in abstract class.")
-
- def test(self):
- raise NotImplementedError("Not implemented in abstract class.")
-
- def plot_figure(self, images, pred, gt, nrow, categories=None, show=True):
- import matplotlib.pyplot as plt
- plot = make_grid(
- images, nrow, padding=4, normalize=True, scale_each=True, pad_value=1)
- if self.num_classes == 1:
- pred = (pred >= 0.5).cpu().numpy()
- else:
- pred = pred.argmax(1).cpu().numpy()
- gt = gt.cpu().numpy()
- if categories is not None:
- pred = [categories[i] for i in pred]
- gt = [categories[i] for i in gt]
- plot = plot.permute([1, 2, 0])
- plot = plot.cpu().numpy()
- ret = plt.figure()
- plt.imshow(plot)
- plt.title("pred: %s\ngt: %s" % (pred, gt))
- plt.axis("off")
- if show:
- plt.savefig(os.path.join(self.dir, "test_image.png"), dpi=300)
- plt.show()
- plt.close()
- else:
- plt.close()
- return ret
diff --git a/video/recce/model_code/trainer/exp_mgpu_trainer.py b/video/recce/model_code/trainer/exp_mgpu_trainer.py
deleted file mode 100644
index f9e2d43c4e040b4390870b7ca5af07809f5e30c3..0000000000000000000000000000000000000000
--- a/video/recce/model_code/trainer/exp_mgpu_trainer.py
+++ /dev/null
@@ -1,370 +0,0 @@
-import os
-import sys
-import time
-import math
-import yaml
-import torch
-import random
-import numpy as np
-
-from tqdm import tqdm
-from pprint import pprint
-from torch.utils import data
-import torch.distributed as dist
-from torch.cuda.amp import autocast, GradScaler
-from tensorboardX import SummaryWriter
-
-from dataset import load_dataset
-from loss import get_loss
-from model import load_model
-from optimizer import get_optimizer
-from scheduler import get_scheduler
-from trainer import AbstractTrainer, LEGAL_METRIC
-from trainer.utils import exp_recons_loss, MLLoss, reduce_tensor, center_print
-from trainer.utils import MODELS_PATH, AccMeter, AUCMeter, AverageMeter, Logger, Timer
-
-
-class ExpMultiGpuTrainer(AbstractTrainer):
- def __init__(self, config, stage="Train"):
- super(ExpMultiGpuTrainer, self).__init__(config, stage)
- np.random.seed(2021)
-
- def _mprint(self, content=""):
- if self.local_rank == 0:
- print(content)
-
- def _initiated_settings(self, model_cfg=None, data_cfg=None, config_cfg=None):
- self.local_rank = config_cfg["local_rank"]
-
- def _train_settings(self, model_cfg, data_cfg, config_cfg):
- # debug mode: no log dir, no train_val operation.
- self.debug = config_cfg["debug"]
- self._mprint(f"Using debug mode: {self.debug}.")
- self._mprint("*" * 20)
-
- self.eval_metric = config_cfg["metric"]
- if self.eval_metric not in LEGAL_METRIC:
- raise ValueError(f"Evaluation metric must be in {LEGAL_METRIC}, but found "
- f"{self.eval_metric}.")
- if self.eval_metric == LEGAL_METRIC[-1]:
- self.best_metric = 1.0e8
-
- # distribution
- dist.init_process_group(config_cfg["distribute"]["backend"])
-
- # load training dataset
- train_dataset = data_cfg["file"]
- branch = data_cfg["train_branch"]
- name = data_cfg["name"]
- with open(train_dataset, "r") as f:
- options = yaml.load(f, Loader=yaml.FullLoader)
- train_options = options[branch]
- self.train_set = load_dataset(name)(train_options)
- # define training sampler
- self.train_sampler = data.distributed.DistributedSampler(self.train_set)
- # wrapped with data loader
- self.train_loader = data.DataLoader(self.train_set, shuffle=False,
- sampler=self.train_sampler,
- num_workers=data_cfg.get("num_workers", 4),
- batch_size=data_cfg["train_batch_size"])
-
- if self.local_rank == 0:
- # load validation dataset
- val_options = options[data_cfg["val_branch"]]
- self.val_set = load_dataset(name)(val_options)
- # wrapped with data loader
- self.val_loader = data.DataLoader(self.val_set, shuffle=True,
- num_workers=data_cfg.get("num_workers", 4),
- batch_size=data_cfg["val_batch_size"])
-
- self.resume = config_cfg.get("resume", False)
-
- if not self.debug:
- time_format = "%Y-%m-%d...%H.%M.%S"
- run_id = time.strftime(time_format, time.localtime(time.time()))
- self.run_id = config_cfg.get("id", run_id)
- self.dir = os.path.join("runs", self.model_name, self.run_id)
-
- if self.local_rank == 0:
- if not self.resume:
- if os.path.exists(self.dir):
- raise ValueError("Error: given id '%s' already exists." % self.run_id)
- os.makedirs(self.dir, exist_ok=True)
- print(f"Writing config file to file directory: {self.dir}.")
- yaml.dump({"config": self.config,
- "train_data": train_options,
- "val_data": val_options},
- open(os.path.join(self.dir, 'train_config.yml'), 'w'))
- # copy the script for the training model
- model_file = MODELS_PATH[self.model_name]
- os.system("cp " + model_file + " " + self.dir)
- else:
- print(f"Resuming the history in file directory: {self.dir}.")
-
- print(f"Logging directory: {self.dir}.")
-
- # redirect the std out stream
- sys.stdout = Logger(os.path.join(self.dir, 'records.txt'))
- center_print('Train configurations begins.')
- pprint(self.config)
- pprint(train_options)
- pprint(val_options)
- center_print('Train configurations ends.')
-
- # load model
- self.num_classes = model_cfg["num_classes"]
- self.device = "cuda:" + str(self.local_rank)
- self.model = load_model(self.model_name)(**model_cfg)
- self.model = torch.nn.SyncBatchNorm.convert_sync_batchnorm(self.model).to(self.device)
- self._mprint(f"Using SyncBatchNorm.")
- self.model = torch.nn.parallel.DistributedDataParallel(
- self.model, device_ids=[self.local_rank], find_unused_parameters=True)
-
- # load optimizer
- optim_cfg = config_cfg.get("optimizer", None)
- optim_name = optim_cfg.pop("name")
- self.optimizer = get_optimizer(optim_name)(self.model.parameters(), **optim_cfg)
- # load scheduler
- self.scheduler = get_scheduler(self.optimizer, config_cfg.get("scheduler", None))
- # load loss
- self.loss_criterion = get_loss(config_cfg.get("loss", None), device=self.device)
-
- # total number of steps (or epoch) to train
- self.num_steps = train_options["num_steps"]
- self.num_epoch = math.ceil(self.num_steps / len(self.train_loader))
-
- # the number of steps to write down a log
- self.log_steps = train_options["log_steps"]
- # the number of steps to validate on val dataset once
- self.val_steps = train_options["val_steps"]
-
- # balance coefficients
- self.lambda_1 = config_cfg["lambda_1"]
- self.lambda_2 = config_cfg["lambda_2"]
- self.warmup_step = config_cfg.get('warmup_step', 0)
-
- self.contra_loss = MLLoss()
- self.acc_meter = AccMeter()
- self.loss_meter = AverageMeter()
- self.recons_loss_meter = AverageMeter()
- self.contra_loss_meter = AverageMeter()
-
- if self.resume and self.local_rank == 0:
- self._load_ckpt(best=config_cfg.get("resume_best", False), train=True)
-
- def _test_settings(self, model_cfg, data_cfg, config_cfg):
- # Not used.
- raise NotImplementedError("The function is not intended to be used here.")
-
- def _load_ckpt(self, best=False, train=False):
- # Not used.
- raise NotImplementedError("The function is not intended to be used here.")
-
- def _save_ckpt(self, step, best=False):
- save_dir = os.path.join(self.dir, f"best_model_{step}.bin" if best else "latest_model.bin")
- torch.save({
- "step": step,
- "best_step": self.best_step,
- "best_metric": self.best_metric,
- "eval_metric": self.eval_metric,
- "model": self.model.module.state_dict(),
- "optimizer": self.optimizer.state_dict(),
- "scheduler": self.scheduler.state_dict(),
- }, save_dir)
-
- def train(self):
- try:
- timer = Timer()
- grad_scalar = GradScaler(2 ** 10)
- if self.local_rank == 0:
- writer = None if self.debug else SummaryWriter(log_dir=self.dir)
- center_print("Training begins......")
- else:
- writer = None
- start_epoch = self.start_step // len(self.train_loader) + 1
- for epoch_idx in range(start_epoch, self.num_epoch + 1):
- # set sampler
- self.train_sampler.set_epoch(epoch_idx)
-
- # reset meter
- self.acc_meter.reset()
- self.loss_meter.reset()
- self.recons_loss_meter.reset()
- self.contra_loss_meter.reset()
- self.optimizer.step()
-
- train_generator = enumerate(self.train_loader, 1)
- # wrap train generator with tqdm for process 0
- if self.local_rank == 0:
- train_generator = tqdm(train_generator, position=0, leave=True)
-
- for batch_idx, train_data in train_generator:
- global_step = (epoch_idx - 1) * len(self.train_loader) + batch_idx
- self.model.train()
- I, Y = train_data
- I = self.train_loader.dataset.load_item(I)
- in_I, Y = self.to_device((I, Y))
-
- # warm-up lr
- if self.warmup_step != 0 and global_step <= self.warmup_step:
- lr = self.config['config']['optimizer']['lr'] * float(global_step) / self.warmup_step
- for param_group in self.optimizer.param_groups:
- param_group['lr'] = lr
-
- self.optimizer.zero_grad()
- with autocast():
- Y_pre = self.model(in_I)
-
- # for BCE Setting:
- if self.num_classes == 1:
- Y_pre = Y_pre.squeeze()
- loss = self.loss_criterion(Y_pre, Y.float())
- Y_pre = torch.sigmoid(Y_pre)
- else:
- loss = self.loss_criterion(Y_pre, Y)
-
- # flood
- loss = (loss - 0.04).abs() + 0.04
- recons_loss = exp_recons_loss(self.model.module.loss_inputs['recons'], (in_I, Y))
- contra_loss = self.contra_loss(self.model.module.loss_inputs['contra'], Y)
- loss += self.lambda_1 * recons_loss + self.lambda_2 * contra_loss
-
- grad_scalar.scale(loss).backward()
- grad_scalar.step(self.optimizer)
- grad_scalar.update()
- if self.warmup_step == 0 or global_step > self.warmup_step:
- self.scheduler.step()
-
- self.acc_meter.update(Y_pre, Y, self.num_classes == 1)
- self.loss_meter.update(reduce_tensor(loss).item())
- self.recons_loss_meter.update(reduce_tensor(recons_loss).item())
- self.contra_loss_meter.update(reduce_tensor(contra_loss).item())
- iter_acc = reduce_tensor(self.acc_meter.mean_acc()).item()
-
- if self.local_rank == 0:
- if global_step % self.log_steps == 0 and writer is not None:
- writer.add_scalar("train/Acc", iter_acc, global_step)
- writer.add_scalar("train/Loss", self.loss_meter.avg, global_step)
- writer.add_scalar("train/Recons_Loss",
- self.recons_loss_meter.avg if self.lambda_1 != 0 else 0.,
- global_step)
- writer.add_scalar("train/Contra_Loss", self.contra_loss_meter.avg, global_step)
- writer.add_scalar("train/LR", self.scheduler.get_last_lr()[0], global_step)
-
- # log training step
- train_generator.set_description(
- "Train Epoch %d (%d/%d), Global Step %d, Loss %.4f, Recons %.4f, con %.4f, "
- "ACC %.4f, LR %.6f" % (
- epoch_idx, batch_idx, len(self.train_loader), global_step,
- self.loss_meter.avg, self.recons_loss_meter.avg, self.contra_loss_meter.avg,
- iter_acc, self.scheduler.get_last_lr()[0])
- )
-
- # validating process
- if global_step % self.val_steps == 0 and not self.debug:
- print()
- self.validate(epoch_idx, global_step, timer, writer)
-
- # when num_steps has been set and the training process will
- # be stopped earlier than the specified num_epochs, then stop.
- if self.num_steps is not None and global_step == self.num_steps:
- if writer is not None:
- writer.close()
- if self.local_rank == 0:
- print()
- center_print("Training process ends.")
- dist.destroy_process_group()
- return
- # close the tqdm bar when one epoch ends
- if self.local_rank == 0:
- train_generator.close()
- print()
- # training ends with integer epochs
- if self.local_rank == 0:
- if writer is not None:
- writer.close()
- center_print("Training process ends.")
- dist.destroy_process_group()
- except Exception as e:
- dist.destroy_process_group()
- raise e
-
- def validate(self, epoch, step, timer, writer):
- v_idx = random.randint(1, len(self.val_loader) + 1)
- categories = self.val_loader.dataset.categories
- self.model.eval()
- with torch.no_grad():
- acc = AccMeter()
- auc = AUCMeter()
- loss_meter = AverageMeter()
- cur_acc = 0.0 # Higher is better
- cur_auc = 0.0 # Higher is better
- cur_loss = 1e8 # Lower is better
- val_generator = tqdm(enumerate(self.val_loader, 1), position=0, leave=True)
- for val_idx, val_data in val_generator:
- I, Y = val_data
- I = self.val_loader.dataset.load_item(I)
- in_I, Y = self.to_device((I, Y))
- Y_pre = self.model(in_I)
-
- # for BCE Setting:
- if self.num_classes == 1:
- Y_pre = Y_pre.squeeze()
- loss = self.loss_criterion(Y_pre, Y.float())
- Y_pre = torch.sigmoid(Y_pre)
- else:
- loss = self.loss_criterion(Y_pre, Y)
-
- acc.update(Y_pre, Y, self.num_classes == 1)
- auc.update(Y_pre, Y, self.num_classes == 1)
- loss_meter.update(loss.item())
-
- cur_acc = acc.mean_acc()
- cur_loss = loss_meter.avg
-
- val_generator.set_description(
- "Eval Epoch %d (%d/%d), Global Step %d, Loss %.4f, ACC %.4f" % (
- epoch, val_idx, len(self.val_loader), step,
- cur_loss, cur_acc)
- )
-
- if val_idx == v_idx or val_idx == 1:
- sample_recons = list()
- for _ in self.model.module.loss_inputs['recons']:
- sample_recons.append(_[:4].to("cpu"))
- # show images
- images = I[:4]
- images = torch.cat([images, *sample_recons], dim=0)
- pred = Y_pre[:4]
- gt = Y[:4]
- figure = self.plot_figure(images, pred, gt, 4, categories, show=False)
-
- cur_auc = auc.mean_auc()
- print("Eval Epoch %d, Loss %.4f, ACC %.4f, AUC %.4f" % (epoch, cur_loss, cur_acc, cur_auc))
- if writer is not None:
- writer.add_scalar("val/Loss", cur_loss, step)
- writer.add_scalar("val/Acc", cur_acc, step)
- writer.add_scalar("val/AUC", cur_auc, step)
- writer.add_figure("val/Figures", figure, step)
- # record the best acc and the corresponding step
- if self.eval_metric == 'Acc' and cur_acc >= self.best_metric:
- self.best_metric = cur_acc
- self.best_step = step
- self._save_ckpt(step, best=True)
- elif self.eval_metric == 'AUC' and cur_auc >= self.best_metric:
- self.best_metric = cur_auc
- self.best_step = step
- self._save_ckpt(step, best=True)
- elif self.eval_metric == 'LogLoss' and cur_loss <= self.best_metric:
- self.best_metric = cur_loss
- self.best_step = step
- self._save_ckpt(step, best=True)
- print("Best Step %d, Best %s %.4f, Running Time: %s, Estimated Time: %s" % (
- self.best_step, self.eval_metric, self.best_metric,
- timer.measure(), timer.measure(step / self.num_steps)
- ))
- self._save_ckpt(step, best=False)
-
- def test(self):
- # Not used.
- raise NotImplementedError("The function is not intended to be used here.")
diff --git a/video/recce/model_code/trainer/exp_tester.py b/video/recce/model_code/trainer/exp_tester.py
deleted file mode 100644
index ee4f219f0ba60c6ded44e3b7c95eb7b21fc0be46..0000000000000000000000000000000000000000
--- a/video/recce/model_code/trainer/exp_tester.py
+++ /dev/null
@@ -1,144 +0,0 @@
-import os
-import sys
-import yaml
-import torch
-import random
-
-from tqdm import tqdm
-from pprint import pprint
-from torch.utils import data
-
-from dataset import load_dataset
-from loss import get_loss
-from model import load_model
-from model.common import freeze_weights
-from trainer import AbstractTrainer
-from trainer.utils import AccMeter, AUCMeter, AverageMeter, Logger, center_print
-
-
-class ExpTester(AbstractTrainer):
- def __init__(self, config, stage="Test"):
- super(ExpTester, self).__init__(config, stage)
-
- if torch.cuda.is_available() and self.device is not None:
- print(f"Using cuda device: {self.device}.")
- self.gpu = True
- self.model = self.model.to(self.device)
- else:
- print("Using cpu device.")
- self.device = torch.device("cpu")
-
- def _initiated_settings(self, model_cfg=None, data_cfg=None, config_cfg=None):
- self.gpu = False
- self.device = config_cfg.get("device", None)
-
- def _train_settings(self, model_cfg=None, data_cfg=None, config_cfg=None):
- # Not used.
- raise NotImplementedError("The function is not intended to be used here.")
-
- def _test_settings(self, model_cfg=None, data_cfg=None, config_cfg=None):
- # load test dataset
- test_dataset = data_cfg["file"]
- branch = data_cfg["test_branch"]
- name = data_cfg["name"]
- with open(test_dataset, "r") as f:
- options = yaml.load(f, Loader=yaml.FullLoader)
- test_options = options[branch]
- self.test_set = load_dataset(name)(test_options)
- # wrapped with data loader
- self.test_batch_size = data_cfg["test_batch_size"]
- self.test_loader = data.DataLoader(self.test_set, shuffle=False,
- batch_size=self.test_batch_size)
- self.run_id = config_cfg["id"]
- self.ckpt_fold = config_cfg.get("ckpt_fold", "runs")
- self.dir = os.path.join(self.ckpt_fold, self.model_name, self.run_id)
-
- # load model
- self.num_classes = model_cfg["num_classes"]
- self.model = load_model(self.model_name)(**model_cfg)
-
- # load loss
- self.loss_criterion = get_loss(config_cfg.get("loss", None))
-
- # redirect the std out stream
- sys.stdout = Logger(os.path.join(self.dir, "test_result.txt"))
- print('Run dir: {}'.format(self.dir))
-
- center_print('Test configurations begins')
- pprint(self.config)
- pprint(test_options)
- center_print('Test configurations ends')
-
- self.ckpt = config_cfg.get("ckpt", "best_model")
- self._load_ckpt(best=True, train=False)
-
- def _save_ckpt(self, step, best=False):
- # Not used.
- raise NotImplementedError("The function is not intended to be used here.")
-
- def _load_ckpt(self, best=False, train=False):
- load_dir = os.path.join(self.dir, self.ckpt + ".bin" if best else "latest_model.bin")
- load_dict = torch.load(load_dir, map_location=self.device)
- self.start_step = load_dict["step"]
- self.best_step = load_dict["best_step"]
- self.best_metric = load_dict.get("best_metric", None)
- if self.best_metric is None:
- self.best_metric = load_dict.get("best_acc")
- self.eval_metric = load_dict.get("eval_metric", None)
- if self.eval_metric is None:
- self.eval_metric = load_dict.get("Acc")
- self.model.load_state_dict(load_dict["model"])
- print(f"Loading checkpoint from {load_dir}, best step: {self.best_step}, "
- f"best {self.eval_metric}: {round(self.best_metric.item(), 4)}.")
-
- def train(self):
- # Not used.
- raise NotImplementedError("The function is not intended to be used here.")
-
- def validate(self, epoch, step, timer, writer):
- # Not used.
- raise NotImplementedError("The function is not intended to be used here.")
-
- def test(self, display_images=False):
- freeze_weights(self.model)
- t_idx = random.randint(1, len(self.test_loader) + 1)
- self.fixed_randomness() # for reproduction
-
- acc = AccMeter()
- auc = AUCMeter()
- logloss = AverageMeter()
- test_generator = tqdm(enumerate(self.test_loader, 1))
- categories = self.test_loader.dataset.categories
- for idx, test_data in test_generator:
- self.model.eval()
- I, Y = test_data
- I = self.test_loader.dataset.load_item(I)
- if self.gpu:
- in_I, Y = self.to_device((I, Y))
- else:
- in_I, Y = (I, Y)
- Y_pre = self.model(in_I)
-
- # for BCE Setting:
- if self.num_classes == 1:
- Y_pre = Y_pre.squeeze()
- loss = self.loss_criterion(Y_pre, Y.float())
- Y_pre = torch.sigmoid(Y_pre)
- else:
- loss = self.loss_criterion(Y_pre, Y)
-
- acc.update(Y_pre, Y, use_bce=self.num_classes == 1)
- auc.update(Y_pre, Y, use_bce=self.num_classes == 1)
- logloss.update(loss.item())
-
- test_generator.set_description("Test %d/%d" % (idx, len(self.test_loader)))
- if display_images and idx == t_idx:
- # show images
- images = I[:4]
- pred = Y_pre[:4]
- gt = Y[:4]
- self.plot_figure(images, pred, gt, 2, categories)
-
- print("Test, FINAL LOSS %.4f, FINAL ACC %.4f, FINAL AUC %.4f" %
- (logloss.avg, acc.mean_acc(), auc.mean_auc()))
- auc.curve(self.dir)
diff --git a/video/recce/model_code/trainer/utils.py b/video/recce/model_code/trainer/utils.py
deleted file mode 100644
index fcf1c4a864d34a18133667293c9dc76215af1011..0000000000000000000000000000000000000000
--- a/video/recce/model_code/trainer/utils.py
+++ /dev/null
@@ -1,183 +0,0 @@
-import os
-import sys
-import time
-import torch
-import torch.nn as nn
-import torch.nn.functional as F
-import torch.distributed as dist
-from collections import OrderedDict
-
-import numpy as np
-from sklearn.metrics import roc_auc_score, roc_curve
-from scipy.optimize import brentq
-from scipy.interpolate import interp1d
-
-# Tracking the path to the definition of the model.
-MODELS_PATH = {
- "Recce": "model/network/Recce.py"
-}
-
-
-def exp_recons_loss(recons, x):
- x, y = x
- loss = torch.tensor(0., device=y.device)
- real_index = torch.where(1 - y)[0]
- for r in recons:
- if real_index.numel() > 0:
- real_x = torch.index_select(x, dim=0, index=real_index)
- real_rec = torch.index_select(r, dim=0, index=real_index)
- real_rec = F.interpolate(real_rec, size=x.shape[-2:], mode='bilinear', align_corners=True)
- loss += torch.mean(torch.abs(real_rec - real_x))
- return loss
-
-
-def center_print(content, around='*', repeat_around=10):
- num = repeat_around
- s = around
- print(num * s + ' %s ' % content + num * s)
-
-
-def reduce_tensor(t):
- rt = t.clone()
- dist.all_reduce(rt)
- rt /= float(dist.get_world_size())
- return rt
-
-
-def tensor2image(tensor):
- image = tensor.permute([1, 2, 0]).cpu().detach().numpy()
- return (image - np.min(image)) / (np.max(image) - np.min(image))
-
-
-def state_dict(state_dict):
- """ Remove 'module' keyword in state dictionary. """
- weights = OrderedDict()
- for k, v in state_dict.items():
- weights.update({k.replace("module.", ""): v})
- return weights
-
-
-class Logger(object):
- def __init__(self, filename):
- self.terminal = sys.stdout
- self.log = open(filename, "a")
-
- def write(self, message):
- self.terminal.write(message)
- self.log.write(message)
- self.log.flush()
-
- def flush(self):
- pass
-
-
-class Timer(object):
- """The class for timer."""
-
- def __init__(self):
- self.o = time.time()
-
- def measure(self, p=1):
- x = (time.time() - self.o) / p
- x = int(x)
- if x >= 3600:
- return '{:.1f}h'.format(x / 3600)
- if x >= 60:
- return '{}m'.format(round(x / 60))
- return '{}s'.format(x)
-
-
-class MLLoss(nn.Module):
- def __init__(self):
- super(MLLoss, self).__init__()
-
- def forward(self, input, target, eps=1e-6):
- # 0 - real; 1 - fake.
- loss = torch.tensor(0., device=target.device)
- batch_size = target.shape[0]
- mat_1 = torch.hstack([target.unsqueeze(-1)] * batch_size)
- mat_2 = torch.vstack([target] * batch_size)
- diff_mat = torch.logical_xor(mat_1, mat_2).float()
- or_mat = torch.logical_or(mat_1, mat_2)
- eye = torch.eye(batch_size, device=target.device)
- or_mat = torch.logical_or(or_mat, eye).float()
- sim_mat = 1. - or_mat
- for _ in input:
- diff = torch.sum(_ * diff_mat, dim=[0, 1]) / (torch.sum(diff_mat, dim=[0, 1]) + eps)
- sim = torch.sum(_ * sim_mat, dim=[0, 1]) / (torch.sum(sim_mat, dim=[0, 1]) + eps)
- partial_loss = 1. - sim + diff
- loss += max(partial_loss, torch.zeros_like(partial_loss))
- return loss
-
-
-class AccMeter(object):
- def __init__(self):
- self.nums = 0
- self.acc = 0
-
- def reset(self):
- self.nums = 0
- self.acc = 0
-
- def update(self, pred, target, use_bce=False):
- if use_bce:
- pred = (pred >= 0.5).int()
- else:
- pred = pred.argmax(1)
- self.nums += target.shape[0]
- self.acc += torch.sum(pred == target)
-
- def mean_acc(self):
- return self.acc / self.nums
-
-
-class AUCMeter(object):
- def __init__(self):
- self.score = None
- self.true = None
-
- def reset(self):
- self.score = None
- self.true = None
-
- def update(self, score, true, use_bce=False):
- if use_bce:
- score = score.detach().cpu().numpy()
- else:
- score = torch.softmax(score.detach(), dim=-1)
- score = torch.select(score, 1, 1).cpu().numpy()
- true = true.flatten().cpu().numpy()
- self.score = score if self.score is None else np.concatenate([self.score, score])
- self.true = true if self.true is None else np.concatenate([self.true, true])
-
- def mean_auc(self):
- return roc_auc_score(self.true, self.score)
-
- def curve(self, prefix):
- fpr, tpr, thresholds = roc_curve(self.true, self.score, pos_label=1)
- eer = brentq(lambda x: 1. - x - interp1d(fpr, tpr)(x), 0., 1.)
- thresh = interp1d(fpr, thresholds)(eer)
- print(f"# EER: {eer:.4f}(thresh: {thresh:.4f})")
- torch.save([fpr, tpr, thresholds], os.path.join(prefix, "roc_curve.pickle"))
-
-
-class AverageMeter(object):
- """Computes and stores the average and current value"""
-
- def __init__(self):
- self.val = 0
- self.avg = 0
- self.sum = 0
- self.count = 0
-
- def reset(self):
- self.val = 0
- self.avg = 0
- self.sum = 0
- self.count = 0
-
- def update(self, val, n=1):
- self.val = val
- self.sum += val * n
- self.count += n
- self.avg = self.sum / self.count
diff --git a/video/recce/requirements.txt b/video/recce/requirements.txt
deleted file mode 100644
index f632c804b4cd0d6f82dde683d120aa6bd51c169e..0000000000000000000000000000000000000000
--- a/video/recce/requirements.txt
+++ /dev/null
@@ -1,11 +0,0 @@
-fastapi
-uvicorn
-pydantic
-python-multipart
-torch>=1.8.0
-torchvision>=0.9.0
-timm
-facenet-pytorch
-opencv-python-headless
-numpy<2.0.0
-Pillow
diff --git a/video/sbi/Dockerfile b/video/sbi/Dockerfile
deleted file mode 100644
index e92bd90622129a0f43116ed6bc470a078df16f2f..0000000000000000000000000000000000000000
--- a/video/sbi/Dockerfile
+++ /dev/null
@@ -1,52 +0,0 @@
-FROM nvidia/cuda:12.1.1-cudnn8-runtime-ubuntu22.04
-
-ENV DEBIAN_FRONTEND=noninteractive
-ENV PYTHONUNBUFFERED=1
-
-WORKDIR /app
-
-# Install Python 3.10 and system dependencies for OpenCV
-RUN apt-get update && apt-get install -y --no-install-recommends \
- python3 python3-pip \
- libgl1 \
- libglib2.0-0 \
- libsm6 \
- libxext6 \
- libxrender-dev \
- && rm -rf /var/lib/apt/lists/*
-
-RUN ln -sf /usr/bin/python3 /usr/bin/python
-
-# Install PyTorch with CUDA 12.1
-RUN pip install --no-cache-dir \
- torch==2.5.1 torchvision==0.20.1 \
- --index-url https://download.pytorch.org/whl/cu121
-
-# Copy requirements and install (facenet-pytorch needs --no-deps due to torch<2.3 pin)
-COPY requirements.txt .
-RUN pip install --no-cache-dir --no-deps facenet-pytorch && \
- pip install --no-cache-dir -r requirements.txt
-
-# Create logs and weights directories
-RUN mkdir -p logs weights
-COPY weights/ /app/weights/
-
-# Copy application code
-COPY app.py .
-
-# Environment variables
-ENV MODEL_PORT=7002
-ENV PRELOAD_MODEL=false
-ENV MODEL_TIMEOUT=1800
-ENV WEIGHTS_PATH=/app/weights/FFc23.tar
-
-# Expose port
-EXPOSE 7002
-
-# Drop root privileges
-RUN adduser --disabled-password --gecos '' appuser && \
- chown -R appuser:appuser /app/logs /app/weights
-USER appuser
-
-# Run the service
-CMD ["python", "app.py"]
diff --git a/video/sbi/app.py b/video/sbi/app.py
deleted file mode 100644
index c6752930d604b9f670c340bb5cac0462d87e4587..0000000000000000000000000000000000000000
--- a/video/sbi/app.py
+++ /dev/null
@@ -1,430 +0,0 @@
-"""SBI (Self-Blended Images) deepfake video detection service.
-
-Wraps the SBI (CVPR 2022) face forgery detection model with a
-FastAPI endpoint. Uses EfficientNet-B4 trained on self-blended image
-augmentation for robust cross-dataset face-swap detection.
-
-Reference: Shiohara & Yamasaki, "Detecting Deepfakes with
-Self-Blended Images", CVPR 2022.
-"""
-
-import base64
-import gc
-import logging
-import os
-import platform
-import sys
-import tempfile
-import threading
-import time
-from typing import Any, Dict, List, Optional, Tuple
-
-import cv2
-import numpy as np
-import torch
-import torch.nn.functional as F
-import uvicorn
-from efficientnet_pytorch import EfficientNet
-from facenet_pytorch import MTCNN
-from fastapi import FastAPI, HTTPException
-from PIL import Image
-from pydantic import BaseModel, ConfigDict, Field
-from torch import nn
-
-logging.basicConfig(level=logging.INFO)
-logger = logging.getLogger(__name__)
-
-MODEL_PORT = int(os.environ.get("MODEL_PORT", 7002))
-PRELOAD_MODEL = os.environ.get("PRELOAD_MODEL", "false").lower() == "true"
-MODEL_TIMEOUT = int(os.environ.get("MODEL_TIMEOUT", 1800))
-WEIGHTS_PATH = os.environ.get("WEIGHTS_PATH", "/app/weights/FFc23.tar")
-
-# SBI uses 380x380 face crops (from configs/sbi/base.json)
-IMAGE_SIZE = (380, 380)
-# Number of frames to sample from each video
-NUM_FRAMES = 32
-# Face crop margin factor (matches SBI preprocess.py test-phase behaviour)
-MARGIN_FACTOR = 0.5
-
-
-class Detector(nn.Module):
- """EfficientNet-B4 binary classifier (real vs fake).
-
- Mirrors the inference-time Detector from the SBI repository
- (src/inference/model.py). The training-time SAM optimizer and
- training_step are intentionally omitted.
- """
-
- def __init__(self):
- super().__init__()
- self.net = EfficientNet.from_pretrained(
- "efficientnet-b4", advprop=True, num_classes=2
- )
-
- def forward(self, x: torch.Tensor) -> torch.Tensor:
- """Forward pass returning 2-class logits."""
- return self.net(x)
-
-
-def _get_device() -> torch.device:
- """Select optimal device: MPS (Apple) > CUDA (NVIDIA) > CPU."""
- override = os.environ.get("DEEPSAFE_DEVICE", "").strip().lower()
- if override == "cpu":
- return torch.device("cpu")
- if override == "cuda" and torch.cuda.is_available():
- return torch.device("cuda")
- if (
- override == "mps"
- and hasattr(torch.backends, "mps")
- and torch.backends.mps.is_available()
- ):
- return torch.device("mps")
- if (
- platform.system() == "Darwin"
- and hasattr(torch.backends, "mps")
- and torch.backends.mps.is_available()
- ):
- return torch.device("mps")
- if torch.cuda.is_available():
- return torch.device("cuda")
- return torch.device("cpu")
-
-
-# ── Global state ────────────────────────────────────────────────────────────
-
-_model: Optional[Detector] = None
-_face_detector: Optional[MTCNN] = None
-_device: Optional[torch.device] = None
-_load_lock = threading.Lock()
-
-
-def _load_models() -> None:
- """Load SBI detector and MTCNN face detector (thread-safe)."""
- global _model, _face_detector, _device
-
- if _model is not None:
- return
-
- with _load_lock:
- # Double-check after acquiring lock
- if _model is not None:
- return
-
- _device = _get_device()
- if _device.type == "cuda":
- torch.backends.cudnn.benchmark = True
- torch.set_float32_matmul_precision("high")
- if _device.type == "cuda":
- logger.info(
- "Device: cuda (%s, %.1f GB VRAM)",
- torch.cuda.get_device_name(0),
- torch.cuda.get_device_properties(0).total_memory / 1024**3,
- )
- else:
- logger.warning(
- "Device: %s (no CUDA available -- check nvidia-container-toolkit)",
- _device,
- )
- logger.info("Loading SBI model on %s ...", _device)
-
- # ── Face detector (MTCNN) ───────────────────────────────────────
- _face_detector = MTCNN(
- keep_all=True,
- device=_device,
- post_process=False,
- )
-
- # ── SBI classifier ──────────────────────────────────────────────
- detector = Detector()
-
- if not os.path.exists(WEIGHTS_PATH):
- raise FileNotFoundError(f"SBI weights not found at {WEIGHTS_PATH}")
-
- checkpoint = torch.load(WEIGHTS_PATH, map_location="cpu", weights_only=False)
- state_dict = checkpoint.get("model", checkpoint)
- detector.load_state_dict(state_dict)
- detector = detector.to(_device)
- detector.eval()
-
- _model = detector
- logger.info("SBI model loaded successfully.")
-
-
-def _is_model_loaded() -> bool:
- """Return True if both the classifier and face detector are loaded."""
- return _model is not None and _face_detector is not None
-
-
-# ── FastAPI app ─────────────────────────────────────────────────────────────
-
-app = FastAPI(
- title="SBI Detection Service",
- description=(
- "Detecting Deepfakes with Self-Blended Images " "(EfficientNet-B4, CVPR 2022)"
- ),
- version="1.0.0",
-)
-
-
-class PredictRequest(BaseModel):
- """Incoming prediction request."""
-
- video_data: str # Base64-encoded video bytes
- threshold: float = 0.5
-
-
-class PredictResponse(BaseModel):
- """Outgoing prediction result."""
-
- model_config = ConfigDict(populate_by_name=True)
-
- model: str = "sbi_detection"
- probability: float
- prediction: int
- class_name: str = Field(..., alias="class")
- inference_time: float
- metadata: Dict[str, Any]
-
-
-@app.on_event("startup")
-async def startup_event():
- """Optionally preload model at startup."""
- if PRELOAD_MODEL:
- _load_models()
-
-
-@app.get("/")
-def root():
- """Service info endpoint."""
- return {
- "service": "sbi_detection",
- "port": MODEL_PORT,
- "model_loaded": _is_model_loaded(),
- "device": str(_device) if _device else "unknown",
- }
-
-
-def _gpu_health_info() -> dict:
- """Return GPU metrics for the health endpoint."""
- if torch.cuda.is_available() and _device is not None and _device.type == "cuda":
- return {
- "gpu_name": torch.cuda.get_device_name(0),
- "vram_used_mb": round(torch.cuda.memory_allocated(0) / 1024**2),
- "vram_total_mb": round(
- torch.cuda.get_device_properties(0).total_memory / 1024**2
- ),
- }
- return {}
-
-
-@app.get("/health")
-def health():
- """Health check endpoint."""
- return {
- "status": "healthy",
- "model": "sbi_detection",
- "device": str(_device) if _device else "cpu",
- "model_loaded": _is_model_loaded(),
- "weights_exist": os.path.exists(WEIGHTS_PATH),
- **_gpu_health_info(),
- }
-
-
-# ── Video / face utilities ──────────────────────────────────────────────────
-
-
-def _extract_frames(video_path: str, num_frames: int = NUM_FRAMES) -> List[np.ndarray]:
- """Uniformly sample *num_frames* RGB frames from a video file.
-
- Args:
- video_path: Path to the video on disk.
- num_frames: Number of frames to extract.
-
- Returns:
- List of RGB uint8 numpy arrays (H, W, 3).
- """
- cap = cv2.VideoCapture(video_path)
- total = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))
-
- if total <= 0:
- cap.release()
- return []
-
- indices = np.linspace(0, total - 1, num_frames, endpoint=True, dtype=int)
- frames: List[np.ndarray] = []
-
- for idx in indices:
- cap.set(cv2.CAP_PROP_POS_FRAMES, int(idx))
- ret, frame = cap.read()
- if ret:
- frames.append(cv2.cvtColor(frame, cv2.COLOR_BGR2RGB))
-
- cap.release()
- return frames
-
-
-def _crop_face(
- img: np.ndarray,
- bbox: Tuple[float, float, float, float],
- margin: float = MARGIN_FACTOR,
-) -> np.ndarray:
- """Crop a face region from an image with a relative margin.
-
- Mirrors the test-phase crop logic from SBI preprocess.py where
- margin multipliers are 0.5 in each direction.
-
- Args:
- img: RGB image array (H, W, 3).
- bbox: (x0, y0, x1, y1) face bounding box.
- margin: Fraction of bbox dimension to add as padding.
-
- Returns:
- Cropped face region as a numpy array.
- """
- h_img, w_img = img.shape[:2]
- x0, y0, x1, y1 = bbox
- w = x1 - x0
- h = y1 - y0
-
- x0_new = max(0, int(x0 - w * margin / 2))
- x1_new = min(w_img, int(x1 + w * margin / 2) + 1)
- y0_new = max(0, int(y0 - h * margin / 2))
- y1_new = min(h_img, int(y1 + h * margin / 2) + 1)
-
- return img[y0_new:y1_new, x0_new:x1_new]
-
-
-def _detect_and_crop_faces(
- frame: np.ndarray,
-) -> List[np.ndarray]:
- """Detect faces in a single frame and return cropped+resized chips.
-
- Uses MTCNN for face detection, then crops with margin and resizes
- to IMAGE_SIZE (380x380).
-
- Args:
- frame: RGB image array (H, W, 3).
-
- Returns:
- List of face crops resized to IMAGE_SIZE, as uint8 arrays.
- """
- assert _face_detector is not None
-
- pil_img = Image.fromarray(frame)
- boxes, _ = _face_detector.detect(pil_img)
-
- if boxes is None or len(boxes) == 0:
- return []
-
- crops: List[np.ndarray] = []
- for box in boxes:
- x0, y0, x1, y1 = box.tolist()
- face = _crop_face(frame, (x0, y0, x1, y1))
- if face.size == 0:
- continue
- resized = cv2.resize(face, IMAGE_SIZE)
- crops.append(resized)
-
- return crops
-
-
-# ── Prediction endpoint ─────────────────────────────────────────────────────
-
-
-@app.post("/predict", response_model=PredictResponse)
-async def predict(request: PredictRequest):
- """Run SBI face-forgery detection on a base64-encoded video.
-
- Pipeline:
- 1. Decode video and write to temp file.
- 2. Extract uniformly-sampled frames.
- 3. Detect and crop faces per frame (MTCNN).
- 4. Classify each face crop with EfficientNet-B4.
- 5. For each frame, take the max probability across faces.
- 6. Average the per-frame max probabilities.
-
- If no faces are detected in any frame the service returns
- probability=0.5 (undetermined) rather than raising an error.
- """
- if not _is_model_loaded():
- _load_models()
-
- start_time = time.time()
-
- # ── Decode video ────────────────────────────────────────────────
- with tempfile.NamedTemporaryFile(suffix=".mp4", delete=False) as tmp:
- try:
- video_bytes = base64.b64decode(request.video_data)
- tmp.write(video_bytes)
- tmp_path = tmp.name
- except Exception as e:
- raise HTTPException(status_code=400, detail=f"Failed to decode video: {e}")
-
- try:
- # ── Extract frames ──────────────────────────────────────────
- frames = _extract_frames(tmp_path, NUM_FRAMES)
- if not frames:
- raise HTTPException(
- status_code=400,
- detail="Could not extract frames from video.",
- )
-
- # ── Detect faces and classify ───────────────────────────────
- per_frame_max: List[float] = []
- total_faces = 0
-
- for frame in frames:
- crops = _detect_and_crop_faces(frame)
- if not crops:
- continue
-
- # Build tensor: (N, C, H, W) float32 in [0, 1]
- batch = np.stack(crops, axis=0) # (N, H, W, C) uint8
- batch_tensor = (
- torch.tensor(batch).permute(0, 3, 1, 2).float().div(255.0).to(_device)
- )
-
- with torch.no_grad():
- logits = _model(batch_tensor) # (N, 2)
- probs = F.softmax(logits, dim=1)[:, 1] # fake prob
-
- frame_max = probs.max().cpu().item()
- per_frame_max.append(frame_max)
- total_faces += len(crops)
-
- # ── Aggregate ───────────────────────────────────────────────
- if per_frame_max:
- probability = float(np.mean(per_frame_max))
- else:
- # No faces detected in any frame — undetermined
- probability = 0.5
-
- prediction = 1 if probability >= request.threshold else 0
- class_name = "fake" if prediction == 1 else "real"
-
- return PredictResponse(
- probability=probability,
- prediction=prediction,
- class_name=class_name,
- inference_time=time.time() - start_time,
- metadata={
- "frames_sampled": len(frames),
- "frames_with_faces": len(per_frame_max),
- "total_faces_detected": total_faces,
- "device": str(_device),
- },
- )
-
- except HTTPException:
- raise
- except Exception as e:
- logger.exception("Error during SBI prediction")
- raise HTTPException(status_code=500, detail=str(e))
- finally:
- if os.path.exists(tmp_path):
- os.remove(tmp_path)
- gc.collect()
-
-
-if __name__ == "__main__":
- uvicorn.run(app, host="0.0.0.0", port=MODEL_PORT)
diff --git a/video/sbi/requirements.txt b/video/sbi/requirements.txt
deleted file mode 100644
index 569d05d0e01da28635f9c28405f3041756d164bb..0000000000000000000000000000000000000000
--- a/video/sbi/requirements.txt
+++ /dev/null
@@ -1,11 +0,0 @@
-fastapi
-uvicorn
-pydantic
-python-multipart
-torch>=1.8.0
-torchvision>=0.9.0
-efficientnet_pytorch
-facenet-pytorch
-opencv-python-headless
-numpy<2.0.0
-Pillow