初始化换发型项目:3个微服务代码 + 部署脚本
包含: - hair_service_sd: 换发型/换发色算法服务 (端口 8801) - photo_service: LoRA 训练调度服务 (端口 32678) - stable-diffusion-webui: SD WebUI 推理服务 (端口 57860) - kohya_ss_home: 训练环境代码 - meidaojia: 监控测试脚本 - setup.sh: 一键部署脚本 (conda环境恢复 + 配置生成 + 完整性检查) - start_all_services.sh: 启动3个服务 - configure.ini.template: 路径模板化 (BASE_DIR自动推导) - conda_envs/py310.yml: py310 环境定义 大文件 (weights/, models/, data/, conda_envs/*.tar.gz 等) 通过 .gitignore 排除, 由网盘单独上传。
This commit is contained in:
@@ -0,0 +1,69 @@
|
||||
import os
|
||||
import time
|
||||
import torch
|
||||
import torch.nn.parallel
|
||||
|
||||
import numpy as np
|
||||
from utils import landmark_processor
|
||||
import cv2
|
||||
|
||||
# modelRoot = "/home/yangchaojie/Desktop/hairstyle/hairstyle_infer/weights"
|
||||
modelRoot = "./weights"
|
||||
|
||||
label_map = [
|
||||
[0, 0, 0],
|
||||
[255, 0, 0],
|
||||
[0, 0, 255],
|
||||
[0, 255, 0],
|
||||
[0, 255, 255],
|
||||
# [0, 255, 0]
|
||||
]
|
||||
class Generator_BaldSeg_5c(object):
|
||||
def __init__(self, gpu_flag, gpu_id):
|
||||
|
||||
if not gpu_flag:
|
||||
self.device = torch.device("cpu")
|
||||
else:
|
||||
self.device = torch.device('cuda:{0}'.format(gpu_id))
|
||||
|
||||
# load seg model
|
||||
self.model_dir = modelRoot
|
||||
self.pre_trained_model = os.path.join(self.model_dir, "ori_hair_checkpoint_7660_0611.pt")
|
||||
self.net = torch.jit.load(self.pre_trained_model, map_location=self.device).to(self.device)
|
||||
self.net.eval()
|
||||
|
||||
self.output_img_size = 512
|
||||
self.img_ratio = 0.4
|
||||
|
||||
def label_to_mask(self, label_np):
|
||||
label_np = label_np.astype(np.int32)[:, :, np.newaxis]
|
||||
mask = np.zeros((label_np.shape[0], label_np.shape[1], 3), dtype=np.uint8)
|
||||
for id, color in enumerate(label_map):
|
||||
index = (label_np == id).all(axis=2)
|
||||
mask[index] = color
|
||||
return mask
|
||||
def forward(self, image, alpha, landmarks1k):
|
||||
image_to_face_mat = landmark_processor.get_transform_mat_full_face_ratio_deeplab(landmarks1k, self.output_img_size, self.img_ratio)
|
||||
img_ = (image * (1 - alpha[:, :, np.newaxis].astype(np.float32) / 255)).astype(np.uint8)
|
||||
img_cuted = cv2.warpAffine(img_, image_to_face_mat, (self.output_img_size, self.output_img_size), flags=cv2.INTER_LANCZOS4, borderMode=cv2.BORDER_CONSTANT, borderValue=[255, 255, 255])
|
||||
inv_image_to_face_mat = cv2.invertAffineTransform(image_to_face_mat)
|
||||
|
||||
img = (img_cuted.astype(np.float32) / 255).transpose((2, 0, 1))
|
||||
img = np.expand_dims(img, axis=0)
|
||||
img = torch.from_numpy(img).to(self.device) #cuda(self.gpu_id)
|
||||
|
||||
with torch.no_grad():
|
||||
output = self.net(img)
|
||||
|
||||
pred = output.detach().cpu().numpy().squeeze().astype(np.float32) #torch.max(output[:1], 1)[1].detach().cpu().numpy().squeeze().astype(np.float32)
|
||||
mask = self.label_to_mask(pred)
|
||||
|
||||
# cv2.imshow("img_cuted: ", img_cuted)
|
||||
# cv2.imshow("mask: ", mask)
|
||||
# cv2.waitKey()
|
||||
|
||||
black_img = np.zeros(image.shape).astype(np.uint8)
|
||||
cv2.warpAffine(mask, inv_image_to_face_mat, (image.shape[1], image.shape[0]),
|
||||
dst=black_img, flags=cv2.INTER_NEAREST, borderMode=cv2.BORDER_TRANSPARENT)
|
||||
|
||||
return black_img
|
||||
@@ -0,0 +1,429 @@
|
||||
import torch.nn as nn
|
||||
import torch.utils.model_zoo as model_zoo
|
||||
import torch
|
||||
import numpy as np
|
||||
import os
|
||||
from utils import landmark_processor
|
||||
from utils.umeyama import umeyama
|
||||
import cv2
|
||||
modelRoot = "./weights"
|
||||
|
||||
__all__ = ['ResNet', 'resnet18', 'resnet34', 'resnet50', 'resnet101',
|
||||
'resnet152']
|
||||
|
||||
|
||||
model_urls = {
|
||||
'resnet18': 'https://download.pytorch.org/models/resnet18-5c106cde.pth',
|
||||
'resnet34': 'https://download.pytorch.org/models/resnet34-333f7ec4.pth',
|
||||
'resnet50': 'https://download.pytorch.org/models/resnet50-19c8e357.pth',
|
||||
'resnet101': 'https://download.pytorch.org/models/resnet101-5d3b4d8f.pth',
|
||||
'resnet152': 'https://download.pytorch.org/models/resnet152-b121ed2d.pth',
|
||||
}
|
||||
|
||||
|
||||
def conv3x3(in_planes, out_planes, stride=1):
|
||||
"""3x3 convolution with padding"""
|
||||
return nn.Conv2d(in_planes, out_planes, kernel_size=3, stride=stride,
|
||||
padding=1, bias=False)
|
||||
|
||||
|
||||
def conv1x1(in_planes, out_planes, stride=1):
|
||||
"""1x1 convolution"""
|
||||
return nn.Conv2d(in_planes, out_planes, kernel_size=1, stride=stride, bias=False)
|
||||
|
||||
|
||||
class BasicBlock(nn.Module):
|
||||
expansion = 1
|
||||
|
||||
def __init__(self, inplanes, planes, stride=1, downsample=None):
|
||||
super(BasicBlock, self).__init__()
|
||||
self.conv1 = conv3x3(inplanes, planes, stride)
|
||||
self.bn1 = nn.BatchNorm2d(planes)
|
||||
self.relu = nn.ReLU(inplace=True)
|
||||
self.conv2 = conv3x3(planes, planes)
|
||||
self.bn2 = nn.BatchNorm2d(planes)
|
||||
self.downsample = downsample
|
||||
self.stride = stride
|
||||
|
||||
def forward(self, x):
|
||||
identity = x
|
||||
|
||||
out = self.conv1(x)
|
||||
out = self.bn1(out)
|
||||
out = self.relu(out)
|
||||
|
||||
out = self.conv2(out)
|
||||
out = self.bn2(out)
|
||||
|
||||
if self.downsample is not None:
|
||||
identity = self.downsample(x)
|
||||
|
||||
out += identity
|
||||
out = self.relu(out)
|
||||
|
||||
return out
|
||||
|
||||
|
||||
class Bottleneck(nn.Module):
|
||||
expansion = 4
|
||||
|
||||
def __init__(self, inplanes, planes, stride=1, downsample=None):
|
||||
super(Bottleneck, self).__init__()
|
||||
self.conv1 = conv1x1(inplanes, planes)
|
||||
self.bn1 = nn.BatchNorm2d(planes)
|
||||
self.conv2 = conv3x3(planes, planes, stride)
|
||||
self.bn2 = nn.BatchNorm2d(planes)
|
||||
self.conv3 = conv1x1(planes, planes * self.expansion)
|
||||
self.bn3 = nn.BatchNorm2d(planes * self.expansion)
|
||||
self.relu = nn.ReLU(inplace=True)
|
||||
self.downsample = downsample
|
||||
self.stride = stride
|
||||
|
||||
def forward(self, x):
|
||||
identity = x
|
||||
|
||||
out = self.conv1(x)
|
||||
out = self.bn1(out)
|
||||
out = self.relu(out)
|
||||
|
||||
out = self.conv2(out)
|
||||
out = self.bn2(out)
|
||||
out = self.relu(out)
|
||||
|
||||
out = self.conv3(out)
|
||||
out = self.bn3(out)
|
||||
|
||||
if self.downsample is not None:
|
||||
identity = self.downsample(x)
|
||||
|
||||
out += identity
|
||||
out = self.relu(out)
|
||||
|
||||
return out
|
||||
|
||||
|
||||
class ResNet(nn.Module):
|
||||
|
||||
def __init__(self, block, layers, num_classes=1000, is_1k=False, zero_init_residual=False):
|
||||
super(ResNet, self).__init__()
|
||||
self.inplanes = 64
|
||||
self.conv1 = nn.Conv2d(3, 64, kernel_size=7, stride=2, padding=3,
|
||||
bias=False)
|
||||
self.bn1 = nn.BatchNorm2d(64)
|
||||
self.relu = nn.ReLU(inplace=True)
|
||||
self.maxpool = nn.MaxPool2d(kernel_size=3, stride=2, padding=1)
|
||||
self.layer1 = self._make_layer(block, 64, layers[0])
|
||||
self.layer2 = self._make_layer(block, 128, layers[1], stride=2)
|
||||
self.layer3 = self._make_layer(block, 256, layers[2], stride=2)
|
||||
self.layer4 = self._make_layer(block, 512, layers[3], stride=2)
|
||||
self.avgpool = nn.AdaptiveAvgPool2d((1, 1))
|
||||
if is_1k:
|
||||
self.fc = nn.Sequential(*[nn.Linear(512 * block.expansion, num_classes), nn.Tanh()])
|
||||
else:
|
||||
self.fc_key = nn.Sequential(*[nn.Linear(256 * block.expansion, 45 * 2), nn.Tanh()])
|
||||
self.fc_ctrl = nn.Sequential(*[nn.Linear(256 * block.expansion, 48 * 2), nn.Tanh()])
|
||||
self.is_1k = is_1k
|
||||
|
||||
for m in self.modules():
|
||||
if isinstance(m, nn.Conv2d):
|
||||
nn.init.kaiming_normal_(m.weight, mode='fan_out', nonlinearity='relu')
|
||||
elif isinstance(m, nn.BatchNorm2d):
|
||||
nn.init.constant_(m.weight, 1)
|
||||
nn.init.constant_(m.bias, 0)
|
||||
|
||||
# Zero-initialize the last BN in each residual branch,
|
||||
# so that the residual branch starts with zeros, and each residual block behaves like an identity.
|
||||
# This improves the model by 0.2~0.3% according to https://arxiv.org/abs/1706.02677
|
||||
if zero_init_residual:
|
||||
for m in self.modules():
|
||||
if isinstance(m, Bottleneck):
|
||||
nn.init.constant_(m.bn3.weight, 0)
|
||||
elif isinstance(m, BasicBlock):
|
||||
nn.init.constant_(m.bn2.weight, 0)
|
||||
|
||||
def _make_layer(self, block, planes, blocks, stride=1):
|
||||
downsample = None
|
||||
if stride != 1 or self.inplanes != planes * block.expansion:
|
||||
downsample = nn.Sequential(
|
||||
conv1x1(self.inplanes, planes * block.expansion, stride),
|
||||
nn.BatchNorm2d(planes * block.expansion),
|
||||
)
|
||||
|
||||
layers = []
|
||||
layers.append(block(self.inplanes, planes, stride, downsample))
|
||||
self.inplanes = planes * block.expansion
|
||||
for _ in range(1, blocks):
|
||||
layers.append(block(self.inplanes, planes))
|
||||
|
||||
return nn.Sequential(*layers)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.conv1(x)
|
||||
x = self.bn1(x)
|
||||
x = self.relu(x)
|
||||
x = self.maxpool(x)
|
||||
|
||||
x = self.layer1(x)
|
||||
x = self.layer2(x)
|
||||
x = self.layer3(x)
|
||||
x = self.layer4(x)
|
||||
|
||||
if self.is_1k:
|
||||
key = self.avgpool(x)
|
||||
key = key.view(key.size(0), -1)
|
||||
key = self.fc(key)
|
||||
return key
|
||||
else:
|
||||
key, ctrl = torch.chunk(x, 2, dim=1)
|
||||
|
||||
key = self.avgpool(key)
|
||||
key = key.view(key.size(0), -1)
|
||||
key = self.fc_key(key)
|
||||
|
||||
ctrl = self.avgpool(ctrl)
|
||||
ctrl = ctrl.view(ctrl.size(0), -1)
|
||||
ctrl = self.fc_ctrl(ctrl)
|
||||
|
||||
return key, ctrl
|
||||
|
||||
def resnet18(pretrained=False, **kwargs):
|
||||
"""Constructs a ResNet-18 model.
|
||||
|
||||
Args:
|
||||
pretrained (bool): If True, returns a model pre-trained on ImageNet
|
||||
"""
|
||||
model = ResNet(BasicBlock, [2, 2, 2, 2], **kwargs)
|
||||
if pretrained:
|
||||
model.load_state_dict(model_zoo.load_url(model_urls['resnet18']), strict=False)
|
||||
return model
|
||||
|
||||
|
||||
def resnet34(pretrained=False, **kwargs):
|
||||
"""Constructs a ResNet-34 model.
|
||||
|
||||
Args:
|
||||
pretrained (bool): If True, returns a model pre-trained on ImageNet
|
||||
"""
|
||||
model = ResNet(BasicBlock, [3, 4, 6, 3], **kwargs)
|
||||
if pretrained:
|
||||
model.load_state_dict(model_zoo.load_url(model_urls['resnet34']))
|
||||
return model
|
||||
|
||||
|
||||
def resnet50(pretrained=False, **kwargs):
|
||||
"""Constructs a ResNet-50 model.
|
||||
|
||||
Args:
|
||||
pretrained (bool): If True, returns a model pre-trained on ImageNet
|
||||
"""
|
||||
model = ResNet(Bottleneck, [3, 4, 6, 3], **kwargs)
|
||||
if pretrained:
|
||||
model.load_state_dict(model_zoo.load_url(model_urls['resnet50']))
|
||||
return model
|
||||
|
||||
|
||||
def resnet101(pretrained=False, **kwargs):
|
||||
"""Constructs a ResNet-101 model.
|
||||
|
||||
Args:
|
||||
pretrained (bool): If True, returns a model pre-trained on ImageNet
|
||||
"""
|
||||
model = ResNet(Bottleneck, [3, 4, 23, 3], **kwargs)
|
||||
if pretrained:
|
||||
model.load_state_dict(model_zoo.load_url(model_urls['resnet101']))
|
||||
return model
|
||||
|
||||
|
||||
def resnet152(pretrained=False, **kwargs):
|
||||
"""Constructs a ResNet-152 model.
|
||||
|
||||
Args:
|
||||
pretrained (bool): If True, returns a model pre-trained on ImageNet
|
||||
"""
|
||||
model = ResNet(Bottleneck, [3, 8, 36, 3], **kwargs)
|
||||
if pretrained:
|
||||
model.load_state_dict(model_zoo.load_url(model_urls['resnet152']))
|
||||
return model
|
||||
|
||||
class Model1k(nn.Module):
|
||||
def __init__(self, gpu_id=None):
|
||||
super(Model1k, self).__init__()
|
||||
|
||||
self.device = torch.device('cuda:{}'.format(gpu_id) if gpu_id is not None else 'cpu')
|
||||
|
||||
self.face_alignment_net = resnet18(pretrained=False, num_classes=1000 * 2, is_1k=True)
|
||||
|
||||
self.model_path, _ = os.path.split(os.path.realpath(__file__))
|
||||
# weights = torch.load(os.path.join(self.model_path, 'face_alignment_1k.pth'), map_location=lambda storage, loc: storage)
|
||||
weights = torch.load(os.path.join(modelRoot, 'face_alignment_1k.pth'), map_location=lambda storage, loc: storage)
|
||||
self.load_state_dict(weights)
|
||||
self.to(self.device)
|
||||
self.eval()
|
||||
|
||||
def forward(self, imgs):
|
||||
pred_key_pts = self.face_alignment_net(imgs)
|
||||
pred_key_pts = pred_key_pts + 0.5
|
||||
return pred_key_pts
|
||||
|
||||
class MomocvFaceAlignment1K(object):
|
||||
def __init__(self, gpu_id=None):
|
||||
self.gpu_id = gpu_id
|
||||
|
||||
self.device = torch.device('cuda:{}'.format(gpu_id) if gpu_id is not None else 'cpu')
|
||||
|
||||
self.face_alignment_net = Model1k(gpu_id)
|
||||
|
||||
self.trackingFaceRects = []
|
||||
|
||||
print('conansherry MomocvFaceAlignment1K')
|
||||
|
||||
def forward(self, img_tensor):
|
||||
fullyconnected1 = self.face_alignment_net(img_tensor).detach().cpu().numpy()
|
||||
return fullyconnected1
|
||||
|
||||
def detect(self, img, landmarks):
|
||||
dst_size = 256
|
||||
landmarks_res = []
|
||||
with torch.no_grad():
|
||||
input_numpy = np.zeros((len(landmarks), 3, dst_size, dst_size), dtype=np.float32)
|
||||
all_mat = []
|
||||
for ix, landmark in enumerate(landmarks):
|
||||
M = landmark_processor.get_transform_mat_full_face(landmark, dst_size)
|
||||
all_mat.append(M)
|
||||
tmp = cv2.warpAffine(img, M, (dst_size, dst_size))
|
||||
|
||||
# cv2.imshow('inp', tmp)
|
||||
# cv2.waitKey()
|
||||
|
||||
input_numpy[ix, :, :, :] = tmp.transpose((2, 0, 1)).astype(np.float32) / 255
|
||||
in_tensor = torch.from_numpy(input_numpy)
|
||||
in_tensor = in_tensor.to(self.device)
|
||||
fullyconnected1 = self.face_alignment_net(in_tensor).detach().cpu().numpy()
|
||||
for ix, pts in enumerate(fullyconnected1):
|
||||
orig_pts = (np.reshape(pts, (2, 1000)).transpose((1, 0)) * dst_size)
|
||||
orig_pts = landmark_processor.transform_points(orig_pts, all_mat[ix], invert=True)
|
||||
landmarks_res.append(orig_pts)
|
||||
return landmarks_res
|
||||
def detect_single_face(self, img):
|
||||
dst_size = 256
|
||||
with torch.no_grad():
|
||||
input_numpy = np.zeros((1, 3, dst_size, dst_size), dtype=np.float32)
|
||||
tmp = cv2.resize(img, (dst_size, dst_size))
|
||||
input_numpy[0, :, :, :] = tmp.transpose((2, 0, 1)).astype(np.float32) / 255
|
||||
in_tensor = torch.from_numpy(input_numpy)
|
||||
in_tensor = in_tensor.to(self.device)
|
||||
fullyconnected1 = self.face_alignment_net(in_tensor).detach().cpu().numpy()
|
||||
orig_pts = (np.reshape(fullyconnected1[0], (2, 1000)).transpose((1, 0)))
|
||||
orig_pts[:, 0] *= img.shape[1]
|
||||
orig_pts[:, 1] *= img.shape[0]
|
||||
return orig_pts
|
||||
|
||||
def detect_single_face_old(self, img):
|
||||
dst_size = 256
|
||||
with torch.no_grad():
|
||||
tmp = cv2.resize(img, (dst_size, dst_size))
|
||||
input_numpy = np.zeros((1, 3, dst_size, dst_size), dtype=np.float32)
|
||||
input_numpy[0, :, :, :] = tmp.transpose((2, 0, 1)).astype(np.float32) / 255
|
||||
in_tensor = torch.from_numpy(input_numpy)
|
||||
in_tensor = in_tensor.to(self.device)
|
||||
fullyconnected1 = self.face_alignment_net(in_tensor).detach().cpu().numpy()
|
||||
orig_pts = (np.reshape(fullyconnected1[0], (2, 1000)).transpose((1, 0)) * img.shape[0])
|
||||
return orig_pts
|
||||
def detect_according_5pts(self, img, pts5):
|
||||
dst_size = 256
|
||||
with torch.no_grad():
|
||||
input_numpy = np.zeros((1, 3, dst_size, dst_size), dtype=np.float32)
|
||||
eye_dis = 0.34
|
||||
mouth_dis = 0.34
|
||||
g_Average_5point_180 = np.array([
|
||||
eye_dis, 0.3,
|
||||
1 - eye_dis, 0.3,
|
||||
0.5, 0.6,
|
||||
mouth_dis, 0.63,
|
||||
1 - mouth_dis, 0.63
|
||||
])
|
||||
# print(g_Average_5point_180)
|
||||
left_eye = np.array([pts5[0], pts5[5]])
|
||||
right_eye = np.array([pts5[1], pts5[6]])
|
||||
nose = np.array([pts5[2], pts5[7]])
|
||||
left_mouth = np.array([pts5[3], pts5[8]])
|
||||
right_mouth = np.array([pts5[4], pts5[9]])
|
||||
|
||||
pts5_src = np.vstack((left_eye, right_eye,
|
||||
nose,
|
||||
left_mouth, right_mouth))
|
||||
pts5_src = np.array(pts5_src).astype(np.int32)
|
||||
pts5_dst = g_Average_5point_180.reshape((5, -1)) * dst_size
|
||||
|
||||
mat = umeyama(pts5_src, pts5_dst, True)[0:2]
|
||||
|
||||
tmp = cv2.warpAffine(img, mat, (dst_size, dst_size))
|
||||
# cv2.imshow("tmp", tmp)
|
||||
# cv2.waitKey()
|
||||
input_numpy[0, :, :, :] = tmp.transpose((2, 0, 1)).astype(np.float32) / 255
|
||||
in_tensor = torch.from_numpy(input_numpy)
|
||||
in_tensor = in_tensor.to(self.device)
|
||||
fullyconnected1 = self.face_alignment_net(in_tensor).detach().cpu().numpy()
|
||||
orig_pts = (np.reshape(fullyconnected1[0], (2, 1000)).transpose((1, 0)) * dst_size)
|
||||
orig_pts = landmark_processor.transform_points(orig_pts, mat, invert=True)
|
||||
return orig_pts
|
||||
|
||||
def stable_forward(self, image, detected_faces, reset=False):
|
||||
if reset is True:
|
||||
self.trackingFaceRects = []
|
||||
|
||||
if len(self.trackingFaceRects) == 0:
|
||||
for face_rect in detected_faces:
|
||||
new_tracking_rect = [face_rect, True, [0, 0], 0, None]
|
||||
self.trackingFaceRects.append(new_tracking_rect)
|
||||
|
||||
with torch.no_grad():
|
||||
landmarks = []
|
||||
for ix, tracking_face_rect in enumerate(self.trackingFaceRects):
|
||||
if tracking_face_rect[1] == True:
|
||||
d = tracking_face_rect[0]
|
||||
src_center = np.array([d[2] - (d[2] - d[0]) / 2.0, d[3] - (d[3] - d[1]) / 2.0])
|
||||
rotate_degree = tracking_face_rect[3]
|
||||
scale = 256 * 0.6 / min(d[2] - d[0], d[3] - d[1])
|
||||
dst_center = np.array([0.5, 0.5]) * 256
|
||||
offset = dst_center - src_center
|
||||
M = cv2.getRotationMatrix2D((src_center[0], src_center[1]), rotate_degree, scale)
|
||||
M[:, 2] += offset
|
||||
else:
|
||||
rotate_degree = 0
|
||||
M = landmark_processor.get_transform_mat_mmcv_bigger(tracking_face_rect[4], 256)
|
||||
inp = cv2.warpAffine(image, M, (256, 256))
|
||||
|
||||
# cv2.imshow('inp_{}'.format(ix), inp)
|
||||
# cv2.waitKey()
|
||||
orig_inp = inp
|
||||
|
||||
inp = inp.transpose((2, 0, 1)).astype(np.float32)
|
||||
inp = inp[np.newaxis, :, :, :] / 255
|
||||
|
||||
in_tensor = torch.from_numpy(inp)
|
||||
in_tensor = in_tensor.cuda(0)
|
||||
fullyconnected1 = self.forward(in_tensor)
|
||||
fullyconnected1 = fullyconnected1[0]
|
||||
orig_pts = (np.reshape(fullyconnected1, (2, 1000)).transpose((1, 0))) * 256
|
||||
|
||||
t2 = cv2.getTickCount()
|
||||
orig_pts = landmark_processor.transform_points(orig_pts, M, invert=True)
|
||||
|
||||
# orig_pts = orig_pts.transpose((1, 0))
|
||||
fullyconnected1 = orig_pts
|
||||
|
||||
# update tracking infos
|
||||
tracking_face_rect[1] = False
|
||||
tracking_face_rect[2] = None
|
||||
tracking_face_rect[3] = rotate_degree
|
||||
tracking_face_rect[4] = fullyconnected1
|
||||
|
||||
# fullyconnected1 = landmark_processor.pts_1k_to_137(fullyconnected1)
|
||||
|
||||
# eye_landmark = self.detect_eye(image, fullyconnected1)
|
||||
# fullyconnected1[87:104] = eye_landmark[0]
|
||||
# fullyconnected1[104:121] = eye_landmark[1]
|
||||
|
||||
landmarks.append(fullyconnected1)
|
||||
return landmarks
|
||||
@@ -0,0 +1,133 @@
|
||||
import numpy as np
|
||||
import cv2
|
||||
|
||||
def nms(boxes, overlap_threshold=0.5, mode='union'):
|
||||
""" Pure Python NMS baseline. """
|
||||
x1 = boxes[:, 0]
|
||||
y1 = boxes[:, 1]
|
||||
x2 = boxes[:, 2]
|
||||
y2 = boxes[:, 3]
|
||||
scores = boxes[:, 4]
|
||||
|
||||
areas = (x2 - x1 + 1) * (y2 - y1 + 1)
|
||||
order = scores.argsort()[::-1]
|
||||
|
||||
keep = []
|
||||
while order.size > 0:
|
||||
i = order[0]
|
||||
keep.append(i)
|
||||
xx1 = np.maximum(x1[i], x1[order[1:]])
|
||||
yy1 = np.maximum(y1[i], y1[order[1:]])
|
||||
xx2 = np.minimum(x2[i], x2[order[1:]])
|
||||
yy2 = np.minimum(y2[i], y2[order[1:]])
|
||||
|
||||
w = np.maximum(0.0, xx2 - xx1 + 1)
|
||||
h = np.maximum(0.0, yy2 - yy1 + 1)
|
||||
inter = w * h
|
||||
|
||||
if mode == 'min':
|
||||
ovr = inter / np.minimum(areas[i], areas[order[1:]])
|
||||
else:
|
||||
ovr = inter / (areas[i] + areas[order[1:]] - inter)
|
||||
|
||||
inds = np.where(ovr <= overlap_threshold)[0]
|
||||
order = order[inds + 1]
|
||||
|
||||
return keep
|
||||
|
||||
|
||||
def convert_to_square(bboxes):
|
||||
"""
|
||||
Convert bounding boxes to a square form.
|
||||
"""
|
||||
square_bboxes = np.zeros_like(bboxes)
|
||||
x1, y1, x2, y2 = [bboxes[:, i] for i in range(4)]
|
||||
h = y2 - y1 + 1.0
|
||||
w = x2 - x1 + 1.0
|
||||
max_side = np.maximum(h, w)
|
||||
square_bboxes[:, 0] = x1 + w*0.5 - max_side*0.5
|
||||
square_bboxes[:, 1] = y1 + h*0.5 - max_side*0.5
|
||||
square_bboxes[:, 2] = square_bboxes[:, 0] + max_side - 1.0
|
||||
square_bboxes[:, 3] = square_bboxes[:, 1] + max_side - 1.0
|
||||
return square_bboxes
|
||||
|
||||
|
||||
def calibrate_box(bboxes, offsets):
|
||||
"""Transform bounding boxes to be more like true bounding boxes.
|
||||
'offsets' is one of the outputs of the nets.
|
||||
"""
|
||||
x1, y1, x2, y2 = [bboxes[:, i] for i in range(4)]
|
||||
w = x2 - x1 + 1.0
|
||||
h = y2 - y1 + 1.0
|
||||
w = np.expand_dims(w, 1)
|
||||
h = np.expand_dims(h, 1)
|
||||
|
||||
translation = np.hstack([w, h, w, h])*offsets
|
||||
bboxes[:, 0:4] = bboxes[:, 0:4] + translation
|
||||
return bboxes
|
||||
|
||||
|
||||
def get_image_boxes(bounding_boxes, img, size=24):
|
||||
"""Cut out boxes from the image.
|
||||
"""
|
||||
num_boxes = len(bounding_boxes)
|
||||
(height, width, _) = img.shape
|
||||
|
||||
[dy, edy, dx, edx, y, ey, x, ex, w, h] = correct_bboxes(bounding_boxes, width, height)
|
||||
img_boxes = np.zeros((num_boxes, 3, size, size), 'float32')
|
||||
|
||||
for i in range(num_boxes):
|
||||
img_box = np.zeros((h[i], w[i], 3), 'uint8')
|
||||
|
||||
img_array = np.asarray(img, 'uint8')
|
||||
img_box[dy[i]:(edy[i] + 1), dx[i]:(edx[i] + 1), :] =\
|
||||
img_array[y[i]:(ey[i] + 1), x[i]:(ex[i] + 1), :]
|
||||
|
||||
img_box = cv2.resize(img_box, (size, size))
|
||||
img_box = np.asarray(img_box, 'float32')
|
||||
|
||||
img_boxes[i, :, :, :] = _preprocess(img_box)
|
||||
|
||||
return img_boxes
|
||||
|
||||
|
||||
def correct_bboxes(bboxes, width, height):
|
||||
"""Crop boxes that are too big and get coordinates
|
||||
with respect to cutouts.
|
||||
"""
|
||||
x1, y1, x2, y2 = [bboxes[:, i] for i in range(4)]
|
||||
w, h = x2 - x1 + 1.0, y2 - y1 + 1.0
|
||||
num_boxes = bboxes.shape[0]
|
||||
|
||||
x, y, ex, ey = x1, y1, x2, y2
|
||||
dx, dy = np.zeros((num_boxes,)), np.zeros((num_boxes,))
|
||||
edx, edy = w.copy() - 1.0, h.copy() - 1.0
|
||||
|
||||
ind = np.where(ex > width - 1.0)[0]
|
||||
edx[ind] = w[ind] + width - 2.0 - ex[ind]
|
||||
ex[ind] = width - 1.0
|
||||
|
||||
ind = np.where(ey > height - 1.0)[0]
|
||||
edy[ind] = h[ind] + height - 2.0 - ey[ind]
|
||||
ey[ind] = height - 1.0
|
||||
|
||||
ind = np.where(x < 0.0)[0]
|
||||
dx[ind] = 0.0 - x[ind]
|
||||
x[ind] = 0.0
|
||||
|
||||
ind = np.where(y < 0.0)[0]
|
||||
dy[ind] = 0.0 - y[ind]
|
||||
y[ind] = 0.0
|
||||
return_list = [dy, edy, dx, edx, y, ey, x, ex, w, h]
|
||||
return_list = [i.astype('int32') for i in return_list]
|
||||
|
||||
return return_list
|
||||
|
||||
|
||||
def _preprocess(img):
|
||||
"""Preprocessing step before feeding the network.
|
||||
"""
|
||||
img = img.transpose((2, 0, 1))
|
||||
img = np.expand_dims(img, 0)
|
||||
img = (img - 127.5)*0.0078125
|
||||
return img
|
||||
@@ -0,0 +1,97 @@
|
||||
# This file contains modules common to various models
|
||||
|
||||
|
||||
from utils.utils import *
|
||||
|
||||
|
||||
def DWConv(c1, c2, k=1, s=1, act=True):
|
||||
# Depthwise convolution
|
||||
return Conv(c1, c2, k, s, g=math.gcd(c1, c2), act=act)
|
||||
|
||||
|
||||
class Conv(nn.Module):
|
||||
# Standard convolution
|
||||
def __init__(self, c1, c2, k=1, s=1, g=1, act=True): # ch_in, ch_out, kernel, stride, groups
|
||||
super(Conv, self).__init__()
|
||||
p = k // 2 if isinstance(k, int) else [x // 2 for x in k] # padding
|
||||
self.conv = nn.Conv2d(c1, c2, k, s, p, groups=g, bias=False)
|
||||
self.bn = nn.BatchNorm2d(c2)
|
||||
self.act = nn.LeakyReLU(0.1, inplace=True) if act else nn.Identity()
|
||||
|
||||
def forward(self, x):
|
||||
return self.act(self.bn(self.conv(x)))
|
||||
|
||||
def fuseforward(self, x):
|
||||
return self.act(self.conv(x))
|
||||
|
||||
|
||||
class Bottleneck(nn.Module):
|
||||
# Standard bottleneck
|
||||
def __init__(self, c1, c2, shortcut=True, g=1, e=0.5): # ch_in, ch_out, shortcut, groups, expansion
|
||||
super(Bottleneck, self).__init__()
|
||||
c_ = int(c2 * e) # hidden channels
|
||||
self.cv1 = Conv(c1, c_, 1, 1)
|
||||
self.cv2 = Conv(c_, c2, 3, 1, g=g)
|
||||
self.add = shortcut and c1 == c2
|
||||
|
||||
def forward(self, x):
|
||||
return x + self.cv2(self.cv1(x)) if self.add else self.cv2(self.cv1(x))
|
||||
|
||||
|
||||
class BottleneckCSP(nn.Module):
|
||||
# CSP Bottleneck https://github.com/WongKinYiu/CrossStagePartialNetworks
|
||||
def __init__(self, c1, c2, n=1, shortcut=True, g=1, e=0.5): # ch_in, ch_out, number, shortcut, groups, expansion
|
||||
super(BottleneckCSP, self).__init__()
|
||||
c_ = int(c2 * e) # hidden channels
|
||||
self.cv1 = Conv(c1, c_, 1, 1)
|
||||
self.cv2 = nn.Conv2d(c1, c_, 1, 1, bias=False)
|
||||
self.cv3 = nn.Conv2d(c_, c_, 1, 1, bias=False)
|
||||
self.cv4 = Conv(c2, c2, 1, 1)
|
||||
self.bn = nn.BatchNorm2d(2 * c_) # applied to cat(cv2, cv3)
|
||||
self.act = nn.LeakyReLU(0.1, inplace=True)
|
||||
self.m = nn.Sequential(*[Bottleneck(c_, c_, shortcut, g, e=1.0) for _ in range(n)])
|
||||
|
||||
def forward(self, x):
|
||||
y1 = self.cv3(self.m(self.cv1(x)))
|
||||
y2 = self.cv2(x)
|
||||
return self.cv4(self.act(self.bn(torch.cat((y1, y2), dim=1))))
|
||||
|
||||
|
||||
class SPP(nn.Module):
|
||||
# Spatial pyramid pooling layer used in YOLOv3-SPP
|
||||
def __init__(self, c1, c2, k=(5, 9, 13)):
|
||||
super(SPP, self).__init__()
|
||||
c_ = c1 // 2 # hidden channels
|
||||
self.cv1 = Conv(c1, c_, 1, 1)
|
||||
self.cv2 = Conv(c_ * (len(k) + 1), c2, 1, 1)
|
||||
self.m = nn.ModuleList([nn.MaxPool2d(kernel_size=x, stride=1, padding=x // 2) for x in k])
|
||||
|
||||
def forward(self, x):
|
||||
x = self.cv1(x)
|
||||
return self.cv2(torch.cat([x] + [m(x) for m in self.m], 1))
|
||||
|
||||
|
||||
class Flatten(nn.Module):
|
||||
# Use after nn.AdaptiveAvgPool2d(1) to remove last 2 dimensions
|
||||
def forward(self, x):
|
||||
return x.view(x.size(0), -1)
|
||||
|
||||
|
||||
class Focus(nn.Module):
|
||||
# Focus wh information into c-space
|
||||
def __init__(self, c1, c2, k=1):
|
||||
super(Focus, self).__init__()
|
||||
self.conv = Conv(c1 * 4, c2, k, 1)
|
||||
|
||||
def forward(self, x): # x(b,c,w,h) -> y(b,4c,w/2,h/2)
|
||||
return self.conv(torch.cat([x[..., ::2, ::2], x[..., 1::2, ::2], x[..., ::2, 1::2], x[..., 1::2, 1::2]], 1))
|
||||
|
||||
|
||||
class Concat(nn.Module):
|
||||
# Concatenate a list of tensors along dimension
|
||||
def __init__(self, dimension=1):
|
||||
super(Concat, self).__init__()
|
||||
self.d = dimension
|
||||
|
||||
def forward(self, x):
|
||||
return torch.cat(x, self.d)
|
||||
@@ -0,0 +1,42 @@
|
||||
# config.py
|
||||
|
||||
cfg_mnet = {
|
||||
'name': 'mobilenet0.25',
|
||||
'min_sizes': [[16, 32], [64, 128], [256, 512]],
|
||||
'steps': [8, 16, 32],
|
||||
'variance': [0.1, 0.2],
|
||||
'clip': False,
|
||||
'loc_weight': 2.0,
|
||||
'gpu_train': True,
|
||||
'batch_size': 32,
|
||||
'ngpu': 1,
|
||||
'epoch': 250,
|
||||
'decay1': 190,
|
||||
'decay2': 220,
|
||||
'image_size': 640,
|
||||
'pretrain': True,
|
||||
'return_layers': {'stage1': 1, 'stage2': 2, 'stage3': 3},
|
||||
'in_channel': 32,
|
||||
'out_channel': 64
|
||||
}
|
||||
|
||||
cfg_re50 = {
|
||||
'name': 'Resnet50',
|
||||
'min_sizes': [[16, 32], [64, 128], [256, 512]],
|
||||
'steps': [8, 16, 32],
|
||||
'variance': [0.1, 0.2],
|
||||
'clip': False,
|
||||
'loc_weight': 2.0,
|
||||
'gpu_train': True,
|
||||
'batch_size': 24,
|
||||
'ngpu': 4,
|
||||
'epoch': 100,
|
||||
'decay1': 70,
|
||||
'decay2': 90,
|
||||
'image_size': 840,
|
||||
'pretrain': True,
|
||||
'return_layers': {'layer2': 1, 'layer3': 2, 'layer4': 3},
|
||||
'in_channel': 256,
|
||||
'out_channel': 256
|
||||
}
|
||||
|
||||
@@ -0,0 +1,353 @@
|
||||
import math
|
||||
import numpy as np
|
||||
from .model import PNet, RNet, ONet
|
||||
from .box_utils import nms, calibrate_box, get_image_boxes, convert_to_square, _preprocess
|
||||
import torch
|
||||
import cv2
|
||||
from .nms.py_cpu_nms import py_cpu_nms
|
||||
from utils import box_utils_Retina
|
||||
from .layers.functions.prior_box import PriorBox
|
||||
from .config import cfg_re50
|
||||
from .retinaface import RetinaFace
|
||||
|
||||
|
||||
def detect_faces(image, min_face_size=20.0, thresholds=[0.6, 0.7, 0.8],
|
||||
nms_thresholds=[0.7, 0.7, 0.7], gpu_id=0):
|
||||
device = torch.device('cuda:{}'.format(gpu_id) if gpu_id is not None else 'cpu')
|
||||
pnet, rnet, onet = PNet(), RNet(), ONet()
|
||||
pnet.to(device)
|
||||
rnet.to(device)
|
||||
onet.to(device)
|
||||
onet.eval()
|
||||
|
||||
image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)
|
||||
(height, width, _) = image.shape
|
||||
min_length = min(height, width)
|
||||
min_detection_size = 12
|
||||
factor = 0.707 # sqrt(0.5)
|
||||
|
||||
scales = []
|
||||
m = min_detection_size / min_face_size
|
||||
min_length *= m
|
||||
|
||||
factor_count = 0
|
||||
while min_length > min_detection_size:
|
||||
scales.append(m * factor ** factor_count)
|
||||
min_length *= factor
|
||||
factor_count += 1
|
||||
|
||||
# STAGE 1
|
||||
bounding_boxes = []
|
||||
for s in scales: # run P-Net on different scales
|
||||
boxes = run_first_stage(image, pnet, scale=s, threshold=thresholds[0], gpu_id=gpu_id)
|
||||
bounding_boxes.append(boxes)
|
||||
bounding_boxes = [i for i in bounding_boxes if i is not None]
|
||||
bounding_boxes = np.vstack(bounding_boxes)
|
||||
|
||||
keep = nms(bounding_boxes[:, 0:5], nms_thresholds[0])
|
||||
bounding_boxes = bounding_boxes[keep]
|
||||
bounding_boxes = calibrate_box(bounding_boxes[:, 0:5], bounding_boxes[:, 5:])
|
||||
bounding_boxes = convert_to_square(bounding_boxes)
|
||||
bounding_boxes[:, 0:4] = np.round(bounding_boxes[:, 0:4])
|
||||
|
||||
# STAGE 2
|
||||
img_boxes = get_image_boxes(bounding_boxes, image, size=24)
|
||||
img_boxes = torch.from_numpy(img_boxes)
|
||||
img_boxes = img_boxes.to(device)
|
||||
output = rnet(img_boxes)
|
||||
offsets = output[0].to('cpu').data.numpy() # shape [n_boxes, 4]
|
||||
probs = output[1].to('cpu').data.numpy() # shape [n_boxes, 2]
|
||||
|
||||
keep = np.where(probs[:, 1] > thresholds[1])[0]
|
||||
bounding_boxes = bounding_boxes[keep]
|
||||
bounding_boxes[:, 4] = probs[keep, 1].reshape((-1,))
|
||||
offsets = offsets[keep]
|
||||
|
||||
keep = nms(bounding_boxes, nms_thresholds[1])
|
||||
bounding_boxes = bounding_boxes[keep]
|
||||
bounding_boxes = calibrate_box(bounding_boxes, offsets[keep])
|
||||
bounding_boxes = convert_to_square(bounding_boxes)
|
||||
bounding_boxes[:, 0:4] = np.round(bounding_boxes[:, 0:4])
|
||||
|
||||
# STAGE 3
|
||||
img_boxes = get_image_boxes(bounding_boxes, image, size=48)
|
||||
if len(img_boxes) == 0:
|
||||
return [], []
|
||||
img_boxes = torch.from_numpy(img_boxes)
|
||||
img_boxes = img_boxes.to(device)
|
||||
output = onet(img_boxes)
|
||||
landmarks = output[0].to('cpu').data.numpy() # shape [n_boxes, 10]
|
||||
offsets = output[1].to('cpu').data.numpy() # shape [n_boxes, 4]
|
||||
probs = output[2].to('cpu').data.numpy() # shape [n_boxes, 2]
|
||||
|
||||
keep = np.where(probs[:, 1] > thresholds[2])[0]
|
||||
bounding_boxes = bounding_boxes[keep]
|
||||
bounding_boxes[:, 4] = probs[keep, 1].reshape((-1,))
|
||||
offsets = offsets[keep]
|
||||
landmarks = landmarks[keep]
|
||||
|
||||
# compute landmark points
|
||||
width = bounding_boxes[:, 2] - bounding_boxes[:, 0] + 1.0
|
||||
height = bounding_boxes[:, 3] - bounding_boxes[:, 1] + 1.0
|
||||
xmin, ymin = bounding_boxes[:, 0], bounding_boxes[:, 1]
|
||||
landmarks[:, 0:5] = np.expand_dims(xmin, 1) + np.expand_dims(width, 1) * landmarks[:, 0:5]
|
||||
landmarks[:, 5:10] = np.expand_dims(ymin, 1) + np.expand_dims(height, 1) * landmarks[:, 5:10]
|
||||
|
||||
bounding_boxes = calibrate_box(bounding_boxes, offsets)
|
||||
keep = nms(bounding_boxes, nms_thresholds[2], mode='min')
|
||||
bounding_boxes = bounding_boxes[keep]
|
||||
landmarks = landmarks[keep]
|
||||
|
||||
return bounding_boxes, landmarks
|
||||
|
||||
|
||||
class RetinaFaceDetector(object):
|
||||
def __init__(self, gpu_id=None):
|
||||
self.gpu_id = gpu_id
|
||||
self.cfg = cfg_re50
|
||||
self.device = torch.device('cuda:{}'.format(gpu_id) if gpu_id is not None else 'cpu')
|
||||
model = RetinaFace(cfg=self.cfg, phase='test')
|
||||
model = self.load_model(model, './weights/Resnet50_Final.pth', True)
|
||||
model.eval()
|
||||
self.net = model.to(self.device)
|
||||
self.resize = 1
|
||||
self.confidence_threshold = 0.02
|
||||
self.top_k = 5000
|
||||
self.nms_threshold = 0.4
|
||||
self.keep_top_k = 750
|
||||
|
||||
def remove_prefix(self, state_dict, prefix):
|
||||
# print('remove prefix \'{}\''.format(prefix))
|
||||
f = lambda x: x.split(prefix, 1)[-1] if x.startswith(prefix) else x
|
||||
return {f(key): value for key, value in state_dict.items()}
|
||||
|
||||
def check_keys(self, model, pretrained_state_dict):
|
||||
ckpt_keys = set(pretrained_state_dict.keys())
|
||||
model_keys = set(model.state_dict().keys())
|
||||
used_pretrained_keys = model_keys & ckpt_keys
|
||||
# unused_pretrained_keys = ckpt_keys - model_keys
|
||||
# missing_keys = model_keys - ckpt_keys
|
||||
assert len(used_pretrained_keys) > 0, 'load NONE from pretrained checkpoint'
|
||||
return True
|
||||
|
||||
def load_model(self, model, pretrained_path, load_to_cpu):
|
||||
# print('Loading pretrained model from {}'.format(pretrained_path))
|
||||
if load_to_cpu:
|
||||
pretrained_dict = torch.load(pretrained_path, map_location=lambda storage, loc: storage)
|
||||
else:
|
||||
device = torch.cuda.current_device()
|
||||
pretrained_dict = torch.load(pretrained_path, map_location=lambda storage, loc: storage.cuda(device))
|
||||
if "state_dict" in pretrained_dict.keys():
|
||||
pretrained_dict = self.remove_prefix(pretrained_dict['state_dict'], 'module.')
|
||||
else:
|
||||
pretrained_dict = self.remove_prefix(pretrained_dict, 'module.')
|
||||
self.check_keys(model, pretrained_dict)
|
||||
model.load_state_dict(pretrained_dict, strict=False)
|
||||
return model
|
||||
|
||||
def forward(self, img_raw, min_face_size=50):
|
||||
img_scale = 640 / max(img_raw.shape[0], img_raw.shape[1])
|
||||
img = cv2.resize(img_raw, (0, 0), fx=img_scale, fy=img_scale)
|
||||
img = np.float32(img)
|
||||
|
||||
im_height, im_width, _ = img.shape
|
||||
scale = torch.Tensor([img.shape[1], img.shape[0], img.shape[1], img.shape[0]])
|
||||
img -= (104, 117, 123)
|
||||
img = img.transpose(2, 0, 1)
|
||||
img = torch.from_numpy(img).unsqueeze(0)
|
||||
img = img.to(self.device)
|
||||
scale = scale.to(self.device)
|
||||
|
||||
loc, conf, landms = self.net(img) # forward pass
|
||||
|
||||
priorbox = PriorBox(self.cfg, image_size=(im_height, im_width))
|
||||
priors = priorbox.forward()
|
||||
priors = priors.to(self.device)
|
||||
prior_data = priors.data
|
||||
boxes = box_utils_Retina.decode(loc.data.squeeze(0), prior_data, self.cfg['variance'])
|
||||
|
||||
boxes = boxes * scale / self.resize
|
||||
boxes = boxes.cpu().numpy()
|
||||
scores = conf.squeeze(0).data.cpu().numpy()[:, 1]
|
||||
landms = box_utils_Retina.decode_landm(landms.data.squeeze(0), prior_data, self.cfg['variance'])
|
||||
scale1 = torch.Tensor([img.shape[3], img.shape[2], img.shape[3], img.shape[2],
|
||||
img.shape[3], img.shape[2], img.shape[3], img.shape[2],
|
||||
img.shape[3], img.shape[2]])
|
||||
scale1 = scale1.to(self.device)
|
||||
landms = landms * scale1 / self.resize
|
||||
landms = landms.cpu().numpy()
|
||||
|
||||
# ignore low scores
|
||||
inds = np.where(scores > self.confidence_threshold)[0]
|
||||
boxes = boxes[inds]
|
||||
landms = landms[inds]
|
||||
scores = scores[inds]
|
||||
|
||||
# keep top-K before NMS
|
||||
order = scores.argsort()[::-1][:self.top_k]
|
||||
boxes = boxes[order]
|
||||
landms = landms[order]
|
||||
scores = scores[order]
|
||||
|
||||
# do NMS
|
||||
dets = np.hstack((boxes, scores[:, np.newaxis])).astype(np.float32, copy=False)
|
||||
keep = py_cpu_nms(dets, self.nms_threshold, min_face_size = min_face_size * img_scale)
|
||||
# keep = nms(dets, args.nms_threshold,force_cpu=args.cpu)
|
||||
dets = dets[keep, :]
|
||||
landms = landms[keep]
|
||||
|
||||
# keep top-K faster NMS
|
||||
dets = dets[:self.keep_top_k, :]
|
||||
landms = landms[:self.keep_top_k, :]
|
||||
|
||||
dets[:, :4] = dets[:, :4] / img_scale
|
||||
landms /= img_scale
|
||||
return dets, landms
|
||||
|
||||
|
||||
class MTCNNFaceDetector(object):
|
||||
def __init__(self, gpu_id=None):
|
||||
self.gpu_id = gpu_id
|
||||
self.device = torch.device('cuda:{}'.format(gpu_id) if gpu_id is not None else 'cpu')
|
||||
self.pnet, self.rnet, self.onet = PNet(), RNet(), ONet()
|
||||
self.pnet.to(self.device)
|
||||
self.rnet.to(self.device)
|
||||
self.onet.to(self.device)
|
||||
self.onet.eval()
|
||||
|
||||
def forward(self, image, min_face_size=20.0, thresholds=[0.6, 0.7, 0.8], nms_thresholds=[0.7, 0.7, 0.7]):
|
||||
image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)
|
||||
(height, width, _) = image.shape
|
||||
min_length = min(height, width)
|
||||
min_detection_size = 12
|
||||
factor = 0.707 # sqrt(0.5)
|
||||
|
||||
scales = []
|
||||
m = min_detection_size / min_face_size
|
||||
min_length *= m
|
||||
|
||||
factor_count = 0
|
||||
while min_length > min_detection_size:
|
||||
scales.append(m * factor ** factor_count)
|
||||
min_length *= factor
|
||||
factor_count += 1
|
||||
|
||||
# STAGE 1
|
||||
bounding_boxes = []
|
||||
for s in scales: # run P-Net on different scales
|
||||
boxes = run_first_stage(image, self.pnet, scale=s, threshold=thresholds[0], gpu_id=self.gpu_id)
|
||||
bounding_boxes.append(boxes)
|
||||
bounding_boxes = [i for i in bounding_boxes if i is not None]
|
||||
if len(bounding_boxes) == 0:
|
||||
return [], []
|
||||
bounding_boxes = np.vstack(bounding_boxes)
|
||||
|
||||
keep = nms(bounding_boxes[:, 0:5], nms_thresholds[0])
|
||||
bounding_boxes = bounding_boxes[keep]
|
||||
bounding_boxes = calibrate_box(bounding_boxes[:, 0:5], bounding_boxes[:, 5:])
|
||||
bounding_boxes = convert_to_square(bounding_boxes)
|
||||
bounding_boxes[:, 0:4] = np.round(bounding_boxes[:, 0:4])
|
||||
|
||||
# STAGE 2
|
||||
img_boxes = get_image_boxes(bounding_boxes, image, size=24)
|
||||
img_boxes = torch.from_numpy(img_boxes)
|
||||
img_boxes = img_boxes.to(self.device)
|
||||
output = self.rnet(img_boxes)
|
||||
offsets = output[0].to('cpu').data.numpy() # shape [n_boxes, 4]
|
||||
probs = output[1].to('cpu').data.numpy() # shape [n_boxes, 2]
|
||||
|
||||
keep = np.where(probs[:, 1] > thresholds[1])[0]
|
||||
bounding_boxes = bounding_boxes[keep]
|
||||
bounding_boxes[:, 4] = probs[keep, 1].reshape((-1,))
|
||||
offsets = offsets[keep]
|
||||
|
||||
keep = nms(bounding_boxes, nms_thresholds[1])
|
||||
bounding_boxes = bounding_boxes[keep]
|
||||
bounding_boxes = calibrate_box(bounding_boxes, offsets[keep])
|
||||
bounding_boxes = convert_to_square(bounding_boxes)
|
||||
bounding_boxes[:, 0:4] = np.round(bounding_boxes[:, 0:4])
|
||||
|
||||
# STAGE 3
|
||||
img_boxes = get_image_boxes(bounding_boxes, image, size=48)
|
||||
if len(img_boxes) == 0:
|
||||
return [], []
|
||||
img_boxes = torch.from_numpy(img_boxes)
|
||||
img_boxes = img_boxes.to(self.device)
|
||||
output = self.onet(img_boxes)
|
||||
landmarks = output[0].to('cpu').data.numpy() # shape [n_boxes, 10]
|
||||
offsets = output[1].to('cpu').data.numpy() # shape [n_boxes, 4]
|
||||
probs = output[2].to('cpu').data.numpy() # shape [n_boxes, 2]
|
||||
|
||||
keep = np.where(probs[:, 1] > thresholds[2])[0]
|
||||
bounding_boxes = bounding_boxes[keep]
|
||||
bounding_boxes[:, 4] = probs[keep, 1].reshape((-1,))
|
||||
offsets = offsets[keep]
|
||||
landmarks = landmarks[keep]
|
||||
|
||||
# compute landmark points
|
||||
width = bounding_boxes[:, 2] - bounding_boxes[:, 0] + 1.0
|
||||
height = bounding_boxes[:, 3] - bounding_boxes[:, 1] + 1.0
|
||||
xmin, ymin = bounding_boxes[:, 0], bounding_boxes[:, 1]
|
||||
landmarks[:, 0:5] = np.expand_dims(xmin, 1) + np.expand_dims(width, 1) * landmarks[:, 0:5]
|
||||
landmarks[:, 5:10] = np.expand_dims(ymin, 1) + np.expand_dims(height, 1) * landmarks[:, 5:10]
|
||||
|
||||
bounding_boxes = calibrate_box(bounding_boxes, offsets)
|
||||
keep = nms(bounding_boxes, nms_thresholds[2], mode='min')
|
||||
bounding_boxes = bounding_boxes[keep]
|
||||
landmarks = landmarks[keep]
|
||||
|
||||
return bounding_boxes, landmarks
|
||||
|
||||
|
||||
def run_first_stage(image, net, scale, threshold, gpu_id=0):
|
||||
"""
|
||||
Run P-Net, generate bounding boxes, and do NMS.
|
||||
"""
|
||||
device = torch.device('cuda:{}'.format(gpu_id) if gpu_id is not None else 'cpu')
|
||||
(height, width, _) = image.shape
|
||||
sw, sh = math.ceil(width * scale), math.ceil(height * scale)
|
||||
img = cv2.resize(image, (sw, sh))
|
||||
# img = image.resize((sw, sh), Image.BILINEAR)
|
||||
img = np.asarray(img, 'float32')
|
||||
img = torch.from_numpy(_preprocess(img))
|
||||
img = img.to(device)
|
||||
|
||||
output = net(img)
|
||||
probs = output[1].to('cpu').data.numpy()[0, 1, :, :]
|
||||
offsets = output[0].to('cpu').data.numpy()
|
||||
|
||||
boxes = _generate_bboxes(probs, offsets, scale, threshold)
|
||||
if len(boxes) == 0:
|
||||
return None
|
||||
|
||||
keep = nms(boxes[:, 0:5], overlap_threshold=0.5)
|
||||
return boxes[keep]
|
||||
|
||||
|
||||
def _generate_bboxes(probs, offsets, scale, threshold):
|
||||
"""
|
||||
Generate bounding boxes at places where there is probably a face.
|
||||
"""
|
||||
stride = 2
|
||||
cell_size = 12
|
||||
|
||||
inds = np.where(probs > threshold)
|
||||
|
||||
if inds[0].size == 0:
|
||||
return np.array([])
|
||||
|
||||
tx1, ty1, tx2, ty2 = [offsets[0, i, inds[0], inds[1]] for i in range(4)]
|
||||
|
||||
offsets = np.array([tx1, ty1, tx2, ty2])
|
||||
score = probs[inds[0], inds[1]]
|
||||
|
||||
# P-Net is applied to scaled images, so we need to rescale bounding boxes back
|
||||
bounding_boxes = np.vstack([
|
||||
np.round((stride * inds[1] + 1.0) / scale),
|
||||
np.round((stride * inds[0] + 1.0) / scale),
|
||||
np.round((stride * inds[1] + 1.0 + cell_size) / scale),
|
||||
np.round((stride * inds[0] + 1.0 + cell_size) / scale),
|
||||
score, offsets
|
||||
])
|
||||
|
||||
return bounding_boxes.T
|
||||
@@ -0,0 +1,85 @@
|
||||
from models.common import *
|
||||
|
||||
|
||||
class Sum(nn.Module):
|
||||
# Weighted sum of 2 or more layers https://arxiv.org/abs/1911.09070
|
||||
def __init__(self, n, weight=False): # n: number of inputs
|
||||
super(Sum, self).__init__()
|
||||
self.weight = weight # apply weights boolean
|
||||
self.iter = range(n - 1) # iter object
|
||||
if weight:
|
||||
self.w = nn.Parameter(-torch.arange(1., n) / 2, requires_grad=True) # layer weights
|
||||
|
||||
def forward(self, x):
|
||||
y = x[0] # no weight
|
||||
if self.weight:
|
||||
w = torch.sigmoid(self.w) * 2
|
||||
for i in self.iter:
|
||||
y = y + x[i + 1] * w[i]
|
||||
else:
|
||||
for i in self.iter:
|
||||
y = y + x[i + 1]
|
||||
return y
|
||||
|
||||
|
||||
class GhostConv(nn.Module):
|
||||
# Ghost Convolution https://github.com/huawei-noah/ghostnet
|
||||
def __init__(self, c1, c2, k=1, s=1, g=1, act=True): # ch_in, ch_out, kernel, stride, groups
|
||||
super(GhostConv, self).__init__()
|
||||
c_ = c2 // 2 # hidden channels
|
||||
self.cv1 = Conv(c1, c_, k, s, g, act)
|
||||
self.cv2 = Conv(c_, c_, 5, 1, c_, act)
|
||||
|
||||
def forward(self, x):
|
||||
y = self.cv1(x)
|
||||
return torch.cat([y, self.cv2(y)], 1)
|
||||
|
||||
|
||||
class GhostBottleneck(nn.Module):
|
||||
# Ghost Bottleneck https://github.com/huawei-noah/ghostnet
|
||||
def __init__(self, c1, c2, k, s):
|
||||
super(GhostBottleneck, self).__init__()
|
||||
c_ = c2 // 2
|
||||
self.conv = nn.Sequential(GhostConv(c1, c_, 1, 1), # pw
|
||||
DWConv(c_, c_, k, s, act=False) if s == 2 else nn.Identity(), # dw
|
||||
GhostConv(c_, c2, 1, 1, act=False)) # pw-linear
|
||||
self.shortcut = nn.Sequential(DWConv(c1, c1, k, s, act=False),
|
||||
Conv(c1, c2, 1, 1, act=False)) if s == 2 else nn.Identity()
|
||||
|
||||
def forward(self, x):
|
||||
return self.conv(x) + self.shortcut(x)
|
||||
|
||||
|
||||
class ConvPlus(nn.Module):
|
||||
# Plus-shaped convolution
|
||||
def __init__(self, c1, c2, k=3, s=1, g=1, bias=True): # ch_in, ch_out, kernel, stride, groups
|
||||
super(ConvPlus, self).__init__()
|
||||
self.cv1 = nn.Conv2d(c1, c2, (k, 1), s, (k // 2, 0), groups=g, bias=bias)
|
||||
self.cv2 = nn.Conv2d(c1, c2, (1, k), s, (0, k // 2), groups=g, bias=bias)
|
||||
|
||||
def forward(self, x):
|
||||
return self.cv1(x) + self.cv2(x)
|
||||
|
||||
|
||||
class MixConv2d(nn.Module):
|
||||
# Mixed Depthwise Conv https://arxiv.org/abs/1907.09595
|
||||
def __init__(self, c1, c2, k=(1, 3), s=1, equal_ch=True):
|
||||
super(MixConv2d, self).__init__()
|
||||
groups = len(k)
|
||||
if equal_ch: # equal c_ per group
|
||||
i = torch.linspace(0, groups - 1E-6, c2).floor() # c2 indices
|
||||
c_ = [(i == g).sum() for g in range(groups)] # intermediate channels
|
||||
else: # equal weight.numel() per group
|
||||
b = [c2] + [0] * groups
|
||||
a = np.eye(groups + 1, groups, k=-1)
|
||||
a -= np.roll(a, 1, axis=1)
|
||||
a *= np.array(k) ** 2
|
||||
a[0] = 1
|
||||
c_ = np.linalg.lstsq(a, b, rcond=None)[0].round() # solve for equal weight indices, ax = b
|
||||
|
||||
self.m = nn.ModuleList([nn.Conv2d(c1, int(c_[g]), k[g], s, k[g] // 2, bias=False) for g in range(groups)])
|
||||
self.bn = nn.BatchNorm2d(c2)
|
||||
self.act = nn.LeakyReLU(0.1, inplace=True)
|
||||
|
||||
def forward(self, x):
|
||||
return x + self.act(self.bn(torch.cat([m(x) for m in self.m], 1)))
|
||||
@@ -0,0 +1,2 @@
|
||||
from .functions import *
|
||||
from .modules import *
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,3 @@
|
||||
from .wider_face import WiderFaceDetection, detection_collate
|
||||
from .data_augment import *
|
||||
from .config import *
|
||||
@@ -0,0 +1,42 @@
|
||||
# config.py
|
||||
|
||||
cfg_mnet = {
|
||||
'name': 'mobilenet0.25',
|
||||
'min_sizes': [[16, 32], [64, 128], [256, 512]],
|
||||
'steps': [8, 16, 32],
|
||||
'variance': [0.1, 0.2],
|
||||
'clip': False,
|
||||
'loc_weight': 2.0,
|
||||
'gpu_train': True,
|
||||
'batch_size': 32,
|
||||
'ngpu': 1,
|
||||
'epoch': 250,
|
||||
'decay1': 190,
|
||||
'decay2': 220,
|
||||
'image_size': 640,
|
||||
'pretrain': True,
|
||||
'return_layers': {'stage1': 1, 'stage2': 2, 'stage3': 3},
|
||||
'in_channel': 32,
|
||||
'out_channel': 64
|
||||
}
|
||||
|
||||
cfg_re50 = {
|
||||
'name': 'Resnet50',
|
||||
'min_sizes': [[16, 32], [64, 128], [256, 512]],
|
||||
'steps': [8, 16, 32],
|
||||
'variance': [0.1, 0.2],
|
||||
'clip': False,
|
||||
'loc_weight': 2.0,
|
||||
'gpu_train': True,
|
||||
'batch_size': 24,
|
||||
'ngpu': 4,
|
||||
'epoch': 100,
|
||||
'decay1': 70,
|
||||
'decay2': 90,
|
||||
'image_size': 840,
|
||||
'pretrain': True,
|
||||
'return_layers': {'layer2': 1, 'layer3': 2, 'layer4': 3},
|
||||
'in_channel': 256,
|
||||
'out_channel': 256
|
||||
}
|
||||
|
||||
@@ -0,0 +1,237 @@
|
||||
import cv2
|
||||
import numpy as np
|
||||
import random
|
||||
from utils.box_utils_Retina import matrix_iof
|
||||
|
||||
|
||||
def _crop(image, boxes, labels, landm, img_dim):
|
||||
height, width, _ = image.shape
|
||||
pad_image_flag = True
|
||||
|
||||
for _ in range(250):
|
||||
"""
|
||||
if random.uniform(0, 1) <= 0.2:
|
||||
scale = 1.0
|
||||
else:
|
||||
scale = random.uniform(0.3, 1.0)
|
||||
"""
|
||||
PRE_SCALES = [0.3, 0.45, 0.6, 0.8, 1.0]
|
||||
scale = random.choice(PRE_SCALES)
|
||||
short_side = min(width, height)
|
||||
w = int(scale * short_side)
|
||||
h = w
|
||||
|
||||
if width == w:
|
||||
l = 0
|
||||
else:
|
||||
l = random.randrange(width - w)
|
||||
if height == h:
|
||||
t = 0
|
||||
else:
|
||||
t = random.randrange(height - h)
|
||||
roi = np.array((l, t, l + w, t + h))
|
||||
|
||||
value = matrix_iof(boxes, roi[np.newaxis])
|
||||
flag = (value >= 1)
|
||||
if not flag.any():
|
||||
continue
|
||||
|
||||
centers = (boxes[:, :2] + boxes[:, 2:]) / 2
|
||||
mask_a = np.logical_and(roi[:2] < centers, centers < roi[2:]).all(axis=1)
|
||||
boxes_t = boxes[mask_a].copy()
|
||||
labels_t = labels[mask_a].copy()
|
||||
landms_t = landm[mask_a].copy()
|
||||
landms_t = landms_t.reshape([-1, 5, 2])
|
||||
|
||||
if boxes_t.shape[0] == 0:
|
||||
continue
|
||||
|
||||
image_t = image[roi[1]:roi[3], roi[0]:roi[2]]
|
||||
|
||||
boxes_t[:, :2] = np.maximum(boxes_t[:, :2], roi[:2])
|
||||
boxes_t[:, :2] -= roi[:2]
|
||||
boxes_t[:, 2:] = np.minimum(boxes_t[:, 2:], roi[2:])
|
||||
boxes_t[:, 2:] -= roi[:2]
|
||||
|
||||
# landm
|
||||
landms_t[:, :, :2] = landms_t[:, :, :2] - roi[:2]
|
||||
landms_t[:, :, :2] = np.maximum(landms_t[:, :, :2], np.array([0, 0]))
|
||||
landms_t[:, :, :2] = np.minimum(landms_t[:, :, :2], roi[2:] - roi[:2])
|
||||
landms_t = landms_t.reshape([-1, 10])
|
||||
|
||||
|
||||
# make sure that the cropped image contains at least one face > 16 pixel at training image scale
|
||||
b_w_t = (boxes_t[:, 2] - boxes_t[:, 0] + 1) / w * img_dim
|
||||
b_h_t = (boxes_t[:, 3] - boxes_t[:, 1] + 1) / h * img_dim
|
||||
mask_b = np.minimum(b_w_t, b_h_t) > 0.0
|
||||
boxes_t = boxes_t[mask_b]
|
||||
labels_t = labels_t[mask_b]
|
||||
landms_t = landms_t[mask_b]
|
||||
|
||||
if boxes_t.shape[0] == 0:
|
||||
continue
|
||||
|
||||
pad_image_flag = False
|
||||
|
||||
return image_t, boxes_t, labels_t, landms_t, pad_image_flag
|
||||
return image, boxes, labels, landm, pad_image_flag
|
||||
|
||||
|
||||
def _distort(image):
|
||||
|
||||
def _convert(image, alpha=1, beta=0):
|
||||
tmp = image.astype(float) * alpha + beta
|
||||
tmp[tmp < 0] = 0
|
||||
tmp[tmp > 255] = 255
|
||||
image[:] = tmp
|
||||
|
||||
image = image.copy()
|
||||
|
||||
if random.randrange(2):
|
||||
|
||||
#brightness distortion
|
||||
if random.randrange(2):
|
||||
_convert(image, beta=random.uniform(-32, 32))
|
||||
|
||||
#contrast distortion
|
||||
if random.randrange(2):
|
||||
_convert(image, alpha=random.uniform(0.5, 1.5))
|
||||
|
||||
image = cv2.cvtColor(image, cv2.COLOR_BGR2HSV)
|
||||
|
||||
#saturation distortion
|
||||
if random.randrange(2):
|
||||
_convert(image[:, :, 1], alpha=random.uniform(0.5, 1.5))
|
||||
|
||||
#hue distortion
|
||||
if random.randrange(2):
|
||||
tmp = image[:, :, 0].astype(int) + random.randint(-18, 18)
|
||||
tmp %= 180
|
||||
image[:, :, 0] = tmp
|
||||
|
||||
image = cv2.cvtColor(image, cv2.COLOR_HSV2BGR)
|
||||
|
||||
else:
|
||||
|
||||
#brightness distortion
|
||||
if random.randrange(2):
|
||||
_convert(image, beta=random.uniform(-32, 32))
|
||||
|
||||
image = cv2.cvtColor(image, cv2.COLOR_BGR2HSV)
|
||||
|
||||
#saturation distortion
|
||||
if random.randrange(2):
|
||||
_convert(image[:, :, 1], alpha=random.uniform(0.5, 1.5))
|
||||
|
||||
#hue distortion
|
||||
if random.randrange(2):
|
||||
tmp = image[:, :, 0].astype(int) + random.randint(-18, 18)
|
||||
tmp %= 180
|
||||
image[:, :, 0] = tmp
|
||||
|
||||
image = cv2.cvtColor(image, cv2.COLOR_HSV2BGR)
|
||||
|
||||
#contrast distortion
|
||||
if random.randrange(2):
|
||||
_convert(image, alpha=random.uniform(0.5, 1.5))
|
||||
|
||||
return image
|
||||
|
||||
|
||||
def _expand(image, boxes, fill, p):
|
||||
if random.randrange(2):
|
||||
return image, boxes
|
||||
|
||||
height, width, depth = image.shape
|
||||
|
||||
scale = random.uniform(1, p)
|
||||
w = int(scale * width)
|
||||
h = int(scale * height)
|
||||
|
||||
left = random.randint(0, w - width)
|
||||
top = random.randint(0, h - height)
|
||||
|
||||
boxes_t = boxes.copy()
|
||||
boxes_t[:, :2] += (left, top)
|
||||
boxes_t[:, 2:] += (left, top)
|
||||
expand_image = np.empty(
|
||||
(h, w, depth),
|
||||
dtype=image.dtype)
|
||||
expand_image[:, :] = fill
|
||||
expand_image[top:top + height, left:left + width] = image
|
||||
image = expand_image
|
||||
|
||||
return image, boxes_t
|
||||
|
||||
|
||||
def _mirror(image, boxes, landms):
|
||||
_, width, _ = image.shape
|
||||
if random.randrange(2):
|
||||
image = image[:, ::-1]
|
||||
boxes = boxes.copy()
|
||||
boxes[:, 0::2] = width - boxes[:, 2::-2]
|
||||
|
||||
# landm
|
||||
landms = landms.copy()
|
||||
landms = landms.reshape([-1, 5, 2])
|
||||
landms[:, :, 0] = width - landms[:, :, 0]
|
||||
tmp = landms[:, 1, :].copy()
|
||||
landms[:, 1, :] = landms[:, 0, :]
|
||||
landms[:, 0, :] = tmp
|
||||
tmp1 = landms[:, 4, :].copy()
|
||||
landms[:, 4, :] = landms[:, 3, :]
|
||||
landms[:, 3, :] = tmp1
|
||||
landms = landms.reshape([-1, 10])
|
||||
|
||||
return image, boxes, landms
|
||||
|
||||
|
||||
def _pad_to_square(image, rgb_mean, pad_image_flag):
|
||||
if not pad_image_flag:
|
||||
return image
|
||||
height, width, _ = image.shape
|
||||
long_side = max(width, height)
|
||||
image_t = np.empty((long_side, long_side, 3), dtype=image.dtype)
|
||||
image_t[:, :] = rgb_mean
|
||||
image_t[0:0 + height, 0:0 + width] = image
|
||||
return image_t
|
||||
|
||||
|
||||
def _resize_subtract_mean(image, insize, rgb_mean):
|
||||
interp_methods = [cv2.INTER_LINEAR, cv2.INTER_CUBIC, cv2.INTER_AREA, cv2.INTER_NEAREST, cv2.INTER_LANCZOS4]
|
||||
interp_method = interp_methods[random.randrange(5)]
|
||||
image = cv2.resize(image, (insize, insize), interpolation=interp_method)
|
||||
image = image.astype(np.float32)
|
||||
image -= rgb_mean
|
||||
return image.transpose(2, 0, 1)
|
||||
|
||||
|
||||
class preproc(object):
|
||||
|
||||
def __init__(self, img_dim, rgb_means):
|
||||
self.img_dim = img_dim
|
||||
self.rgb_means = rgb_means
|
||||
|
||||
def __call__(self, image, targets):
|
||||
assert targets.shape[0] > 0, "this image does not have gt"
|
||||
|
||||
boxes = targets[:, :4].copy()
|
||||
labels = targets[:, -1].copy()
|
||||
landm = targets[:, 4:-1].copy()
|
||||
|
||||
image_t, boxes_t, labels_t, landm_t, pad_image_flag = _crop(image, boxes, labels, landm, self.img_dim)
|
||||
image_t = _distort(image_t)
|
||||
image_t = _pad_to_square(image_t,self.rgb_means, pad_image_flag)
|
||||
image_t, boxes_t, landm_t = _mirror(image_t, boxes_t, landm_t)
|
||||
height, width, _ = image_t.shape
|
||||
image_t = _resize_subtract_mean(image_t, self.img_dim, self.rgb_means)
|
||||
boxes_t[:, 0::2] /= width
|
||||
boxes_t[:, 1::2] /= height
|
||||
|
||||
landm_t[:, 0::2] /= width
|
||||
landm_t[:, 1::2] /= height
|
||||
|
||||
labels_t = np.expand_dims(labels_t, 1)
|
||||
targets_t = np.hstack((boxes_t, landm_t, labels_t))
|
||||
|
||||
return image_t, targets_t
|
||||
@@ -0,0 +1,101 @@
|
||||
import os
|
||||
import os.path
|
||||
import sys
|
||||
import torch
|
||||
import torch.utils.data as data
|
||||
import cv2
|
||||
import numpy as np
|
||||
|
||||
class WiderFaceDetection(data.Dataset):
|
||||
def __init__(self, txt_path, preproc=None):
|
||||
self.preproc = preproc
|
||||
self.imgs_path = []
|
||||
self.words = []
|
||||
f = open(txt_path,'r')
|
||||
lines = f.readlines()
|
||||
isFirst = True
|
||||
labels = []
|
||||
for line in lines:
|
||||
line = line.rstrip()
|
||||
if line.startswith('#'):
|
||||
if isFirst is True:
|
||||
isFirst = False
|
||||
else:
|
||||
labels_copy = labels.copy()
|
||||
self.words.append(labels_copy)
|
||||
labels.clear()
|
||||
path = line[2:]
|
||||
path = txt_path.replace('label.txt','images/') + path
|
||||
self.imgs_path.append(path)
|
||||
else:
|
||||
line = line.split(' ')
|
||||
label = [float(x) for x in line]
|
||||
labels.append(label)
|
||||
|
||||
self.words.append(labels)
|
||||
|
||||
def __len__(self):
|
||||
return len(self.imgs_path)
|
||||
|
||||
def __getitem__(self, index):
|
||||
img = cv2.imread(self.imgs_path[index])
|
||||
height, width, _ = img.shape
|
||||
|
||||
labels = self.words[index]
|
||||
annotations = np.zeros((0, 15))
|
||||
if len(labels) == 0:
|
||||
return annotations
|
||||
for idx, label in enumerate(labels):
|
||||
annotation = np.zeros((1, 15))
|
||||
# bbox
|
||||
annotation[0, 0] = label[0] # x1
|
||||
annotation[0, 1] = label[1] # y1
|
||||
annotation[0, 2] = label[0] + label[2] # x2
|
||||
annotation[0, 3] = label[1] + label[3] # y2
|
||||
|
||||
# landmarks
|
||||
annotation[0, 4] = label[4] # l0_x
|
||||
annotation[0, 5] = label[5] # l0_y
|
||||
annotation[0, 6] = label[7] # l1_x
|
||||
annotation[0, 7] = label[8] # l1_y
|
||||
annotation[0, 8] = label[10] # l2_x
|
||||
annotation[0, 9] = label[11] # l2_y
|
||||
annotation[0, 10] = label[13] # l3_x
|
||||
annotation[0, 11] = label[14] # l3_y
|
||||
annotation[0, 12] = label[16] # l4_x
|
||||
annotation[0, 13] = label[17] # l4_y
|
||||
if (annotation[0, 4]<0):
|
||||
annotation[0, 14] = -1
|
||||
else:
|
||||
annotation[0, 14] = 1
|
||||
|
||||
annotations = np.append(annotations, annotation, axis=0)
|
||||
target = np.array(annotations)
|
||||
if self.preproc is not None:
|
||||
img, target = self.preproc(img, target)
|
||||
|
||||
return torch.from_numpy(img), target
|
||||
|
||||
def detection_collate(batch):
|
||||
"""Custom collate fn for dealing with batches of images that have a different
|
||||
number of associated object annotations (bounding boxes).
|
||||
|
||||
Arguments:
|
||||
batch: (tuple) A tuple of tensor images and lists of annotations
|
||||
|
||||
Return:
|
||||
A tuple containing:
|
||||
1) (tensor) batch of images stacked on their 0 dim
|
||||
2) (list of tensors) annotations for a given image are stacked on 0 dim
|
||||
"""
|
||||
targets = []
|
||||
imgs = []
|
||||
for _, sample in enumerate(batch):
|
||||
for _, tup in enumerate(sample):
|
||||
if torch.is_tensor(tup):
|
||||
imgs.append(tup)
|
||||
elif isinstance(tup, type(np.empty(0))):
|
||||
annos = torch.from_numpy(tup).float()
|
||||
targets.append(annos)
|
||||
|
||||
return (torch.stack(imgs, 0), targets)
|
||||
@@ -0,0 +1,34 @@
|
||||
import torch
|
||||
from itertools import product as product
|
||||
import numpy as np
|
||||
from math import ceil
|
||||
|
||||
|
||||
class PriorBox(object):
|
||||
def __init__(self, cfg, image_size=None, phase='train'):
|
||||
super(PriorBox, self).__init__()
|
||||
self.min_sizes = cfg['min_sizes']
|
||||
self.steps = cfg['steps']
|
||||
self.clip = cfg['clip']
|
||||
self.image_size = image_size
|
||||
self.feature_maps = [[ceil(self.image_size[0]/step), ceil(self.image_size[1]/step)] for step in self.steps]
|
||||
self.name = "s"
|
||||
|
||||
def forward(self):
|
||||
anchors = []
|
||||
for k, f in enumerate(self.feature_maps):
|
||||
min_sizes = self.min_sizes[k]
|
||||
for i, j in product(range(f[0]), range(f[1])):
|
||||
for min_size in min_sizes:
|
||||
s_kx = min_size / self.image_size[1]
|
||||
s_ky = min_size / self.image_size[0]
|
||||
dense_cx = [x * self.steps[k] / self.image_size[1] for x in [j + 0.5]]
|
||||
dense_cy = [y * self.steps[k] / self.image_size[0] for y in [i + 0.5]]
|
||||
for cy, cx in product(dense_cy, dense_cx):
|
||||
anchors += [cx, cy, s_kx, s_ky]
|
||||
|
||||
# back to torch land
|
||||
output = torch.Tensor(anchors).view(-1, 4)
|
||||
if self.clip:
|
||||
output.clamp_(max=1, min=0)
|
||||
return output
|
||||
@@ -0,0 +1,3 @@
|
||||
from .multibox_loss import MultiBoxLoss
|
||||
|
||||
__all__ = ['MultiBoxLoss']
|
||||
@@ -0,0 +1,125 @@
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torch.nn.functional as F
|
||||
from torch.autograd import Variable
|
||||
from utils.box_utils_Retina import match, log_sum_exp
|
||||
from models.layers.data import cfg_mnet
|
||||
GPU = cfg_mnet['gpu_train']
|
||||
|
||||
class MultiBoxLoss(nn.Module):
|
||||
"""SSD Weighted Loss Function
|
||||
Compute Targets:
|
||||
1) Produce Confidence Target Indices by matching ground truth boxes
|
||||
with (default) 'priorboxes' that have jaccard index > threshold parameter
|
||||
(default threshold: 0.5).
|
||||
2) Produce localization target by 'encoding' variance into offsets of ground
|
||||
truth boxes and their matched 'priorboxes'.
|
||||
3) Hard negative mining to filter the excessive number of negative examples
|
||||
that comes with using a large number of default bounding boxes.
|
||||
(default negative:positive ratio 3:1)
|
||||
Objective Loss:
|
||||
L(x,c,l,g) = (Lconf(x, c) + αLloc(x,l,g)) / N
|
||||
Where, Lconf is the CrossEntropy Loss and Lloc is the SmoothL1 Loss
|
||||
weighted by α which is set to 1 by cross val.
|
||||
Args:
|
||||
c: class confidences,
|
||||
l: predicted boxes,
|
||||
g: ground truth boxes
|
||||
N: number of matched default boxes
|
||||
See: https://arxiv.org/pdf/1512.02325.pdf for more details.
|
||||
"""
|
||||
|
||||
def __init__(self, num_classes, overlap_thresh, prior_for_matching, bkg_label, neg_mining, neg_pos, neg_overlap, encode_target):
|
||||
super(MultiBoxLoss, self).__init__()
|
||||
self.num_classes = num_classes
|
||||
self.threshold = overlap_thresh
|
||||
self.background_label = bkg_label
|
||||
self.encode_target = encode_target
|
||||
self.use_prior_for_matching = prior_for_matching
|
||||
self.do_neg_mining = neg_mining
|
||||
self.negpos_ratio = neg_pos
|
||||
self.neg_overlap = neg_overlap
|
||||
self.variance = [0.1, 0.2]
|
||||
|
||||
def forward(self, predictions, priors, targets):
|
||||
"""Multibox Loss
|
||||
Args:
|
||||
predictions (tuple): A tuple containing loc preds, conf preds,
|
||||
and prior boxes from SSD net.
|
||||
conf shape: torch.size(batch_size,num_priors,num_classes)
|
||||
loc shape: torch.size(batch_size,num_priors,4)
|
||||
priors shape: torch.size(num_priors,4)
|
||||
|
||||
ground_truth (tensor): Ground truth boxes and labels for a batch,
|
||||
shape: [batch_size,num_objs,5] (last idx is the label).
|
||||
"""
|
||||
|
||||
loc_data, conf_data, landm_data = predictions
|
||||
priors = priors
|
||||
num = loc_data.size(0)
|
||||
num_priors = (priors.size(0))
|
||||
|
||||
# match priors (default boxes) and ground truth boxes
|
||||
loc_t = torch.Tensor(num, num_priors, 4)
|
||||
landm_t = torch.Tensor(num, num_priors, 10)
|
||||
conf_t = torch.LongTensor(num, num_priors)
|
||||
for idx in range(num):
|
||||
truths = targets[idx][:, :4].data
|
||||
labels = targets[idx][:, -1].data
|
||||
landms = targets[idx][:, 4:14].data
|
||||
defaults = priors.data
|
||||
match(self.threshold, truths, defaults, self.variance, labels, landms, loc_t, conf_t, landm_t, idx)
|
||||
if GPU:
|
||||
loc_t = loc_t.cuda()
|
||||
conf_t = conf_t.cuda()
|
||||
landm_t = landm_t.cuda()
|
||||
|
||||
zeros = torch.tensor(0).cuda()
|
||||
# landm Loss (Smooth L1)
|
||||
# Shape: [batch,num_priors,10]
|
||||
pos1 = conf_t > zeros
|
||||
num_pos_landm = pos1.long().sum(1, keepdim=True)
|
||||
N1 = max(num_pos_landm.data.sum().float(), 1)
|
||||
pos_idx1 = pos1.unsqueeze(pos1.dim()).expand_as(landm_data)
|
||||
landm_p = landm_data[pos_idx1].view(-1, 10)
|
||||
landm_t = landm_t[pos_idx1].view(-1, 10)
|
||||
loss_landm = F.smooth_l1_loss(landm_p, landm_t, reduction='sum')
|
||||
|
||||
|
||||
pos = conf_t != zeros
|
||||
conf_t[pos] = 1
|
||||
|
||||
# Localization Loss (Smooth L1)
|
||||
# Shape: [batch,num_priors,4]
|
||||
pos_idx = pos.unsqueeze(pos.dim()).expand_as(loc_data)
|
||||
loc_p = loc_data[pos_idx].view(-1, 4)
|
||||
loc_t = loc_t[pos_idx].view(-1, 4)
|
||||
loss_l = F.smooth_l1_loss(loc_p, loc_t, reduction='sum')
|
||||
|
||||
# Compute max conf across batch for hard negative mining
|
||||
batch_conf = conf_data.view(-1, self.num_classes)
|
||||
loss_c = log_sum_exp(batch_conf) - batch_conf.gather(1, conf_t.view(-1, 1))
|
||||
|
||||
# Hard Negative Mining
|
||||
loss_c[pos.view(-1, 1)] = 0 # filter out pos boxes for now
|
||||
loss_c = loss_c.view(num, -1)
|
||||
_, loss_idx = loss_c.sort(1, descending=True)
|
||||
_, idx_rank = loss_idx.sort(1)
|
||||
num_pos = pos.long().sum(1, keepdim=True)
|
||||
num_neg = torch.clamp(self.negpos_ratio*num_pos, max=pos.size(1)-1)
|
||||
neg = idx_rank < num_neg.expand_as(idx_rank)
|
||||
|
||||
# Confidence Loss Including Positive and Negative Examples
|
||||
pos_idx = pos.unsqueeze(2).expand_as(conf_data)
|
||||
neg_idx = neg.unsqueeze(2).expand_as(conf_data)
|
||||
conf_p = conf_data[(pos_idx+neg_idx).gt(0)].view(-1,self.num_classes)
|
||||
targets_weighted = conf_t[(pos+neg).gt(0)]
|
||||
loss_c = F.cross_entropy(conf_p, targets_weighted, reduction='sum')
|
||||
|
||||
# Sum of losses: L(x,c,l,g) = (Lconf(x, c) + αLloc(x,l,g)) / N
|
||||
N = max(num_pos.data.sum().float(), 1)
|
||||
loss_l /= N
|
||||
loss_c /= N
|
||||
loss_landm /= N1
|
||||
|
||||
return loss_l, loss_c, loss_landm
|
||||
@@ -0,0 +1,105 @@
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torch.nn.functional as F
|
||||
from collections import OrderedDict
|
||||
import numpy as np
|
||||
import os
|
||||
|
||||
class Flatten(nn.Module):
|
||||
def __init__(self):
|
||||
super(Flatten, self).__init__()
|
||||
def forward(self, x):
|
||||
x = x.transpose(3, 2).contiguous()
|
||||
return x.view(x.size(0), -1)
|
||||
|
||||
class PNet(nn.Module):
|
||||
def __init__(self):
|
||||
super(PNet, self).__init__()
|
||||
self.model_path,_ = os.path.split(os.path.realpath(__file__))
|
||||
self.features = nn.Sequential(OrderedDict([
|
||||
('conv1', nn.Conv2d(3, 10, 3, 1)),
|
||||
('prelu1', nn.PReLU(10)),
|
||||
('pool1', nn.MaxPool2d(2, 2, ceil_mode=True)),
|
||||
('conv2', nn.Conv2d(10, 16, 3, 1)),
|
||||
('prelu2', nn.PReLU(16)),
|
||||
('conv3', nn.Conv2d(16, 32, 3, 1)),
|
||||
('prelu3', nn.PReLU(32))
|
||||
]))
|
||||
self.conv4_1 = nn.Conv2d(32, 2, 1, 1)
|
||||
self.conv4_2 = nn.Conv2d(32, 4, 1, 1)
|
||||
weights = np.load('./weights/pnet.npy', allow_pickle=True)[()]
|
||||
for n, p in self.named_parameters():
|
||||
p.data = torch.FloatTensor(weights[n])
|
||||
|
||||
def forward(self, x):
|
||||
x = self.features(x)
|
||||
a = self.conv4_1(x)
|
||||
b = self.conv4_2(x)
|
||||
a = F.softmax(a, dim=1)
|
||||
return b, a
|
||||
|
||||
class RNet(nn.Module):
|
||||
def __init__(self):
|
||||
super(RNet, self).__init__()
|
||||
self.model_path,_ = os.path.split(os.path.realpath(__file__))
|
||||
self.features = nn.Sequential(OrderedDict([
|
||||
('conv1', nn.Conv2d(3, 28, 3, 1)),
|
||||
('prelu1', nn.PReLU(28)),
|
||||
('pool1', nn.MaxPool2d(3, 2, ceil_mode=True)),
|
||||
('conv2', nn.Conv2d(28, 48, 3, 1)),
|
||||
('prelu2', nn.PReLU(48)),
|
||||
('pool2', nn.MaxPool2d(3, 2, ceil_mode=True)),
|
||||
('conv3', nn.Conv2d(48, 64, 2, 1)),
|
||||
('prelu3', nn.PReLU(64)),
|
||||
('flatten', Flatten()),
|
||||
('conv4', nn.Linear(576, 128)),
|
||||
('prelu4', nn.PReLU(128))
|
||||
]))
|
||||
self.conv5_1 = nn.Linear(128, 2)
|
||||
self.conv5_2 = nn.Linear(128, 4)
|
||||
weights = np.load('./weights/rnet.npy', allow_pickle=True)[()]
|
||||
for n, p in self.named_parameters():
|
||||
p.data = torch.FloatTensor(weights[n])
|
||||
|
||||
def forward(self, x):
|
||||
x = self.features(x)
|
||||
a = self.conv5_1(x)
|
||||
b = self.conv5_2(x)
|
||||
a = F.softmax(a, dim=1)
|
||||
return b, a
|
||||
|
||||
class ONet(nn.Module):
|
||||
def __init__(self):
|
||||
super(ONet, self).__init__()
|
||||
self.model_path,_ = os.path.split(os.path.realpath(__file__))
|
||||
self.features = nn.Sequential(OrderedDict([
|
||||
('conv1', nn.Conv2d(3, 32, 3, 1)),
|
||||
('prelu1', nn.PReLU(32)),
|
||||
('pool1', nn.MaxPool2d(3, 2, ceil_mode=True)),
|
||||
('conv2', nn.Conv2d(32, 64, 3, 1)),
|
||||
('prelu2', nn.PReLU(64)),
|
||||
('pool2', nn.MaxPool2d(3, 2, ceil_mode=True)),
|
||||
('conv3', nn.Conv2d(64, 64, 3, 1)),
|
||||
('prelu3', nn.PReLU(64)),
|
||||
('pool3', nn.MaxPool2d(2, 2, ceil_mode=True)),
|
||||
('conv4', nn.Conv2d(64, 128, 2, 1)),
|
||||
('prelu4', nn.PReLU(128)),
|
||||
('flatten', Flatten()),
|
||||
('conv5', nn.Linear(1152, 256)),
|
||||
('drop5', nn.Dropout(0.25)),
|
||||
('prelu5', nn.PReLU(256)),
|
||||
]))
|
||||
self.conv6_1 = nn.Linear(256, 2)
|
||||
self.conv6_2 = nn.Linear(256, 4)
|
||||
self.conv6_3 = nn.Linear(256, 10)
|
||||
weights = np.load('./weights/onet.npy', allow_pickle=True)[()]
|
||||
for n, p in self.named_parameters():
|
||||
p.data = torch.FloatTensor(weights[n])
|
||||
|
||||
def forward(self, x):
|
||||
x = self.features(x)
|
||||
a = self.conv6_1(x)
|
||||
b = self.conv6_2(x)
|
||||
c = self.conv6_3(x)
|
||||
a = F.softmax(a, dim=1)
|
||||
return c, b, a
|
||||
@@ -0,0 +1,137 @@
|
||||
import time
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torchvision.models._utils as _utils
|
||||
import torchvision.models as models
|
||||
import torch.nn.functional as F
|
||||
from torch.autograd import Variable
|
||||
|
||||
def conv_bn(inp, oup, stride = 1, leaky = 0):
|
||||
return nn.Sequential(
|
||||
nn.Conv2d(inp, oup, 3, stride, 1, bias=False),
|
||||
nn.BatchNorm2d(oup),
|
||||
nn.LeakyReLU(negative_slope=leaky, inplace=True)
|
||||
)
|
||||
|
||||
def conv_bn_no_relu(inp, oup, stride):
|
||||
return nn.Sequential(
|
||||
nn.Conv2d(inp, oup, 3, stride, 1, bias=False),
|
||||
nn.BatchNorm2d(oup),
|
||||
)
|
||||
|
||||
def conv_bn1X1(inp, oup, stride, leaky=0):
|
||||
return nn.Sequential(
|
||||
nn.Conv2d(inp, oup, 1, stride, padding=0, bias=False),
|
||||
nn.BatchNorm2d(oup),
|
||||
nn.LeakyReLU(negative_slope=leaky, inplace=True)
|
||||
)
|
||||
|
||||
def conv_dw(inp, oup, stride, leaky=0.1):
|
||||
return nn.Sequential(
|
||||
nn.Conv2d(inp, inp, 3, stride, 1, groups=inp, bias=False),
|
||||
nn.BatchNorm2d(inp),
|
||||
nn.LeakyReLU(negative_slope= leaky,inplace=True),
|
||||
|
||||
nn.Conv2d(inp, oup, 1, 1, 0, bias=False),
|
||||
nn.BatchNorm2d(oup),
|
||||
nn.LeakyReLU(negative_slope= leaky,inplace=True),
|
||||
)
|
||||
|
||||
class SSH(nn.Module):
|
||||
def __init__(self, in_channel, out_channel):
|
||||
super(SSH, self).__init__()
|
||||
assert out_channel % 4 == 0
|
||||
leaky = 0
|
||||
if (out_channel <= 64):
|
||||
leaky = 0.1
|
||||
self.conv3X3 = conv_bn_no_relu(in_channel, out_channel//2, stride=1)
|
||||
|
||||
self.conv5X5_1 = conv_bn(in_channel, out_channel//4, stride=1, leaky = leaky)
|
||||
self.conv5X5_2 = conv_bn_no_relu(out_channel//4, out_channel//4, stride=1)
|
||||
|
||||
self.conv7X7_2 = conv_bn(out_channel//4, out_channel//4, stride=1, leaky = leaky)
|
||||
self.conv7x7_3 = conv_bn_no_relu(out_channel//4, out_channel//4, stride=1)
|
||||
|
||||
def forward(self, input):
|
||||
conv3X3 = self.conv3X3(input)
|
||||
|
||||
conv5X5_1 = self.conv5X5_1(input)
|
||||
conv5X5 = self.conv5X5_2(conv5X5_1)
|
||||
|
||||
conv7X7_2 = self.conv7X7_2(conv5X5_1)
|
||||
conv7X7 = self.conv7x7_3(conv7X7_2)
|
||||
|
||||
out = torch.cat([conv3X3, conv5X5, conv7X7], dim=1)
|
||||
out = F.relu(out)
|
||||
return out
|
||||
|
||||
class FPN(nn.Module):
|
||||
def __init__(self,in_channels_list,out_channels):
|
||||
super(FPN,self).__init__()
|
||||
leaky = 0
|
||||
if (out_channels <= 64):
|
||||
leaky = 0.1
|
||||
self.output1 = conv_bn1X1(in_channels_list[0], out_channels, stride = 1, leaky = leaky)
|
||||
self.output2 = conv_bn1X1(in_channels_list[1], out_channels, stride = 1, leaky = leaky)
|
||||
self.output3 = conv_bn1X1(in_channels_list[2], out_channels, stride = 1, leaky = leaky)
|
||||
|
||||
self.merge1 = conv_bn(out_channels, out_channels, leaky = leaky)
|
||||
self.merge2 = conv_bn(out_channels, out_channels, leaky = leaky)
|
||||
|
||||
def forward(self, input):
|
||||
# names = list(input.keys())
|
||||
input = list(input.values())
|
||||
|
||||
output1 = self.output1(input[0])
|
||||
output2 = self.output2(input[1])
|
||||
output3 = self.output3(input[2])
|
||||
|
||||
up3 = F.interpolate(output3, size=[output2.size(2), output2.size(3)], mode="nearest")
|
||||
output2 = output2 + up3
|
||||
output2 = self.merge2(output2)
|
||||
|
||||
up2 = F.interpolate(output2, size=[output1.size(2), output1.size(3)], mode="nearest")
|
||||
output1 = output1 + up2
|
||||
output1 = self.merge1(output1)
|
||||
|
||||
out = [output1, output2, output3]
|
||||
return out
|
||||
|
||||
|
||||
|
||||
class MobileNetV1(nn.Module):
|
||||
def __init__(self):
|
||||
super(MobileNetV1, self).__init__()
|
||||
self.stage1 = nn.Sequential(
|
||||
conv_bn(3, 8, 2, leaky = 0.1), # 3
|
||||
conv_dw(8, 16, 1), # 7
|
||||
conv_dw(16, 32, 2), # 11
|
||||
conv_dw(32, 32, 1), # 19
|
||||
conv_dw(32, 64, 2), # 27
|
||||
conv_dw(64, 64, 1), # 43
|
||||
)
|
||||
self.stage2 = nn.Sequential(
|
||||
conv_dw(64, 128, 2), # 43 + 16 = 59
|
||||
conv_dw(128, 128, 1), # 59 + 32 = 91
|
||||
conv_dw(128, 128, 1), # 91 + 32 = 123
|
||||
conv_dw(128, 128, 1), # 123 + 32 = 155
|
||||
conv_dw(128, 128, 1), # 155 + 32 = 187
|
||||
conv_dw(128, 128, 1), # 187 + 32 = 219
|
||||
)
|
||||
self.stage3 = nn.Sequential(
|
||||
conv_dw(128, 256, 2), # 219 +3 2 = 241
|
||||
conv_dw(256, 256, 1), # 241 + 64 = 301
|
||||
)
|
||||
self.avg = nn.AdaptiveAvgPool2d((1,1))
|
||||
self.fc = nn.Linear(256, 1000)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.stage1(x)
|
||||
x = self.stage2(x)
|
||||
x = self.stage3(x)
|
||||
x = self.avg(x)
|
||||
# x = self.model(x)
|
||||
x = x.view(-1, 256)
|
||||
x = self.fc(x)
|
||||
return x
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
# --------------------------------------------------------
|
||||
# Fast R-CNN
|
||||
# Copyright (c) 2015 Microsoft
|
||||
# Licensed under The MIT License [see LICENSE for details]
|
||||
# Written by Ross Girshick
|
||||
# --------------------------------------------------------
|
||||
|
||||
import numpy as np
|
||||
|
||||
def py_cpu_nms(dets, thresh, min_face_size = 50):
|
||||
"""Pure Python NMS baseline."""
|
||||
x1 = dets[:, 0]
|
||||
y1 = dets[:, 1]
|
||||
x2 = dets[:, 2]
|
||||
y2 = dets[:, 3]
|
||||
scores = dets[:, 4]
|
||||
|
||||
areas = (x2 - x1 + 1) * (y2 - y1 + 1)
|
||||
order = scores.argsort()[::-1]
|
||||
|
||||
keep = []
|
||||
while order.size > 0:
|
||||
i = order[0]
|
||||
keep.append(i)
|
||||
xx1 = np.maximum(x1[i], x1[order[1:]])
|
||||
yy1 = np.maximum(y1[i], y1[order[1:]])
|
||||
xx2 = np.minimum(x2[i], x2[order[1:]])
|
||||
yy2 = np.minimum(y2[i], y2[order[1:]])
|
||||
|
||||
w = np.maximum(0.0, xx2 - xx1 + 1)
|
||||
h = np.maximum(0.0, yy2 - yy1 + 1)
|
||||
inter = w * h
|
||||
ovr = inter / (areas[i] + areas[order[1:]] - inter)
|
||||
|
||||
inds = np.where(ovr <= thresh)[0]
|
||||
order = order[inds + 1]
|
||||
|
||||
#filter_small_faces
|
||||
filter_list = []
|
||||
for idx in keep:
|
||||
w = np.abs(dets[idx, 2] - dets[idx, 0])
|
||||
h = np.abs(dets[idx, 3] - dets[idx, 1])
|
||||
if max(w, h) < min_face_size: continue
|
||||
filter_list.append(idx)
|
||||
|
||||
return filter_list
|
||||
@@ -0,0 +1,223 @@
|
||||
import torch.nn as nn
|
||||
import torch.utils.model_zoo as model_zoo
|
||||
|
||||
|
||||
__all__ = ['ResNet', 'resnet18', 'resnet34', 'resnet50', 'resnet101',
|
||||
'resnet152']
|
||||
|
||||
|
||||
model_urls = {
|
||||
'resnet18': 'https://download.pytorch.org/models/resnet18-5c106cde.pth',
|
||||
'resnet34': 'https://download.pytorch.org/models/resnet34-333f7ec4.pth',
|
||||
'resnet50': 'https://download.pytorch.org/models/resnet50-19c8e357.pth',
|
||||
'resnet101': 'https://download.pytorch.org/models/resnet101-5d3b4d8f.pth',
|
||||
'resnet152': 'https://download.pytorch.org/models/resnet152-b121ed2d.pth',
|
||||
}
|
||||
|
||||
|
||||
def conv3x3(in_planes, out_planes, stride=1):
|
||||
"""3x3 convolution with padding"""
|
||||
return nn.Conv2d(in_planes, out_planes, kernel_size=3, stride=stride,
|
||||
padding=1, bias=False)
|
||||
|
||||
|
||||
def conv1x1(in_planes, out_planes, stride=1):
|
||||
"""1x1 convolution"""
|
||||
return nn.Conv2d(in_planes, out_planes, kernel_size=1, stride=stride, bias=False)
|
||||
|
||||
|
||||
class BasicBlock(nn.Module):
|
||||
expansion = 1
|
||||
|
||||
def __init__(self, inplanes, planes, stride=1, downsample=None):
|
||||
super(BasicBlock, self).__init__()
|
||||
self.conv1 = conv3x3(inplanes, planes, stride)
|
||||
self.bn1 = nn.BatchNorm2d(planes)
|
||||
self.relu = nn.ReLU(inplace=True)
|
||||
self.conv2 = conv3x3(planes, planes)
|
||||
self.bn2 = nn.BatchNorm2d(planes)
|
||||
self.downsample = downsample
|
||||
self.stride = stride
|
||||
|
||||
def forward(self, x):
|
||||
identity = x
|
||||
|
||||
out = self.conv1(x)
|
||||
out = self.bn1(out)
|
||||
out = self.relu(out)
|
||||
|
||||
out = self.conv2(out)
|
||||
out = self.bn2(out)
|
||||
|
||||
if self.downsample is not None:
|
||||
identity = self.downsample(x)
|
||||
|
||||
out += identity
|
||||
out = self.relu(out)
|
||||
|
||||
return out
|
||||
|
||||
|
||||
class Bottleneck(nn.Module):
|
||||
expansion = 4
|
||||
|
||||
def __init__(self, inplanes, planes, stride=1, downsample=None):
|
||||
super(Bottleneck, self).__init__()
|
||||
self.conv1 = conv1x1(inplanes, planes)
|
||||
self.bn1 = nn.BatchNorm2d(planes)
|
||||
self.conv2 = conv3x3(planes, planes, stride)
|
||||
self.bn2 = nn.BatchNorm2d(planes)
|
||||
self.conv3 = conv1x1(planes, planes * self.expansion)
|
||||
self.bn3 = nn.BatchNorm2d(planes * self.expansion)
|
||||
self.relu = nn.ReLU(inplace=True)
|
||||
self.downsample = downsample
|
||||
self.stride = stride
|
||||
|
||||
def forward(self, x):
|
||||
identity = x
|
||||
|
||||
out = self.conv1(x)
|
||||
out = self.bn1(out)
|
||||
out = self.relu(out)
|
||||
|
||||
out = self.conv2(out)
|
||||
out = self.bn2(out)
|
||||
out = self.relu(out)
|
||||
|
||||
out = self.conv3(out)
|
||||
out = self.bn3(out)
|
||||
|
||||
if self.downsample is not None:
|
||||
identity = self.downsample(x)
|
||||
|
||||
out += identity
|
||||
out = self.relu(out)
|
||||
|
||||
return out
|
||||
|
||||
|
||||
class ResNet(nn.Module):
|
||||
|
||||
def __init__(self, block, layers, num_classes=1000, zero_init_residual=False):
|
||||
super(ResNet, self).__init__()
|
||||
self.inplanes = 64
|
||||
self.conv1 = nn.Conv2d(3, 64, kernel_size=7, stride=2, padding=3,
|
||||
bias=False)
|
||||
self.bn1 = nn.BatchNorm2d(64)
|
||||
self.relu = nn.ReLU(inplace=True)
|
||||
self.maxpool = nn.MaxPool2d(kernel_size=3, stride=2, padding=1)
|
||||
self.layer1 = self._make_layer(block, 64, layers[0])
|
||||
self.layer2 = self._make_layer(block, 128, layers[1], stride=2)
|
||||
self.layer3 = self._make_layer(block, 256, layers[2], stride=2)
|
||||
self.layer4 = self._make_layer(block, 512, layers[3], stride=2)
|
||||
self.avgpool = nn.AdaptiveAvgPool2d((1, 1))
|
||||
self.fc = nn.Linear(512 * block.expansion, num_classes)
|
||||
|
||||
for m in self.modules():
|
||||
if isinstance(m, nn.Conv2d):
|
||||
nn.init.kaiming_normal_(m.weight, mode='fan_out', nonlinearity='relu')
|
||||
elif isinstance(m, nn.BatchNorm2d):
|
||||
nn.init.constant_(m.weight, 1)
|
||||
nn.init.constant_(m.bias, 0)
|
||||
|
||||
# Zero-initialize the last BN in each residual branch,
|
||||
# so that the residual branch starts with zeros, and each residual block behaves like an identity.
|
||||
# This improves the model by 0.2~0.3% according to https://arxiv.org/abs/1706.02677
|
||||
if zero_init_residual:
|
||||
for m in self.modules():
|
||||
if isinstance(m, Bottleneck):
|
||||
nn.init.constant_(m.bn3.weight, 0)
|
||||
elif isinstance(m, BasicBlock):
|
||||
nn.init.constant_(m.bn2.weight, 0)
|
||||
|
||||
def _make_layer(self, block, planes, blocks, stride=1):
|
||||
downsample = None
|
||||
if stride != 1 or self.inplanes != planes * block.expansion:
|
||||
downsample = nn.Sequential(
|
||||
conv1x1(self.inplanes, planes * block.expansion, stride),
|
||||
nn.BatchNorm2d(planes * block.expansion),
|
||||
)
|
||||
|
||||
layers = []
|
||||
layers.append(block(self.inplanes, planes, stride, downsample))
|
||||
self.inplanes = planes * block.expansion
|
||||
for _ in range(1, blocks):
|
||||
layers.append(block(self.inplanes, planes))
|
||||
|
||||
return nn.Sequential(*layers)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.conv1(x)
|
||||
x = self.bn1(x)
|
||||
x = self.relu(x)
|
||||
x = self.maxpool(x)
|
||||
|
||||
x = self.layer1(x)
|
||||
x = self.layer2(x)
|
||||
x = self.layer3(x)
|
||||
x = self.layer4(x)
|
||||
|
||||
x = self.avgpool(x)
|
||||
x = x.view(-1, 512)
|
||||
|
||||
return self.fc(x)
|
||||
|
||||
|
||||
def resnet18(pretrained=False, **kwargs):
|
||||
"""Constructs a ResNet-18 model.
|
||||
|
||||
Args:
|
||||
pretrained (bool): If True, returns a model pre-trained on ImageNet
|
||||
"""
|
||||
model = ResNet(BasicBlock, [2, 2, 2, 2], **kwargs)
|
||||
if pretrained:
|
||||
model.load_state_dict(model_zoo.load_url(model_urls['resnet18']))
|
||||
return model
|
||||
|
||||
|
||||
def resnet34(pretrained=False, **kwargs):
|
||||
"""Constructs a ResNet-34 model.
|
||||
|
||||
Args:
|
||||
pretrained (bool): If True, returns a model pre-trained on ImageNet
|
||||
"""
|
||||
model = ResNet(BasicBlock, [3, 4, 6, 3], **kwargs)
|
||||
if pretrained:
|
||||
model.load_state_dict(model_zoo.load_url(model_urls['resnet34']))
|
||||
return model
|
||||
|
||||
|
||||
def resnet50(pretrained=False, **kwargs):
|
||||
"""Constructs a ResNet-50 model.
|
||||
|
||||
Args:
|
||||
pretrained (bool): If True, returns a model pre-trained on ImageNet
|
||||
"""
|
||||
model = ResNet(Bottleneck, [3, 4, 6, 3], **kwargs)
|
||||
if pretrained:
|
||||
model.load_state_dict(model_zoo.load_url(model_urls['resnet50']))
|
||||
return model
|
||||
|
||||
|
||||
def resnet101(pretrained=False, **kwargs):
|
||||
"""Constructs a ResNet-101 model.
|
||||
|
||||
Args:
|
||||
pretrained (bool): If True, returns a model pre-trained on ImageNet
|
||||
"""
|
||||
model = ResNet(Bottleneck, [3, 4, 23, 3], **kwargs)
|
||||
if pretrained:
|
||||
model.load_state_dict(model_zoo.load_url(model_urls['resnet101']))
|
||||
return model
|
||||
|
||||
|
||||
def resnet152(pretrained=False, **kwargs):
|
||||
"""Constructs a ResNet-152 model.
|
||||
|
||||
Args:
|
||||
pretrained (bool): If True, returns a model pre-trained on ImageNet
|
||||
"""
|
||||
model = ResNet(Bottleneck, [3, 8, 36, 3], **kwargs)
|
||||
if pretrained:
|
||||
model.load_state_dict(model_zoo.load_url(model_urls['resnet152']))
|
||||
return model
|
||||
@@ -0,0 +1,127 @@
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torchvision.models.detection.backbone_utils as backbone_utils
|
||||
import torchvision.models._utils as _utils
|
||||
import torch.nn.functional as F
|
||||
from collections import OrderedDict
|
||||
|
||||
from models.net import MobileNetV1 as MobileNetV1
|
||||
from models.net import FPN as FPN
|
||||
from models.net import SSH as SSH
|
||||
|
||||
|
||||
|
||||
class ClassHead(nn.Module):
|
||||
def __init__(self,inchannels=512,num_anchors=3):
|
||||
super(ClassHead,self).__init__()
|
||||
self.num_anchors = num_anchors
|
||||
self.conv1x1 = nn.Conv2d(inchannels,self.num_anchors*2,kernel_size=(1,1),stride=1,padding=0)
|
||||
|
||||
def forward(self,x):
|
||||
out = self.conv1x1(x)
|
||||
out = out.permute(0,2,3,1).contiguous()
|
||||
|
||||
return out.view(out.shape[0], -1, 2)
|
||||
|
||||
class BboxHead(nn.Module):
|
||||
def __init__(self,inchannels=512,num_anchors=3):
|
||||
super(BboxHead,self).__init__()
|
||||
self.conv1x1 = nn.Conv2d(inchannels,num_anchors*4,kernel_size=(1,1),stride=1,padding=0)
|
||||
|
||||
def forward(self,x):
|
||||
out = self.conv1x1(x)
|
||||
out = out.permute(0,2,3,1).contiguous()
|
||||
|
||||
return out.view(out.shape[0], -1, 4)
|
||||
|
||||
class LandmarkHead(nn.Module):
|
||||
def __init__(self,inchannels=512,num_anchors=3):
|
||||
super(LandmarkHead,self).__init__()
|
||||
self.conv1x1 = nn.Conv2d(inchannels,num_anchors*10,kernel_size=(1,1),stride=1,padding=0)
|
||||
|
||||
def forward(self,x):
|
||||
out = self.conv1x1(x)
|
||||
out = out.permute(0,2,3,1).contiguous()
|
||||
|
||||
return out.view(out.shape[0], -1, 10)
|
||||
|
||||
class RetinaFace(nn.Module):
|
||||
def __init__(self, cfg = None, phase = 'train'):
|
||||
"""
|
||||
:param cfg: Network related settings.
|
||||
:param phase: train or test.
|
||||
"""
|
||||
super(RetinaFace,self).__init__()
|
||||
self.phase = phase
|
||||
backbone = None
|
||||
if cfg['name'] == 'mobilenet0.25':
|
||||
backbone = MobileNetV1()
|
||||
if cfg['pretrain']:
|
||||
checkpoint = torch.load("./weights/mobilenetV1X0.25_pretrain.tar", map_location=torch.device('cpu'))
|
||||
from collections import OrderedDict
|
||||
new_state_dict = OrderedDict()
|
||||
for k, v in checkpoint['state_dict'].items():
|
||||
name = k[7:] # remove module.
|
||||
new_state_dict[name] = v
|
||||
# load params
|
||||
backbone.load_state_dict(new_state_dict)
|
||||
elif cfg['name'] == 'Resnet50':
|
||||
import torchvision.models as models
|
||||
backbone = models.resnet50(pretrained=cfg['pretrain'])
|
||||
|
||||
self.body = _utils.IntermediateLayerGetter(backbone, cfg['return_layers'])
|
||||
in_channels_stage2 = cfg['in_channel']
|
||||
in_channels_list = [
|
||||
in_channels_stage2 * 2,
|
||||
in_channels_stage2 * 4,
|
||||
in_channels_stage2 * 8,
|
||||
]
|
||||
out_channels = cfg['out_channel']
|
||||
self.fpn = FPN(in_channels_list,out_channels)
|
||||
self.ssh1 = SSH(out_channels, out_channels)
|
||||
self.ssh2 = SSH(out_channels, out_channels)
|
||||
self.ssh3 = SSH(out_channels, out_channels)
|
||||
|
||||
self.ClassHead = self._make_class_head(fpn_num=3, inchannels=cfg['out_channel'])
|
||||
self.BboxHead = self._make_bbox_head(fpn_num=3, inchannels=cfg['out_channel'])
|
||||
self.LandmarkHead = self._make_landmark_head(fpn_num=3, inchannels=cfg['out_channel'])
|
||||
|
||||
def _make_class_head(self,fpn_num=3,inchannels=64,anchor_num=2):
|
||||
classhead = nn.ModuleList()
|
||||
for i in range(fpn_num):
|
||||
classhead.append(ClassHead(inchannels,anchor_num))
|
||||
return classhead
|
||||
|
||||
def _make_bbox_head(self,fpn_num=3,inchannels=64,anchor_num=2):
|
||||
bboxhead = nn.ModuleList()
|
||||
for i in range(fpn_num):
|
||||
bboxhead.append(BboxHead(inchannels,anchor_num))
|
||||
return bboxhead
|
||||
|
||||
def _make_landmark_head(self,fpn_num=3,inchannels=64,anchor_num=2):
|
||||
landmarkhead = nn.ModuleList()
|
||||
for i in range(fpn_num):
|
||||
landmarkhead.append(LandmarkHead(inchannels,anchor_num))
|
||||
return landmarkhead
|
||||
|
||||
def forward(self,inputs):
|
||||
out = self.body(inputs)
|
||||
|
||||
# FPN
|
||||
fpn = self.fpn(out)
|
||||
|
||||
# SSH
|
||||
feature1 = self.ssh1(fpn[0])
|
||||
feature2 = self.ssh2(fpn[1])
|
||||
feature3 = self.ssh3(fpn[2])
|
||||
features = [feature1, feature2, feature3]
|
||||
|
||||
bbox_regressions = torch.cat([self.BboxHead[i](feature) for i, feature in enumerate(features)], dim=1)
|
||||
classifications = torch.cat([self.ClassHead[i](feature) for i, feature in enumerate(features)],dim=1)
|
||||
ldm_regressions = torch.cat([self.LandmarkHead[i](feature) for i, feature in enumerate(features)], dim=1)
|
||||
|
||||
if self.phase == 'train':
|
||||
output = (bbox_regressions, classifications, ldm_regressions)
|
||||
else:
|
||||
output = (bbox_regressions, F.softmax(classifications, dim=-1), ldm_regressions)
|
||||
return output
|
||||
@@ -0,0 +1,688 @@
|
||||
# ------------------------------------------------------------------------------
|
||||
# Copyright (c) Microsoft
|
||||
# Licensed under the MIT License.
|
||||
# Written by Ke Sun (sunk@mail.ustc.edu.cn), Jingyi Xie (hsfzxjy@gmail.com)
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import os
|
||||
import logging
|
||||
import functools
|
||||
|
||||
import numpy as np
|
||||
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torch._utils
|
||||
import torch.nn.functional as F
|
||||
|
||||
ALIGN_CORNERS = True
|
||||
BN_MOMENTUM = 0.1
|
||||
logger = logging.getLogger(__name__)
|
||||
BatchNorm2d_class = BatchNorm2d = torch.nn.SyncBatchNorm
|
||||
relu_inplace = True
|
||||
|
||||
|
||||
class ModuleHelper:
|
||||
|
||||
@staticmethod
|
||||
def BNReLU(num_features, bn_type=None, **kwargs):
|
||||
return nn.Sequential(
|
||||
BatchNorm2d(num_features, **kwargs),
|
||||
nn.ReLU()
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def BatchNorm2d(*args, **kwargs):
|
||||
return BatchNorm2d
|
||||
|
||||
|
||||
def conv3x3(in_planes, out_planes, stride=1):
|
||||
"""3x3 convolution with padding"""
|
||||
return nn.Conv2d(in_planes, out_planes, kernel_size=3, stride=stride,
|
||||
padding=1, bias=False)
|
||||
|
||||
|
||||
class SpatialGather_Module(nn.Module):
|
||||
"""
|
||||
Aggregate the context features according to the initial
|
||||
predicted probability distribution.
|
||||
Employ the soft-weighted method to aggregate the context.
|
||||
"""
|
||||
|
||||
def __init__(self, cls_num=0, scale=1):
|
||||
super(SpatialGather_Module, self).__init__()
|
||||
self.cls_num = cls_num
|
||||
self.scale = scale
|
||||
|
||||
def forward(self, feats, probs):
|
||||
batch_size, c, h, w = probs.size(0), probs.size(1), probs.size(2), probs.size(3)
|
||||
probs = probs.view(batch_size, c, -1)
|
||||
feats = feats.view(batch_size, feats.size(1), -1)
|
||||
feats = feats.permute(0, 2, 1) # batch x hw x c
|
||||
probs = F.softmax(self.scale * probs, dim=2) # batch x k x hw
|
||||
ocr_context = torch.matmul(probs, feats) \
|
||||
.permute(0, 2, 1).unsqueeze(3) # batch x k x c
|
||||
return ocr_context
|
||||
|
||||
|
||||
class _ObjectAttentionBlock(nn.Module):
|
||||
'''
|
||||
The basic implementation for object context block
|
||||
Input:
|
||||
N X C X H X W
|
||||
Parameters:
|
||||
in_channels : the dimension of the input feature map
|
||||
key_channels : the dimension after the key/query transform
|
||||
scale : choose the scale to downsample the input feature maps (save memory cost)
|
||||
bn_type : specify the bn type
|
||||
Return:
|
||||
N X C X H X W
|
||||
'''
|
||||
|
||||
def __init__(self,
|
||||
in_channels,
|
||||
key_channels,
|
||||
scale=1,
|
||||
bn_type=None):
|
||||
super(_ObjectAttentionBlock, self).__init__()
|
||||
self.scale = scale
|
||||
self.in_channels = in_channels
|
||||
self.key_channels = key_channels
|
||||
self.pool = nn.MaxPool2d(kernel_size=(scale, scale))
|
||||
self.f_pixel = nn.Sequential(
|
||||
nn.Conv2d(in_channels=self.in_channels, out_channels=self.key_channels,
|
||||
kernel_size=1, stride=1, padding=0, bias=False),
|
||||
ModuleHelper.BNReLU(self.key_channels, bn_type=bn_type),
|
||||
nn.Conv2d(in_channels=self.key_channels, out_channels=self.key_channels,
|
||||
kernel_size=1, stride=1, padding=0, bias=False),
|
||||
ModuleHelper.BNReLU(self.key_channels, bn_type=bn_type),
|
||||
)
|
||||
self.f_object = nn.Sequential(
|
||||
nn.Conv2d(in_channels=self.in_channels, out_channels=self.key_channels,
|
||||
kernel_size=1, stride=1, padding=0, bias=False),
|
||||
ModuleHelper.BNReLU(self.key_channels, bn_type=bn_type),
|
||||
nn.Conv2d(in_channels=self.key_channels, out_channels=self.key_channels,
|
||||
kernel_size=1, stride=1, padding=0, bias=False),
|
||||
ModuleHelper.BNReLU(self.key_channels, bn_type=bn_type),
|
||||
)
|
||||
self.f_down = nn.Sequential(
|
||||
nn.Conv2d(in_channels=self.in_channels, out_channels=self.key_channels,
|
||||
kernel_size=1, stride=1, padding=0, bias=False),
|
||||
ModuleHelper.BNReLU(self.key_channels, bn_type=bn_type),
|
||||
)
|
||||
self.f_up = nn.Sequential(
|
||||
nn.Conv2d(in_channels=self.key_channels, out_channels=self.in_channels,
|
||||
kernel_size=1, stride=1, padding=0, bias=False),
|
||||
ModuleHelper.BNReLU(self.in_channels, bn_type=bn_type),
|
||||
)
|
||||
|
||||
def forward(self, x, proxy):
|
||||
batch_size, h, w = x.size(0), x.size(2), x.size(3)
|
||||
if self.scale > 1:
|
||||
x = self.pool(x)
|
||||
|
||||
query = self.f_pixel(x).view(batch_size, self.key_channels, -1)
|
||||
query = query.permute(0, 2, 1)
|
||||
key = self.f_object(proxy).view(batch_size, self.key_channels, -1)
|
||||
value = self.f_down(proxy).view(batch_size, self.key_channels, -1)
|
||||
value = value.permute(0, 2, 1)
|
||||
|
||||
sim_map = torch.matmul(query, key)
|
||||
sim_map = (self.key_channels ** -.5) * sim_map
|
||||
sim_map = F.softmax(sim_map, dim=-1)
|
||||
|
||||
# add bg context ...
|
||||
context = torch.matmul(sim_map, value)
|
||||
context = context.permute(0, 2, 1).contiguous()
|
||||
context = context.view(batch_size, self.key_channels, *x.size()[2:])
|
||||
context = self.f_up(context)
|
||||
if self.scale > 1:
|
||||
context = F.interpolate(input=context, size=(h, w), mode='bilinear', align_corners=ALIGN_CORNERS)
|
||||
|
||||
return context
|
||||
|
||||
|
||||
class ObjectAttentionBlock2D(_ObjectAttentionBlock):
|
||||
def __init__(self,
|
||||
in_channels,
|
||||
key_channels,
|
||||
scale=1,
|
||||
bn_type=None):
|
||||
super(ObjectAttentionBlock2D, self).__init__(in_channels,
|
||||
key_channels,
|
||||
scale,
|
||||
bn_type=bn_type)
|
||||
|
||||
|
||||
class SpatialOCR_Module(nn.Module):
|
||||
"""
|
||||
Implementation of the OCR module:
|
||||
We aggregate the global object representation to update the representation for each pixel.
|
||||
"""
|
||||
|
||||
def __init__(self,
|
||||
in_channels,
|
||||
key_channels,
|
||||
out_channels,
|
||||
scale=1,
|
||||
dropout=0.1,
|
||||
bn_type=None):
|
||||
super(SpatialOCR_Module, self).__init__()
|
||||
self.object_context_block = ObjectAttentionBlock2D(in_channels,
|
||||
key_channels,
|
||||
scale,
|
||||
bn_type)
|
||||
_in_channels = 2 * in_channels
|
||||
|
||||
self.conv_bn_dropout = nn.Sequential(
|
||||
nn.Conv2d(_in_channels, out_channels, kernel_size=1, padding=0, bias=False),
|
||||
ModuleHelper.BNReLU(out_channels, bn_type=bn_type),
|
||||
nn.Dropout2d(dropout)
|
||||
)
|
||||
|
||||
def forward(self, feats, proxy_feats):
|
||||
context = self.object_context_block(feats, proxy_feats)
|
||||
|
||||
output = self.conv_bn_dropout(torch.cat([context, feats], 1))
|
||||
|
||||
return output
|
||||
|
||||
|
||||
class BasicBlock(nn.Module):
|
||||
expansion = 1
|
||||
|
||||
def __init__(self, inplanes, planes, stride=1, downsample=None):
|
||||
super(BasicBlock, self).__init__()
|
||||
self.conv1 = conv3x3(inplanes, planes, stride)
|
||||
self.bn1 = BatchNorm2d(planes, momentum=BN_MOMENTUM)
|
||||
self.relu = nn.ReLU(inplace=relu_inplace)
|
||||
self.conv2 = conv3x3(planes, planes)
|
||||
self.bn2 = BatchNorm2d(planes, momentum=BN_MOMENTUM)
|
||||
self.downsample = downsample
|
||||
self.stride = stride
|
||||
|
||||
def forward(self, x):
|
||||
residual = x
|
||||
|
||||
out = self.conv1(x)
|
||||
out = self.bn1(out)
|
||||
out = self.relu(out)
|
||||
|
||||
out = self.conv2(out)
|
||||
out = self.bn2(out)
|
||||
|
||||
if self.downsample is not None:
|
||||
residual = self.downsample(x)
|
||||
|
||||
out = out + residual
|
||||
out = self.relu(out)
|
||||
|
||||
return out
|
||||
|
||||
|
||||
class Bottleneck(nn.Module):
|
||||
expansion = 4
|
||||
|
||||
def __init__(self, inplanes, planes, stride=1, downsample=None):
|
||||
super(Bottleneck, self).__init__()
|
||||
self.conv1 = nn.Conv2d(inplanes, planes, kernel_size=1, bias=False)
|
||||
self.bn1 = BatchNorm2d(planes, momentum=BN_MOMENTUM)
|
||||
self.conv2 = nn.Conv2d(planes, planes, kernel_size=3, stride=stride,
|
||||
padding=1, bias=False)
|
||||
self.bn2 = BatchNorm2d(planes, momentum=BN_MOMENTUM)
|
||||
self.conv3 = nn.Conv2d(planes, planes * self.expansion, kernel_size=1,
|
||||
bias=False)
|
||||
self.bn3 = BatchNorm2d(planes * self.expansion,
|
||||
momentum=BN_MOMENTUM)
|
||||
self.relu = nn.ReLU(inplace=relu_inplace)
|
||||
self.downsample = downsample
|
||||
self.stride = stride
|
||||
|
||||
def forward(self, x):
|
||||
residual = x
|
||||
|
||||
out = self.conv1(x)
|
||||
out = self.bn1(out)
|
||||
out = self.relu(out)
|
||||
|
||||
out = self.conv2(out)
|
||||
out = self.bn2(out)
|
||||
out = self.relu(out)
|
||||
|
||||
out = self.conv3(out)
|
||||
out = self.bn3(out)
|
||||
|
||||
if self.downsample is not None:
|
||||
residual = self.downsample(x)
|
||||
|
||||
out = out + residual
|
||||
out = self.relu(out)
|
||||
|
||||
return out
|
||||
|
||||
|
||||
class HighResolutionModule(nn.Module):
|
||||
def __init__(self, num_branches, blocks, num_blocks, num_inchannels,
|
||||
num_channels, fuse_method, multi_scale_output=True):
|
||||
super(HighResolutionModule, self).__init__()
|
||||
self._check_branches(
|
||||
num_branches, blocks, num_blocks, num_inchannels, num_channels)
|
||||
|
||||
self.num_inchannels = num_inchannels
|
||||
self.fuse_method = fuse_method
|
||||
self.num_branches = num_branches
|
||||
|
||||
self.multi_scale_output = multi_scale_output
|
||||
|
||||
self.branches = self._make_branches(
|
||||
num_branches, blocks, num_blocks, num_channels)
|
||||
self.fuse_layers = self._make_fuse_layers()
|
||||
self.relu = nn.ReLU(inplace=relu_inplace)
|
||||
|
||||
def _check_branches(self, num_branches, blocks, num_blocks,
|
||||
num_inchannels, num_channels):
|
||||
if num_branches != len(num_blocks):
|
||||
error_msg = 'NUM_BRANCHES({}) <> NUM_BLOCKS({})'.format(
|
||||
num_branches, len(num_blocks))
|
||||
logger.error(error_msg)
|
||||
raise ValueError(error_msg)
|
||||
|
||||
if num_branches != len(num_channels):
|
||||
error_msg = 'NUM_BRANCHES({}) <> NUM_CHANNELS({})'.format(
|
||||
num_branches, len(num_channels))
|
||||
logger.error(error_msg)
|
||||
raise ValueError(error_msg)
|
||||
|
||||
if num_branches != len(num_inchannels):
|
||||
error_msg = 'NUM_BRANCHES({}) <> NUM_INCHANNELS({})'.format(
|
||||
num_branches, len(num_inchannels))
|
||||
logger.error(error_msg)
|
||||
raise ValueError(error_msg)
|
||||
|
||||
def _make_one_branch(self, branch_index, block, num_blocks, num_channels,
|
||||
stride=1):
|
||||
downsample = None
|
||||
if stride != 1 or \
|
||||
self.num_inchannels[branch_index] != num_channels[branch_index] * block.expansion:
|
||||
downsample = nn.Sequential(
|
||||
nn.Conv2d(self.num_inchannels[branch_index],
|
||||
num_channels[branch_index] * block.expansion,
|
||||
kernel_size=1, stride=stride, bias=False),
|
||||
BatchNorm2d(num_channels[branch_index] * block.expansion,
|
||||
momentum=BN_MOMENTUM),
|
||||
)
|
||||
|
||||
layers = []
|
||||
layers.append(block(self.num_inchannels[branch_index],
|
||||
num_channels[branch_index], stride, downsample))
|
||||
self.num_inchannels[branch_index] = \
|
||||
num_channels[branch_index] * block.expansion
|
||||
for i in range(1, num_blocks[branch_index]):
|
||||
layers.append(block(self.num_inchannels[branch_index],
|
||||
num_channels[branch_index]))
|
||||
|
||||
return nn.Sequential(*layers)
|
||||
|
||||
def _make_branches(self, num_branches, block, num_blocks, num_channels):
|
||||
branches = []
|
||||
|
||||
for i in range(num_branches):
|
||||
branches.append(
|
||||
self._make_one_branch(i, block, num_blocks, num_channels))
|
||||
|
||||
return nn.ModuleList(branches)
|
||||
|
||||
def _make_fuse_layers(self):
|
||||
if self.num_branches == 1:
|
||||
return None
|
||||
|
||||
num_branches = self.num_branches
|
||||
num_inchannels = self.num_inchannels
|
||||
fuse_layers = []
|
||||
for i in range(num_branches if self.multi_scale_output else 1):
|
||||
fuse_layer = []
|
||||
for j in range(num_branches):
|
||||
if j > i:
|
||||
fuse_layer.append(nn.Sequential(
|
||||
nn.Conv2d(num_inchannels[j],
|
||||
num_inchannels[i],
|
||||
1,
|
||||
1,
|
||||
0,
|
||||
bias=False),
|
||||
BatchNorm2d(num_inchannels[i], momentum=BN_MOMENTUM)))
|
||||
elif j == i:
|
||||
fuse_layer.append(None)
|
||||
else:
|
||||
conv3x3s = []
|
||||
for k in range(i - j):
|
||||
if k == i - j - 1:
|
||||
num_outchannels_conv3x3 = num_inchannels[i]
|
||||
conv3x3s.append(nn.Sequential(
|
||||
nn.Conv2d(num_inchannels[j],
|
||||
num_outchannels_conv3x3,
|
||||
3, 2, 1, bias=False),
|
||||
BatchNorm2d(num_outchannels_conv3x3,
|
||||
momentum=BN_MOMENTUM)))
|
||||
else:
|
||||
num_outchannels_conv3x3 = num_inchannels[j]
|
||||
conv3x3s.append(nn.Sequential(
|
||||
nn.Conv2d(num_inchannels[j],
|
||||
num_outchannels_conv3x3,
|
||||
3, 2, 1, bias=False),
|
||||
BatchNorm2d(num_outchannels_conv3x3,
|
||||
momentum=BN_MOMENTUM),
|
||||
nn.ReLU(inplace=relu_inplace)))
|
||||
fuse_layer.append(nn.Sequential(*conv3x3s))
|
||||
fuse_layers.append(nn.ModuleList(fuse_layer))
|
||||
|
||||
return nn.ModuleList(fuse_layers)
|
||||
|
||||
def get_num_inchannels(self):
|
||||
return self.num_inchannels
|
||||
|
||||
def forward(self, x):
|
||||
if self.num_branches == 1:
|
||||
return [self.branches[0](x[0])]
|
||||
|
||||
for i in range(self.num_branches):
|
||||
x[i] = self.branches[i](x[i])
|
||||
|
||||
x_fuse = []
|
||||
for i in range(len(self.fuse_layers)):
|
||||
y = x[0] if i == 0 else self.fuse_layers[i][0](x[0])
|
||||
for j in range(1, self.num_branches):
|
||||
if i == j:
|
||||
y = y + x[j]
|
||||
elif j > i:
|
||||
width_output = x[i].shape[-1]
|
||||
height_output = x[i].shape[-2]
|
||||
y = y + F.interpolate(
|
||||
self.fuse_layers[i][j](x[j]),
|
||||
size=[height_output, width_output],
|
||||
mode='bilinear', align_corners=ALIGN_CORNERS)
|
||||
else:
|
||||
y = y + self.fuse_layers[i][j](x[j])
|
||||
x_fuse.append(self.relu(y))
|
||||
|
||||
return x_fuse
|
||||
|
||||
|
||||
blocks_dict = {
|
||||
'BASIC': BasicBlock,
|
||||
'BOTTLENECK': Bottleneck
|
||||
}
|
||||
|
||||
|
||||
class HighResolutionNet(nn.Module):
|
||||
|
||||
def __init__(self, config, **kwargs):
|
||||
global ALIGN_CORNERS
|
||||
extra = config.MODEL.EXTRA
|
||||
super(HighResolutionNet, self).__init__()
|
||||
ALIGN_CORNERS = config.MODEL.ALIGN_CORNERS
|
||||
|
||||
# stem net
|
||||
self.conv1 = nn.Conv2d(config.MODEL.INPUTC, 64, kernel_size=3, stride=2, padding=1,
|
||||
bias=False)
|
||||
self.bn1 = BatchNorm2d(64, momentum=BN_MOMENTUM)
|
||||
self.conv2 = nn.Conv2d(64, 64, kernel_size=3, stride=2, padding=1,
|
||||
bias=False)
|
||||
self.bn2 = BatchNorm2d(64, momentum=BN_MOMENTUM)
|
||||
self.relu = nn.ReLU(inplace=relu_inplace)
|
||||
|
||||
self.stage1_cfg = extra['STAGE1']
|
||||
num_channels = self.stage1_cfg['NUM_CHANNELS'][0]
|
||||
block = blocks_dict[self.stage1_cfg['BLOCK']]
|
||||
num_blocks = self.stage1_cfg['NUM_BLOCKS'][0]
|
||||
self.layer1 = self._make_layer(block, 64, num_channels, num_blocks)
|
||||
stage1_out_channel = block.expansion * num_channels
|
||||
|
||||
self.stage2_cfg = extra['STAGE2']
|
||||
num_channels = self.stage2_cfg['NUM_CHANNELS']
|
||||
block = blocks_dict[self.stage2_cfg['BLOCK']]
|
||||
num_channels = [
|
||||
num_channels[i] * block.expansion for i in range(len(num_channels))]
|
||||
self.transition1 = self._make_transition_layer(
|
||||
[stage1_out_channel], num_channels)
|
||||
self.stage2, pre_stage_channels = self._make_stage(
|
||||
self.stage2_cfg, num_channels)
|
||||
|
||||
self.stage3_cfg = extra['STAGE3']
|
||||
num_channels = self.stage3_cfg['NUM_CHANNELS']
|
||||
block = blocks_dict[self.stage3_cfg['BLOCK']]
|
||||
num_channels = [
|
||||
num_channels[i] * block.expansion for i in range(len(num_channels))]
|
||||
self.transition2 = self._make_transition_layer(
|
||||
pre_stage_channels, num_channels)
|
||||
self.stage3, pre_stage_channels = self._make_stage(
|
||||
self.stage3_cfg, num_channels)
|
||||
|
||||
self.stage4_cfg = extra['STAGE4']
|
||||
num_channels = self.stage4_cfg['NUM_CHANNELS']
|
||||
block = blocks_dict[self.stage4_cfg['BLOCK']]
|
||||
num_channels = [
|
||||
num_channels[i] * block.expansion for i in range(len(num_channels))]
|
||||
self.transition3 = self._make_transition_layer(
|
||||
pre_stage_channels, num_channels)
|
||||
self.stage4, pre_stage_channels = self._make_stage(
|
||||
self.stage4_cfg, num_channels, multi_scale_output=True)
|
||||
|
||||
last_inp_channels = np.int(np.sum(pre_stage_channels))
|
||||
ocr_mid_channels = config.MODEL.OCR.MID_CHANNELS
|
||||
ocr_key_channels = config.MODEL.OCR.KEY_CHANNELS
|
||||
|
||||
self.conv3x3_ocr = nn.Sequential(
|
||||
nn.Conv2d(last_inp_channels, ocr_mid_channels,
|
||||
kernel_size=3, stride=1, padding=1),
|
||||
BatchNorm2d(ocr_mid_channels),
|
||||
nn.ReLU(inplace=relu_inplace),
|
||||
)
|
||||
self.ocr_gather_head = SpatialGather_Module(config.DATASET.NUM_CLASSES)
|
||||
|
||||
self.ocr_distri_head = SpatialOCR_Module(in_channels=ocr_mid_channels,
|
||||
key_channels=ocr_key_channels,
|
||||
out_channels=ocr_mid_channels,
|
||||
scale=1,
|
||||
dropout=0.05,
|
||||
)
|
||||
self.cls_head = nn.Conv2d(
|
||||
ocr_mid_channels, config.DATASET.NUM_CLASSES, kernel_size=1, stride=1, padding=0, bias=True)
|
||||
|
||||
self.aux_head = nn.Sequential(
|
||||
nn.Conv2d(last_inp_channels, last_inp_channels,
|
||||
kernel_size=1, stride=1, padding=0),
|
||||
BatchNorm2d(last_inp_channels),
|
||||
nn.ReLU(inplace=relu_inplace),
|
||||
nn.Conv2d(last_inp_channels, config.DATASET.NUM_CLASSES,
|
||||
kernel_size=1, stride=1, padding=0, bias=True)
|
||||
)
|
||||
|
||||
def _make_transition_layer(
|
||||
self, num_channels_pre_layer, num_channels_cur_layer):
|
||||
num_branches_cur = len(num_channels_cur_layer)
|
||||
num_branches_pre = len(num_channels_pre_layer)
|
||||
|
||||
transition_layers = []
|
||||
for i in range(num_branches_cur):
|
||||
if i < num_branches_pre:
|
||||
if num_channels_cur_layer[i] != num_channels_pre_layer[i]:
|
||||
transition_layers.append(nn.Sequential(
|
||||
nn.Conv2d(num_channels_pre_layer[i],
|
||||
num_channels_cur_layer[i],
|
||||
3,
|
||||
1,
|
||||
1,
|
||||
bias=False),
|
||||
BatchNorm2d(
|
||||
num_channels_cur_layer[i], momentum=BN_MOMENTUM),
|
||||
nn.ReLU(inplace=relu_inplace)))
|
||||
else:
|
||||
transition_layers.append(None)
|
||||
else:
|
||||
conv3x3s = []
|
||||
for j in range(i + 1 - num_branches_pre):
|
||||
inchannels = num_channels_pre_layer[-1]
|
||||
outchannels = num_channels_cur_layer[i] \
|
||||
if j == i - num_branches_pre else inchannels
|
||||
conv3x3s.append(nn.Sequential(
|
||||
nn.Conv2d(
|
||||
inchannels, outchannels, 3, 2, 1, bias=False),
|
||||
BatchNorm2d(outchannels, momentum=BN_MOMENTUM),
|
||||
nn.ReLU(inplace=relu_inplace)))
|
||||
transition_layers.append(nn.Sequential(*conv3x3s))
|
||||
|
||||
return nn.ModuleList(transition_layers)
|
||||
|
||||
def _make_layer(self, block, inplanes, planes, blocks, stride=1):
|
||||
downsample = None
|
||||
if stride != 1 or inplanes != planes * block.expansion:
|
||||
downsample = nn.Sequential(
|
||||
nn.Conv2d(inplanes, planes * block.expansion,
|
||||
kernel_size=1, stride=stride, bias=False),
|
||||
BatchNorm2d(planes * block.expansion, momentum=BN_MOMENTUM),
|
||||
)
|
||||
|
||||
layers = []
|
||||
layers.append(block(inplanes, planes, stride, downsample))
|
||||
inplanes = planes * block.expansion
|
||||
for i in range(1, blocks):
|
||||
layers.append(block(inplanes, planes))
|
||||
|
||||
return nn.Sequential(*layers)
|
||||
|
||||
def _make_stage(self, layer_config, num_inchannels,
|
||||
multi_scale_output=True):
|
||||
num_modules = layer_config['NUM_MODULES']
|
||||
num_branches = layer_config['NUM_BRANCHES']
|
||||
num_blocks = layer_config['NUM_BLOCKS']
|
||||
num_channels = layer_config['NUM_CHANNELS']
|
||||
block = blocks_dict[layer_config['BLOCK']]
|
||||
fuse_method = layer_config['FUSE_METHOD']
|
||||
|
||||
modules = []
|
||||
for i in range(num_modules):
|
||||
# multi_scale_output is only used last module
|
||||
if not multi_scale_output and i == num_modules - 1:
|
||||
reset_multi_scale_output = False
|
||||
else:
|
||||
reset_multi_scale_output = True
|
||||
modules.append(
|
||||
HighResolutionModule(num_branches,
|
||||
block,
|
||||
num_blocks,
|
||||
num_inchannels,
|
||||
num_channels,
|
||||
fuse_method,
|
||||
reset_multi_scale_output)
|
||||
)
|
||||
num_inchannels = modules[-1].get_num_inchannels()
|
||||
|
||||
return nn.Sequential(*modules), num_inchannels
|
||||
|
||||
def forward(self, x):
|
||||
x = self.conv1(x)
|
||||
x = self.bn1(x)
|
||||
x = self.relu(x)
|
||||
x = self.conv2(x)
|
||||
x = self.bn2(x)
|
||||
x = self.relu(x)
|
||||
x = self.layer1(x)
|
||||
|
||||
x_list = []
|
||||
for i in range(self.stage2_cfg['NUM_BRANCHES']):
|
||||
if self.transition1[i] is not None:
|
||||
x_list.append(self.transition1[i](x))
|
||||
else:
|
||||
x_list.append(x)
|
||||
y_list = self.stage2(x_list)
|
||||
|
||||
x_list = []
|
||||
for i in range(self.stage3_cfg['NUM_BRANCHES']):
|
||||
if self.transition2[i] is not None:
|
||||
if i < self.stage2_cfg['NUM_BRANCHES']:
|
||||
x_list.append(self.transition2[i](y_list[i]))
|
||||
else:
|
||||
x_list.append(self.transition2[i](y_list[-1]))
|
||||
else:
|
||||
x_list.append(y_list[i])
|
||||
y_list = self.stage3(x_list)
|
||||
|
||||
x_list = []
|
||||
for i in range(self.stage4_cfg['NUM_BRANCHES']):
|
||||
if self.transition3[i] is not None:
|
||||
if i < self.stage3_cfg['NUM_BRANCHES']:
|
||||
x_list.append(self.transition3[i](y_list[i]))
|
||||
else:
|
||||
x_list.append(self.transition3[i](y_list[-1]))
|
||||
else:
|
||||
x_list.append(y_list[i])
|
||||
x = self.stage4(x_list)
|
||||
|
||||
# Upsampling
|
||||
x0_h, x0_w = x[0].size(2), x[0].size(3)
|
||||
x1 = F.interpolate(x[1], size=(x0_h, x0_w),
|
||||
mode='bilinear', align_corners=ALIGN_CORNERS)
|
||||
x2 = F.interpolate(x[2], size=(x0_h, x0_w),
|
||||
mode='bilinear', align_corners=ALIGN_CORNERS)
|
||||
x3 = F.interpolate(x[3], size=(x0_h, x0_w),
|
||||
mode='bilinear', align_corners=ALIGN_CORNERS)
|
||||
|
||||
feats = torch.cat([x[0], x1, x2, x3], 1)
|
||||
|
||||
# out_aux_seg = []
|
||||
|
||||
# ocr
|
||||
out_aux = self.aux_head(feats)
|
||||
# compute contrast feature
|
||||
feats = self.conv3x3_ocr(feats)
|
||||
|
||||
context = self.ocr_gather_head(feats, out_aux)
|
||||
feats = self.ocr_distri_head(feats, context)
|
||||
|
||||
out = self.cls_head(feats)
|
||||
|
||||
out = F.interpolate(out, size=(out.size(2) * 4, out.size(3) * 4),
|
||||
mode='bilinear', align_corners=ALIGN_CORNERS)
|
||||
|
||||
return out
|
||||
|
||||
def init_weights(self, pretrained='', ):
|
||||
logger.info('=> init weights from normal distribution')
|
||||
for name, m in self.named_modules():
|
||||
if any(part in name for part in {'cls', 'aux', 'ocr'}):
|
||||
# print('skipped', name)
|
||||
continue
|
||||
if isinstance(m, nn.Conv2d):
|
||||
nn.init.normal_(m.weight, std=0.001)
|
||||
elif isinstance(m, BatchNorm2d_class):
|
||||
nn.init.constant_(m.weight, 1)
|
||||
nn.init.constant_(m.bias, 0)
|
||||
if os.path.isfile(pretrained):
|
||||
pretrained_dict = torch.load(pretrained, map_location={'cuda:0': 'cpu'})
|
||||
logger.info('=> loading pretrained model {}'.format(pretrained))
|
||||
model_dict = self.state_dict()
|
||||
pretrained_dict = {k.replace('last_layer', 'aux_head').replace('model.', ''): v for k, v in
|
||||
pretrained_dict.items()}
|
||||
print(set(model_dict) - set(pretrained_dict))
|
||||
print(set(pretrained_dict) - set(model_dict))
|
||||
pretrained_dict = {k: v for k, v in pretrained_dict.items()
|
||||
if k in model_dict.keys() and not k.startswith('conv1')}
|
||||
# for k, _ in pretrained_dict.items():
|
||||
# logger.info(
|
||||
# '=> loading {} pretrained model {}'.format(k, pretrained))
|
||||
model_dict.update(pretrained_dict)
|
||||
self.load_state_dict(model_dict)
|
||||
elif pretrained:
|
||||
raise RuntimeError('No such file {}'.format(pretrained))
|
||||
|
||||
|
||||
def get_seg_model(cfg, **kwargs):
|
||||
model = HighResolutionNet(cfg, **kwargs)
|
||||
# model.init_weights(cfg.MODEL.PRETRAINED)
|
||||
|
||||
return model
|
||||
@@ -0,0 +1,233 @@
|
||||
import argparse
|
||||
|
||||
from models.experimental import *
|
||||
|
||||
|
||||
class Detect(nn.Module):
|
||||
def __init__(self, nc=80, anchors=()): # detection layer
|
||||
super(Detect, self).__init__()
|
||||
self.stride = None # strides computed during build
|
||||
self.nc = nc # number of classes
|
||||
self.no = nc + 5 # number of outputs per anchor
|
||||
self.nl = len(anchors) # number of detection layers
|
||||
self.na = len(anchors[0]) // 2 # number of anchors
|
||||
self.grid = [torch.zeros(1)] * self.nl # init grid
|
||||
a = torch.tensor(anchors).float().view(self.nl, -1, 2)
|
||||
self.register_buffer('anchors', a) # shape(nl,na,2)
|
||||
self.register_buffer('anchor_grid', a.clone().view(self.nl, 1, -1, 1, 1, 2)) # shape(nl,1,na,1,1,2)
|
||||
self.export = False # onnx export
|
||||
|
||||
def forward(self, x):
|
||||
# x = x.copy() # for profiling
|
||||
z = [] # inference output
|
||||
self.training |= self.export
|
||||
for i in range(self.nl):
|
||||
bs, _, ny, nx = x[i].shape # x(bs,255,20,20) to x(bs,3,20,20,85)
|
||||
x[i] = x[i].view(bs, self.na, self.no, ny, nx).permute(0, 1, 3, 4, 2).contiguous()
|
||||
|
||||
if not self.training: # inference
|
||||
if self.grid[i].shape[2:4] != x[i].shape[2:4]:
|
||||
self.grid[i] = self._make_grid(nx, ny).to(x[i].device)
|
||||
|
||||
y = x[i].sigmoid()
|
||||
y[..., 0:2] = (y[..., 0:2] * 2. - 0.5 + self.grid[i].to(x[i].device)) * self.stride[i] # xy
|
||||
y[..., 2:4] = (y[..., 2:4] * 2) ** 2 * self.anchor_grid[i] # wh
|
||||
z.append(y.view(bs, -1, self.no))
|
||||
|
||||
return x if self.training else (torch.cat(z, 1), x)
|
||||
|
||||
@staticmethod
|
||||
def _make_grid(nx=20, ny=20):
|
||||
yv, xv = torch.meshgrid([torch.arange(ny), torch.arange(nx)])
|
||||
return torch.stack((xv, yv), 2).view((1, 1, ny, nx, 2)).float()
|
||||
|
||||
|
||||
class Model(nn.Module):
|
||||
def __init__(self, model_cfg='yolov5s.yaml', ch=3, nc=None): # model, input channels, number of classes
|
||||
super(Model, self).__init__()
|
||||
if type(model_cfg) is dict:
|
||||
self.md = model_cfg # model dict
|
||||
else: # is *.yaml
|
||||
with open(model_cfg) as f:
|
||||
self.md = yaml.load(f, Loader=yaml.FullLoader) # model dict
|
||||
|
||||
# Define model
|
||||
if nc:
|
||||
self.md['nc'] = nc # override yaml value
|
||||
self.model, self.save = parse_model(self.md, ch=[ch]) # model, savelist, ch_out
|
||||
# print([x.shape for x in self.forward(torch.zeros(1, ch, 64, 64))])
|
||||
|
||||
# Build strides, anchors
|
||||
m = self.model[-1] # Detect()
|
||||
m.stride = torch.tensor([128 / x.shape[-2] for x in self.forward(torch.zeros(1, ch, 128, 128))]) # forward
|
||||
m.anchors /= m.stride.view(-1, 1, 1)
|
||||
check_anchor_order(m)
|
||||
self.stride = m.stride
|
||||
|
||||
# Init weights, biases
|
||||
torch_utils.initialize_weights(self)
|
||||
self._initialize_biases() # only run once
|
||||
torch_utils.model_info(self)
|
||||
print('')
|
||||
|
||||
def forward(self, x, augment=False, profile=False):
|
||||
if augment:
|
||||
img_size = x.shape[-2:] # height, width
|
||||
s = [0.83, 0.67] # scales
|
||||
y = []
|
||||
for i, xi in enumerate((x,
|
||||
torch_utils.scale_img(x.flip(3), s[0]), # flip-lr and scale
|
||||
torch_utils.scale_img(x, s[1]), # scale
|
||||
)):
|
||||
# cv2.imwrite('img%g.jpg' % i, 255 * xi[0].numpy().transpose((1, 2, 0))[:, :, ::-1])
|
||||
y.append(self.forward_once(xi)[0])
|
||||
|
||||
y[1][..., :4] /= s[0] # scale
|
||||
y[1][..., 0] = img_size[1] - y[1][..., 0] # flip lr
|
||||
y[2][..., :4] /= s[1] # scale
|
||||
return torch.cat(y, 1), None # augmented inference, train
|
||||
else:
|
||||
return self.forward_once(x, profile) # single-scale inference, train
|
||||
|
||||
def forward_once(self, x, profile=False):
|
||||
y, dt = [], [] # outputs
|
||||
for m in self.model:
|
||||
if m.f != -1: # if not from previous layer
|
||||
x = y[m.f] if isinstance(m.f, int) else [x if j == -1 else y[j] for j in m.f] # from earlier layers
|
||||
|
||||
if profile:
|
||||
try:
|
||||
import thop
|
||||
o = thop.profile(m, inputs=(x,), verbose=False)[0] / 1E9 * 2 # FLOPS
|
||||
except:
|
||||
o = 0
|
||||
t = torch_utils.time_synchronized()
|
||||
for _ in range(10):
|
||||
_ = m(x)
|
||||
dt.append((torch_utils.time_synchronized() - t) * 100)
|
||||
print('%10.1f%10.0f%10.1fms %-40s' % (o, m.np, dt[-1], m.type))
|
||||
|
||||
x = m(x) # run
|
||||
y.append(x if m.i in self.save else None) # save output
|
||||
|
||||
if profile:
|
||||
print('%.1fms total' % sum(dt))
|
||||
return x
|
||||
|
||||
def _initialize_biases(self, cf=None): # initialize biases into Detect(), cf is class frequency
|
||||
# cf = torch.bincount(torch.tensor(np.concatenate(dataset.labels, 0)[:, 0]).long(), minlength=nc) + 1.
|
||||
m = self.model[-1] # Detect() module
|
||||
for f, s in zip(m.f, m.stride): # from
|
||||
mi = self.model[f % m.i]
|
||||
b = mi.bias.view(m.na, -1) # conv.bias(255) to (3,85)
|
||||
b[:, 4] += math.log(8 / (640 / s) ** 2) # obj (8 objects per 640 image)
|
||||
b[:, 5:] += math.log(0.6 / (m.nc - 0.99)) if cf is None else torch.log(cf / cf.sum()) # cls
|
||||
mi.bias = torch.nn.Parameter(b.view(-1), requires_grad=True)
|
||||
|
||||
def _print_biases(self):
|
||||
m = self.model[-1] # Detect() module
|
||||
for f in sorted([x % m.i for x in m.f]): # from
|
||||
b = self.model[f].bias.detach().view(m.na, -1).T # conv.bias(255) to (3,85)
|
||||
print(('%g Conv2d.bias:' + '%10.3g' * 6) % (f, *b[:5].mean(1).tolist(), b[5:].mean()))
|
||||
|
||||
# def _print_weights(self):
|
||||
# for m in self.model.modules():
|
||||
# if type(m) is Bottleneck:
|
||||
# print('%10.3g' % (m.w.detach().sigmoid() * 2)) # shortcut weights
|
||||
|
||||
def fuse(self): # fuse model Conv2d() + BatchNorm2d() layers
|
||||
print('Fusing layers...')
|
||||
for m in self.model.modules():
|
||||
if type(m) is Conv:
|
||||
m.conv = torch_utils.fuse_conv_and_bn(m.conv, m.bn) # update conv
|
||||
m.bn = None # remove batchnorm
|
||||
m.forward = m.fuseforward # update forward
|
||||
torch_utils.model_info(self)
|
||||
|
||||
|
||||
def parse_model(md, ch): # model_dict, input_channels(3)
|
||||
print('\n%3s%15s%3s%10s %-40s%-30s' % ('', 'from', 'n', 'params', 'module', 'arguments'))
|
||||
anchors, nc, gd, gw = md['anchors'], md['nc'], md['depth_multiple'], md['width_multiple']
|
||||
na = (len(anchors[0]) // 2) # number of anchors
|
||||
no = na * (nc + 5) # number of outputs = anchors * (classes + 5)
|
||||
|
||||
layers, save, c2 = [], [], ch[-1] # layers, savelist, ch out
|
||||
for i, (f, n, m, args) in enumerate(md['backbone'] + md['head']): # from, number, module, args
|
||||
m = eval(m) if isinstance(m, str) else m # eval strings
|
||||
for j, a in enumerate(args):
|
||||
try:
|
||||
args[j] = eval(a) if isinstance(a, str) else a # eval strings
|
||||
except:
|
||||
pass
|
||||
|
||||
n = max(round(n * gd), 1) if n > 1 else n # depth gain
|
||||
if m in [nn.Conv2d, Conv, Bottleneck, SPP, DWConv, MixConv2d, Focus, ConvPlus, BottleneckCSP]:
|
||||
c1, c2 = ch[f], args[0]
|
||||
|
||||
# Normal
|
||||
# if i > 0 and args[0] != no: # channel expansion factor
|
||||
# ex = 1.75 # exponential (default 2.0)
|
||||
# e = math.log(c2 / ch[1]) / math.log(2)
|
||||
# c2 = int(ch[1] * ex ** e)
|
||||
# if m != Focus:
|
||||
c2 = make_divisible(c2 * gw, 8) if c2 != no else c2
|
||||
|
||||
# Experimental
|
||||
# if i > 0 and args[0] != no: # channel expansion factor
|
||||
# ex = 1 + gw # exponential (default 2.0)
|
||||
# ch1 = 32 # ch[1]
|
||||
# e = math.log(c2 / ch1) / math.log(2) # level 1-n
|
||||
# c2 = int(ch1 * ex ** e)
|
||||
# if m != Focus:
|
||||
# c2 = make_divisible(c2, 8) if c2 != no else c2
|
||||
|
||||
args = [c1, c2, *args[1:]]
|
||||
if m is BottleneckCSP:
|
||||
args.insert(2, n)
|
||||
n = 1
|
||||
elif m is nn.BatchNorm2d:
|
||||
args = [ch[f]]
|
||||
elif m is Concat:
|
||||
c2 = sum([ch[-1 if x == -1 else x + 1] for x in f])
|
||||
elif m is Detect:
|
||||
f = f or list(reversed([(-1 if j == i else j - 1) for j, x in enumerate(ch) if x == no]))
|
||||
else:
|
||||
c2 = ch[f]
|
||||
|
||||
m_ = nn.Sequential(*[m(*args) for _ in range(n)]) if n > 1 else m(*args) # module
|
||||
t = str(m)[8:-2].replace('__main__.', '') # module type
|
||||
np = sum([x.numel() for x in m_.parameters()]) # number params
|
||||
m_.i, m_.f, m_.type, m_.np = i, f, t, np # attach index, 'from' index, type, number params
|
||||
print('%3s%15s%3s%10.0f %-40s%-30s' % (i, f, n, np, t, args)) # print
|
||||
save.extend(x % i for x in ([f] if isinstance(f, int) else f) if x != -1) # append to savelist
|
||||
layers.append(m_)
|
||||
ch.append(c2)
|
||||
return nn.Sequential(*layers), sorted(save)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument('--cfg', type=str, default='yolov5s.yaml', help='model.yaml')
|
||||
parser.add_argument('--device', default='', help='cuda device, i.e. 0 or 0,1,2,3 or cpu')
|
||||
opt = parser.parse_args()
|
||||
opt.cfg = check_file(opt.cfg) # check file
|
||||
device = torch_utils.select_device(opt.device)
|
||||
|
||||
# Create model
|
||||
model = Model(opt.cfg).to(device)
|
||||
model.train()
|
||||
|
||||
# Profile
|
||||
# img = torch.rand(8 if torch.cuda.is_available() else 1, 3, 640, 640).to(device)
|
||||
# y = model(img, profile=True)
|
||||
|
||||
# ONNX export
|
||||
# model.model[-1].export = True
|
||||
# torch.onnx.export(model, img, opt.cfg.replace('.yaml', '.onnx'), verbose=True, opset_version=11)
|
||||
|
||||
# Tensorboard
|
||||
# from torch.utils.tensorboard import SummaryWriter
|
||||
# tb_writer = SummaryWriter()
|
||||
# print("Run 'tensorboard --logdir=models/runs' to view tensorboard at http://localhost:6006/")
|
||||
# tb_writer.add_graph(model.model, img) # add model to tensorboard
|
||||
# tb_writer.add_image('test', img[0], dataformats='CWH') # add model to tensorboard
|
||||
Reference in New Issue
Block a user