SSD_vgg_300网络
import tensorflow as tf
import numpy as np
import cv2
slim = tf.contrib.slim
class ssd(object):
def __init__(self):
self.num_boxes = []
self.feature_map_size = [(38, 38), (19, 19), (10, 10), (5, 5), (3, 3), (1, 1)] # 特征图大小
self.classes = ['aeroplane', 'bicycle', 'bird', 'boat', 'bottle', 'bus', 'car', 'cat', 'chair', 'cow',
'diningtable', 'dog', 'horse', 'motorbike', 'person', 'pottedplant', 'sheep', 'spfa', 'train', 'tvmonitor']
self.feature_layers = ['block4', 'block7', 'block8', 'block9', 'block10', 'block11']
self.img_size = (300, 300)
self.num_classes = 21
self.boxes_len = [4, 6, 6, 6, 4, 4] # 锚框个数
self.isL2norm = [True, False, False, False, False, False]
self.anchor_size = [(21., 45.), (45., 99.), (99., 153.), (153., 207.), (207., 261), (261.,315.)] # 最后一个有问题
self.anchor_ratios = [[2, .5], [2, .5, 3, 1./3], [2, .5, 3, 1./3], [2, .5, 3, 1./3], [2, .5], [2, .5]]
self.anchor_steps = [8, 16, 32, 64, 100, 300]
self.prior_scaling = [0.1, 0.1, 0.2, 0.2] # 特征图先验框缩放比例 (x, y, w, h)
self.n_boxes = [5776, 2166, 600, 150, 36, 4] # 8732个
self.threshold = 0.2
def l2norm(self, x, trainable=True, scope='L2Normlization'):
n_channels = x.get_shape().as_list()[-1] # 通道数
l2_norm = tf.nn.l2_normalize(x, dim=[3], epsilon=1e-12) # 只对每个像素点在channels上做归一化
with tf.variable_scope(scope):
gamma = tf.get_variable('gamma', shape=[n_channels, ], dtype=tf.float32,
trainable=trainable)
return l2_norm * gamma
def ssd_prediction(self, x, num_classes, box_len, isL2norm, scope='multibox'):
reshape = [-1] + x.get_shape().as_list()[1:-1] # 得到维度并且形成列表 取第一个到最后一个 [-1, w, h, c]
with tf.variable_scope(scope):
if isL2norm:
x = self.l2norm(x)
# 预测位置 --> 坐标和大小 回归问题
location_pred = self.conv2d(x, filter=box_len*4, k_size=[3, 3], activation=None, scope='conv_loc')
location_pred = tf.reshape(location_pred, reshape+[box_len, 4])
# 预测类别 --> 分类 softmax
class_pred = self.conv2d(x, filter=box_len*num_classes, k_size=[3, 3], activation=None, scope='conv_cls')
class_pred = tf.reshape(class_pred, reshape+[box_len, num_classes])
print(location_pred, class_pred)
return location_pred, class_pred
def conv2d(self, x, filter, k_size, stride=[1, 1], padding='same', dilatio=1, activation=tf.nn.relu, scope='conv2d'):
return tf.layers.conv2d(inputs=x, filters=filter, kernel_size=k_size, strides=stride, padding=padding, dilation_rate=dilatio, name=scope, activation=activation)
def max_pool2d(self, x, pool_size, stride, scope='max_pool2d'):
return tf.layers.max_pooling2d(inputs=x, pool_size=pool_size, strides=stride, name=scope, padding='same')
def pad2d(self, x, pad):
return tf.pad(x, paddings=[[0, 0], [pad, pad], [pad, pad], [0, 0]])
def dropout(self, x, d_rate=0.5):
return tf.layers.dropout(inputs=x, rate=d_rate)
def set_net(self):
check_points = {}
predictions = []
locations = []
x = tf.placeholder(dtype=tf.float32, shape=[None, 300, 300, 3])
with tf.variable_scope('ssd_300_vgg'):
# b1
net = self.conv2d(x, 64, [3, 3], scope='conv1_1')
net = self.conv2d(net, 64, [3, 3], scope='conv1_2')
print(net)
net = self.max_pool2d(net, pool_size=[2, 2], stride=[2, 2], scope='pool1')
# b2
net = self.conv2d(net, 128, [3, 3], scope='conv2_1')
net = self.conv2d(net, 128, [3, 3], scope='conv2_2')
print(net)
net = self.max_pool2d(net, pool_size=[2, 2], stride=[2, 2], scope='pool2')
# b3
net = self.conv2d(net, 256, [3, 3], scope='conv3_1')
net = self.conv2d(net, 256, [3, 3], scope='conv3_2')
net = self.conv2d(net, 256, [3, 3], scope='conv3_3')
print(net)
net = self.max_pool2d(net, pool_size=[2, 2], stride=[2, 2], scope='pool3')
# b4
net = self.conv2d(net, 512, [3, 3], scope='conv4_1')
net = self.conv2d(net, 512, [3, 3], scope='conv4_2')
net = self.conv2d(net, 512, [3, 3], scope='conv4_3')
print(net)
check_points['block4'] = net
net = self.max_pool2d(net, pool_size=[2, 2], stride=[2, 2], scope='pool4')
# b5
net = self.conv2d(net, 512, [3, 3], scope='conv5_1')
net = self.conv2d(net, 512, [3, 3], scope='conv5_2')
net = self.conv2d(net, 512, [3, 3], scope='conv5_3')
print(net)
net = self.max_pool2d(net, pool_size=[3, 3], stride=[1, 1], scope='pool5')
# b6
net = self.conv2d(net, 1024, [3, 3], dilatio=[6, 6], scope='conv6')
print(net)
# b7
net = self.conv2d(net, 1024, [3, 3], scope='conv7')
check_points['block7'] = net
print(net)
# b8
net = self.conv2d(net, 256, [1, 1], scope='conv8_1x1')
net = self.conv2d(self.pad2d(net, 1), 512, [3, 3], stride=[2, 2], scope='conv8_3x3', padding='valid')
check_points['block8'] = net
print(net)
# b9
net = self.conv2d(net, 128, [1, 1], scope='conv9_1x1')
net = self.conv2d(self.pad2d(net, 1), 256, [3, 3], stride=[2, 2], scope='conv9_3x3', padding='valid')
check_points['block9'] = net
print(net)
# b10
net = self.conv2d(net, 128, [1, 1], scope='conv10_1x1')
net = self.conv2d(net, 256, [3, 3], scope='conv10_3x3', padding='valid')
check_points['block10'] = net
print(net)
# b11
net = self.conv2d(net, 128, [1, 1], scope='conv11_1x1')
net = self.conv2d(net, 256, [3, 3], scope='conv11_3x3', padding='valid')
check_points['block11'] = net
print(net)
#---------------------------------------------------------------------------------------------------------------------------
# # b1
# net = slim.repeat(x, 2, slim.conv2d, 64, [3, 3], scope='conv1')
# print(net)
# net = slim.max_pool2d(net, [2, 2], scope='pool1')
# # b2
# net = slim.repeat(net, 2, slim.conv2d, 128, [3, 3], scope='conv2')
# print(net)
# net = slim.max_pool2d(net, [2, 2], scope='pool2')
# # b3.
# net = slim.repeat(net, 3, slim.conv2d, 256, [3, 3], scope='conv3')
# print(net)
# net = slim.max_pool2d(net, [2, 2], scope='pool3')
# # b4.
# net = slim.repeat(net, 3, slim.conv2d, 512, [3, 3], scope='conv4')
# print(net)
# check_points['block4'] = net
# net = slim.max_pool2d(net, [2, 2], scope='pool4')
# # b5.
# net = slim.repeat(net, 3, slim.conv2d, 512, [3, 3], scope='conv5')
# print(net)
# net = slim.max_pool2d(net, [3, 3], stride=1, scope='pool5')
# # b6
# net = slim.conv2d(net, 1024, [3, 3], rate=6, scope='conv6')
# print(net)
# # b7
# net = slim.conv2d(net, 1024, [3, 3], scope='conv7')
# print(net)
# check_points['block7'] = net
# # b8
# end_point = 'block8'
# with tf.variable_scope(end_point):
# net = slim.conv2d(net, 256, [1, 1], scope='conv1x1')
# net = self.pad2d(net, 1)
# net = slim.conv2d(net, 512, [3, 3], stride=2, scope='conv3x3', padding='VALID')
# check_points[end_point] = net
# print(net)
# # b9
# end_point = 'block9'
# with tf.variable_scope(end_point):
# net = slim.conv2d(net, 128, [1, 1], scope='conv1x1')
# net = self.pad2d(net, 1)
# net = slim.conv2d(net, 256, [3, 3], stride=2, scope='conv3x3', padding='VALID')
# check_points[end_point] = net
# print(net)
# # b10
# end_point = 'block10'
# with tf.variable_scope(end_point):
# net = slim.conv2d(net, 128, [1, 1], scope='conv1x1')
# net = slim.conv2d(net, 256, [3, 3], scope='conv3x3', padding='VALID')
# check_points[end_point] = net
# print(net)
# # b11
# end_point = 'block11'
# with tf.variable_scope(end_point):
# net = slim.conv2d(net, 128, [1, 1], scope='conv1x1')
# net = slim.conv2d(net, 256, [3, 3], scope='conv3x3', padding='VALID')
# check_points[end_point] = net
# print(net)
for i, j in enumerate(self.feature_layers):
loc, cls = self.ssd_prediction(check_points[j],
num_classes=self.num_classes,
box_len=self.boxes_len[i],
isL2norm=self.isL2norm[i],
scope = j + '_box')
predictions.append(tf.nn.softmax(cls)) # 类别预测需要经过softmax层
locations.append(loc)
return locations, predictions, x
# -------------------------SSD网络框架部分结束——----------------------------
# 先验框生成
def ssd_anchor_layer(self, img_size, feature_map_size, anchor_size, anchor_ratio, anchor_step, box_num, offset=0.5):
# 提取feature map的每个坐标
y, x = np.mgrid[0: feature_map_size[0], 0: feature_map_size[1]]
y = (y + offset) * anchor_step / img_size[0]
x = (x + offset) * anchor_step / img_size[1]
# 计算俩个长宽比为1的 h, w
h = np.zeros((box_num, ), np.float32)
w = np.zeros((box_num, ), np.float32)
h[0] = anchor_size[0] / img_size[0]
w[0] = anchor_size[0] / img_size[0]
h[1] = (anchor_size[0] * anchor_size[1]) ** 0.5 / img_size[0]
w[1] = (anchor_size[0] * anchor_size[1]) ** 0.5 / img_size[0]
for i, j in enumerate(anchor_ratio):
h[i + 2] = anchor_size[0] / img_size[0] / (j ** .5)
w[i + 2] = anchor_size[0] / img_size[0] * (j ** .5)
return np.expand_dims(x, axis=-1), np.expand_dims(y, axis=-1), h, w
# 解码网络 解码先验框
def ssd_decode(self, location, box, prior_scaling):
x, y, w, h = box
cx = location[:, :, :, :, 0] * x * prior_scaling[0] + x
cy = location[:, :, :, :, 1] * y * prior_scaling[1] + y
cw = w * tf.exp(location[:, :, :, :, 2] * prior_scaling[2])
ch = w * tf.exp(location[:, :, :, :, 3] * prior_scaling[3])
boxes = tf.stack([cy-ch/2, cx-cw/2, cy+ch/2, cx+cw/2], axis=-1)
return boxes
# 先验框筛选
def choose_anchor_boxes(self, predictions, anchor_box, n_box):
anchor_box = tf.reshape(anchor_box, [n_box, 4])
predictions = tf.reshape(predictions, [n_box, 21])[:, 1:]
classes = tf.argmax(predictions, axis=1) + 1 # 得到最大类别的索引
scores = tf.reduce_max(predictions, axis=1) # 得到最大类别的分数
filter_mask = scores > self.threshold
classes = tf.boolean_mask(classes, filter_mask) # 将分数超过0.2阈值的类别都保留下来
scores = tf.boolean_mask(scores, filter_mask)
anchor_box = tf.boolean_mask(anchor_box, filter_mask)
return classes, scores, anchor_box
#-----------------------------------先验框部分结束----------------------------------------------------
# 训练部分
# 对框进行排序,取score值高的前四百个框
def bboxes_sort(self, classes, scores, bboxes, top_k=400):
idxes = np.argsort(-scores)
classes = classes[idxes][:top_k]
scores = scores[idxes][:top_k]
bboxes = bboxes[idxes][:top_k]
return classes, scores, bboxes
# 计算IoU
def bboxes_iou(self, bboxes1, bboxes2):
bboxes1 = np.transpose(bboxes1)
bboxes2 = np.transpose(bboxes2)
# 计算俩个box的交集:交集左上角的点取俩个box的max,交集右下角的点取俩个box的min
int_ymin = np.maximum(bboxes1[0], bboxes2[0])
int_xmin = np.maximum(bboxes1[1], bboxes2[1])
int_ymax = np.minimum(bboxes1[2], bboxes2[2])
int_xmax = np.minimum(bboxes1[3], bboxes2[3])
# 计算俩个交集的w, h
int_h = np.maximum(int_ymax - int_ymin, 0.)
int_w = np.maximum(int_xmax - int_xmin, 0.)
# 计算IoU
int_vol = int_h * int_w # 交集面积
vol1 = (bboxes1[2] - bboxes1[0]) * (bboxes1[3] - bboxes1[1]) # bboxes1的面积
vol2 = (bboxes2[2] - bboxes2[0]) * (bboxes2[3] - bboxes2[1]) # bboxes2的面积
iou = int_vol / (vol1 + vol2) # 交并比
return iou
# NMS
def bboxes_nms(self, classes, scores, bboxes, nms_threshold=0.5):
keep_bboxes = np.ones(scores.shape, dtype=np.bool)
for i in range(scores.size - 1):
if keep_bboxes[i]:
overlap = self.bboxes_iou(bboxes[i], bboxes[(i+1): ])
keep_overlap = np.logical_or(overlap < nms_threshold, classes[(i+1): ] != classes[i])
keep_bboxes[(i+1): ] = np.logical_and(keep_bboxes[(i+1): ], keep_overlap)
idxes = np.where(keep_bboxes)
return classes[idxes], scores[idxes], bboxes[idxes]
# ------------------------------------训练部分结束---------------------------------------------------------------
# 图像均值化
def handle_img(self, img_path):
means = np.array((123., 117., 104.))
self.img = cv2.imread(img_path)
img = np.expand_dims(cv2.resize(cv2.cvtColor(self.img, cv2.COLOR_BGR2RGB) - means, self.img_size), axis=0)
return img
# 画图
def draw_rectangle(self, img, classes, scores, bboxes, colors, thickness=2):
shape = img.shape
for i in range(bboxes.shape[0]):
bbox = bboxes[i]
p1 = (int(bbox[0] * shape[0]), int(bbox[1] * shape[1]))
p2 = (int(bbox[2] * shape[0]), int(bbox[3] * shape[1]))
cv2.rectangle(img, p1[::-1], p2[::-1], colors[0], thickness)
# Draw text...
s = '%s/%.3f'%(self.classes[classes[i] - 1], scores[i])
p1 = (p1[0]-5, p1[1])
cv2.putText(img, s, p1[::-1], cv2.FONT_HERSHEY_DUPLEX, 0.5, colors[i], 1)
cv2.namedWindow('img', 0)
cv2.resizeWindow('img', 640, 480)
cv2.imshow('img', img)
cv2.waitKey(0)
cv2.destroyAllWindows()
def run_this(self, locations, predictions):
layers_anchors = []
classes_list = []
scores_list = []
bboxes_list = []
for i, s in enumerate(self.feature_map_size):
anchor_bboxes = self.ssd_anchor_layer(self.img_size, s,
self.anchor_size[i],
self.anchor_ratios[i],
self.anchor_steps[i],
self.boxes_len[i])
layers_anchors.append(anchor_bboxes)
for i in range(len(predictions)):
d_box = self.ssd_decode(locations[i], layers_anchors[i], self.prior_scaling)
cls, sco, box = self.choose_anchor_boxes(predictions[i], d_box, self.n_boxes[i])
classes_list.append(cls)
scores_list.append(sco)
bboxes_list.append(box)
classes = tf.concat(classes_list, axis=0)
scores = tf.concat(scores_list, axis=0)
bboxes = tf.concat(bboxes_list, axis=0)
return classes, scores, bboxes
if __name__ == '__main__':
sd = ssd()
locations, predictions, x = sd.set_net()
classes, scores, bboxes = sd.run_this(locations, predictions)
sess = tf.Session()
# ckpt_filename = '../checkpoint/ssd_300_vgg.ckpt.index'
sess.run(tf.global_variables_initializer())
saver = tf.train.Saver()
saver.restore(sess, '../checkpoint/')
img = sd.handle_img('0.jpg')
rclasses, rscores, rbboxes = sess.run([classes, scores, bboxes], feed_dict={x: img})
rclasses, rscores, rbboxes = sd.bboxes_sort(rclasses, rscores, rbboxes)
rclasses, rscores, rbboxes = sd.bboxes_nms(rclasses, rscores, rbboxes)
sd.draw_rectangle(sd.img, rclasses, rscores, rbboxes, [[0, 0, 255], [255, 0, 0]])
输出结果:

posted on 2020-09-26 17:57 WhitePinkJ 阅读(476) 评论(0) 收藏 举报
浙公网安备 33010602011771号