本文所使用的開源數(shù)據(jù)集(kaggle貓狗大戰(zhàn)):
https://www.kaggle.com/c/dogs-vs-cats
國內(nèi)百度網(wǎng)盤下載地址:
https://pan.baidu.com/s/12ab32UNYI_6o4i9HyX-n8Q
利用本文代碼訓(xùn)練并生成的模型(對應(yīng)項(xiàng)目中的model文件夾):
https://pan.baidu.com/s/1tBkVQKoH2Fc_HsBxtdAxnQ
簡單介紹:
(需要預(yù)先安裝pip install opencv-python, pip install flask, pip install tensorflow/pip install tensorflow-gpu)
本文使用Python3娃善,TensorFlow實(shí)現(xiàn)適合新手的VGG16模型(不了解VGG16的同學(xué)可以自行百度一下,本文沒有使用slim或者keras實(shí)現(xiàn)误债,對VGG16逐層實(shí)現(xiàn),便于新手理解,有經(jīng)驗(yàn)的同學(xué)可以用高級庫重寫這部分)可應(yīng)用于單標(biāo)簽分類(一張圖片要么是貓遣钳,要么是狗)任務(wù)饱岸。
(預(yù)告:之后會寫一篇多標(biāo)簽分類任務(wù)驻龟,與單標(biāo)簽分類有些區(qū)別http://www.reibang.com/p/596db72a7e00)
整體訓(xùn)練邏輯:
0撤摸,使用pipeline方式異步讀取訓(xùn)練集圖片,節(jié)省內(nèi)存消耗褒纲,提高效率
1准夷,將圖像傳入到CNN(VGG16)中提取特征
2,將特征圖拉伸輸入到FC layer中得出分類預(yù)測向量
3莺掠,通過softmax交叉熵函數(shù)對預(yù)測向量和標(biāo)簽向量進(jìn)行訓(xùn)練衫嵌,得出最終模型
整體預(yù)測邏輯:
1,將圖像傳入到CNN(VGG16)中提取特征
2彻秆,將特征圖拉伸輸入到FC layer中得出分類預(yù)測向量
3楔绞,將預(yù)測向量做softmax操作,取向量中的最大值唇兑,并映射到對應(yīng)類別中
制作成web服務(wù):
利用flask框架將整個(gè)項(xiàng)目啟動(dòng)成web服務(wù)酒朵,使得項(xiàng)目支持http方式調(diào)用
啟動(dòng)服務(wù)后調(diào)用以下地址測試
http://127.0.0.1:5050/dogOrCat?img_path=./data/test1/1.jpg
http://127.0.0.1:5050/dogOrCat?img_path=./data/test1/5.jpg
后續(xù)優(yōu)化邏輯:
可以采用遷移學(xué)習(xí),模型融合等方案進(jìn)一步提高acc
可以左右翻轉(zhuǎn)圖片扎附,將訓(xùn)練集翻倍
運(yùn)行命令:
對數(shù)據(jù)集進(jìn)行訓(xùn)練:python DogVsCat.py train
對新的圖片進(jìn)行測試:python DogVsCat.py test
啟動(dòng)成http服務(wù):python DogVsCat.py start
項(xiàng)目整體目錄結(jié)構(gòu):
model結(jié)構(gòu):
訓(xùn)練過程:
整體代碼如下:
# coding:utf-8
import tensorflow as tf
import os, sys, random
import numpy as np
import cv2
from flask import request
from flask import Flask
import json
app = Flask(__name__)
class DogVsCat:
def __init__(self):
# 可調(diào)參數(shù)
self.save_epoch = 1 # 每相隔多少個(gè)epoch保存一次模型
self.train_max_num = 25000 # 訓(xùn)練時(shí)讀取的最大圖片數(shù)目 0~25000之間蔫耽,內(nèi)存不足的可以調(diào)小
self.epoch_max = 13 # 最大迭代epoch次數(shù)
self.batch_size = 16 # 訓(xùn)練時(shí)每個(gè)批次參與訓(xùn)練的圖像數(shù)目,顯存不足的可以調(diào)小
self.class_num = 2 # 分類數(shù)目留夜,貓狗共兩類
self.val_num = 20 * self.batch_size # 不能大于self.train_max_num 做驗(yàn)證集用
self.lr = 1e-4 # 初始學(xué)習(xí)率
# 無需修改參數(shù)
self.x_val = []
self.y_val = []
self.x = None # 每批次的圖像數(shù)據(jù)
self.y = None # 每批次的one-hot標(biāo)簽
self.learning_rate = None # 學(xué)習(xí)率
self.sess = None # 持久化的tf.session
self.pred = None # cnn網(wǎng)絡(luò)結(jié)構(gòu)的預(yù)測
self.keep_drop = tf.placeholder(tf.float32) # dropout比例
def dogOrCat(self, img_path):
"""
貓狗分類
:param img_path:
:return:
"""
im = cv2.imread(img_path)
im = cv2.resize(im, (224, 224))
im = [im]
im = np.array(im, dtype=np.float32)
im -= 147
output = self.sess.run(self.output, feed_dict={self.x: im, self.keep_drop: 1.})
ret = output.tolist()[0]
ret = 'It is a cat' if ret[0] <= ret[1] else 'It is a dog'
return ret
def test(self, img_path):
"""
測試接口
:param img_path:
:return:
"""
self.x = tf.placeholder(tf.float32, [None, 224, 224, 3]) # 輸入數(shù)據(jù)
self.pred = self.CNN()
self.output = tf.nn.softmax(self.pred)
saver = tf.train.Saver()
# tfconfig = tf.ConfigProto(allow_soft_placement=True)
# tfconfig.gpu_options.per_process_gpu_memory_fraction = 0.3 # 占用顯存的比例
# self.ses = tf.Session(config=tfconfig)
self.sess = tf.Session()
self.sess.run(tf.global_variables_initializer()) # 全局tf變量初始化
# 加載w,b參數(shù)
saver.restore(self.sess, './model/DogVsCat-13')
im = cv2.imread(img_path)
im = cv2.resize(im, (224, 224))
im = [im]
im = np.array(im, dtype=np.float32)
im -= 147
output = self.sess.run(self.output, feed_dict={self.x: im, self.keep_drop: 1.})
ret = output.tolist()[0]
ret = 'It is a cat' if ret[0] <= ret[1] else 'It is a dog'
print(ret)
def train(self):
"""
開始訓(xùn)練
:return:
"""
self.x = tf.placeholder(tf.float32, [None, 224, 224, 3]) # 輸入數(shù)據(jù)
self.y = tf.placeholder(tf.float32, [None, self.class_num]) # 標(biāo)簽數(shù)據(jù)
self.learning_rate = tf.placeholder(tf.float32) # 學(xué)習(xí)率
# 生成訓(xùn)練用數(shù)據(jù)集
x_train_list, y_train_list, x_val_list, y_val_list = self.getTrainDataset()
print('開始轉(zhuǎn)換tensor隊(duì)列')
x_train_list_tensor = tf.convert_to_tensor(x_train_list, dtype=tf.string)
y_train_list_tensor = tf.convert_to_tensor(y_train_list, dtype=tf.float32)
x_val_list_tensor = tf.convert_to_tensor(x_val_list, dtype=tf.string)
y_val_list_tensor = tf.convert_to_tensor(y_val_list, dtype=tf.float32)
x_train_queue = tf.train.slice_input_producer(tensor_list=[x_train_list_tensor], shuffle=False)
y_train_queue = tf.train.slice_input_producer(tensor_list=[y_train_list_tensor], shuffle=False)
x_val_queue = tf.train.slice_input_producer(tensor_list=[x_val_list_tensor], shuffle=False)
y_val_queue = tf.train.slice_input_producer(tensor_list=[y_val_list_tensor], shuffle=False)
train_im, train_label = self.dataset_opt(x_train_queue, y_train_queue)
train_batch = tf.train.batch(tensors=[train_im, train_label], batch_size=self.batch_size, num_threads=2)
val_im, val_label = self.dataset_opt(x_val_queue, y_val_queue)
val_batch = tf.train.batch(tensors=[val_im, val_label], batch_size=self.batch_size, num_threads=2)
# VGG16網(wǎng)絡(luò)
print('開始加載網(wǎng)絡(luò)')
self.pred = self.CNN()
# 損失函數(shù)
self.loss = tf.nn.softmax_cross_entropy_with_logits(logits=self.pred, labels=self.y)
# 優(yōu)化器
self.opt = tf.train.AdamOptimizer(learning_rate=self.learning_rate).minimize(self.loss)
# acc
self.acc_tf = tf.equal(tf.argmax(self.pred, 1), tf.argmax(self.y, 1))
self.acc = tf.reduce_mean(tf.cast(self.acc_tf, tf.float32))
with tf.Session() as self.sess:
# 全局tf變量初始化
self.sess.run(tf.global_variables_initializer())
coordinator = tf.train.Coordinator()
threads = tf.train.start_queue_runners(sess=self.sess, coord=coordinator)
# 模型保存
saver = tf.train.Saver()
batch_max = len(x_train_list) // self.batch_size
total_step = 1
for epoch_num in range(self.epoch_max):
lr = self.lr * (1 - (epoch_num/self.epoch_max) ** 2) # 動(dòng)態(tài)學(xué)習(xí)率
for batch_num in range(batch_max):
x_train_tmp, y_train_tmp = self.sess.run(train_batch)
self.sess.run(self.opt, feed_dict={self.x: x_train_tmp, self.y: y_train_tmp, self.learning_rate: lr, self.keep_drop: 0.5})
# 輸出評價(jià)標(biāo)準(zhǔn)
if total_step % 20 == 0 or total_step == 1:
print()
print('epoch:%d/%d batch:%d/%d step:%d lr:%.10f' % ((epoch_num + 1), self.epoch_max, (batch_num + 1), batch_max, total_step, lr))
# 輸出訓(xùn)練集評價(jià)
train_loss, train_acc = self.sess.run([self.loss, self.acc], feed_dict={self.x: x_train_tmp, self.y: y_train_tmp, self.keep_drop: 1.})
print('train_loss:%.10f train_acc:%.10f' % (np.mean(train_loss), train_acc))
# 輸出驗(yàn)證集評價(jià)
val_loss_list, val_acc_list = [], []
for i in range(int(self.val_num/self.batch_size)):
x_val_tmp, y_val_tmp = self.sess.run(val_batch)
val_loss, val_acc = self.sess.run([self.loss, self.acc], feed_dict={self.x: x_val_tmp, self.y: y_val_tmp, self.keep_drop: 1.})
val_loss_list.append(np.mean(val_loss))
val_acc_list.append(np.mean(val_acc))
print(' val_loss:%.10f val_acc:%.10f' % (np.mean(val_loss), np.mean(val_acc)))
total_step += 1
# 保存模型
if (epoch_num + 1) % self.save_epoch == 0:
print('正在保存模型:')
saver.save(self.sess, './model/DogVsCat', global_step=(epoch_num + 1))
coordinator.request_stop()
coordinator.join(threads)
def CNN(self):
"""
VGG16 + FC
:return:
"""
# 權(quán)重
weight = {
# 輸入 batch_size*224*224*3
# 第一層
'wc1_1': tf.get_variable('wc1_1', [3, 3, 3, 64]), # 卷積 輸出:batch_size*224*224*64
'wc1_2': tf.get_variable('wc1_2', [3, 3, 64, 64]), # 卷積 輸出:batch_size*224*224*64
# 池化 輸出:112*112*64
# 第二層
'wc2_1': tf.get_variable('wc2_1', [3, 3, 64, 128]), # 卷積 輸出:batch_size*112*112*128
'wc2_2': tf.get_variable('wc2_2', [3, 3, 128, 128]), # 卷積 輸出:batch_size*112*112*128
# 池化 輸出:56*56*128
# 第三層
'wc3_1': tf.get_variable('wc3_1', [3, 3, 128, 256]), # 卷積 輸出:batch_size*56*56*256
'wc3_2': tf.get_variable('wc3_2', [3, 3, 256, 256]), # 卷積 輸出:batch_size*56*56*256
'wc3_3': tf.get_variable('wc3_3', [3, 3, 256, 256]), # 卷積 輸出:batch_size*56*56*256
# 池化 輸出:28*28*256
# 第四層
'wc4_1': tf.get_variable('wc4_1', [3, 3, 256, 512]), # 卷積 輸出:batch_size*28*28*512
'wc4_2': tf.get_variable('wc4_2', [3, 3, 512, 512]), # 卷積 輸出:batch_size*28*28*512
'wc4_3': tf.get_variable('wc4_3', [3, 3, 512, 512]), # 卷積 輸出:batch_size*28*28*512
# 池化 輸出:14*14*512
# 第五層
'wc5_1': tf.get_variable('wc5_1', [3, 3, 512, 512]), # 卷積 輸出:batch_size*14*14*512
'wc5_2': tf.get_variable('wc5_2', [3, 3, 512, 512]), # 卷積 輸出:batch_size*14*14*512
'wc5_3': tf.get_variable('wc5_3', [3, 3, 512, 512]), # 卷積 輸出:batch_size*14*14*512
# 池化 輸出:7*7*512
# 全鏈接第一層
'wfc_1': tf.get_variable('wfc_1', [7*7*512, 4096]),
# 全鏈接第二層
'wfc_2': tf.get_variable('wfc_2', [4096, 4096]),
# 全鏈接第三層
'wfc_3': tf.get_variable('wfc_3', [4096, self.class_num]),
}
# 偏移量
biase = {
# 第一層
'bc1_1': tf.get_variable('bc1_1', [64]),
'bc1_2': tf.get_variable('bc1_2', [64]),
# 第二層
'bc2_1': tf.get_variable('bc2_1', [128]),
'bc2_2': tf.get_variable('bc2_2', [128]),
# 第三層
'bc3_1': tf.get_variable('bc3_1', [256]),
'bc3_2': tf.get_variable('bc3_2', [256]),
'bc3_3': tf.get_variable('bc3_3', [256]),
# 第四層
'bc4_1': tf.get_variable('bc4_1', [512]),
'bc4_2': tf.get_variable('bc4_2', [512]),
'bc4_3': tf.get_variable('bc4_3', [512]),
# 第五層
'bc5_1': tf.get_variable('bc5_1', [512]),
'bc5_2': tf.get_variable('bc5_2', [512]),
'bc5_3': tf.get_variable('bc5_3', [512]),
# 全鏈接第一層
'bfc_1': tf.get_variable('bfc_1', [4096]),
# 全鏈接第二層
'bfc_2': tf.get_variable('bfc_2', [4096]),
# 全鏈接第三層
'bfc_3': tf.get_variable('bfc_3', [self.class_num]),
}
# 第一層
net = tf.nn.conv2d(input=self.x, filter=weight['wc1_1'], strides=[1, 1, 1, 1], padding='SAME') # 卷積
net = tf.nn.leaky_relu(tf.nn.bias_add(net, biase['bc1_1'])) # 加b 然后 激活
net = tf.nn.conv2d(net, filter=weight['wc1_2'], strides=[1, 1, 1, 1], padding='SAME') # 卷積
net = tf.nn.leaky_relu(tf.nn.bias_add(net, biase['bc1_2'])) # 加b 然后 激活
net = tf.nn.max_pool(value=net, ksize=[1, 2, 2, 1], strides=[1, 2, 2, 1], padding='VALID') # 池化
# 第二層
net = tf.nn.conv2d(net, weight['wc2_1'], [1, 1, 1, 1], padding='SAME') # 卷積
net = tf.nn.leaky_relu(tf.nn.bias_add(net, biase['bc2_1'])) # 加b 然后 激活
net = tf.nn.conv2d(net, weight['wc2_2'], [1, 1, 1, 1], padding='SAME') # 卷積
net = tf.nn.leaky_relu(tf.nn.bias_add(net, biase['bc2_2'])) # 加b 然后 激活
net = tf.nn.max_pool(net, [1, 2, 2, 1], [1, 2, 2, 1], padding='VALID') # 池化
# 第三層
net = tf.nn.conv2d(net, weight['wc3_1'], [1, 1, 1, 1], padding='SAME') # 卷積
net = tf.nn.leaky_relu(tf.nn.bias_add(net, biase['bc3_1'])) # 加b 然后 激活
net = tf.nn.conv2d(net, weight['wc3_2'], [1, 1, 1, 1], padding='SAME') # 卷積
net = tf.nn.leaky_relu(tf.nn.bias_add(net, biase['bc3_2'])) # 加b 然后 激活
net = tf.nn.conv2d(net, weight['wc3_3'], [1, 1, 1, 1], padding='SAME') # 卷積
net = tf.nn.leaky_relu(tf.nn.bias_add(net, biase['bc3_3'])) # 加b 然后 激活
net = tf.nn.max_pool(net, [1, 2, 2, 1], [1, 2, 2, 1], padding='VALID') # 池化
# 第四層
net = tf.nn.conv2d(net, weight['wc4_1'], [1, 1, 1, 1], padding='SAME') # 卷積
net = tf.nn.leaky_relu(tf.nn.bias_add(net, biase['bc4_1'])) # 加b 然后 激活
net = tf.nn.conv2d(net, weight['wc4_2'], [1, 1, 1, 1], padding='SAME') # 卷積
net = tf.nn.leaky_relu(tf.nn.bias_add(net, biase['bc4_2'])) # 加b 然后 激活
net = tf.nn.conv2d(net, weight['wc4_3'], [1, 1, 1, 1], padding='SAME') # 卷積
net = tf.nn.leaky_relu(tf.nn.bias_add(net, biase['bc4_3'])) # 加b 然后 激活
net = tf.nn.max_pool(net, [1, 2, 2, 1], [1, 2, 2, 1], padding='VALID') # 池化
# 第五層
net = tf.nn.conv2d(net, weight['wc5_1'], [1, 1, 1, 1], padding='SAME') # 卷積
net = tf.nn.leaky_relu(tf.nn.bias_add(net, biase['bc5_1'])) # 加b 然后 激活
net = tf.nn.conv2d(net, weight['wc5_2'], [1, 1, 1, 1], padding='SAME') # 卷積
net = tf.nn.leaky_relu(tf.nn.bias_add(net, biase['bc5_2'])) # 加b 然后 激活
net = tf.nn.conv2d(net, weight['wc5_3'], [1, 1, 1, 1], padding='SAME') # 卷積
net = tf.nn.leaky_relu(tf.nn.bias_add(net, biase['bc5_3'])) # 加b 然后 激活
net = tf.nn.max_pool(net, [1, 2, 2, 1], [1, 2, 2, 1], padding='VALID') # 池化
print('last-net', net)
# 拉伸flatten匙铡,把多個(gè)圖片同時(shí)分別拉伸成一條向量
net = tf.reshape(net, shape=[-1, weight['wfc_1'].get_shape()[0]])
print(weight['wfc_1'].get_shape()[0])
print('拉伸flatten', net)
# 全鏈接層
# fc第一層
net = tf.matmul(net, weight['wfc_1']) + biase['bfc_1']
net = tf.nn.dropout(net, self.keep_drop)
net = tf.nn.relu(net)
print('fc第一層', net)
# fc第二層
net = tf.matmul(net, weight['wfc_2']) + biase['bfc_2']
net = tf.nn.dropout(net, self.keep_drop)
net = tf.nn.relu(net)
print('fc第二層', net)
# fc第三層
net = tf.matmul(net, weight['wfc_3']) + biase['bfc_3']
print('fc第三層', net)
return net
def getTrainDataset(self):
"""
整理數(shù)據(jù)集,把圖像resize為224*224*3碍粥,訓(xùn)練集做成25000*224*224*3鳖眼,把label做成one-hot形式
:return:
"""
train_data_list = os.listdir('./data/train_data/')
print('共有%d張訓(xùn)練圖片, 讀取%d張:' % (len(train_data_list), self.train_max_num))
random.shuffle(train_data_list) # 打亂順序
x_val_list = train_data_list[:self.val_num]
y_val_list = [[0, 1] if file_name.find('cat') > -1 else [1, 0] for file_name in x_val_list]
x_train_list = train_data_list[self.val_num:self.train_max_num]
y_train_list = [[0, 1] if file_name.find('cat') > -1 else [1, 0] for file_name in x_train_list]
return x_train_list, y_train_list, x_val_list, y_val_list
def dataset_opt(self, x_train_queue, y_train_queue):
"""
處理圖片和標(biāo)簽
:param queue:
:return:
"""
queue = x_train_queue[0]
contents = tf.read_file('./data/train_data/' + queue)
im = tf.image.decode_jpeg(contents)
im = tf.image.resize_images(images=im, size=[224, 224])
im = tf.reshape(im, tf.stack([224, 224, 3]))
im -= 147 # 去均值化
# im /= 255 # 將像素處理在0~1之間嚼摩,加速收斂
# im -= 0.5 # 將像素處理在-0.5~0.5之間
return im, y_train_queue[0]
if __name__ == '__main__':
opt_type = sys.argv[1:][0]
instance = DogVsCat()
if opt_type == 'train':
instance.train()
elif opt_type == 'test':
instance.test('./data/test1/1.jpg')
elif opt_type == 'start':
# 將session持久化到內(nèi)存中
instance.test('./data/test1/1.jpg')
# 啟動(dòng)web服務(wù)
# http://127.0.0.1:5050/dogOrCat?img_path=./data/test1/1.jpg
@app.route('/dogOrCat', methods=['GET', 'POST'])
def dogOrCat():
img_path = ''
if request.method == 'POST':
img_path = request.form.to_dict().get('img_path')
elif request.method == 'GET':
# img_path = request.args.get('img_path')
img_path = request.args.to_dict().get('img_path')
print(img_path)
ret = instance.dogOrCat(img_path)
print(ret)
return json.dumps({'type': ret})
app.run(host='0.0.0.0', port=5050, debug=False)