代码全部来自OpenCV+TensorFlow 深度学习与计算机视觉实战
import cv2
import os
def rebuild(dir):
for root, dirs, files in os.walk(dir):
for file in files:
filepath = os.path.join(root, file)
# 读取文件,剪裁文件大小并重新写入文件
image = cv2.imread(filepath)
dim = (227, 227)
resized = cv2.resize(image, dim)
path = 'C:\\cat_and_dog\\dog_r\\' + file
cv.imwrite(path, resized)
def get_file(file_dir):
images = []
temp = []
for root, sub_folders, files in os.walk(file_dir):
for name in files:
images.append(os.path.join(root, name))
#get 10 sub-folder names
for name in sub_folders:
temp.append(os.path.join(root, name))
#assign 10 labels based on the folder names,这里设置了新的标签labels,并将temp里面的文件夹标签按照0或者1形式存入
labels = []
for one_folder in temp:
n_img = len(os.listdir(one_folder))
letter = one_folder.split('\\')[-1]
if letter == 'cat':
labels = np.append(labels, n_img*[0])
labels = np.append(labels, n_img*[1])
#shuffle 设置相应的图片列表和标签列表
temp = np.array([images, labels])
temp = temp.transpose()
image_list = list(temp[:, 0])
label_list = list(temp[:, 1])
label_list = [int(float(i)) for i in label_list]
return image_list, label_list
def get_batch(image_list, label_list, img_width, img_height, batch_size, capacity):
image = tf.cast(image_list, tf.string)
label = tf.cast(label_list, tf.int32)
input_queue = tf.train.slice_input_producer([image, label])
label = input_queue[1]
image_contents = tf.read_file(input_queue[0])
image = tf.image.decode_jpeg(image_contents, channels = 3)
image = tf.image.resize_image_width_crop_or_pad(image, img_width, img_height)
image = tf.image.per_image_standardization(image) #将图片标准化
image_batch, label_batch = tf.train.batch([image, label], batch_size = batch_size, num_threads = 64, capacity = capacity)
label_batch = tf.reshape(label_batch, [batch_size])
return image_batch, label_batch
在这里get_batch(image_list, label_list, img_width, img_height, batch_size, capacity)函数中有6个参数,前两个分别为图片列表和标签列表(图片列表和标签列表的生成方式在前文的代码段中已经说明)。 img_width和img_height分别为生成图片的大小,这里按照模型的需求指定。batch_size和capacity分别是每次生成的图片数量和内存中存储的最大数据容量,这里根据不同硬件配置。
import tensorflow as tf
import numpy as np
import matplotlib.pyplot as plt
import time
import create_and_read_TFRecord2 as reader2
import os
# 猫狗大战的数据集下载地址为 http://www.kaggle.com/c/dogs-vs-cats
X_train, y_train = reader2.get_file('c:\\cat_and_dog_r')
image_batch, label_batch = reader2.get_batch(X_train, y_train, 227, 227, 200, 2048)
def batch_norm(inputs, is_training, is_conv_out = True, decay = 0.999):
scale = tf.Variable(tf.ones([inputs.get_shape()[-1]]))
beta = tf.Variable(tf.zeros([inputs.get_shape()[-1]]))
pop_mean = tf.Variable(tf.zeros([inputs.get_shape()[-1]]), trainable = False)
pop_var = tf.Variable(tf.noes([inputs.get_shape()[-1]]), trainable = False)
if is_training:
if is_conv_out:
batch_mean, batch_var = tf.nn.monents(inputs,[0,1,2])
batch_mean, batch_var = tf.nn.monents(inputs,[0])
train_mean = tf.assign(pop_mean, pop_mean * decay + batch_mean * (1-decay))
train_var = tf.assign(pop_var, pop_var * decay +batch_var * (1-decay))
with tf.control_dependencies([train_mean, train_var]):
return tf.nn.batch_normalization(inputs, batch_mean, batch_var, beta, scale, 0.001)
return tf.nn.batch_normalization(inputs, pop_mean, pop_var, beta, scale, 0.001)
with tf.device('/cpu:0'):
learning_rate = 1e-4
training_iters = 200
batch_size = 200
display_step = 5
n_classes = 2
n_fcl = 4096
n_fc2 = 2048
x = tf.placeholder(tf.float32, [None, 227, 227, 3])
y = tf.placeholder(tf.int32, [None, n_classes])
W_conv = {'conv1': tf.Variable(tf.truncated_normal([11, 11, 3, 96], stddev = 0.0001)),
'conv2': tf.Variable(tf.truncated_normal([5, 5, 96, 255], stddev = 0.01)),
'conv3': tf.Variable(tf.truncated_normal([3, 3, 256, 384], stddev = 0.01)),
'conv4': tf.Variable(tf.truncated_normal([3, 3, 384, 384], stddev = 0.01)),
'conv5': tf.Variable(tf.truncated_normal([3, 3, 384, 256], stddev = 0.01)),
'fc1': tf.Variable(tf.truncated_normal([13 * 13 * 256, n_fc1], stddev = 0.1)),
'fc2': tf.Variable(tf.truncated_normal([n_fc1, n_fc2], stddev = 0.1)),
'fc3': tf.Variable(tf.truncated_normal([n_fc2, n_classes], stddev = 0.1))}
b_conv = {'conv1': tf.Variable(tf.constant(0.0, dtype = tf.float32, shape = [96])),
'conv2': tf.Variable(tf.constant(0.1, dtype = tf.float32, shape = [256])),
'conv3': tf.Variable(tf.constant(0.1, dtype = tf.float32, shape = [384])),
'conv4': tf.Variable(tf.constant(0.1, dtype = tf.float32, shape = [384])),
'conv2': tf.Variable(tf.constant(0.1, dtype = tf.float32, shape = [256])),
'fc1': tf.Variable(tf.constant(0.1, dtype = tf.float32, shape = [n_fc1])),
'fc2': tf.Variable(tf.constant(0.1, dtype = tf.float32, shape = [n_fc2])),
'fc3': tf.Variable(tf.constant(0.0, dtype = tf.float32, shape = [n_classes]))}
x_image = tf.reshape(x, [-1, 227, 227,3])
#卷积层 1
conv1 = tf.nn.conv2d(x_image, W_conv['conv1'], strides = [1, 4, 4, 1], padding = 'VALID')
conv1 = tf.nn.bias_add(conv1, b_conv['conv1'])
conv1 = tf.nn.relu(conv1)
pool1 = tf.nn.avg_pool(conv1, ksize = [1, 3, 3, 1], strides = [1, 2, 2, 1], padding = 'VALID')
#LRN层, Local Response Normalization
norm1 = tf.nn.lrn(pool1, 5, bias = 1.0, alpha = 0.001/9.0, beta = 0.75)
conv2 = tf.nn.conv2d(norm1, W_conv['conv2'], strides = [1, 1, 1,], padding = 'SAME')
conv2 = tf.nn.bias_add(conv2, b_conv['conv2'])
conv2 = tf.nn.relu(conv2)
pool2 = tf.nn.avg_pool(conv2,ksize = [1, 3, 3, 1], strides = [1, 2, 2, 1], padding = 'VALID')
#LRN层,Local Response Normalization
norm2 = tf.nn.lrn(pool2, 5, bias = 1.0, alpha = 0.001/9.0, beta = 0.75)
conv3 = tf.nn.conv2d(norm2, W_conv['conv3'], strides = [1, 1, 1, 1], padding = 'SAME')
conv3 = tf.nn.bias_add(conv3, b_conv['conv3'])
conv3 = tf.nn.relu(conv3)
conv4 = tf.nn.conv2d(conv3, W_conv['conv4'], strides = [1, 1, 1, 1], padding = 'SAME')
conv4 = tf.nn.bias_add(conv4, b_conv['conv4'])
conv4 = tf.nn.relu(conv4)
conv5 = tf.nn.conv2d(conv4, W_conv['conv5'], strides = [1, 1, 1, 1], padding = 'SAME')
conv5 = tf.nn.bias_add(conv5, b_conv['conv5'])
conv5 = tf.nn.relu(conv5)
pool5 = tf.nn.avg_pool(conv5, ksize = [1, 3, 3, 1], strides = [1, 2, 2, 1], padding = 'VALID')
reshape = tf.reshape(pool5, [-1, 13 * 13 * 256])
fc1 = tf.add(tf.matmul(reshape, W_conv['fc1']), b_conv['fc1'])
fc1 = tf.nn.relu(fc1)
fc1 = tf.nn.dropout(fc1, 0.5)
fc2 = tf.add(tf.matmul(fc1, W_conv['fc2']), b_conv['fc2'])
fc2 = tf.nn.relu(fc2)
fc2 = tf.nn.dropout(fc2, 0.5)
fc3 = tf.add(tf.matmul(fc2, W_conv['fc3']), b_conv['fc3'])
loss = tf.reduce_mean(tf.nn.softmax_cross_entropy_with_logits(fc3, y))
optimizer = tf.train.GradientDescentOptimizer(learning_rate = learning_rate).minimize(loss)
correct_pred = tf.equal(tf.argmax(fc3, 1), tf.argmax(y, 1))
accuracy = tf.reduce_mean(tf.cast(correct_pred, tf.float32))
init = tf.global_variables_initializer()
def onehot(labels):
'''one-hot 编码'''
n_sample = len(labels)
n_class = max(labels)+1
onehot_labels = np.zeros((n_sample, n_class))
onehot_labels[np.arange(n_sample), labels] = 1
return onehot_labels
save_model = './/model//AlexNetModel.ckpt'
def train(opech):
with tf.Session() as sess:
train_writer = tf.summary.FileWriter('.//log', sess.graph) #输出日志
saver = tf.train.Saver()
c = []
start_time = time.time()
coord = tf.train.Coordinator()
threads = tf.train.start_queue_runners(coord = coord)
step = 0
for i in range(opech):
step = i
image, label = sess.run([image_batch, label_batch])
labels = onehot(label)
sess.run(optimizer, feed_dict = {x:image, y: labels})
loss_record = sess.run(loss, feed_dict = {x: image, y: labels})
print('now the loss is %f ' % loss_record)
end_time = time.time()
print('time: ', (end_time - start_time))
start_time = end_time
print('--------------%d onpech is finished---------------' %i)
print('Optimization Finished!')
saver.save(sess, save_model)
print('Model Save Finished!')
plt.title('1r = %f, ti = %d, bs = %d' % (learning_rate, training_iters, batch_size))
plt.savefig('cat_and_dog_AlexNet.jpg', dpi = 200)
from PIL import Image
def per_class(imagefile):
image = Image.open(imagefile)
image = Image.resize([227, 227])
image_array = np.array(image)
image = tf.cast(image_array, tf.float 32)
image = tf.image.per_image_standardization(image)
image = tf.reshape(image, [1, 227, 227, 3])
saver = tf.train.Saver()
with tf.Session() as sess:
save_model = tf.train.latest_checkpoint('.//model')
saver.restore(sess, save_model)
image = tf.reshape(image, [1, 227, 227, 3])
image = sess.run(image)
prediction = sess.run(fc3, feed_dict = {x: image})
max_index = np.argmax(prediction)
if max_index == 0:
return 'cat'
return 'dog'