【发布时间】:2018-03-03 21:34:23
【问题描述】:
我在使用 tensorflow 的批量标准化时遇到了问题。我已经建立了以下模型:
def weight_variable(kernal_shape):
weights = tf.get_variable(name='weights', shape=kernal_shape, dtype=tf.float32, trainable=True,
initializer=tf.truncated_normal_initializer(stddev=0.02))
return weights
def bias_variable(shape):
initial = tf.constant(0.0, shape=shape)
return tf.Variable(initial)
# return 1 conv layer
def conv_layer(x, w_shape, b_shape, is_training, padding='SAME'):
W = weight_variable(w_shape)
tf.summary.histogram("weights", W)
b = bias_variable(b_shape)
tf.summary.histogram("biases", b)
# Note that I used a stride of 2 on purpose in order not to use max pool layer.
conv = tf.nn.conv2d(x, W, strides=[1, 2, 2, 1], padding=padding) + b
conv = tf.contrib.layers.batch_norm(conv, scale=True, is_training=is_training)
activations = tf.nn.relu(conv)
tf.summary.histogram("activations", activations)
return activations
# return deconv layer
def deconv_layer(x, w_shape, b_shape, is_training, padding="SAME", activation='relu'):
W = weight_variable(w_shape)
tf.summary.histogram("weights", W)
b = bias_variable(b_shape)
tf.summary.histogram('biases', b)
x_shape = tf.shape(x)
# output shape: [batch_size, h * 2, w * 2, input_shape from w].
out_shape = tf.stack([x_shape[0], x_shape[1] * 2, x_shape[2] * 2, w_shape[2]])
# Note that I have used a stride of 2 since I used a stride of 2 in conv layer.
conv_trans = tf.nn.conv2d_transpose(x, W, out_shape, [1, 2, 2, 1], padding=padding) + b
conv_trans = tf.contrib.layers.batch_norm(conv_trans, scale=True, is_training=is_training)
if activation == 'relu':
transposed_activations = tf.nn.relu(conv_trans)
else:
transposed_activations = tf.nn.sigmoid(conv_trans)
tf.summary.histogram("transpose_activation", transposed_activations)
return transposed_activations
def model(input):
with tf.variable_scope('conv1'):
conv1 = conv_layer(input, [4, 4, 3, 32], [32], is_training=phase_train) # image size: [56, 56]
with tf.variable_scope('conv2'):
conv2 = conv_layer(conv1, [4, 4, 32, 64], [64], is_training=phase_train) # image size: [28, 28]
with tf.variable_scope('conv3'):
conv3 = conv_layer(conv2, [4, 4, 64, 128], [128], is_training=phase_train) # image size: [14, 14]
with tf.variable_scope('conv4'):
conv4 = conv_layer(conv3, [4, 4, 128, 256], [256], is_training=phase_train) # image size: [7, 7]
conv4_reshaped = tf.reshape(conv4, [batch_size * num_participants, 7 * 7 * 256], name='conv4_reshaped')
w_c_mu = tf.Variable(tf.truncated_normal([7 * 7 * 256, latent_dim], stddev=0.1), name='weight_fc_mu')
b_c_mu = tf.Variable(tf.constant(0.1, shape=[latent_dim]), name='biases_fc_mu')
w_c_sig = tf.Variable(tf.truncated_normal([7 * 7 * 256, latent_dim], stddev=0.1), name='weight_fc_sig')
b_c_sig = tf.Variable(tf.constant(0.1, shape=[latent_dim]), name='biases_fc_sig')
epsilon = tf.random_normal([1, latent_dim])
tf.summary.histogram('weights_c_mu', w_c_mu)
tf.summary.histogram('biases_c_mu', b_c_mu)
tf.summary.histogram('weights_c_sig', w_c_sig)
tf.summary.histogram('biases_c_sig', b_c_sig)
with tf.variable_scope('mu'):
mu = tf.nn.bias_add(tf.matmul(conv4_reshaped, w_c_mu), b_c_mu)
tf.summary.histogram('mu', mu)
with tf.variable_scope('stddev'):
stddev = tf.nn.bias_add(tf.matmul(conv4_reshaped, w_c_sig), b_c_sig)
tf.summary.histogram('stddev', stddev)
with tf.variable_scope('z'):
# This formula was adopted from the following paper: http://ieeexplore.ieee.org/stamp/stamp.jsp?arnumber=7979344
latent_var = mu + tf.multiply(tf.sqrt(tf.exp(stddev)), epsilon)
tf.summary.histogram('features_sig', stddev)
with tf.variable_scope('GRU'):
print(latent_var.get_shape().as_list())
latent_var = tf.reshape(latent_var, shape=[int(batch_size / 100)* num_participants, time_steps, latent_dim])
cell = tf.nn.rnn_cell.GRUCell(cell_size) # state_size of cell_size.
H, C = tf.nn.dynamic_rnn(cell, latent_var, dtype=tf.float32) # H size: [batch_size * num_participants, SEQLEN, cell_size]
H = tf.reshape(H, [batch_size * num_participants, cell_size])
with tf.variable_scope('output'):
# output layer.
w_output = tf.Variable(tf.truncated_normal([cell_size, 1], mean=0, stddev=0.01, dtype=tf.float32, name='w_output'))
tf.summary.histogram('w_output', w_output)
b_output = tf.get_variable('b_output', shape=[1], dtype=tf.float32,
initializer=tf.constant_initializer(0.0))
predictions = tf.add(tf.matmul(H, w_output), b_output, name='softmax_output')
tf.summary.histogram('output', predictions)
var_list = [v for v in tf.global_variables() if 'GRU' in v.name]
var_list.append([w_output, b_output])
return predictions, var_list
另外,我正在恢复模型参数如下:
saver_torestore = tf.train.Saver()
with tf.Session() as sess:
train_writer = tf.summary.FileWriter(events_path, sess.graph)
merged = tf.summary.merge_all()
to_run_list = [merged, RMSE]
# Initialize `iterator` with training data.
sess.run(init_op)
# Note that the last name "Graph_model" is the name of the saved checkpoints file => the ckpt is saved
# under tensorboard_logs.
ckpt = tf.train.get_checkpoint_state(
os.path.dirname(model_path))
if ckpt and ckpt.model_checkpoint_path:
saver_torestore.restore(sess, ckpt.model_checkpoint_path)
print('checkpoints are saved!!!')
else:
print('No stored checkpoints')
counter = 0
for _ in range(num_epoch):
sess.run(iterator.initializer)
print('epoch:', _)
# This while loop will run indefinitly until the end of the first epoch
while True:
try:
summary, loss_ = sess.run(to_run_list, feed_dict={phase_train: False})
print('loss: ' + str(loss_))
losses.append(loss_)
counter += 1
train_writer.add_summary(summary, counter)
except tf.errors.OutOfRangeError:
print('error, ignore ;) ')
break
print('average losses:', np.average(losses))
train_writer.close()
我确保变量被保存。所以我运行了以下命令:
def assign_values_to_batchNorm():
vars = [v for v in tf.global_variables() if "BatchNorm" in v.name and "Adam" not in v.name]
file_names = [(v.name[:-2].replace("/", "_") + ".txt") for v in vars]
for var, file_name in zip(vars, file_names):
lst = open(file_name).read().split(";")[:-1]
print(lst)
values = list(map(np.float32, lst))
tf.assign(var, values)
请注意,我使用此方法是为了手动恢复移动均值和移动方差的值。但我得到了同样的结果。
我在会话下调用了assign_values_to_batchNorm()。我得到了一些值 => 似乎移动平均线、移动方差、伽马和斗鱼都被保存了。
现在请注意,我正在使用 Windows 10,并且我有 tensorflow 版本 1.3。
所以,每当我在会话下运行summary, loss_ = sess.run(to_run_list, feed_dict={phase_train: True}) 时,在初始化/恢复所有变量后,我得到的 RMSE 为 0.022,这与训练模型结束时得到的错误相同。现在,如果我将 phase_train 设置为 false,我得到的 RMSE 为 0.038。请注意,同时我只是在测试网络。因此,即使我使用训练数据集进行测试,但我的目的只是在训练/测试时测试网络的行为。所以这对我来说太奇怪了。请注意,阶段是占位符。我的代码如下:
phase_train = tf.placeholder(dtype=tf.bool, name='phase')
另外,这里是优化器的代码sn-p:
with tf.name_scope('optimizer'):
update_ops = tf.get_collection(tf.GraphKeys.UPDATE_OPS)
with tf.control_dependencies(update_ops):
optimizer = tf.train.AdamOptimizer(0.00001).minimize(RMSE)
主要问题: RMSE = 0.038(相位 = 假)和 0.022(相位 = 真)。
非常感谢任何帮助!
【问题讨论】:
-
我仍然对您的问题/期望是什么感到困惑。据我所知,在训练模式下运行具有批处理规范的网络将更新运行平均值,因此,如果您这样做一次,那么即使您在测试模式下再次输入同一批次,您也可能不会再获得相同的结果之后。
-
再次,问题是我还没有使用测试数据集。我使用训练数据集训练了网络,然后在 phase_train 为 False 时使用训练数据集再次运行它。我得到了上面报告的 RMSE
-
tf.assign(var, values)返回一个操作,这意味着它不会真正将values分配给var,除非您在会话下运行此操作。但我不确定这是不是问题,因为saver_torestore.restore默认应该加载所有变量,包括moving_means 和moving_vars。关于您得到的 RMSE 的另一个问题:您使用所有训练数据进行测试?您每次迭代都使用所有训练数据进行训练,而不是使用小批量? -
是的,我使用小批量测试了 1 个 epoch 的所有训练数据。我也使用小批量训练了 150 个 epoch 的所有训练数据。请注意,数据集由人脸组成。所以我不确定处理面孔是否有什么不同??
-
训练和测试时一个小批量有多少样例?训练阶段和测试阶段的预处理输入有什么区别吗?为什么您认为 RMSE=0.038 和 RMSE=0.022 有很大不同?你期待什么结果? (它们不会一样,就像这里的第一条评论所说,“你可能不会再得到相同的结果了”。)
标签: python tensorflow batch-normalization