Skip to content

Commit 59dfcaf

Browse files
author
Ryo Miyajima
committed
works with a large learning rate
1 parent b3d53bf commit 59dfcaf

2 files changed

Lines changed: 17 additions & 10 deletions

File tree

‎fully_connected_feed.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -188,7 +188,7 @@ def run_training():
188188
summary_writer.add_summary(summary_str, step)
189189

190190
# Save a checkpoint and evaluate the model periodically.
191-
if (step + 1) % 10000 == 0 or (step + 1) == FLAGS.max_steps:
191+
if (step + 1) % 1000 == 0 or (step + 1) == FLAGS.max_steps:
192192
saver.save(sess, FLAGS.train_dir, global_step=step)
193193
# Evaluate against the training set.
194194
print('Training Data Eval:')

‎mnist.py‎

Lines changed: 16 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -43,6 +43,13 @@ def conv2d(x, W):
4343
def max_pool_2x2(x):
4444
return tf.nn.max_pool(x, ksize=[1, 2, 2, 1], strides=[1,2,2,1], padding='SAME')
4545

46+
def batch_normalization(shape, input):
47+
eps = 1e-5
48+
gamma = weight_variable([shape])
49+
beta = weight_variable([shape])
50+
mean, variance = tf.nn.moments(input, [0])
51+
return gamma * (input - mean) / tf.sqrt(variance + eps) + beta
52+
4653
def inference(images, keep_pl):
4754
# FIXME: deprecated documentation
4855
"""Build the MNIST model up to where it may be used for inference.
@@ -58,21 +65,21 @@ def inference(images, keep_pl):
5865

5966
with tf.name_scope('first_convolutional_layer') as scope:
6067
W_conv1 = weight_variable([5, 5, 1, 32])
61-
b_conv1 = bias_variable([32])
62-
h_conv1 = tf.nn.relu(conv2d(x_image, W_conv1) + b_conv1)
63-
h_pool1 = max_pool_2x2(h_conv1)
68+
h_conv1 = conv2d(x_image, W_conv1)
69+
bn1 = batch_normalization(32, h_conv1)
70+
h_pool1 = max_pool_2x2(tf.nn.relu(bn1))
6471

6572
with tf.name_scope('second_convolutional_layer') as scope:
6673
W_conv2 = weight_variable([5, 5, 32, 64])
67-
b_conv2 = bias_variable([64])
68-
h_conv2 = tf.nn.relu(conv2d(h_pool1, W_conv2) + b_conv2)
69-
h_pool2 = max_pool_2x2(h_conv2)
74+
h_conv2 = conv2d(h_pool1, W_conv2)
75+
bn2 = batch_normalization(64, h_conv2)
76+
h_pool2 = max_pool_2x2(tf.nn.relu(bn2))
7077

7178
with tf.name_scope('densely_connected_layer') as scope:
7279
W_fc1 = weight_variable([7*7*64, 1024])
73-
b_fc1 = bias_variable([1024])
7480
h_pool2_flat = tf.reshape(h_pool2, [-1, 7*7*64])
75-
h_fc1 = tf.nn.relu(tf.matmul(h_pool2_flat, W_fc1) + b_fc1)
81+
bn3 = batch_normalization(1024, tf.matmul(h_pool2_flat, W_fc1))
82+
h_fc1 = tf.nn.relu(bn3)
7683

7784
with tf.name_scope('dropout') as scope:
7885
h_fc1_drop = tf.nn.dropout(h_fc1, keep_pl)
@@ -131,7 +138,7 @@ def training(loss):
131138
# Add a scalar summary for the snapshot loss.
132139
tf.scalar_summary(loss.op.name, loss)
133140
# Create the gradient descent optimizer with the given learning rate.
134-
optimizer = tf.train.AdamOptimizer(1e-4)
141+
optimizer = tf.train.AdamOptimizer(1e-3)
135142
# Create a variable to track the global step.
136143
global_step = tf.Variable(0, name='global_step', trainable=False)
137144
# Use the optimizer to apply the gradients that minimize the loss

0 commit comments

Comments
 (0)