From b00b4492417f394e91ff229e8da3ccde2b311754 Mon Sep 17 00:00:00 2001 From: Luuk <321luuk@live.nl> Date: Sat, 15 Aug 2020 22:13:37 +0200 Subject: [PATCH] Some basic changes to start using only network1 and to print some things during training --- network.py | 54 ++++++++++++- network_ori.py | 149 +++++++++++++++++++++++++++++++++++ test.py | 22 ++++-- test_ori.py | 210 +++++++++++++++++++++++++++++++++++++++++++++++++ 4 files changed, 425 insertions(+), 10 deletions(-) create mode 100644 network_ori.py create mode 100644 test_ori.py diff --git a/network.py b/network.py index ad9f26e..7d9ef14 100644 --- a/network.py +++ b/network.py @@ -62,19 +62,23 @@ def SGD(self, training_data, epochs, mini_batch_size, eta, test_data = list(test_data) n_test = len(test_data) + print_thingy = True # Print thingy + for j in range(epochs): random.shuffle(training_data) mini_batches = [ training_data[k:k+mini_batch_size] for k in range(0, n, mini_batch_size)] + self.evaluate(test_data) for mini_batch in mini_batches: - self.update_mini_batch(mini_batch, eta) + self.update_mini_batch(mini_batch, eta, print_thingy=print_thingy) # Print thingy + print_thingy = False # Print thingy if test_data: print("Epoch {} : {} / {}".format(j,self.evaluate(test_data),n_test)); else: print("Epoch {} complete".format(j)) - def update_mini_batch(self, mini_batch, eta): + def update_mini_batch(self, mini_batch, eta, print_thingy): # Print thingy """Update the network's weights and biases by applying gradient descent using backpropagation to a single mini batch. The ``mini_batch`` is a list of tuples ``(x, y)``, and ``eta`` @@ -87,6 +91,10 @@ def update_mini_batch(self, mini_batch, eta): nabla_w = [nw+dnw for nw, dnw in zip(nabla_w, delta_nabla_w)] self.weights = [w-(eta/len(mini_batch))*nw for w, nw in zip(self.weights, nabla_w)] + if print_thingy: # Print thingy + for i in range(100): + print(nabla_w[0][0][i]) + print() self.biases = [b-(eta/len(mini_batch))*nb for b, nb in zip(self.biases, nabla_b)] @@ -110,7 +118,23 @@ def backprop(self, x, y): delta = self.cost_derivative(activations[-1], y) * \ sigmoid_prime(zs[-1]) nabla_b[-1] = delta + """ + delta = [ + [d0] + [d1] + [d2] + [d3] + ] + """ nabla_w[-1] = np.dot(delta, activations[-2].transpose()) + """ + nabla_w[-1] = [ + [d0a0, d0a1, d0a2, d0a3] + [d1a0, d1a1, d1a2, d1a3] + [d2a0, d2a1, d2a2, d2a3] + [d3a0, d3a1, d3a2, d3a3] + ] + """ # Note that the variable l in the loop below is used a little # differently to the notation in Chapter 2 of the book. Here, # l = 1 means the last layer of neurons, l = 2 is the @@ -121,6 +145,28 @@ def backprop(self, x, y): z = zs[-l] sp = sigmoid_prime(z) delta = np.dot(self.weights[-l+1].transpose(), delta) * sp + """ + weights[l+1] = [ + [w0,0 w0,1 w0,2 w0,3] + [w1,0 w1,1 w1,2 w1,3] + [w2,0 w2,1 w2,2 w2,3] + [w3,0 w3,1 w3,2 w3,3] + ] + + weights[l+1].transpose() = [ + [w0,0 w1,0 w2,0 w3,0] + [w0,1 w1,1 w2,1 w3,1] + [w0,2 w1,2 w2,2 w3,2] + [w0,3 w1,3 w2,3 w3,3] + ] + + delta = [ + [d0w0,0 + d1w1,0 + d2w2,0 + d3w3,0] + [d0w0,1 + d1w1,1 + d2w2,1 + d3w3,1] + [d0w0,2 + d1w1,2 + d2w2,2 + d3w3,2] + [d0w0,3 + d1w1,3 + d2w2,3 + d3w3,3] + ] + """ nabla_b[-l] = delta nabla_w[-l] = np.dot(delta, activations[-l-1].transpose()) return (nabla_b, nabla_w) @@ -132,6 +178,10 @@ def evaluate(self, test_data): neuron in the final layer has the highest activation.""" test_results = [(np.argmax(self.feedforward(x)), y) for (x, y) in test_data] + for i in range(3): # Also a print thingy + x, y = test_data[i] + print(self.feedforward(x)) + print(y) return sum(int(x == y) for (x, y) in test_results) def cost_derivative(self, output_activations, y): diff --git a/network_ori.py b/network_ori.py new file mode 100644 index 0000000..ad9f26e --- /dev/null +++ b/network_ori.py @@ -0,0 +1,149 @@ +# %load network.py + +""" +network.py +~~~~~~~~~~ +IT WORKS + +A module to implement the stochastic gradient descent learning +algorithm for a feedforward neural network. Gradients are calculated +using backpropagation. Note that I have focused on making the code +simple, easily readable, and easily modifiable. It is not optimized, +and omits many desirable features. +""" + +#### Libraries +# Standard library +import random + +# Third-party libraries +import numpy as np + +class Network(object): + + def __init__(self, sizes): + """The list ``sizes`` contains the number of neurons in the + respective layers of the network. For example, if the list + was [2, 3, 1] then it would be a three-layer network, with the + first layer containing 2 neurons, the second layer 3 neurons, + and the third layer 1 neuron. The biases and weights for the + network are initialized randomly, using a Gaussian + distribution with mean 0, and variance 1. Note that the first + layer is assumed to be an input layer, and by convention we + won't set any biases for those neurons, since biases are only + ever used in computing the outputs from later layers.""" + self.num_layers = len(sizes) + self.sizes = sizes + self.biases = [np.random.randn(y, 1) for y in sizes[1:]] + self.weights = [np.random.randn(y, x) + for x, y in zip(sizes[:-1], sizes[1:])] + + def feedforward(self, a): + """Return the output of the network if ``a`` is input.""" + for b, w in zip(self.biases, self.weights): + a = sigmoid(np.dot(w, a)+b) + return a + + def SGD(self, training_data, epochs, mini_batch_size, eta, + test_data=None): + """Train the neural network using mini-batch stochastic + gradient descent. The ``training_data`` is a list of tuples + ``(x, y)`` representing the training inputs and the desired + outputs. The other non-optional parameters are + self-explanatory. If ``test_data`` is provided then the + network will be evaluated against the test data after each + epoch, and partial progress printed out. This is useful for + tracking progress, but slows things down substantially.""" + + training_data = list(training_data) + n = len(training_data) + + if test_data: + test_data = list(test_data) + n_test = len(test_data) + + for j in range(epochs): + random.shuffle(training_data) + mini_batches = [ + training_data[k:k+mini_batch_size] + for k in range(0, n, mini_batch_size)] + for mini_batch in mini_batches: + self.update_mini_batch(mini_batch, eta) + if test_data: + print("Epoch {} : {} / {}".format(j,self.evaluate(test_data),n_test)); + else: + print("Epoch {} complete".format(j)) + + def update_mini_batch(self, mini_batch, eta): + """Update the network's weights and biases by applying + gradient descent using backpropagation to a single mini batch. + The ``mini_batch`` is a list of tuples ``(x, y)``, and ``eta`` + is the learning rate.""" + nabla_b = [np.zeros(b.shape) for b in self.biases] + nabla_w = [np.zeros(w.shape) for w in self.weights] + for x, y in mini_batch: + delta_nabla_b, delta_nabla_w = self.backprop(x, y) + nabla_b = [nb+dnb for nb, dnb in zip(nabla_b, delta_nabla_b)] + nabla_w = [nw+dnw for nw, dnw in zip(nabla_w, delta_nabla_w)] + self.weights = [w-(eta/len(mini_batch))*nw + for w, nw in zip(self.weights, nabla_w)] + self.biases = [b-(eta/len(mini_batch))*nb + for b, nb in zip(self.biases, nabla_b)] + + def backprop(self, x, y): + """Return a tuple ``(nabla_b, nabla_w)`` representing the + gradient for the cost function C_x. ``nabla_b`` and + ``nabla_w`` are layer-by-layer lists of numpy arrays, similar + to ``self.biases`` and ``self.weights``.""" + nabla_b = [np.zeros(b.shape) for b in self.biases] + nabla_w = [np.zeros(w.shape) for w in self.weights] + # feedforward + activation = x + activations = [x] # list to store all the activations, layer by layer + zs = [] # list to store all the z vectors, layer by layer + for b, w in zip(self.biases, self.weights): + z = np.dot(w, activation)+b + zs.append(z) + activation = sigmoid(z) + activations.append(activation) + # backward pass + delta = self.cost_derivative(activations[-1], y) * \ + sigmoid_prime(zs[-1]) + nabla_b[-1] = delta + nabla_w[-1] = np.dot(delta, activations[-2].transpose()) + # Note that the variable l in the loop below is used a little + # differently to the notation in Chapter 2 of the book. Here, + # l = 1 means the last layer of neurons, l = 2 is the + # second-last layer, and so on. It's a renumbering of the + # scheme in the book, used here to take advantage of the fact + # that Python can use negative indices in lists. + for l in range(2, self.num_layers): + z = zs[-l] + sp = sigmoid_prime(z) + delta = np.dot(self.weights[-l+1].transpose(), delta) * sp + nabla_b[-l] = delta + nabla_w[-l] = np.dot(delta, activations[-l-1].transpose()) + return (nabla_b, nabla_w) + + def evaluate(self, test_data): + """Return the number of test inputs for which the neural + network outputs the correct result. Note that the neural + network's output is assumed to be the index of whichever + neuron in the final layer has the highest activation.""" + test_results = [(np.argmax(self.feedforward(x)), y) + for (x, y) in test_data] + return sum(int(x == y) for (x, y) in test_results) + + def cost_derivative(self, output_activations, y): + """Return the vector of partial derivatives \partial C_x / + \partial a for the output activations.""" + return (output_activations-y) + +#### Miscellaneous functions +def sigmoid(z): + """The sigmoid function.""" + return 1.0/(1.0+np.exp(-z)) + +def sigmoid_prime(z): + """Derivative of the sigmoid function.""" + return sigmoid(z)*(1-sigmoid(z)) diff --git a/test.py b/test.py index d1f4996..6effda0 100644 --- a/test.py +++ b/test.py @@ -17,19 +17,19 @@ # ---------------------- # - read the input data: -''' + import mnist_loader training_data, validation_data, test_data = mnist_loader.load_data_wrapper() training_data = list(training_data) -''' + # --------------------- # - network.py example: -#import network +import network + + +net = network.Network([784, 10]) +net.SGD(training_data, 1, 100, 1.0, test_data=test_data) -''' -net = network.Network([784, 30, 10]) -net.SGD(training_data, 30, 10, 3.0, test_data=test_data) -''' # ---------------------- # - network2.py example: @@ -124,6 +124,8 @@ """ + +''' def testTheano(): from theano import function, config, shared, sandbox import theano.tensor as T @@ -149,10 +151,11 @@ def testTheano(): print('Used the gpu') # Perform check: #testTheano() - +''' # ---------------------- # - network3.py example: +''' import network3 from network3 import Network, ConvPoolLayer, FullyConnectedLayer, SoftmaxLayer # softmax plus log-likelihood cost is more common in modern image classification networks. @@ -160,6 +163,7 @@ def testTheano(): training_data, validation_data, test_data = network3.load_data_shared() # mini-batch size: mini_batch_size = 10 +''' # chapter 6 - shallow architecture using just a single hidden layer, containing 100 hidden neurons. ''' @@ -195,6 +199,7 @@ def testTheano(): ''' # chapter 6 - rectified linear units and some l2 regularization (lmbda=0.1) => even better accuracy +''' from network3 import ReLU net = Network([ ConvPoolLayer(image_shape=(mini_batch_size, 1, 28, 28), @@ -208,3 +213,4 @@ def testTheano(): FullyConnectedLayer(n_in=40*4*4, n_out=100, activation_fn=ReLU), SoftmaxLayer(n_in=100, n_out=10)], mini_batch_size) net.SGD(training_data, 60, mini_batch_size, 0.03, validation_data, test_data, lmbda=0.1) +''' \ No newline at end of file diff --git a/test_ori.py b/test_ori.py new file mode 100644 index 0000000..d1f4996 --- /dev/null +++ b/test_ori.py @@ -0,0 +1,210 @@ +""" + Testing code for different neural network configurations. + Adapted for Python 3.5.2 + + Usage in shell: + python3.5 test.py + + Network (network.py and network2.py) parameters: + 2nd param is epochs count + 3rd param is batch size + 4th param is learning rate (eta) + + Author: + Michał Dobrzański, 2016 + dobrzanski.michal.daniel@gmail.com +""" + +# ---------------------- +# - read the input data: +''' +import mnist_loader +training_data, validation_data, test_data = mnist_loader.load_data_wrapper() +training_data = list(training_data) +''' +# --------------------- +# - network.py example: +#import network + +''' +net = network.Network([784, 30, 10]) +net.SGD(training_data, 30, 10, 3.0, test_data=test_data) +''' + +# ---------------------- +# - network2.py example: +#import network2 + +''' +net = network2.Network([784, 30, 10], cost=network2.CrossEntropyCost) +#net.large_weight_initializer() +net.SGD(training_data, 30, 10, 0.1, lmbda = 5.0,evaluation_data=validation_data, + monitor_evaluation_accuracy=True) +''' + +# chapter 3 - Overfitting example - too many epochs of learning applied on small (1k samples) amount od data. +# Overfitting is treating noise as a signal. +''' +net = network2.Network([784, 30, 10], cost=network2.CrossEntropyCost) +net.large_weight_initializer() +net.SGD(training_data[:1000], 400, 10, 0.5, evaluation_data=test_data, + monitor_evaluation_accuracy=True, + monitor_training_cost=True) +''' + +# chapter 3 - Regularization (weight decay) example 1 (only 1000 of training data and 30 hidden neurons) +''' +net = network2.Network([784, 30, 10], cost=network2.CrossEntropyCost) +net.large_weight_initializer() +net.SGD(training_data[:1000], 400, 10, 0.5, + evaluation_data=test_data, + lmbda = 0.1, # this is a regularization parameter + monitor_evaluation_cost=True, + monitor_evaluation_accuracy=True, + monitor_training_cost=True, + monitor_training_accuracy=True) +''' + +# chapter 3 - Early stopping implemented +''' +net = network2.Network([784, 30, 10], cost=network2.CrossEntropyCost) +net.SGD(training_data[:1000], 30, 10, 0.5, + lmbda=5.0, + evaluation_data=validation_data, + monitor_evaluation_accuracy=True, + monitor_training_cost=True, + early_stopping_n=10) +''' + +# chapter 4 - The vanishing gradient problem - deep networks are hard to train with simple SGD algorithm +# this network learns much slower than a shallow one. +''' +net = network2.Network([784, 30, 30, 30, 30, 10], cost=network2.CrossEntropyCost) +net.SGD(training_data, 30, 10, 0.1, + lmbda=5.0, + evaluation_data=validation_data, + monitor_evaluation_accuracy=True) +''' + + +# ---------------------- +# Theano and CUDA +# ---------------------- + +""" + This deep network uses Theano with GPU acceleration support. + I am using Ubuntu 16.04 with CUDA 7.5. + Tutorial: + http://deeplearning.net/software/theano/install_ubuntu.html#install-ubuntu + + The following command will update only Theano: + sudo pip install --upgrade --no-deps theano + + The following command will update Theano and Numpy/Scipy (warning bellow): + sudo pip install --upgrade theano + +""" + +""" + Below, there is a testing function to check whether your computations have been made on CPU or GPU. + If the result is 'Used the cpu' and you want to have it in gpu, do the following: + 1) install theano: + sudo python3.5 -m pip install Theano + 2) download and install the latest cuda: + https://developer.nvidia.com/cuda-downloads + I had some issues with that, so I followed this idea (better option is to download the 1,1GB package as .run file): + http://askubuntu.com/questions/760242/how-can-i-force-16-04-to-add-a-repository-even-if-it-isnt-considered-secure-eno + You may also want to grab the proper NVidia driver, choose it form there: + System Settings > Software & Updates > Additional Drivers. + 3) should work, run it with: + THEANO_FLAGS=mode=FAST_RUN,device=gpu,floatX=float32 python3.5 test.py + http://deeplearning.net/software/theano/tutorial/using_gpu.html + 4) Optionally, you can add cuDNN support from: + https://developer.nvidia.com/cudnn + + +""" +def testTheano(): + from theano import function, config, shared, sandbox + import theano.tensor as T + import numpy + import time + print("Testing Theano library...") + vlen = 10 * 30 * 768 # 10 x #cores x # threads per core + iters = 1000 + + rng = numpy.random.RandomState(22) + x = shared(numpy.asarray(rng.rand(vlen), config.floatX)) + f = function([], T.exp(x)) + print(f.maker.fgraph.toposort()) + t0 = time.time() + for i in range(iters): + r = f() + t1 = time.time() + print("Looping %d times took %f seconds" % (iters, t1 - t0)) + print("Result is %s" % (r,)) + if numpy.any([isinstance(x.op, T.Elemwise) for x in f.maker.fgraph.toposort()]): + print('Used the cpu') + else: + print('Used the gpu') +# Perform check: +#testTheano() + + +# ---------------------- +# - network3.py example: +import network3 +from network3 import Network, ConvPoolLayer, FullyConnectedLayer, SoftmaxLayer # softmax plus log-likelihood cost is more common in modern image classification networks. + +# read data: +training_data, validation_data, test_data = network3.load_data_shared() +# mini-batch size: +mini_batch_size = 10 + +# chapter 6 - shallow architecture using just a single hidden layer, containing 100 hidden neurons. +''' +net = Network([ + FullyConnectedLayer(n_in=784, n_out=100), + SoftmaxLayer(n_in=100, n_out=10)], mini_batch_size) +net.SGD(training_data, 60, mini_batch_size, 0.1, validation_data, test_data) +''' + +# chapter 6 - 5x5 local receptive fields, 20 feature maps, max-pooling layer 2x2 +''' +net = Network([ + ConvPoolLayer(image_shape=(mini_batch_size, 1, 28, 28), + filter_shape=(20, 1, 5, 5), + poolsize=(2, 2)), + FullyConnectedLayer(n_in=20*12*12, n_out=100), + SoftmaxLayer(n_in=100, n_out=10)], mini_batch_size) +net.SGD(training_data, 60, mini_batch_size, 0.1, validation_data, test_data) +''' + +# chapter 6 - inserting a second convolutional-pooling layer to the previous example => better accuracy +''' +net = Network([ + ConvPoolLayer(image_shape=(mini_batch_size, 1, 28, 28), + filter_shape=(20, 1, 5, 5), + poolsize=(2, 2)), + ConvPoolLayer(image_shape=(mini_batch_size, 20, 12, 12), + filter_shape=(40, 20, 5, 5), + poolsize=(2, 2)), + FullyConnectedLayer(n_in=40*4*4, n_out=100), + SoftmaxLayer(n_in=100, n_out=10)], mini_batch_size) +net.SGD(training_data, 60, mini_batch_size, 0.1, validation_data, test_data) +''' + +# chapter 6 - rectified linear units and some l2 regularization (lmbda=0.1) => even better accuracy +from network3 import ReLU +net = Network([ + ConvPoolLayer(image_shape=(mini_batch_size, 1, 28, 28), + filter_shape=(20, 1, 5, 5), + poolsize=(2, 2), + activation_fn=ReLU), + ConvPoolLayer(image_shape=(mini_batch_size, 20, 12, 12), + filter_shape=(40, 20, 5, 5), + poolsize=(2, 2), + activation_fn=ReLU), + FullyConnectedLayer(n_in=40*4*4, n_out=100, activation_fn=ReLU), + SoftmaxLayer(n_in=100, n_out=10)], mini_batch_size) +net.SGD(training_data, 60, mini_batch_size, 0.03, validation_data, test_data, lmbda=0.1)