From def09cb9b2062c34e5ec3e5e37e78c160102375a Mon Sep 17 00:00:00 2001 From: WangLiang Date: Thu, 25 Apr 2019 14:15:13 +0800 Subject: [PATCH] Update network.py which rewrite by liang, test pass, and inital README --- README.md | 11 +- network.py | 229 +++++++++++++-------------------- test.py | 367 ++++++++++++++++++++++++++--------------------------- 3 files changed, 275 insertions(+), 332 deletions(-) diff --git a/README.md b/README.md index aa618b0..e3232a8 100644 --- a/README.md +++ b/README.md @@ -1,12 +1,17 @@ ## Overview -### neuralnetworksanddeeplearning.com integrated scripts for Python 3.5.2 and Theano with CUDA support +### Neural Network for Deeplearning integrated scripts for Python 3.6.7 and TensorFlow2.0 Env + +### Optional succeed earlier Theano with CUDA support + + +These scrips are updated ones from the **MichalDanielDobrzanski/DeepLearningPython35 +** gitHub repository in order to work with Python 3.6 and TensorFlow2.0 -These scrips are updated ones from the **neuralnetworksanddeeplearning.com** gitHub repository in order to work with Python 3.5.2 The testing file (**test.py**) contains all three networks (network.py, network2.py, network3.py) from the book and it is the starting point to run (i.e. *train and evaluate*) them. -## Just type at shell: **python3.5 test.py** +## Just type at shell: **python3.6 test.py** In test.py there are examples of networks configurations with proper comments. I did that to relate with particular chapters from the book. diff --git a/network.py b/network.py index ad9f26e..c9be4e1 100644 --- a/network.py +++ b/network.py @@ -1,149 +1,90 @@ -# %load network.py - -""" -network.py -~~~~~~~~~~ -IT WORKS - -A module to implement the stochastic gradient descent learning -algorithm for a feedforward neural network. Gradients are calculated -using backpropagation. Note that I have focused on making the code -simple, easily readable, and easily modifiable. It is not optimized, -and omits many desirable features. -""" - -#### Libraries -# Standard library import random -# Third-party libraries import numpy as np class Network(object): - - def __init__(self, sizes): - """The list ``sizes`` contains the number of neurons in the - respective layers of the network. For example, if the list - was [2, 3, 1] then it would be a three-layer network, with the - first layer containing 2 neurons, the second layer 3 neurons, - and the third layer 1 neuron. The biases and weights for the - network are initialized randomly, using a Gaussian - distribution with mean 0, and variance 1. Note that the first - layer is assumed to be an input layer, and by convention we - won't set any biases for those neurons, since biases are only - ever used in computing the outputs from later layers.""" - self.num_layers = len(sizes) - self.sizes = sizes - self.biases = [np.random.randn(y, 1) for y in sizes[1:]] - self.weights = [np.random.randn(y, x) - for x, y in zip(sizes[:-1], sizes[1:])] - - def feedforward(self, a): - """Return the output of the network if ``a`` is input.""" - for b, w in zip(self.biases, self.weights): - a = sigmoid(np.dot(w, a)+b) - return a - - def SGD(self, training_data, epochs, mini_batch_size, eta, - test_data=None): - """Train the neural network using mini-batch stochastic - gradient descent. The ``training_data`` is a list of tuples - ``(x, y)`` representing the training inputs and the desired - outputs. The other non-optional parameters are - self-explanatory. If ``test_data`` is provided then the - network will be evaluated against the test data after each - epoch, and partial progress printed out. This is useful for - tracking progress, but slows things down substantially.""" - - training_data = list(training_data) - n = len(training_data) - - if test_data: - test_data = list(test_data) - n_test = len(test_data) - - for j in range(epochs): - random.shuffle(training_data) - mini_batches = [ - training_data[k:k+mini_batch_size] - for k in range(0, n, mini_batch_size)] - for mini_batch in mini_batches: - self.update_mini_batch(mini_batch, eta) - if test_data: - print("Epoch {} : {} / {}".format(j,self.evaluate(test_data),n_test)); - else: - print("Epoch {} complete".format(j)) - - def update_mini_batch(self, mini_batch, eta): - """Update the network's weights and biases by applying - gradient descent using backpropagation to a single mini batch. - The ``mini_batch`` is a list of tuples ``(x, y)``, and ``eta`` - is the learning rate.""" - nabla_b = [np.zeros(b.shape) for b in self.biases] - nabla_w = [np.zeros(w.shape) for w in self.weights] - for x, y in mini_batch: - delta_nabla_b, delta_nabla_w = self.backprop(x, y) - nabla_b = [nb+dnb for nb, dnb in zip(nabla_b, delta_nabla_b)] - nabla_w = [nw+dnw for nw, dnw in zip(nabla_w, delta_nabla_w)] - self.weights = [w-(eta/len(mini_batch))*nw - for w, nw in zip(self.weights, nabla_w)] - self.biases = [b-(eta/len(mini_batch))*nb - for b, nb in zip(self.biases, nabla_b)] - - def backprop(self, x, y): - """Return a tuple ``(nabla_b, nabla_w)`` representing the - gradient for the cost function C_x. ``nabla_b`` and - ``nabla_w`` are layer-by-layer lists of numpy arrays, similar - to ``self.biases`` and ``self.weights``.""" - nabla_b = [np.zeros(b.shape) for b in self.biases] - nabla_w = [np.zeros(w.shape) for w in self.weights] - # feedforward - activation = x - activations = [x] # list to store all the activations, layer by layer - zs = [] # list to store all the z vectors, layer by layer - for b, w in zip(self.biases, self.weights): - z = np.dot(w, activation)+b - zs.append(z) - activation = sigmoid(z) - activations.append(activation) - # backward pass - delta = self.cost_derivative(activations[-1], y) * \ - sigmoid_prime(zs[-1]) - nabla_b[-1] = delta - nabla_w[-1] = np.dot(delta, activations[-2].transpose()) - # Note that the variable l in the loop below is used a little - # differently to the notation in Chapter 2 of the book. Here, - # l = 1 means the last layer of neurons, l = 2 is the - # second-last layer, and so on. It's a renumbering of the - # scheme in the book, used here to take advantage of the fact - # that Python can use negative indices in lists. - for l in range(2, self.num_layers): - z = zs[-l] - sp = sigmoid_prime(z) - delta = np.dot(self.weights[-l+1].transpose(), delta) * sp - nabla_b[-l] = delta - nabla_w[-l] = np.dot(delta, activations[-l-1].transpose()) - return (nabla_b, nabla_w) - - def evaluate(self, test_data): - """Return the number of test inputs for which the neural - network outputs the correct result. Note that the neural - network's output is assumed to be the index of whichever - neuron in the final layer has the highest activation.""" - test_results = [(np.argmax(self.feedforward(x)), y) - for (x, y) in test_data] - return sum(int(x == y) for (x, y) in test_results) - - def cost_derivative(self, output_activations, y): - """Return the vector of partial derivatives \partial C_x / - \partial a for the output activations.""" - return (output_activations-y) - -#### Miscellaneous functions -def sigmoid(z): - """The sigmoid function.""" - return 1.0/(1.0+np.exp(-z)) - -def sigmoid_prime(z): - """Derivative of the sigmoid function.""" - return sigmoid(z)*(1-sigmoid(z)) + def __init__(self, sizes): + self.layers=len(sizes) + self.sizes=sizes + # sizes[:-1] means input side ,from first layer to the second-last + # sizes[1:] means output side, from second layer to the last layer + self.bias=[np.random.randn(b,1) for b in sizes[1:]] + # all weights are refer to second layer to the last layer. + self.weights=[np.random.randn(x,y) for x, y in zip(sizes[1:], sizes[:-1])] + + def feedforward(self, x): + for w, b in zip(self.weights, self.bias): + x = activateFunc(np.dot(w,x) + b) + return x + + def evaluate(self, test_data): + cal = 0 + for (x, y) in test_data: + if np.argmax(self.feedforward(x)) == y: + cal+=1 + return cal + + def backprop(self, x, y): + nabla_b = [np.zeros(b.shape) for b in self.bias] + nabla_w = [np.zeros(w.shape) for w in self.weights] + # feedforward + feedin = x + activates = [x] + zs = [] + for b, w in zip(self.bias, self.weights): + z = np.dot(w, feedin) + b + zs.append(z) + activated = activateFunc(z) + feedin=activated + activates.append(activated) + # backward pass + delta = (activates[-1] - y) * activateFunc_der(zs[-1]) + nabla_b[-1] = delta + nabla_w[-1] = np.dot(delta, activates[-2].transpose()) + + for l in range(2, self.layers): + z = zs[-l] + der = activateFunc_der(z) + delta = np.dot(self.weights[-l+1].transpose(), delta) * der + nabla_b[-l] = delta + nabla_w[-l] = np.dot(delta, activates[-l-1].transpose()) + return (nabla_b, nabla_w) + + def update_wb_by_batch(self, batch_size, rate): + nabla_b = [np.zeros(b.shape) for b in self.bias] + nabla_w = [np.zeros(w.shape) for w in self.weights] + for x, y in batch_size: + delta_b, delta_w =self.backprop(x,y) + # accumalte + nabla_b = [nb+dnb for nb, dnb in zip(nabla_b, delta_b)] + nabla_w = [nw+dnw for nw, dnw in zip(nabla_w, delta_w)] + self.bias=[b-(rate/len(batch_size)) * del_b for b, del_b in zip(self.bias, nabla_b)] + self.weights=[w-(rate/len(batch_size)) * del_w for w, del_w in zip(self.weights, nabla_w)] + + def SGD(self, tr_data, epochs, batch_size, rate, te_data=None): + tr_data = list(tr_data) + n = len(tr_data) + + if te_data: + te_data = list(te_data) + n_te = len(te_data) + + for j in range(epochs): + random.shuffle(tr_data) + batches=[ tr_data[k:k+batch_size] for k in range(0, n, batch_size)] + for batch in batches: + self.update_wb_by_batch(batch,rate) + if te_data: + print("Epoch {} : {} /{}".format(j, self.evaluate(te_data), n_te)) + else: + print("Epoch {} complete".format(j)) + + + +# this activateFunc is sigmod function +def activateFunc(z): + return 1.0 / (1+np.exp(-z)) + +def activateFunc_der(z): + return activateFunc(z)*(1-activateFunc(z)) + diff --git a/test.py b/test.py index d1f4996..05ef358 100644 --- a/test.py +++ b/test.py @@ -17,194 +17,191 @@ # ---------------------- # - read the input data: -''' import mnist_loader training_data, validation_data, test_data = mnist_loader.load_data_wrapper() training_data = list(training_data) -''' # --------------------- # - network.py example: -#import network - -''' -net = network.Network([784, 30, 10]) -net.SGD(training_data, 30, 10, 3.0, test_data=test_data) -''' - -# ---------------------- -# - network2.py example: -#import network2 - -''' -net = network2.Network([784, 30, 10], cost=network2.CrossEntropyCost) +import network + +net = network.Network([784, 40, 10]) +#net.SGD(training_data, 5, 10, 3.0, test_data=test_data) +net.SGD(training_data, 5, 10, 2.0, test_data) + +## ---------------------- +## - network2.py example: +##import network2 +# +#''' +#net = network2.Network([784, 30, 10], cost=network2.CrossEntropyCost) +##net.large_weight_initializer() +#net.SGD(training_data, 30, 10, 0.1, lmbda = 5.0,evaluation_data=validation_data, +# monitor_evaluation_accuracy=True) +#''' +# +## chapter 3 - Overfitting example - too many epochs of learning applied on small (1k samples) amount od data. +## Overfitting is treating noise as a signal. +#''' +#net = network2.Network([784, 30, 10], cost=network2.CrossEntropyCost) #net.large_weight_initializer() -net.SGD(training_data, 30, 10, 0.1, lmbda = 5.0,evaluation_data=validation_data, - monitor_evaluation_accuracy=True) -''' - -# chapter 3 - Overfitting example - too many epochs of learning applied on small (1k samples) amount od data. -# Overfitting is treating noise as a signal. -''' -net = network2.Network([784, 30, 10], cost=network2.CrossEntropyCost) -net.large_weight_initializer() -net.SGD(training_data[:1000], 400, 10, 0.5, evaluation_data=test_data, - monitor_evaluation_accuracy=True, - monitor_training_cost=True) -''' - -# chapter 3 - Regularization (weight decay) example 1 (only 1000 of training data and 30 hidden neurons) -''' -net = network2.Network([784, 30, 10], cost=network2.CrossEntropyCost) -net.large_weight_initializer() -net.SGD(training_data[:1000], 400, 10, 0.5, - evaluation_data=test_data, - lmbda = 0.1, # this is a regularization parameter - monitor_evaluation_cost=True, - monitor_evaluation_accuracy=True, - monitor_training_cost=True, - monitor_training_accuracy=True) -''' - -# chapter 3 - Early stopping implemented -''' -net = network2.Network([784, 30, 10], cost=network2.CrossEntropyCost) -net.SGD(training_data[:1000], 30, 10, 0.5, - lmbda=5.0, - evaluation_data=validation_data, - monitor_evaluation_accuracy=True, - monitor_training_cost=True, - early_stopping_n=10) -''' - -# chapter 4 - The vanishing gradient problem - deep networks are hard to train with simple SGD algorithm -# this network learns much slower than a shallow one. -''' -net = network2.Network([784, 30, 30, 30, 30, 10], cost=network2.CrossEntropyCost) -net.SGD(training_data, 30, 10, 0.1, - lmbda=5.0, - evaluation_data=validation_data, - monitor_evaluation_accuracy=True) -''' - - -# ---------------------- -# Theano and CUDA -# ---------------------- - -""" - This deep network uses Theano with GPU acceleration support. - I am using Ubuntu 16.04 with CUDA 7.5. - Tutorial: - http://deeplearning.net/software/theano/install_ubuntu.html#install-ubuntu - - The following command will update only Theano: - sudo pip install --upgrade --no-deps theano - - The following command will update Theano and Numpy/Scipy (warning bellow): - sudo pip install --upgrade theano - -""" - -""" - Below, there is a testing function to check whether your computations have been made on CPU or GPU. - If the result is 'Used the cpu' and you want to have it in gpu, do the following: - 1) install theano: - sudo python3.5 -m pip install Theano - 2) download and install the latest cuda: - https://developer.nvidia.com/cuda-downloads - I had some issues with that, so I followed this idea (better option is to download the 1,1GB package as .run file): - http://askubuntu.com/questions/760242/how-can-i-force-16-04-to-add-a-repository-even-if-it-isnt-considered-secure-eno - You may also want to grab the proper NVidia driver, choose it form there: - System Settings > Software & Updates > Additional Drivers. - 3) should work, run it with: - THEANO_FLAGS=mode=FAST_RUN,device=gpu,floatX=float32 python3.5 test.py - http://deeplearning.net/software/theano/tutorial/using_gpu.html - 4) Optionally, you can add cuDNN support from: - https://developer.nvidia.com/cudnn - - -""" -def testTheano(): - from theano import function, config, shared, sandbox - import theano.tensor as T - import numpy - import time - print("Testing Theano library...") - vlen = 10 * 30 * 768 # 10 x #cores x # threads per core - iters = 1000 - - rng = numpy.random.RandomState(22) - x = shared(numpy.asarray(rng.rand(vlen), config.floatX)) - f = function([], T.exp(x)) - print(f.maker.fgraph.toposort()) - t0 = time.time() - for i in range(iters): - r = f() - t1 = time.time() - print("Looping %d times took %f seconds" % (iters, t1 - t0)) - print("Result is %s" % (r,)) - if numpy.any([isinstance(x.op, T.Elemwise) for x in f.maker.fgraph.toposort()]): - print('Used the cpu') - else: - print('Used the gpu') -# Perform check: -#testTheano() - - -# ---------------------- -# - network3.py example: -import network3 -from network3 import Network, ConvPoolLayer, FullyConnectedLayer, SoftmaxLayer # softmax plus log-likelihood cost is more common in modern image classification networks. - -# read data: -training_data, validation_data, test_data = network3.load_data_shared() -# mini-batch size: -mini_batch_size = 10 - -# chapter 6 - shallow architecture using just a single hidden layer, containing 100 hidden neurons. -''' -net = Network([ - FullyConnectedLayer(n_in=784, n_out=100), - SoftmaxLayer(n_in=100, n_out=10)], mini_batch_size) -net.SGD(training_data, 60, mini_batch_size, 0.1, validation_data, test_data) -''' - -# chapter 6 - 5x5 local receptive fields, 20 feature maps, max-pooling layer 2x2 -''' -net = Network([ - ConvPoolLayer(image_shape=(mini_batch_size, 1, 28, 28), - filter_shape=(20, 1, 5, 5), - poolsize=(2, 2)), - FullyConnectedLayer(n_in=20*12*12, n_out=100), - SoftmaxLayer(n_in=100, n_out=10)], mini_batch_size) -net.SGD(training_data, 60, mini_batch_size, 0.1, validation_data, test_data) -''' - -# chapter 6 - inserting a second convolutional-pooling layer to the previous example => better accuracy -''' -net = Network([ - ConvPoolLayer(image_shape=(mini_batch_size, 1, 28, 28), - filter_shape=(20, 1, 5, 5), - poolsize=(2, 2)), - ConvPoolLayer(image_shape=(mini_batch_size, 20, 12, 12), - filter_shape=(40, 20, 5, 5), - poolsize=(2, 2)), - FullyConnectedLayer(n_in=40*4*4, n_out=100), - SoftmaxLayer(n_in=100, n_out=10)], mini_batch_size) -net.SGD(training_data, 60, mini_batch_size, 0.1, validation_data, test_data) -''' - -# chapter 6 - rectified linear units and some l2 regularization (lmbda=0.1) => even better accuracy -from network3 import ReLU -net = Network([ - ConvPoolLayer(image_shape=(mini_batch_size, 1, 28, 28), - filter_shape=(20, 1, 5, 5), - poolsize=(2, 2), - activation_fn=ReLU), - ConvPoolLayer(image_shape=(mini_batch_size, 20, 12, 12), - filter_shape=(40, 20, 5, 5), - poolsize=(2, 2), - activation_fn=ReLU), - FullyConnectedLayer(n_in=40*4*4, n_out=100, activation_fn=ReLU), - SoftmaxLayer(n_in=100, n_out=10)], mini_batch_size) -net.SGD(training_data, 60, mini_batch_size, 0.03, validation_data, test_data, lmbda=0.1) +#net.SGD(training_data[:1000], 400, 10, 0.5, evaluation_data=test_data, +# monitor_evaluation_accuracy=True, +# monitor_training_cost=True) +#''' +# +## chapter 3 - Regularization (weight decay) example 1 (only 1000 of training data and 30 hidden neurons) +#''' +#net = network2.Network([784, 30, 10], cost=network2.CrossEntropyCost) +#net.large_weight_initializer() +#net.SGD(training_data[:1000], 400, 10, 0.5, +# evaluation_data=test_data, +# lmbda = 0.1, # this is a regularization parameter +# monitor_evaluation_cost=True, +# monitor_evaluation_accuracy=True, +# monitor_training_cost=True, +# monitor_training_accuracy=True) +#''' +# +## chapter 3 - Early stopping implemented +#''' +#net = network2.Network([784, 30, 10], cost=network2.CrossEntropyCost) +#net.SGD(training_data[:1000], 30, 10, 0.5, +# lmbda=5.0, +# evaluation_data=validation_data, +# monitor_evaluation_accuracy=True, +# monitor_training_cost=True, +# early_stopping_n=10) +#''' +# +## chapter 4 - The vanishing gradient problem - deep networks are hard to train with simple SGD algorithm +## this network learns much slower than a shallow one. +#''' +#net = network2.Network([784, 30, 30, 30, 30, 10], cost=network2.CrossEntropyCost) +#net.SGD(training_data, 30, 10, 0.1, +# lmbda=5.0, +# evaluation_data=validation_data, +# monitor_evaluation_accuracy=True) +#''' +# +# +## ---------------------- +## Theano and CUDA +## ---------------------- +# +#""" +# This deep network uses Theano with GPU acceleration support. +# I am using Ubuntu 16.04 with CUDA 7.5. +# Tutorial: +# http://deeplearning.net/software/theano/install_ubuntu.html#install-ubuntu +# +# The following command will update only Theano: +# sudo pip install --upgrade --no-deps theano +# +# The following command will update Theano and Numpy/Scipy (warning bellow): +# sudo pip install --upgrade theano +# +#""" +# +#""" +# Below, there is a testing function to check whether your computations have been made on CPU or GPU. +# If the result is 'Used the cpu' and you want to have it in gpu, do the following: +# 1) install theano: +# sudo python3.5 -m pip install Theano +# 2) download and install the latest cuda: +# https://developer.nvidia.com/cuda-downloads +# I had some issues with that, so I followed this idea (better option is to download the 1,1GB package as .run file): +# http://askubuntu.com/questions/760242/how-can-i-force-16-04-to-add-a-repository-even-if-it-isnt-considered-secure-eno +# You may also want to grab the proper NVidia driver, choose it form there: +# System Settings > Software & Updates > Additional Drivers. +# 3) should work, run it with: +# THEANO_FLAGS=mode=FAST_RUN,device=gpu,floatX=float32 python3.5 test.py +# http://deeplearning.net/software/theano/tutorial/using_gpu.html +# 4) Optionally, you can add cuDNN support from: +# https://developer.nvidia.com/cudnn +# +# +#""" +#def testTheano(): +# from theano import function, config, shared, sandbox +# import theano.tensor as T +# import numpy +# import time +# print("Testing Theano library...") +# vlen = 10 * 30 * 768 # 10 x #cores x # threads per core +# iters = 1000 +# +# rng = numpy.random.RandomState(22) +# x = shared(numpy.asarray(rng.rand(vlen), config.floatX)) +# f = function([], T.exp(x)) +# print(f.maker.fgraph.toposort()) +# t0 = time.time() +# for i in range(iters): +# r = f() +# t1 = time.time() +# print("Looping %d times took %f seconds" % (iters, t1 - t0)) +# print("Result is %s" % (r,)) +# if numpy.any([isinstance(x.op, T.Elemwise) for x in f.maker.fgraph.toposort()]): +# print('Used the cpu') +# else: +# print('Used the gpu') +## Perform check: +##testTheano() +# +# +## ---------------------- +## - network3.py example: +#import network3 +#from network3 import Network, ConvPoolLayer, FullyConnectedLayer, SoftmaxLayer # softmax plus log-likelihood cost is more common in modern image classification networks. +# +## read data: +#training_data, validation_data, test_data = network3.load_data_shared() +## mini-batch size: +#mini_batch_size = 10 +# +## chapter 6 - shallow architecture using just a single hidden layer, containing 100 hidden neurons. +#''' +#net = Network([ +# FullyConnectedLayer(n_in=784, n_out=100), +# SoftmaxLayer(n_in=100, n_out=10)], mini_batch_size) +#net.SGD(training_data, 60, mini_batch_size, 0.1, validation_data, test_data) +#''' +# +## chapter 6 - 5x5 local receptive fields, 20 feature maps, max-pooling layer 2x2 +#''' +#net = Network([ +# ConvPoolLayer(image_shape=(mini_batch_size, 1, 28, 28), +# filter_shape=(20, 1, 5, 5), +# poolsize=(2, 2)), +# FullyConnectedLayer(n_in=20*12*12, n_out=100), +# SoftmaxLayer(n_in=100, n_out=10)], mini_batch_size) +#net.SGD(training_data, 60, mini_batch_size, 0.1, validation_data, test_data) +#''' +# +## chapter 6 - inserting a second convolutional-pooling layer to the previous example => better accuracy +#''' +#net = Network([ +# ConvPoolLayer(image_shape=(mini_batch_size, 1, 28, 28), +# filter_shape=(20, 1, 5, 5), +# poolsize=(2, 2)), +# ConvPoolLayer(image_shape=(mini_batch_size, 20, 12, 12), +# filter_shape=(40, 20, 5, 5), +# poolsize=(2, 2)), +# FullyConnectedLayer(n_in=40*4*4, n_out=100), +# SoftmaxLayer(n_in=100, n_out=10)], mini_batch_size) +#net.SGD(training_data, 60, mini_batch_size, 0.1, validation_data, test_data) +#''' +# +## chapter 6 - rectified linear units and some l2 regularization (lmbda=0.1) => even better accuracy +#from network3 import ReLU +#net = Network([ +# ConvPoolLayer(image_shape=(mini_batch_size, 1, 28, 28), +# filter_shape=(20, 1, 5, 5), +# poolsize=(2, 2), +# activation_fn=ReLU), +# ConvPoolLayer(image_shape=(mini_batch_size, 20, 12, 12), +# filter_shape=(40, 20, 5, 5), +# poolsize=(2, 2), +# activation_fn=ReLU), +# FullyConnectedLayer(n_in=40*4*4, n_out=100, activation_fn=ReLU), +# SoftmaxLayer(n_in=100, n_out=10)], mini_batch_size) +#net.SGD(training_data, 10, mini_batch_size, 0.03, validation_data, test_data, lmbda=0.1)