looking at mee and poisson equation
This commit is contained in:
File diff suppressed because one or more lines are too long
@@ -0,0 +1,501 @@
|
||||
TITLE: Poisson equation in one dimension
|
||||
AUTHOR: Using standard methods and neural networks
|
||||
DATE: today
|
||||
|
||||
===== Solving the one dimensional Poisson equation =====
|
||||
|
||||
The Poisson equation for $g(x)$ in one dimension is
|
||||
|
||||
!bt
|
||||
\begin{equation} \label{poisson}
|
||||
-g''(x) = f(x)
|
||||
\end{equation}
|
||||
!et
|
||||
|
||||
where $f(x)$ is a given function for $x \in (0,1)$.
|
||||
|
||||
The boundary conditions that $g(x)$ is chosen to fulfill, are
|
||||
!bt
|
||||
\begin{align*}
|
||||
g(0) &= 0 \\
|
||||
g(1) &= 0
|
||||
\end{align*}
|
||||
!et
|
||||
|
||||
This equation can be solved numerically using programs where e.g Autograd and TensorFlow are used.
|
||||
The results from the networks can then be compared to the analytical solution.
|
||||
In addition, it could be interesting to see how a typical method for numerically solving second order ODEs compares to the neural networks.
|
||||
|
||||
!split
|
||||
===== The specific equation to solve for =====
|
||||
|
||||
Here, the function $g(x)$ to solve for follows the equation
|
||||
|
||||
!bt
|
||||
-g''(x) = f(x),\qquad x \in (0,1)
|
||||
!et
|
||||
|
||||
where $f(x)$ is a given function, along with the chosen conditions
|
||||
|
||||
!bt
|
||||
\begin{aligned}
|
||||
g(0) = g(1) = 0
|
||||
\end{aligned}\label{cond}
|
||||
!et
|
||||
|
||||
In this example, we consider the case when $f(x) = (3x + x^2)\exp(x)$.
|
||||
|
||||
For this case, a possible trial solution satisfying the conditions could be
|
||||
|
||||
!bt
|
||||
g_t(x) = x \cdot (1-x) \cdot N(P,x)
|
||||
!et
|
||||
|
||||
The analytical solution for this problem is
|
||||
|
||||
!bt
|
||||
g(x) = x(1 - x)\exp(x)
|
||||
!et
|
||||
|
||||
!split
|
||||
===== Solving the equation using Autograd =====
|
||||
|
||||
!bc pycod
|
||||
import autograd.numpy as np
|
||||
from autograd import grad, elementwise_grad
|
||||
import autograd.numpy.random as npr
|
||||
from matplotlib import pyplot as plt
|
||||
|
||||
def sigmoid(z):
|
||||
return 1/(1 + np.exp(-z))
|
||||
|
||||
def deep_neural_network(deep_params, x):
|
||||
# N_hidden is the number of hidden layers
|
||||
N_hidden = np.size(deep_params) - 1 # -1 since params consist of parameters to all the hidden layers AND the output layer
|
||||
|
||||
# Assumes input x being an one-dimensional array
|
||||
num_values = np.size(x)
|
||||
x = x.reshape(-1, num_values)
|
||||
|
||||
# Assume that the input layer does nothing to the input x
|
||||
x_input = x
|
||||
|
||||
# Due to multiple hidden layers, define a variable referencing to the
|
||||
# output of the previous layer:
|
||||
x_prev = x_input
|
||||
|
||||
## Hidden layers:
|
||||
|
||||
for l in range(N_hidden):
|
||||
# From the list of parameters P; find the correct weigths and bias for this layer
|
||||
w_hidden = deep_params[l]
|
||||
|
||||
# Add a row of ones to include bias
|
||||
x_prev = np.concatenate((np.ones((1,num_values)), x_prev ), axis = 0)
|
||||
|
||||
z_hidden = np.matmul(w_hidden, x_prev)
|
||||
x_hidden = sigmoid(z_hidden)
|
||||
|
||||
# Update x_prev such that next layer can use the output from this layer
|
||||
x_prev = x_hidden
|
||||
|
||||
## Output layer:
|
||||
|
||||
# Get the weights and bias for this layer
|
||||
w_output = deep_params[-1]
|
||||
|
||||
# Include bias:
|
||||
x_prev = np.concatenate((np.ones((1,num_values)), x_prev), axis = 0)
|
||||
|
||||
z_output = np.matmul(w_output, x_prev)
|
||||
x_output = z_output
|
||||
|
||||
return x_output
|
||||
|
||||
def solve_ode_deep_neural_network(x, num_neurons, num_iter, lmb):
|
||||
# num_hidden_neurons is now a list of number of neurons within each hidden layer
|
||||
|
||||
# Find the number of hidden layers:
|
||||
N_hidden = np.size(num_neurons)
|
||||
|
||||
## Set up initial weigths and biases
|
||||
|
||||
# Initialize the list of parameters:
|
||||
P = [None]*(N_hidden + 1) # + 1 to include the output layer
|
||||
|
||||
P[0] = npr.randn(num_neurons[0], 2 )
|
||||
for l in range(1,N_hidden):
|
||||
P[l] = npr.randn(num_neurons[l], num_neurons[l-1] + 1) # +1 to include bias
|
||||
|
||||
# For the output layer
|
||||
P[-1] = npr.randn(1, num_neurons[-1] + 1 ) # +1 since bias is included
|
||||
|
||||
print('Initial cost: %g'%cost_function_deep(P, x))
|
||||
|
||||
## Start finding the optimal weigths using gradient descent
|
||||
|
||||
# Find the Python function that represents the gradient of the cost function
|
||||
# w.r.t the 0-th input argument -- that is the weights and biases in the hidden and output layer
|
||||
cost_function_deep_grad = grad(cost_function_deep,0)
|
||||
|
||||
# Let the update be done num_iter times
|
||||
for i in range(num_iter):
|
||||
# Evaluate the gradient at the current weights and biases in P.
|
||||
# The cost_grad consist now of N_hidden + 1 arrays; the gradient w.r.t the weights and biases
|
||||
# in the hidden layers and output layers evaluated at x.
|
||||
cost_deep_grad = cost_function_deep_grad(P, x)
|
||||
|
||||
for l in range(N_hidden+1):
|
||||
P[l] = P[l] - lmb * cost_deep_grad[l]
|
||||
|
||||
print('Final cost: %g'%cost_function_deep(P, x))
|
||||
|
||||
return P
|
||||
|
||||
## Set up the cost function specified for this Poisson equation:
|
||||
|
||||
# The right side of the ODE
|
||||
def f(x):
|
||||
return (3*x + x**2)*np.exp(x)
|
||||
|
||||
def cost_function_deep(P, x):
|
||||
|
||||
# Evaluate the trial function with the current parameters P
|
||||
g_t = g_trial_deep(x,P)
|
||||
|
||||
# Find the derivative w.r.t x of the trial function
|
||||
d2_g_t = elementwise_grad(elementwise_grad(g_trial_deep,0))(x,P)
|
||||
|
||||
right_side = f(x)
|
||||
|
||||
err_sqr = (-d2_g_t - right_side)**2
|
||||
cost_sum = np.sum(err_sqr)
|
||||
|
||||
return cost_sum/np.size(err_sqr)
|
||||
|
||||
# The trial solution:
|
||||
def g_trial_deep(x,P):
|
||||
return x*(1-x)*deep_neural_network(P,x)
|
||||
|
||||
# The analytic solution;
|
||||
def g_analytic(x):
|
||||
return x*(1-x)*np.exp(x)
|
||||
|
||||
if __name__ == '__main__':
|
||||
npr.seed(4155)
|
||||
|
||||
## Decide the vales of arguments to the function to solve
|
||||
Nx = 10
|
||||
x = np.linspace(0,1, Nx)
|
||||
|
||||
## Set up the initial parameters
|
||||
num_hidden_neurons = [200,100]
|
||||
num_iter = 1000
|
||||
lmb = 1e-3
|
||||
|
||||
P = solve_ode_deep_neural_network(x, num_hidden_neurons, num_iter, lmb)
|
||||
|
||||
g_dnn_ag = g_trial_deep(x,P)
|
||||
g_analytical = g_analytic(x)
|
||||
|
||||
# Find the maximum absolute difference between the solutons:
|
||||
max_diff = np.max(np.abs(g_dnn_ag - g_analytical))
|
||||
print("The max absolute difference between the solutions is: %g"%max_diff)
|
||||
|
||||
plt.figure(figsize=(10,10))
|
||||
|
||||
plt.title('Performance of neural network solving an ODE compared to the analytical solution')
|
||||
plt.plot(x, g_analytical)
|
||||
plt.plot(x, g_dnn_ag[0,:])
|
||||
plt.legend(['analytical','nn'])
|
||||
plt.xlabel('x')
|
||||
plt.ylabel('g(x)')
|
||||
plt.show()
|
||||
!ec
|
||||
|
||||
!split
|
||||
===== Comparing with a numerical scheme =====
|
||||
|
||||
The Poisson equation is possible to solve using a standard Taylor expansion to approximate the second derivative.
|
||||
|
||||
Using Taylor series, the second derivative can be expressed as
|
||||
|
||||
$$
|
||||
g''(x) = \frac{g(x + \Delta x) - 2g(x) + g(x-\Delta x)}{\Delta x^2} + E_{\Delta x}(x)
|
||||
$$
|
||||
|
||||
where $\Delta x$ is a small step size and $E_{\Delta x}(x)$ being the error term.
|
||||
|
||||
Looking away from the error terms gives an approximation to the second derivative:
|
||||
|
||||
!bt
|
||||
\begin{equation} \label{approx}
|
||||
g''(x) \approx \frac{g(x + \Delta x) - 2g(x) + g(x-\Delta x)}{\Delta x^2}
|
||||
\end{equation}
|
||||
!et
|
||||
|
||||
If $x_i = i \Delta x = x_{i-1} + \Delta x$ and $g_i = g(x_i)$ for $i = 1,\dots N_x - 2$ with $N_x$ being the number of values for $x$, (ref{approx}) becomes
|
||||
|
||||
!bt
|
||||
\begin{aligned}
|
||||
g''(x_i) &\approx \frac{g(x_i + \Delta x) - 2g(x_i) + g(x_i -\Delta x)}{\Delta x^2} \\
|
||||
&= \frac{g_{i+1} - 2g_i + g_{i-1}}{\Delta x^2}
|
||||
\end{aligned}
|
||||
!et
|
||||
|
||||
Since we know from our problem that
|
||||
|
||||
!bt
|
||||
\begin{aligned}
|
||||
-g''(x) &= f(x) \\
|
||||
&= (3x + x^2)\exp(x)
|
||||
\end{aligned}
|
||||
!et
|
||||
|
||||
along with the conditions $g(0) = g(1) = 0$,
|
||||
the following scheme can be used to find an approximate solution for $g(x)$ numerically:
|
||||
|
||||
!bt
|
||||
\begin{equation}
|
||||
\begin{aligned}
|
||||
-\Big( \frac{g_{i+1} - 2g_i + g_{i-1}}{\Delta x^2} \Big) &= f(x_i) \\
|
||||
-g_{i+1} + 2g_i - g_{i-1} &= \Delta x^2 f(x_i)
|
||||
\end{aligned}
|
||||
\end{equation} \label{odesys}
|
||||
!et
|
||||
|
||||
for $i = 1, \dots, N_x - 2$ where $g_0 = g_{N_x - 1} = 0$ and $f(x_i) = (3x_i + x_i^2)\exp(x_i)$, which is given for our specific problem.
|
||||
|
||||
The equation can be rewritten into a matrix equation and this is where the MEE could come into play.
|
||||
|
||||
!bt
|
||||
\begin{aligned}
|
||||
\begin{pmatrix}
|
||||
2 & -1 & 0 & \dots & 0 \\
|
||||
-1 & 2 & -1 & \dots & 0 \\
|
||||
\vdots & & \ddots & & \vdots \\
|
||||
0 & \dots & -1 & 2 & -1 \\
|
||||
0 & \dots & 0 & -1 & 2\\
|
||||
\end{pmatrix}
|
||||
\begin{pmatrix}
|
||||
g_1 \\
|
||||
g_2 \\
|
||||
\vdots \\
|
||||
g_{N_x - 3} \\
|
||||
g_{N_x - 2}
|
||||
\end{pmatrix}
|
||||
&=
|
||||
\Delta x^2
|
||||
\begin{pmatrix}
|
||||
f(x_1) \\
|
||||
f(x_2) \\
|
||||
\vdots \\
|
||||
f(x_{N_x - 3}) \\
|
||||
f(x_{N_x - 2})
|
||||
\end{pmatrix} \\
|
||||
\bm{A}\bm{g} &= \bm{f},
|
||||
\end{aligned}
|
||||
!et
|
||||
|
||||
which makes it possible to solve for the vector $\bm{g}$.
|
||||
|
||||
!split
|
||||
===== Setting up the code =====
|
||||
|
||||
We can then compare the result from this numerical scheme with the output from our network using Autograd:
|
||||
|
||||
!bc pycod
|
||||
import autograd.numpy as np
|
||||
from autograd import grad, elementwise_grad
|
||||
import autograd.numpy.random as npr
|
||||
from matplotlib import pyplot as plt
|
||||
|
||||
def sigmoid(z):
|
||||
return 1/(1 + np.exp(-z))
|
||||
|
||||
def deep_neural_network(deep_params, x):
|
||||
# N_hidden is the number of hidden layers
|
||||
N_hidden = np.size(deep_params) - 1 # -1 since params consist of parameters to all the hidden layers AND the output layer
|
||||
|
||||
# Assumes input x being an one-dimensional array
|
||||
num_values = np.size(x)
|
||||
x = x.reshape(-1, num_values)
|
||||
|
||||
# Assume that the input layer does nothing to the input x
|
||||
x_input = x
|
||||
|
||||
# Due to multiple hidden layers, define a variable referencing to the
|
||||
# output of the previous layer:
|
||||
x_prev = x_input
|
||||
|
||||
## Hidden layers:
|
||||
|
||||
for l in range(N_hidden):
|
||||
# From the list of parameters P; find the correct weigths and bias for this layer
|
||||
w_hidden = deep_params[l]
|
||||
|
||||
# Add a row of ones to include bias
|
||||
x_prev = np.concatenate((np.ones((1,num_values)), x_prev ), axis = 0)
|
||||
|
||||
z_hidden = np.matmul(w_hidden, x_prev)
|
||||
x_hidden = sigmoid(z_hidden)
|
||||
|
||||
# Update x_prev such that next layer can use the output from this layer
|
||||
x_prev = x_hidden
|
||||
|
||||
## Output layer:
|
||||
|
||||
# Get the weights and bias for this layer
|
||||
w_output = deep_params[-1]
|
||||
|
||||
# Include bias:
|
||||
x_prev = np.concatenate((np.ones((1,num_values)), x_prev), axis = 0)
|
||||
|
||||
z_output = np.matmul(w_output, x_prev)
|
||||
x_output = z_output
|
||||
|
||||
return x_output
|
||||
|
||||
def solve_ode_deep_neural_network(x, num_neurons, num_iter, lmb):
|
||||
# num_hidden_neurons is now a list of number of neurons within each hidden layer
|
||||
|
||||
# Find the number of hidden layers:
|
||||
N_hidden = np.size(num_neurons)
|
||||
|
||||
## Set up initial weigths and biases
|
||||
|
||||
# Initialize the list of parameters:
|
||||
P = [None]*(N_hidden + 1) # + 1 to include the output layer
|
||||
|
||||
P[0] = npr.randn(num_neurons[0], 2 )
|
||||
for l in range(1,N_hidden):
|
||||
P[l] = npr.randn(num_neurons[l], num_neurons[l-1] + 1) # +1 to include bias
|
||||
|
||||
# For the output layer
|
||||
P[-1] = npr.randn(1, num_neurons[-1] + 1 ) # +1 since bias is included
|
||||
|
||||
print('Initial cost: %g'%cost_function_deep(P, x))
|
||||
|
||||
## Start finding the optimal weigths using gradient descent
|
||||
|
||||
# Find the Python function that represents the gradient of the cost function
|
||||
# w.r.t the 0-th input argument -- that is the weights and biases in the hidden and output layer
|
||||
cost_function_deep_grad = grad(cost_function_deep,0)
|
||||
|
||||
# Let the update be done num_iter times
|
||||
for i in range(num_iter):
|
||||
# Evaluate the gradient at the current weights and biases in P.
|
||||
# The cost_grad consist now of N_hidden + 1 arrays; the gradient w.r.t the weights and biases
|
||||
# in the hidden layers and output layers evaluated at x.
|
||||
cost_deep_grad = cost_function_deep_grad(P, x)
|
||||
|
||||
for l in range(N_hidden+1):
|
||||
P[l] = P[l] - lmb * cost_deep_grad[l]
|
||||
|
||||
print('Final cost: %g'%cost_function_deep(P, x))
|
||||
|
||||
return P
|
||||
|
||||
## Set up the cost function specified for this Poisson equation:
|
||||
|
||||
# The right side of the ODE
|
||||
def f(x):
|
||||
return (3*x + x**2)*np.exp(x)
|
||||
|
||||
def cost_function_deep(P, x):
|
||||
|
||||
# Evaluate the trial function with the current parameters P
|
||||
g_t = g_trial_deep(x,P)
|
||||
|
||||
# Find the derivative w.r.t x of the trial function
|
||||
d2_g_t = elementwise_grad(elementwise_grad(g_trial_deep,0))(x,P)
|
||||
|
||||
right_side = f(x)
|
||||
|
||||
err_sqr = (-d2_g_t - right_side)**2
|
||||
cost_sum = np.sum(err_sqr)
|
||||
|
||||
return cost_sum/np.size(err_sqr)
|
||||
|
||||
# The trial solution:
|
||||
def g_trial_deep(x,P):
|
||||
return x*(1-x)*deep_neural_network(P,x)
|
||||
|
||||
# The analytic solution;
|
||||
def g_analytic(x):
|
||||
return x*(1-x)*np.exp(x)
|
||||
|
||||
if __name__ == '__main__':
|
||||
npr.seed(4155)
|
||||
|
||||
## Decide the vales of arguments to the function to solve
|
||||
Nx = 10
|
||||
x = np.linspace(0,1, Nx)
|
||||
|
||||
## Set up the initial parameters
|
||||
num_hidden_neurons = [200,100]
|
||||
num_iter = 1000
|
||||
lmb = 1e-3
|
||||
|
||||
P = solve_ode_deep_neural_network(x, num_hidden_neurons, num_iter, lmb)
|
||||
|
||||
g_dnn_ag = g_trial_deep(x,P)
|
||||
g_analytical = g_analytic(x)
|
||||
|
||||
# Find the maximum absolute difference between the solutons:
|
||||
|
||||
plt.figure(figsize=(10,10))
|
||||
|
||||
plt.title('Performance of neural network solving an ODE compared to the analytical solution')
|
||||
plt.plot(x, g_analytical)
|
||||
plt.plot(x, g_dnn_ag[0,:])
|
||||
plt.legend(['analytical','nn'])
|
||||
plt.xlabel('x')
|
||||
plt.ylabel('g(x)')
|
||||
|
||||
## Perform the computation using the numerical scheme
|
||||
|
||||
dx = 1/(Nx - 1)
|
||||
|
||||
# Set up the matrix A
|
||||
A = np.zeros((Nx-2,Nx-2))
|
||||
|
||||
A[0,0] = 2
|
||||
A[0,1] = -1
|
||||
|
||||
for i in range(1,Nx-3):
|
||||
A[i,i-1] = -1
|
||||
A[i,i] = 2
|
||||
A[i,i+1] = -1
|
||||
|
||||
A[Nx - 3, Nx - 4] = -1
|
||||
A[Nx - 3, Nx - 3] = 2
|
||||
|
||||
# Set up the vector f
|
||||
f_vec = dx**2 * f(x[1:-1])
|
||||
|
||||
# Solve the equation
|
||||
g_res = np.linalg.solve(A,f_vec)
|
||||
|
||||
g_vec = np.zeros(Nx)
|
||||
g_vec[1:-1] = g_res
|
||||
|
||||
# Print the differences between each method
|
||||
max_diff1 = np.max(np.abs(g_dnn_ag - g_analytical))
|
||||
max_diff2 = np.max(np.abs(g_vec - g_analytical))
|
||||
print("The max absolute difference between the analytical solution and DNN Autograd: %g"%max_diff1)
|
||||
print("The max absolute difference between the analytical solution and numerical scheme: %g"%max_diff2)
|
||||
|
||||
# Plot the results
|
||||
plt.figure(figsize=(10,10))
|
||||
|
||||
plt.plot(x,g_vec)
|
||||
plt.plot(x,g_analytical)
|
||||
plt.plot(x,g_dnn_ag[0,:])
|
||||
|
||||
plt.legend(['numerical scheme','analytical','dnn'])
|
||||
plt.show()
|
||||
|
||||
!ec
|
||||
|
||||
|
||||
File diff suppressed because one or more lines are too long
Reference in New Issue
Block a user