import numpy as np
import torch
import matplotlib.pyplot as plt
import torch.nn as nn
import torch.nn.functional as F
import time
import math 
plt.ion()

# GD (gradient descent) is one of the most important concepts of deep learning

# basic steps of model learning:
# 1. guess solution (forward)
# 2. calculate the error of this solution (predict vs. real)
# 3. learn from errors and modify parameters to make the solution better in the next iteration (backprop)

# we need a mathematical description of the error - we can represent it as a mathematical function ("error landscape") and then we need to find the minimum of this function

# how to find the minimum of the function? 
# can we use the derivative in the 1D case
# if we have more dimensions we are talking about gradient

# look at a simple example to find the minimum of a function using GD

def fx(x):
    return x**2 - 6*x + 1
    
def deriv(x):
    return 2*x - 6 
    
x = np.linspace(-14, 20, 2000)
print(x)

localmin = 15.0 #np.random.choice(x,1)
print("localmin", localmin)

plt.plot(x, fx(x))
plt.plot(localmin, fx(localmin), 'ro')
plt.show()
plt.pause(5)

learning_rate = 0.1 #1.0 #0.01
training_epochs = 50

for i in range(training_epochs):
    print("---------------- epoch", i)
    grad = deriv(localmin)
    move = learning_rate * grad
    localmin = localmin - move
    print("grad", grad)
    print("localmin new", localmin)   
    print("localmin fx", fx(localmin))
    plt.plot(localmin, fx(localmin), 'go')
    plt.show()
    plt.pause(5.0)
print(localmin)  

plt.savefig("out.png")


# https://discuss.pytorch.org/t/how-will-sgd-work-under-batch-data/167252/2
