import numpy as np
from sklearn.linear_model import LogisticRegression
import scipy.optimize as opt
import matplotlib.pyplot as plt
%matplotlib inline
Creating your own Logistic Regression Creating the functions:
These are just the definintions for the functions, we will use them later after reading in the data Create the Hypothesis Function
#FILL IN THE CODE THAT IS MISSING
def h(X,theta):
z = x*theta.T
#np.exp(value) will evaluate e^(value)
sigmoid = 1/1+np.exp(-z)
return sigmoid
Create the Cost Function
#FILL IN THE CODE THAT IS MISSING
def J(theta, X, y):
#convert theta, X, y to a np matrix using np.matrix(listName)
theta = np.matrix(theta)
X = np.matrix(X)
y = np.matrix(y)
#The first part of the Sum Function ylog(h(x,theta) use np.multiply(matrix1,matrix2)
z = X*theta.T
sigmoid = 1 / (1 + np.exp(-z))
first = np.multiply(y, np.log(h(x,theta)))
#The second part of the Sum Function (1-y)log(1-h(x,theta)) use np.multiply(matrix1,matrix2)
second = np.multiply((1-y),np.log(1-h(x,theta)))
total = np.sum(first - second) / (len(X))
return total
Create the Gradient Function
#FILL IN THE CODE THAT IS MISSING
def G(theta, X_train, y_train):
#Convert X_train, y_train, theta to a matrix using np.matrix(listName)
theta = np.matrix(theta)
X_train = np.matrix(X_train)
y_train = np.matrix(y_train)
parameters = int(theta.ravel().shape[1])
grad = np.zeros(parameters)
#Compute the error h(xtrain,theta) - ytrain
error = h(X_train,theta) - ytrain
for i in range(parameters):
term = np.multiply(error, X_train[:,i])
grad[i] = np.sum(term) / len(X_train)
return grad
Create a prediction function for a piece of data
#FILL IN THE CODE THAT IS MISSING
def predict(theta,xtest):
theta = np.matrix(theta)
#Calculate the probabilities of all the test data sent by calling the hypothesis function sending xtest and theta
probabilities = h(xtest,theta)
#Calculate the prediction based on the probabilities
predictions = []
for probability in probabilities:
#prediction is equal to 1 if the probability is >= 0.5 otherwise prediction is equal to 0
if probability >=0.5 :
prediction = 1
else:
prediction = 0
predictions.append(prediction)
return predictions
Create the Accuracy Function Takes two arrays and compares the results to check how many are equal
def score(theta, xtest, ytest):
predictions = predict(theta, xtest)
correct = 0
numValues = len(ytest)
for i in range(0,numValues):
if predictions[i] == ytest[i]:
correct = correct + 1
a = correct / numValues * 100
return a
Using the Functions:
Read all the data in from the ex2data1.txt that is included in your download. Store your data in regular python lists named:
x1 which represents the score on Exam 1 (First column) x2 which represents the score on Exam 2 (2nd column) y which respresents if they were accepted or not. 1 = accepted, 0 = not accepted (3rd Column)
#FILL IN THE CODE To Read in the data to x1, x2, y
import os
path = os.getcwd() + '\ex2data1.txt'
x1=[]
x2=[]
y=[]
file = open(path,"r")
while True:
line = file.readline()
if not line:
break
else:
values = line.split(",")
x1.append(float(values[0]))
x2.append(float(values[1]))
y.append(float(values[2].rstrip("\n")))
file.close()
Get the size of the data set using the len(listName) function
#FILL IN THE CODE THAT IS MISSING
m1 = len(x1)
print(m1)
m2 = len(x2)
print(m2)
m3 = len(y)
print(m3)
Convert the x1 and x2 data to a np array using np.array(listName)
#FILL IN THE CODE THAT IS MISSING
x1 = np.array(x1)
x2 = np.array(x2)
Create an array of ones using np.ones((#ofRows,1))
#FILL IN THE CODE THAT IS MISSING
ones = np.ones((100,1))
Combine them together using np.column_stack((col1Name, col2Name, col3Name)). The ones are first, then x1, then x2
#FILL IN THE CODE THAT IS MISSING
x = np.column_stack((ones,x1,x2))
Convert the y data to the transpose of a np array (Needed for the matrix multiplication that the functions do)
y = np.array([y]).T
Create the theta array filled with zeros as the initial guesses using np.zeros(#ofWeights)
#FILL IN THE CODE THAT IS MISSING
theta = np.zeros(3)
print(theta)
Test the cost function
This is the original cost value with the theta values as 0 If your code is correct it should display approximately 0.69
cost = J(theta,x,y)
print(cost)
My output: nan
RuntimeWarning: invalid value encountered in log second = np.multiply((1-y),np.log(1-h(x,theta)))
This is what someone said I need to do:
Logistic regression
in the h function you need to type in the formulas for z and sigmoid exactly as they are written in the equations above it.
In the J function you need to calculate
first -> use -y as the first matrix, and use log (h(x,theta)) as the second matrix in the np multiplication
second -> use (1-y) as the first matrix and use log (1-h(x,theta)) as the second matrix in the multiplication
In the predict function
to calculate the probabilities you just call the h function and send xtest and theta to it
What went wrong?
answer clearly and show me with my code, please.