commit 1239a1d9372508a499d4044d91173229409bcb35 Author: Dib, Gerges Date: Fri Feb 23 17:16:58 2018 -0800 First commit diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..848d2a0 --- /dev/null +++ b/.gitignore @@ -0,0 +1,110 @@ + +### Python template +# Byte-compiled / optimized / DLL files +__pycache__/ +*.py[cod] +*$py.class + +# C extensions +*.so + +# Distribution / packaging +.Python +build/ +develop-eggs/ +dist/ +downloads/ +eggs/ +.eggs/ +lib/ +lib64/ +parts/ +sdist/ +var/ +wheels/ +*.egg-info/ +.installed.cfg +*.egg +MANIFEST + +# PyInstaller +# Usually these files are written by a python script from a template +# before PyInstaller builds the exe, so as to inject date/other infos into it. +*.manifest +*.spec + +# Installer logs +pip-log.txt +pip-delete-this-directory.txt + +# Unit test / coverage reports +htmlcov/ +.tox/ +.coverage +.coverage.* +.cache +nosetests.xml +coverage.xml +*.cover +.hypothesis/ + +# Translations +*.mo +*.pot + +# Django stuff: +*.log +.static_storage/ +.media/ +local_settings.py + +# Flask stuff: +instance/ +.webassets-cache + +# Scrapy stuff: +.scrapy + +# Sphinx documentation +docs/_build/ + +# PyBuilder +target/ + +# Jupyter Notebook +.ipynb_checkpoints + +# pyenv +.python-version + +# celery beat schedule file +celerybeat-schedule + +# SageMath parsed files +*.sage.py + +# Environments +.env +.venv +env/ +venv/ +ENV/ +env.bak/ +venv.bak/ + +# Spyder project settings +.spyderproject +.spyproject + +# Rope project settings +.ropeproject + +# mkdocs documentation +/site + +# mypy +.mypy_cache/ + +*.pkl +*-solved.ipynb + diff --git a/Exercise1/Data/ex1data1.txt b/Exercise1/Data/ex1data1.txt new file mode 100755 index 0000000..0f88ccb --- /dev/null +++ b/Exercise1/Data/ex1data1.txt @@ -0,0 +1,97 @@ +6.1101,17.592 +5.5277,9.1302 +8.5186,13.662 +7.0032,11.854 +5.8598,6.8233 +8.3829,11.886 +7.4764,4.3483 +8.5781,12 +6.4862,6.5987 +5.0546,3.8166 +5.7107,3.2522 +14.164,15.505 +5.734,3.1551 +8.4084,7.2258 +5.6407,0.71618 +5.3794,3.5129 +6.3654,5.3048 +5.1301,0.56077 +6.4296,3.6518 +7.0708,5.3893 +6.1891,3.1386 +20.27,21.767 +5.4901,4.263 +6.3261,5.1875 +5.5649,3.0825 +18.945,22.638 +12.828,13.501 +10.957,7.0467 +13.176,14.692 +22.203,24.147 +5.2524,-1.22 +6.5894,5.9966 +9.2482,12.134 +5.8918,1.8495 +8.2111,6.5426 +7.9334,4.5623 +8.0959,4.1164 +5.6063,3.3928 +12.836,10.117 +6.3534,5.4974 +5.4069,0.55657 +6.8825,3.9115 +11.708,5.3854 +5.7737,2.4406 +7.8247,6.7318 +7.0931,1.0463 +5.0702,5.1337 +5.8014,1.844 +11.7,8.0043 +5.5416,1.0179 +7.5402,6.7504 +5.3077,1.8396 +7.4239,4.2885 +7.6031,4.9981 +6.3328,1.4233 +6.3589,-1.4211 +6.2742,2.4756 +5.6397,4.6042 +9.3102,3.9624 +9.4536,5.4141 +8.8254,5.1694 +5.1793,-0.74279 +21.279,17.929 +14.908,12.054 +18.959,17.054 +7.2182,4.8852 +8.2951,5.7442 +10.236,7.7754 +5.4994,1.0173 +20.341,20.992 +10.136,6.6799 +7.3345,4.0259 +6.0062,1.2784 +7.2259,3.3411 +5.0269,-2.6807 +6.5479,0.29678 +7.5386,3.8845 +5.0365,5.7014 +10.274,6.7526 +5.1077,2.0576 +5.7292,0.47953 +5.1884,0.20421 +6.3557,0.67861 +9.7687,7.5435 +6.5159,5.3436 +8.5172,4.2415 +9.1802,6.7981 +6.002,0.92695 +5.5204,0.152 +5.0594,2.8214 +5.7077,1.8451 +7.6366,4.2959 +5.8707,7.2029 +5.3054,1.9869 +8.2934,0.14454 +13.394,9.0551 +5.4369,0.61705 diff --git a/Exercise1/Data/ex1data2.txt b/Exercise1/Data/ex1data2.txt new file mode 100755 index 0000000..79e9a80 --- /dev/null +++ b/Exercise1/Data/ex1data2.txt @@ -0,0 +1,47 @@ +2104,3,399900 +1600,3,329900 +2400,3,369000 +1416,2,232000 +3000,4,539900 +1985,4,299900 +1534,3,314900 +1427,3,198999 +1380,3,212000 +1494,3,242500 +1940,4,239999 +2000,3,347000 +1890,3,329999 +4478,5,699900 +1268,3,259900 +2300,4,449900 +1320,2,299900 +1236,3,199900 +2609,4,499998 +3031,4,599000 +1767,3,252900 +1888,2,255000 +1604,3,242900 +1962,4,259900 +3890,3,573900 +1100,3,249900 +1458,3,464500 +2526,3,469000 +2200,3,475000 +2637,3,299900 +1839,2,349900 +1000,1,169900 +2040,4,314900 +3137,3,579900 +1811,4,285900 +1437,3,249900 +1239,3,229900 +2132,4,345000 +4215,4,549000 +2162,4,287000 +1664,2,368500 +2238,3,329900 +2567,4,314000 +1200,3,299000 +852,2,179900 +1852,4,299900 +1203,3,239500 diff --git a/Exercise1/Figures/cost_function.png b/Exercise1/Figures/cost_function.png new file mode 100755 index 0000000..b5f175f Binary files /dev/null and b/Exercise1/Figures/cost_function.png differ diff --git a/Exercise1/Figures/dataset1.png b/Exercise1/Figures/dataset1.png new file mode 100755 index 0000000..8bded89 Binary files /dev/null and b/Exercise1/Figures/dataset1.png differ diff --git a/Exercise1/Figures/learning_rate.png b/Exercise1/Figures/learning_rate.png new file mode 100755 index 0000000..8701bb9 Binary files /dev/null and b/Exercise1/Figures/learning_rate.png differ diff --git a/Exercise1/Figures/regression_result.png b/Exercise1/Figures/regression_result.png new file mode 100755 index 0000000..622a6ec Binary files /dev/null and b/Exercise1/Figures/regression_result.png differ diff --git a/Exercise1/exercise1.ipynb b/Exercise1/exercise1.ipynb new file mode 100755 index 0000000..fe13e3d --- /dev/null +++ b/Exercise1/exercise1.ipynb @@ -0,0 +1,1307 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Programming Exercise 1: Linear Regression\n", + "\n", + "## Introduction\n", + "\n", + "In this exercise, you will implement linear regression and get to see it work on data. Before starting on this programming exercise, we strongly recommend watching the video lectures and completing the review questions for the associated topics.\n", + "\n", + "All the information you need for solving this assignment is in this notebook, and all the code you will be implementing will take place within this notebook. The assignment can be promptly submitted to the coursera grader directly from this notebook (code and instructions are included below).\n", + "\n", + "Before we begin with the exercises, we need to import all libraries required for this programming exercise. Throughout the course, we will be using [`numpy`](http://www.numpy.org/) for all arrays and matrix operations, and [`matplotlib`](https://matplotlib.org/) for plotting.\n", + "\n", + "You can find instructions on how to install required libraries in the README file in the [github repository](https://github.com/dibgerge/ml-coursera-python-assignments)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# used for manipulating directory paths\n", + "import os\n", + "\n", + "# Scientific and vector computation for python\n", + "import numpy as np\n", + "\n", + "# Plotting library\n", + "from matplotlib import pyplot\n", + "from mpl_toolkits.mplot3d import Axes3D # needed to plot 3-D surfaces\n", + "\n", + "# library written for this exercise providing additional functions for assignment submission, and others\n", + "import utils \n", + "\n", + "# define the submission/grader object for this exercise\n", + "grader = utils.Grader()\n", + "\n", + "# tells matplotlib to embed plots within the notebook\n", + "%matplotlib inline" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Submission and Grading\n", + "\n", + "After completing each part of the assignment, be sure to submit your solutions to the grader.\n", + "\n", + "For this programming exercise, you are only required to complete the first part of the exercise to implement linear regression with one variable. The second part of the exercise, which is optional, covers linear regression with multiple variables. The following is a breakdown of how each part of this exercise is scored.\n", + "\n", + "**Required Exercises**\n", + "\n", + "| Section | Part |Submitted Function | Points \n", + "|---------|:- |:- | :-: \n", + "| 1 | [Warm up exercise](#section1) | [`warmUpExercise`](#warmUpExercise) | 10 \n", + "| 2 | [Compute cost for one variable](#section2) | [`computeCost`](#computeCost) | 40 \n", + "| 3 | [Gradient descent for one variable](#section3) | [`gradientDescent`](#gradientDescent) | 50 \n", + "| | Total Points | | 100 \n", + "\n", + "**Optional Exercises**\n", + "\n", + "| Section | Part | Submitted Function | Points |\n", + "|:-------:|:- |:-: | :-: |\n", + "| 4 | [Feature normalization](#section4) | [`featureNormalize`](#featureNormalize) | 0 |\n", + "| 5 | [Compute cost for multiple variables](#section5) | [`computeCostMulti`](#computeCostMulti) | 0 |\n", + "| 6 | [Gradient descent for multiple variables](#section5) | [`gradientDescentMulti`](#gradientDescentMulti) |0 |\n", + "| 7 | [Normal Equations](#section7) | [`normalEqn`](#normalEqn) | 0 |\n", + "\n", + "You are allowed to submit your solutions multiple times, and we will take only the highest score into consideration.\n", + "\n", + "
\n", + "At the end of each section in this notebook, we have a cell which contains code for submitting the solutions thus far to the grader. Execute the cell to see your score up to the current section. For all your work to be submitted properly, you must execute those cells at least once. They must also be re-executed everytime the submitted function is updated.\n", + "
\n", + "\n", + "\n", + "## Debugging\n", + "\n", + "Here are some things to keep in mind throughout this exercise:\n", + "\n", + "- Python array indices start from zero, not one (contrary to OCTAVE/MATLAB). \n", + "\n", + "- There is an important distinction between python arrays (called `list` or `tuple`) and `numpy` arrays. You should use `numpy` arrays in all your computations. Vector/matrix operations work only with `numpy` arrays. Python lists do not support vector operations (you need to use for loops).\n", + "\n", + "- If you are seeing many errors at runtime, inspect your matrix operations to make sure that you are adding and multiplying matrices of compatible dimensions. Printing the dimensions of `numpy` arrays using the `shape` property will help you debug.\n", + "\n", + "- By default, `numpy` interprets math operators to be element-wise operators. If you want to do matrix multiplication, you need to use the `dot` function in `numpy`. For, example if `A` and `B` are two `numpy` matrices, then the matrix operation AB is `np.dot(A, B)`. Note that for 2-dimensional matrices or vectors (1-dimensional), this is also equivalent to `A@B` (requires python >= 3.5)." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "## 1 Simple python and `numpy` function\n", + "\n", + "The first part of this assignment gives you practice with python and `numpy` syntax and the homework submission process. In the next cell, you will find the outline of a `python` function. Modify it to return a 5 x 5 identity matrix by filling in the following code:\n", + "\n", + "```python\n", + "A = np.eye(5)\n", + "```\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def warmUpExercise():\n", + " \"\"\"\n", + " Example function in Python which computes the identity matrix.\n", + " \n", + " Returns\n", + " -------\n", + " A : array_like\n", + " The 5x5 identity matrix.\n", + " \n", + " Instructions\n", + " ------------\n", + " Return the 5x5 identity matrix.\n", + " \"\"\" \n", + " # ======== YOUR CODE HERE ======\n", + " A = [] # modify this line\n", + " \n", + " # ==============================\n", + " return A" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The previous cell only defines the function `warmUpExercise`. We can now run it by executing the following cell to see its output. You should see output similar to the following:\n", + "\n", + "```python\n", + "array([[ 1., 0., 0., 0., 0.],\n", + " [ 0., 1., 0., 0., 0.],\n", + " [ 0., 0., 1., 0., 0.],\n", + " [ 0., 0., 0., 1., 0.],\n", + " [ 0., 0., 0., 0., 1.]])\n", + "```" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "warmUpExercise()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 1.1 Submitting solutions\n", + "\n", + "After completing a part of the exercise, you can submit your solutions for grading by first adding the function you modified to the grader object, and then sending your function to Coursera for grading. \n", + "\n", + "The grader will prompt you for your login e-mail and submission token. You can obtain a submission token from the web page for the assignment. You are allowed to submit your solutions multiple times, and we will take only the highest score into consideration.\n", + "\n", + "Execute the next cell to grade your solution to the first part of this exercise.\n", + "\n", + "*You should now submit you solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# appends the implemented function in part 1 to the grader object\n", + "grader[1] = warmUpExercise\n", + "\n", + "# send the added functions to coursera grader for getting a grade on this part\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 2 Linear regression with one variable\n", + "\n", + "Now you will implement linear regression with one variable to predict profits for a food truck. Suppose you are the CEO of a restaurant franchise and are considering different cities for opening a new outlet. The chain already has trucks in various cities and you have data for profits and populations from the cities. You would like to use this data to help you select which city to expand to next. \n", + "\n", + "The file `Data/ex1data1.txt` contains the dataset for our linear regression problem. The first column is the population of a city and the second column is the profit of a food truck in that city. A negative value for profit indicates a loss. \n", + "\n", + "We provide you with the code needed to load this data. The dataset is loaded from the data file into the variables `x` and `y`:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Read comma separated data\n", + "data = np.loadtxt(os.path.join('Data', 'ex1data1.txt'), delimiter=',')\n", + "X, y = data[:, 0], data[:, 1]\n", + "\n", + "m = y.size # number of training examples" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2.1 Plotting the Data\n", + "\n", + "Before starting on any task, it is often useful to understand the data by visualizing it. For this dataset, you can use a scatter plot to visualize the data, since it has only two properties to plot (profit and population). Many other problems that you will encounter in real life are multi-dimensional and cannot be plotted on a 2-d plot. There are many plotting libraries in python (see this [blog post](https://blog.modeanalytics.com/python-data-visualization-libraries/) for a good summary of the most popular ones). \n", + "\n", + "In this course, we will be exclusively using `matplotlib` to do all our plotting. `matplotlib` is one of the most popular scientific plotting libraries in python and has extensive tools and functions to make beautiful plots. `pyplot` is a module within `matplotlib` which provides a simplified interface to `matplotlib`'s most common plotting tasks, mimicking MATLAB's plotting interface.\n", + "\n", + "
\n", + "You might have noticed that we have imported the `pyplot` module at the beginning of this exercise using the command `from matplotlib import pyplot`. This is rather uncommon, and if you look at python code elsewhere or in the `matplotlib` tutorials, you will see that the module is named `plt`. This is used by module renaming by using the import command `import matplotlib.pyplot as plt`. We will not using the short name of `pyplot` module in this class exercises, but you should be aware of this deviation from norm.\n", + "
\n", + "\n", + "\n", + "In the following part, your first job is to complete the `plotData` function below. Modify the function and fill in the following code:\n", + "\n", + "```python\n", + " pyplot.plot(x, y, 'ro', ms=10, mec='k')\n", + " pyplot.ylabel('Profit in $10,000')\n", + " pyplot.xlabel('Population of City in 10,000s')\n", + "```" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def plotData(x, y):\n", + " \"\"\"\n", + " Plots the data points x and y into a new figure. Plots the data \n", + " points and gives the figure axes labels of population and profit.\n", + " \n", + " Parameters\n", + " ----------\n", + " x : array_like\n", + " Data point values for x-axis.\n", + "\n", + " y : array_like\n", + " Data point values for y-axis. Note x and y should have the same size.\n", + " \n", + " Instructions\n", + " ------------\n", + " Plot the training data into a figure using the \"figure\" and \"plot\"\n", + " functions. Set the axes labels using the \"xlabel\" and \"ylabel\" functions.\n", + " Assume the population and revenue data have been passed in as the x\n", + " and y arguments of this function. \n", + " \n", + " Hint\n", + " ----\n", + " You can use the 'ro' option with plot to have the markers\n", + " appear as red circles. Furthermore, you can make the markers larger by\n", + " using plot(..., 'ro', ms=10), where `ms` refers to marker size. You \n", + " can also set the marker edge color using the `mec` property.\n", + " \"\"\"\n", + " fig = pyplot.figure() # open a new figure\n", + " \n", + " # ====================== YOUR CODE HERE ======================= \n", + " \n", + "\n", + " # =============================================================\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Now run the defined function with the loaded data to visualize the data. The end result should look like the following figure:\n", + "\n", + "![](Figures/dataset1.png)\n", + "\n", + "Execute the next cell to visualize the data." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "plotData(X, y)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "To quickly learn more about the `matplotlib` plot function and what arguments you can provide to it, you can type `?pyplot.plot` in a cell within the jupyter notebook. This opens a separate page showing the documentation for the requested function. You can also search online for plotting documentation. \n", + "\n", + "To set the markers to red circles, we used the option `'or'` within the `plot` function." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "?pyplot.plot" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 2.2 Gradient Descent\n", + "\n", + "In this part, you will fit the linear regression parameters $\\theta$ to our dataset using gradient descent.\n", + "\n", + "#### 2.2.1 Update Equations\n", + "\n", + "The objective of linear regression is to minimize the cost function\n", + "\n", + "$$ J(\\theta) = \\frac{1}{2m} \\sum_{i=1}^m \\left( h_{\\theta}(x^{(i)}) - y^{(i)}\\right)^2$$\n", + "\n", + "where the hypothesis $h_\\theta(x)$ is given by the linear model\n", + "$$ h_\\theta(x) = \\theta^Tx = \\theta_0 + \\theta_1 x_1$$\n", + "\n", + "Recall that the parameters of your model are the $\\theta_j$ values. These are\n", + "the values you will adjust to minimize cost $J(\\theta)$. One way to do this is to\n", + "use the batch gradient descent algorithm. In batch gradient descent, each\n", + "iteration performs the update\n", + "\n", + "$$ \\theta_j = \\theta_j - \\alpha \\frac{1}{m} \\sum_{i=1}^m \\left( h_\\theta(x^{(i)}) - y^{(i)}\\right)x_j^{(i)} \\qquad \\text{simultaneously update } \\theta_j \\text{ for all } j$$\n", + "\n", + "With each step of gradient descent, your parameters $\\theta_j$ come closer to the optimal values that will achieve the lowest cost J($\\theta$).\n", + "\n", + "
\n", + "**Implementation Note:** We store each example as a row in the the $X$ matrix in Python `numpy`. To take into account the intercept term ($\\theta_0$), we add an additional first column to $X$ and set it to all ones. This allows us to treat $\\theta_0$ as simply another 'feature'.\n", + "
\n", + "\n", + "\n", + "#### 2.2.2 Implementation\n", + "\n", + "We have already set up the data for linear regression. In the following cell, we add another dimension to our data to accommodate the $\\theta_0$ intercept term. Do NOT execute this cell more than once." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Add a column of ones to X. The numpy function stack joins arrays along a given axis. \n", + "# The first axis (axis=0) refers to rows (training examples) \n", + "# and second axis (axis=1) refers to columns (features).\n", + "X = np.stack([np.ones(m), X], axis=1)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "#### 2.2.3 Computing the cost $J(\\theta)$\n", + "\n", + "As you perform gradient descent to learn minimize the cost function $J(\\theta)$, it is helpful to monitor the convergence by computing the cost. In this section, you will implement a function to calculate $J(\\theta)$ so you can check the convergence of your gradient descent implementation. \n", + "\n", + "Your next task is to complete the code for the function `computeCost` which computes $J(\\theta)$. As you are doing this, remember that the variables $X$ and $y$ are not scalar values. $X$ is a matrix whose rows represent the examples from the training set and $y$ is a vector whose each elemennt represent the value at a given row of $X$.\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def computeCost(X, y, theta):\n", + " \"\"\"\n", + " Compute cost for linear regression. Computes the cost of using theta as the\n", + " parameter for linear regression to fit the data points in X and y.\n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like\n", + " The input dataset of shape (m x n+1), where m is the number of examples,\n", + " and n is the number of features. We assume a vector of one's already \n", + " appended to the features so we have n+1 columns.\n", + " \n", + " y : array_like\n", + " The values of the function at each data point. This is a vector of\n", + " shape (m, ).\n", + " \n", + " theta : array_like\n", + " The parameters for the regression function. This is a vector of \n", + " shape (n+1, ).\n", + " \n", + " Returns\n", + " -------\n", + " J : float\n", + " The value of the regression cost function.\n", + " \n", + " Instructions\n", + " ------------\n", + " Compute the cost of a particular choice of theta. \n", + " You should set J to the cost.\n", + " \"\"\"\n", + " \n", + " # initialize some useful values\n", + " m = y.size # number of training examples\n", + " \n", + " # You need to return the following variables correctly\n", + " J = 0\n", + " \n", + " # ====================== YOUR CODE HERE =====================\n", + "\n", + " \n", + " # ===========================================================\n", + " return J" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Once you have completed the function, the next step will run `computeCost` two times using two different initializations of $\\theta$. You will see the cost printed to the screen." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "J = computeCost(X, y, theta=np.array([0.0, 0.0]))\n", + "print('With theta = [0, 0] \\nCost computed = %.2f' % J)\n", + "print('Expected cost value (approximately) 32.07\\n')\n", + "\n", + "# further testing of the cost function\n", + "J = computeCost(X, y, theta=np.array([-1, 2]))\n", + "print('With theta = [-1, 2]\\nCost computed = %.2f' % J)\n", + "print('Expected cost value (approximately) 54.24')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions by executing the following cell.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[2] = computeCost\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "#### 2.2.4 Gradient descent\n", + "\n", + "Next, you will complete a function which implements gradient descent.\n", + "The loop structure has been written for you, and you only need to supply the updates to $\\theta$ within each iteration. \n", + "\n", + "As you program, make sure you understand what you are trying to optimize and what is being updated. Keep in mind that the cost $J(\\theta)$ is parameterized by the vector $\\theta$, not $X$ and $y$. That is, we minimize the value of $J(\\theta)$ by changing the values of the vector $\\theta$, not by changing $X$ or $y$. [Refer to the equations in this notebook](#section2) and to the video lectures if you are uncertain. A good way to verify that gradient descent is working correctly is to look at the value of $J(\\theta)$ and check that it is decreasing with each step. \n", + "\n", + "The starter code for the function `gradientDescent` calls `computeCost` on every iteration and saves the cost to a `python` list. Assuming you have implemented gradient descent and `computeCost` correctly, your value of $J(\\theta)$ should never increase, and should converge to a steady value by the end of the algorithm.\n", + "\n", + "
\n", + "**Vectors and matrices in `numpy`** - Important implementation notes\n", + "\n", + "A vector in `numpy` is a one dimensional array, for example `np.array([1, 2, 3])` is a vector. A matrix in `numpy` is a two dimensional array, for example `np.array([[1, 2, 3], [4, 5, 6]])`. However, the following is still considered a matrix `np.array([[1, 2, 3]])` since it has two dimensions, even if it has a shape of 1x3 (which looks like a vector).\n", + "\n", + "Given the above, the function `np.dot` which we will use for all matrix/vector multiplication has the following properties:\n", + "- It always performs inner products on vectors. If `x=np.array([1, 2, 3])`, then `np.dot(x, x)` is a scalar.\n", + "- For matrix-vector multiplication, so if $X$ is a $m\\times n$ matrix and $y$ is a vector of length $m$, then the operation `np.dot(y, X)` considers $y$ as a $1 \\times m$ vector. On the other hand, if $y$ is a vector of length $n$, then the operation `np.dot(X, y)` considers $y$ as a $n \\times 1$ vector.\n", + "- A vector can be promoted to a matrix using `y[None]` or `[y[np.newaxis]`. That is, if `y = np.array([1, 2, 3])` is a vector of size 3, then `y[None, :]` is a matrix of shape $1 \\times 3$. We can use `y[:, None]` to obtain a shape of $3 \\times 1$.\n", + "
\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def gradientDescent(X, y, theta, alpha, num_iters):\n", + " \"\"\"\n", + " Performs gradient descent to learn `theta`. Updates theta by taking `num_iters`\n", + " gradient steps with learning rate `alpha`.\n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like\n", + " The input dataset of shape (m x n+1).\n", + " \n", + " y : arra_like\n", + " Value at given features. A vector of shape (m, ).\n", + " \n", + " theta : array_like\n", + " Initial values for the linear regression parameters. \n", + " A vector of shape (n+1, ).\n", + " \n", + " alpha : float\n", + " The learning rate.\n", + " \n", + " num_iters : int\n", + " The number of iterations for gradient descent. \n", + " \n", + " Returns\n", + " -------\n", + " theta : array_like\n", + " The learned linear regression parameters. A vector of shape (n+1, ).\n", + " \n", + " J_history : list\n", + " A python list for the values of the cost function after each iteration.\n", + " \n", + " Instructions\n", + " ------------\n", + " Peform a single gradient step on the parameter vector theta.\n", + "\n", + " While debugging, it can be useful to print out the values of \n", + " the cost function (computeCost) and gradient here.\n", + " \"\"\"\n", + " # Initialize some useful values\n", + " m = y.shape[0] # number of training examples\n", + " \n", + " # make a copy of theta, to avoid changing the original array, since numpy arrays\n", + " # are passed by reference to functions\n", + " theta = theta.copy()\n", + " \n", + " J_history = [] # Use a python list to save cost in every iteration\n", + " \n", + " for i in range(num_iters):\n", + " # ==================== YOUR CODE HERE =================================\n", + " \n", + "\n", + " # =====================================================================\n", + " \n", + " # save the cost J in every iteration\n", + " J_history.append(computeCost(X, y, theta))\n", + " \n", + " return theta, J_history" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "After you are finished call the implemented `gradientDescent` function and print the computed $\\theta$. We initialize the $\\theta$ parameters to 0 and the learning rate $\\alpha$ to 0.01. Execute the following cell to check your code." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# initialize fitting parameters\n", + "theta = np.zeros(2)\n", + "\n", + "# some gradient descent settings\n", + "iterations = 1500\n", + "alpha = 0.01\n", + "\n", + "theta, J_history = gradientDescent(X ,y, theta, alpha, iterations)\n", + "print('Theta found by gradient descent: {:.4f}, {:.4f}'.format(*theta))\n", + "print('Expected theta values (approximately): [-3.6303, 1.1664]')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We will use your final parameters to plot the linear fit. The results should look like the following figure.\n", + "\n", + "![](Figures/regression_result.png)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# plot the linear fit\n", + "plotData(X[:, 1], y)\n", + "pyplot.plot(X[:, 1], np.dot(X, theta), '-')\n", + "pyplot.legend(['Training data', 'Linear regression']);" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Your final values for $\\theta$ will also be used to make predictions on profits in areas of 35,000 and 70,000 people.\n", + "\n", + "
\n", + "Note the way that the following lines use matrix multiplication, rather than explicit summation or looping, to calculate the predictions. This is an example of code vectorization in `numpy`.\n", + "
\n", + "\n", + "
\n", + "Note that the first argument to the `numpy` function `dot` is a python list. `numpy` can internally converts **valid** python lists to numpy arrays when explicitly provided as arguments to `numpy` functions.\n", + "
\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Predict values for population sizes of 35,000 and 70,000\n", + "predict1 = np.dot([1, 3.5], theta)\n", + "print('For population = 35,000, we predict a profit of {:.2f}\\n'.format(predict1*10000))\n", + "\n", + "predict2 = np.dot([1, 7], theta)\n", + "print('For population = 70,000, we predict a profit of {:.2f}\\n'.format(predict2*10000))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions by executing the next cell.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[3] = gradientDescent\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2.4 Visualizing $J(\\theta)$\n", + "\n", + "To understand the cost function $J(\\theta)$ better, you will now plot the cost over a 2-dimensional grid of $\\theta_0$ and $\\theta_1$ values. You will not need to code anything new for this part, but you should understand how the code you have written already is creating these images.\n", + "\n", + "In the next cell, the code is set up to calculate $J(\\theta)$ over a grid of values using the `computeCost` function that you wrote. After executing the following cell, you will have a 2-D array of $J(\\theta)$ values. Then, those values are used to produce surface and contour plots of $J(\\theta)$ using the matplotlib `plot_surface` and `contourf` functions. The plots should look something like the following:\n", + "\n", + "![](Figures/cost_function.png)\n", + "\n", + "The purpose of these graphs is to show you how $J(\\theta)$ varies with changes in $\\theta_0$ and $\\theta_1$. The cost function $J(\\theta)$ is bowl-shaped and has a global minimum. (This is easier to see in the contour plot than in the 3D surface plot). This minimum is the optimal point for $\\theta_0$ and $\\theta_1$, and each step of gradient descent moves closer to this point." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# grid over which we will calculate J\n", + "theta0_vals = np.linspace(-10, 10, 100)\n", + "theta1_vals = np.linspace(-1, 4, 100)\n", + "\n", + "# initialize J_vals to a matrix of 0's\n", + "J_vals = np.zeros((theta0_vals.shape[0], theta1_vals.shape[0]))\n", + "\n", + "# Fill out J_vals\n", + "for i, theta0 in enumerate(theta0_vals):\n", + " for j, theta1 in enumerate(theta1_vals):\n", + " J_vals[i, j] = computeCost(X, y, [theta0, theta1])\n", + " \n", + "# Because of the way meshgrids work in the surf command, we need to\n", + "# transpose J_vals before calling surf, or else the axes will be flipped\n", + "J_vals = J_vals.T\n", + "\n", + "# surface plot\n", + "fig = pyplot.figure(figsize=(12, 5))\n", + "ax = fig.add_subplot(121, projection='3d')\n", + "ax.plot_surface(theta0_vals, theta1_vals, J_vals, cmap='viridis')\n", + "pyplot.xlabel('theta0')\n", + "pyplot.ylabel('theta1')\n", + "pyplot.title('Surface')\n", + "\n", + "# contour plot\n", + "# Plot J_vals as 15 contours spaced logarithmically between 0.01 and 100\n", + "ax = pyplot.subplot(122)\n", + "pyplot.contour(theta0_vals, theta1_vals, J_vals, linewidths=2, cmap='viridis', levels=np.logspace(-2, 3, 20))\n", + "pyplot.xlabel('theta0')\n", + "pyplot.ylabel('theta1')\n", + "pyplot.plot(theta[0], theta[1], 'ro', ms=10, lw=2)\n", + "pyplot.title('Contour, showing minimum')\n", + "pass" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Optional Exercises\n", + "\n", + "If you have successfully completed the material above, congratulations! You now understand linear regression and should able to start using it on your own datasets.\n", + "\n", + "For the rest of this programming exercise, we have included the following optional exercises. These exercises will help you gain a deeper understanding of the material, and if you are able to do so, we encourage you to complete them as well. You can still submit your solutions to these exercises to check if your answers are correct.\n", + "\n", + "## 3 Linear regression with multiple variables\n", + "\n", + "In this part, you will implement linear regression with multiple variables to predict the prices of houses. Suppose you are selling your house and you want to know what a good market price would be. One way to do this is to first collect information on recent houses sold and make a model of housing prices.\n", + "\n", + "The file `Data/ex1data2.txt` contains a training set of housing prices in Portland, Oregon. The first column is the size of the house (in square feet), the second column is the number of bedrooms, and the third column is the price\n", + "of the house. \n", + "\n", + "\n", + "### 3.1 Feature Normalization\n", + "\n", + "We start by loading and displaying some values from this dataset. By looking at the values, note that house sizes are about 1000 times the number of bedrooms. When features differ by orders of magnitude, first performing feature scaling can make gradient descent converge much more quickly." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Load data\n", + "data = np.loadtxt(os.path.join('Data', 'ex1data2.txt'), delimiter=',')\n", + "X = data[:, :2]\n", + "y = data[:, 2]\n", + "m = y.size\n", + "\n", + "# print out some data points\n", + "print('{:>8s}{:>8s}{:>10s}'.format('X[:,0]', 'X[:, 1]', 'y'))\n", + "print('-'*26)\n", + "for i in range(10):\n", + " print('{:8.0f}{:8.0f}{:10.0f}'.format(X[i, 0], X[i, 1], y[i]))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Your task here is to complete the code in `featureNormalize` function:\n", + "- Subtract the mean value of each feature from the dataset.\n", + "- After subtracting the mean, additionally scale (divide) the feature values by their respective “standard deviations.”\n", + "\n", + "The standard deviation is a way of measuring how much variation there is in the range of values of a particular feature (most data points will lie within ±2 standard deviations of the mean); this is an alternative to taking the range of values (max-min). In `numpy`, you can use the `std` function to compute the standard deviation. \n", + "\n", + "For example, the quantity `X[:, 0]` contains all the values of $x_1$ (house sizes) in the training set, so `np.std(X[:, 0])` computes the standard deviation of the house sizes.\n", + "At the time that the function `featureNormalize` is called, the extra column of 1’s corresponding to $x_0 = 1$ has not yet been added to $X$. \n", + "\n", + "You will do this for all the features and your code should work with datasets of all sizes (any number of features / examples). Note that each column of the matrix $X$ corresponds to one feature.\n", + "\n", + "
\n", + "**Implementation Note:** When normalizing the features, it is important\n", + "to store the values used for normalization - the mean value and the standard deviation used for the computations. After learning the parameters\n", + "from the model, we often want to predict the prices of houses we have not\n", + "seen before. Given a new x value (living room area and number of bedrooms), we must first normalize x using the mean and standard deviation that we had previously computed from the training set.\n", + "
\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def featureNormalize(X):\n", + " \"\"\"\n", + " Normalizes the features in X. returns a normalized version of X where\n", + " the mean value of each feature is 0 and the standard deviation\n", + " is 1. This is often a good preprocessing step to do when working with\n", + " learning algorithms.\n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like\n", + " The dataset of shape (m x n).\n", + " \n", + " Returns\n", + " -------\n", + " X_norm : array_like\n", + " The normalized dataset of shape (m x n).\n", + " \n", + " Instructions\n", + " ------------\n", + " First, for each feature dimension, compute the mean of the feature\n", + " and subtract it from the dataset, storing the mean value in mu. \n", + " Next, compute the standard deviation of each feature and divide\n", + " each feature by it's standard deviation, storing the standard deviation \n", + " in sigma. \n", + " \n", + " Note that X is a matrix where each column is a feature and each row is\n", + " an example. You needto perform the normalization separately for each feature. \n", + " \n", + " Hint\n", + " ----\n", + " You might find the 'np.mean' and 'np.std' functions useful.\n", + " \"\"\"\n", + " # You need to set these values correctly\n", + " X_norm = X.copy()\n", + " mu = np.zeros(X.shape[1])\n", + " sigma = np.zeros(X.shape[1])\n", + "\n", + " # =========================== YOUR CODE HERE =====================\n", + "\n", + " \n", + " # ================================================================\n", + " return X_norm, mu, sigma" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Execute the next cell to run the implemented `featureNormalize` function." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# call featureNormalize on the loaded data\n", + "X_norm, mu, sigma = featureNormalize(X)\n", + "\n", + "print('Computed mean:', mu)\n", + "print('Computed standard deviation:', sigma)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should not submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[4] = featureNormalize\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "After the `featureNormalize` function is tested, we now add the intercept term to `X_norm`:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Add intercept term to X\n", + "X = np.concatenate([np.ones((m, 1)), X_norm], axis=1)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 3.2 Gradient Descent\n", + "\n", + "Previously, you implemented gradient descent on a univariate regression problem. The only difference now is that there is one more feature in the matrix $X$. The hypothesis function and the batch gradient descent update\n", + "rule remain unchanged. \n", + "\n", + "You should complete the code for the functions `computeCostMulti` and `gradientDescentMulti` to implement the cost function and gradient descent for linear regression with multiple variables. If your code in the previous part (single variable) already supports multiple variables, you can use it here too.\n", + "Make sure your code supports any number of features and is well-vectorized.\n", + "You can use the `shape` property of `numpy` arrays to find out how many features are present in the dataset.\n", + "\n", + "
\n", + "**Implementation Note:** In the multivariate case, the cost function can\n", + "also be written in the following vectorized form:\n", + "\n", + "$$ J(\\theta) = \\frac{1}{2m}(X\\theta - \\vec{y})^T(X\\theta - \\vec{y}) $$\n", + "\n", + "where \n", + "\n", + "$$ X = \\begin{pmatrix}\n", + " - (x^{(1)})^T - \\\\\n", + " - (x^{(2)})^T - \\\\\n", + " \\vdots \\\\\n", + " - (x^{(m)})^T - \\\\ \\\\\n", + " \\end{pmatrix} \\qquad \\mathbf{y} = \\begin{bmatrix} y^{(1)} \\\\ y^{(2)} \\\\ \\vdots \\\\ y^{(m)} \\\\\\end{bmatrix}$$\n", + "\n", + "the vectorized version is efficient when you are working with numerical computing tools like `numpy`. If you are an expert with matrix operations, you can prove to yourself that the two forms are equivalent.\n", + "
\n", + "\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def computeCostMulti(X, y, theta):\n", + " \"\"\"\n", + " Compute cost for linear regression with multiple variables.\n", + " Computes the cost of using theta as the parameter for linear regression to fit the data points in X and y.\n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like\n", + " The dataset of shape (m x n+1).\n", + " \n", + " y : array_like\n", + " A vector of shape (m, ) for the values at a given data point.\n", + " \n", + " theta : array_like\n", + " The linear regression parameters. A vector of shape (n+1, )\n", + " \n", + " Returns\n", + " -------\n", + " J : float\n", + " The value of the cost function. \n", + " \n", + " Instructions\n", + " ------------\n", + " Compute the cost of a particular choice of theta. You should set J to the cost.\n", + " \"\"\"\n", + " # Initialize some useful values\n", + " m = y.shape[0] # number of training examples\n", + " \n", + " # You need to return the following variable correctly\n", + " J = 0\n", + " \n", + " # ======================= YOUR CODE HERE ===========================\n", + "\n", + " \n", + " # ==================================================================\n", + " return J\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[5] = computeCostMulti\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def gradientDescentMulti(X, y, theta, alpha, num_iters):\n", + " \"\"\"\n", + " Performs gradient descent to learn theta.\n", + " Updates theta by taking num_iters gradient steps with learning rate alpha.\n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like\n", + " The dataset of shape (m x n+1).\n", + " \n", + " y : array_like\n", + " A vector of shape (m, ) for the values at a given data point.\n", + " \n", + " theta : array_like\n", + " The linear regression parameters. A vector of shape (n+1, )\n", + " \n", + " alpha : float\n", + " The learning rate for gradient descent. \n", + " \n", + " num_iters : int\n", + " The number of iterations to run gradient descent. \n", + " \n", + " Returns\n", + " -------\n", + " theta : array_like\n", + " The learned linear regression parameters. A vector of shape (n+1, ).\n", + " \n", + " J_history : list\n", + " A python list for the values of the cost function after each iteration.\n", + " \n", + " Instructions\n", + " ------------\n", + " Peform a single gradient step on the parameter vector theta.\n", + "\n", + " While debugging, it can be useful to print out the values of \n", + " the cost function (computeCost) and gradient here.\n", + " \"\"\"\n", + " # Initialize some useful values\n", + " m = y.shape[0] # number of training examples\n", + " \n", + " # make a copy of theta, which will be updated by gradient descent\n", + " theta = theta.copy()\n", + " \n", + " J_history = []\n", + " \n", + " for i in range(num_iters):\n", + " # ======================= YOUR CODE HERE ==========================\n", + "\n", + " \n", + " # =================================================================\n", + " \n", + " # save the cost J in every iteration\n", + " J_history.append(computeCostMulti(X, y, theta))\n", + " \n", + " return theta, J_history" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[6] = gradientDescentMulti\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "#### 3.2.1 Optional (ungraded) exercise: Selecting learning rates\n", + "\n", + "In this part of the exercise, you will get to try out different learning rates for the dataset and find a learning rate that converges quickly. You can change the learning rate by modifying the following code and changing the part of the code that sets the learning rate.\n", + "\n", + "Use your implementation of `gradientDescentMulti` function and run gradient descent for about 50 iterations at the chosen learning rate. The function should also return the history of $J(\\theta)$ values in a vector $J$.\n", + "\n", + "After the last iteration, plot the J values against the number of the iterations.\n", + "\n", + "If you picked a learning rate within a good range, your plot look similar as the following Figure. \n", + "\n", + "![](Figures/learning_rate.png)\n", + "\n", + "If your graph looks very different, especially if your value of $J(\\theta)$ increases or even blows up, adjust your learning rate and try again. We recommend trying values of the learning rate $\\alpha$ on a log-scale, at multiplicative steps of about 3 times the previous value (i.e., 0.3, 0.1, 0.03, 0.01 and so on). You may also want to adjust the number of iterations you are running if that will help you see the overall trend in the curve.\n", + "\n", + "
\n", + "**Implementation Note:** If your learning rate is too large, $J(\\theta)$ can diverge and ‘blow up’, resulting in values which are too large for computer calculations. In these situations, `numpy` will tend to return\n", + "NaNs. NaN stands for ‘not a number’ and is often caused by undefined operations that involve −∞ and +∞.\n", + "
\n", + "\n", + "
\n", + "**MATPLOTLIB tip:** To compare how different learning learning rates affect convergence, it is helpful to plot $J$ for several learning rates on the same figure. This can be done by making `alpha` a python list, and looping across the values within this list, and calling the plot function in every iteration of the loop. It is also useful to have a legend to distinguish the different lines within the plot. Search online for `pyplot.legend` for help on showing legends in `matplotlib`.\n", + "
\n", + "\n", + "Notice the changes in the convergence curves as the learning rate changes. With a small learning rate, you should find that gradient descent takes a very long time to converge to the optimal value. Conversely, with a large learning rate, gradient descent might not converge or might even diverge!\n", + "Using the best learning rate that you found, run the script\n", + "to run gradient descent until convergence to find the final values of $\\theta$. Next,\n", + "use this value of $\\theta$ to predict the price of a house with 1650 square feet and\n", + "3 bedrooms. You will use value later to check your implementation of the normal equations. Don’t forget to normalize your features when you make this prediction!" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "\"\"\"\n", + "Instructions\n", + "------------\n", + "We have provided you with the following starter code that runs\n", + "gradient descent with a particular learning rate (alpha). \n", + "\n", + "Your task is to first make sure that your functions - `computeCost`\n", + "and `gradientDescent` already work with this starter code and\n", + "support multiple variables.\n", + "\n", + "After that, try running gradient descent with different values of\n", + "alpha and see which one gives you the best result.\n", + "\n", + "Finally, you should complete the code at the end to predict the price\n", + "of a 1650 sq-ft, 3 br house.\n", + "\n", + "Hint\n", + "----\n", + "At prediction, make sure you do the same feature normalization.\n", + "\"\"\"\n", + "# Choose some alpha value - change this\n", + "alpha = 0.1\n", + "num_iters = 400\n", + "\n", + "# init theta and run gradient descent\n", + "theta = np.zeros(3)\n", + "theta, J_history = gradientDescentMulti(X, y, theta, alpha, num_iters)\n", + "\n", + "# Plot the convergence graph\n", + "pyplot.plot(np.arange(len(J_history)), J_history, lw=2)\n", + "pyplot.xlabel('Number of iterations')\n", + "pyplot.ylabel('Cost J')\n", + "\n", + "# Display the gradient descent's result\n", + "print('theta computed from gradient descent: {:s}'.format(str(theta)))\n", + "\n", + "# Estimate the price of a 1650 sq-ft, 3 br house\n", + "# ======================= YOUR CODE HERE ===========================\n", + "# Recall that the first column of X is all-ones. \n", + "# Thus, it does not need to be normalized.\n", + "\n", + "price = 0 # You should change this\n", + "\n", + "# ===================================================================\n", + "\n", + "print('Predicted price of a 1650 sq-ft, 3 br house (using gradient descent): ${:.0f}'.format(price))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You do not need to submit any solutions for this optional (ungraded) part.*" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 3.3 Normal Equations\n", + "\n", + "In the lecture videos, you learned that the closed-form solution to linear regression is\n", + "\n", + "$$ \\theta = \\left( X^T X\\right)^{-1} X^T\\vec{y}$$\n", + "\n", + "Using this formula does not require any feature scaling, and you will get an exact solution in one calculation: there is no “loop until convergence” like in gradient descent. \n", + "\n", + "First, we will reload the data to ensure that the variables have not been modified. Remember that while you do not need to scale your features, we still need to add a column of 1’s to the $X$ matrix to have an intercept term ($\\theta_0$). The code in the next cell will add the column of 1’s to X for you." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Load data\n", + "data = np.loadtxt(os.path.join('Data', 'ex1data2.txt'), delimiter=',')\n", + "X = data[:, :2]\n", + "y = data[:, 2]\n", + "m = y.size\n", + "X = np.concatenate([np.ones((m, 1)), X], axis=1)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Complete the code for the function `normalEqn` below to use the formula above to calculate $\\theta$. \n", + "\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def normalEqn(X, y):\n", + " \"\"\"\n", + " Computes the closed-form solution to linear regression using the normal equations.\n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like\n", + " The dataset of shape (m x n+1).\n", + " \n", + " y : array_like\n", + " The value at each data point. A vector of shape (m, ).\n", + " \n", + " Returns\n", + " -------\n", + " theta : array_like\n", + " Estimated linear regression parameters. A vector of shape (n+1, ).\n", + " \n", + " Instructions\n", + " ------------\n", + " Complete the code to compute the closed form solution to linear\n", + " regression and put the result in theta.\n", + " \n", + " Hint\n", + " ----\n", + " Look up the function `np.linalg.pinv` for computing matrix inverse.\n", + " \"\"\"\n", + " theta = np.zeros(X.shape[1])\n", + " \n", + " # ===================== YOUR CODE HERE ============================\n", + "\n", + " \n", + " # =================================================================\n", + " return theta" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[7] = normalEqn\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Optional (ungraded) exercise: Now, once you have found $\\theta$ using this\n", + "method, use it to make a price prediction for a 1650-square-foot house with\n", + "3 bedrooms. You should find that gives the same predicted price as the value\n", + "you obtained using the model fit with gradient descent (in Section 3.2.1)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Calculate the parameters from the normal equation\n", + "theta = normalEqn(X, y);\n", + "\n", + "# Display normal equation's result\n", + "print('Theta computed from the normal equations: {:s}'.format(str(theta)));\n", + "\n", + "# Estimate the price of a 1650 sq-ft, 3 br house\n", + "# ====================== YOUR CODE HERE ======================\n", + "\n", + "price = 0 # You should change this\n", + "\n", + "# ============================================================\n", + "\n", + "print('Predicted price of a 1650 sq-ft, 3 br house (using normal equations): ${:.0f}'.format(price))" + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.6.4" + } + }, + "nbformat": 4, + "nbformat_minor": 2 +} diff --git a/Exercise1/utils.py b/Exercise1/utils.py new file mode 100755 index 0000000..d0c909d --- /dev/null +++ b/Exercise1/utils.py @@ -0,0 +1,48 @@ +import numpy as np +import sys +sys.path.append('..') + +from submission import SubmissionBase + + +class Grader(SubmissionBase): + X1 = np.column_stack((np.ones(20), np.exp(1) + np.exp(2) * np.linspace(0.1, 2, 20))) + Y1 = X1[:, 1] + np.sin(X1[:, 0]) + np.cos(X1[:, 1]) + X2 = np.column_stack((X1, X1[:, 1]**0.5, X1[:, 1]**0.25)) + Y2 = np.power(Y1, 0.5) + Y1 + + def __init__(self): + part_names = ['Warm up exercise', + 'Computing Cost (for one variable)', + 'Gradient Descent (for one variable)', + 'Feature Normalization', + 'Computing Cost (for multiple variables)', + 'Gradient Descent (for multiple variables)', + 'Normal Equations'] + super().__init__('linear-regression', part_names) + + def __iter__(self): + for part_id in range(1, 8): + try: + func = self.functions[part_id] + + # Each part has different expected arguments/different function + if part_id == 1: + res = func() + elif part_id == 2: + res = func(self.X1, self.Y1, np.array([0.5, -0.5])) + elif part_id == 3: + res = func(self.X1, self.Y1, np.array([0.5, -0.5]), 0.01, 10) + elif part_id == 4: + res = func(self.X2[:, 1:4]) + elif part_id == 5: + res = func(self.X2, self.Y2, np.array([0.1, 0.2, 0.3, 0.4])) + elif part_id == 6: + res = func(self.X2, self.Y2, np.array([-0.1, -0.2, -0.3, -0.4]), 0.01, 10) + elif part_id == 7: + res = func(self.X2, self.Y2) + else: + raise KeyError + yield part_id, res + except KeyError: + yield part_id, 0 diff --git a/Exercise2/Data/ex2data1.txt b/Exercise2/Data/ex2data1.txt new file mode 100755 index 0000000..3a5f952 --- /dev/null +++ b/Exercise2/Data/ex2data1.txt @@ -0,0 +1,100 @@ +34.62365962451697,78.0246928153624,0 +30.28671076822607,43.89499752400101,0 +35.84740876993872,72.90219802708364,0 +60.18259938620976,86.30855209546826,1 +79.0327360507101,75.3443764369103,1 +45.08327747668339,56.3163717815305,0 +61.10666453684766,96.51142588489624,1 +75.02474556738889,46.55401354116538,1 +76.09878670226257,87.42056971926803,1 +84.43281996120035,43.53339331072109,1 +95.86155507093572,38.22527805795094,0 +75.01365838958247,30.60326323428011,0 +82.30705337399482,76.48196330235604,1 +69.36458875970939,97.71869196188608,1 +39.53833914367223,76.03681085115882,0 +53.9710521485623,89.20735013750205,1 +69.07014406283025,52.74046973016765,1 +67.94685547711617,46.67857410673128,0 +70.66150955499435,92.92713789364831,1 +76.97878372747498,47.57596364975532,1 +67.37202754570876,42.83843832029179,0 +89.67677575072079,65.79936592745237,1 +50.534788289883,48.85581152764205,0 +34.21206097786789,44.20952859866288,0 +77.9240914545704,68.9723599933059,1 +62.27101367004632,69.95445795447587,1 +80.1901807509566,44.82162893218353,1 +93.114388797442,38.80067033713209,0 +61.83020602312595,50.25610789244621,0 +38.78580379679423,64.99568095539578,0 +61.379289447425,72.80788731317097,1 +85.40451939411645,57.05198397627122,1 +52.10797973193984,63.12762376881715,0 +52.04540476831827,69.43286012045222,1 +40.23689373545111,71.16774802184875,0 +54.63510555424817,52.21388588061123,0 +33.91550010906887,98.86943574220611,0 +64.17698887494485,80.90806058670817,1 +74.78925295941542,41.57341522824434,0 +34.1836400264419,75.2377203360134,0 +83.90239366249155,56.30804621605327,1 +51.54772026906181,46.85629026349976,0 +94.44336776917852,65.56892160559052,1 +82.36875375713919,40.61825515970618,0 +51.04775177128865,45.82270145776001,0 +62.22267576120188,52.06099194836679,0 +77.19303492601364,70.45820000180959,1 +97.77159928000232,86.7278223300282,1 +62.07306379667647,96.76882412413983,1 +91.56497449807442,88.69629254546599,1 +79.94481794066932,74.16311935043758,1 +99.2725269292572,60.99903099844988,1 +90.54671411399852,43.39060180650027,1 +34.52451385320009,60.39634245837173,0 +50.2864961189907,49.80453881323059,0 +49.58667721632031,59.80895099453265,0 +97.64563396007767,68.86157272420604,1 +32.57720016809309,95.59854761387875,0 +74.24869136721598,69.82457122657193,1 +71.79646205863379,78.45356224515052,1 +75.3956114656803,85.75993667331619,1 +35.28611281526193,47.02051394723416,0 +56.25381749711624,39.26147251058019,0 +30.05882244669796,49.59297386723685,0 +44.66826172480893,66.45008614558913,0 +66.56089447242954,41.09209807936973,0 +40.45755098375164,97.53518548909936,1 +49.07256321908844,51.88321182073966,0 +80.27957401466998,92.11606081344084,1 +66.74671856944039,60.99139402740988,1 +32.72283304060323,43.30717306430063,0 +64.0393204150601,78.03168802018232,1 +72.34649422579923,96.22759296761404,1 +60.45788573918959,73.09499809758037,1 +58.84095621726802,75.85844831279042,1 +99.82785779692128,72.36925193383885,1 +47.26426910848174,88.47586499559782,1 +50.45815980285988,75.80985952982456,1 +60.45555629271532,42.50840943572217,0 +82.22666157785568,42.71987853716458,0 +88.9138964166533,69.80378889835472,1 +94.83450672430196,45.69430680250754,1 +67.31925746917527,66.58935317747915,1 +57.23870631569862,59.51428198012956,1 +80.36675600171273,90.96014789746954,1 +68.46852178591112,85.59430710452014,1 +42.0754545384731,78.84478600148043,0 +75.47770200533905,90.42453899753964,1 +78.63542434898018,96.64742716885644,1 +52.34800398794107,60.76950525602592,0 +94.09433112516793,77.15910509073893,1 +90.44855097096364,87.50879176484702,1 +55.48216114069585,35.57070347228866,0 +74.49269241843041,84.84513684930135,1 +89.84580670720979,45.35828361091658,1 +83.48916274498238,48.38028579728175,1 +42.2617008099817,87.10385094025457,1 +99.31500880510394,68.77540947206617,1 +55.34001756003703,64.9319380069486,1 +74.77589300092767,89.52981289513276,1 diff --git a/Exercise2/Data/ex2data2.txt b/Exercise2/Data/ex2data2.txt new file mode 100755 index 0000000..a888992 --- /dev/null +++ b/Exercise2/Data/ex2data2.txt @@ -0,0 +1,118 @@ +0.051267,0.69956,1 +-0.092742,0.68494,1 +-0.21371,0.69225,1 +-0.375,0.50219,1 +-0.51325,0.46564,1 +-0.52477,0.2098,1 +-0.39804,0.034357,1 +-0.30588,-0.19225,1 +0.016705,-0.40424,1 +0.13191,-0.51389,1 +0.38537,-0.56506,1 +0.52938,-0.5212,1 +0.63882,-0.24342,1 +0.73675,-0.18494,1 +0.54666,0.48757,1 +0.322,0.5826,1 +0.16647,0.53874,1 +-0.046659,0.81652,1 +-0.17339,0.69956,1 +-0.47869,0.63377,1 +-0.60541,0.59722,1 +-0.62846,0.33406,1 +-0.59389,0.005117,1 +-0.42108,-0.27266,1 +-0.11578,-0.39693,1 +0.20104,-0.60161,1 +0.46601,-0.53582,1 +0.67339,-0.53582,1 +-0.13882,0.54605,1 +-0.29435,0.77997,1 +-0.26555,0.96272,1 +-0.16187,0.8019,1 +-0.17339,0.64839,1 +-0.28283,0.47295,1 +-0.36348,0.31213,1 +-0.30012,0.027047,1 +-0.23675,-0.21418,1 +-0.06394,-0.18494,1 +0.062788,-0.16301,1 +0.22984,-0.41155,1 +0.2932,-0.2288,1 +0.48329,-0.18494,1 +0.64459,-0.14108,1 +0.46025,0.012427,1 +0.6273,0.15863,1 +0.57546,0.26827,1 +0.72523,0.44371,1 +0.22408,0.52412,1 +0.44297,0.67032,1 +0.322,0.69225,1 +0.13767,0.57529,1 +-0.0063364,0.39985,1 +-0.092742,0.55336,1 +-0.20795,0.35599,1 +-0.20795,0.17325,1 +-0.43836,0.21711,1 +-0.21947,-0.016813,1 +-0.13882,-0.27266,1 +0.18376,0.93348,0 +0.22408,0.77997,0 +0.29896,0.61915,0 +0.50634,0.75804,0 +0.61578,0.7288,0 +0.60426,0.59722,0 +0.76555,0.50219,0 +0.92684,0.3633,0 +0.82316,0.27558,0 +0.96141,0.085526,0 +0.93836,0.012427,0 +0.86348,-0.082602,0 +0.89804,-0.20687,0 +0.85196,-0.36769,0 +0.82892,-0.5212,0 +0.79435,-0.55775,0 +0.59274,-0.7405,0 +0.51786,-0.5943,0 +0.46601,-0.41886,0 +0.35081,-0.57968,0 +0.28744,-0.76974,0 +0.085829,-0.75512,0 +0.14919,-0.57968,0 +-0.13306,-0.4481,0 +-0.40956,-0.41155,0 +-0.39228,-0.25804,0 +-0.74366,-0.25804,0 +-0.69758,0.041667,0 +-0.75518,0.2902,0 +-0.69758,0.68494,0 +-0.4038,0.70687,0 +-0.38076,0.91886,0 +-0.50749,0.90424,0 +-0.54781,0.70687,0 +0.10311,0.77997,0 +0.057028,0.91886,0 +-0.10426,0.99196,0 +-0.081221,1.1089,0 +0.28744,1.087,0 +0.39689,0.82383,0 +0.63882,0.88962,0 +0.82316,0.66301,0 +0.67339,0.64108,0 +1.0709,0.10015,0 +-0.046659,-0.57968,0 +-0.23675,-0.63816,0 +-0.15035,-0.36769,0 +-0.49021,-0.3019,0 +-0.46717,-0.13377,0 +-0.28859,-0.060673,0 +-0.61118,-0.067982,0 +-0.66302,-0.21418,0 +-0.59965,-0.41886,0 +-0.72638,-0.082602,0 +-0.83007,0.31213,0 +-0.72062,0.53874,0 +-0.59389,0.49488,0 +-0.48445,0.99927,0 +-0.0063364,0.99927,0 +0.63265,-0.030612,0 diff --git a/Exercise2/Figures/decision_boundary1.png b/Exercise2/Figures/decision_boundary1.png new file mode 100755 index 0000000..f1399c5 Binary files /dev/null and b/Exercise2/Figures/decision_boundary1.png differ diff --git a/Exercise2/Figures/decision_boundary2.png b/Exercise2/Figures/decision_boundary2.png new file mode 100755 index 0000000..52c9f75 Binary files /dev/null and b/Exercise2/Figures/decision_boundary2.png differ diff --git a/Exercise2/Figures/decision_boundary3.png b/Exercise2/Figures/decision_boundary3.png new file mode 100755 index 0000000..82b6ea1 Binary files /dev/null and b/Exercise2/Figures/decision_boundary3.png differ diff --git a/Exercise2/Figures/decision_boundary4.png b/Exercise2/Figures/decision_boundary4.png new file mode 100755 index 0000000..b6c1370 Binary files /dev/null and b/Exercise2/Figures/decision_boundary4.png differ diff --git a/Exercise2/exercise2.ipynb b/Exercise2/exercise2.ipynb new file mode 100755 index 0000000..39983d9 --- /dev/null +++ b/Exercise2/exercise2.ipynb @@ -0,0 +1,965 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Programming Exercise 2: Logistic Regression\n", + "\n", + "## Introduction\n", + "\n", + "In this exercise, you will implement logistic regression and apply it to two different datasets. Before starting on the programming exercise, we strongly recommend watching the video lectures and completing the review questions for the associated topics.\n", + "\n", + "All the information you need for solving this assignment is in this notebook, and all the code you will be implementing will take place within this notebook. The assignment can be promptly submitted to the coursera grader directly from this notebook (code and instructions are included below).\n", + "\n", + "Before we begin with the exercises, we need to import all libraries required for this programming exercise. Throughout the course, we will be using [`numpy`](http://www.numpy.org/) for all arrays and matrix operations, and [`matplotlib`](https://matplotlib.org/) for plotting. In this assignment, we will also use [`scipy`](https://docs.scipy.org/doc/scipy/reference/), which contains scientific and numerical computation functions and tools. \n", + "\n", + "You can find instructions on how to install required libraries in the README file in the [github repository](https://github.com/dibgerge/ml-coursera-python-assignments)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# used for manipulating directory paths\n", + "import os\n", + "\n", + "# Scientific and vector computation for python\n", + "import numpy as np\n", + "\n", + "# Plotting library\n", + "from matplotlib import pyplot\n", + "\n", + "# Optimization module in scipy\n", + "from scipy import optimize\n", + "\n", + "# library written for this exercise providing additional functions for assignment submission, and others\n", + "import utils\n", + "\n", + "# define the submission/grader object for this exercise\n", + "grader = utils.Grader()\n", + "\n", + "# tells matplotlib to embed plots within the notebook\n", + "%matplotlib inline" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Submission and Grading\n", + "\n", + "\n", + "After completing each part of the assignment, be sure to submit your solutions to the grader. The following is a breakdown of how each part of this exercise is scored.\n", + "\n", + "\n", + "| Section | Part | Submission function | Points \n", + "| :- |:- | :- | :-:\n", + "| 1 | [Sigmoid Function](#section1) | [`sigmoid`](#sigmoid) | 5 \n", + "| 2 | [Compute cost for logistic regression](#section2) | [`costFunction`](#costFunction) | 30 \n", + "| 3 | [Gradient for logistic regression](#section2) | [`costFunction`](#costFunction) | 30 \n", + "| 4 | [Predict Function](#section4) | [`predict`](#predict) | 5 \n", + "| 5 | [Compute cost for regularized LR](#section5) | [`costFunctionReg`](#costFunctionReg) | 15 \n", + "| 6 | [Gradient for regularized LR](#section5) | [`costFunctionReg`](#costFunctionReg) | 15 \n", + "| | Total Points | | 100 \n", + "\n", + "\n", + "\n", + "You are allowed to submit your solutions multiple times, and we will take only the highest score into consideration.\n", + "\n", + "
\n", + "At the end of each section in this notebook, we have a cell which contains code for submitting the solutions thus far to the grader. Execute the cell to see your score up to the current section. For all your work to be submitted properly, you must execute those cells at least once. They must also be re-executed everytime the submitted function is updated.\n", + "
\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 1 Logistic Regression\n", + "\n", + "In this part of the exercise, you will build a logistic regression model to predict whether a student gets admitted into a university. Suppose that you are the administrator of a university department and\n", + "you want to determine each applicant’s chance of admission based on their results on two exams. You have historical data from previous applicants that you can use as a training set for logistic regression. For each training example, you have the applicant’s scores on two exams and the admissions\n", + "decision. Your task is to build a classification model that estimates an applicant’s probability of admission based the scores from those two exams. \n", + "\n", + "The following cell will load the data and corresponding labels:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Load data\n", + "# The first two columns contains the exam scores and the third column\n", + "# contains the label.\n", + "data = np.loadtxt(os.path.join('Data', 'ex2data1.txt'), delimiter=',')\n", + "X, y = data[:, 0:2], data[:, 2]" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 1.1 Visualizing the data\n", + "\n", + "Before starting to implement any learning algorithm, it is always good to visualize the data if possible. We display the data on a 2-dimensional plot by calling the function `plotData`. You will now complete the code in `plotData` so that it displays a figure where the axes are the two exam scores, and the positive and negative examples are shown with different markers.\n", + "\n", + "To help you get more familiar with plotting, we have left `plotData` empty so you can try to implement it yourself. However, this is an optional (ungraded) exercise. We also provide our implementation below so you can\n", + "copy it or refer to it. If you choose to copy our example, make sure you learn\n", + "what each of its commands is doing by consulting the `matplotlib` and `numpy` documentation.\n", + "\n", + "```python\n", + "# Find Indices of Positive and Negative Examples\n", + "pos = y == 1\n", + "neg = y == 0\n", + "\n", + "# Plot Examples\n", + "pyplot.plot(X[pos, 0], X[pos, 1], 'k*', lw=2, ms=10)\n", + "pyplot.plot(X[neg, 0], X[neg, 1], 'ko', mfc='y', ms=8, mec='k', mew=1)\n", + "```" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def plotData(X, y):\n", + " \"\"\"\n", + " Plots the data points X and y into a new figure. Plots the data \n", + " points with * for the positive examples and o for the negative examples.\n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like\n", + " An Mx2 matrix representing the dataset. \n", + " \n", + " y : array_like\n", + " Label values for the dataset. A vector of size (M, ).\n", + " \n", + " Instructions\n", + " ------------\n", + " Plot the positive and negative examples on a 2D plot, using the\n", + " option 'k*' for the positive examples and 'ko' for the negative examples. \n", + " \"\"\"\n", + " # Create New Figure\n", + " fig = pyplot.figure()\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + " \n", + " # ============================================================" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Now, we call the implemented function to display the loaded data:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "plotData(X, y)\n", + "# add axes labels\n", + "pyplot.xlabel('Exam 1 score')\n", + "pyplot.ylabel('Exam 2 score')\n", + "pyplot.legend(['Admitted', 'Not admitted'])\n", + "pass" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 1.2 Implementation\n", + "\n", + "#### 1.2.1 Warmup exercise: sigmoid function\n", + "\n", + "Before you start with the actual cost function, recall that the logistic regression hypothesis is defined as:\n", + "\n", + "$$ h_\\theta(x) = g(\\theta^T x)$$\n", + "\n", + "where function $g$ is the sigmoid function. The sigmoid function is defined as: \n", + "\n", + "$$g(z) = \\frac{1}{1+e^{-z}}$$.\n", + "\n", + "Your first step is to implement this function `sigmoid` so it can be\n", + "called by the rest of your program. When you are finished, try testing a few\n", + "values by calling `sigmoid(x)` in a new cell. For large positive values of `x`, the sigmoid should be close to 1, while for large negative values, the sigmoid should be close to 0. Evaluating `sigmoid(0)` should give you exactly 0.5. Your code should also work with vectors and matrices. **For a matrix, your function should perform the sigmoid function on every element.**\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def sigmoid(z):\n", + " \"\"\"\n", + " Compute sigmoid function given the input z.\n", + " \n", + " Parameters\n", + " ----------\n", + " z : array_like\n", + " The input to the sigmoid function. This can be a 1-D vector \n", + " or a 2-D matrix. \n", + " \n", + " Returns\n", + " -------\n", + " g : array_like\n", + " The computed sigmoid function. g has the same shape as z, since\n", + " the sigmoid is computed element-wise on z.\n", + " \n", + " Instructions\n", + " ------------\n", + " Compute the sigmoid of each value of z (z can be a matrix, vector or scalar).\n", + " \"\"\"\n", + " # convert input to a numpy array\n", + " z = np.array(z)\n", + " \n", + " # You need to return the following variables correctly \n", + " g = np.zeros(z.shape)\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + " \n", + "\n", + " # =============================================================\n", + " return g" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The following cell evaluates the sigmoid function at `z=0`. You should get a value of 0.5. You can also try different values for `z` to experiment with the sigmoid function." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Test the implementation of sigmoid function here\n", + "z = 0\n", + "g = sigmoid(z)\n", + "\n", + "print('g(', z, ') = ', g)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "After completing a part of the exercise, you can submit your solutions for grading by first adding the function you modified to the submission object, and then sending your function to Coursera for grading. \n", + "\n", + "The submission script will prompt you for your login e-mail and submission token. You can obtain a submission token from the web page for the assignment. You are allowed to submit your solutions multiple times, and we will take only the highest score into consideration.\n", + "\n", + "Execute the following cell to grade your solution to the first part of this exercise.\n", + "\n", + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# appends the implemented function in part 1 to the grader object\n", + "grader[1] = sigmoid\n", + "\n", + "# send the added functions to coursera grader for getting a grade on this part\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "#### 1.2.2 Cost function and gradient\n", + "\n", + "Now you will implement the cost function and gradient for logistic regression. Before proceeding we add the intercept term to X. " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Setup the data matrix appropriately, and add ones for the intercept term\n", + "m, n = X.shape\n", + "\n", + "# Add intercept term to X\n", + "X = np.concatenate([np.ones((m, 1)), X], axis=1)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Now, complete the code for the function `costFunction` to return the cost and gradient. Recall that the cost function in logistic regression is\n", + "\n", + "$$ J(\\theta) = \\frac{1}{m} \\sum_{i=1}^{m} \\left[ -y^{(i)} \\log\\left(h_\\theta\\left( x^{(i)} \\right) \\right) - \\left( 1 - y^{(i)}\\right) \\log \\left( 1 - h_\\theta\\left( x^{(i)} \\right) \\right) \\right]$$\n", + "\n", + "and the gradient of the cost is a vector of the same length as $\\theta$ where the $j^{th}$\n", + "element (for $j = 0, 1, \\cdots , n$) is defined as follows:\n", + "\n", + "$$ \\frac{\\partial J(\\theta)}{\\partial \\theta_j} = \\frac{1}{m} \\sum_{i=1}^m \\left( h_\\theta \\left( x^{(i)} \\right) - y^{(i)} \\right) x_j^{(i)} $$\n", + "\n", + "Note that while this gradient looks identical to the linear regression gradient, the formula is actually different because linear and logistic regression have different definitions of $h_\\theta(x)$.\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def costFunction(theta, X, y):\n", + " \"\"\"\n", + " Compute cost and gradient for logistic regression. \n", + " \n", + " Parameters\n", + " ----------\n", + " theta : array_like\n", + " The parameters for logistic regression. This a vector\n", + " of shape (n+1, ).\n", + " \n", + " X : array_like\n", + " The input dataset of shape (m x n+1) where m is the total number\n", + " of data points and n is the number of features. We assume the \n", + " intercept has already been added to the input.\n", + " \n", + " y : arra_like\n", + " Labels for the input. This is a vector of shape (m, ).\n", + " \n", + " Returns\n", + " -------\n", + " J : float\n", + " The computed value for the cost function. \n", + " \n", + " grad : array_like\n", + " A vector of shape (n+1, ) which is the gradient of the cost\n", + " function with respect to theta, at the current values of theta.\n", + " \n", + " Instructions\n", + " ------------\n", + " Compute the cost of a particular choice of theta. You should set J to \n", + " the cost. Compute the partial derivatives and set grad to the partial\n", + " derivatives of the cost w.r.t. each parameter in theta.\n", + " \"\"\"\n", + " # Initialize some useful values\n", + " m = y.size # number of training examples\n", + "\n", + " # You need to return the following variables correctly \n", + " J = 0\n", + " grad = np.zeros(theta.shape)\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + " \n", + " \n", + " # =============================================================\n", + " return J, grad" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Once you are done call your `costFunction` using two test cases for $\\theta$ by executing the next cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Initialize fitting parameters\n", + "initial_theta = np.zeros(n+1)\n", + "\n", + "cost, grad = costFunction(initial_theta, X, y)\n", + "\n", + "print('Cost at initial theta (zeros): {:.3f}'.format(cost))\n", + "print('Expected cost (approx): 0.693\\n')\n", + "\n", + "print('Gradient at initial theta (zeros):')\n", + "print('\\t[{:.4f}, {:.4f}, {:.4f}]'.format(*grad))\n", + "print('Expected gradients (approx):\\n\\t[-0.1000, -12.0092, -11.2628]\\n')\n", + "\n", + "# Compute and display cost and gradient with non-zero theta\n", + "test_theta = np.array([-24, 0.2, 0.2])\n", + "cost, grad = costFunction(test_theta, X, y)\n", + "\n", + "print('Cost at test theta: {:.3f}'.format(cost))\n", + "print('Expected cost (approx): 0.218\\n')\n", + "\n", + "print('Gradient at test theta:')\n", + "print('\\t[{:.3f}, {:.3f}, {:.3f}]'.format(*grad))\n", + "print('Expected gradients (approx):\\n\\t[0.043, 2.566, 2.647]')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[2] = costFunction\n", + "grader[3] = costFunction\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "#### 1.2.3 Learning parameters using `scipy.optimize`\n", + "\n", + "In the previous assignment, you found the optimal parameters of a linear regression model by implementing gradient descent. You wrote a cost function and calculated its gradient, then took a gradient descent step accordingly. This time, instead of taking gradient descent steps, you will use the [`scipy.optimize` module](https://docs.scipy.org/doc/scipy/reference/optimize.html). SciPy is a numerical computing library for `python`. It provides an optimization module for root finding and minimization. As of `scipy 1.0`, the function `scipy.optimize.minimize` is the method to use for optimization problems(both constrained and unconstrained).\n", + "\n", + "For logistic regression, you want to optimize the cost function $J(\\theta)$ with parameters $\\theta$.\n", + "Concretely, you are going to use `optimize.minimize` to find the best parameters $\\theta$ for the logistic regression cost function, given a fixed dataset (of X and y values). You will pass to `optimize.minimize` the following inputs:\n", + "- `costFunction`: A cost function that, when given the training set and a particular $\\theta$, computes the logistic regression cost and gradient with respect to $\\theta$ for the dataset (X, y). It is important to note that we only pass the name of the function without the parenthesis. This indicates that we are only providing a reference to this function, and not evaluating the result from this function.\n", + "- `initial_theta`: The initial values of the parameters we are trying to optimize.\n", + "- `(X, y)`: These are additional arguments to the cost function.\n", + "- `jac`: Indication if the cost function returns the Jacobian (gradient) along with cost value. (True)\n", + "- `method`: Optimization method/algorithm to use\n", + "- `options`: Additional options which might be specific to the specific optimization method. In the following, we only tell the algorithm the maximum number of iterations before it terminates.\n", + "\n", + "If you have completed the `costFunction` correctly, `optimize.minimize` will converge on the right optimization parameters and return the final values of the cost and $\\theta$ in a class object. Notice that by using `optimize.minimize`, you did not have to write any loops yourself, or set a learning rate like you did for gradient descent. This is all done by `optimize.minimize`: you only needed to provide a function calculating the cost and the gradient.\n", + "\n", + "In the following, we already have code written to call `optimize.minimize` with the correct arguments." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# set options for optimize.minimize\n", + "options= {'maxiter': 400}\n", + "\n", + "# see documention for scipy's optimize.minimize for description about\n", + "# the different parameters\n", + "# The function returns an object `OptimizeResult`\n", + "# We use truncated Newton algorithm for optimization which is \n", + "# equivalent to MATLAB's fminunc\n", + "# See https://stackoverflow.com/questions/18801002/fminunc-alternate-in-numpy\n", + "res = optimize.minimize(costFunction,\n", + " initial_theta,\n", + " (X, y),\n", + " jac=True,\n", + " method='TNC',\n", + " options=options)\n", + "\n", + "# the fun property of `OptimizeResult` object returns\n", + "# the value of costFunction at optimized theta\n", + "cost = res.fun\n", + "\n", + "# the optimized theta is in the x property\n", + "theta = res.x\n", + "\n", + "# Print theta to screen\n", + "print('Cost at theta found by optimize.minimize: {:.3f}'.format(cost))\n", + "print('Expected cost (approx): 0.203\\n');\n", + "\n", + "print('theta:')\n", + "print('\\t[{:.3f}, {:.3f}, {:.3f}]'.format(*theta))\n", + "print('Expected theta (approx):\\n\\t[-25.161, 0.206, 0.201]')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Once `optimize.minimize` completes, we want to use the final value for $\\theta$ to visualize the decision boundary on the training data as shown in the figure below. \n", + "\n", + "![](Figures/decision_boundary1.png)\n", + "\n", + "To do so, we have written a function `plotDecisionBoundary` for plotting the decision boundary on top of training data. You do not need to write any code for plotting the decision boundary, but we also encourage you to look at the code in `plotDecisionBoundary` to see how to plot such a boundary using the $\\theta$ values. You can find this function in the `utils.py` file which comes with this assignment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Plot Boundary\n", + "utils.plotDecisionBoundary(plotData, theta, X, y)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "#### 1.2.4 Evaluating logistic regression\n", + "\n", + "After learning the parameters, you can use the model to predict whether a particular student will be admitted. For a student with an Exam 1 score of 45 and an Exam 2 score of 85, you should expect to see an admission\n", + "probability of 0.776. Another way to evaluate the quality of the parameters we have found is to see how well the learned model predicts on our training set. In this part, your task is to complete the code in function `predict`. The predict function will produce “1” or “0” predictions given a dataset and a learned parameter vector $\\theta$. \n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def predict(theta, X):\n", + " \"\"\"\n", + " Predict whether the label is 0 or 1 using learned logistic regression.\n", + " Computes the predictions for X using a threshold at 0.5 \n", + " (i.e., if sigmoid(theta.T*x) >= 0.5, predict 1)\n", + " \n", + " Parameters\n", + " ----------\n", + " theta : array_like\n", + " Parameters for logistic regression. A vecotor of shape (n+1, ).\n", + " \n", + " X : array_like\n", + " The data to use for computing predictions. The rows is the number \n", + " of points to compute predictions, and columns is the number of\n", + " features.\n", + "\n", + " Returns\n", + " -------\n", + " p : array_like\n", + " Predictions and 0 or 1 for each row in X. \n", + " \n", + " Instructions\n", + " ------------\n", + " Complete the following code to make predictions using your learned \n", + " logistic regression parameters.You should set p to a vector of 0's and 1's \n", + " \"\"\"\n", + " m = X.shape[0] # Number of training examples\n", + "\n", + " # You need to return the following variables correctly\n", + " p = np.zeros(m)\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + " \n", + " \n", + " # ============================================================\n", + " return p" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "After you have completed the code in `predict`, we proceed to report the training accuracy of your classifier by computing the percentage of examples it got correct." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Predict probability for a student with score 45 on exam 1 \n", + "# and score 85 on exam 2 \n", + "prob = sigmoid(np.dot([1, 45, 85], theta))\n", + "print('For a student with scores 45 and 85,'\n", + " 'we predict an admission probability of {:.3f}'.format(prob))\n", + "print('Expected value: 0.775 +/- 0.002\\n')\n", + "\n", + "# Compute accuracy on our training set\n", + "p = predict(theta, X)\n", + "print('Train Accuracy: {:.2f} %'.format(np.mean(p == y) * 100))\n", + "print('Expected accuracy (approx): 89.00 %')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[4] = predict\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 2 Regularized logistic regression\n", + "\n", + "In this part of the exercise, you will implement regularized logistic regression to predict whether microchips from a fabrication plant passes quality assurance (QA). During QA, each microchip goes through various tests to ensure it is functioning correctly.\n", + "Suppose you are the product manager of the factory and you have the test results for some microchips on two different tests. From these two tests, you would like to determine whether the microchips should be accepted or rejected. To help you make the decision, you have a dataset of test results on past microchips, from which you can build a logistic regression model.\n", + "\n", + "First, we load the data from a CSV file:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Load Data\n", + "# The first two columns contains the X values and the third column\n", + "# contains the label (y).\n", + "data = np.loadtxt(os.path.join('Data', 'ex2data2.txt'), delimiter=',')\n", + "X = data[:, :2]\n", + "y = data[:, 2]" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2.1 Visualize the data\n", + "\n", + "Similar to the previous parts of this exercise, `plotData` is used to generate a figure, where the axes are the two test scores, and the positive (y = 1, accepted) and negative (y = 0, rejected) examples are shown with\n", + "different markers." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "plotData(X, y)\n", + "# Labels and Legend\n", + "pyplot.xlabel('Microchip Test 1')\n", + "pyplot.ylabel('Microchip Test 2')\n", + "\n", + "# Specified in plot order\n", + "pyplot.legend(['y = 1', 'y = 0'], loc='upper right')\n", + "pass" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The above figure shows that our dataset cannot be separated into positive and negative examples by a straight-line through the plot. Therefore, a straight-forward application of logistic regression will not perform well on this dataset since logistic regression will only be able to find a linear decision boundary.\n", + "\n", + "### 2.2 Feature mapping\n", + "\n", + "One way to fit the data better is to create more features from each data point. In the function `mapFeature` defined in the file `utils.py`, we will map the features into all polynomial terms of $x_1$ and $x_2$ up to the sixth power.\n", + "\n", + "$$ \\text{mapFeature}(x) = \\begin{bmatrix} 1 & x_1 & x_2 & x_1^2 & x_1 x_2 & x_2^2 & x_1^3 & \\dots & x_1 x_2^5 & x_2^6 \\end{bmatrix}^T $$\n", + "\n", + "As a result of this mapping, our vector of two features (the scores on two QA tests) has been transformed into a 28-dimensional vector. A logistic regression classifier trained on this higher-dimension feature vector will have a more complex decision boundary and will appear nonlinear when drawn in our 2-dimensional plot.\n", + "While the feature mapping allows us to build a more expressive classifier, it also more susceptible to overfitting. In the next parts of the exercise, you will implement regularized logistic regression to fit the data and also see for yourself how regularization can help combat the overfitting problem.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Note that mapFeature also adds a column of ones for us, so the intercept\n", + "# term is handled\n", + "X = utils.mapFeature(X[:, 0], X[:, 1])" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 2.3 Cost function and gradient\n", + "\n", + "Now you will implement code to compute the cost function and gradient for regularized logistic regression. Complete the code for the function `costFunctionReg` below to return the cost and gradient.\n", + "\n", + "Recall that the regularized cost function in logistic regression is\n", + "\n", + "$$ J(\\theta) = \\frac{1}{m} \\sum_{i=1}^m \\left[ -y^{(i)}\\log \\left( h_\\theta \\left(x^{(i)} \\right) \\right) - \\left( 1 - y^{(i)} \\right) \\log \\left( 1 - h_\\theta \\left( x^{(i)} \\right) \\right) \\right] + \\frac{\\lambda}{2m} \\sum_{j=1}^n \\theta_j^2 $$\n", + "\n", + "Note that you should not regularize the parameters $\\theta_0$. The gradient of the cost function is a vector where the $j^{th}$ element is defined as follows:\n", + "\n", + "$$ \\frac{\\partial J(\\theta)}{\\partial \\theta_0} = \\frac{1}{m} \\sum_{i=1}^m \\left( h_\\theta \\left(x^{(i)}\\right) - y^{(i)} \\right) x_j^{(i)} \\qquad \\text{for } j =0 $$\n", + "\n", + "$$ \\frac{\\partial J(\\theta)}{\\partial \\theta_j} = \\left( \\frac{1}{m} \\sum_{i=1}^m \\left( h_\\theta \\left(x^{(i)}\\right) - y^{(i)} \\right) x_j^{(i)} \\right) + \\frac{\\lambda}{m}\\theta_j \\qquad \\text{for } j \\ge 1 $$\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def costFunctionReg(theta, X, y, lambda_):\n", + " \"\"\"\n", + " Compute cost and gradient for logistic regression with regularization.\n", + " \n", + " Parameters\n", + " ----------\n", + " theta : array_like\n", + " Logistic regression parameters. A vector with shape (n, ). n is \n", + " the number of features including any intercept. If we have mapped\n", + " our initial features into polynomial features, then n is the total \n", + " number of polynomial features. \n", + " \n", + " X : array_like\n", + " The data set with shape (m x n). m is the number of examples, and\n", + " n is the number of features (after feature mapping).\n", + " \n", + " y : array_like\n", + " The data labels. A vector with shape (m, ).\n", + " \n", + " lambda_ : float\n", + " The regularization parameter. \n", + " \n", + " Returns\n", + " -------\n", + " J : float\n", + " The computed value for the regularized cost function. \n", + " \n", + " grad : array_like\n", + " A vector of shape (n, ) which is the gradient of the cost\n", + " function with respect to theta, at the current values of theta.\n", + " \n", + " Instructions\n", + " ------------\n", + " Compute the cost `J` of a particular choice of theta.\n", + " Compute the partial derivatives and set `grad` to the partial\n", + " derivatives of the cost w.r.t. each parameter in theta.\n", + " \"\"\"\n", + " # Initialize some useful values\n", + " m = y.size # number of training examples\n", + "\n", + " # You need to return the following variables correctly \n", + " J = 0\n", + " grad = np.zeros(theta.shape)\n", + "\n", + " # ===================== YOUR CODE HERE ======================\n", + "\n", + " \n", + " \n", + " # =============================================================\n", + " return J, grad" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Once you are done with the `costFunctionReg`, we call it below using the initial value of $\\theta$ (initialized to all zeros), and also another test case where $\\theta$ is all ones." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Initialize fitting parameters\n", + "initial_theta = np.zeros(X.shape[1])\n", + "\n", + "# Set regularization parameter lambda to 1\n", + "# DO NOT use `lambda` as a variable name in python\n", + "# because it is a python keyword\n", + "lambda_ = 1\n", + "\n", + "# Compute and display initial cost and gradient for regularized logistic\n", + "# regression\n", + "cost, grad = costFunctionReg(initial_theta, X, y, lambda_)\n", + "\n", + "print('Cost at initial theta (zeros): {:.3f}'.format(cost))\n", + "print('Expected cost (approx) : 0.693\\n')\n", + "\n", + "print('Gradient at initial theta (zeros) - first five values only:')\n", + "print('\\t[{:.4f}, {:.4f}, {:.4f}, {:.4f}, {:.4f}]'.format(*grad[:5]))\n", + "print('Expected gradients (approx) - first five values only:')\n", + "print('\\t[0.0085, 0.0188, 0.0001, 0.0503, 0.0115]\\n')\n", + "\n", + "\n", + "# Compute and display cost and gradient\n", + "# with all-ones theta and lambda = 10\n", + "test_theta = np.ones(X.shape[1])\n", + "cost, grad = costFunctionReg(test_theta, X, y, 10)\n", + "\n", + "print('------------------------------\\n')\n", + "print('Cost at test theta : {:.2f}'.format(cost))\n", + "print('Expected cost (approx): 3.16\\n')\n", + "\n", + "print('Gradient at initial theta (zeros) - first five values only:')\n", + "print('\\t[{:.4f}, {:.4f}, {:.4f}, {:.4f}, {:.4f}]'.format(*grad[:5]))\n", + "print('Expected gradients (approx) - first five values only:')\n", + "print('\\t[0.3460, 0.1614, 0.1948, 0.2269, 0.0922]')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[5] = costFunctionReg\n", + "grader[6] = costFunctionReg\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "#### 2.3.1 Learning parameters using `scipy.optimize.minimize`\n", + "\n", + "Similar to the previous parts, you will use `optimize.minimize` to learn the optimal parameters $\\theta$. If you have completed the cost and gradient for regularized logistic regression (`costFunctionReg`) correctly, you should be able to step through the next part of to learn the parameters $\\theta$ using `optimize.minimize`." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2.4 Plotting the decision boundary\n", + "\n", + "To help you visualize the model learned by this classifier, we have provided the function `plotDecisionBoundary` which plots the (non-linear) decision boundary that separates the positive and negative examples. In `plotDecisionBoundary`, we plot the non-linear decision boundary by computing the classifier’s predictions on an evenly spaced grid and then and draw a contour plot where the predictions change from y = 0 to y = 1. " + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2.5 Optional (ungraded) exercises\n", + "\n", + "In this part of the exercise, you will get to try out different regularization parameters for the dataset to understand how regularization prevents overfitting.\n", + "\n", + "Notice the changes in the decision boundary as you vary $\\lambda$. With a small\n", + "$\\lambda$, you should find that the classifier gets almost every training example correct, but draws a very complicated boundary, thus overfitting the data. See the following figures for the decision boundaries you should get for different values of $\\lambda$. \n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
\n", + " No regularization (overfitting)\n", + " \n", + " Decision boundary with regularization\n", + " \n", + " \n", + " Decision boundary with too much regularization\n", + " \n", + "
\n", + "\n", + "This is not a good decision boundary: for example, it predicts that a point at $x = (−0.25, 1.5)$ is accepted $(y = 1)$, which seems to be an incorrect decision given the training set.\n", + "With a larger $\\lambda$, you should see a plot that shows an simpler decision boundary which still separates the positives and negatives fairly well. However, if $\\lambda$ is set to too high a value, you will not get a good fit and the decision boundary will not follow the data so well, thus underfitting the data." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Initialize fitting parameters\n", + "initial_theta = np.zeros(X.shape[1])\n", + "\n", + "# Set regularization parameter lambda to 1 (you should vary this)\n", + "lambda_ = 1\n", + "\n", + "# set options for optimize.minimize\n", + "options= {'maxiter': 100}\n", + "\n", + "res = optimize.minimize(costFunctionReg,\n", + " initial_theta,\n", + " (X, y, lambda_),\n", + " jac=True,\n", + " method='TNC',\n", + " options=options)\n", + "\n", + "# the fun property of OptimizeResult object returns\n", + "# the value of costFunction at optimized theta\n", + "cost = res.fun\n", + "\n", + "# the optimized theta is in the x property of the result\n", + "theta = res.x\n", + "\n", + "utils.plotDecisionBoundary(plotData, theta, X, y)\n", + "pyplot.xlabel('Microchip Test 1')\n", + "pyplot.ylabel('Microchip Test 2')\n", + "pyplot.legend(['y = 1', 'y = 0'])\n", + "pyplot.grid(False)\n", + "pyplot.title('lambda = %0.2f' % lambda_)\n", + "\n", + "# Compute accuracy on our training set\n", + "p = predict(theta, X)\n", + "\n", + "print('Train Accuracy: %.1f %%' % (np.mean(p == y) * 100))\n", + "print('Expected accuracy (with lambda = 1): 83.1 % (approx)\\n')\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You do not need to submit any solutions for these optional (ungraded) exercises.*" + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.6.4" + } + }, + "nbformat": 4, + "nbformat_minor": 2 +} diff --git a/Exercise2/utils.py b/Exercise2/utils.py new file mode 100755 index 0000000..7c52dbe --- /dev/null +++ b/Exercise2/utils.py @@ -0,0 +1,147 @@ +import sys +import numpy as np +from matplotlib import pyplot + +sys.path.append('..') +from submission import SubmissionBase + + +def mapFeature(X1, X2, degree=6): + """ + Maps the two input features to quadratic features used in the regularization exercise. + + Returns a new feature array with more features, comprising of + X1, X2, X1.^2, X2.^2, X1*X2, X1*X2.^2, etc.. + + Parameters + ---------- + X1 : array_like + A vector of shape (m, 1), containing one feature for all examples. + + X2 : array_like + A vector of shape (m, 1), containing a second feature for all examples. + Inputs X1, X2 must be the same size. + + degree: int, optional + The polynomial degree. + + Returns + ------- + : array_like + A matrix of of m rows, and columns depend on the degree of polynomial. + """ + if X1.ndim > 0: + out = [np.ones(X1.shape[0])] + else: + out = [np.ones(1)] + + for i in range(1, degree + 1): + for j in range(i + 1): + out.append((X1 ** (i - j)) * (X2 ** j)) + + if X1.ndim > 0: + return np.stack(out, axis=1) + else: + return np.array(out) + + +def plotDecisionBoundary(plotData, theta, X, y): + """ + Plots the data points X and y into a new figure with the decision boundary defined by theta. + Plots the data points with * for the positive examples and o for the negative examples. + + Parameters + ---------- + plotData : func + A function reference for plotting the X, y data. + + theta : array_like + Parameters for logistic regression. A vector of shape (n+1, ). + + X : array_like + The input dataset. X is assumed to be a either: + 1) Mx3 matrix, where the first column is an all ones column for the intercept. + 2) MxN, N>3 matrix, where the first column is all ones. + + y : array_like + Vector of data labels of shape (m, ). + """ + # make sure theta is a numpy array + theta = np.array(theta) + + # Plot Data (remember first column in X is the intercept) + plotData(X[:, 1:3], y) + + if X.shape[1] <= 3: + # Only need 2 points to define a line, so choose two endpoints + plot_x = np.array([np.min(X[:, 1]) - 2, np.max(X[:, 1]) + 2]) + + # Calculate the decision boundary line + plot_y = (-1. / theta[2]) * (theta[1] * plot_x + theta[0]) + + # Plot, and adjust axes for better viewing + pyplot.plot(plot_x, plot_y) + + # Legend, specific for the exercise + pyplot.legend(['Admitted', 'Not admitted', 'Decision Boundary']) + pyplot.xlim([30, 100]) + pyplot.ylim([30, 100]) + else: + # Here is the grid range + u = np.linspace(-1, 1.5, 50) + v = np.linspace(-1, 1.5, 50) + + z = np.zeros((u.size, v.size)) + # Evaluate z = theta*x over the grid + for i, ui in enumerate(u): + for j, vj in enumerate(v): + z[i, j] = np.dot(mapFeature(ui, vj), theta) + + z = z.T # important to transpose z before calling contour + # print(z) + + # Plot z = 0 + pyplot.contour(u, v, z, levels=[0], linewidths=2, colors='g') + pyplot.contourf(u, v, z, levels=[np.min(z), 0, np.max(z)], cmap='Greens', alpha=0.4) + + +class Grader(SubmissionBase): + X = np.stack([np.ones(20), + np.exp(1) * np.sin(np.arange(1, 21)), + np.exp(0.5) * np.cos(np.arange(1, 21))], axis=1) + + y = (np.sin(X[:, 0] + X[:, 1]) > 0).astype(float) + + def __init__(self): + part_names = ['Sigmoid Function', + 'Logistic Regression Cost', + 'Logistic Regression Gradient', + 'Predict', + 'Regularized Logistic Regression Cost', + 'Regularized Logistic Regression Gradient'] + super().__init__('logistic-regression', part_names) + + def __iter__(self): + for part_id in range(1, 7): + try: + func = self.functions[part_id] + + # Each part has different expected arguments/different function + if part_id == 1: + res = func(self.X) + elif part_id == 2: + res = func(np.array([0.25, 0.5, -0.5]), self.X, self.y) + elif part_id == 3: + J, grad = func(np.array([0.25, 0.5, -0.5]), self.X, self.y) + res = grad + elif part_id == 4: + res = func(np.array([0.25, 0.5, -0.5]), self.X) + elif part_id == 5: + res = func(np.array([0.25, 0.5, -0.5]), self.X, self.y, 0.1) + elif part_id == 6: + res = func(np.array([0.25, 0.5, -0.5]), self.X, self.y, 0.1)[1] + else: + raise KeyError + yield part_id, res + except KeyError: + yield part_id, 0 diff --git a/Exercise3/Data/ex3data1.mat b/Exercise3/Data/ex3data1.mat new file mode 100755 index 0000000..371bd0c Binary files /dev/null and b/Exercise3/Data/ex3data1.mat differ diff --git a/Exercise3/Data/ex3weights.mat b/Exercise3/Data/ex3weights.mat new file mode 100755 index 0000000..ace2a09 Binary files /dev/null and b/Exercise3/Data/ex3weights.mat differ diff --git a/Exercise3/Figures/neuralnetwork.png b/Exercise3/Figures/neuralnetwork.png new file mode 100755 index 0000000..140fdb0 Binary files /dev/null and b/Exercise3/Figures/neuralnetwork.png differ diff --git a/Exercise3/exercise3.ipynb b/Exercise3/exercise3.ipynb new file mode 100755 index 0000000..c577171 --- /dev/null +++ b/Exercise3/exercise3.ipynb @@ -0,0 +1,923 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Programming Exercise 3\n", + "# Multi-class Classification and Neural Networks\n", + "\n", + "## Introduction\n", + "\n", + "\n", + "In this exercise, you will implement one-vs-all logistic regression and neural networks to recognize handwritten digits. Before starting the programming exercise, we strongly recommend watching the video lectures and completing the review questions for the associated topics. \n", + "\n", + "All the information you need for solving this assignment is in this notebook, and all the code you will be implementing will take place within this notebook. The assignment can be promptly submitted to the coursera grader directly from this notebook (code and instructions are included below).\n", + "\n", + "Before we begin with the exercises, we need to import all libraries required for this programming exercise. Throughout the course, we will be using [`numpy`](http://www.numpy.org/) for all arrays and matrix operations, [`matplotlib`](https://matplotlib.org/) for plotting, and [`scipy`](https://docs.scipy.org/doc/scipy/reference/) for scientific and numerical computation functions and tools. You can find instructions on how to install required libraries in the README file in the [github repository](https://github.com/dibgerge/ml-coursera-python-assignments)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# used for manipulating directory paths\n", + "import os\n", + "\n", + "# Scientific and vector computation for python\n", + "import numpy as np\n", + "\n", + "# Plotting library\n", + "from matplotlib import pyplot\n", + "\n", + "# Optimization module in scipy\n", + "from scipy import optimize\n", + "\n", + "# will be used to load MATLAB mat datafile format\n", + "from scipy.io import loadmat\n", + "\n", + "# library written for this exercise providing additional functions for assignment submission, and others\n", + "import utils\n", + "\n", + "# define the submission/grader object for this exercise\n", + "grader = utils.Grader()\n", + "\n", + "# tells matplotlib to embed plots within the notebook\n", + "%matplotlib inline" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Submission and Grading\n", + "\n", + "\n", + "After completing each part of the assignment, be sure to submit your solutions to the grader. The following is a breakdown of how each part of this exercise is scored.\n", + "\n", + "\n", + "| Section | Part | Submission function | Points \n", + "| :- |:- | :- | :-: \n", + "| 1 | [Regularized Logistic Regression](#section1) | [`lrCostFunction`](#lrCostFunction) | 30 \n", + "| 2 | [One-vs-all classifier training](#section2) | [`oneVsAll`](#oneVsAll) | 20 \n", + "| 3 | [One-vs-all classifier prediction](#section3) | [`predictOneVsAll`](#predictOneVsAll) | 20 \n", + "| 4 | [Neural Network Prediction Function](#section4) | [`predict`](#predict) | 30\n", + "| | Total Points | | 100 \n", + "\n", + "\n", + "You are allowed to submit your solutions multiple times, and we will take only the highest score into consideration.\n", + "\n", + "
\n", + "At the end of each section in this notebook, we have a cell which contains code for submitting the solutions thus far to the grader. Execute the cell to see your score up to the current section. For all your work to be submitted properly, you must execute those cells at least once. They must also be re-executed everytime the submitted function is updated.\n", + "
" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 1 Multi-class Classification\n", + "\n", + "For this exercise, you will use logistic regression and neural networks to recognize handwritten digits (from 0 to 9). Automated handwritten digit recognition is widely used today - from recognizing zip codes (postal codes)\n", + "on mail envelopes to recognizing amounts written on bank checks. This exercise will show you how the methods you have learned can be used for this classification task.\n", + "\n", + "In the first part of the exercise, you will extend your previous implementation of logistic regression and apply it to one-vs-all classification.\n", + "\n", + "### 1.1 Dataset\n", + "\n", + "You are given a data set in `ex3data1.mat` that contains 5000 training examples of handwritten digits (This is a subset of the [MNIST](http://yann.lecun.com/exdb/mnist) handwritten digit dataset). The `.mat` format means that that the data has been saved in a native Octave/MATLAB matrix format, instead of a text (ASCII) format like a csv-file. We use the `.mat` format here because this is the dataset provided in the MATLAB version of this assignment. Fortunately, python provides mechanisms to load MATLAB native format using the `loadmat` function within the `scipy.io` module. This function returns a python dictionary with keys containing the variable names within the `.mat` file. \n", + "\n", + "There are 5000 training examples in `ex3data1.mat`, where each training example is a 20 pixel by 20 pixel grayscale image of the digit. Each pixel is represented by a floating point number indicating the grayscale intensity at that location. The 20 by 20 grid of pixels is “unrolled” into a 400-dimensional vector. Each of these training examples becomes a single row in our data matrix `X`. This gives us a 5000 by 400 matrix `X` where every row is a training example for a handwritten digit image.\n", + "\n", + "$$ X = \\begin{bmatrix} - \\: (x^{(1)})^T \\: - \\\\ -\\: (x^{(2)})^T \\:- \\\\ \\vdots \\\\ - \\: (x^{(m)})^T \\:- \\end{bmatrix} $$\n", + "\n", + "The second part of the training set is a 5000-dimensional vector `y` that contains labels for the training set. \n", + "We start the exercise by first loading the dataset. Execute the cell below, you do not need to write any code here." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# 20x20 Input Images of Digits\n", + "input_layer_size = 400\n", + "\n", + "# 10 labels, from 1 to 10 (note that we have mapped \"0\" to label 10)\n", + "num_labels = 10\n", + "\n", + "# training data stored in arrays X, y\n", + "data = loadmat(os.path.join('Data', 'ex3data1.mat'))\n", + "X, y = data['X'], data['y'].ravel()\n", + "\n", + "# set the zero digit to 0, rather than its mapped 10 in this dataset\n", + "# This is an artifact due to the fact that this dataset was used in \n", + "# MATLAB where there is no index 0\n", + "y[y == 10] = 0\n", + "\n", + "m = y.size" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 1.2 Visualizing the data\n", + "\n", + "You will begin by visualizing a subset of the training set. In the following cell, the code randomly selects selects 100 rows from `X` and passes those rows to the `displayData` function. This function maps each row to a 20 pixel by 20 pixel grayscale image and displays the images together. We have provided the `displayData` function in the file `utils.py`. You are encouraged to examine the code to see how it works. Run the following cell to visualize the data." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Randomly select 100 data points to display\n", + "rand_indices = np.random.choice(m, 100, replace=False)\n", + "sel = X[rand_indices, :]\n", + "\n", + "utils.displayData(sel)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "collapsed": true + }, + "source": [ + "### 1.3 Vectorizing Logistic Regression\n", + "\n", + "You will be using multiple one-vs-all logistic regression models to build a multi-class classifier. Since there are 10 classes, you will need to train 10 separate logistic regression classifiers. To make this training efficient, it is important to ensure that your code is well vectorized. In this section, you will implement a vectorized version of logistic regression that does not employ any `for` loops. You can use your code in the previous exercise as a starting point for this exercise. \n", + "\n", + "To test your vectorized logistic regression, we will use custom data as defined in the following cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# test values for the parameters theta\n", + "theta_t = np.array([-2, -1, 1, 2], dtype=float)\n", + "\n", + "# test values for the inputs\n", + "X_t = np.concatenate([np.ones((5, 1)), np.arange(1, 16).reshape(5, 3, order='F')/10.0], axis=1)\n", + "\n", + "# test values for the labels\n", + "y_t = np.array([1, 0, 1, 0, 1])\n", + "\n", + "# test value for the regularization parameter\n", + "lambda_t = 3" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "#### 1.3.1 Vectorizing the cost function \n", + "\n", + "We will begin by writing a vectorized version of the cost function. Recall that in (unregularized) logistic regression, the cost function is\n", + "\n", + "$$ J(\\theta) = \\frac{1}{m} \\sum_{i=1}^m \\left[ -y^{(i)} \\log \\left( h_\\theta\\left( x^{(i)} \\right) \\right) - \\left(1 - y^{(i)} \\right) \\log \\left(1 - h_\\theta \\left( x^{(i)} \\right) \\right) \\right] $$\n", + "\n", + "To compute each element in the summation, we have to compute $h_\\theta(x^{(i)})$ for every example $i$, where $h_\\theta(x^{(i)}) = g(\\theta^T x^{(i)})$ and $g(z) = \\frac{1}{1+e^{-z}}$ is the sigmoid function. It turns out that we can compute this quickly for all our examples by using matrix multiplication. Let us define $X$ and $\\theta$ as\n", + "\n", + "$$ X = \\begin{bmatrix} - \\left( x^{(1)} \\right)^T - \\\\ - \\left( x^{(2)} \\right)^T - \\\\ \\vdots \\\\ - \\left( x^{(m)} \\right)^T - \\end{bmatrix} \\qquad \\text{and} \\qquad \\theta = \\begin{bmatrix} \\theta_0 \\\\ \\theta_1 \\\\ \\vdots \\\\ \\theta_n \\end{bmatrix} $$\n", + "\n", + "Then, by computing the matrix product $X\\theta$, we have: \n", + "\n", + "$$ X\\theta = \\begin{bmatrix} - \\left( x^{(1)} \\right)^T\\theta - \\\\ - \\left( x^{(2)} \\right)^T\\theta - \\\\ \\vdots \\\\ - \\left( x^{(m)} \\right)^T\\theta - \\end{bmatrix} = \\begin{bmatrix} - \\theta^T x^{(1)} - \\\\ - \\theta^T x^{(2)} - \\\\ \\vdots \\\\ - \\theta^T x^{(m)} - \\end{bmatrix} $$\n", + "\n", + "In the last equality, we used the fact that $a^Tb = b^Ta$ if $a$ and $b$ are vectors. This allows us to compute the products $\\theta^T x^{(i)}$ for all our examples $i$ in one line of code.\n", + "\n", + "#### 1.3.2 Vectorizing the gradient\n", + "\n", + "Recall that the gradient of the (unregularized) logistic regression cost is a vector where the $j^{th}$ element is defined as\n", + "\n", + "$$ \\frac{\\partial J }{\\partial \\theta_j} = \\frac{1}{m} \\sum_{i=1}^m \\left( \\left( h_\\theta\\left(x^{(i)}\\right) - y^{(i)} \\right)x_j^{(i)} \\right) $$\n", + "\n", + "To vectorize this operation over the dataset, we start by writing out all the partial derivatives explicitly for all $\\theta_j$,\n", + "\n", + "$$\n", + "\\begin{align*}\n", + "\\begin{bmatrix} \n", + "\\frac{\\partial J}{\\partial \\theta_0} \\\\\n", + "\\frac{\\partial J}{\\partial \\theta_1} \\\\\n", + "\\frac{\\partial J}{\\partial \\theta_2} \\\\\n", + "\\vdots \\\\\n", + "\\frac{\\partial J}{\\partial \\theta_n}\n", + "\\end{bmatrix} = &\n", + "\\frac{1}{m} \\begin{bmatrix}\n", + "\\sum_{i=1}^m \\left( \\left(h_\\theta\\left(x^{(i)}\\right) - y^{(i)} \\right)x_0^{(i)}\\right) \\\\\n", + "\\sum_{i=1}^m \\left( \\left(h_\\theta\\left(x^{(i)}\\right) - y^{(i)} \\right)x_1^{(i)}\\right) \\\\\n", + "\\sum_{i=1}^m \\left( \\left(h_\\theta\\left(x^{(i)}\\right) - y^{(i)} \\right)x_2^{(i)}\\right) \\\\\n", + "\\vdots \\\\\n", + "\\sum_{i=1}^m \\left( \\left(h_\\theta\\left(x^{(i)}\\right) - y^{(i)} \\right)x_n^{(i)}\\right) \\\\\n", + "\\end{bmatrix} \\\\\n", + "= & \\frac{1}{m} \\sum_{i=1}^m \\left( \\left(h_\\theta\\left(x^{(i)}\\right) - y^{(i)} \\right)x^{(i)}\\right) \\\\\n", + "= & \\frac{1}{m} X^T \\left( h_\\theta(x) - y\\right)\n", + "\\end{align*}\n", + "$$\n", + "\n", + "where\n", + "\n", + "$$ h_\\theta(x) - y = \n", + "\\begin{bmatrix}\n", + "h_\\theta\\left(x^{(1)}\\right) - y^{(1)} \\\\\n", + "h_\\theta\\left(x^{(2)}\\right) - y^{(2)} \\\\\n", + "\\vdots \\\\\n", + "h_\\theta\\left(x^{(m)}\\right) - y^{(m)} \n", + "\\end{bmatrix} $$\n", + "\n", + "Note that $x^{(i)}$ is a vector, while $h_\\theta\\left(x^{(i)}\\right) - y^{(i)}$ is a scalar (single number).\n", + "To understand the last step of the derivation, let $\\beta_i = (h_\\theta\\left(x^{(m)}\\right) - y^{(m)})$ and\n", + "observe that:\n", + "\n", + "$$ \\sum_i \\beta_ix^{(i)} = \\begin{bmatrix} \n", + "| & | & & | \\\\\n", + "x^{(1)} & x^{(2)} & \\cdots & x^{(m)} \\\\\n", + "| & | & & | \n", + "\\end{bmatrix}\n", + "\\begin{bmatrix}\n", + "\\beta_1 \\\\\n", + "\\beta_2 \\\\\n", + "\\vdots \\\\\n", + "\\beta_m\n", + "\\end{bmatrix} = x^T \\beta\n", + "$$\n", + "\n", + "where the values $\\beta_i = \\left( h_\\theta(x^{(i)} - y^{(i)} \\right)$.\n", + "\n", + "The expression above allows us to compute all the partial derivatives\n", + "without any loops. If you are comfortable with linear algebra, we encourage you to work through the matrix multiplications above to convince yourself that the vectorized version does the same computations. \n", + "\n", + "Your job is to write the unregularized cost function `lrCostFunction` which returns both the cost function $J(\\theta)$ and its gradient $\\frac{\\partial J}{\\partial \\theta}$. Your implementation should use the strategy we presented above to calculate $\\theta^T x^{(i)}$. You should also use a vectorized approach for the rest of the cost function. A fully vectorized version of `lrCostFunction` should not contain any loops.\n", + "\n", + "
\n", + "**Debugging Tip:** Vectorizing code can sometimes be tricky. One common strategy for debugging is to print out the sizes of the matrices you are working with using the `shape` property of `numpy` arrays. For example, given a data matrix $X$ of size $100 \\times 20$ (100 examples, 20 features) and $\\theta$, a vector with size $20$, you can observe that `np.dot(X, theta)` is a valid multiplication operation, while `np.dot(theta, X)` is not. Furthermore, if you have a non-vectorized version of your code, you can compare the output of your vectorized code and non-vectorized code to make sure that they produce the same outputs.\n", + "
\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def lrCostFunction(theta, X, y, lambda_):\n", + " \"\"\"\n", + " Computes the cost of using theta as the parameter for regularized\n", + " logistic regression and the gradient of the cost w.r.t. to the parameters.\n", + " \n", + " Parameters\n", + " ----------\n", + " theta : array_like\n", + " Logistic regression parameters. A vector with shape (n, ). n is \n", + " the number of features including any intercept. \n", + " \n", + " X : array_like\n", + " The data set with shape (m x n). m is the number of examples, and\n", + " n is the number of features (including intercept).\n", + " \n", + " y : array_like\n", + " The data labels. A vector with shape (m, ).\n", + " \n", + " lambda_ : float\n", + " The regularization parameter. \n", + " \n", + " Returns\n", + " -------\n", + " J : float\n", + " The computed value for the regularized cost function. \n", + " \n", + " grad : array_like\n", + " A vector of shape (n, ) which is the gradient of the cost\n", + " function with respect to theta, at the current values of theta.\n", + " \n", + " Instructions\n", + " ------------\n", + " Compute the cost of a particular choice of theta. You should set J to the cost.\n", + " Compute the partial derivatives and set grad to the partial\n", + " derivatives of the cost w.r.t. each parameter in theta\n", + " \n", + " Hint 1\n", + " ------\n", + " The computation of the cost function and gradients can be efficiently\n", + " vectorized. For example, consider the computation\n", + " \n", + " sigmoid(X * theta)\n", + " \n", + " Each row of the resulting matrix will contain the value of the prediction\n", + " for that example. You can make use of this to vectorize the cost function\n", + " and gradient computations. \n", + " \n", + " Hint 2\n", + " ------\n", + " When computing the gradient of the regularized cost function, there are\n", + " many possible vectorized solutions, but one solution looks like:\n", + " \n", + " grad = (unregularized gradient for logistic regression)\n", + " temp = theta \n", + " temp[0] = 0 # because we don't add anything for j = 0\n", + " grad = grad + YOUR_CODE_HERE (using the temp variable)\n", + " \n", + " Hint 3\n", + " ------\n", + " We have provided the implementatation of the sigmoid function within \n", + " the file `utils.py`. At the start of the notebook, we imported this file\n", + " as a module. Thus to access the sigmoid function within that file, you can\n", + " do the following: `utils.sigmoid(z)`.\n", + " \n", + " \"\"\"\n", + " #Initialize some useful values\n", + " m = y.size\n", + " \n", + " # convert labels to ints if their type is bool\n", + " if y.dtype == bool:\n", + " y = y.astype(int)\n", + " \n", + " # You need to return the following variables correctly\n", + " J = 0\n", + " grad = np.zeros(theta.shape)\n", + " \n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + "\n", + " \n", + " # =============================================================\n", + " return J, grad" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "#### 1.3.3 Vectorizing regularized logistic regression\n", + "\n", + "After you have implemented vectorization for logistic regression, you will now\n", + "add regularization to the cost function. Recall that for regularized logistic\n", + "regression, the cost function is defined as\n", + "\n", + "$$ J(\\theta) = \\frac{1}{m} \\sum_{i=1}^m \\left[ -y^{(i)} \\log \\left(h_\\theta\\left(x^{(i)} \\right)\\right) - \\left( 1 - y^{(i)} \\right) \\log\\left(1 - h_\\theta \\left(x^{(i)} \\right) \\right) \\right] + \\frac{\\lambda}{2m} \\sum_{j=1}^n \\theta_j^2 $$\n", + "\n", + "Note that you should not be regularizing $\\theta_0$ which is used for the bias term.\n", + "Correspondingly, the partial derivative of regularized logistic regression cost for $\\theta_j$ is defined as\n", + "\n", + "$$\n", + "\\begin{align*}\n", + "& \\frac{\\partial J(\\theta)}{\\partial \\theta_0} = \\frac{1}{m} \\sum_{i=1}^m \\left( h_\\theta\\left( x^{(i)} \\right) - y^{(i)} \\right) x_j^{(i)} & \\text{for } j = 0 \\\\\n", + "& \\frac{\\partial J(\\theta)}{\\partial \\theta_0} = \\left( \\frac{1}{m} \\sum_{i=1}^m \\left( h_\\theta\\left( x^{(i)} \\right) - y^{(i)} \\right) x_j^{(i)} \\right) + \\frac{\\lambda}{m} \\theta_j & \\text{for } j \\ge 1\n", + "\\end{align*}\n", + "$$\n", + "\n", + "Now modify your code in lrCostFunction in the [**previous cell**](#lrCostFunction) to account for regularization. Once again, you should not put any loops into your code.\n", + "\n", + "
\n", + "**python/numpy Tip:** When implementing the vectorization for regularized logistic regression, you might often want to only sum and update certain elements of $\\theta$. In `numpy`, you can index into the matrices to access and update only certain elements. For example, A[:, 3:5]\n", + "= B[:, 1:3] will replaces the columns with index 3 to 5 of A with the columns with index 1 to 3 from B. To select columns (or rows) until the end of the matrix, you can leave the right hand side of the colon blank. For example, A[:, 2:] will only return elements from the $3^{rd}$ to last columns of $A$. If you leave the left hand size of the colon blank, you will select elements from the beginning of the matrix. For example, A[:, :2] selects the first two columns, and is equivalent to A[:, 0:2]. In addition, you can use negative indices to index arrays from the end. Thus, A[:, :-1] selects all columns of A except the last column, and A[:, -5:] selects the $5^{th}$ column from the end to the last column. Thus, you could use this together with the sum and power ($^{**}$) operations to compute the sum of only the elements you are interested in (e.g., `np.sum(z[1:]**2)`). In the starter code, `lrCostFunction`, we have also provided hints on yet another possible method computing the regularized gradient.\n", + "
\n", + "\n", + "Once you finished your implementation, you can call the function `lrCostFunction` to test your solution using the following cell:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "J, grad = lrCostFunction(theta_t, X_t, y_t, lambda_t)\n", + "\n", + "print('Cost : {:.6f}'.format(J))\n", + "print('Expected cost: 2.534819')\n", + "print('-----------------------')\n", + "print('Gradients:')\n", + "print(' [{:.6f}, {:.6f}, {:.6f}, {:.6f}]'.format(*grad))\n", + "print('Expected gradients:')\n", + "print(' [0.146561, -0.548558, 0.724722, 1.398003]');" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "After completing a part of the exercise, you can submit your solutions for grading by first adding the function you modified to the submission object, and then sending your function to Coursera for grading. \n", + "\n", + "The submission script will prompt you for your login e-mail and submission token. You can obtain a submission token from the web page for the assignment. You are allowed to submit your solutions multiple times, and we will take only the highest score into consideration.\n", + "\n", + "*Execute the following cell to grade your solution to the first part of this exercise.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# appends the implemented function in part 1 to the grader object\n", + "grader[1] = lrCostFunction\n", + "\n", + "# send the added functions to coursera grader for getting a grade on this part\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 1.4 One-vs-all Classification\n", + "\n", + "In this part of the exercise, you will implement one-vs-all classification by training multiple regularized logistic regression classifiers, one for each of the $K$ classes in our dataset. In the handwritten digits dataset, $K = 10$, but your code should work for any value of $K$. \n", + "\n", + "You should now complete the code for the function `oneVsAll` below, to train one classifier for each class. In particular, your code should return all the classifier parameters in a matrix $\\theta \\in \\mathbb{R}^{K \\times (N +1)}$, where each row of $\\theta$ corresponds to the learned logistic regression parameters for one class. You can do this with a “for”-loop from $0$ to $K-1$, training each classifier independently.\n", + "\n", + "Note that the `y` argument to this function is a vector of labels from 0 to 9. When training the classifier for class $k \\in \\{0, ..., K-1\\}$, you will want a K-dimensional vector of labels $y$, where $y_j \\in 0, 1$ indicates whether the $j^{th}$ training instance belongs to class $k$ $(y_j = 1)$, or if it belongs to a different\n", + "class $(y_j = 0)$. You may find logical arrays helpful for this task. \n", + "\n", + "Furthermore, you will be using scipy's `optimize.minimize` for this exercise. \n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def oneVsAll(X, y, num_labels, lambda_):\n", + " \"\"\"\n", + " Trains num_labels logistic regression classifiers and returns\n", + " each of these classifiers in a matrix all_theta, where the i-th\n", + " row of all_theta corresponds to the classifier for label i.\n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like\n", + " The input dataset of shape (m x n). m is the number of \n", + " data points, and n is the number of features. Note that we \n", + " do not assume that the intercept term (or bias) is in X, however\n", + " we provide the code below to add the bias term to X. \n", + " \n", + " y : array_like\n", + " The data labels. A vector of shape (m, ).\n", + " \n", + " num_labels : int\n", + " Number of possible labels.\n", + " \n", + " lambda_ : float\n", + " The logistic regularization parameter.\n", + " \n", + " Returns\n", + " -------\n", + " all_theta : array_like\n", + " The trained parameters for logistic regression for each class.\n", + " This is a matrix of shape (K x n+1) where K is number of classes\n", + " (ie. `numlabels`) and n is number of features without the bias.\n", + " \n", + " Instructions\n", + " ------------\n", + " You should complete the following code to train `num_labels`\n", + " logistic regression classifiers with regularization parameter `lambda_`. \n", + " \n", + " Hint\n", + " ----\n", + " You can use y == c to obtain a vector of 1's and 0's that tell you\n", + " whether the ground truth is true/false for this class.\n", + " \n", + " Note\n", + " ----\n", + " For this assignment, we recommend using `scipy.optimize.minimize(method='CG')`\n", + " to optimize the cost function. It is okay to use a for-loop \n", + " (`for c in range(num_labels):`) to loop over the different classes.\n", + " \n", + " Example Code\n", + " ------------\n", + " \n", + " # Set Initial theta\n", + " initial_theta = np.zeros(n + 1)\n", + " \n", + " # Set options for minimize\n", + " options = {'maxiter', 50}\n", + " \n", + " # Run minimize to obtain the optimal theta. This function will \n", + " # return a class object where theta is in `res.x` and cost in `res.fun`\n", + " res = optimize.minimize(lrCostFunction, \n", + " initial_theta, \n", + " (X, (y == c), lambda_), \n", + " jac=True, \n", + " method='TNC')\n", + " options=options) \n", + " \"\"\"\n", + " # Some useful variables\n", + " m, n = X.shape\n", + " \n", + " # You need to return the following variables correctly \n", + " all_theta = np.zeros((num_labels, n + 1))\n", + "\n", + " # Add ones to the X data matrix\n", + " X = np.concatenate([np.ones((m, 1)), X], axis=1)\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + " \n", + "\n", + "\n", + " # ============================================================\n", + " return all_theta" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "After you have completed the code for `oneVsAll`, the following cell will use your implementation to train a multi-class classifier. " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "lambda_ = 0.1\n", + "all_theta = oneVsAll(X, y, num_labels, lambda_)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[2] = oneVsAll\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "#### 1.4.1 One-vs-all Prediction\n", + "\n", + "After training your one-vs-all classifier, you can now use it to predict the digit contained in a given image. For each input, you should compute the “probability” that it belongs to each class using the trained logistic regression classifiers. Your one-vs-all prediction function will pick the class for which the corresponding logistic regression classifier outputs the highest probability and return the class label (0, 1, ..., K-1) as the prediction for the input example. You should now complete the code in the function `predictOneVsAll` to use the one-vs-all classifier for making predictions. \n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def predictOneVsAll(all_theta, X):\n", + " \"\"\"\n", + " Return a vector of predictions for each example in the matrix X. \n", + " Note that X contains the examples in rows. all_theta is a matrix where\n", + " the i-th row is a trained logistic regression theta vector for the \n", + " i-th class. You should set p to a vector of values from 0..K-1 \n", + " (e.g., p = [0, 2, 0, 1] predicts classes 0, 2, 0, 1 for 4 examples) .\n", + " \n", + " Parameters\n", + " ----------\n", + " all_theta : array_like\n", + " The trained parameters for logistic regression for each class.\n", + " This is a matrix of shape (K x n+1) where K is number of classes\n", + " and n is number of features without the bias.\n", + " \n", + " X : array_like\n", + " Data points to predict their labels. This is a matrix of shape \n", + " (m x n) where m is number of data points to predict, and n is number \n", + " of features without the bias term. Note we add the bias term for X in \n", + " this function. \n", + " \n", + " Returns\n", + " -------\n", + " p : array_like\n", + " The predictions for each data point in X. This is a vector of shape (m, ).\n", + " \n", + " Instructions\n", + " ------------\n", + " Complete the following code to make predictions using your learned logistic\n", + " regression parameters (one-vs-all). You should set p to a vector of predictions\n", + " (from 0 to num_labels-1).\n", + " \n", + " Hint\n", + " ----\n", + " This code can be done all vectorized using the numpy argmax function.\n", + " In particular, the argmax function returns the index of the max element,\n", + " for more information see '?np.argmax' or search online. If your examples\n", + " are in rows, then, you can use np.argmax(A, axis=1) to obtain the index \n", + " of the max for each row.\n", + " \"\"\"\n", + " m = X.shape[0];\n", + " num_labels = all_theta.shape[0]\n", + "\n", + " # You need to return the following variables correctly \n", + " p = np.zeros(m)\n", + "\n", + " # Add ones to the X data matrix\n", + " X = np.concatenate([np.ones((m, 1)), X], axis=1)\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + "\n", + " \n", + " # ============================================================\n", + " return p" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Once you are done, call your `predictOneVsAll` function using the learned value of $\\theta$. You should see that the training set accuracy is about 95.1% (i.e., it classifies 95.1% of the examples in the training set correctly)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "pred = predictOneVsAll(all_theta, X)\n", + "print('Training Set Accuracy: {:.2f}%'.format(np.mean(pred == y) * 100))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[3] = predictOneVsAll\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 2 Neural Networks\n", + "\n", + "In the previous part of this exercise, you implemented multi-class logistic regression to recognize handwritten digits. However, logistic regression cannot form more complex hypotheses as it is only a linear classifier (You could add more features - such as polynomial features - to logistic regression, but that can be very expensive to train).\n", + "\n", + "In this part of the exercise, you will implement a neural network to recognize handwritten digits using the same training set as before. The neural network will be able to represent complex models that form non-linear hypotheses. For this week, you will be using parameters from a neural network that we have already trained. Your goal is to implement the feedforward propagation algorithm to use our weights for prediction. In next week’s exercise, you will write the backpropagation algorithm for learning the neural network parameters. \n", + "\n", + "We start by first reloading and visualizing the dataset which contains the MNIST handwritten digits (this is the same as we did in the first part of this exercise, we reload it here to ensure the variables have not been modified). " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# training data stored in arrays X, y\n", + "data = loadmat(os.path.join('Data', 'ex3data1.mat'))\n", + "X, y = data['X'], data['y'].ravel()\n", + "\n", + "# set the zero digit to 0, rather than its mapped 10 in this dataset\n", + "# This is an artifact due to the fact that this dataset was used in \n", + "# MATLAB where there is no index 0\n", + "y[y == 10] = 0\n", + "\n", + "# get number of examples in dataset\n", + "m = y.size\n", + "\n", + "# randomly permute examples, to be used for visualizing one \n", + "# picture at a time\n", + "indices = np.random.permutation(m)\n", + "\n", + "# Randomly select 100 data points to display\n", + "rand_indices = np.random.choice(m, 100, replace=False)\n", + "sel = X[rand_indices, :]\n", + "\n", + "utils.displayData(sel)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 2.1 Model representation \n", + "\n", + "Our neural network is shown in the following figure.\n", + "\n", + "![Neural network](Figures/neuralnetwork.png)\n", + "\n", + "It has 3 layers: an input layer, a hidden layer and an output layer. Recall that our inputs are pixel values of digit images. Since the images are of size 20×20, this gives us 400 input layer units (excluding the extra bias unit which always outputs +1). As before, the training data will be loaded into the variables X and y. \n", + "\n", + "You have been provided with a set of network parameters ($\\Theta^{(1)}$, $\\Theta^{(2)}$) already trained by us. These are stored in `ex3weights.mat`. The following cell loads those parameters into `Theta1` and `Theta2`. The parameters have dimensions that are sized for a neural network with 25 units in the second layer and 10 output units (corresponding to the 10 digit classes)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Setup the parameters you will use for this exercise\n", + "input_layer_size = 400 # 20x20 Input Images of Digits\n", + "hidden_layer_size = 25 # 25 hidden units\n", + "num_labels = 10 # 10 labels, from 0 to 9\n", + "\n", + "# Load the .mat file, which returns a dictionary \n", + "weights = loadmat(os.path.join('Data', 'ex3weights.mat'))\n", + "\n", + "# get the model weights from the dictionary\n", + "# Theta1 has size 25 x 401\n", + "# Theta2 has size 10 x 26\n", + "Theta1, Theta2 = weights['Theta1'], weights['Theta2']\n", + "\n", + "# swap first and last columns of Theta2, due to legacy from MATLAB indexing, \n", + "# since the weight file ex3weights.mat was saved based on MATLAB indexing\n", + "Theta2 = np.roll(Theta2, 1, axis=0)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 2.2 Feedforward Propagation and Prediction\n", + "\n", + "Now you will implement feedforward propagation for the neural network. You will need to complete the code in the function `predict` to return the neural network’s prediction. You should implement the feedforward computation that computes $h_\\theta(x^{(i)})$ for every example $i$ and returns the associated predictions. Similar to the one-vs-all classification strategy, the prediction from the neural network will be the label that has the largest output $\\left( h_\\theta(x) \\right)_k$.\n", + "\n", + "
\n", + "**Implementation Note:** The matrix $X$ contains the examples in rows. When you complete the code in the function `predict`, you will need to add the column of 1’s to the matrix. The matrices `Theta1` and `Theta2` contain the parameters for each unit in rows. Specifically, the first row of `Theta1` corresponds to the first hidden unit in the second layer. In `numpy`, when you compute $z^{(2)} = \\theta^{(1)}a^{(1)}$, be sure that you index (and if necessary, transpose) $X$ correctly so that you get $a^{(l)}$ as a 1-D vector.\n", + "
\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def predict(Theta1, Theta2, X):\n", + " \"\"\"\n", + " Predict the label of an input given a trained neural network.\n", + " \n", + " Parameters\n", + " ----------\n", + " Theta1 : array_like\n", + " Weights for the first layer in the neural network.\n", + " It has shape (2nd hidden layer size x input size)\n", + " \n", + " Theta2: array_like\n", + " Weights for the second layer in the neural network. \n", + " It has shape (output layer size x 2nd hidden layer size)\n", + " \n", + " X : array_like\n", + " The image inputs having shape (number of examples x image dimensions).\n", + " \n", + " Return \n", + " ------\n", + " p : array_like\n", + " Predictions vector containing the predicted label for each example.\n", + " It has a length equal to the number of examples.\n", + " \n", + " Instructions\n", + " ------------\n", + " Complete the following code to make predictions using your learned neural\n", + " network. You should set p to a vector containing labels \n", + " between 0 to (num_labels-1).\n", + " \n", + " Hint\n", + " ----\n", + " This code can be done all vectorized using the numpy argmax function.\n", + " In particular, the argmax function returns the index of the max element,\n", + " for more information see '?np.argmax' or search online. If your examples\n", + " are in rows, then, you can use np.argmax(A, axis=1) to obtain the index\n", + " of the max for each row.\n", + " \n", + " Note\n", + " ----\n", + " Remember, we have supplied the `sigmoid` function in the `utils.py` file. \n", + " You can use this function by calling `utils.sigmoid(z)`, where you can \n", + " replace `z` by the required input variable to sigmoid.\n", + " \"\"\"\n", + " # Make sure the input has two dimensions\n", + " if X.ndim == 1:\n", + " X = X[None] # promote to 2-dimensions\n", + " \n", + " # useful variables\n", + " m = X.shape[0]\n", + " num_labels = Theta2.shape[0]\n", + "\n", + " # You need to return the following variables correctly \n", + " p = np.zeros(X.shape[0])\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + "\n", + "\n", + " # =============================================================\n", + " return p" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Once you are done, call your predict function using the loaded set of parameters for `Theta1` and `Theta2`. You should see that the accuracy is about 97.5%." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "pred = predict(Theta1, Theta2, X)\n", + "print('Training Set Accuracy: {:.1f}%'.format(np.mean(pred == y) * 100))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "After that, we will display images from the training set one at a time, while at the same time printing out the predicted label for the displayed image. \n", + "\n", + "Run the following cell to display a single image the the neural network's prediction. You can run the cell multiple time to see predictions for different images." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "if indices.size > 0:\n", + " i, indices = indices[0], indices[1:]\n", + " utils.displayData(X[i, :], figsize=(4, 4))\n", + " pred = predict(Theta1, Theta2, X[i, :])\n", + " print('Neural Network Prediction: {}'.format(*pred))\n", + "else:\n", + " print('No more images to display!')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[4] = predict\n", + "grader.grade()" + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.6.4" + } + }, + "nbformat": 4, + "nbformat_minor": 2 +} diff --git a/Exercise3/utils.py b/Exercise3/utils.py new file mode 100755 index 0000000..633a563 --- /dev/null +++ b/Exercise3/utils.py @@ -0,0 +1,104 @@ +import sys +import numpy as np +from matplotlib import pyplot + +sys.path.append('..') +from submission import SubmissionBase + + +def displayData(X, example_width=None, figsize=(10, 10)): + """ + Displays 2D data stored in X in a nice grid. + """ + # Compute rows, cols + if X.ndim == 2: + m, n = X.shape + elif X.ndim == 1: + n = X.size + m = 1 + X = X[None] # Promote to a 2 dimensional array + else: + raise IndexError('Input X should be 1 or 2 dimensional.') + + example_width = example_width or int(np.round(np.sqrt(n))) + example_height = n / example_width + + # Compute number of items to display + display_rows = int(np.floor(np.sqrt(m))) + display_cols = int(np.ceil(m / display_rows)) + + fig, ax_array = pyplot.subplots(display_rows, display_cols, figsize=figsize) + fig.subplots_adjust(wspace=0.025, hspace=0.025) + + ax_array = [ax_array] if m == 1 else ax_array.ravel() + + for i, ax in enumerate(ax_array): + ax.imshow(X[i].reshape(example_width, example_width, order='F'), + cmap='Greys', extent=[0, 1, 0, 1]) + ax.axis('off') + + +def sigmoid(z): + """ + Computes the sigmoid of z. + """ + return 1.0 / (1.0 + np.exp(-z)) + + +class Grader(SubmissionBase): + # Random Test Cases + X = np.stack([np.ones(20), + np.exp(1) * np.sin(np.arange(1, 21)), + np.exp(0.5) * np.cos(np.arange(1, 21))], axis=1) + + y = (np.sin(X[:, 0] + X[:, 1]) > 0).astype(float) + + Xm = np.array([[-1, -1], + [-1, -2], + [-2, -1], + [-2, -2], + [1, 1], + [1, 2], + [2, 1], + [2, 2], + [-1, 1], + [-1, 2], + [-2, 1], + [-2, 2], + [1, -1], + [1, -2], + [-2, -1], + [-2, -2]]) + ym = np.array([0, 0, 0, 0, 1, 1, 1, 1, 2, 2, 2, 2, 3, 3, 3, 3]) + + t1 = np.sin(np.reshape(np.arange(1, 25, 2), (4, 3), order='F')) + t2 = np.cos(np.reshape(np.arange(1, 41, 2), (4, 5), order='F')) + + def __init__(self): + part_names = ['Regularized Logistic Regression', + 'One-vs-All Classifier Training', + 'One-vs-All Classifier Prediction', + 'Neural Network Prediction Function'] + + super().__init__('multi-class-classification-and-neural-networks', part_names) + + def __iter__(self): + for part_id in range(1, 5): + try: + func = self.functions[part_id] + + # Each part has different expected arguments/different function + if part_id == 1: + res = func(np.array([0.25, 0.5, -0.5]), self.X, self.y, 0.1) + res = np.hstack(res).tolist() + elif part_id == 2: + res = func(self.Xm, self.ym, 4, 0.1) + elif part_id == 3: + res = func(self.t1, self.Xm) + 1 + elif part_id == 4: + res = func(self.t1, self.t2, self.Xm) + 1 + else: + raise KeyError + yield part_id, res + except KeyError: + yield part_id, 0 diff --git a/Exercise4/Data/ex4data1.mat b/Exercise4/Data/ex4data1.mat new file mode 100755 index 0000000..371bd0c Binary files /dev/null and b/Exercise4/Data/ex4data1.mat differ diff --git a/Exercise4/Data/ex4weights.mat b/Exercise4/Data/ex4weights.mat new file mode 100755 index 0000000..ace2a09 Binary files /dev/null and b/Exercise4/Data/ex4weights.mat differ diff --git a/Exercise4/Figures/ex4-backpropagation.png b/Exercise4/Figures/ex4-backpropagation.png new file mode 100755 index 0000000..62e1861 Binary files /dev/null and b/Exercise4/Figures/ex4-backpropagation.png differ diff --git a/Exercise4/Figures/neural_network.png b/Exercise4/Figures/neural_network.png new file mode 100755 index 0000000..140fdb0 Binary files /dev/null and b/Exercise4/Figures/neural_network.png differ diff --git a/Exercise4/exercise4.ipynb b/Exercise4/exercise4.ipynb new file mode 100755 index 0000000..d8ebee0 --- /dev/null +++ b/Exercise4/exercise4.ipynb @@ -0,0 +1,924 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Programming Exercise 4: Neural Networks Learning\n", + "\n", + "## Introduction\n", + "\n", + "In this exercise, you will implement the backpropagation algorithm for neural networks and apply it to the task of hand-written digit recognition. Before starting on the programming exercise, we strongly recommend watching the video lectures and completing the review questions for the associated topics.\n", + "\n", + "\n", + "All the information you need for solving this assignment is in this notebook, and all the code you will be implementing will take place within this notebook. The assignment can be promptly submitted to the coursera grader directly from this notebook (code and instructions are included below).\n", + "\n", + "Before we begin with the exercises, we need to import all libraries required for this programming exercise. Throughout the course, we will be using [`numpy`](http://www.numpy.org/) for all arrays and matrix operations, [`matplotlib`](https://matplotlib.org/) for plotting, and [`scipy`](https://docs.scipy.org/doc/scipy/reference/) for scientific and numerical computation functions and tools. You can find instructions on how to install required libraries in the README file in the [github repository](https://github.com/dibgerge/ml-coursera-python-assignments)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# used for manipulating directory paths\n", + "import os\n", + "\n", + "# Scientific and vector computation for python\n", + "import numpy as np\n", + "\n", + "# Plotting library\n", + "from matplotlib import pyplot\n", + "\n", + "# Optimization module in scipy\n", + "from scipy import optimize\n", + "\n", + "# will be used to load MATLAB mat datafile format\n", + "from scipy.io import loadmat\n", + "\n", + "# library written for this exercise providing additional functions for assignment submission, and others\n", + "import utils\n", + "\n", + "# define the submission/grader object for this exercise\n", + "grader = utils.Grader()\n", + "\n", + "# tells matplotlib to embed plots within the notebook\n", + "%matplotlib inline" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Submission and Grading\n", + "\n", + "\n", + "After completing each part of the assignment, be sure to submit your solutions to the grader. The following is a breakdown of how each part of this exercise is scored.\n", + "\n", + "\n", + "| Section | Part | Submission function | Points \n", + "| :- |:- | :- | :-: \n", + "| 1 | [Feedforward and Cost Function](#section1) | [`nnCostFunction`](#nnCostFunction) | 30 \n", + "| 2 | [Regularized Cost Function](#section2) | [`nnCostFunction`](#nnCostFunction) | 15 \n", + "| 3 | [Sigmoid Gradient](#section3) | [`sigmoidGradient`](#sigmoidGradient) | 5 \n", + "| 4 | [Neural Net Gradient Function (Backpropagation)](#section4) | [`nnCostFunction`](#nnCostFunction) | 40 \n", + "| 5 | [Regularized Gradient](#section5) | [`nnCostFunction`](#nnCostFunction) |10 \n", + "| | Total Points | | 100 \n", + "\n", + "\n", + "You are allowed to submit your solutions multiple times, and we will take only the highest score into consideration.\n", + "\n", + "
\n", + "At the end of each section in this notebook, we have a cell which contains code for submitting the solutions thus far to the grader. Execute the cell to see your score up to the current section. For all your work to be submitted properly, you must execute those cells at least once.\n", + "
" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Neural Networks\n", + "\n", + "In the previous exercise, you implemented feedforward propagation for neural networks and used it to predict handwritten digits with the weights we provided. In this exercise, you will implement the backpropagation algorithm to learn the parameters for the neural network.\n", + "\n", + "We start the exercise by first loading the dataset. " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# training data stored in arrays X, y\n", + "data = loadmat(os.path.join('Data', 'ex4data1.mat'))\n", + "X, y = data['X'], data['y'].ravel()\n", + "\n", + "# set the zero digit to 0, rather than its mapped 10 in this dataset\n", + "# This is an artifact due to the fact that this dataset was used in \n", + "# MATLAB where there is no index 0\n", + "y[y == 10] = 0\n", + "\n", + "# Number of training examples\n", + "m = y.size" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 1.1 Visualizing the data\n", + "\n", + "You will begin by visualizing a subset of the training set, using the function `displayData`, which is the same function we used in Exercise 3. It is provided in the `utils.py` file for this assignment as well. The dataset is also the same one you used in the previous exercise.\n", + "\n", + "There are 5000 training examples in `ex4data1.mat`, where each training example is a 20 pixel by 20 pixel grayscale image of the digit. Each pixel is represented by a floating point number indicating the grayscale intensity at that location. The 20 by 20 grid of pixels is “unrolled” into a 400-dimensional vector. Each\n", + "of these training examples becomes a single row in our data matrix $X$. This gives us a 5000 by 400 matrix $X$ where every row is a training example for a handwritten digit image.\n", + "\n", + "$$ X = \\begin{bmatrix} - \\left(x^{(1)} \\right)^T - \\\\\n", + "- \\left(x^{(2)} \\right)^T - \\\\\n", + "\\vdots \\\\\n", + "- \\left(x^{(m)} \\right)^T - \\\\\n", + "\\end{bmatrix}\n", + "$$\n", + "\n", + "The second part of the training set is a 5000-dimensional vector `y` that contains labels for the training set. \n", + "The following cell randomly selects 100 images from the dataset and plots them." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Randomly select 100 data points to display\n", + "rand_indices = np.random.choice(m, 100, replace=False)\n", + "sel = X[rand_indices, :]\n", + "\n", + "utils.displayData(sel)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 1.2 Model representation\n", + "\n", + "Our neural network is shown in the following figure.\n", + "\n", + "![](Figures/neural_network.png)\n", + "\n", + "It has 3 layers - an input layer, a hidden layer and an output layer. Recall that our inputs are pixel values\n", + "of digit images. Since the images are of size $20 \\times 20$, this gives us 400 input layer units (not counting the extra bias unit which always outputs +1). The training data was loaded into the variables `X` and `y` above.\n", + "\n", + "You have been provided with a set of network parameters ($\\Theta^{(1)}, \\Theta^{(2)}$) already trained by us. These are stored in `ex4weights.mat` and will be loaded in the next cell of this notebook into `Theta1` and `Theta2`. The parameters have dimensions that are sized for a neural network with 25 units in the second layer and 10 output units (corresponding to the 10 digit classes)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Setup the parameters you will use for this exercise\n", + "input_layer_size = 400 # 20x20 Input Images of Digits\n", + "hidden_layer_size = 25 # 25 hidden units\n", + "num_labels = 10 # 10 labels, from 0 to 9\n", + "\n", + "# Load the weights into variables Theta1 and Theta2\n", + "weights = loadmat(os.path.join('Data', 'ex4weights.mat'))\n", + "\n", + "# Theta1 has size 25 x 401\n", + "# Theta2 has size 10 x 26\n", + "Theta1, Theta2 = weights['Theta1'], weights['Theta2']\n", + "\n", + "# swap first and last columns of Theta2, due to legacy from MATLAB indexing, \n", + "# since the weight file ex3weights.mat was saved based on MATLAB indexing\n", + "Theta2 = np.roll(Theta2, 1, axis=0)\n", + "\n", + "# Unroll parameters \n", + "nn_params = np.concatenate([Theta1.ravel(), Theta2.ravel()])" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 1.3 Feedforward and cost function\n", + "\n", + "Now you will implement the cost function and gradient for the neural network. First, complete the code for the function `nnCostFunction` in the next cell to return the cost.\n", + "\n", + "Recall that the cost function for the neural network (without regularization) is:\n", + "\n", + "$$ J(\\theta) = \\frac{1}{m} \\sum_{i=1}^{m}\\sum_{k=1}^{K} \\left[ - y_k^{(i)} \\log \\left( \\left( h_\\theta \\left( x^{(i)} \\right) \\right)_k \\right) - \\left( 1 - y_k^{(i)} \\right) \\log \\left( 1 - \\left( h_\\theta \\left( x^{(i)} \\right) \\right)_k \\right) \\right]$$\n", + "\n", + "where $h_\\theta \\left( x^{(i)} \\right)$ is computed as shown in the neural network figure above, and K = 10 is the total number of possible labels. Note that $h_\\theta(x^{(i)})_k = a_k^{(3)}$ is the activation (output\n", + "value) of the $k^{th}$ output unit. Also, recall that whereas the original labels (in the variable y) were 0, 1, ..., 9, for the purpose of training a neural network, we need to encode the labels as vectors containing only values 0 or 1, so that\n", + "\n", + "$$ y = \n", + "\\begin{bmatrix} 1 \\\\ 0 \\\\ 0 \\\\\\vdots \\\\ 0 \\end{bmatrix}, \\quad\n", + "\\begin{bmatrix} 0 \\\\ 1 \\\\ 0 \\\\ \\vdots \\\\ 0 \\end{bmatrix}, \\quad \\cdots \\quad \\text{or} \\qquad\n", + "\\begin{bmatrix} 0 \\\\ 0 \\\\ 0 \\\\ \\vdots \\\\ 1 \\end{bmatrix}.\n", + "$$\n", + "\n", + "For example, if $x^{(i)}$ is an image of the digit 5, then the corresponding $y^{(i)}$ (that you should use with the cost function) should be a 10-dimensional vector with $y_5 = 1$, and the other elements equal to 0.\n", + "\n", + "You should implement the feedforward computation that computes $h_\\theta(x^{(i)})$ for every example $i$ and sum the cost over all examples. **Your code should also work for a dataset of any size, with any number of labels** (you can assume that there are always at least $K \\ge 3$ labels).\n", + "\n", + "
\n", + "**Implementation Note:** The matrix $X$ contains the examples in rows (i.e., X[i,:] is the i-th training example $x^{(i)}$, expressed as a $n \\times 1$ vector.) When you complete the code in `nnCostFunction`, you will need to add the column of 1’s to the X matrix. The parameters for each unit in the neural network is represented in Theta1 and Theta2 as one row. Specifically, the first row of Theta1 corresponds to the first hidden unit in the second layer. You can use a for-loop over the examples to compute the cost.\n", + "
\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def nnCostFunction(nn_params,\n", + " input_layer_size,\n", + " hidden_layer_size,\n", + " num_labels,\n", + " X, y, lambda_=0.0):\n", + " \"\"\"\n", + " Implements the neural network cost function and gradient for a two layer neural \n", + " network which performs classification. \n", + " \n", + " Parameters\n", + " ----------\n", + " nn_params : array_like\n", + " The parameters for the neural network which are \"unrolled\" into \n", + " a vector. This needs to be converted back into the weight matrices Theta1\n", + " and Theta2.\n", + " \n", + " input_layer_size : int\n", + " Number of features for the input layer. \n", + " \n", + " hidden_layer_size : int\n", + " Number of hidden units in the second layer.\n", + " \n", + " num_labels : int\n", + " Total number of labels, or equivalently number of units in output layer. \n", + " \n", + " X : array_like\n", + " Input dataset. A matrix of shape (m x input_layer_size).\n", + " \n", + " y : array_like\n", + " Dataset labels. A vector of shape (m,).\n", + " \n", + " lambda_ : float, optional\n", + " Regularization parameter.\n", + " \n", + " Returns\n", + " -------\n", + " J : float\n", + " The computed value for the cost function at the current weight values.\n", + " \n", + " grad : array_like\n", + " An \"unrolled\" vector of the partial derivatives of the concatenatation of\n", + " neural network weights Theta1 and Theta2.\n", + " \n", + " Instructions\n", + " ------------\n", + " You should complete the code by working through the following parts.\n", + " \n", + " - Part 1: Feedforward the neural network and return the cost in the \n", + " variable J. After implementing Part 1, you can verify that your\n", + " cost function computation is correct by verifying the cost\n", + " computed in the following cell.\n", + " \n", + " - Part 2: Implement the backpropagation algorithm to compute the gradients\n", + " Theta1_grad and Theta2_grad. You should return the partial derivatives of\n", + " the cost function with respect to Theta1 and Theta2 in Theta1_grad and\n", + " Theta2_grad, respectively. After implementing Part 2, you can check\n", + " that your implementation is correct by running checkNNGradients provided\n", + " in the utils.py module.\n", + " \n", + " Note: The vector y passed into the function is a vector of labels\n", + " containing values from 0..K-1. You need to map this vector into a \n", + " binary vector of 1's and 0's to be used with the neural network\n", + " cost function.\n", + " \n", + " Hint: We recommend implementing backpropagation using a for-loop\n", + " over the training examples if you are implementing it for the \n", + " first time.\n", + " \n", + " - Part 3: Implement regularization with the cost function and gradients.\n", + " \n", + " Hint: You can implement this around the code for\n", + " backpropagation. That is, you can compute the gradients for\n", + " the regularization separately and then add them to Theta1_grad\n", + " and Theta2_grad from Part 2.\n", + " \n", + " Note \n", + " ----\n", + " We have provided an implementation for the sigmoid function in the file \n", + " `utils.py` accompanying this assignment.\n", + " \"\"\"\n", + " # Reshape nn_params back into the parameters Theta1 and Theta2, the weight matrices\n", + " # for our 2 layer neural network\n", + " Theta1 = np.reshape(nn_params[:hidden_layer_size * (input_layer_size + 1)],\n", + " (hidden_layer_size, (input_layer_size + 1)))\n", + "\n", + " Theta2 = np.reshape(nn_params[(hidden_layer_size * (input_layer_size + 1)):],\n", + " (num_labels, (hidden_layer_size + 1)))\n", + "\n", + " # Setup some useful variables\n", + " m = y.size\n", + " \n", + " # You need to return the following variables correctly \n", + " J = 0\n", + " Theta1_grad = np.zeros(Theta1.shape)\n", + " Theta2_grad = np.zeros(Theta2.shape)\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + " \n", + " \n", + " # ================================================================\n", + " # Unroll gradients\n", + " # grad = np.concatenate([Theta1_grad.ravel(order=order), Theta2_grad.ravel(order=order)])\n", + " grad = np.concatenate([Theta1_grad.ravel(), Theta2_grad.ravel()])\n", + "\n", + " return J, grad" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "
\n", + "Use the following links to go back to the different parts of this exercise that require to modify the function `nnCostFunction`.
\n", + "\n", + "Back to:\n", + "- [Feedforward and cost function](#section1)\n", + "- [Regularized cost](#section2)\n", + "- [Neural Network Gradient (Backpropagation)](#section4)\n", + "- [Regularized Gradient](#section5)\n", + "
" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Once you are done, call your `nnCostFunction` using the loaded set of parameters for `Theta1` and `Theta2`. You should see that the cost is about 0.287629." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "lambda_ = 0\n", + "J, _ = nnCostFunction(nn_params, input_layer_size, hidden_layer_size,\n", + " num_labels, X, y, lambda_)\n", + "print('Cost at parameters (loaded from ex4weights): %.6f ' % J)\n", + "print('The cost should be about : 0.287629.')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader = utils.Grader()\n", + "grader[1] = nnCostFunction\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 1.4 Regularized cost function\n", + "\n", + "The cost function for neural networks with regularization is given by:\n", + "\n", + "\n", + "$$ J(\\theta) = \\frac{1}{m} \\sum_{i=1}^{m}\\sum_{k=1}^{K} \\left[ - y_k^{(i)} \\log \\left( \\left( h_\\theta \\left( x^{(i)} \\right) \\right)_k \\right) - \\left( 1 - y_k^{(i)} \\right) \\log \\left( 1 - \\left( h_\\theta \\left( x^{(i)} \\right) \\right)_k \\right) \\right] + \\frac{\\lambda}{2 m} \\left[ \\sum_{j=1}^{25} \\sum_{k=1}^{400} \\left( \\Theta_{j,k}^{(1)} \\right)^2 + \\sum_{j=1}^{10} \\sum_{k=1}^{25} \\left( \\Theta_{j,k}^{(2)} \\right)^2 \\right] $$\n", + "\n", + "You can assume that the neural network will only have 3 layers - an input layer, a hidden layer and an output layer. However, your code should work for any number of input units, hidden units and outputs units. While we\n", + "have explicitly listed the indices above for $\\Theta^{(1)}$ and $\\Theta^{(2)}$ for clarity, do note that your code should in general work with $\\Theta^{(1)}$ and $\\Theta^{(2)}$ of any size. Note that you should not be regularizing the terms that correspond to the bias. For the matrices `Theta1` and `Theta2`, this corresponds to the first column of each matrix. You should now add regularization to your cost function. Notice that you can first compute the unregularized cost function $J$ using your existing `nnCostFunction` and then later add the cost for the regularization terms.\n", + "\n", + "[Click here to go back to `nnCostFunction` for editing.](#nnCostFunction)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Once you are done, the next cell will call your `nnCostFunction` using the loaded set of parameters for `Theta1` and `Theta2`, and $\\lambda = 1$. You should see that the cost is about 0.383770." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Weight regularization parameter (we set this to 1 here).\n", + "lambda_ = 1\n", + "J, _ = nnCostFunction(nn_params, input_layer_size, hidden_layer_size,\n", + " num_labels, X, y, lambda_)\n", + "\n", + "print('Cost at parameters (loaded from ex4weights): %.6f' % J)\n", + "print('This value should be about : 0.383770.')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[2] = nnCostFunction\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 2 Backpropagation\n", + "\n", + "In this part of the exercise, you will implement the backpropagation algorithm to compute the gradient for the neural network cost function. You will need to update the function `nnCostFunction` so that it returns an appropriate value for `grad`. Once you have computed the gradient, you will be able to train the neural network by minimizing the cost function $J(\\theta)$ using an advanced optimizer such as `scipy`'s `optimize.minimize`.\n", + "You will first implement the backpropagation algorithm to compute the gradients for the parameters for the (unregularized) neural network. After you have verified that your gradient computation for the unregularized case is correct, you will implement the gradient for the regularized neural network." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 2.1 Sigmoid Gradient\n", + "\n", + "To help you get started with this part of the exercise, you will first implement\n", + "the sigmoid gradient function. The gradient for the sigmoid function can be\n", + "computed as\n", + "\n", + "$$ g'(z) = \\frac{d}{dz} g(z) = g(z)\\left(1-g(z)\\right) $$\n", + "\n", + "where\n", + "\n", + "$$ \\text{sigmoid}(z) = g(z) = \\frac{1}{1 + e^{-z}} $$\n", + "\n", + "Now complete the implementation of `sigmoidGradient` in the next cell.\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def sigmoidGradient(z):\n", + " \"\"\"\n", + " Computes the gradient of the sigmoid function evaluated at z. \n", + " This should work regardless if z is a matrix or a vector. \n", + " In particular, if z is a vector or matrix, you should return\n", + " the gradient for each element.\n", + " \n", + " Parameters\n", + " ----------\n", + " z : array_like\n", + " A vector or matrix as input to the sigmoid function. \n", + " \n", + " Returns\n", + " --------\n", + " g : array_like\n", + " Gradient of the sigmoid function. Has the same shape as z. \n", + " \n", + " Instructions\n", + " ------------\n", + " Compute the gradient of the sigmoid function evaluated at\n", + " each value of z (z can be a matrix, vector or scalar).\n", + " \n", + " Note\n", + " ----\n", + " We have provided an implementation of the sigmoid function \n", + " in `utils.py` file accompanying this assignment.\n", + " \"\"\"\n", + "\n", + " g = np.zeros(z.shape)\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + "\n", + "\n", + " # =============================================================\n", + " return g" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "When you are done, the following cell call `sigmoidGradient` on a given vector `z`. Try testing a few values by calling `sigmoidGradient(z)`. For large values (both positive and negative) of z, the gradient should be close to 0. When $z = 0$, the gradient should be exactly 0.25. Your code should also work with vectors and matrices. For a matrix, your function should perform the sigmoid gradient function on every element." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "z = np.array([-1, -0.5, 0, 0.5, 1])\n", + "g = sigmoidGradient(z)\n", + "print('Sigmoid gradient evaluated at [-1 -0.5 0 0.5 1]:\\n ')\n", + "print(g)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[3] = sigmoidGradient\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 2.2 Random Initialization\n", + "\n", + "When training neural networks, it is important to randomly initialize the parameters for symmetry breaking. One effective strategy for random initialization is to randomly select values for $\\Theta^{(l)}$ uniformly in the range $[-\\epsilon_{init}, \\epsilon_{init}]$. You should use $\\epsilon_{init} = 0.12$. This range of values ensures that the parameters are kept small and makes the learning more efficient.\n", + "\n", + "
\n", + "One effective strategy for choosing $\\epsilon_{init}$ is to base it on the number of units in the network. A good choice of $\\epsilon_{init}$ is $\\epsilon_{init} = \\frac{\\sqrt{6}}{\\sqrt{L_{in} + L_{out}}}$ where $L_{in} = s_l$ and $L_{out} = s_{l+1}$ are the number of units in the layers adjacent to $\\Theta^{l}$.\n", + "
\n", + "\n", + "Your job is to complete the function `randInitializeWeights` to initialize the weights for $\\Theta$. Modify the function by filling in the following code:\n", + "\n", + "```python\n", + "# Randomly initialize the weights to small values\n", + "W = np.random.rand(L_out, 1 + L_in) * 2 * epsilon_init - epsilon_init\n", + "```\n", + "Note that we give the function an argument for $\\epsilon$ with default value `epsilon_init = 0.12`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def randInitializeWeights(L_in, L_out, epsilon_init=0.12):\n", + " \"\"\"\n", + " Randomly initialize the weights of a layer in a neural network.\n", + " \n", + " Parameters\n", + " ----------\n", + " L_in : int\n", + " Number of incomming connections.\n", + " \n", + " L_out : int\n", + " Number of outgoing connections. \n", + " \n", + " epsilon_init : float, optional\n", + " Range of values which the weight can take from a uniform \n", + " distribution.\n", + " \n", + " Returns\n", + " -------\n", + " W : array_like\n", + " The weight initialiatized to random values. Note that W should\n", + " be set to a matrix of size(L_out, 1 + L_in) as\n", + " the first column of W handles the \"bias\" terms.\n", + " \n", + " Instructions\n", + " ------------\n", + " Initialize W randomly so that we break the symmetry while training\n", + " the neural network. Note that the first column of W corresponds \n", + " to the parameters for the bias unit.\n", + " \"\"\"\n", + "\n", + " # You need to return the following variables correctly \n", + " W = np.zeros((L_out, 1 + L_in))\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + "\n", + "\n", + " # ============================================================\n", + " return W" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You do not need to submit any code for this part of the exercise.*\n", + "\n", + "Execute the following cell to initialize the weights for the 2 layers in the neural network using the `randInitializeWeights` function." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "print('Initializing Neural Network Parameters ...')\n", + "\n", + "initial_Theta1 = randInitializeWeights(input_layer_size, hidden_layer_size)\n", + "initial_Theta2 = randInitializeWeights(hidden_layer_size, num_labels)\n", + "\n", + "# Unroll parameters\n", + "initial_nn_params = np.concatenate([initial_Theta1.ravel(), initial_Theta2.ravel()], axis=0)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 2.4 Backpropagation\n", + "\n", + "![](Figures/ex4-backpropagation.png)\n", + "\n", + "Now, you will implement the backpropagation algorithm. Recall that the intuition behind the backpropagation algorithm is as follows. Given a training example $(x^{(t)}, y^{(t)})$, we will first run a “forward pass” to compute all the activations throughout the network, including the output value of the hypothesis $h_\\theta(x)$. Then, for each node $j$ in layer $l$, we would like to compute an “error term” $\\delta_j^{(l)}$ that measures how much that node was “responsible” for any errors in our output.\n", + "\n", + "For an output node, we can directly measure the difference between the network’s activation and the true target value, and use that to define $\\delta_j^{(3)}$ (since layer 3 is the output layer). For the hidden units, you will compute $\\delta_j^{(l)}$ based on a weighted average of the error terms of the nodes in layer $(l+1)$. In detail, here is the backpropagation algorithm (also depicted in the figure above). You should implement steps 1 to 4 in a loop that processes one example at a time. Concretely, you should implement a for-loop `for t in range(m)` and place steps 1-4 below inside the for-loop, with the $t^{th}$ iteration performing the calculation on the $t^{th}$ training example $(x^{(t)}, y^{(t)})$. Step 5 will divide the accumulated gradients by $m$ to obtain the gradients for the neural network cost function.\n", + "\n", + "1. Set the input layer’s values $(a^{(1)})$ to the $t^{th }$training example $x^{(t)}$. Perform a feedforward pass, computing the activations $(z^{(2)}, a^{(2)}, z^{(3)}, a^{(3)})$ for layers 2 and 3. Note that you need to add a `+1` term to ensure that the vectors of activations for layers $a^{(1)}$ and $a^{(2)}$ also include the bias unit. In `numpy`, if a 1 is a column matrix, adding one corresponds to `a_1 = np.concatenate([np.ones((m, 1)), a_1], axis=1)`.\n", + "\n", + "1. For each output unit $k$ in layer 3 (the output layer), set \n", + "$$\\delta_k^{(3)} = \\left(a_k^{(3)} - y_k \\right)$$\n", + "where $y_k \\in \\{0, 1\\}$ indicates whether the current training example belongs to class $k$ $(y_k = 1)$, or if it belongs to a different class $(y_k = 0)$. You may find logical arrays helpful for this task (explained in the previous programming exercise).\n", + "\n", + "1. For the hidden layer $l = 2$, set \n", + "$$ \\delta^{(2)} = \\left( \\Theta^{(2)} \\right)^T \\delta^{(3)} * g'\\left(z^{(2)} \\right)$$\n", + "Note that the symbol $*$ performs element wise multiplication in `numpy`.\n", + "\n", + "1. Accumulate the gradient from this example using the following formula. Note that you should skip or remove $\\delta_0^{(2)}$. In `numpy`, removing $\\delta_0^{(2)}$ corresponds to `delta_2 = delta_2[1:]`.\n", + "\n", + "1. Obtain the (unregularized) gradient for the neural network cost function by dividing the accumulated gradients by $\\frac{1}{m}$:\n", + "$$ \\frac{\\partial}{\\partial \\Theta_{ij}^{(l)}} J(\\Theta) = D_{ij}^{(l)} = \\frac{1}{m} \\Delta_{ij}^{(l)}$$\n", + "\n", + "
\n", + "**Python/Numpy tip**: You should implement the backpropagation algorithm only after you have successfully completed the feedforward and cost functions. While implementing the backpropagation alogrithm, it is often useful to use the `shape` function to print out the shapes of the variables you are working with if you run into dimension mismatch errors.\n", + "
\n", + "\n", + "[Click here to go back and update the function `nnCostFunction` with the backpropagation algorithm](#nnCostFunction)." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "After you have implemented the backpropagation algorithm, we will proceed to run gradient checking on your implementation. The gradient check will allow you to increase your confidence that your code is\n", + "computing the gradients correctly.\n", + "\n", + "### 2.4 Gradient checking \n", + "\n", + "In your neural network, you are minimizing the cost function $J(\\Theta)$. To perform gradient checking on your parameters, you can imagine “unrolling” the parameters $\\Theta^{(1)}$, $\\Theta^{(2)}$ into a long vector $\\theta$. By doing so, you can think of the cost function being $J(\\Theta)$ instead and use the following gradient checking procedure.\n", + "\n", + "Suppose you have a function $f_i(\\theta)$ that purportedly computes $\\frac{\\partial}{\\partial \\theta_i} J(\\theta)$; you’d like to check if $f_i$ is outputting correct derivative values.\n", + "\n", + "$$\n", + "\\text{Let } \\theta^{(i+)} = \\theta + \\begin{bmatrix} 0 \\\\ 0 \\\\ \\vdots \\\\ \\epsilon \\\\ \\vdots \\\\ 0 \\end{bmatrix}\n", + "\\quad \\text{and} \\quad \\theta^{(i-)} = \\theta - \\begin{bmatrix} 0 \\\\ 0 \\\\ \\vdots \\\\ \\epsilon \\\\ \\vdots \\\\ 0 \\end{bmatrix}\n", + "$$\n", + "\n", + "So, $\\theta^{(i+)}$ is the same as $\\theta$, except its $i^{th}$ element has been incremented by $\\epsilon$. Similarly, $\\theta^{(i−)}$ is the corresponding vector with the $i^{th}$ element decreased by $\\epsilon$. You can now numerically verify $f_i(\\theta)$’s correctness by checking, for each $i$, that:\n", + "\n", + "$$ f_i\\left( \\theta \\right) \\approx \\frac{J\\left( \\theta^{(i+)}\\right) - J\\left( \\theta^{(i-)} \\right)}{2\\epsilon} $$\n", + "\n", + "The degree to which these two values should approximate each other will depend on the details of $J$. But assuming $\\epsilon = 10^{-4}$, you’ll usually find that the left- and right-hand sides of the above will agree to at least 4 significant digits (and often many more).\n", + "\n", + "We have implemented the function to compute the numerical gradient for you in `computeNumericalGradient` (within the file `utils.py`). While you are not required to modify the file, we highly encourage you to take a look at the code to understand how it works.\n", + "\n", + "In the next cell we will run the provided function `checkNNGradients` which will create a small neural network and dataset that will be used for checking your gradients. If your backpropagation implementation is correct,\n", + "you should see a relative difference that is less than 1e-9.\n", + "\n", + "
\n", + "**Practical Tip**: When performing gradient checking, it is much more efficient to use a small neural network with a relatively small number of input units and hidden units, thus having a relatively small number\n", + "of parameters. Each dimension of $\\theta$ requires two evaluations of the cost function and this can be expensive. In the function `checkNNGradients`, our code creates a small random model and dataset which is used with `computeNumericalGradient` for gradient checking. Furthermore, after you are confident that your gradient computations are correct, you should turn off gradient checking before running your learning algorithm.\n", + "
\n", + "\n", + "
\n", + "**Practical Tip:** Gradient checking works for any function where you are computing the cost and the gradient. Concretely, you can use the same `computeNumericalGradient` function to check if your gradient implementations for the other exercises are correct too (e.g., logistic regression’s cost function).\n", + "
" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "utils.checkNNGradients(nnCostFunction)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*Once your cost function passes the gradient check for the (unregularized) neural network cost function, you should submit the neural network gradient function (backpropagation).*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[4] = nnCostFunction\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 2.5 Regularized Neural Network\n", + "\n", + "After you have successfully implemented the backpropagation algorithm, you will add regularization to the gradient. To account for regularization, it turns out that you can add this as an additional term *after* computing the gradients using backpropagation.\n", + "\n", + "Specifically, after you have computed $\\Delta_{ij}^{(l)}$ using backpropagation, you should add regularization using\n", + "\n", + "$$ \\begin{align} \n", + "& \\frac{\\partial}{\\partial \\Theta_{ij}^{(l)}} J(\\Theta) = D_{ij}^{(l)} = \\frac{1}{m} \\Delta_{ij}^{(l)} & \\qquad \\text{for } j = 0 \\\\\n", + "& \\frac{\\partial}{\\partial \\Theta_{ij}^{(l)}} J(\\Theta) = D_{ij}^{(l)} = \\frac{1}{m} \\Delta_{ij}^{(l)} + \\frac{\\lambda}{m} \\Theta_{ij}^{(l)} & \\qquad \\text{for } j \\ge 1\n", + "\\end{align}\n", + "$$\n", + "\n", + "Note that you should *not* be regularizing the first column of $\\Theta^{(l)}$ which is used for the bias term. Furthermore, in the parameters $\\Theta_{ij}^{(l)}$, $i$ is indexed starting from 1, and $j$ is indexed starting from 0. Thus, \n", + "\n", + "$$\n", + "\\Theta^{(l)} = \\begin{bmatrix}\n", + "\\Theta_{1,0}^{(i)} & \\Theta_{1,1}^{(l)} & \\cdots \\\\\n", + "\\Theta_{2,0}^{(i)} & \\Theta_{2,1}^{(l)} & \\cdots \\\\\n", + "\\vdots & ~ & \\ddots\n", + "\\end{bmatrix}\n", + "$$\n", + "\n", + "[Now modify your code that computes grad in `nnCostFunction` to account for regularization.](#nnCostFunction)\n", + "\n", + "After you are done, the following cell runs gradient checking on your implementation. If your code is correct, you should expect to see a relative difference that is less than 1e-9." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Check gradients by running checkNNGradients\n", + "lambda_ = 3\n", + "utils.checkNNGradients(nnCostFunction, lambda_)\n", + "\n", + "# Also output the costFunction debugging values\n", + "debug_J, _ = nnCostFunction(nn_params, input_layer_size,\n", + " hidden_layer_size, num_labels, X, y, lambda_)\n", + "\n", + "print('\\n\\nCost at (fixed) debugging parameters (w/ lambda = %f): %f ' % (lambda_, debug_J))\n", + "print('(for lambda = 3, this value should be about 0.576051)')" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[5] = nnCostFunction\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2.6 Learning parameters using `scipy.optimize.minimize`\n", + "\n", + "After you have successfully implemented the neural network cost function\n", + "and gradient computation, the next step we will use `scipy`'s minimization to learn a good set parameters." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# After you have completed the assignment, change the maxiter to a larger\n", + "# value to see how more training helps.\n", + "options= {'maxiter': 100}\n", + "\n", + "# You should also try different values of lambda\n", + "lambda_ = 1\n", + "\n", + "# Create \"short hand\" for the cost function to be minimized\n", + "costFunction = lambda p: nnCostFunction(p, input_layer_size,\n", + " hidden_layer_size,\n", + " num_labels, X, y, lambda_)\n", + "\n", + "# Now, costFunction is a function that takes in only one argument\n", + "# (the neural network parameters)\n", + "res = optimize.minimize(costFunction,\n", + " initial_nn_params,\n", + " jac=True,\n", + " method='TNC',\n", + " options=options)\n", + "\n", + "# get the solution of the optimization\n", + "nn_params = res.x\n", + " \n", + "# Obtain Theta1 and Theta2 back from nn_params\n", + "Theta1 = np.reshape(nn_params[:hidden_layer_size * (input_layer_size + 1)],\n", + " (hidden_layer_size, (input_layer_size + 1)))\n", + "\n", + "Theta2 = np.reshape(nn_params[(hidden_layer_size * (input_layer_size + 1)):],\n", + " (num_labels, (hidden_layer_size + 1)))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "After the training completes, we will proceed to report the training accuracy of your classifier by computing the percentage of examples it got correct. If your implementation is correct, you should see a reported\n", + "training accuracy of about 95.3% (this may vary by about 1% due to the random initialization). It is possible to get higher training accuracies by training the neural network for more iterations. We encourage you to try\n", + "training the neural network for more iterations (e.g., set `maxiter` to 400) and also vary the regularization parameter $\\lambda$. With the right learning settings, it is possible to get the neural network to perfectly fit the training set." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "pred = utils.predict(Theta1, Theta2, X)\n", + "print('Training Set Accuracy: %f' % (np.mean(pred == y) * 100))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 3 Visualizing the Hidden Layer\n", + "\n", + "One way to understand what your neural network is learning is to visualize what the representations captured by the hidden units. Informally, given a particular hidden unit, one way to visualize what it computes is to find an input $x$ that will cause it to activate (that is, to have an activation value \n", + "($a_i^{(l)}$) close to 1). For the neural network you trained, notice that the $i^{th}$ row of $\\Theta^{(1)}$ is a 401-dimensional vector that represents the parameter for the $i^{th}$ hidden unit. If we discard the bias term, we get a 400 dimensional vector that represents the weights from each input pixel to the hidden unit.\n", + "\n", + "Thus, one way to visualize the “representation” captured by the hidden unit is to reshape this 400 dimensional vector into a 20 × 20 image and display it (It turns out that this is equivalent to finding the input that gives the highest activation for the hidden unit, given a “norm” constraint on the input (i.e., $||x||_2 \\le 1$)). \n", + "\n", + "The next cell does this by using the `displayData` function and it will show you an image with 25 units,\n", + "each corresponding to one hidden unit in the network. In your trained network, you should find that the hidden units corresponds roughly to detectors that look for strokes and other patterns in the input." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "utils.displayData(Theta1[:, 1:])" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 3.1 Optional (ungraded) exercise\n", + "\n", + "In this part of the exercise, you will get to try out different learning settings for the neural network to see how the performance of the neural network varies with the regularization parameter $\\lambda$ and number of training steps (the `maxiter` option when using `scipy.optimize.minimize`). Neural networks are very powerful models that can form highly complex decision boundaries. Without regularization, it is possible for a neural network to “overfit” a training set so that it obtains close to 100% accuracy on the training set but does not as well on new examples that it has not seen before. You can set the regularization $\\lambda$ to a smaller value and the `maxiter` parameter to a higher number of iterations to see this for youself." + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.6.4" + } + }, + "nbformat": 4, + "nbformat_minor": 2 +} diff --git a/Exercise4/utils.py b/Exercise4/utils.py new file mode 100755 index 0000000..6b7c3bd --- /dev/null +++ b/Exercise4/utils.py @@ -0,0 +1,226 @@ +import sys +import numpy as np +from matplotlib import pyplot + +sys.path.append('..') +from submission import SubmissionBase + + +def displayData(X, example_width=None, figsize=(10, 10)): + """ + Displays 2D data stored in X in a nice grid. + """ + # Compute rows, cols + if X.ndim == 2: + m, n = X.shape + elif X.ndim == 1: + n = X.size + m = 1 + X = X[None] # Promote to a 2 dimensional array + else: + raise IndexError('Input X should be 1 or 2 dimensional.') + + example_width = example_width or int(np.round(np.sqrt(n))) + example_height = n / example_width + + # Compute number of items to display + display_rows = int(np.floor(np.sqrt(m))) + display_cols = int(np.ceil(m / display_rows)) + + fig, ax_array = pyplot.subplots(display_rows, display_cols, figsize=figsize) + fig.subplots_adjust(wspace=0.025, hspace=0.025) + + ax_array = [ax_array] if m == 1 else ax_array.ravel() + + for i, ax in enumerate(ax_array): + # Display Image + h = ax.imshow(X[i].reshape(example_width, example_width, order='F'), + cmap='Greys', extent=[0, 1, 0, 1]) + ax.axis('off') + + +def predict(Theta1, Theta2, X): + """ + Predict the label of an input given a trained neural network + Outputs the predicted label of X given the trained weights of a neural + network(Theta1, Theta2) + """ + # Useful values + m = X.shape[0] + num_labels = Theta2.shape[0] + + # You need to return the following variables correctly + p = np.zeros(m) + h1 = sigmoid(np.dot(np.concatenate([np.ones((m, 1)), X], axis=1), Theta1.T)) + h2 = sigmoid(np.dot(np.concatenate([np.ones((m, 1)), h1], axis=1), Theta2.T)) + p = np.argmax(h2, axis=1) + return p + + +def debugInitializeWeights(fan_out, fan_in): + """ + Initialize the weights of a layer with fan_in incoming connections and fan_out outgoings + connections using a fixed strategy. This will help you later in debugging. + + Note that W should be set a matrix of size (1+fan_in, fan_out) as the first row of W handles + the "bias" terms. + + Parameters + ---------- + fan_out : int + The number of outgoing connections. + + fan_in : int + The number of incoming connections. + + Returns + ------- + W : array_like (1+fan_in, fan_out) + The initialized weights array given the dimensions. + """ + # Initialize W using "sin". This ensures that W is always of the same values and will be + # useful for debugging + W = np.sin(np.arange(1, 1 + (1+fan_in)*fan_out))/10.0 + W = W.reshape(fan_out, 1+fan_in, order='F') + return W + + +def computeNumericalGradient(J, theta, e=1e-4): + """ + Computes the gradient using "finite differences" and gives us a numerical estimate of the + gradient. + + Parameters + ---------- + J : func + The cost function which will be used to estimate its numerical gradient. + + theta : array_like + The one dimensional unrolled network parameters. The numerical gradient is computed at + those given parameters. + + e : float (optional) + The value to use for epsilon for computing the finite difference. + + Notes + ----- + The following code implements numerical gradient checking, and + returns the numerical gradient. It sets `numgrad[i]` to (a numerical + approximation of) the partial derivative of J with respect to the + i-th input argument, evaluated at theta. (i.e., `numgrad[i]` should + be the (approximately) the partial derivative of J with respect + to theta[i].) + """ + numgrad = np.zeros(theta.shape) + perturb = np.diag(e * np.ones(theta.shape)) + for i in range(theta.size): + loss1, _ = J(theta - perturb[:, i]) + loss2, _ = J(theta + perturb[:, i]) + numgrad[i] = (loss2 - loss1)/(2*e) + return numgrad + + +def checkNNGradients(nnCostFunction, lambda_=0): + """ + Creates a small neural network to check the backpropagation gradients. It will output the + analytical gradients produced by your backprop code and the numerical gradients + (computed using computeNumericalGradient). These two gradient computations should result in + very similar values. + + Parameters + ---------- + nnCostFunction : func + A reference to the cost function implemented by the student. + + lambda_ : float (optional) + The regularization parameter value. + """ + input_layer_size = 3 + hidden_layer_size = 5 + num_labels = 3 + m = 5 + + # We generate some 'random' test data + Theta1 = debugInitializeWeights(hidden_layer_size, input_layer_size) + Theta2 = debugInitializeWeights(num_labels, hidden_layer_size) + + # Reusing debugInitializeWeights to generate X + X = debugInitializeWeights(m, input_layer_size - 1) + y = np.arange(1, 1+m) % num_labels + # print(y) + # Unroll parameters + nn_params = np.concatenate([Theta1.ravel(), Theta2.ravel()]) + + # short hand for cost function + costFunc = lambda p: nnCostFunction(p, input_layer_size, hidden_layer_size, + num_labels, X, y, lambda_) + cost, grad = costFunc(nn_params) + numgrad = computeNumericalGradient(costFunc, nn_params) + + # Visually examine the two gradient computations.The two columns you get should be very similar. + print(np.stack([numgrad, grad], axis=1)) + print('The above two columns you get should be very similar.') + print('(Left-Your Numerical Gradient, Right-Analytical Gradient)\n') + + # Evaluate the norm of the difference between two the solutions. If you have a correct + # implementation, and assuming you used e = 0.0001 in computeNumericalGradient, then diff + # should be less than 1e-9. + diff = np.linalg.norm(numgrad - grad)/np.linalg.norm(numgrad + grad) + + print('If your backpropagation implementation is correct, then \n' + 'the relative difference will be small (less than 1e-9). \n' + 'Relative Difference: %g' % diff) + + +def sigmoid(z): + """ + Computes the sigmoid of z. + """ + return 1.0 / (1.0 + np.exp(-z)) + + +class Grader(SubmissionBase): + X = np.reshape(3 * np.sin(np.arange(1, 31)), (3, 10), order='F') + Xm = np.reshape(np.sin(np.arange(1, 33)), (16, 2), order='F') / 5 + ym = np.arange(1, 17) % 4 + t1 = np.sin(np.reshape(np.arange(1, 25, 2), (4, 3), order='F')) + t2 = np.cos(np.reshape(np.arange(1, 41, 2), (4, 5), order='F')) + t = np.concatenate([t1.ravel(), t2.ravel()], axis=0) + + def __init__(self): + part_names = ['Feedforward and Cost Function', + 'Regularized Cost Function', + 'Sigmoid Gradient', + 'Neural Network Gradient (Backpropagation)', + 'Regularized Gradient'] + super().__init__('neural-network-learning', part_names) + + def __iter__(self): + for part_id in range(1, 6): + try: + func = self.functions[part_id] + + # Each part has different expected arguments/different function + if part_id == 1: + res = func(self.t, 2, 4, 4, self.Xm, self.ym, 0)[0] + elif part_id == 2: + res = func(self.t, 2, 4, 4, self.Xm, self.ym, 1.5) + elif part_id == 3: + res = func(self.X, ) + elif part_id == 4: + J, grad = func(self.t, 2, 4, 4, self.Xm, self.ym, 0) + grad1 = np.reshape(grad[:12], (4, 3)) + grad2 = np.reshape(grad[12:], (4, 5)) + grad = np.concatenate([grad1.ravel('F'), grad2.ravel('F')]) + res = np.hstack([J, grad]).tolist() + elif part_id == 5: + J, grad = func(self.t, 2, 4, 4, self.Xm, self.ym, 1.5) + grad1 = np.reshape(grad[:12], (4, 3)) + grad2 = np.reshape(grad[12:], (4, 5)) + grad = np.concatenate([grad1.ravel('F'), grad2.ravel('F')]) + res = np.hstack([J, grad]).tolist() + else: + raise KeyError + yield part_id, res + except KeyError: + yield part_id, 0 diff --git a/Exercise5/Data/ex4data1.mat b/Exercise5/Data/ex4data1.mat new file mode 100755 index 0000000..371bd0c Binary files /dev/null and b/Exercise5/Data/ex4data1.mat differ diff --git a/Exercise5/Data/ex4weights.mat b/Exercise5/Data/ex4weights.mat new file mode 100755 index 0000000..ace2a09 Binary files /dev/null and b/Exercise5/Data/ex4weights.mat differ diff --git a/Exercise5/Data/ex5data1.mat b/Exercise5/Data/ex5data1.mat new file mode 100755 index 0000000..5a17abd Binary files /dev/null and b/Exercise5/Data/ex5data1.mat differ diff --git a/Exercise5/Figures/cross_validation.png b/Exercise5/Figures/cross_validation.png new file mode 100644 index 0000000..e6a8f28 Binary files /dev/null and b/Exercise5/Figures/cross_validation.png differ diff --git a/Exercise5/Figures/learning_curve.png b/Exercise5/Figures/learning_curve.png new file mode 100755 index 0000000..c4d3e1f Binary files /dev/null and b/Exercise5/Figures/learning_curve.png differ diff --git a/Exercise5/Figures/learning_curve_random.png b/Exercise5/Figures/learning_curve_random.png new file mode 100644 index 0000000..ee96525 Binary files /dev/null and b/Exercise5/Figures/learning_curve_random.png differ diff --git a/Exercise5/Figures/linear_fit.png b/Exercise5/Figures/linear_fit.png new file mode 100755 index 0000000..826912f Binary files /dev/null and b/Exercise5/Figures/linear_fit.png differ diff --git a/Exercise5/Figures/polynomial_learning_curve.png b/Exercise5/Figures/polynomial_learning_curve.png new file mode 100644 index 0000000..39e4af4 Binary files /dev/null and b/Exercise5/Figures/polynomial_learning_curve.png differ diff --git a/Exercise5/Figures/polynomial_learning_curve_reg_1.png b/Exercise5/Figures/polynomial_learning_curve_reg_1.png new file mode 100644 index 0000000..01b52b0 Binary files /dev/null and b/Exercise5/Figures/polynomial_learning_curve_reg_1.png differ diff --git a/Exercise5/Figures/polynomial_regression.png b/Exercise5/Figures/polynomial_regression.png new file mode 100644 index 0000000..530ae53 Binary files /dev/null and b/Exercise5/Figures/polynomial_regression.png differ diff --git a/Exercise5/Figures/polynomial_regression_reg_1.png b/Exercise5/Figures/polynomial_regression_reg_1.png new file mode 100644 index 0000000..e27bb13 Binary files /dev/null and b/Exercise5/Figures/polynomial_regression_reg_1.png differ diff --git a/Exercise5/Figures/polynomial_regression_reg_100.png b/Exercise5/Figures/polynomial_regression_reg_100.png new file mode 100644 index 0000000..cb060bc Binary files /dev/null and b/Exercise5/Figures/polynomial_regression_reg_100.png differ diff --git a/Exercise5/exercise5.ipynb b/Exercise5/exercise5.ipynb new file mode 100755 index 0000000..66c4500 --- /dev/null +++ b/Exercise5/exercise5.ipynb @@ -0,0 +1,927 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Programming Exercise 5:\n", + "# Regularized Linear Regression and Bias vs Variance\n", + "\n", + "## Introduction\n", + "\n", + "In this exercise, you will implement regularized linear regression and use it to study models with different bias-variance properties. Before starting on the programming exercise, we strongly recommend watching the video lectures and completing the review questions for the associated topics.\n", + "\n", + "All the information you need for solving this assignment is in this notebook, and all the code you will be implementing will take place within this notebook. The assignment can be promptly submitted to the coursera grader directly from this notebook (code and instructions are included below).\n", + "\n", + "Before we begin with the exercises, we need to import all libraries required for this programming exercise. Throughout the course, we will be using [`numpy`](http://www.numpy.org/) for all arrays and matrix operations, [`matplotlib`](https://matplotlib.org/) for plotting, and [`scipy`](https://docs.scipy.org/doc/scipy/reference/) for scientific and numerical computation functions and tools. You can find instructions on how to install required libraries in the README file in the [github repository](https://github.com/dibgerge/ml-coursera-python-assignments)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": true + }, + "outputs": [], + "source": [ + "# used for manipulating directory paths\n", + "import os\n", + "\n", + "# Scientific and vector computation for python\n", + "import numpy as np\n", + "\n", + "# Plotting library\n", + "from matplotlib import pyplot\n", + "\n", + "# Optimization module in scipy\n", + "from scipy import optimize\n", + "\n", + "# will be used to load MATLAB mat datafile format\n", + "from scipy.io import loadmat\n", + "\n", + "# library written for this exercise providing additional functions for assignment submission, and others\n", + "import utils\n", + "\n", + "# define the submission/grader object for this exercise\n", + "grader = utils.Grader()\n", + "\n", + "# tells matplotlib to embed plots within the notebook\n", + "%matplotlib inline" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Submission and Grading\n", + "\n", + "\n", + "After completing each part of the assignment, be sure to submit your solutions to the grader. The following is a breakdown of how each part of this exercise is scored.\n", + "\n", + "\n", + "| Section | Part | Submitted Function | Points |\n", + "| :- |:- |:- | :-: |\n", + "| 1 | [Regularized Linear Regression Cost Function](#section1) | [`linearRegCostFunction`](#linearRegCostFunction) | 25 |\n", + "| 2 | [Regularized Linear Regression Gradient](#section2) | [`linearRegCostFunction`](#linearRegCostFunction) |25 |\n", + "| 3 | [Learning Curve](#section3) | [`learningCurve`](#func2) | 20 |\n", + "| 4 | [Polynomial Feature Mapping](#section4) | [`polyFeatures`](#polyFeatures) | 10 |\n", + "| 5 | [Cross Validation Curve](#section5) | [`validationCurve`](#validationCurve) | 20 |\n", + "| | Total Points | |100 |\n", + "\n", + "\n", + "You are allowed to submit your solutions multiple times, and we will take only the highest score into consideration.\n", + "\n", + "
\n", + "At the end of each section in this notebook, we have a cell which contains code for submitting the solutions thus far to the grader. Execute the cell to see your score up to the current section. For all your work to be submitted properly, you must execute those cells at least once.\n", + "
" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "## 1 Regularized Linear Regression\n", + "\n", + "In the first half of the exercise, you will implement regularized linear regression to predict the amount of water flowing out of a dam using the change of water level in a reservoir. In the next half, you will go through some diagnostics of debugging learning algorithms and examine the effects of bias v.s.\n", + "variance. \n", + "\n", + "### 1.1 Visualizing the dataset\n", + "\n", + "We will begin by visualizing the dataset containing historical records on the change in the water level, $x$, and the amount of water flowing out of the dam, $y$. This dataset is divided into three parts:\n", + "\n", + "- A **training** set that your model will learn on: `X`, `y`\n", + "- A **cross validation** set for determining the regularization parameter: `Xval`, `yval`\n", + "- A **test** set for evaluating performance. These are “unseen” examples which your model did not see during training: `Xtest`, `ytest`\n", + "\n", + "Run the next cell to plot the training data. In the following parts, you will implement linear regression and use that to fit a straight line to the data and plot learning curves. Following that, you will implement polynomial regression to find a better fit to the data." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Load from ex5data1.mat, where all variables will be store in a dictionary\n", + "data = loadmat(os.path.join('Data', 'ex5data1.mat'))\n", + "\n", + "# Extract train, test, validation data from dictionary\n", + "# and also convert y's form 2-D matrix (MATLAB format) to a numpy vector\n", + "X, y = data['X'], data['y'][:, 0]\n", + "Xtest, ytest = data['Xtest'], data['ytest'][:, 0]\n", + "Xval, yval = data['Xval'], data['yval'][:, 0]\n", + "\n", + "# m = Number of examples\n", + "m = y.size\n", + "\n", + "# Plot training data\n", + "pyplot.plot(X, y, 'ro', ms=10, mec='k', mew=1)\n", + "pyplot.xlabel('Change in water level (x)')\n", + "pyplot.ylabel('Water flowing out of the dam (y)');" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 1.2 Regularized linear regression cost function\n", + "\n", + "Recall that regularized linear regression has the following cost function:\n", + "\n", + "$$ J(\\theta) = \\frac{1}{2m} \\left( \\sum_{i=1}^m \\left( h_\\theta\\left( x^{(i)} \\right) - y^{(i)} \\right)^2 \\right) + \\frac{\\lambda}{2m} \\left( \\sum_{j=1}^n \\theta_j^2 \\right)$$\n", + "\n", + "where $\\lambda$ is a regularization parameter which controls the degree of regularization (thus, help preventing overfitting). The regularization term puts a penalty on the overall cost J. As the magnitudes of the model parameters $\\theta_j$ increase, the penalty increases as well. Note that you should not regularize\n", + "the $\\theta_0$ term.\n", + "\n", + "You should now complete the code in the function `linearRegCostFunction` in the next cell. Your task is to calculate the regularized linear regression cost function. If possible, try to vectorize your code and avoid writing loops.\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": true + }, + "outputs": [], + "source": [ + "def linearRegCostFunction(X, y, theta, lambda_=0.0):\n", + " \"\"\"\n", + " Compute cost and gradient for regularized linear regression \n", + " with multiple variables. Computes the cost of using theta as\n", + " the parameter for linear regression to fit the data points in X and y. \n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like\n", + " The dataset. Matrix with shape (m x n + 1) where m is the \n", + " total number of examples, and n is the number of features \n", + " before adding the bias term.\n", + " \n", + " y : array_like\n", + " The functions values at each datapoint. A vector of\n", + " shape (m, ).\n", + " \n", + " theta : array_like\n", + " The parameters for linear regression. A vector of shape (n+1,).\n", + " \n", + " lambda_ : float, optional\n", + " The regularization parameter.\n", + " \n", + " Returns\n", + " -------\n", + " J : float\n", + " The computed cost function. \n", + " \n", + " grad : array_like\n", + " The value of the cost function gradient w.r.t theta. \n", + " A vector of shape (n+1, ).\n", + " \n", + " Instructions\n", + " ------------\n", + " Compute the cost and gradient of regularized linear regression for\n", + " a particular choice of theta.\n", + " You should set J to the cost and grad to the gradient.\n", + " \"\"\"\n", + " # Initialize some useful values\n", + " m = y.size # number of training examples\n", + "\n", + " # You need to return the following variables correctly \n", + " J = 0\n", + " grad = np.zeros(theta.shape)\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + "\n", + "\n", + " # ============================================================\n", + " return J, grad" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "When you are finished, the next cell will run your cost function using `theta` initialized at `[1, 1]`. You should expect to see an output of 303.993." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "theta = np.array([1, 1])\n", + "J, _ = linearRegCostFunction(np.concatenate([np.ones((m, 1)), X], axis=1), y, theta, 1)\n", + "\n", + "print('Cost at theta = [1, 1]:\\t %f ' % J)\n", + "print('This value should be about 303.993192)\\n' % J)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "After completing a part of the exercise, you can submit your solutions for grading by first adding the function you modified to the submission object, and then sending your function to Coursera for grading. \n", + "\n", + "The submission script will prompt you for your login e-mail and submission token. You can obtain a submission token from the web page for the assignment. You are allowed to submit your solutions multiple times, and we will take only the highest score into consideration.\n", + "\n", + "*Execute the following cell to grade your solution to the first part of this exercise.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[1] = linearRegCostFunction\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 1.3 Regularized linear regression gradient\n", + "\n", + "Correspondingly, the partial derivative of the cost function for regularized linear regression is defined as:\n", + "\n", + "$$\n", + "\\begin{align}\n", + "& \\frac{\\partial J(\\theta)}{\\partial \\theta_0} = \\frac{1}{m} \\sum_{i=1}^m \\left( h_\\theta \\left(x^{(i)} \\right) - y^{(i)} \\right) x_j^{(i)} & \\qquad \\text{for } j = 0 \\\\\n", + "& \\frac{\\partial J(\\theta)}{\\partial \\theta_j} = \\left( \\frac{1}{m} \\sum_{i=1}^m \\left( h_\\theta \\left( x^{(i)} \\right) - y^{(i)} \\right) x_j^{(i)} \\right) + \\frac{\\lambda}{m} \\theta_j & \\qquad \\text{for } j \\ge 1\n", + "\\end{align}\n", + "$$\n", + "\n", + "In the function [`linearRegCostFunction`](#linearRegCostFunction) above, add code to calculate the gradient, returning it in the variable `grad`. Do not forget to re-execute the cell containing this function to update the function's definition.\n", + "\n", + "\n", + "When you are finished, use the next cell to run your gradient function using theta initialized at `[1, 1]`. You should expect to see a gradient of `[-15.30, 598.250]`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "theta = np.array([1, 1])\n", + "J, grad = linearRegCostFunction(np.concatenate([np.ones((m, 1)), X], axis=1), y, theta, 1)\n", + "\n", + "print('Gradient at theta = [1, 1]: [{:.6f}, {:.6f}] '.format(*grad))\n", + "print(' (this value should be about [-15.303016, 598.250744])\\n')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[2] = linearRegCostFunction\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### Fitting linear regression\n", + "\n", + "Once your cost function and gradient are working correctly, the next cell will run the code in `trainLinearReg` (found in the module `utils.py`) to compute the optimal values of $\\theta$. This training function uses `scipy`'s optimization module to minimize the cost function.\n", + "\n", + "In this part, we set regularization parameter $\\lambda$ to zero. Because our current implementation of linear regression is trying to fit a 2-dimensional $\\theta$, regularization will not be incredibly helpful for a $\\theta$ of such low dimension. In the later parts of the exercise, you will be using polynomial regression with regularization.\n", + "\n", + "Finally, the code in the next cell should also plot the best fit line, which should look like the figure below. \n", + "\n", + "![](Figures/linear_fit.png)\n", + "\n", + "The best fit line tells us that the model is not a good fit to the data because the data has a non-linear pattern. While visualizing the best fit as shown is one possible way to debug your learning algorithm, it is not always easy to visualize the data and model. In the next section, you will implement a function to generate learning curves that can help you debug your learning algorithm even if it is not easy to visualize the\n", + "data." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# add a columns of ones for the y-intercept\n", + "X_aug = np.concatenate([np.ones((m, 1)), X], axis=1)\n", + "theta = utils.trainLinearReg(linearRegCostFunction, X_aug, y, lambda_=0)\n", + "\n", + "# Plot fit over the data\n", + "pyplot.plot(X, y, 'ro', ms=10, mec='k', mew=1.5)\n", + "pyplot.xlabel('Change in water level (x)')\n", + "pyplot.ylabel('Water flowing out of the dam (y)')\n", + "pyplot.plot(X, np.dot(X_aug, theta), '--', lw=2);" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "## 2 Bias-variance\n", + "\n", + "An important concept in machine learning is the bias-variance tradeoff. Models with high bias are not complex enough for the data and tend to underfit, while models with high variance overfit to the training data.\n", + "\n", + "In this part of the exercise, you will plot training and test errors on a learning curve to diagnose bias-variance problems.\n", + "\n", + "### 2.1 Learning Curves\n", + "\n", + "You will now implement code to generate the learning curves that will be useful in debugging learning algorithms. Recall that a learning curve plots training and cross validation error as a function of training set size. Your job is to fill in the function `learningCurve` in the next cell, so that it returns a vector of errors for the training set and cross validation set.\n", + "\n", + "To plot the learning curve, we need a training and cross validation set error for different training set sizes. To obtain different training set sizes, you should use different subsets of the original training set `X`. Specifically, for a training set size of $i$, you should use the first $i$ examples (i.e., `X[:i, :]`\n", + "and `y[:i]`).\n", + "\n", + "You can use the `trainLinearReg` function (by calling `utils.trainLinearReg(...)`) to find the $\\theta$ parameters. Note that the `lambda_` is passed as a parameter to the `learningCurve` function.\n", + "After learning the $\\theta$ parameters, you should compute the error on the training and cross validation sets. Recall that the training error for a dataset is defined as\n", + "\n", + "$$ J_{\\text{train}} = \\frac{1}{2m} \\left[ \\sum_{i=1}^m \\left(h_\\theta \\left( x^{(i)} \\right) - y^{(i)} \\right)^2 \\right] $$\n", + "\n", + "In particular, note that the training error does not include the regularization term. One way to compute the training error is to use your existing cost function and set $\\lambda$ to 0 only when using it to compute the training error and cross validation error. When you are computing the training set error, make sure you compute it on the training subset (i.e., `X[:n,:]` and `y[:n]`) instead of the entire training set. However, for the cross validation error, you should compute it over the entire cross validation set. You should store\n", + "the computed errors in the vectors error train and error val.\n", + "\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": true + }, + "outputs": [], + "source": [ + "def learningCurve(X, y, Xval, yval, lambda_=0):\n", + " \"\"\"\n", + " Generates the train and cross validation set errors needed to plot a learning curve\n", + " returns the train and cross validation set errors for a learning curve. \n", + " \n", + " In this function, you will compute the train and test errors for\n", + " dataset sizes from 1 up to m. In practice, when working with larger\n", + " datasets, you might want to do this in larger intervals.\n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like\n", + " The training dataset. Matrix with shape (m x n + 1) where m is the \n", + " total number of examples, and n is the number of features \n", + " before adding the bias term.\n", + " \n", + " y : array_like\n", + " The functions values at each training datapoint. A vector of\n", + " shape (m, ).\n", + " \n", + " Xval : array_like\n", + " The validation dataset. Matrix with shape (m_val x n + 1) where m is the \n", + " total number of examples, and n is the number of features \n", + " before adding the bias term.\n", + " \n", + " yval : array_like\n", + " The functions values at each validation datapoint. A vector of\n", + " shape (m_val, ).\n", + " \n", + " lambda_ : float, optional\n", + " The regularization parameter.\n", + " \n", + " Returns\n", + " -------\n", + " error_train : array_like\n", + " A vector of shape m. error_train[i] contains the training error for\n", + " i examples.\n", + " error_val : array_like\n", + " A vecotr of shape m. error_val[i] contains the validation error for\n", + " i training examples.\n", + " \n", + " Instructions\n", + " ------------\n", + " Fill in this function to return training errors in error_train and the\n", + " cross validation errors in error_val. i.e., error_train[i] and \n", + " error_val[i] should give you the errors obtained after training on i examples.\n", + " \n", + " Notes\n", + " -----\n", + " - You should evaluate the training error on the first i training\n", + " examples (i.e., X[:i, :] and y[:i]).\n", + " \n", + " For the cross-validation error, you should instead evaluate on\n", + " the _entire_ cross validation set (Xval and yval).\n", + " \n", + " - If you are using your cost function (linearRegCostFunction) to compute\n", + " the training and cross validation error, you should call the function with\n", + " the lambda argument set to 0. Do note that you will still need to use\n", + " lambda when running the training to obtain the theta parameters.\n", + " \n", + " Hint\n", + " ----\n", + " You can loop over the examples with the following:\n", + " \n", + " for i in range(1, m+1):\n", + " # Compute train/cross validation errors using training examples \n", + " # X[:i, :] and y[:i], storing the result in \n", + " # error_train[i-1] and error_val[i-1]\n", + " .... \n", + " \"\"\"\n", + " # Number of training examples\n", + " m = y.size\n", + "\n", + " # You need to return these values correctly\n", + " error_train = np.zeros(m)\n", + " error_val = np.zeros(m)\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + " \n", + "\n", + " \n", + " # =============================================================\n", + " return error_train, error_val" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "When you are finished implementing the function `learningCurve`, executing the next cell prints the learning curves and produce a plot similar to the figure below. \n", + "\n", + "![](Figures/learning_curve.png)\n", + "\n", + "In the learning curve figure, you can observe that both the train error and cross validation error are high when the number of training examples is increased. This reflects a high bias problem in the model - the linear regression model is too simple and is unable to fit our dataset well. In the next section, you will implement polynomial regression to fit a better model for this dataset." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "X_aug = np.concatenate([np.ones((m, 1)), X], axis=1)\n", + "Xval_aug = np.concatenate([np.ones((yval.size, 1)), Xval], axis=1)\n", + "error_train, error_val = learningCurve(X_aug, y, Xval_aug, yval, lambda_=0)\n", + "\n", + "pyplot.plot(np.arange(1, m+1), error_train, np.arange(1, m+1), error_val, lw=2)\n", + "pyplot.title('Learning curve for linear regression')\n", + "pyplot.legend(['Train', 'Cross Validation'])\n", + "pyplot.xlabel('Number of training examples')\n", + "pyplot.ylabel('Error')\n", + "pyplot.axis([0, 13, 0, 150])\n", + "\n", + "print('# Training Examples\\tTrain Error\\tCross Validation Error')\n", + "for i in range(m):\n", + " print(' \\t%d\\t\\t%f\\t%f' % (i+1, error_train[i], error_val[i]))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[3] = learningCurve\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "\n", + "## 3 Polynomial regression\n", + "\n", + "The problem with our linear model was that it was too simple for the data\n", + "and resulted in underfitting (high bias). In this part of the exercise, you will address this problem by adding more features. For polynomial regression, our hypothesis has the form:\n", + "\n", + "$$\n", + "\\begin{align}\n", + "h_\\theta(x) &= \\theta_0 + \\theta_1 \\times (\\text{waterLevel}) + \\theta_2 \\times (\\text{waterLevel})^2 + \\cdots + \\theta_p \\times (\\text{waterLevel})^p \\\\\n", + "& = \\theta_0 + \\theta_1 x_1 + \\theta_2 x_2 + \\cdots + \\theta_p x_p\n", + "\\end{align}\n", + "$$\n", + "\n", + "Notice that by defining $x_1 = (\\text{waterLevel})$, $x_2 = (\\text{waterLevel})^2$ , $\\cdots$, $x_p =\n", + "(\\text{waterLevel})^p$, we obtain a linear regression model where the features are the various powers of the original value (waterLevel).\n", + "\n", + "Now, you will add more features using the higher powers of the existing feature $x$ in the dataset. Your task in this part is to complete the code in the function `polyFeatures` in the next cell. The function should map the original training set $X$ of size $m \\times 1$ into its higher powers. Specifically, when a training set $X$ of size $m \\times 1$ is passed into the function, the function should return a $m \\times p$ matrix `X_poly`, where column 1 holds the original values of X, column 2 holds the values of $X^2$, column 3 holds the values of $X^3$, and so on. Note that you don’t have to account for the zero-eth power in this function.\n", + "\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": true + }, + "outputs": [], + "source": [ + "def polyFeatures(X, p):\n", + " \"\"\"\n", + " Maps X (1D vector) into the p-th power.\n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like\n", + " A data vector of size m, where m is the number of examples.\n", + " \n", + " p : int\n", + " The polynomial power to map the features. \n", + " \n", + " Returns \n", + " -------\n", + " X_poly : array_like\n", + " A matrix of shape (m x p) where p is the polynomial \n", + " power and m is the number of examples. That is:\n", + " \n", + " X_poly[i, :] = [X[i], X[i]**2, X[i]**3 ... X[i]**p]\n", + " \n", + " Instructions\n", + " ------------\n", + " Given a vector X, return a matrix X_poly where the p-th column of\n", + " X contains the values of X to the p-th power.\n", + " \"\"\"\n", + " # You need to return the following variables correctly.\n", + " X_poly = np.zeros((X.shape[0], p))\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + "\n", + "\n", + " # ============================================================\n", + " return X_poly" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Now you have a function that will map features to a higher dimension. The next cell will apply it to the training set, the test set, and the cross validation set." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "p = 8\n", + "\n", + "# Map X onto Polynomial Features and Normalize\n", + "X_poly = polyFeatures(X, p)\n", + "X_poly, mu, sigma = utils.featureNormalize(X_poly)\n", + "X_poly = np.concatenate([np.ones((m, 1)), X_poly], axis=1)\n", + "\n", + "# Map X_poly_test and normalize (using mu and sigma)\n", + "X_poly_test = polyFeatures(Xtest, p)\n", + "X_poly_test -= mu\n", + "X_poly_test /= sigma\n", + "X_poly_test = np.concatenate([np.ones((ytest.size, 1)), X_poly_test], axis=1)\n", + "\n", + "# Map X_poly_val and normalize (using mu and sigma)\n", + "X_poly_val = polyFeatures(Xval, p)\n", + "X_poly_val -= mu\n", + "X_poly_val /= sigma\n", + "X_poly_val = np.concatenate([np.ones((yval.size, 1)), X_poly_val], axis=1)\n", + "\n", + "print('Normalized Training Example 1:')\n", + "X_poly[0, :]" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[4] = polyFeatures\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 3.1 Learning Polynomial Regression\n", + "\n", + "After you have completed the function `polyFeatures`, we will proceed to train polynomial regression using your linear regression cost function.\n", + "\n", + "Keep in mind that even though we have polynomial terms in our feature vector, we are still solving a linear regression optimization problem. The polynomial terms have simply turned into features that we can use for linear regression. We are using the same cost function and gradient that you wrote for the earlier part of this exercise.\n", + "\n", + "For this part of the exercise, you will be using a polynomial of degree 8. It turns out that if we run the training directly on the projected data, will not work well as the features would be badly scaled (e.g., an example with $x = 40$ will now have a feature $x_8 = 40^8 = 6.5 \\times 10^{12}$). Therefore, you will\n", + "need to use feature normalization.\n", + "\n", + "Before learning the parameters $\\theta$ for the polynomial regression, we first call `featureNormalize` and normalize the features of the training set, storing the mu, sigma parameters separately. We have already implemented this function for you (in `utils.py` module) and it is the same function from the first exercise.\n", + "\n", + "After learning the parameters $\\theta$, you should see two plots generated for polynomial regression with $\\lambda = 0$, which should be similar to the ones here:\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + "
\n", + "\n", + "You should see that the polynomial fit is able to follow the datapoints very well, thus, obtaining a low training error. The figure on the right shows that the training error essentially stays zero for all numbers of training samples. However, the polynomial fit is very complex and even drops off at the extremes. This is an indicator that the polynomial regression model is overfitting the training data and will not generalize well.\n", + "\n", + "To better understand the problems with the unregularized ($\\lambda = 0$) model, you can see that the learning curve shows the same effect where the training error is low, but the cross validation error is high. There is a gap between the training and cross validation errors, indicating a high variance problem." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "lambda_ = 100\n", + "theta = utils.trainLinearReg(linearRegCostFunction, X_poly, y,\n", + " lambda_=lambda_, maxiter=55)\n", + "\n", + "# Plot training data and fit\n", + "pyplot.plot(X, y, 'ro', ms=10, mew=1.5, mec='k')\n", + "\n", + "utils.plotFit(polyFeatures, np.min(X), np.max(X), mu, sigma, theta, p)\n", + "\n", + "pyplot.xlabel('Change in water level (x)')\n", + "pyplot.ylabel('Water flowing out of the dam (y)')\n", + "pyplot.title('Polynomial Regression Fit (lambda = %f)' % lambda_)\n", + "pyplot.ylim([-20, 50])\n", + "\n", + "pyplot.figure()\n", + "error_train, error_val = learningCurve(X_poly, y, X_poly_val, yval, lambda_)\n", + "pyplot.plot(np.arange(1, 1+m), error_train, np.arange(1, 1+m), error_val)\n", + "\n", + "pyplot.title('Polynomial Regression Learning Curve (lambda = %f)' % lambda_)\n", + "pyplot.xlabel('Number of training examples')\n", + "pyplot.ylabel('Error')\n", + "pyplot.axis([0, 13, 0, 100])\n", + "pyplot.legend(['Train', 'Cross Validation'])\n", + "\n", + "print('Polynomial Regression (lambda = %f)\\n' % lambda_)\n", + "print('# Training Examples\\tTrain Error\\tCross Validation Error')\n", + "for i in range(m):\n", + " print(' \\t%d\\t\\t%f\\t%f' % (i+1, error_train[i], error_val[i]))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "One way to combat the overfitting (high-variance) problem is to add regularization to the model. In the next section, you will get to try different $\\lambda$ parameters to see how regularization can lead to a better model.\n", + "\n", + "### 3.2 Optional (ungraded) exercise: Adjusting the regularization parameter\n", + "\n", + "In this section, you will get to observe how the regularization parameter affects the bias-variance of regularized polynomial regression. You should now modify the the lambda parameter and try $\\lambda = 1, 100$. For each of these values, the script should generate a polynomial fit to the data and also a learning curve.\n", + "\n", + "For $\\lambda = 1$, the generated plots should look like the the figure below. You should see a polynomial fit that follows the data trend well (left) and a learning curve (right) showing that both the cross validation and training error converge to a relatively low value. This shows the $\\lambda = 1$ regularized polynomial regression model does not have the high-bias or high-variance problems. In effect, it achieves a good trade-off between bias and variance.\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + "
\n", + "\n", + "For $\\lambda = 100$, you should see a polynomial fit (figure below) that does not follow the data well. In this case, there is too much regularization and the model is unable to fit the training data.\n", + "\n", + "![](Figures/polynomial_regression_reg_100.png)\n", + "\n", + "*You do not need to submit any solutions for this optional (ungraded) exercise.*" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 3.3 Selecting $\\lambda$ using a cross validation set\n", + "\n", + "From the previous parts of the exercise, you observed that the value of $\\lambda$ can significantly affect the results of regularized polynomial regression on the training and cross validation set. In particular, a model without regularization ($\\lambda = 0$) fits the training set well, but does not generalize. Conversely, a model with too much regularization ($\\lambda = 100$) does not fit the training set and testing set well. A good choice of $\\lambda$ (e.g., $\\lambda = 1$) can provide a good fit to the data.\n", + "\n", + "In this section, you will implement an automated method to select the $\\lambda$ parameter. Concretely, you will use a cross validation set to evaluate how good each $\\lambda$ value is. After selecting the best $\\lambda$ value using the cross validation set, we can then evaluate the model on the test set to estimate\n", + "how well the model will perform on actual unseen data. \n", + "\n", + "Your task is to complete the code in the function `validationCurve`. Specifically, you should should use the `utils.trainLinearReg` function to train the model using different values of $\\lambda$ and compute the training error and cross validation error. You should try $\\lambda$ in the following range: {0, 0.001, 0.003, 0.01, 0.03, 0.1, 0.3, 1, 3, 10}.\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": true + }, + "outputs": [], + "source": [ + "def validationCurve(X, y, Xval, yval):\n", + " \"\"\"\n", + " Generate the train and validation errors needed to plot a validation\n", + " curve that we can use to select lambda_.\n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like\n", + " The training dataset. Matrix with shape (m x n) where m is the \n", + " total number of training examples, and n is the number of features \n", + " including any polynomial features.\n", + " \n", + " y : array_like\n", + " The functions values at each training datapoint. A vector of\n", + " shape (m, ).\n", + " \n", + " Xval : array_like\n", + " The validation dataset. Matrix with shape (m_val x n) where m is the \n", + " total number of validation examples, and n is the number of features \n", + " including any polynomial features.\n", + " \n", + " yval : array_like\n", + " The functions values at each validation datapoint. A vector of\n", + " shape (m_val, ).\n", + " \n", + " Returns\n", + " -------\n", + " lambda_vec : list\n", + " The values of the regularization parameters which were used in \n", + " cross validation.\n", + " \n", + " error_train : list\n", + " The training error computed at each value for the regularization\n", + " parameter.\n", + " \n", + " error_val : list\n", + " The validation error computed at each value for the regularization\n", + " parameter.\n", + " \n", + " Instructions\n", + " ------------\n", + " Fill in this function to return training errors in `error_train` and\n", + " the validation errors in `error_val`. The vector `lambda_vec` contains\n", + " the different lambda parameters to use for each calculation of the\n", + " errors, i.e, `error_train[i]`, and `error_val[i]` should give you the\n", + " errors obtained after training with `lambda_ = lambda_vec[i]`.\n", + "\n", + " Note\n", + " ----\n", + " You can loop over lambda_vec with the following:\n", + " \n", + " for i in range(len(lambda_vec))\n", + " lambda = lambda_vec[i]\n", + " # Compute train / val errors when training linear \n", + " # regression with regularization parameter lambda_\n", + " # You should store the result in error_train[i]\n", + " # and error_val[i]\n", + " ....\n", + " \"\"\"\n", + " # Selected values of lambda (you should not change this)\n", + " lambda_vec = [0, 0.001, 0.003, 0.01, 0.03, 0.1, 0.3, 1, 3, 10]\n", + "\n", + " # You need to return these variables correctly.\n", + " error_train = np.zeros(len(lambda_vec))\n", + " error_val = np.zeros(len(lambda_vec))\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + "\n", + "\n", + " # ============================================================\n", + " return lambda_vec, error_train, error_val" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "After you have completed the code, the next cell will run your function and plot a cross validation curve of error v.s. $\\lambda$ that allows you select which $\\lambda$ parameter to use. You should see a plot similar to the figure below. \n", + "\n", + "![](Figures/cross_validation.png)\n", + "\n", + "In this figure, we can see that the best value of $\\lambda$ is around 3. Due to randomness\n", + "in the training and validation splits of the dataset, the cross validation error can sometimes be lower than the training error." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "lambda_vec, error_train, error_val = validationCurve(X_poly, y, X_poly_val, yval)\n", + "\n", + "pyplot.plot(lambda_vec, error_train, '-o', lambda_vec, error_val, '-o', lw=2)\n", + "pyplot.legend(['Train', 'Cross Validation'])\n", + "pyplot.xlabel('lambda')\n", + "pyplot.ylabel('Error')\n", + "\n", + "print('lambda\\t\\tTrain Error\\tValidation Error')\n", + "for i in range(len(lambda_vec)):\n", + " print(' %f\\t%f\\t%f' % (lambda_vec[i], error_train[i], error_val[i]))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[5] = validationCurve\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 3.4 Optional (ungraded) exercise: Computing test set error\n", + "\n", + "In the previous part of the exercise, you implemented code to compute the cross validation error for various values of the regularization parameter $\\lambda$. However, to get a better indication of the model’s performance in the real world, it is important to evaluate the “final” model on a test set that was not used in any part of training (that is, it was neither used to select the $\\lambda$ parameters, nor to learn the model parameters $\\theta$). For this optional (ungraded) exercise, you should compute the test error using the best value of $\\lambda$ you found. In our cross validation, we obtained a test error of 3.8599 for $\\lambda = 3$.\n", + "\n", + "*You do not need to submit any solutions for this optional (ungraded) exercise.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 3.5 Optional (ungraded) exercise: Plotting learning curves with randomly selected examples\n", + "\n", + "In practice, especially for small training sets, when you plot learning curves to debug your algorithms, it is often helpful to average across multiple sets of randomly selected examples to determine the training error and cross validation error.\n", + "\n", + "Concretely, to determine the training error and cross validation error for $i$ examples, you should first randomly select $i$ examples from the training set and $i$ examples from the cross validation set. You will then learn the parameters $\\theta$ using the randomly chosen training set and evaluate the parameters $\\theta$ on the randomly chosen training set and cross validation set. The above steps should then be repeated multiple times (say 50) and the averaged error should be used to determine the training error and cross validation error for $i$ examples.\n", + "\n", + "For this optional (ungraded) exercise, you should implement the above strategy for computing the learning curves. For reference, the figure below shows the learning curve we obtained for polynomial regression with $\\lambda = 0.01$. Your figure may differ slightly due to the random selection of examples.\n", + "\n", + "![](Figures/learning_curve_random.png)\n", + "\n", + "*You do not need to submit any solutions for this optional (ungraded) exercise.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": true + }, + "outputs": [], + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.6.4" + } + }, + "nbformat": 4, + "nbformat_minor": 2 +} diff --git a/Exercise5/utils.py b/Exercise5/utils.py new file mode 100755 index 0000000..b2340ad --- /dev/null +++ b/Exercise5/utils.py @@ -0,0 +1,164 @@ +import sys +import numpy as np +from scipy import optimize +from matplotlib import pyplot + +sys.path.append('..') +from submission import SubmissionBase + + +def trainLinearReg(linearRegCostFunction, X, y, lambda_=0.0, maxiter=200): + """ + Trains linear regression using scipy's optimize.minimize. + + Parameters + ---------- + X : array_like + The dataset with shape (m x n+1). The bias term is assumed to be concatenated. + + y : array_like + Function values at each datapoint. A vector of shape (m,). + + lambda_ : float, optional + The regularization parameter. + + maxiter : int, optional + Maximum number of iteration for the optimization algorithm. + + Returns + ------- + theta : array_like + The parameters for linear regression. This is a vector of shape (n+1,). + """ + # Initialize Theta + initial_theta = np.zeros(X.shape[1]) + + # Create "short hand" for the cost function to be minimized + costFunction = lambda t: linearRegCostFunction(X, y, t, lambda_) + + # Now, costFunction is a function that takes in only one argument + options = {'maxiter': maxiter} + + # Minimize using scipy + res = optimize.minimize(costFunction, initial_theta, jac=True, method='TNC', options=options) + return res.x + + +def featureNormalize(X): + """ + Normalizes the features in X returns a normalized version of X where the mean value of each + feature is 0 and the standard deviation is 1. This is often a good preprocessing step to do when + working with learning algorithms. + + Parameters + ---------- + X : array_like + An dataset which is a (m x n) matrix, where m is the number of examples, + and n is the number of dimensions for each example. + + Returns + ------- + X_norm : array_like + The normalized input dataset. + + mu : array_like + A vector of size n corresponding to the mean for each dimension across all examples. + + sigma : array_like + A vector of size n corresponding to the standard deviations for each dimension across + all examples. + """ + mu = np.mean(X, axis=0) + X_norm = X - mu + + sigma = np.std(X_norm, axis=0, ddof=1) + X_norm /= sigma + return X_norm, mu, sigma + + +def plotFit(polyFeatures, min_x, max_x, mu, sigma, theta, p): + """ + Plots a learned polynomial regression fit over an existing figure. + Also works with linear regression. + Plots the learned polynomial fit with power p and feature normalization (mu, sigma). + + Parameters + ---------- + polyFeatures : func + A function which generators polynomial features from a single feature. + + min_x : float + The minimum value for the feature. + + max_x : float + The maximum value for the feature. + + mu : float + The mean feature value over the training dataset. + + sigma : float + The feature standard deviation of the training dataset. + + theta : array_like + The parameters for the trained polynomial linear regression. + + p : int + The polynomial order. + """ + # We plot a range slightly bigger than the min and max values to get + # an idea of how the fit will vary outside the range of the data points + x = np.arange(min_x - 15, max_x + 25, 0.05).reshape(-1, 1) + + # Map the X values + X_poly = polyFeatures(x, p) + X_poly -= mu + X_poly /= sigma + + # Add ones + X_poly = np.concatenate([np.ones((x.shape[0], 1)), X_poly], axis=1) + + # Plot + pyplot.plot(x, np.dot(X_poly, theta), '--', lw=2) + + +class Grader(SubmissionBase): + # Random test cases + X = np.vstack([np.ones(10), + np.sin(np.arange(1, 15, 1.5)), + np.cos(np.arange(1, 15, 1.5))]).T + y = np.sin(np.arange(1, 31, 3)) + Xval = np.vstack([np.ones(10), + np.sin(np.arange(0, 14, 1.5)), + np.cos(np.arange(0, 14, 1.5))]).T + yval = np.sin(np.arange(1, 11)) + + def __init__(self): + part_names = ['Regularized Linear Regression Cost Function', + 'Regularized Linear Regression Gradient', + 'Learning Curve', + 'Polynomial Feature Mapping', + 'Validation Curve'] + super().__init__('regularized-linear-regression-and-bias-variance', part_names) + + def __iter__(self): + for part_id in range(1, 6): + try: + func = self.functions[part_id] + # Each part has different expected arguments/different function + if part_id == 1: + res = func(self.X, self.y, np.array([0.1, 0.2, 0.3]), 0.5) + elif part_id == 2: + theta = np.array([0.1, 0.2, 0.3]) + res = func(self.X, self.y, theta, 0.5)[1] + elif part_id == 3: + res = np.hstack(func(self.X, self.y, self.Xval, self.yval, 1)).tolist() + elif part_id == 4: + res = func(self.X[1, :].reshape(-1, 1), 8) + elif part_id == 5: + res = np.hstack(func(self.X, self.y, self.Xval, self.yval)).tolist() + else: + raise KeyError + except KeyError: + yield part_id, 0 + yield part_id, res + diff --git a/Exercise6/Data/emailSample1.txt b/Exercise6/Data/emailSample1.txt new file mode 100755 index 0000000..eac52a3 --- /dev/null +++ b/Exercise6/Data/emailSample1.txt @@ -0,0 +1,10 @@ +> Anyone knows how much it costs to host a web portal ? +> +Well, it depends on how many visitors you're expecting. +This can be anywhere from less than 10 bucks a month to a couple of $100. +You should checkout http://www.rackspace.com/ or perhaps Amazon EC2 +if youre running something big.. + +To unsubscribe yourself from this mailing list, send an email to: +groupname-unsubscribe@egroups.com + diff --git a/Exercise6/Data/emailSample2.txt b/Exercise6/Data/emailSample2.txt new file mode 100755 index 0000000..e47acda --- /dev/null +++ b/Exercise6/Data/emailSample2.txt @@ -0,0 +1,34 @@ +Folks, + +my first time posting - have a bit of Unix experience, but am new to Linux. + + +Just got a new PC at home - Dell box with Windows XP. Added a second hard disk +for Linux. Partitioned the disk and have installed Suse 7.2 from CD, which went +fine except it didn't pick up my monitor. + +I have a Dell branded E151FPp 15" LCD flat panel monitor and a nVidia GeForce4 +Ti4200 video card, both of which are probably too new to feature in Suse's default +set. I downloaded a driver from the nVidia website and installed it using RPM. +Then I ran Sax2 (as was recommended in some postings I found on the net), but +it still doesn't feature my video card in the available list. What next? + +Another problem. I have a Dell branded keyboard and if I hit Caps-Lock twice, +the whole machine crashes (in Linux, not Windows) - even the on/off switch is +inactive, leaving me to reach for the power cable instead. + +If anyone can help me in any way with these probs., I'd be really grateful - +I've searched the 'net but have run out of ideas. + +Or should I be going for a different version of Linux such as RedHat? Opinions +welcome. + +Thanks a lot, +Peter + +-- +Irish Linux Users' Group: ilug@linux.ie +http://www.linux.ie/mailman/listinfo/ilug for (un)subscription information. +List maintainer: listmaster@linux.ie + + diff --git a/Exercise6/Data/ex6data1.mat b/Exercise6/Data/ex6data1.mat new file mode 100755 index 0000000..ae0d2aa Binary files /dev/null and b/Exercise6/Data/ex6data1.mat differ diff --git a/Exercise6/Data/ex6data2.mat b/Exercise6/Data/ex6data2.mat new file mode 100755 index 0000000..c6ad661 Binary files /dev/null and b/Exercise6/Data/ex6data2.mat differ diff --git a/Exercise6/Data/ex6data3.mat b/Exercise6/Data/ex6data3.mat new file mode 100755 index 0000000..a0441ac Binary files /dev/null and b/Exercise6/Data/ex6data3.mat differ diff --git a/Exercise6/Data/spamSample1.txt b/Exercise6/Data/spamSample1.txt new file mode 100755 index 0000000..bab0ca2 --- /dev/null +++ b/Exercise6/Data/spamSample1.txt @@ -0,0 +1,42 @@ +Do You Want To Make $1000 Or More Per Week? + + + +If you are a motivated and qualified individual - I +will personally demonstrate to you a system that will +make you $1,000 per week or more! This is NOT mlm. + + + +Call our 24 hour pre-recorded number to get the +details. + + + +000-456-789 + + + +I need people who want to make serious money. Make +the call and get the facts. + +Invest 2 minutes in yourself now! + + + +000-456-789 + + + +Looking forward to your call and I will introduce you +to people like yourself who +are currently making $10,000 plus per week! + + + +000-456-789 + + + +3484lJGv6-241lEaN9080lRmS6-271WxHo7524qiyT5-438rjUv5615hQcf0-662eiDB9057dMtVl72 + diff --git a/Exercise6/Data/spamSample2.txt b/Exercise6/Data/spamSample2.txt new file mode 100755 index 0000000..f8e8fce --- /dev/null +++ b/Exercise6/Data/spamSample2.txt @@ -0,0 +1,8 @@ +Best Buy Viagra Generic Online + +Viagra 100mg x 60 Pills $125, Free Pills & Reorder Discount, Top Selling 100% Quality & Satisfaction guaranteed! + +We accept VISA, Master & E-Check Payments, 90000+ Satisfied Customers! +http://medphysitcstech.ru + + diff --git a/Exercise6/Data/spamTest.mat b/Exercise6/Data/spamTest.mat new file mode 100755 index 0000000..b7bf953 Binary files /dev/null and b/Exercise6/Data/spamTest.mat differ diff --git a/Exercise6/Data/spamTrain.mat b/Exercise6/Data/spamTrain.mat new file mode 100755 index 0000000..1b9c81f Binary files /dev/null and b/Exercise6/Data/spamTrain.mat differ diff --git a/Exercise6/Data/vocab.txt b/Exercise6/Data/vocab.txt new file mode 100755 index 0000000..27f64a3 --- /dev/null +++ b/Exercise6/Data/vocab.txt @@ -0,0 +1,1899 @@ +1 aa +2 ab +3 abil +4 abl +5 about +6 abov +7 absolut +8 abus +9 ac +10 accept +11 access +12 accord +13 account +14 achiev +15 acquir +16 across +17 act +18 action +19 activ +20 actual +21 ad +22 adam +23 add +24 addit +25 address +26 administr +27 adult +28 advanc +29 advantag +30 advertis +31 advic +32 advis +33 ae +34 af +35 affect +36 affili +37 afford +38 africa +39 after +40 ag +41 again +42 against +43 agenc +44 agent +45 ago +46 agre +47 agreement +48 aid +49 air +50 al +51 alb +52 align +53 all +54 allow +55 almost +56 alon +57 along +58 alreadi +59 alsa +60 also +61 altern +62 although +63 alwai +64 am +65 amaz +66 america +67 american +68 among +69 amount +70 amp +71 an +72 analysi +73 analyst +74 and +75 ani +76 anim +77 announc +78 annual +79 annuiti +80 anoth +81 answer +82 anti +83 anumb +84 anybodi +85 anymor +86 anyon +87 anyth +88 anywai +89 anywher +90 aol +91 ap +92 apolog +93 app +94 appar +95 appear +96 appl +97 appli +98 applic +99 appreci +100 approach +101 approv +102 apt +103 ar +104 archiv +105 area +106 aren +107 argument +108 arial +109 arm +110 around +111 arrai +112 arriv +113 art +114 articl +115 artist +116 as +117 ascii +118 ask +119 asset +120 assist +121 associ +122 assum +123 assur +124 at +125 atol +126 attach +127 attack +128 attempt +129 attent +130 attornei +131 attract +132 audio +133 aug +134 august +135 author +136 auto +137 autom +138 automat +139 avail +140 averag +141 avoid +142 awai +143 awar +144 award +145 ba +146 babi +147 back +148 background +149 backup +150 bad +151 balanc +152 ban +153 bank +154 bar +155 base +156 basenumb +157 basi +158 basic +159 bb +160 bc +161 bd +162 be +163 beat +164 beberg +165 becaus +166 becom +167 been +168 befor +169 begin +170 behalf +171 behavior +172 behind +173 believ +174 below +175 benefit +176 best +177 beta +178 better +179 between +180 bf +181 big +182 bill +183 billion +184 bin +185 binari +186 bit +187 black +188 blank +189 block +190 blog +191 blood +192 blue +193 bnumber +194 board +195 bodi +196 boi +197 bonu +198 book +199 boot +200 border +201 boss +202 boston +203 botan +204 both +205 bottl +206 bottom +207 boundari +208 box +209 brain +210 brand +211 break +212 brian +213 bring +214 broadcast +215 broker +216 browser +217 bug +218 bui +219 build +220 built +221 bulk +222 burn +223 bush +224 busi +225 but +226 button +227 by +228 byte +229 ca +230 cabl +231 cach +232 calcul +233 california +234 call +235 came +236 camera +237 campaign +238 can +239 canada +240 cannot +241 canon +242 capabl +243 capillari +244 capit +245 car +246 card +247 care +248 career +249 carri +250 cartridg +251 case +252 cash +253 cat +254 catch +255 categori +256 caus +257 cb +258 cc +259 cd +260 ce +261 cell +262 cent +263 center +264 central +265 centuri +266 ceo +267 certain +268 certainli +269 cf +270 challeng +271 chanc +272 chang +273 channel +274 char +275 charact +276 charg +277 charset +278 chat +279 cheap +280 check +281 cheer +282 chief +283 children +284 china +285 chip +286 choic +287 choos +288 chri +289 citi +290 citizen +291 civil +292 claim +293 class +294 classifi +295 clean +296 clear +297 clearli +298 click +299 client +300 close +301 clue +302 cnet +303 cnumber +304 co +305 code +306 collect +307 colleg +308 color +309 com +310 combin +311 come +312 comfort +313 command +314 comment +315 commentari +316 commerci +317 commiss +318 commit +319 common +320 commun +321 compani +322 compar +323 comparison +324 compat +325 compet +326 competit +327 compil +328 complet +329 comprehens +330 comput +331 concentr +332 concept +333 concern +334 condit +335 conf +336 confer +337 confid +338 confidenti +339 config +340 configur +341 confirm +342 conflict +343 confus +344 congress +345 connect +346 consid +347 consolid +348 constitut +349 construct +350 consult +351 consum +352 contact +353 contain +354 content +355 continu +356 contract +357 contribut +358 control +359 conveni +360 convers +361 convert +362 cool +363 cooper +364 copi +365 copyright +366 core +367 corpor +368 correct +369 correspond +370 cost +371 could +372 couldn +373 count +374 countri +375 coupl +376 cours +377 court +378 cover +379 coverag +380 crash +381 creat +382 creativ +383 credit +384 critic +385 cross +386 cultur +387 current +388 custom +389 cut +390 cv +391 da +392 dagga +393 dai +394 daili +395 dan +396 danger +397 dark +398 data +399 databas +400 datapow +401 date +402 dave +403 david +404 dc +405 de +406 dead +407 deal +408 dear +409 death +410 debt +411 decad +412 decid +413 decis +414 declar +415 declin +416 decor +417 default +418 defend +419 defens +420 defin +421 definit +422 degre +423 delai +424 delet +425 deliv +426 deliveri +427 dell +428 demand +429 democrat +430 depart +431 depend +432 deposit +433 describ +434 descript +435 deserv +436 design +437 desir +438 desktop +439 despit +440 detail +441 detect +442 determin +443 dev +444 devel +445 develop +446 devic +447 di +448 dial +449 did +450 didn +451 diet +452 differ +453 difficult +454 digit +455 direct +456 directli +457 director +458 directori +459 disabl +460 discount +461 discov +462 discoveri +463 discuss +464 disk +465 displai +466 disposit +467 distanc +468 distribut +469 dn +470 dnumber +471 do +472 doc +473 document +474 doe +475 doer +476 doesn +477 dollar +478 dollarac +479 dollarnumb +480 domain +481 don +482 done +483 dont +484 doubl +485 doubt +486 down +487 download +488 dr +489 draw +490 dream +491 drive +492 driver +493 drop +494 drug +495 due +496 dure +497 dvd +498 dw +499 dynam +500 ea +501 each +502 earli +503 earlier +504 earn +505 earth +506 easi +507 easier +508 easili +509 eat +510 eb +511 ebai +512 ec +513 echo +514 econom +515 economi +516 ed +517 edg +518 edit +519 editor +520 educ +521 eff +522 effect +523 effici +524 effort +525 either +526 el +527 electron +528 elimin +529 els +530 email +531 emailaddr +532 emerg +533 empir +534 employ +535 employe +536 en +537 enabl +538 encod +539 encourag +540 end +541 enemi +542 enenkio +543 energi +544 engin +545 english +546 enhanc +547 enjoi +548 enough +549 ensur +550 enter +551 enterpris +552 entertain +553 entir +554 entri +555 enumb +556 environ +557 equal +558 equip +559 equival +560 error +561 especi +562 essenti +563 establish +564 estat +565 estim +566 et +567 etc +568 euro +569 europ +570 european +571 even +572 event +573 eventu +574 ever +575 everi +576 everyon +577 everyth +578 evid +579 evil +580 exactli +581 exampl +582 excel +583 except +584 exchang +585 excit +586 exclus +587 execut +588 exercis +589 exist +590 exmh +591 expand +592 expect +593 expens +594 experi +595 expert +596 expir +597 explain +598 explor +599 express +600 extend +601 extens +602 extra +603 extract +604 extrem +605 ey +606 fa +607 face +608 fact +609 factor +610 fail +611 fair +612 fall +613 fals +614 famili +615 faq +616 far +617 fast +618 faster +619 fastest +620 fat +621 father +622 favorit +623 fax +624 fb +625 fd +626 featur +627 feder +628 fee +629 feed +630 feedback +631 feel +632 femal +633 few +634 ffffff +635 ffnumber +636 field +637 fight +638 figur +639 file +640 fill +641 film +642 filter +643 final +644 financ +645 financi +646 find +647 fine +648 finish +649 fire +650 firewal +651 firm +652 first +653 fit +654 five +655 fix +656 flag +657 flash +658 flow +659 fnumber +660 focu +661 folder +662 folk +663 follow +664 font +665 food +666 for +667 forc +668 foreign +669 forev +670 forget +671 fork +672 form +673 format +674 former +675 fortun +676 forward +677 found +678 foundat +679 four +680 franc +681 free +682 freedom +683 french +684 freshrpm +685 fri +686 fridai +687 friend +688 from +689 front +690 ftoc +691 ftp +692 full +693 fulli +694 fun +695 function +696 fund +697 further +698 futur +699 ga +700 gain +701 game +702 gari +703 garrigu +704 gave +705 gcc +706 geek +707 gener +708 get +709 gif +710 gift +711 girl +712 give +713 given +714 global +715 gnome +716 gnu +717 gnupg +718 go +719 goal +720 god +721 goe +722 gold +723 gone +724 good +725 googl +726 got +727 govern +728 gpl +729 grand +730 grant +731 graphic +732 great +733 greater +734 ground +735 group +736 grow +737 growth +738 gt +739 guarante +740 guess +741 gui +742 guid +743 ha +744 hack +745 had +746 half +747 ham +748 hand +749 handl +750 happen +751 happi +752 hard +753 hardwar +754 hat +755 hate +756 have +757 haven +758 he +759 head +760 header +761 headlin +762 health +763 hear +764 heard +765 heart +766 heaven +767 hei +768 height +769 held +770 hello +771 help +772 helvetica +773 her +774 herba +775 here +776 hermio +777 hettinga +778 hi +779 high +780 higher +781 highli +782 highlight +783 him +784 histori +785 hit +786 hold +787 home +788 honor +789 hope +790 host +791 hot +792 hour +793 hous +794 how +795 howev +796 hp +797 html +798 http +799 httpaddr +800 huge +801 human +802 hundr +803 ibm +804 id +805 idea +806 ident +807 identifi +808 idnumb +809 ie +810 if +811 ignor +812 ii +813 iii +814 iiiiiiihnumberjnumberhnumberjnumberhnumb +815 illeg +816 im +817 imag +818 imagin +819 immedi +820 impact +821 implement +822 import +823 impress +824 improv +825 in +826 inc +827 includ +828 incom +829 increas +830 incred +831 inde +832 independ +833 index +834 india +835 indian +836 indic +837 individu +838 industri +839 info +840 inform +841 initi +842 inlin +843 innov +844 input +845 insert +846 insid +847 instal +848 instanc +849 instant +850 instead +851 institut +852 instruct +853 insur +854 int +855 integr +856 intel +857 intellig +858 intend +859 interact +860 interest +861 interfac +862 intern +863 internet +864 interview +865 into +866 intro +867 introduc +868 inumb +869 invest +870 investig +871 investor +872 invok +873 involv +874 ip +875 ireland +876 irish +877 is +878 island +879 isn +880 iso +881 isp +882 issu +883 it +884 item +885 itself +886 jabber +887 jame +888 java +889 jim +890 jnumberiiiiiiihepihepihf +891 job +892 joe +893 john +894 join +895 journal +896 judg +897 judgment +898 jul +899 juli +900 jump +901 june +902 just +903 justin +904 keep +905 kei +906 kept +907 kernel +908 kevin +909 keyboard +910 kid +911 kill +912 kind +913 king +914 kingdom +915 knew +916 know +917 knowledg +918 known +919 la +920 lack +921 land +922 languag +923 laptop +924 larg +925 larger +926 largest +927 laser +928 last +929 late +930 later +931 latest +932 launch +933 law +934 lawrenc +935 le +936 lead +937 leader +938 learn +939 least +940 leav +941 left +942 legal +943 lender +944 length +945 less +946 lesson +947 let +948 letter +949 level +950 lib +951 librari +952 licens +953 life +954 lifetim +955 light +956 like +957 limit +958 line +959 link +960 linux +961 list +962 listen +963 littl +964 live +965 ll +966 lo +967 load +968 loan +969 local +970 locat +971 lock +972 lockergnom +973 log +974 long +975 longer +976 look +977 lose +978 loss +979 lost +980 lot +981 love +982 low +983 lower +984 lowest +985 lt +986 ma +987 mac +988 machin +989 made +990 magazin +991 mai +992 mail +993 mailer +994 main +995 maintain +996 major +997 make +998 maker +999 male +1000 man +1001 manag +1002 mani +1003 manual +1004 manufactur +1005 map +1006 march +1007 margin +1008 mark +1009 market +1010 marshal +1011 mass +1012 master +1013 match +1014 materi +1015 matter +1016 matthia +1017 mayb +1018 me +1019 mean +1020 measur +1021 mechan +1022 media +1023 medic +1024 meet +1025 member +1026 membership +1027 memori +1028 men +1029 mention +1030 menu +1031 merchant +1032 messag +1033 method +1034 mh +1035 michael +1036 microsoft +1037 middl +1038 might +1039 mike +1040 mile +1041 militari +1042 million +1043 mime +1044 mind +1045 mine +1046 mini +1047 minimum +1048 minut +1049 miss +1050 mistak +1051 mobil +1052 mode +1053 model +1054 modem +1055 modifi +1056 modul +1057 moment +1058 mon +1059 mondai +1060 monei +1061 monitor +1062 month +1063 monthli +1064 more +1065 morn +1066 mortgag +1067 most +1068 mostli +1069 mother +1070 motiv +1071 move +1072 movi +1073 mpnumber +1074 mr +1075 ms +1076 msg +1077 much +1078 multi +1079 multipart +1080 multipl +1081 murphi +1082 music +1083 must +1084 my +1085 myself +1086 name +1087 nation +1088 natur +1089 nbsp +1090 near +1091 nearli +1092 necessari +1093 need +1094 neg +1095 net +1096 netscap +1097 network +1098 never +1099 new +1100 newslett +1101 next +1102 nextpart +1103 nice +1104 nigeria +1105 night +1106 no +1107 nobodi +1108 non +1109 none +1110 nor +1111 normal +1112 north +1113 not +1114 note +1115 noth +1116 notic +1117 now +1118 nt +1119 null +1120 number +1121 numbera +1122 numberam +1123 numberanumb +1124 numberb +1125 numberbit +1126 numberc +1127 numbercb +1128 numbercbr +1129 numbercfont +1130 numbercli +1131 numbercnumb +1132 numbercp +1133 numberctd +1134 numberd +1135 numberdari +1136 numberdnumb +1137 numberenumb +1138 numberf +1139 numberfb +1140 numberff +1141 numberffont +1142 numberfp +1143 numberftd +1144 numberk +1145 numberm +1146 numbermb +1147 numberp +1148 numberpd +1149 numberpm +1150 numberpx +1151 numberst +1152 numberth +1153 numbertnumb +1154 numberx +1155 object +1156 oblig +1157 obtain +1158 obvious +1159 occur +1160 oct +1161 octob +1162 of +1163 off +1164 offer +1165 offic +1166 offici +1167 often +1168 oh +1169 ok +1170 old +1171 on +1172 onc +1173 onli +1174 onlin +1175 open +1176 oper +1177 opinion +1178 opportun +1179 opt +1180 optim +1181 option +1182 or +1183 order +1184 org +1185 organ +1186 origin +1187 os +1188 osdn +1189 other +1190 otherwis +1191 our +1192 out +1193 outlook +1194 output +1195 outsid +1196 over +1197 own +1198 owner +1199 oz +1200 pacif +1201 pack +1202 packag +1203 page +1204 pai +1205 paid +1206 pain +1207 palm +1208 panel +1209 paper +1210 paragraph +1211 parent +1212 part +1213 parti +1214 particip +1215 particular +1216 particularli +1217 partit +1218 partner +1219 pass +1220 password +1221 past +1222 patch +1223 patent +1224 path +1225 pattern +1226 paul +1227 payment +1228 pc +1229 peac +1230 peopl +1231 per +1232 percent +1233 percentag +1234 perfect +1235 perfectli +1236 perform +1237 perhap +1238 period +1239 perl +1240 perman +1241 permiss +1242 person +1243 pgp +1244 phone +1245 photo +1246 php +1247 phrase +1248 physic +1249 pick +1250 pictur +1251 piec +1252 piiiiiiii +1253 pipe +1254 pjnumber +1255 place +1256 plai +1257 plain +1258 plan +1259 planet +1260 plant +1261 planta +1262 platform +1263 player +1264 pleas +1265 plu +1266 plug +1267 pm +1268 pocket +1269 point +1270 polic +1271 polici +1272 polit +1273 poor +1274 pop +1275 popul +1276 popular +1277 port +1278 posit +1279 possibl +1280 post +1281 potenti +1282 pound +1283 powel +1284 power +1285 powershot +1286 practic +1287 pre +1288 predict +1289 prefer +1290 premium +1291 prepar +1292 present +1293 presid +1294 press +1295 pretti +1296 prevent +1297 previou +1298 previous +1299 price +1300 principl +1301 print +1302 printabl +1303 printer +1304 privaci +1305 privat +1306 prize +1307 pro +1308 probabl +1309 problem +1310 procedur +1311 process +1312 processor +1313 procmail +1314 produc +1315 product +1316 profession +1317 profil +1318 profit +1319 program +1320 programm +1321 progress +1322 project +1323 promis +1324 promot +1325 prompt +1326 properti +1327 propos +1328 proprietari +1329 prospect +1330 protect +1331 protocol +1332 prove +1333 proven +1334 provid +1335 proxi +1336 pub +1337 public +1338 publish +1339 pudg +1340 pull +1341 purchas +1342 purpos +1343 put +1344 python +1345 qnumber +1346 qualifi +1347 qualiti +1348 quarter +1349 question +1350 quick +1351 quickli +1352 quit +1353 quot +1354 radio +1355 ragga +1356 rais +1357 random +1358 rang +1359 rate +1360 rather +1361 ratio +1362 razor +1363 razornumb +1364 re +1365 reach +1366 read +1367 reader +1368 readi +1369 real +1370 realiz +1371 realli +1372 reason +1373 receiv +1374 recent +1375 recipi +1376 recommend +1377 record +1378 red +1379 redhat +1380 reduc +1381 refer +1382 refin +1383 reg +1384 regard +1385 region +1386 regist +1387 regul +1388 regular +1389 rel +1390 relat +1391 relationship +1392 releas +1393 relev +1394 reliabl +1395 remain +1396 rememb +1397 remot +1398 remov +1399 replac +1400 repli +1401 report +1402 repositori +1403 repres +1404 republ +1405 request +1406 requir +1407 research +1408 reserv +1409 resid +1410 resourc +1411 respect +1412 respond +1413 respons +1414 rest +1415 result +1416 retail +1417 return +1418 reveal +1419 revenu +1420 revers +1421 review +1422 revok +1423 rh +1424 rich +1425 right +1426 risk +1427 road +1428 robert +1429 rock +1430 role +1431 roll +1432 rom +1433 roman +1434 room +1435 root +1436 round +1437 rpm +1438 rss +1439 rule +1440 run +1441 sa +1442 safe +1443 sai +1444 said +1445 sale +1446 same +1447 sampl +1448 san +1449 saou +1450 sat +1451 satellit +1452 save +1453 saw +1454 scan +1455 schedul +1456 school +1457 scienc +1458 score +1459 screen +1460 script +1461 se +1462 search +1463 season +1464 second +1465 secret +1466 section +1467 secur +1468 see +1469 seed +1470 seek +1471 seem +1472 seen +1473 select +1474 self +1475 sell +1476 seminar +1477 send +1478 sender +1479 sendmail +1480 senior +1481 sens +1482 sensit +1483 sent +1484 sep +1485 separ +1486 septemb +1487 sequenc +1488 seri +1489 serif +1490 seriou +1491 serv +1492 server +1493 servic +1494 set +1495 setup +1496 seven +1497 seventh +1498 sever +1499 sex +1500 sexual +1501 sf +1502 shape +1503 share +1504 she +1505 shell +1506 ship +1507 shop +1508 short +1509 shot +1510 should +1511 show +1512 side +1513 sign +1514 signatur +1515 signific +1516 similar +1517 simpl +1518 simpli +1519 sinc +1520 sincer +1521 singl +1522 sit +1523 site +1524 situat +1525 six +1526 size +1527 skeptic +1528 skill +1529 skin +1530 skip +1531 sleep +1532 slow +1533 small +1534 smart +1535 smoke +1536 smtp +1537 snumber +1538 so +1539 social +1540 societi +1541 softwar +1542 sold +1543 solut +1544 solv +1545 some +1546 someon +1547 someth +1548 sometim +1549 son +1550 song +1551 soni +1552 soon +1553 sorri +1554 sort +1555 sound +1556 sourc +1557 south +1558 space +1559 spain +1560 spam +1561 spamassassin +1562 spamd +1563 spammer +1564 speak +1565 spec +1566 special +1567 specif +1568 specifi +1569 speech +1570 speed +1571 spend +1572 sponsor +1573 sport +1574 spot +1575 src +1576 ssh +1577 st +1578 stabl +1579 staff +1580 stai +1581 stand +1582 standard +1583 star +1584 start +1585 state +1586 statement +1587 statu +1588 step +1589 steve +1590 still +1591 stock +1592 stop +1593 storag +1594 store +1595 stori +1596 strategi +1597 stream +1598 street +1599 string +1600 strip +1601 strong +1602 structur +1603 studi +1604 stuff +1605 stupid +1606 style +1607 subject +1608 submit +1609 subscrib +1610 subscript +1611 substanti +1612 success +1613 such +1614 suffer +1615 suggest +1616 suit +1617 sum +1618 summari +1619 summer +1620 sun +1621 super +1622 suppli +1623 support +1624 suppos +1625 sure +1626 surpris +1627 suse +1628 suspect +1629 sweet +1630 switch +1631 system +1632 tab +1633 tabl +1634 tablet +1635 tag +1636 take +1637 taken +1638 talk +1639 tape +1640 target +1641 task +1642 tax +1643 teach +1644 team +1645 tech +1646 technic +1647 techniqu +1648 technolog +1649 tel +1650 telecom +1651 telephon +1652 tell +1653 temperatur +1654 templ +1655 ten +1656 term +1657 termin +1658 terror +1659 terrorist +1660 test +1661 texa +1662 text +1663 than +1664 thank +1665 that +1666 the +1667 thei +1668 their +1669 them +1670 themselv +1671 then +1672 theori +1673 there +1674 therefor +1675 these +1676 thi +1677 thing +1678 think +1679 thinkgeek +1680 third +1681 those +1682 though +1683 thought +1684 thousand +1685 thread +1686 threat +1687 three +1688 through +1689 thu +1690 thursdai +1691 ti +1692 ticket +1693 tim +1694 time +1695 tip +1696 tire +1697 titl +1698 tm +1699 to +1700 todai +1701 togeth +1702 token +1703 told +1704 toll +1705 tom +1706 toner +1707 toni +1708 too +1709 took +1710 tool +1711 top +1712 topic +1713 total +1714 touch +1715 toward +1716 track +1717 trade +1718 tradit +1719 traffic +1720 train +1721 transact +1722 transfer +1723 travel +1724 treat +1725 tree +1726 tri +1727 trial +1728 trick +1729 trip +1730 troubl +1731 true +1732 truli +1733 trust +1734 truth +1735 try +1736 tue +1737 tuesdai +1738 turn +1739 tv +1740 two +1741 type +1742 uk +1743 ultim +1744 un +1745 under +1746 understand +1747 unfortun +1748 uniqu +1749 unison +1750 unit +1751 univers +1752 unix +1753 unless +1754 unlik +1755 unlimit +1756 unseen +1757 unsolicit +1758 unsubscrib +1759 until +1760 up +1761 updat +1762 upgrad +1763 upon +1764 urgent +1765 url +1766 us +1767 usa +1768 usag +1769 usb +1770 usd +1771 usdollarnumb +1772 useless +1773 user +1774 usr +1775 usual +1776 util +1777 vacat +1778 valid +1779 valu +1780 valuabl +1781 var +1782 variabl +1783 varieti +1784 variou +1785 ve +1786 vendor +1787 ventur +1788 veri +1789 verifi +1790 version +1791 via +1792 video +1793 view +1794 virtual +1795 visa +1796 visit +1797 visual +1798 vnumber +1799 voic +1800 vote +1801 vs +1802 vulner +1803 wa +1804 wai +1805 wait +1806 wake +1807 walk +1808 wall +1809 want +1810 war +1811 warm +1812 warn +1813 warranti +1814 washington +1815 wasn +1816 wast +1817 watch +1818 water +1819 we +1820 wealth +1821 weapon +1822 web +1823 weblog +1824 websit +1825 wed +1826 wednesdai +1827 week +1828 weekli +1829 weight +1830 welcom +1831 well +1832 went +1833 were +1834 west +1835 what +1836 whatev +1837 when +1838 where +1839 whether +1840 which +1841 while +1842 white +1843 whitelist +1844 who +1845 whole +1846 whose +1847 why +1848 wi +1849 wide +1850 width +1851 wife +1852 will +1853 william +1854 win +1855 window +1856 wing +1857 winner +1858 wireless +1859 wish +1860 with +1861 within +1862 without +1863 wnumberp +1864 woman +1865 women +1866 won +1867 wonder +1868 word +1869 work +1870 worker +1871 world +1872 worldwid +1873 worri +1874 worst +1875 worth +1876 would +1877 wouldn +1878 write +1879 written +1880 wrong +1881 wrote +1882 www +1883 ximian +1884 xml +1885 xp +1886 yahoo +1887 ye +1888 yeah +1889 year +1890 yesterdai +1891 yet +1892 york +1893 you +1894 young +1895 your +1896 yourself +1897 zdnet +1898 zero +1899 zip diff --git a/Exercise6/Figures/dataset1.png b/Exercise6/Figures/dataset1.png new file mode 100755 index 0000000..1746db6 Binary files /dev/null and b/Exercise6/Figures/dataset1.png differ diff --git a/Exercise6/Figures/dataset2.png b/Exercise6/Figures/dataset2.png new file mode 100644 index 0000000..65ee2e3 Binary files /dev/null and b/Exercise6/Figures/dataset2.png differ diff --git a/Exercise6/Figures/dataset3.png b/Exercise6/Figures/dataset3.png new file mode 100644 index 0000000..57c5338 Binary files /dev/null and b/Exercise6/Figures/dataset3.png differ diff --git a/Exercise6/Figures/email.png b/Exercise6/Figures/email.png new file mode 100644 index 0000000..c025f36 Binary files /dev/null and b/Exercise6/Figures/email.png differ diff --git a/Exercise6/Figures/email_cleaned.png b/Exercise6/Figures/email_cleaned.png new file mode 100644 index 0000000..6ba070b Binary files /dev/null and b/Exercise6/Figures/email_cleaned.png differ diff --git a/Exercise6/Figures/svm_c1.png b/Exercise6/Figures/svm_c1.png new file mode 100644 index 0000000..ccc160c Binary files /dev/null and b/Exercise6/Figures/svm_c1.png differ diff --git a/Exercise6/Figures/svm_c100.png b/Exercise6/Figures/svm_c100.png new file mode 100644 index 0000000..d8be018 Binary files /dev/null and b/Exercise6/Figures/svm_c100.png differ diff --git a/Exercise6/Figures/svm_dataset2.png b/Exercise6/Figures/svm_dataset2.png new file mode 100644 index 0000000..5acf2b0 Binary files /dev/null and b/Exercise6/Figures/svm_dataset2.png differ diff --git a/Exercise6/Figures/svm_dataset3_best.png b/Exercise6/Figures/svm_dataset3_best.png new file mode 100644 index 0000000..0cc45b5 Binary files /dev/null and b/Exercise6/Figures/svm_dataset3_best.png differ diff --git a/Exercise6/Figures/svm_predictors.png b/Exercise6/Figures/svm_predictors.png new file mode 100755 index 0000000..4917f69 Binary files /dev/null and b/Exercise6/Figures/svm_predictors.png differ diff --git a/Exercise6/Figures/vocab.png b/Exercise6/Figures/vocab.png new file mode 100644 index 0000000..f7157d5 Binary files /dev/null and b/Exercise6/Figures/vocab.png differ diff --git a/Exercise6/Figures/word_indices.png b/Exercise6/Figures/word_indices.png new file mode 100644 index 0000000..7c91ef7 Binary files /dev/null and b/Exercise6/Figures/word_indices.png differ diff --git a/Exercise6/exercise6.ipynb b/Exercise6/exercise6.ipynb new file mode 100755 index 0000000..cdc4397 --- /dev/null +++ b/Exercise6/exercise6.ipynb @@ -0,0 +1,1029 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Programming Exercise 6:\n", + "# Support Vector Machines\n", + "\n", + "## Introduction\n", + "\n", + "In this exercise, you will be using support vector machines (SVMs) to build a spam classifier. Before starting on the programming exercise, we strongly recommend watching the video lectures and completing the review questions for the associated topics.\n", + "\n", + "All the information you need for solving this assignment is in this notebook, and all the code you will be implementing will take place within this notebook. The assignment can be promptly submitted to the coursera grader directly from this notebook (code and instructions are included below).\n", + "\n", + "Before we begin with the exercises, we need to import all libraries required for this programming exercise. Throughout the course, we will be using [`numpy`](http://www.numpy.org/) for all arrays and matrix operations, [`matplotlib`](https://matplotlib.org/) for plotting, and [`scipy`](https://docs.scipy.org/doc/scipy/reference/) for scientific and numerical computation functions and tools. You can find instructions on how to install required libraries in the README file in the [github repository](https://github.com/dibgerge/ml-coursera-python-assignments)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# used for manipulating directory paths\n", + "import os\n", + "\n", + "# Scientific and vector computation for python\n", + "import numpy as np\n", + "\n", + "# Import regular expressions to process emails\n", + "import re\n", + "\n", + "# Plotting library\n", + "from matplotlib import pyplot\n", + "\n", + "# Optimization module in scipy\n", + "from scipy import optimize\n", + "\n", + "# will be used to load MATLAB mat datafile format\n", + "from scipy.io import loadmat\n", + "\n", + "# library written for this exercise providing additional functions for assignment submission, and others\n", + "import utils\n", + "\n", + "# define the submission/grader object for this exercise\n", + "grader = utils.Grader()\n", + "\n", + "# tells matplotlib to embed plots within the notebook\n", + "%matplotlib inline" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Submission and Grading\n", + "\n", + "\n", + "After completing each part of the assignment, be sure to submit your solutions to the grader. The following is a breakdown of how each part of this exercise is scored.\n", + "\n", + "\n", + "| Section | Part | Submitted Function | Points |\n", + "| :- |:- |:- | :-: |\n", + "| 1 | [Gaussian Kernel](#section1) | [`gaussianKernel`](#gaussianKernel) | 25 |\n", + "| 2 | [Parameters (C, $\\sigma$) for Dataset 3](#section2)| [`dataset3Params`](#dataset3Params) | 25 |\n", + "| 3 | [Email Preprocessing](#section3) | [`processEmail`](#processEmail) | 25 |\n", + "| 4 | [Email Feature Extraction](#section4) | [`emailFeatures`](#emailFeatures) | 25 |\n", + "| | Total Points | |100 |\n", + "\n", + "\n", + "You are allowed to submit your solutions multiple times, and we will take only the highest score into consideration.\n", + "\n", + "
\n", + "At the end of each section in this notebook, we have a cell which contains code for submitting the solutions thus far to the grader. Execute the cell to see your score up to the current section. For all your work to be submitted properly, you must execute those cells at least once.\n", + "
" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 1 Support Vector Machines\n", + "\n", + "In the first half of this exercise, you will be using support vector machines (SVMs) with various example 2D datasets. Experimenting with these datasets will help you gain an intuition of how SVMs work and how to use a Gaussian kernel with SVMs. In the next half of the exercise, you will be using support\n", + "vector machines to build a spam classifier." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 1.1 Example Dataset 1\n", + "\n", + "We will begin by with a 2D example dataset which can be separated by a linear boundary. The following cell plots the training data, which should look like this:\n", + "\n", + "![Dataset 1 training data](Figures/dataset1.png)\n", + "\n", + "In this dataset, the positions of the positive examples (indicated with `x`) and the negative examples (indicated with `o`) suggest a natural separation indicated by the gap. However, notice that there is an outlier positive example `x` on the far left at about (0.1, 4.1). As part of this exercise, you will also see how this outlier affects the SVM decision boundary." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Load from ex6data1\n", + "# You will have X, y as keys in the dict data\n", + "data = loadmat(os.path.join('Data', 'ex6data1.mat'))\n", + "X, y = data['X'], data['y'][:, 0]\n", + "\n", + "# Plot training data\n", + "utils.plotData(X, y)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In this part of the exercise, you will try using different values of the $C$ parameter with SVMs. Informally, the $C$ parameter is a positive value that controls the penalty for misclassified training examples. A large $C$ parameter tells the SVM to try to classify all the examples correctly. $C$ plays a role similar to $1/\\lambda$, where $\\lambda$ is the regularization parameter that we were using previously for logistic regression.\n", + "\n", + "\n", + "The following cell will run the SVM training (with $C=1$) using SVM software that we have included with the starter code (function `svmTrain` within the `utils` module of this exercise). When $C=1$, you should find that the SVM puts the decision boundary in the gap between the two datasets and *misclassifies* the data point on the far left, as shown in the figure (left) below.\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
SVM Decision boundary for example dataset 1
C=1C=100
\n", + "\n", + "
\n", + "In order to minimize the dependency of this assignment on external libraries, we have included this implementation of an SVM learning algorithm in utils.svmTrain. However, this particular implementation is not very efficient (it was originally chosen to maximize compatibility between Octave/MATLAB for the first version of this assignment set). If you are training an SVM on a real problem, especially if you need to scale to a larger dataset, we strongly recommend instead using a highly optimized SVM toolbox such as [LIBSVM](https://www.csie.ntu.edu.tw/~cjlin/libsvm/). The python machine learning library [scikit-learn](http://scikit-learn.org/stable/index.html) provides wrappers for the LIBSVM library.\n", + "
\n", + "
\n", + "
\n", + "**Implementation Note:** Most SVM software packages (including the function `utils.svmTrain`) automatically add the extra feature $x_0$ = 1 for you and automatically take care of learning the intercept term $\\theta_0$. So when passing your training data to the SVM software, there is no need to add this extra feature $x_0 = 1$ yourself. In particular, in python your code should be working with training examples $x \\in \\mathcal{R}^n$ (rather than $x \\in \\mathcal{R}^{n+1}$); for example, in the first example dataset $x \\in \\mathcal{R}^2$.\n", + "
\n", + "\n", + "Your task is to try different values of $C$ on this dataset. Specifically, you should change the value of $C$ in the next cell to $C = 100$ and run the SVM training again. When $C = 100$, you should find that the SVM now classifies every single example correctly, but has a decision boundary that does not\n", + "appear to be a natural fit for the data." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# You should try to change the C value below and see how the decision\n", + "# boundary varies (e.g., try C = 1000)\n", + "C = 1\n", + "\n", + "model = utils.svmTrain(X, y, C, utils.linearKernel, 1e-3, 20)\n", + "utils.visualizeBoundaryLinear(X, y, model)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 1.2 SVM with Gaussian Kernels\n", + "\n", + "In this part of the exercise, you will be using SVMs to do non-linear classification. In particular, you will be using SVMs with Gaussian kernels on datasets that are not linearly separable.\n", + "\n", + "#### 1.2.1 Gaussian Kernel\n", + "\n", + "To find non-linear decision boundaries with the SVM, we need to first implement a Gaussian kernel. You can think of the Gaussian kernel as a similarity function that measures the “distance” between a pair of examples,\n", + "($x^{(i)}$, $x^{(j)}$). The Gaussian kernel is also parameterized by a bandwidth parameter, $\\sigma$, which determines how fast the similarity metric decreases (to 0) as the examples are further apart.\n", + "You should now complete the code in `gaussianKernel` to compute the Gaussian kernel between two examples, ($x^{(i)}$, $x^{(j)}$). The Gaussian kernel function is defined as:\n", + "\n", + "$$ K_{\\text{gaussian}} \\left( x^{(i)}, x^{(j)} \\right) = \\exp \\left( - \\frac{\\left\\lvert\\left\\lvert x^{(i)} - x^{(j)}\\right\\lvert\\right\\lvert^2}{2\\sigma^2} \\right) = \\exp \\left( -\\frac{\\sum_{k=1}^n \\left( x_k^{(i)} - x_k^{(j)}\\right)^2}{2\\sigma^2} \\right)$$\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def gaussianKernel(x1, x2, sigma):\n", + " \"\"\"\n", + " Computes the radial basis function\n", + " Returns a radial basis function kernel between x1 and x2.\n", + " \n", + " Parameters\n", + " ----------\n", + " x1 : numpy ndarray\n", + " A vector of size (n, ), representing the first datapoint.\n", + " \n", + " x2 : numpy ndarray\n", + " A vector of size (n, ), representing the second datapoint.\n", + " \n", + " sigma : float\n", + " The bandwidth parameter for the Gaussian kernel.\n", + "\n", + " Returns\n", + " -------\n", + " sim : float\n", + " The computed RBF between the two provided data points.\n", + " \n", + " Instructions\n", + " ------------\n", + " Fill in this function to return the similarity between `x1` and `x2`\n", + " computed using a Gaussian kernel with bandwidth `sigma`.\n", + " \"\"\"\n", + " sim = 0\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + "\n", + "\n", + " # =============================================================\n", + " return sim" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Once you have completed the function `gaussianKernel` the following cell will test your kernel function on two provided examples and you should expect to see a value of 0.324652." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "x1 = np.array([1, 2, 1])\n", + "x2 = np.array([0, 4, -1])\n", + "sigma = 2\n", + "\n", + "sim = gaussianKernel(x1, x2, sigma)\n", + "\n", + "print('Gaussian Kernel between x1 = [1, 2, 1], x2 = [0, 4, -1], sigma = %0.2f:'\n", + " '\\n\\t%f\\n(for sigma = 2, this value should be about 0.324652)\\n' % (sigma, sim))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[1] = gaussianKernel\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 1.2.2 Example Dataset 2\n", + "\n", + "The next part in this notebook will load and plot dataset 2, as shown in the figure below. \n", + "\n", + "![Dataset 2](Figures/dataset2.png)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Load from ex6data2\n", + "# You will have X, y as keys in the dict data\n", + "data = loadmat(os.path.join('Data', 'ex6data2.mat'))\n", + "X, y = data['X'], data['y'][:, 0]\n", + "\n", + "# Plot training data\n", + "utils.plotData(X, y)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "From the figure, you can obserse that there is no linear decision boundary that separates the positive and negative examples for this dataset. However, by using the Gaussian kernel with the SVM, you will be able to learn a non-linear decision boundary that can perform reasonably well for the dataset. If you have correctly implemented the Gaussian kernel function, the following cell will proceed to train the SVM with the Gaussian kernel on this dataset.\n", + "\n", + "You should get a decision boundary as shown in the figure below, as computed by the SVM with a Gaussian kernel. The decision boundary is able to separate most of the positive and negative examples correctly and follows the contours of the dataset well.\n", + "\n", + "![Dataset 2 decision boundary](Figures/svm_dataset2.png)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# SVM Parameters\n", + "C = 1\n", + "sigma = 0.1\n", + "\n", + "model= utils.svmTrain(X, y, C, gaussianKernel, args=(sigma,))\n", + "utils.visualizeBoundary(X, y, model)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "#### 1.2.3 Example Dataset 3\n", + "\n", + "In this part of the exercise, you will gain more practical skills on how to use a SVM with a Gaussian kernel. The next cell will load and display a third dataset, which should look like the figure below.\n", + "\n", + "![Dataset 3](Figures/dataset3.png)\n", + "\n", + "You will be using the SVM with the Gaussian kernel with this dataset. In the provided dataset, `ex6data3.mat`, you are given the variables `X`, `y`, `Xval`, `yval`. " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Load from ex6data3\n", + "# You will have X, y, Xval, yval as keys in the dict data\n", + "data = loadmat(os.path.join('Data', 'ex6data3.mat'))\n", + "X, y, Xval, yval = data['X'], data['y'][:, 0], data['Xval'], data['yval'][:, 0]\n", + "\n", + "# Plot training data\n", + "utils.plotData(X, y)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Your task is to use the cross validation set `Xval`, `yval` to determine the best $C$ and $\\sigma$ parameter to use. You should write any additional code necessary to help you search over the parameters $C$ and $\\sigma$. For both $C$ and $\\sigma$, we suggest trying values in multiplicative steps (e.g., 0.01, 0.03, 0.1, 0.3, 1, 3, 10, 30).\n", + "Note that you should try all possible pairs of values for $C$ and $\\sigma$ (e.g., $C = 0.3$ and $\\sigma = 0.1$). For example, if you try each of the 8 values listed above for $C$ and for $\\sigma^2$, you would end up training and evaluating (on the cross validation set) a total of $8^2 = 64$ different models. After you have determined the best $C$ and $\\sigma$ parameters to use, you should modify the code in `dataset3Params`, filling in the best parameters you found. For our best parameters, the SVM returned a decision boundary shown in the figure below. \n", + "\n", + "![](Figures/svm_dataset3_best.png)\n", + "\n", + "
\n", + "**Implementation Tip:** When implementing cross validation to select the best $C$ and $\\sigma$ parameter to use, you need to evaluate the error on the cross validation set. Recall that for classification, the error is defined as the fraction of the cross validation examples that were classified incorrectly. In `numpy`, you can compute this error using `np.mean(predictions != yval)`, where `predictions` is a vector containing all the predictions from the SVM, and `yval` are the true labels from the cross validation set. You can use the `utils.svmPredict` function to generate the predictions for the cross validation set.\n", + "
\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def dataset3Params(X, y, Xval, yval):\n", + " \"\"\"\n", + " Returns your choice of C and sigma for Part 3 of the exercise \n", + " where you select the optimal (C, sigma) learning parameters to use for SVM\n", + " with RBF kernel.\n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like\n", + " (m x n) matrix of training data where m is number of training examples, and \n", + " n is the number of features.\n", + " \n", + " y : array_like\n", + " (m, ) vector of labels for ther training data.\n", + " \n", + " Xval : array_like\n", + " (mv x n) matrix of validation data where mv is the number of validation examples\n", + " and n is the number of features\n", + " \n", + " yval : array_like\n", + " (mv, ) vector of labels for the validation data.\n", + " \n", + " Returns\n", + " -------\n", + " C, sigma : float, float\n", + " The best performing values for the regularization parameter C and \n", + " RBF parameter sigma.\n", + " \n", + " Instructions\n", + " ------------\n", + " Fill in this function to return the optimal C and sigma learning \n", + " parameters found using the cross validation set.\n", + " You can use `svmPredict` to predict the labels on the cross\n", + " validation set. For example, \n", + " \n", + " predictions = svmPredict(model, Xval)\n", + "\n", + " will return the predictions on the cross validation set.\n", + " \n", + " Note\n", + " ----\n", + " You can compute the prediction error using \n", + " \n", + " np.mean(predictions != yval)\n", + " \"\"\"\n", + " # You need to return the following variables correctly.\n", + " C = 1\n", + " sigma = 0.3\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + " \n", + " \n", + " # ============================================================\n", + " return C, sigma" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The provided code in the next cell trains the SVM classifier using the training set $(X, y)$ using parameters loaded from `dataset3Params`. Note that this might take a few minutes to execute." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Try different SVM Parameters here\n", + "C, sigma = dataset3Params(X, y, Xval, yval)\n", + "\n", + "# Train the SVM\n", + "# model = utils.svmTrain(X, y, C, lambda x1, x2: gaussianKernel(x1, x2, sigma))\n", + "model = utils.svmTrain(X, y, C, gaussianKernel, args=(sigma,))\n", + "utils.visualizeBoundary(X, y, model)\n", + "print(C, sigma)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "One you have computed the values `C` and `sigma` in the cell above, we will submit those values for grading.\n", + "\n", + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[2] = lambda : (C, sigma)\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "## 2 Spam Classification\n", + "\n", + "Many email services today provide spam filters that are able to classify emails into spam and non-spam email with high accuracy. In this part of the exercise, you will use SVMs to build your own spam filter.\n", + "\n", + "You will be training a classifier to classify whether a given email, $x$, is spam ($y = 1$) or non-spam ($y = 0$). In particular, you need to convert each email into a feature vector $x \\in \\mathbb{R}^n$ . The following parts of the exercise will walk you through how such a feature vector can be constructed from an email.\n", + "\n", + "The dataset included for this exercise is based on a a subset of the [SpamAssassin Public Corpus](http://spamassassin.apache.org/old/publiccorpus/). For the purpose of this exercise, you will only be using the body of the email (excluding the email headers)." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2.1 Preprocessing Emails\n", + "\n", + "Before starting on a machine learning task, it is usually insightful to take a look at examples from the dataset. The figure below shows a sample email that contains a URL, an email address (at the end), numbers, and dollar\n", + "amounts.\n", + "\n", + "\n", + "\n", + "While many emails would contain similar types of entities (e.g., numbers, other URLs, or other email addresses), the specific entities (e.g., the specific URL or specific dollar amount) will be different in almost every\n", + "email. Therefore, one method often employed in processing emails is to “normalize” these values, so that all URLs are treated the same, all numbers are treated the same, etc. For example, we could replace each URL in the\n", + "email with the unique string “httpaddr” to indicate that a URL was present.\n", + "\n", + "This has the effect of letting the spam classifier make a classification decision based on whether any URL was present, rather than whether a specific URL was present. This typically improves the performance of a spam classifier, since spammers often randomize the URLs, and thus the odds of seeing any particular URL again in a new piece of spam is very small. \n", + "\n", + "In the function `processEmail` below, we have implemented the following email preprocessing and normalization steps:\n", + "\n", + "- **Lower-casing**: The entire email is converted into lower case, so that captialization is ignored (e.g., IndIcaTE is treated the same as Indicate).\n", + "\n", + "- **Stripping HTML**: All HTML tags are removed from the emails. Many emails often come with HTML formatting; we remove all the HTML tags, so that only the content remains.\n", + "\n", + "- **Normalizing URLs**: All URLs are replaced with the text “httpaddr”.\n", + "\n", + "- **Normalizing Email Addresses**: All email addresses are replaced with the text “emailaddr”.\n", + "\n", + "- **Normalizing Numbers**: All numbers are replaced with the text “number”.\n", + "\n", + "- **Normalizing Dollars**: All dollar signs ($) are replaced with the text “dollar”.\n", + "\n", + "- **Word Stemming**: Words are reduced to their stemmed form. For example, “discount”, “discounts”, “discounted” and “discounting” are all replaced with “discount”. Sometimes, the Stemmer actually strips off additional characters from the end, so “include”, “includes”, “included”, and “including” are all replaced with “includ”.\n", + "\n", + "- **Removal of non-words**: Non-words and punctuation have been removed. All white spaces (tabs, newlines, spaces) have all been trimmed to a single space character.\n", + "\n", + "The result of these preprocessing steps is shown in the figure below. \n", + "\n", + "\"email\n", + "\n", + "While preprocessing has left word fragments and non-words, this form turns out to be much easier to work with for performing feature extraction." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "#### 2.1.1 Vocabulary List\n", + "\n", + "After preprocessing the emails, we have a list of words for each email. The next step is to choose which words we would like to use in our classifier and which we would want to leave out.\n", + "\n", + "For this exercise, we have chosen only the most frequently occuring words as our set of words considered (the vocabulary list). Since words that occur rarely in the training set are only in a few emails, they might cause the\n", + "model to overfit our training set. The complete vocabulary list is in the file `vocab.txt` (inside the `Data` directory for this exercise) and also shown in the figure below.\n", + "\n", + "\"Vocab\"\n", + "\n", + "Our vocabulary list was selected by choosing all words which occur at least a 100 times in the spam corpus,\n", + "resulting in a list of 1899 words. In practice, a vocabulary list with about 10,000 to 50,000 words is often used.\n", + "Given the vocabulary list, we can now map each word in the preprocessed emails into a list of word indices that contains the index of the word in the vocabulary dictionary. The figure below shows the mapping for the sample email. Specifically, in the sample email, the word “anyone” was first normalized to “anyon” and then mapped onto the index 86 in the vocabulary list.\n", + "\n", + "\"word\n", + "\n", + "Your task now is to complete the code in the function `processEmail` to perform this mapping. In the code, you are given a string `word` which is a single word from the processed email. You should look up the word in the vocabulary list `vocabList`. If the word exists in the list, you should add the index of the word into the `word_indices` variable. If the word does not exist, and is therefore not in the vocabulary, you can skip the word.\n", + "\n", + "
\n", + "**python tip**: In python, you can find the index of the first occurence of an item in `list` using the `index` attribute. In the provided code for `processEmail`, `vocabList` is a python list containing the words in the vocabulary. To find the index of a word, we can use `vocabList.index(word)` which would return a number indicating the index of the word within the list. If the word does not exist in the list, a `ValueError` exception is raised. In python, we can use the `try/except` statement to catch exceptions which we do not want to stop the program from running. You can think of the `try/except` statement to be the same as an `if/else` statement, but it asks for forgiveness rather than permission.\n", + "\n", + "An example would be:\n", + "
\n", + "\n", + "```\n", + "try:\n", + " do stuff here\n", + "except ValueError:\n", + " pass\n", + " # do nothing (forgive me) if a ValueError exception occured within the try statement\n", + "```\n", + "
\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def processEmail(email_contents, verbose=True):\n", + " \"\"\"\n", + " Preprocesses the body of an email and returns a list of indices \n", + " of the words contained in the email. \n", + " \n", + " Parameters\n", + " ----------\n", + " email_contents : str\n", + " A string containing one email. \n", + " \n", + " verbose : bool\n", + " If True, print the resulting email after processing.\n", + " \n", + " Returns\n", + " -------\n", + " word_indices : list\n", + " A list of integers containing the index of each word in the \n", + " email which is also present in the vocabulary.\n", + " \n", + " Instructions\n", + " ------------\n", + " Fill in this function to add the index of word to word_indices \n", + " if it is in the vocabulary. At this point of the code, you have \n", + " a stemmed word from the email in the variable word.\n", + " You should look up word in the vocabulary list (vocabList). \n", + " If a match exists, you should add the index of the word to the word_indices\n", + " list. Concretely, if word = 'action', then you should\n", + " look up the vocabulary list to find where in vocabList\n", + " 'action' appears. For example, if vocabList[18] =\n", + " 'action', then, you should add 18 to the word_indices \n", + " vector (e.g., word_indices.append(18)).\n", + " \n", + " Notes\n", + " -----\n", + " - vocabList[idx] returns a the word with index idx in the vocabulary list.\n", + " \n", + " - vocabList.index(word) return index of word `word` in the vocabulary list.\n", + " (A ValueError exception is raised if the word does not exist.)\n", + " \"\"\"\n", + " # Load Vocabulary\n", + " vocabList = utils.getVocabList()\n", + "\n", + " # Init return value\n", + " word_indices = []\n", + "\n", + " # ========================== Preprocess Email ===========================\n", + " # Find the Headers ( \\n\\n and remove )\n", + " # Uncomment the following lines if you are working with raw emails with the\n", + " # full headers\n", + " # hdrstart = email_contents.find(chr(10) + chr(10))\n", + " # email_contents = email_contents[hdrstart:]\n", + "\n", + " # Lower case\n", + " email_contents = email_contents.lower()\n", + " \n", + " # Strip all HTML\n", + " # Looks for any expression that starts with < and ends with > and replace\n", + " # and does not have any < or > in the tag it with a space\n", + " email_contents =re.compile('<[^<>]+>').sub(' ', email_contents)\n", + "\n", + " # Handle Numbers\n", + " # Look for one or more characters between 0-9\n", + " email_contents = re.compile('[0-9]+').sub(' number ', email_contents)\n", + "\n", + " # Handle URLS\n", + " # Look for strings starting with http:// or https://\n", + " email_contents = re.compile('(http|https)://[^\\s]*').sub(' httpaddr ', email_contents)\n", + "\n", + " # Handle Email Addresses\n", + " # Look for strings with @ in the middle\n", + " email_contents = re.compile('[^\\s]+@[^\\s]+').sub(' emailaddr ', email_contents)\n", + " \n", + " # Handle $ sign\n", + " email_contents = re.compile('[$]+').sub(' dollar ', email_contents)\n", + " \n", + " # get rid of any punctuation\n", + " email_contents = re.split('[ @$/#.-:&*+=\\[\\]?!(){},''\">_<;%\\n\\r]', email_contents)\n", + "\n", + " # remove any empty word string\n", + " email_contents = [word for word in email_contents if len(word) > 0]\n", + " \n", + " # Stem the email contents word by word\n", + " stemmer = utils.PorterStemmer()\n", + " processed_email = []\n", + " for word in email_contents:\n", + " # Remove any remaining non alphanumeric characters in word\n", + " word = re.compile('[^a-zA-Z0-9]').sub('', word).strip()\n", + " word = stemmer.stem(word)\n", + " processed_email.append(word)\n", + "\n", + " if len(word) < 1:\n", + " continue\n", + "\n", + " # Look up the word in the dictionary and add to word_indices if found\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + " \n", + "\n", + " # =============================================================\n", + "\n", + " if verbose:\n", + " print('----------------')\n", + " print('Processed email:')\n", + " print('----------------')\n", + " print(' '.join(processed_email))\n", + " return word_indices" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Once you have implemented `processEmail`, the following cell will run your code on the email sample and you should see an output of the processed email and the indices list mapping." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# To use an SVM to classify emails into Spam v.s. Non-Spam, you first need\n", + "# to convert each email into a vector of features. In this part, you will\n", + "# implement the preprocessing steps for each email. You should\n", + "# complete the code in processEmail.m to produce a word indices vector\n", + "# for a given email.\n", + "\n", + "# Extract Features\n", + "with open(os.path.join('Data', 'emailSample1.txt')) as fid:\n", + " file_contents = fid.read()\n", + "\n", + "word_indices = processEmail(file_contents)\n", + "\n", + "#Print Stats\n", + "print('-------------')\n", + "print('Word Indices:')\n", + "print('-------------')\n", + "print(word_indices)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[3] = processEmail\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 2.2 Extracting Features from Emails\n", + "\n", + "You will now implement the feature extraction that converts each email into a vector in $\\mathbb{R}^n$. For this exercise, you will be using n = # words in vocabulary list. Specifically, the feature $x_i \\in \\{0, 1\\}$ for an email corresponds to whether the $i^{th}$ word in the dictionary occurs in the email. That is, $x_i = 1$ if the $i^{th}$ word is in the email and $x_i = 0$ if the $i^{th}$ word is not present in the email.\n", + "\n", + "Thus, for a typical email, this feature would look like:\n", + "\n", + "$$ x = \\begin{bmatrix} \n", + "0 & \\dots & 1 & 0 & \\dots & 1 & 0 & \\dots & 0 \n", + "\\end{bmatrix}^T \\in \\mathbb{R}^n\n", + "$$\n", + "\n", + "You should now complete the code in the function `emailFeatures` to generate a feature vector for an email, given the `word_indices`.\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def emailFeatures(word_indices):\n", + " \"\"\"\n", + " Takes in a word_indices vector and produces a feature vector from the word indices. \n", + " \n", + " Parameters\n", + " ----------\n", + " word_indices : list\n", + " A list of word indices from the vocabulary list.\n", + " \n", + " Returns\n", + " -------\n", + " x : list \n", + " The computed feature vector.\n", + " \n", + " Instructions\n", + " ------------\n", + " Fill in this function to return a feature vector for the\n", + " given email (word_indices). To help make it easier to process \n", + " the emails, we have have already pre-processed each email and converted\n", + " each word in the email into an index in a fixed dictionary (of 1899 words).\n", + " The variable `word_indices` contains the list of indices of the words \n", + " which occur in one email.\n", + " \n", + " Concretely, if an email has the text:\n", + "\n", + " The quick brown fox jumped over the lazy dog.\n", + "\n", + " Then, the word_indices vector for this text might look like:\n", + " \n", + " 60 100 33 44 10 53 60 58 5\n", + "\n", + " where, we have mapped each word onto a number, for example:\n", + "\n", + " the -- 60\n", + " quick -- 100\n", + " ...\n", + "\n", + " Note\n", + " ----\n", + " The above numbers are just an example and are not the actual mappings.\n", + "\n", + " Your task is take one such `word_indices` vector and construct\n", + " a binary feature vector that indicates whether a particular\n", + " word occurs in the email. That is, x[i] = 1 when word i\n", + " is present in the email. Concretely, if the word 'the' (say,\n", + " index 60) appears in the email, then x[60] = 1. The feature\n", + " vector should look like:\n", + " x = [ 0 0 0 0 1 0 0 0 ... 0 0 0 0 1 ... 0 0 0 1 0 ..]\n", + " \"\"\"\n", + " # Total number of words in the dictionary\n", + " n = 1899\n", + "\n", + " # You need to return the following variables correctly.\n", + " x = np.zeros(n)\n", + "\n", + " # ===================== YOUR CODE HERE ======================\n", + "\n", + " \n", + " \n", + " # ===========================================================\n", + " \n", + " return x" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Once you have implemented `emailFeatures`, the next cell will run your code on the email sample. You should see that the feature vector had length 1899 and 45 non-zero entries." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Extract Features\n", + "with open(os.path.join('Data', 'emailSample1.txt')) as fid:\n", + " file_contents = fid.read()\n", + "\n", + "word_indices = processEmail(file_contents)\n", + "features = emailFeatures(word_indices)\n", + "\n", + "# Print Stats\n", + "print('\\nLength of feature vector: %d' % len(features))\n", + "print('Number of non-zero entries: %d' % sum(features > 0))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[4] = emailFeatures\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2.3 Training SVM for Spam Classification\n", + "\n", + "In the following section we will load a preprocessed training dataset that will be used to train a SVM classifier. The file `spamTrain.mat` (within the `Data` folder for this exercise) contains 4000 training examples of spam and non-spam email, while `spamTest.mat` contains 1000 test examples. Each\n", + "original email was processed using the `processEmail` and `emailFeatures` functions and converted into a vector $x^{(i)} \\in \\mathbb{R}^{1899}$.\n", + "\n", + "After loading the dataset, the next cell proceed to train a linear SVM to classify between spam ($y = 1$) and non-spam ($y = 0$) emails. Once the training completes, you should see that the classifier gets a training accuracy of about 99.8% and a test accuracy of about 98.5%." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Load the Spam Email dataset\n", + "# You will have X, y in your environment\n", + "data = loadmat(os.path.join('Data', 'spamTrain.mat'))\n", + "X, y= data['X'].astype(float), data['y'][:, 0]\n", + "\n", + "print('Training Linear SVM (Spam Classification)')\n", + "print('This may take 1 to 2 minutes ...\\n')\n", + "\n", + "C = 0.1\n", + "model = utils.svmTrain(X, y, C, utils.linearKernel)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Compute the training accuracy\n", + "p = utils.svmPredict(model, X)\n", + "\n", + "print('Training Accuracy: %.2f' % (np.mean(p == y) * 100))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Execute the following cell to load the test set and compute the test accuracy." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Load the test dataset\n", + "# You will have Xtest, ytest in your environment\n", + "data = loadmat(os.path.join('Data', 'spamTest.mat'))\n", + "Xtest, ytest = data['Xtest'].astype(float), data['ytest'][:, 0]\n", + "\n", + "print('Evaluating the trained Linear SVM on a test set ...')\n", + "p = utils.svmPredict(model, Xtest)\n", + "\n", + "print('Test Accuracy: %.2f' % (np.mean(p == ytest) * 100))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2.4 Top Predictors for Spam\n", + "\n", + "To better understand how the spam classifier works, we can inspect the parameters to see which words the classifier thinks are the most predictive of spam. The next cell finds the parameters with the largest positive values in the classifier and displays the corresponding words similar to the ones shown in the figure below.\n", + "\n", + "
\n", + "our click remov guarante visit basenumb dollar pleas price will nbsp most lo ga hour\n", + "
\n", + "\n", + "Thus, if an email contains words such as “guarantee”, “remove”, “dollar”, and “price” (the top predictors shown in the figure), it is likely to be classified as spam.\n", + "\n", + "Since the model we are training is a linear SVM, we can inspect the weights learned by the model to understand better how it is determining whether an email is spam or not. The following code finds the words with the highest weights in the classifier. Informally, the classifier 'thinks' that these words are the most likely indicators of spam." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Sort the weights and obtin the vocabulary list\n", + "# NOTE some words have the same weights, \n", + "# so their order might be different than in the text above\n", + "idx = np.argsort(model['w'])\n", + "top_idx = idx[-15:][::-1]\n", + "vocabList = utils.getVocabList()\n", + "\n", + "print('Top predictors of spam:')\n", + "print('%-15s %-15s' % ('word', 'weight'))\n", + "print('----' + ' '*12 + '------')\n", + "for word, w in zip(np.array(vocabList)[top_idx], model['w'][top_idx]):\n", + " print('%-15s %0.2f' % (word, w))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2.5 Optional (ungraded) exercise: Try your own emails\n", + "\n", + "Now that you have trained a spam classifier, you can start trying it out on your own emails. In the starter code, we have included two email examples (`emailSample1.txt` and `emailSample2.txt`) and two spam examples (`spamSample1.txt` and `spamSample2.txt`). The next cell runs the spam classifier over the first spam example and classifies it using the learned SVM. You should now try the other examples we have provided and see if the classifier gets them right. You can also try your own emails by replacing the examples (plain text files) with your own emails.\n", + "\n", + "*You do not need to submit any solutions for this optional (ungraded) exercise.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "filename = os.path.join('Data', 'emailSample1.txt')\n", + "\n", + "with open(filename) as fid:\n", + " file_contents = fid.read()\n", + "\n", + "word_indices = processEmail(file_contents, verbose=False)\n", + "x = emailFeatures(word_indices)\n", + "p = utils.svmPredict(model, x)\n", + "\n", + "print('\\nProcessed %s\\nSpam Classification: %s' % (filename, 'spam' if p else 'not spam'))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2.6 Optional (ungraded) exercise: Build your own dataset\n", + "\n", + "In this exercise, we provided a preprocessed training set and test set. These datasets were created using the same functions (`processEmail` and `emailFeatures`) that you now have completed. For this optional (ungraded) exercise, you will build your own dataset using the original emails from the SpamAssassin Public Corpus.\n", + "\n", + "Your task in this optional (ungraded) exercise is to download the original\n", + "files from the public corpus and extract them. After extracting them, you should run the `processEmail` and `emailFeatures` functions on each email to extract a feature vector from each email. This will allow you to build a dataset `X`, `y` of examples. You should then randomly divide up the dataset into a training set, a cross validation set and a test set.\n", + "\n", + "While you are building your own dataset, we also encourage you to try building your own vocabulary list (by selecting the high frequency words that occur in the dataset) and adding any additional features that you think\n", + "might be useful. Finally, we also suggest trying to use highly optimized SVM toolboxes such as [`LIBSVM`](https://www.csie.ntu.edu.tw/~cjlin/libsvm/) or [`scikit-learn`](http://scikit-learn.org/stable/modules/classes.html#module-sklearn.svm).\n", + "\n", + "*You do not need to submit any solutions for this optional (ungraded) exercise.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.6.4" + } + }, + "nbformat": 4, + "nbformat_minor": 2 +} diff --git a/Exercise6/utils.py b/Exercise6/utils.py new file mode 100755 index 0000000..6d99cec --- /dev/null +++ b/Exercise6/utils.py @@ -0,0 +1,718 @@ +import sys + +sys.path.append('..') +from submission import SubmissionBase +import numpy as np +from scipy.io import loadmat +from os.path import join +from matplotlib import pyplot + + +def plotData(X, y, grid=False): + """ + Plots the data points X and y into a new figure. Uses `+` for positive examples, and `o` for + negative examples. `X` is assumed to be a Mx2 matrix + + Parameters + ---------- + X : numpy ndarray + X is assumed to be a Mx2 matrix. + + y : numpy ndarray + The data labels. + + grid : bool (Optional) + Specify whether or not to show the grid in the plot. It is False by default. + + Notes + ----- + This was slightly modified such that it expects y=1 or y=0. + """ + # Find Indices of Positive and Negative Examples + pos = y == 1 + neg = y == 0 + + # Plot Examples + pyplot.plot(X[pos, 0], X[pos, 1], 'X', mew=1, ms=10, mec='k') + pyplot.plot(X[neg, 0], X[neg, 1], 'o', mew=1, mfc='y', ms=10, mec='k') + pyplot.grid(grid) + + +def svmTrain(X, Y, C, kernelFunction, tol=1e-3, max_passes=5, args=()): + """ + Trains an SVM classifier using a simplified version of the SMO algorithm. + + Parameters + --------- + X : numpy ndarray + (m x n) Matrix of training examples. Each row is a training example, and the + jth column holds the jth feature. + + Y : numpy ndarray + (m, ) A vector (1-D numpy array) containing 1 for positive examples and 0 for negative examples. + + C : float + The standard SVM regularization parameter. + + kernelFunction : func + A function handle which computes the kernel. The function should accept two vectors as + inputs, and returns a scalar as output. + + tol : float, optional + Tolerance value used for determining equality of floating point numbers. + + max_passes : int, optional + Controls the number of iterations over the dataset (without changes to alpha) + before the algorithm quits. + + args : tuple + Extra arguments required for the kernel function, such as the sigma parameter for a + Gaussian kernel. + + Returns + ------- + model : + The trained SVM model. + + Notes + ----- + This is a simplified version of the SMO algorithm for training SVMs. In practice, if + you want to train an SVM classifier, we recommend using an optimized package such as: + + - LIBSVM (http://www.csie.ntu.edu.tw/~cjlin/libsvm/) + - SVMLight (http://svmlight.joachims.org/) + - scikit-learn (http://scikit-learn.org/stable/modules/svm.html) which contains python wrappers + for the LIBSVM library. + """ + # make sure data is signed int + Y = Y.astype(int) + # Dataset size parameters + m, n = X.shape + + passes = 0 + E = np.zeros(m) + alphas = np.zeros(m) + b = 0 + + # Map 0 to -1 + Y[Y == 0] = -1 + + # Pre-compute the Kernel Matrix since our dataset is small + # (in practice, optimized SVM packages that handle large datasets + # gracefully will **not** do this) + + # We have implemented the optimized vectorized version of the Kernels here so + # that the SVM training will run faster + if kernelFunction.__name__ == 'linearKernel': + # Vectorized computation for the linear kernel + # This is equivalent to computing the kernel on every pair of examples + K = np.dot(X, X.T) + elif kernelFunction.__name__ == 'gaussianKernel': + # vectorized RBF Kernel + # This is equivalent to computing the kernel on every pair of examples + X2 = np.sum(X**2, axis=1) + K = X2 + X2[:, None] - 2 * np.dot(X, X.T) + + if len(args) > 0: + K /= 2*args[0]**2 + + K = np.exp(-K) + else: + K = np.zeros((m, m)) + for i in range(m): + for j in range(i, m): + K[i, j] = kernelFunction(X[i, :], X[j, :]) + K[j, i] = K[i, j] + + while passes < max_passes: + num_changed_alphas = 0 + for i in range(m): + E[i] = b + np.sum(alphas * Y * K[:, i]) - Y[i] + + if (Y[i]*E[i] < -tol and alphas[i] < C) or (Y[i]*E[i] > tol and alphas[i] > 0): + # select the alpha_j randomly + j = np.random.choice(list(range(i)) + list(range(i+1, m)), size=1)[0] + + E[j] = b + np.sum(alphas * Y * K[:, j]) - Y[j] + + alpha_i_old = alphas[i] + alpha_j_old = alphas[j] + + if Y[i] == Y[j]: + L = max(0, alphas[j] + alphas[i] - C) + H = min(C, alphas[j] + alphas[i]) + else: + L = max(0, alphas[j] - alphas[i]) + H = min(C, C + alphas[j] - alphas[i]) + + if L == H: + continue + + eta = 2 * K[i, j] - K[i, i] - K[j, j] + + # objective function positive definite, there will be a minimum along the direction + # of linear equality constrain, and eta will be greater than zero + # we are actually computing -eta here (so we skip of eta >= 0) + if eta >= 0: + continue + + alphas[j] -= Y[j] * (E[i] - E[j])/eta + alphas[j] = max(L, min(H, alphas[j])) + + if abs(alphas[j] - alpha_j_old) < tol: + alphas[j] = alpha_j_old + continue + alphas[i] += Y[i]*Y[j]*(alpha_j_old - alphas[j]) + + b1 = b - E[i] - Y[i]*(alphas[i] - alpha_i_old) * K[i, j] \ + - Y[j] * (alphas[j] - alpha_j_old) * K[i, j] + + b2 = b - E[j] - Y[i]*(alphas[i] - alpha_i_old) * K[i, j] \ + - Y[j] * (alphas[j] - alpha_j_old) * K[j, j] + + if 0 < alphas[i] < C: + b = b1 + elif 0 < alphas[j] < C: + b = b2 + else: + b = (b1 + b2)/2 + + num_changed_alphas += 1 + if num_changed_alphas == 0: + passes += 1 + else: + passes = 0 + + idx = alphas > 0 + model = {'X': X[idx, :], + 'y': Y[idx], + 'kernelFunction': kernelFunction, + 'b': b, + 'args': args, + 'alphas': alphas[idx], + 'w': np.dot(alphas * Y, X)} + return model + + +def svmPredict(model, X): + """ + Returns a vector of predictions using a trained SVM model. + + Parameters + ---------- + model : dict + The parameters of the trained svm model, as returned by the function svmTrain + + X : array_like + A (m x n) matrix where each example is a row. + + Returns + ------- + pred : array_like + A (m,) sized vector of predictions {0, 1} values. + """ + # check if we are getting a vector. If so, then assume we only need to do predictions + # for a single example + if X.ndim == 1: + X = X[np.newaxis, :] + + m = X.shape[0] + p = np.zeros(m) + pred = np.zeros(m) + + if model['kernelFunction'].__name__ == 'linearKernel': + # we can use the weights and bias directly if working with the linear kernel + p = np.dot(X, model['w']) + model['b'] + elif model['kernelFunction'].__name__ == 'gaussianKernel': + # vectorized RBF Kernel + # This is equivalent to computing the kernel on every pair of examples + X1 = np.sum(X**2, 1) + X2 = np.sum(model['X']**2, 1) + K = X2 + X1[:, None] - 2 * np.dot(X, model['X'].T) + + if len(model['args']) > 0: + K /= 2*model['args'][0]**2 + + K = np.exp(-K) + p = np.dot(K, model['alphas']*model['y']) + model['b'] + else: + # other non-linear kernel + for i in range(m): + predictions = 0 + for j in range(model['X'].shape[0]): + predictions += model['alphas'][j] * model['y'][j] \ + * model['kernelFunction'](X[i, :], model['X'][j, :]) + p[i] = predictions + + pred[p >= 0] = 1 + return pred + + +def linearKernel(x1, x2): + """ + Returns a linear kernel between x1 and x2. + + Parameters + ---------- + x1 : numpy ndarray + A 1-D vector. + + x2 : numpy ndarray + A 1-D vector of same size as x1. + + Returns + ------- + : float + The scalar amplitude. + """ + return np.dot(x1, x2) + + +def visualizeBoundaryLinear(X, y, model): + """ + Plots a linear decision boundary learned by the SVM. + + Parameters + ---------- + X : array_like + (m x 2) The training data with two features (to plot in a 2-D plane). + + y : array_like + (m, ) The data labels. + + model : dict + Dictionary of model variables learned by SVM. + """ + w, b = model['w'], model['b'] + xp = np.linspace(min(X[:, 0]), max(X[:, 0]), 100) + yp = -(w[0] * xp + b)/w[1] + + plotData(X, y) + pyplot.plot(xp, yp, '-b') + + +def visualizeBoundary(X, y, model): + """ + Plots a non-linear decision boundary learned by the SVM and overlays the data on it. + + Parameters + ---------- + X : array_like + (m x 2) The training data with two features (to plot in a 2-D plane). + + y : array_like + (m, ) The data labels. + + model : dict + Dictionary of model variables learned by SVM. + """ + plotData(X, y) + + # make classification predictions over a grid of values + x1plot = np.linspace(min(X[:, 0]), max(X[:, 0]), 100) + x2plot = np.linspace(min(X[:, 1]), max(X[:, 1]), 100) + X1, X2 = np.meshgrid(x1plot, x2plot) + + vals = np.zeros(X1.shape) + for i in range(X1.shape[1]): + this_X = np.stack((X1[:, i], X2[:, i]), axis=1) + vals[:, i] = svmPredict(model, this_X) + + pyplot.contour(X1, X2, vals, colors='y', linewidths=2) + pyplot.pcolormesh(X1, X2, vals, cmap='YlGnBu', alpha=0.25, edgecolors='None', lw=0) + pyplot.grid(False) + + +def getVocabList(): + """ + Reads the fixed vocabulary list in vocab.txt and returns a cell array of the words + % vocabList = GETVOCABLIST() reads the fixed vocabulary list in vocab.txt + % and returns a cell array of the words in vocabList. + + :return: + """ + vocabList = np.genfromtxt(join('Data', 'vocab.txt'), dtype=object) + return list(vocabList[:, 1].astype(str)) + + +class PorterStemmer: + """ + Porter Stemming Algorithm + + This is the Porter stemming algorithm, ported to Python from the + version coded up in ANSI C by the author. It may be be regarded + as canonical, in that it follows the algorithm presented in + + Porter, 1980, An algorithm for suffix stripping, Program, Vol. 14, + no. 3, pp 130-137, + + only differing from it at the points maked --DEPARTURE-- below. + + See also http://www.tartarus.org/~martin/PorterStemmer + + The algorithm as described in the paper could be exactly replicated + by adjusting the points of DEPARTURE, but this is barely necessary, + because (a) the points of DEPARTURE are definitely improvements, and + (b) no encoding of the Porter stemmer I have seen is anything like + as exact as this version, even with the points of DEPARTURE! + + Vivake Gupta (v@nano.com) + + Release 1: January 2001 + + Further adjustments by Santiago Bruno (bananabruno@gmail.com) + to allow word input not restricted to one word per line, leading + to: + + release 2: July 2008 + """ + def __init__(self): + """ + The main part of the stemming algorithm starts here. + b is a buffer holding a word to be stemmed. The letters are in b[k0], + b[k0+1] ... ending at b[k]. In fact k0 = 0 in this demo program. k is + readjusted downwards as the stemming progresses. Zero termination is + not in fact used in the algorithm. + + Note that only lower case sequences are stemmed. Forcing to lower case + should be done before stem(...) is called. + """ + self.b = "" # buffer for word to be stemmed + self.k = 0 + self.k0 = 0 + self.j = 0 # j is a general offset into the string + + def cons(self, i): + """cons(i) is TRUE <=> b[i] is a consonant.""" + if self.b[i] in 'aeiou': + return 0 + if self.b[i] == 'y': + if i == self.k0: + return 1 + else: + return not self.cons(i - 1) + return 1 + + def m(self): + """ + m() measures the number of consonant sequences between k0 and j. + if c is a consonant sequence and v a vowel sequence, and <..> + indicates arbitrary presence, + + gives 0 + vc gives 1 + vcvc gives 2 + vcvcvc gives 3 + .... + """ + n = 0 + i = self.k0 + while 1: + if i > self.j: + return n + if not self.cons(i): + break + i = i + 1 + i = i + 1 + while 1: + while 1: + if i > self.j: + return n + if self.cons(i): + break + i = i + 1 + i = i + 1 + n = n + 1 + while 1: + if i > self.j: + return n + if not self.cons(i): + break + i = i + 1 + i = i + 1 + + def vowelinstem(self): + """vowelinstem() is TRUE <=> k0,...j contains a vowel""" + for i in range(self.k0, self.j + 1): + if not self.cons(i): + return 1 + return 0 + + def doublec(self, j): + """ doublec(j) is TRUE <=> j,(j-1) contain a double consonant. """ + if j < (self.k0 + 1): + return 0 + if self.b[j] != self.b[j-1]: + return 0 + return self.cons(j) + + def cvc(self, i): + """ + cvc(i) is TRUE <=> i-2,i-1,i has the form consonant - vowel - consonant + and also if the second c is not w,x or y. this is used when trying to + restore an e at the end of a short e.g. + + cav(e), lov(e), hop(e), crim(e), but + snow, box, tray. + """ + if i < (self.k0 + 2) or not self.cons(i) or self.cons(i-1) or not self.cons(i-2): + return 0 + ch = self.b[i] + if ch in 'wxy': + return 0 + return 1 + + def ends(self, s): + """ends(s) is TRUE <=> k0,...k ends with the string s.""" + length = len(s) + if s[length - 1] != self.b[self.k]: # tiny speed-up + return 0 + if length > (self.k - self.k0 + 1): + return 0 + if self.b[self.k-length+1:self.k+1] != s: + return 0 + self.j = self.k - length + return 1 + + def setto(self, s): + """setto(s) sets (j+1),...k to the characters in the string s, readjusting k.""" + length = len(s) + self.b = self.b[:self.j+1] + s + self.b[self.j+length+1:] + self.k = self.j + length + + def r(self, s): + """r(s) is used further down.""" + if self.m() > 0: + self.setto(s) + + def step1ab(self): + """step1ab() gets rid of plurals and -ed or -ing. e.g. + + caresses -> caress + ponies -> poni + ties -> ti + caress -> caress + cats -> cat + + feed -> feed + agreed -> agree + disabled -> disable + + matting -> mat + mating -> mate + meeting -> meet + milling -> mill + messing -> mess + + meetings -> meet + """ + if self.b[self.k] == 's': + if self.ends("sses"): + self.k = self.k - 2 + elif self.ends("ies"): + self.setto("i") + elif self.b[self.k - 1] != 's': + self.k = self.k - 1 + if self.ends("eed"): + if self.m() > 0: + self.k = self.k - 1 + elif (self.ends("ed") or self.ends("ing")) and self.vowelinstem(): + self.k = self.j + if self.ends("at"): + self.setto("ate") + elif self.ends("bl"): + self.setto("ble") + elif self.ends("iz"): + self.setto("ize") + elif self.doublec(self.k): + self.k = self.k - 1 + ch = self.b[self.k] + if ch in 'lsz': + self.k += 1 + elif self.m() == 1 and self.cvc(self.k): + self.setto("e") + + def step1c(self): + """step1c() turns terminal y to i when there is another vowel in the stem.""" + if self.ends("y") and self.vowelinstem(): + self.b = self.b[:self.k] + 'i' + self.b[self.k+1:] + + def step2(self): + """step2() maps double suffices to single ones. + so -ization ( = -ize plus -ation) maps to -ize etc. note that the + string before the suffix must give m() > 0. + """ + if self.b[self.k - 1] == 'a': + if self.ends("ational"): self.r("ate") + elif self.ends("tional"): self.r("tion") + elif self.b[self.k - 1] == 'c': + if self.ends("enci"): self.r("ence") + elif self.ends("anci"): self.r("ance") + elif self.b[self.k - 1] == 'e': + if self.ends("izer"): self.r("ize") + elif self.b[self.k - 1] == 'l': + if self.ends("bli"): self.r("ble") # --DEPARTURE-- + # To match the published algorithm, replace this phrase with + # if self.ends("abli"): self.r("able") + elif self.ends("alli"): self.r("al") + elif self.ends("entli"): self.r("ent") + elif self.ends("eli"): self.r("e") + elif self.ends("ousli"): self.r("ous") + elif self.b[self.k - 1] == 'o': + if self.ends("ization"): self.r("ize") + elif self.ends("ation"): self.r("ate") + elif self.ends("ator"): self.r("ate") + elif self.b[self.k - 1] == 's': + if self.ends("alism"): self.r("al") + elif self.ends("iveness"): self.r("ive") + elif self.ends("fulness"): self.r("ful") + elif self.ends("ousness"): self.r("ous") + elif self.b[self.k - 1] == 't': + if self.ends("aliti"): self.r("al") + elif self.ends("iviti"): self.r("ive") + elif self.ends("biliti"): self.r("ble") + elif self.b[self.k - 1] == 'g': # --DEPARTURE-- + if self.ends("logi"): self.r("log") + # To match the published algorithm, delete this phrase + + def step3(self): + """step3() dels with -ic-, -full, -ness etc. similar strategy to step2.""" + if self.b[self.k] == 'e': + if self.ends("icate"): self.r("ic") + elif self.ends("ative"): self.r("") + elif self.ends("alize"): self.r("al") + elif self.b[self.k] == 'i': + if self.ends("iciti"): self.r("ic") + elif self.b[self.k] == 'l': + if self.ends("ical"): self.r("ic") + elif self.ends("ful"): self.r("") + elif self.b[self.k] == 's': + if self.ends("ness"): self.r("") + + def step4(self): + """step4() takes off -ant, -ence etc., in context vcvc.""" + if self.b[self.k - 1] == 'a': + if self.ends("al"): pass + else: return + elif self.b[self.k - 1] == 'c': + if self.ends("ance"): pass + elif self.ends("ence"): pass + else: return + elif self.b[self.k - 1] == 'e': + if self.ends("er"): pass + else: return + elif self.b[self.k - 1] == 'i': + if self.ends("ic"): pass + else: return + elif self.b[self.k - 1] == 'l': + if self.ends("able"): pass + elif self.ends("ible"): pass + else: return + elif self.b[self.k - 1] == 'n': + if self.ends("ant"): pass + elif self.ends("ement"): pass + elif self.ends("ment"): pass + elif self.ends("ent"): pass + else: return + elif self.b[self.k - 1] == 'o': + if self.ends("ion") and (self.b[self.j] == 's' or self.b[self.j] == 't'): pass + elif self.ends("ou"): pass + # takes care of -ous + else: return + elif self.b[self.k - 1] == 's': + if self.ends("ism"): pass + else: return + elif self.b[self.k - 1] == 't': + if self.ends("ate"): pass + elif self.ends("iti"): pass + else: return + elif self.b[self.k - 1] == 'u': + if self.ends("ous"): pass + else: return + elif self.b[self.k - 1] == 'v': + if self.ends("ive"): pass + else: return + elif self.b[self.k - 1] == 'z': + if self.ends("ize"): pass + else: return + else: + return + if self.m() > 1: + self.k = self.j + + def step5(self): + """step5() removes a final -e if m() > 1, and changes -ll to -l if + m() > 1. + """ + self.j = self.k + if self.b[self.k] == 'e': + a = self.m() + if a > 1 or (a == 1 and not self.cvc(self.k-1)): + self.k = self.k - 1 + if self.b[self.k] == 'l' and self.doublec(self.k) and self.m() > 1: + self.k = self.k -1 + + def stem(self, p, i=0, j=None): + """In stem(p,i,j), p is a char pointer, and the string to be stemmed + is from p[i] to p[j] inclusive. Typically i is zero and j is the + offset to the last character of a string, (p[j+1] == '\0'). The + stemmer adjusts the characters p[i] ... p[j] and returns the new + end-point of the string, k. Stemming never increases word length, so + i <= k <= j. To turn the stemmer into a module, declare 'stem' as + extern, and delete the remainder of this file. + """ + # copy the parameters into statics + self.b = p + self.k = j or len(p) - 1 + self.k0 = i + if self.k <= self.k0 + 1: + return self.b # --DEPARTURE-- + + # With this line, strings of length 1 or 2 don't go through the + # stemming process, although no mention is made of this in the + # published algorithm. Remove the line to match the published + # algorithm. + + self.step1ab() + self.step1c() + self.step2() + self.step3() + self.step4() + self.step5() + return self.b[self.k0:self.k+1] + + +class Grader(SubmissionBase): + # Random Test Cases + x1 = np.sin(np.arange(1, 11)) + x2 = np.cos(np.arange(1, 11)) + ec = 'the quick brown fox jumped over the lazy dog' + wi = np.abs(np.round(x1 * 1863)).astype(int) + wi = np.concatenate([wi, wi]) + + def __init__(self): + part_names = ['Gaussian Kernel', + 'Parameters (C, sigma) for Dataset 3', + 'Email Processing', + 'Email Feature Extraction'] + super().__init__('support-vector-machines', part_names) + + def __iter__(self): + for part_id in range(1, 5): + try: + func = self.functions[part_id] + # Each part has different expected arguments/different function + if part_id == 1: + res = func(self.x1, self.x2, 2) + elif part_id == 2: + res = np.hstack(func()).tolist() + elif part_id == 3: + # add one to be compatible with matlab grader + res = [ind+1 for ind in func(self.ec, False)] + elif part_id == 4: + res = func(self.wi) + else: + raise KeyError + yield part_id, res + except KeyError: + yield part_id, 0 diff --git a/Exercise7/Data/bird_small.mat b/Exercise7/Data/bird_small.mat new file mode 100755 index 0000000..04c224c Binary files /dev/null and b/Exercise7/Data/bird_small.mat differ diff --git a/Exercise7/Data/bird_small.png b/Exercise7/Data/bird_small.png new file mode 100755 index 0000000..a3cd00c Binary files /dev/null and b/Exercise7/Data/bird_small.png differ diff --git a/Exercise7/Data/ex7data1.mat b/Exercise7/Data/ex7data1.mat new file mode 100755 index 0000000..f9c3961 Binary files /dev/null and b/Exercise7/Data/ex7data1.mat differ diff --git a/Exercise7/Data/ex7data2.mat b/Exercise7/Data/ex7data2.mat new file mode 100755 index 0000000..de3f5b9 Binary files /dev/null and b/Exercise7/Data/ex7data2.mat differ diff --git a/Exercise7/Data/ex7faces.mat b/Exercise7/Data/ex7faces.mat new file mode 100755 index 0000000..3965bd1 Binary files /dev/null and b/Exercise7/Data/ex7faces.mat differ diff --git a/Exercise7/Figures/bird_compression.png b/Exercise7/Figures/bird_compression.png new file mode 100755 index 0000000..3f1a60d Binary files /dev/null and b/Exercise7/Figures/bird_compression.png differ diff --git a/Exercise7/Figures/faces.png b/Exercise7/Figures/faces.png new file mode 100755 index 0000000..de33f11 Binary files /dev/null and b/Exercise7/Figures/faces.png differ diff --git a/Exercise7/Figures/faces_original.png b/Exercise7/Figures/faces_original.png new file mode 100755 index 0000000..6bbcdde Binary files /dev/null and b/Exercise7/Figures/faces_original.png differ diff --git a/Exercise7/Figures/faces_reconstructed.png b/Exercise7/Figures/faces_reconstructed.png new file mode 100755 index 0000000..603f75c Binary files /dev/null and b/Exercise7/Figures/faces_reconstructed.png differ diff --git a/Exercise7/Figures/kmeans_result.png b/Exercise7/Figures/kmeans_result.png new file mode 100755 index 0000000..1ca9107 Binary files /dev/null and b/Exercise7/Figures/kmeans_result.png differ diff --git a/Exercise7/Figures/pca_components.png b/Exercise7/Figures/pca_components.png new file mode 100755 index 0000000..d47078e Binary files /dev/null and b/Exercise7/Figures/pca_components.png differ diff --git a/Exercise7/Figures/pca_reconstruction.png b/Exercise7/Figures/pca_reconstruction.png new file mode 100755 index 0000000..f7cd238 Binary files /dev/null and b/Exercise7/Figures/pca_reconstruction.png differ diff --git a/Exercise7/exercise7.ipynb b/Exercise7/exercise7.ipynb new file mode 100755 index 0000000..0668112 --- /dev/null +++ b/Exercise7/exercise7.ipynb @@ -0,0 +1,1194 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Programming Exercise 7:\n", + "# K-means Clustering and Principal Component Analysis\n", + "\n", + "## Introduction\n", + "\n", + "In this exercise, you will implement the K-means clustering algorithm and apply it to compress an image. In the second part, you will use principal component analysis to find a low-dimensional representation of face images. Before starting on the programming exercise, we strongly recommend watching the video lectures and completing the review questions for the associated topics.\n", + "\n", + "All the information you need for solving this assignment is in this notebook, and all the code you will be implementing will take place within this notebook. The assignment can be promptly submitted to the coursera grader directly from this notebook (code and instructions are included below).\n", + "\n", + "Before we begin with the exercises, we need to import all libraries required for this programming exercise. Throughout the course, we will be using [`numpy`](http://www.numpy.org/) for all arrays and matrix operations, [`matplotlib`](https://matplotlib.org/) for plotting, and [`scipy`](https://docs.scipy.org/doc/scipy/reference/) for scientific and numerical computation functions and tools. You can find instructions on how to install required libraries in the README file in the [github repository](https://github.com/dibgerge/ml-coursera-python-assignments)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# used for manipulating directory paths\n", + "import os\n", + "\n", + "# Scientific and vector computation for python\n", + "import numpy as np\n", + "\n", + "# Import regular expressions to process emails\n", + "import re\n", + "\n", + "# Plotting library\n", + "from matplotlib import pyplot\n", + "from mpl_toolkits.mplot3d import Axes3D\n", + "import matplotlib as mpl\n", + "\n", + "from IPython.display import HTML, display, clear_output\n", + "\n", + "try:\n", + " pyplot.rcParams[\"animation.html\"] = \"jshtml\"\n", + "except ValueError:\n", + " pyplot.rcParams[\"animation.html\"] = \"html5\"\n", + "\n", + "# Optimization module in scipy\n", + "from scipy import optimize\n", + "\n", + "# will be used to load MATLAB mat datafile format\n", + "from scipy.io import loadmat\n", + "\n", + "# library written for this exercise providing additional functions for assignment submission, and others\n", + "import utils\n", + "\n", + "%load_ext autoreload \n", + "%autoreload 2\n", + "\n", + "# define the submission/grader object for this exercise\n", + "grader = utils.Grader()\n", + "\n", + "# tells matplotlib to embed plots within the notebook\n", + "%matplotlib inline" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Submission and Grading\n", + "\n", + "\n", + "After completing each part of the assignment, be sure to submit your solutions to the grader. The following is a breakdown of how each part of this exercise is scored.\n", + "\n", + "\n", + "| Section | Part | Submitted Function | Points |\n", + "| :- |:- |:- | :-: |\n", + "| 1 | [Find Closest Centroids](#section1) | [`findClosestCentroids`](#findClosestCentroids) | 30 |\n", + "| 2 | [Computed Centroid Means](#section2) | [`computeCentroids`](#computeCentroids) | 30 |\n", + "| 3 | [PCA](#section3) | [`pca`](#pca) | 20 |\n", + "| 4 | [Project Data](#section4) | [`projectData`](#projectData) | 10 |\n", + "| 5 | [Recover Data](#section5) | [`recoverData`](#recoverData) | 10 |\n", + "| | Total Points | |100 |\n", + "\n", + "\n", + "You are allowed to submit your solutions multiple times, and we will take only the highest score into consideration.\n", + "\n", + "
\n", + "At the end of each section in this notebook, we have a cell which contains code for submitting the solutions thus far to the grader. Execute the cell to see your score up to the current section. For all your work to be submitted properly, you must execute those cells at least once.\n", + "
" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 1 K-means Clustering\n", + "\n", + "In this exercise, you will implement K-means algorithm and use it for image compression. You will first start on an example 2D dataset that will help you gain an intuition of how the K-means algorithm works. After\n", + "that, you wil use the K-means algorithm for image compression by reducing the number of colors that occur in an image to only those that are most common in that image.\n", + "\n", + "### 1.1 Implementing K-means\n", + "\n", + "The K-means algorithm is a method to automatically cluster similar data examples together. Concretely, you are given a training set $\\{x^{(1)} , \\cdots, x^{(m)}\\}$ (where $x^{(i)} \\in \\mathbb{R}^n$), and want to group the data into a few cohesive “clusters”. The intuition behind K-means is an iterative procedure that starts by guessing the initial centroids, and then refines this guess by repeatedly assigning examples to their closest centroids and then recomputing the centroids based on the assignments.\n", + "\n", + "The K-means algorithm is as follows:\n", + "\n", + "```python\n", + "centroids = kMeansInitCentroids(X, K)\n", + "for i in range(iterations):\n", + " # Cluster assignment step: Assign each data point to the\n", + " # closest centroid. idx[i] corresponds to cˆ(i), the index\n", + " # of the centroid assigned to example i\n", + " idx = findClosestCentroids(X, centroids)\n", + " \n", + " # Move centroid step: Compute means based on centroid\n", + " # assignments\n", + " centroids = computeMeans(X, idx, K)\n", + "```\n", + "\n", + "The inner-loop of the algorithm repeatedly carries out two steps: (1) Assigning each training example $x^{(i)}$ to its closest centroid, and (2) Recomputing the mean of each centroid using the points assigned to it. The K-means algorithm will always converge to some final set of means for the centroids. Note that the converged solution may not always be ideal and depends on the initial setting of the centroids. Therefore, in practice the K-means algorithm is usually run a few times with different random initializations. One way to choose between these different solutions from different random initializations is to choose the one with the lowest cost function value (distortion). You will implement the two phases of the K-means algorithm separately\n", + "in the next sections." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "#### 1.1.1 Finding closest centroids\n", + "\n", + "In the “cluster assignment” phase of the K-means algorithm, the algorithm assigns every training example $x^{(i)}$ to its closest centroid, given the current positions of centroids. Specifically, for every example $i$ we set\n", + "\n", + "$$c^{(i)} := j \\quad \\text{that minimizes} \\quad \\lvert\\rvert x^{(i)} - \\mu_j \\lvert\\rvert^2, $$\n", + "\n", + "where $c^{(i)}$ is the index of the centroid that is closest to $x^{(i)}$, and $\\mu_j$ is the position (value) of the $j^{th}$ centroid. Note that $c^{(i)}$ corresponds to `idx[i]` in the starter code.\n", + "\n", + "Your task is to complete the code in the function `findClosestCentroids`. This function takes the data matrix `X` and the locations of all centroids inside `centroids` and should output a one-dimensional array `idx` that holds the index (a value in $\\{1, ..., K\\}$, where $K$ is total number of centroids) of the closest centroid to every training example.\n", + "\n", + "You can implement this using a loop over every training example and every centroid.\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def findClosestCentroids(X, centroids):\n", + " \"\"\"\n", + " Computes the centroid memberships for every example.\n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like\n", + " The dataset of size (m, n) where each row is a single example. \n", + " That is, we have m examples each of n dimensions.\n", + " \n", + " centroids : array_like\n", + " The k-means centroids of size (K, n). K is the number\n", + " of clusters, and n is the the data dimension.\n", + " \n", + " Returns\n", + " -------\n", + " idx : array_like\n", + " A vector of size (m, ) which holds the centroids assignment for each\n", + " example (row) in the dataset X.\n", + " \n", + " Instructions\n", + " ------------\n", + " Go over every example, find its closest centroid, and store\n", + " the index inside `idx` at the appropriate location.\n", + " Concretely, idx[i] should contain the index of the centroid\n", + " closest to example i. Hence, it should be a value in the \n", + " range 0..K-1\n", + "\n", + " Note\n", + " ----\n", + " You can use a for-loop over the examples to compute this.\n", + " \"\"\"\n", + " # Set K\n", + " K = centroids.shape[0]\n", + "\n", + " # You need to return the following variables correctly.\n", + " idx = np.zeros(X.shape[0], dtype=int)\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + " \n", + " \n", + " # =============================================================\n", + " return idx" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Once you have completed the code in `findClosestCentroids`, the following cell will run your code and you should see the output `[0 2 1]` corresponding to the centroid assignments for the first 3 examples." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Load an example dataset that we will be using\n", + "data = loadmat(os.path.join('Data', 'ex7data2.mat'))\n", + "X = data['X']\n", + "\n", + "# Select an initial set of centroids\n", + "K = 3 # 3 Centroids\n", + "initial_centroids = np.array([[3, 3], [6, 2], [8, 5]])\n", + "\n", + "# Find the closest centroids for the examples using the initial_centroids\n", + "idx = findClosestCentroids(X, initial_centroids)\n", + "\n", + "print('Closest centroids for the first 3 examples:')\n", + "print(idx[:3])\n", + "print('(the closest centroids should be 0, 2, 1 respectively)')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[1] = findClosestCentroids\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 1.1.2 Computing centroid means\n", + "\n", + "Given assignments of every point to a centroid, the second phase of the algorithm recomputes, for each centroid, the mean of the points that were assigned to it. Specifically, for every centroid $k$ we set\n", + "\n", + "$$ \\mu_k := \\frac{1}{\\left| C_k\\right|} \\sum_{i \\in C_k} x^{(i)}$$\n", + "\n", + "where $C_k$ is the set of examples that are assigned to centroid $k$. Concretely, if two examples say $x^{(3)}$ and $x^{(5)}$ are assigned to centroid $k = 2$, then you should update $\\mu_2 = \\frac{1}{2} \\left( x^{(3)} + x^{(5)} \\right)$.\n", + "\n", + "You should now complete the code in the function `computeCentroids`. You can implement this function using a loop over the centroids. You can also use a loop over the examples; but if you can use a vectorized implementation that does not use such a loop, your code may run faster.\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def computeCentroids(X, idx, K):\n", + " \"\"\"\n", + " Returns the new centroids by computing the means of the data points\n", + " assigned to each centroid.\n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like\n", + " The datset where each row is a single data point. That is, it \n", + " is a matrix of size (m, n) where there are m datapoints each\n", + " having n dimensions. \n", + " \n", + " idx : array_like \n", + " A vector (size m) of centroid assignments (i.e. each entry in range [0 ... K-1])\n", + " for each example.\n", + " \n", + " K : int\n", + " Number of clusters\n", + " \n", + " Returns\n", + " -------\n", + " centroids : array_like\n", + " A matrix of size (K, n) where each row is the mean of the data \n", + " points assigned to it.\n", + " \n", + " Instructions\n", + " ------------\n", + " Go over every centroid and compute mean of all points that\n", + " belong to it. Concretely, the row vector centroids[i, :]\n", + " should contain the mean of the data points assigned to\n", + " cluster i.\n", + "\n", + " Note:\n", + " -----\n", + " You can use a for-loop over the centroids to compute this.\n", + " \"\"\"\n", + " # Useful variables\n", + " m, n = X.shape\n", + " # You need to return the following variables correctly.\n", + " centroids = np.zeros((K, n))\n", + "\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + " \n", + " \n", + " # =============================================================\n", + " return centroids" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Once you have completed the code in `computeCentroids`, the following cell will run your code and output the centroids after the first step of K-means." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Compute means based on the closest centroids found in the previous part.\n", + "centroids = computeCentroids(X, idx, K)\n", + "\n", + "print('Centroids computed after initial finding of closest centroids:')\n", + "print(centroids)\n", + "print('\\nThe centroids should be')\n", + "print(' [ 2.428301 3.157924 ]')\n", + "print(' [ 5.813503 2.633656 ]')\n", + "print(' [ 7.119387 3.616684 ]')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[2] = computeCentroids\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 1.2 K-means on example dataset \n", + "\n", + "After you have completed the two functions (`findClosestCentroids` and `computeCentroids`), you have all the necessary pieces to run the K-means algorithm. The next cell will run the K-means algorithm on a toy 2D dataset to help you understand how K-means works. Your functions are called from inside the `runKmeans` function (in this assignment's `utils.py` module). We encourage you to take a look at the function to understand how it works. Notice that the code calls the two functions you implemented in a loop.\n", + "\n", + "When you run the next step, the K-means code will produce an animation that steps you through the progress of the algorithm at each iteration. At the end, your figure should look as the one displayed below.\n", + "\n", + "![](Figures/kmeans_result.png)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "scrolled": false + }, + "outputs": [], + "source": [ + "# Load an example dataset\n", + "data = loadmat(os.path.join('Data', 'ex7data2.mat'))\n", + "\n", + "# Settings for running K-Means\n", + "K = 3\n", + "max_iters = 10\n", + "\n", + "# For consistency, here we set centroids to specific values\n", + "# but in practice you want to generate them automatically, such as by\n", + "# settings them to be random examples (as can be seen in\n", + "# kMeansInitCentroids).\n", + "initial_centroids = np.array([[3, 3], [6, 2], [8, 5]])\n", + "\n", + "\n", + "# Run K-Means algorithm. The 'true' at the end tells our function to plot\n", + "# the progress of K-Means\n", + "centroids, idx, anim = utils.runkMeans(X, initial_centroids,\n", + " findClosestCentroids, computeCentroids, max_iters, True)\n", + "anim" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 1.3 Random initialization \n", + "\n", + "The initial assignments of centroids for the example dataset in the previous cell were designed so that you will see the same figure as that shown in the cell above. In practice, a\n", + "good strategy for initializing the centroids is to select random examples from the training set.\n", + "\n", + "In this part of the exercise, you should complete the function `kMeansInitCentroids` with the following code:\n", + "\n", + "```python\n", + "# Initialize the centroids to be random examples\n", + "\n", + "# Randomly reorder the indices of examples\n", + "randidx = np.random.permutation(X.shape[0])\n", + "# Take the first K examples as centroids\n", + "centroids = X[randidx[:K], :]\n", + "```\n", + "\n", + "The code above first randomly permutes the indices of the examples (using `permute` within the `numpy.random` module). Then, it selects the first $K$ examples based on the random permutation of the indices. This allows the examples to be selected at random without the risk of selecting the same example twice.\n", + "\n", + "*You do not need to make any submission for this part of the exercise*\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def kMeansInitCentroids(X, K):\n", + " \"\"\"\n", + " This function initializes K centroids that are to be used in K-means on the dataset x.\n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like \n", + " The dataset of size (m x n).\n", + " \n", + " K : int\n", + " The number of clusters.\n", + " \n", + " Returns\n", + " -------\n", + " centroids : array_like\n", + " Centroids of the clusters. This is a matrix of size (K x n).\n", + " \n", + " Instructions\n", + " ------------\n", + " You should set centroids to randomly chosen examples from the dataset X.\n", + " \"\"\"\n", + " m, n = X.shape\n", + " \n", + " # You should return this values correctly\n", + " centroids = np.zeros((K, n))\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + "\n", + " \n", + " # =============================================================\n", + " return centroids" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 1.4 Image compression with K-means\n", + "\n", + "In this exercise, you will apply K-means to image compression. We will use the image below as an example (property of Frank Wouters with permission to this class).\n", + "\n", + "![](Data/bird_small.png)\n", + "\n", + "In a straightforward 24-bit color representation of an image, each pixel is represented as three 8-bit unsigned integers (ranging from 0 to 255) that specify the red, green and blue intensity values. This encoding is often referred to as the RGB encoding. Our image contains thousands of colors, and in this part of the exercise, you will reduce the number of colors to 16 colors.\n", + "\n", + "By making this reduction, it is possible to represent (compress) the photo in an efficient way. Specifically, you only need to store the RGB values of the 16 selected colors, and for each pixel in the image you now need to only store the index of the color at that location (where only 4 bits are necessary to represent 16 possibilities).\n", + "\n", + "In this exercise, you will use the K-means algorithm to select the 16 colors that will be used to represent the compressed image. Concretely, you will treat every pixel in the original image as a data example and use the K-means algorithm to find the 16 colors that best group (cluster) the pixels in the 3-dimensional RGB space. Once you have computed the cluster centroids on the image, you will then use the 16 colors to replace the pixels in the original image.\n", + "\n", + "#### 1.4.1 K-means on pixels\n", + "\n", + "In python, images can be read in as follows:\n", + "\n", + "```python\n", + "# Load 128x128 color image (bird_small.png)\n", + "img = mpl.image.imread(os.path.join('Data', 'bird_small.png'))\n", + "\n", + "# We have already imported matplotlib as mpl at the beginning of this notebook.\n", + "```\n", + "This creates a three-dimensional matrix `A` whose first two indices identify a pixel position and whose last index represents red, green, or blue. For example, A[50, 33, 2] gives the blue intensity of the pixel at row 51 and column 34.\n", + "\n", + "The code in the following cell first loads the image, and then reshapes it to create an m x 3 matrix of pixel colors (where m = 16384 = 128 x 128), and calls your K-means function on it.\n", + "\n", + "After finding the top K = 16 colors to represent the image, you can now assign each pixel position to its closest centroid using the `findClosestCentroids` function. This allows you to represent the original image using the centroid assignments of each pixel. Notice that you have significantly reduced the number of bits that are required to describe the image. The original image required 24 bits for each one of the 128 x 128 pixel locations, resulting in total size of 128 x 128 x 24 = 393,216 bits. The new representation requires some overhead storage in form of a dictionary of 16 colors, each of which require 24 bits, but the image itself then only requires 4 bits per pixel location. The final number of bits used is therefore 16 x 24 + 128 x 128 x 4 = 65,920 bits, which corresponds to compressing the original image by about a factor of 6.\n", + "\n", + "Finally, you can view the effects of the compression by reconstructing the image based only on the centroid assignments. Specifically, you can replace each pixel location with the mean of the centroid assigned to it. The figure below shows the reconstruction we obtained. \n", + "\n", + "![](Figures/bird_compression.png)\n", + "\n", + "Even though the resulting image retains most of the characteristics of the original, we also see some compression artifacts.\n", + "\n", + "Run the following cell to compute the centroids and the centroid allocation of each pixel in the image." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# ======= Experiment with these parameters ================\n", + "# You should try different values for those parameters\n", + "K = 16\n", + "max_iters = 10\n", + "\n", + "# Load an image of a bird\n", + "# Change the file name and path to experiment with your own images\n", + "A = mpl.image.imread(os.path.join('Data', 'bird_small.png'))\n", + "# ==========================================================\n", + "\n", + "# Divide by 255 so that all values are in the range 0 - 1\n", + "A /= 255\n", + "\n", + "# Reshape the image into an Nx3 matrix where N = number of pixels.\n", + "# Each row will contain the Red, Green and Blue pixel values\n", + "# This gives us our dataset matrix X that we will use K-Means on.\n", + "X = A.reshape(-1, 3)\n", + "\n", + "# When using K-Means, it is important to randomly initialize centroids\n", + "# You should complete the code in kMeansInitCentroids above before proceeding\n", + "initial_centroids = kMeansInitCentroids(X, K)\n", + "\n", + "# Run K-Means\n", + "centroids, idx = utils.runkMeans(X, initial_centroids,\n", + " findClosestCentroids,\n", + " computeCentroids,\n", + " max_iters)\n", + "\n", + "# We can now recover the image from the indices (idx) by mapping each pixel\n", + "# (specified by its index in idx) to the centroid value\n", + "# Reshape the recovered image into proper dimensions\n", + "X_recovered = centroids[idx, :].reshape(A.shape)\n", + "\n", + "# Display the original image, rescale back by 255\n", + "fig, ax = pyplot.subplots(1, 2, figsize=(8, 4))\n", + "ax[0].imshow(A*255)\n", + "ax[0].set_title('Original')\n", + "ax[0].grid(False)\n", + "\n", + "# Display compressed image, rescale back by 255\n", + "ax[1].imshow(X_recovered*255)\n", + "ax[1].set_title('Compressed, with %d colors' % K)\n", + "ax[1].grid(False)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You do not need to make any submissions for this part of the exercise.*" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 1.5 Optional (ungraded) exercise: Use your own image\n", + "\n", + "In this exercise, modify the code we have supplied in the previous cell to run on one of your own images. Note that if your image is very large, then K-means can take a long time to run. Therefore, we recommend that you resize your images to\n", + "manageable sizes before running the code. You can also try to vary $K$ to see the effects on the compression." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 2 Principal Component Analysis\n", + "\n", + "In this exercise, you will use principal component analysis (PCA) to perform dimensionality reduction. You will first experiment with an example 2D dataset to get intuition on how PCA works, and then use it on a bigger dataset of 5000 face image dataset.\n", + "\n", + "### 2.1 Example Dataset\n", + "\n", + "To help you understand how PCA works, you will first start with a 2D dataset which has one direction of large variation and one of smaller variation. The cell below will plot the training data, also shown in here:\n", + "\n", + "In this part of the exercise, you will visualize what happens when you use PCA to reduce the data from 2D to 1D. In practice, you might want to reduce data from 256 to 50 dimensions, say; but using lower dimensional data in this example allows us to visualize the algorithms better." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Load the dataset into the variable X \n", + "data = loadmat(os.path.join('Data', 'ex7data1.mat'))\n", + "X = data['X']\n", + "\n", + "# Visualize the example dataset\n", + "pyplot.plot(X[:, 0], X[:, 1], 'bo', ms=10, mec='k', mew=1)\n", + "pyplot.axis([0.5, 6.5, 2, 8])\n", + "pyplot.gca().set_aspect('equal')\n", + "pyplot.grid(False)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 2.2 Implementing PCA\n", + "\n", + "In this part of the exercise, you will implement PCA. PCA consists of two computational steps: \n", + "\n", + "1. Compute the covariance matrix of the data.\n", + "2. Use SVD (in python we use numpy's implementation `np.linalg.svd`) to compute the eigenvectors $U_1$, $U_2$, $\\dots$, $U_n$. These will correspond to the principal components of variation in the data.\n", + "\n", + "First, you should compute the covariance matrix of the data, which is given by:\n", + "\n", + "$$ \\Sigma = \\frac{1}{m} X^T X$$\n", + "\n", + "where $X$ is the data matrix with examples in rows, and $m$ is the number of examples. Note that $\\Sigma$ is a $n \\times n$ matrix and not the summation operator. \n", + "\n", + "After computing the covariance matrix, you can run SVD on it to compute the principal components. In python and `numpy` (or `scipy`), you can run SVD with the following command: `U, S, V = np.linalg.svd(Sigma)`, where `U` will contain the principal components and `S` will contain a diagonal matrix. Note that the `scipy` library also has a similar function to compute SVD `scipy.linalg.svd`. The functions in the two libraries use the same C-based library (LAPACK) for the SVD computation, but the `scipy` version provides more options and arguments to control SVD computation. In this exercise, we will stick with the `numpy` implementation of SVD.\n", + "\n", + "Complete the code in the following cell to implemente PCA.\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def pca(X):\n", + " \"\"\"\n", + " Run principal component analysis.\n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like\n", + " The dataset to be used for computing PCA. It has dimensions (m x n)\n", + " where m is the number of examples (observations) and n is \n", + " the number of features.\n", + " \n", + " Returns\n", + " -------\n", + " U : array_like\n", + " The eigenvectors, representing the computed principal components\n", + " of X. U has dimensions (n x n) where each column is a single \n", + " principal component.\n", + " \n", + " S : array_like\n", + " A vector of size n, contaning the singular values for each\n", + " principal component. Note this is the diagonal of the matrix we \n", + " mentioned in class.\n", + " \n", + " Instructions\n", + " ------------\n", + " You should first compute the covariance matrix. Then, you\n", + " should use the \"svd\" function to compute the eigenvectors\n", + " and eigenvalues of the covariance matrix. \n", + "\n", + " Notes\n", + " -----\n", + " When computing the covariance matrix, remember to divide by m (the\n", + " number of examples).\n", + " \"\"\"\n", + " # Useful values\n", + " m, n = X.shape\n", + "\n", + " # You need to return the following variables correctly.\n", + " U = np.zeros(n)\n", + " S = np.zeros(n)\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + " \n", + " \n", + " # ============================================================\n", + " return U, S" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Before using PCA, it is important to first normalize the data by subtracting the mean value of each feature from the dataset, and scaling each dimension so that they are in the same range.\n", + "\n", + "In the next cell, this normalization will be performed for you using the `utils.featureNormalize` function.\n", + "After normalizing the data, you can run PCA to compute the principal components. Your task is to complete the code in the function `pca` to compute the principal components of the dataset. \n", + "\n", + "Once you have completed the function `pca`, the following cell will run PCA on the example dataset and plot the corresponding principal components found similar to the figure below. \n", + "\n", + "![](Figures/pca_components.png)\n", + "\n", + "\n", + "The following cell will also output the top principal component (eigenvector) found, and you should expect to see an output of about `[-0.707 -0.707]`. (It is possible that `numpy` may instead output the negative of this, since $U_1$ and $-U_1$ are equally valid choices for the first principal component.)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Before running PCA, it is important to first normalize X\n", + "X_norm, mu, sigma = utils.featureNormalize(X)\n", + "\n", + "# Run PCA\n", + "U, S = pca(X_norm)\n", + "\n", + "# Draw the eigenvectors centered at mean of data. These lines show the\n", + "# directions of maximum variations in the dataset.\n", + "fig, ax = pyplot.subplots()\n", + "ax.plot(X[:, 0], X[:, 1], 'bo', ms=10, mec='k', mew=0.25)\n", + "\n", + "for i in range(2):\n", + " ax.arrow(mu[0], mu[1], 1.5 * S[i]*U[0, i], 1.5 * S[i]*U[1, i],\n", + " head_width=0.25, head_length=0.2, fc='k', ec='k', lw=2, zorder=1000)\n", + "\n", + "ax.axis([0.5, 6.5, 2, 8])\n", + "ax.set_aspect('equal')\n", + "ax.grid(False)\n", + "\n", + "print('Top eigenvector: U[:, 0] = [{:.6f} {:.6f}]'.format(U[0, 0], U[1, 0]))\n", + "print(' (you should expect to see [-0.707107 -0.707107])')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[3] = pca\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2.3 Dimensionality Reduction with PCA\n", + "\n", + "After computing the principal components, you can use them to reduce the feature dimension of your dataset by projecting each example onto a lower dimensional space, $x^{(i)} \\rightarrow z^{(i)}$ (e.g., projecting the data from 2D to 1D). In this part of the exercise, you will use the eigenvectors returned by PCA and\n", + "project the example dataset into a 1-dimensional space. In practice, if you were using a learning algorithm such as linear regression or perhaps neural networks, you could now use the projected data instead of the original data. By using the projected data, you can train your model faster as there are less dimensions in the input.\n", + "\n", + "\n", + "\n", + "#### 2.3.1 Projecting the data onto the principal components\n", + "\n", + "You should now complete the code in the function `projectData`. Specifically, you are given a dataset `X`, the principal components `U`, and the desired number of dimensions to reduce to `K`. You should project each example in `X` onto the top `K` components in `U`. Note that the top `K` components in `U` are given by\n", + "the first `K` columns of `U`, that is `Ureduce = U[:, :K]`.\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def projectData(X, U, K):\n", + " \"\"\"\n", + " Computes the reduced data representation when projecting only \n", + " on to the top K eigenvectors.\n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like\n", + " The input dataset of shape (m x n). The dataset is assumed to be \n", + " normalized.\n", + " \n", + " U : array_like\n", + " The computed eigenvectors using PCA. This is a matrix of \n", + " shape (n x n). Each column in the matrix represents a single\n", + " eigenvector (or a single principal component).\n", + " \n", + " K : int\n", + " Number of dimensions to project onto. Must be smaller than n.\n", + " \n", + " Returns\n", + " -------\n", + " Z : array_like\n", + " The projects of the dataset onto the top K eigenvectors. \n", + " This will be a matrix of shape (m x k).\n", + " \n", + " Instructions\n", + " ------------\n", + " Compute the projection of the data using only the top K \n", + " eigenvectors in U (first K columns). \n", + " For the i-th example X[i,:], the projection on to the k-th \n", + " eigenvector is given as follows:\n", + " \n", + " x = X[i, :]\n", + " projection_k = np.dot(x, U[:, k])\n", + "\n", + " \"\"\"\n", + " # You need to return the following variables correctly.\n", + " Z = np.zeros((X.shape[0], K))\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + "\n", + " \n", + " # =============================================================\n", + " return Z" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Once you have completed the code in `projectData`, the following cell will project the first example onto the first dimension and you should see a value of about 1.481 (or possibly -1.481, if you got $-U_1$ instead of $U_1$)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Project the data onto K = 1 dimension\n", + "K = 1\n", + "Z = projectData(X_norm, U, K)\n", + "print('Projection of the first example: {:.6f}'.format(Z[0, 0]))\n", + "print('(this value should be about : 1.481274)')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[4] = projectData\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "#### 2.3.2 Reconstructing an approximation of the data\n", + "\n", + "After projecting the data onto the lower dimensional space, you can approximately recover the data by projecting them back onto the original high dimensional space. Your task is to complete the function `recoverData` to project each example in `Z` back onto the original space and return the recovered approximation in `Xrec`.\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def recoverData(Z, U, K):\n", + " \"\"\"\n", + " Recovers an approximation of the original data when using the \n", + " projected data.\n", + " \n", + " Parameters\n", + " ----------\n", + " Z : array_like\n", + " The reduced data after applying PCA. This is a matrix\n", + " of shape (m x K).\n", + " \n", + " U : array_like\n", + " The eigenvectors (principal components) computed by PCA.\n", + " This is a matrix of shape (n x n) where each column represents\n", + " a single eigenvector.\n", + " \n", + " K : int\n", + " The number of principal components retained\n", + " (should be less than n).\n", + " \n", + " Returns\n", + " -------\n", + " X_rec : array_like\n", + " The recovered data after transformation back to the original \n", + " dataset space. This is a matrix of shape (m x n), where m is \n", + " the number of examples and n is the dimensions (number of\n", + " features) of original datatset.\n", + " \n", + " Instructions\n", + " ------------\n", + " Compute the approximation of the data by projecting back\n", + " onto the original space using the top K eigenvectors in U.\n", + " For the i-th example Z[i,:], the (approximate)\n", + " recovered data for dimension j is given as follows:\n", + "\n", + " v = Z[i, :]\n", + " recovered_j = np.dot(v, U[j, :K])\n", + "\n", + " Notice that U[j, :K] is a vector of size K.\n", + " \"\"\"\n", + " # You need to return the following variables correctly.\n", + " X_rec = np.zeros((Z.shape[0], U.shape[0]))\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + " \n", + "\n", + " # =============================================================\n", + " return X_rec" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Once you have completed the code in `recoverData`, the following cell will recover an approximation of the first example and you should see a value of about `[-1.047 -1.047]`. The code will then plot the data in this reduced dimension space. This will show you what the data looks like when using only the corresponding eigenvectors to reconstruct it. An example of what you should get for PCA projection is shown in this figure: \n", + "\n", + "![](Figures/pca_reconstruction.png)\n", + "\n", + "In the figure above, the original data points are indicated with the blue circles, while the projected data points are indicated with the red circles. The projection effectively only retains the information in the direction given by $U_1$. The dotted lines show the distance from the data points in original space to the projected space. Those dotted lines represent the error measure due to PCA projection." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "X_rec = recoverData(Z, U, K)\n", + "print('Approximation of the first example: [{:.6f} {:.6f}]'.format(X_rec[0, 0], X_rec[0, 1]))\n", + "print(' (this value should be about [-1.047419 -1.047419])')\n", + "\n", + "# Plot the normalized dataset (returned from featureNormalize)\n", + "fig, ax = pyplot.subplots(figsize=(5, 5))\n", + "ax.plot(X_norm[:, 0], X_norm[:, 1], 'bo', ms=8, mec='b', mew=0.5)\n", + "ax.set_aspect('equal')\n", + "ax.grid(False)\n", + "pyplot.axis([-3, 2.75, -3, 2.75])\n", + "\n", + "# Draw lines connecting the projected points to the original points\n", + "ax.plot(X_rec[:, 0], X_rec[:, 1], 'ro', mec='r', mew=2, mfc='none')\n", + "for xnorm, xrec in zip(X_norm, X_rec):\n", + " ax.plot([xnorm[0], xrec[0]], [xnorm[1], xrec[1]], '--k', lw=1)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[5] = recoverData\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2.4 Face Image Dataset\n", + "\n", + "In this part of the exercise, you will run PCA on face images to see how it can be used in practice for dimension reduction. The dataset `ex7faces.mat` contains a dataset `X` of face images, each $32 \\times 32$ in grayscale. This dataset was based on a [cropped version](http://conradsanderson.id.au/lfwcrop/) of the [labeled faces in the wild](http://vis-www.cs.umass.edu/lfw/) dataset. Each row of `X` corresponds to one face image (a row vector of length 1024). \n", + "\n", + "The next cell will load and visualize the first 100 of these face images similar to what is shown in this figure:\n", + "\n", + "![Faces](Figures/faces.png)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Load Face dataset\n", + "data = loadmat(os.path.join('Data', 'ex7faces.mat'))\n", + "X = data['X']\n", + "\n", + "# Display the first 100 faces in the dataset\n", + "utils.displayData(X[:100, :], figsize=(8, 8))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "#### 2.4.1 PCA on Faces\n", + "\n", + "To run PCA on the face dataset, we first normalize the dataset by subtracting the mean of each feature from the data matrix `X`. After running PCA, you will obtain the principal components of the dataset. Notice that each principal component in `U` (each column) is a vector of length $n$ (where for the face dataset, $n = 1024$). It turns out that we can visualize these principal components by reshaping each of them into a $32 \\times 32$ matrix that corresponds to the pixels in the original dataset. \n", + "\n", + "The following cell will first normalize the dataset for you and then run your PCA code. Then, the first 36 principal components (conveniently called eigenfaces) that describe the largest variations are displayed. If you want, you can also change the code to display more principal components to see how they capture more and more details." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# normalize X by subtracting the mean value from each feature\n", + "X_norm, mu, sigma = utils.featureNormalize(X)\n", + "\n", + "# Run PCA\n", + "U, S = pca(X_norm)\n", + "\n", + "# Visualize the top 36 eigenvectors found\n", + "utils.displayData(U[:, :36].T, figsize=(8, 8))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "#### 2.4.2 Dimensionality Reduction\n", + "\n", + "Now that you have computed the principal components for the face dataset, you can use it to reduce the dimension of the face dataset. This allows you to use your learning algorithm with a smaller input size (e.g., 100 dimensions) instead of the original 1024 dimensions. This can help speed up your learning algorithm.\n", + "\n", + "The next cell will project the face dataset onto only the first 100 principal components. Concretely, each face image is now described by a vector $z^{(i)} \\in \\mathbb{R}^{100}$. To understand what is lost in the dimension reduction, you can recover the data using only the projected dataset." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Project images to the eigen space using the top k eigenvectors \n", + "# If you are applying a machine learning algorithm \n", + "K = 100\n", + "Z = projectData(X_norm, U, K)\n", + "\n", + "print('The projected data Z has a shape of: ', Z.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In the next cell, an approximate recovery of the data is performed and the original and projected face images\n", + "are displayed similar to what is shown here:\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + "
\n", + "\n", + "From the reconstruction, you can observe that the general structure and appearance of the face are kept while the fine details are lost. This is a remarkable reduction (more than 10x) in the dataset size that can help speed up your learning algorithm significantly. For example, if you were training a neural network to perform person recognition (given a face image, predict the identity of the person), you can use the dimension reduced input of only a 100 dimensions instead of the original pixels." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Project images to the eigen space using the top K eigen vectors and \n", + "# visualize only using those K dimensions\n", + "# Compare to the original input, which is also displayed\n", + "K = 100\n", + "X_rec = recoverData(Z, U, K)\n", + "\n", + "# Display normalized data\n", + "utils.displayData(X_norm[:100, :], figsize=(6, 6))\n", + "pyplot.gcf().suptitle('Original faces')\n", + "\n", + "# Display reconstructed data from only k eigenfaces\n", + "utils.displayData(X_rec[:100, :], figsize=(6, 6))\n", + "pyplot.gcf().suptitle('Recovered faces')\n", + "pass" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2.5 Optional (ungraded) exercise: PCA for visualization\n", + "\n", + "In the earlier K-means image compression exercise, you used the K-means algorithm in the 3-dimensional RGB space. We reduced each pixel of the RGB image to be represented by 16 clusters. In the next cell, we have provided code to visualize the final pixel assignments in this 3D space. Each data point is colored according to the cluster it has been assigned to. You can drag your mouse on the figure to rotate and inspect this data in 3 dimensions." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# this allows to have interactive plot to rotate the 3-D plot\n", + "# The double identical statement is on purpose\n", + "# see: https://stackoverflow.com/questions/43545050/using-matplotlib-notebook-after-matplotlib-inline-in-jupyter-notebook-doesnt\n", + "%matplotlib notebook\n", + "%matplotlib notebook\n", + "from matplotlib import pyplot\n", + "\n", + "\n", + "A = mpl.image.imread(os.path.join('Data', 'bird_small.png'))\n", + "A /= 255\n", + "X = A.reshape(-1, 3)\n", + "\n", + "# perform the K-means clustering again here\n", + "K = 16\n", + "max_iters = 10\n", + "initial_centroids = kMeansInitCentroids(X, K)\n", + "centroids, idx = utils.runkMeans(X, initial_centroids,\n", + " findClosestCentroids,\n", + " computeCentroids, max_iters)\n", + "\n", + "# Sample 1000 random indexes (since working with all the data is\n", + "# too expensive. If you have a fast computer, you may increase this.\n", + "sel = np.random.choice(X.shape[0], size=1000)\n", + "\n", + "fig = pyplot.figure(figsize=(6, 6))\n", + "ax = fig.add_subplot(111, projection='3d')\n", + "\n", + "ax.scatter(X[sel, 0], X[sel, 1], X[sel, 2], cmap='rainbow', c=idx[sel], s=8**2)\n", + "ax.set_title('Pixel dataset plotted in 3D.\\nColor shows centroid memberships')\n", + "pass" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "It turns out that visualizing datasets in 3 dimensions or greater can be cumbersome. Therefore, it is often desirable to only display the data in 2D even at the cost of losing some information. In practice, PCA is often used to reduce the dimensionality of data for visualization purposes. \n", + "\n", + "In the next cell,we will apply your implementation of PCA to the 3-dimensional data to reduce it to 2 dimensions and visualize the result in a 2D scatter plot. The PCA projection can be thought of as a rotation that selects the view that maximizes the spread of the data, which often corresponds to the “best” view." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Subtract the mean to use PCA\n", + "X_norm, mu, sigma = utils.featureNormalize(X)\n", + "\n", + "# PCA and project the data to 2D\n", + "U, S = pca(X_norm)\n", + "Z = projectData(X_norm, U, 2)\n", + "\n", + "# Reset matplotlib to non-interactive\n", + "%matplotlib inline\n", + "\n", + "fig = pyplot.figure(figsize=(6, 6))\n", + "ax = fig.add_subplot(111)\n", + "\n", + "ax.scatter(Z[sel, 0], Z[sel, 1], cmap='rainbow', c=idx[sel], s=64)\n", + "ax.set_title('Pixel dataset plotted in 2D, using PCA for dimensionality reduction')\n", + "ax.grid(False)\n", + "pass" + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.6.4" + } + }, + "nbformat": 4, + "nbformat_minor": 2 +} diff --git a/Exercise7/utils.py b/Exercise7/utils.py new file mode 100755 index 0000000..8110917 --- /dev/null +++ b/Exercise7/utils.py @@ -0,0 +1,236 @@ +import sys +import numpy as np +from matplotlib import pyplot +from matplotlib.animation import FuncAnimation +import matplotlib as mpl + +sys.path.append('..') +from submission import SubmissionBase + + +def displayData(X, example_width=None, figsize=(10, 10)): + """ + Displays 2D data in a nice grid. + + Parameters + ---------- + X : array_like + The input data of size (m x n) where m is the number of examples and n is the number of + features. + + example_width : int, optional + THe width of each 2-D image in pixels. If not provided, the image is assumed to be square, + and the width is the floor of the square root of total number of pixels. + + figsize : tuple, optional + A 2-element tuple indicating the width and height of figure in inches. + """ + # Compute rows, cols + if X.ndim == 2: + m, n = X.shape + elif X.ndim == 1: + n = X.size + m = 1 + X = X[None] # Promote to a 2 dimensional array + else: + raise IndexError('Input X should be 1 or 2 dimensional.') + + example_width = example_width or int(np.round(np.sqrt(n))) + example_height = int(n / example_width) + + # Compute number of items to display + display_rows = int(np.floor(np.sqrt(m))) + display_cols = int(np.ceil(m / display_rows)) + + fig, ax_array = pyplot.subplots(display_rows, display_cols, figsize=figsize) + fig.subplots_adjust(wspace=0.025, hspace=0.025) + + ax_array = [ax_array] if m == 1 else ax_array.ravel() + + for i, ax in enumerate(ax_array): + ax.imshow(X[i].reshape(example_height, example_width, order='F'), cmap='gray') + ax.axis('off') + + +def featureNormalize(X): + """ + Normalizes the features in X returns a normalized version of X where the mean value of each + feature is 0 and the standard deviation is 1. This is often a good preprocessing step to do when + working with learning algorithms. + + Parameters + ---------- + X : array_like + An dataset which is a (m x n) matrix, where m is the number of examples, + and n is the number of dimensions for each example. + + Returns + ------- + X_norm : array_like + The normalized input dataset. + + mu : array_like + A vector of size n corresponding to the mean for each dimension across all examples. + + sigma : array_like + A vector of size n corresponding to the standard deviations for each dimension across + all examples. + """ + mu = np.mean(X, axis=0) + X_norm = X - mu + + sigma = np.std(X_norm, axis=0, ddof=1) + X_norm /= sigma + return X_norm, mu, sigma + + +def plotProgresskMeans(i, X, centroid_history, idx_history): + """ + A helper function that displays the progress of k-Means as it is running. It is intended for use + only with 2D data. It plots data points with colors assigned to each centroid. With the + previous centroids, it also plots a line between the previous locations and current locations + of the centroids. + + Parameters + ---------- + i : int + Current iteration number of k-means. Used for matplotlib animation function. + + X : array_like + The dataset, which is a matrix (m x n). Note since the plot only supports 2D data, n should + be equal to 2. + + centroid_history : list + A list of computed centroids for all iteration. + + idx_history : list + A list of computed assigned indices for all iterations. + """ + K = centroid_history[0].shape[0] + pyplot.gcf().clf() + cmap = pyplot.cm.rainbow + norm = mpl.colors.Normalize(vmin=0, vmax=2) + + for k in range(K): + current = np.stack([c[k, :] for c in centroid_history[:i+1]], axis=0) + pyplot.plot(current[:, 0], current[:, 1], + '-Xk', + mec='k', + lw=2, + ms=10, + mfc=cmap(norm(k)), + mew=2) + + pyplot.scatter(X[:, 0], X[:, 1], + c=idx_history[i], + cmap=cmap, + marker='o', + s=8**2, + linewidths=1,) + pyplot.grid(False) + pyplot.title('Iteration number %d' % (i+1)) + + +def runkMeans(X, centroids, findClosestCentroids, computeCentroids, + max_iters=10, plot_progress=False): + """ + Runs the K-means algorithm. + + Parameters + ---------- + X : array_like + The data set of size (m, n). Each row of X is a single example of n dimensions. The + data set is a total of m examples. + + centroids : array_like + Initial centroid location for each clusters. This is a matrix of size (K, n). K is the total + number of clusters and n is the dimensions of each data point. + + findClosestCentroids : func + A function (implemented by student) reference which computes the cluster assignment for + each example. + + computeCentroids : func + A function(implemented by student) reference which computes the centroid of each cluster. + + max_iters : int, optional + Specifies the total number of interactions of K-Means to execute. + + plot_progress : bool, optional + A flag that indicates if the function should also plot its progress as the learning happens. + This is set to false by default. + + Returns + ------- + centroids : array_like + A (K x n) matrix of the computed (updated) centroids. + idx : array_like + A vector of size (m,) for cluster assignment for each example in the dataset. Each entry + in idx is within the range [0 ... K-1]. + + anim : FuncAnimation, optional + A matplotlib animation object which can be used to embed a video within the jupyter + notebook. This is only returned if `plot_progress` is `True`. + """ + K = centroids.shape[0] + idx = None + idx_history = [] + centroid_history = [] + + for i in range(max_iters): + idx = findClosestCentroids(X, centroids) + + if plot_progress: + idx_history.append(idx) + centroid_history.append(centroids) + + centroids = computeCentroids(X, idx, K) + + if plot_progress: + fig = pyplot.figure() + anim = FuncAnimation(fig, plotProgresskMeans, + frames=max_iters, + interval=500, + repeat_delay=2, + fargs=(X, centroid_history, idx_history)) + return centroids, idx, anim + + return centroids, idx + + +class Grader(SubmissionBase): + # Random Test Cases + X = np.sin(np.arange(1, 166)).reshape(15, 11, order='F') + Z = np.cos(np.arange(1, 122)).reshape(11, 11, order='F') + C = Z[:5, :] + idx = np.arange(1, 16) % 3 + + def __init__(self): + part_names = ['Find Closest Centroids (k-Means)', + 'Compute Centroid Means (k-Means)', + 'PCA', + 'Project Data (PCA)', + 'Recover Data (PCA)'] + super().__init__('k-means-clustering-and-pca', part_names) + + def __iter__(self): + for part_id in range(1, 6): + try: + func = self.functions[part_id] + # Each part has different expected arguments/different function + if part_id == 1: + res = 1 + func(self.X, self.C) + elif part_id == 2: + res = func(self.X, self.idx, 3) + elif part_id == 3: + U, S = func(self.X) + res = np.hstack([U.ravel('F'), np.diag(S).ravel('F')]).tolist() + elif part_id == 4: + res = func(self.X, self.Z, 5) + elif part_id == 5: + res = func(self.X[:, :5], self.Z, 5) + else: + raise KeyError + yield part_id, res + except KeyError: + yield part_id, 0 diff --git a/Exercise8/Data/ex8_movieParams.mat b/Exercise8/Data/ex8_movieParams.mat new file mode 100755 index 0000000..2dea689 Binary files /dev/null and b/Exercise8/Data/ex8_movieParams.mat differ diff --git a/Exercise8/Data/ex8_movies.mat b/Exercise8/Data/ex8_movies.mat new file mode 100755 index 0000000..31ecd00 Binary files /dev/null and b/Exercise8/Data/ex8_movies.mat differ diff --git a/Exercise8/Data/ex8data1.mat b/Exercise8/Data/ex8data1.mat new file mode 100755 index 0000000..1f08123 Binary files /dev/null and b/Exercise8/Data/ex8data1.mat differ diff --git a/Exercise8/Data/ex8data2.mat b/Exercise8/Data/ex8data2.mat new file mode 100755 index 0000000..fe48db3 Binary files /dev/null and b/Exercise8/Data/ex8data2.mat differ diff --git a/Exercise8/Data/movie_ids.txt b/Exercise8/Data/movie_ids.txt new file mode 100755 index 0000000..392427a --- /dev/null +++ b/Exercise8/Data/movie_ids.txt @@ -0,0 +1,1682 @@ +1 Toy Story (1995) +2 GoldenEye (1995) +3 Four Rooms (1995) +4 Get Shorty (1995) +5 Copycat (1995) +6 Shanghai Triad (Yao a yao yao dao waipo qiao) (1995) +7 Twelve Monkeys (1995) +8 Babe (1995) +9 Dead Man Walking (1995) +10 Richard III (1995) +11 Seven (Se7en) (1995) +12 Usual Suspects, The (1995) +13 Mighty Aphrodite (1995) +14 Postino, Il (1994) +15 Mr. Holland's Opus (1995) +16 French Twist (Gazon maudit) (1995) +17 From Dusk Till Dawn (1996) +18 White Balloon, The (1995) +19 Antonia's Line (1995) +20 Angels and Insects (1995) +21 Muppet Treasure Island (1996) +22 Braveheart (1995) +23 Taxi Driver (1976) +24 Rumble in the Bronx (1995) +25 Birdcage, The (1996) +26 Brothers McMullen, The (1995) +27 Bad Boys (1995) +28 Apollo 13 (1995) +29 Batman Forever (1995) +30 Belle de jour (1967) +31 Crimson Tide (1995) +32 Crumb (1994) +33 Desperado (1995) +34 Doom Generation, The (1995) +35 Free Willy 2: The Adventure Home (1995) +36 Mad Love (1995) +37 Nadja (1994) +38 Net, The (1995) +39 Strange Days (1995) +40 To Wong Foo, Thanks for Everything! Julie Newmar (1995) +41 Billy Madison (1995) +42 Clerks (1994) +43 Disclosure (1994) +44 Dolores Claiborne (1994) +45 Eat Drink Man Woman (1994) +46 Exotica (1994) +47 Ed Wood (1994) +48 Hoop Dreams (1994) +49 I.Q. (1994) +50 Star Wars (1977) +51 Legends of the Fall (1994) +52 Madness of King George, The (1994) +53 Natural Born Killers (1994) +54 Outbreak (1995) +55 Professional, The (1994) +56 Pulp Fiction (1994) +57 Priest (1994) +58 Quiz Show (1994) +59 Three Colors: Red (1994) +60 Three Colors: Blue (1993) +61 Three Colors: White (1994) +62 Stargate (1994) +63 Santa Clause, The (1994) +64 Shawshank Redemption, The (1994) +65 What's Eating Gilbert Grape (1993) +66 While You Were Sleeping (1995) +67 Ace Ventura: Pet Detective (1994) +68 Crow, The (1994) +69 Forrest Gump (1994) +70 Four Weddings and a Funeral (1994) +71 Lion King, The (1994) +72 Mask, The (1994) +73 Maverick (1994) +74 Faster Pussycat! Kill! Kill! (1965) +75 Brother Minister: The Assassination of Malcolm X (1994) +76 Carlito's Way (1993) +77 Firm, The (1993) +78 Free Willy (1993) +79 Fugitive, The (1993) +80 Hot Shots! Part Deux (1993) +81 Hudsucker Proxy, The (1994) +82 Jurassic Park (1993) +83 Much Ado About Nothing (1993) +84 Robert A. Heinlein's The Puppet Masters (1994) +85 Ref, The (1994) +86 Remains of the Day, The (1993) +87 Searching for Bobby Fischer (1993) +88 Sleepless in Seattle (1993) +89 Blade Runner (1982) +90 So I Married an Axe Murderer (1993) +91 Nightmare Before Christmas, The (1993) +92 True Romance (1993) +93 Welcome to the Dollhouse (1995) +94 Home Alone (1990) +95 Aladdin (1992) +96 Terminator 2: Judgment Day (1991) +97 Dances with Wolves (1990) +98 Silence of the Lambs, The (1991) +99 Snow White and the Seven Dwarfs (1937) +100 Fargo (1996) +101 Heavy Metal (1981) +102 Aristocats, The (1970) +103 All Dogs Go to Heaven 2 (1996) +104 Theodore Rex (1995) +105 Sgt. Bilko (1996) +106 Diabolique (1996) +107 Moll Flanders (1996) +108 Kids in the Hall: Brain Candy (1996) +109 Mystery Science Theater 3000: The Movie (1996) +110 Operation Dumbo Drop (1995) +111 Truth About Cats & Dogs, The (1996) +112 Flipper (1996) +113 Horseman on the Roof, The (Hussard sur le toit, Le) (1995) +114 Wallace & Gromit: The Best of Aardman Animation (1996) +115 Haunted World of Edward D. Wood Jr., The (1995) +116 Cold Comfort Farm (1995) +117 Rock, The (1996) +118 Twister (1996) +119 Maya Lin: A Strong Clear Vision (1994) +120 Striptease (1996) +121 Independence Day (ID4) (1996) +122 Cable Guy, The (1996) +123 Frighteners, The (1996) +124 Lone Star (1996) +125 Phenomenon (1996) +126 Spitfire Grill, The (1996) +127 Godfather, The (1972) +128 Supercop (1992) +129 Bound (1996) +130 Kansas City (1996) +131 Breakfast at Tiffany's (1961) +132 Wizard of Oz, The (1939) +133 Gone with the Wind (1939) +134 Citizen Kane (1941) +135 2001: A Space Odyssey (1968) +136 Mr. Smith Goes to Washington (1939) +137 Big Night (1996) +138 D3: The Mighty Ducks (1996) +139 Love Bug, The (1969) +140 Homeward Bound: The Incredible Journey (1993) +141 20,000 Leagues Under the Sea (1954) +142 Bedknobs and Broomsticks (1971) +143 Sound of Music, The (1965) +144 Die Hard (1988) +145 Lawnmower Man, The (1992) +146 Unhook the Stars (1996) +147 Long Kiss Goodnight, The (1996) +148 Ghost and the Darkness, The (1996) +149 Jude (1996) +150 Swingers (1996) +151 Willy Wonka and the Chocolate Factory (1971) +152 Sleeper (1973) +153 Fish Called Wanda, A (1988) +154 Monty Python's Life of Brian (1979) +155 Dirty Dancing (1987) +156 Reservoir Dogs (1992) +157 Platoon (1986) +158 Weekend at Bernie's (1989) +159 Basic Instinct (1992) +160 Glengarry Glen Ross (1992) +161 Top Gun (1986) +162 On Golden Pond (1981) +163 Return of the Pink Panther, The (1974) +164 Abyss, The (1989) +165 Jean de Florette (1986) +166 Manon of the Spring (Manon des sources) (1986) +167 Private Benjamin (1980) +168 Monty Python and the Holy Grail (1974) +169 Wrong Trousers, The (1993) +170 Cinema Paradiso (1988) +171 Delicatessen (1991) +172 Empire Strikes Back, The (1980) +173 Princess Bride, The (1987) +174 Raiders of the Lost Ark (1981) +175 Brazil (1985) +176 Aliens (1986) +177 Good, The Bad and The Ugly, The (1966) +178 12 Angry Men (1957) +179 Clockwork Orange, A (1971) +180 Apocalypse Now (1979) +181 Return of the Jedi (1983) +182 GoodFellas (1990) +183 Alien (1979) +184 Army of Darkness (1993) +185 Psycho (1960) +186 Blues Brothers, The (1980) +187 Godfather: Part II, The (1974) +188 Full Metal Jacket (1987) +189 Grand Day Out, A (1992) +190 Henry V (1989) +191 Amadeus (1984) +192 Raging Bull (1980) +193 Right Stuff, The (1983) +194 Sting, The (1973) +195 Terminator, The (1984) +196 Dead Poets Society (1989) +197 Graduate, The (1967) +198 Nikita (La Femme Nikita) (1990) +199 Bridge on the River Kwai, The (1957) +200 Shining, The (1980) +201 Evil Dead II (1987) +202 Groundhog Day (1993) +203 Unforgiven (1992) +204 Back to the Future (1985) +205 Patton (1970) +206 Akira (1988) +207 Cyrano de Bergerac (1990) +208 Young Frankenstein (1974) +209 This Is Spinal Tap (1984) +210 Indiana Jones and the Last Crusade (1989) +211 M*A*S*H (1970) +212 Unbearable Lightness of Being, The (1988) +213 Room with a View, A (1986) +214 Pink Floyd - The Wall (1982) +215 Field of Dreams (1989) +216 When Harry Met Sally... (1989) +217 Bram Stoker's Dracula (1992) +218 Cape Fear (1991) +219 Nightmare on Elm Street, A (1984) +220 Mirror Has Two Faces, The (1996) +221 Breaking the Waves (1996) +222 Star Trek: First Contact (1996) +223 Sling Blade (1996) +224 Ridicule (1996) +225 101 Dalmatians (1996) +226 Die Hard 2 (1990) +227 Star Trek VI: The Undiscovered Country (1991) +228 Star Trek: The Wrath of Khan (1982) +229 Star Trek III: The Search for Spock (1984) +230 Star Trek IV: The Voyage Home (1986) +231 Batman Returns (1992) +232 Young Guns (1988) +233 Under Siege (1992) +234 Jaws (1975) +235 Mars Attacks! (1996) +236 Citizen Ruth (1996) +237 Jerry Maguire (1996) +238 Raising Arizona (1987) +239 Sneakers (1992) +240 Beavis and Butt-head Do America (1996) +241 Last of the Mohicans, The (1992) +242 Kolya (1996) +243 Jungle2Jungle (1997) +244 Smilla's Sense of Snow (1997) +245 Devil's Own, The (1997) +246 Chasing Amy (1997) +247 Turbo: A Power Rangers Movie (1997) +248 Grosse Pointe Blank (1997) +249 Austin Powers: International Man of Mystery (1997) +250 Fifth Element, The (1997) +251 Shall We Dance? (1996) +252 Lost World: Jurassic Park, The (1997) +253 Pillow Book, The (1995) +254 Batman & Robin (1997) +255 My Best Friend's Wedding (1997) +256 When the Cats Away (Chacun cherche son chat) (1996) +257 Men in Black (1997) +258 Contact (1997) +259 George of the Jungle (1997) +260 Event Horizon (1997) +261 Air Bud (1997) +262 In the Company of Men (1997) +263 Steel (1997) +264 Mimic (1997) +265 Hunt for Red October, The (1990) +266 Kull the Conqueror (1997) +267 unknown +268 Chasing Amy (1997) +269 Full Monty, The (1997) +270 Gattaca (1997) +271 Starship Troopers (1997) +272 Good Will Hunting (1997) +273 Heat (1995) +274 Sabrina (1995) +275 Sense and Sensibility (1995) +276 Leaving Las Vegas (1995) +277 Restoration (1995) +278 Bed of Roses (1996) +279 Once Upon a Time... When We Were Colored (1995) +280 Up Close and Personal (1996) +281 River Wild, The (1994) +282 Time to Kill, A (1996) +283 Emma (1996) +284 Tin Cup (1996) +285 Secrets & Lies (1996) +286 English Patient, The (1996) +287 Marvin's Room (1996) +288 Scream (1996) +289 Evita (1996) +290 Fierce Creatures (1997) +291 Absolute Power (1997) +292 Rosewood (1997) +293 Donnie Brasco (1997) +294 Liar Liar (1997) +295 Breakdown (1997) +296 Promesse, La (1996) +297 Ulee's Gold (1997) +298 Face/Off (1997) +299 Hoodlum (1997) +300 Air Force One (1997) +301 In & Out (1997) +302 L.A. Confidential (1997) +303 Ulee's Gold (1997) +304 Fly Away Home (1996) +305 Ice Storm, The (1997) +306 Mrs. Brown (Her Majesty, Mrs. Brown) (1997) +307 Devil's Advocate, The (1997) +308 FairyTale: A True Story (1997) +309 Deceiver (1997) +310 Rainmaker, The (1997) +311 Wings of the Dove, The (1997) +312 Midnight in the Garden of Good and Evil (1997) +313 Titanic (1997) +314 3 Ninjas: High Noon At Mega Mountain (1998) +315 Apt Pupil (1998) +316 As Good As It Gets (1997) +317 In the Name of the Father (1993) +318 Schindler's List (1993) +319 Everyone Says I Love You (1996) +320 Paradise Lost: The Child Murders at Robin Hood Hills (1996) +321 Mother (1996) +322 Murder at 1600 (1997) +323 Dante's Peak (1997) +324 Lost Highway (1997) +325 Crash (1996) +326 G.I. Jane (1997) +327 Cop Land (1997) +328 Conspiracy Theory (1997) +329 Desperate Measures (1998) +330 187 (1997) +331 Edge, The (1997) +332 Kiss the Girls (1997) +333 Game, The (1997) +334 U Turn (1997) +335 How to Be a Player (1997) +336 Playing God (1997) +337 House of Yes, The (1997) +338 Bean (1997) +339 Mad City (1997) +340 Boogie Nights (1997) +341 Critical Care (1997) +342 Man Who Knew Too Little, The (1997) +343 Alien: Resurrection (1997) +344 Apostle, The (1997) +345 Deconstructing Harry (1997) +346 Jackie Brown (1997) +347 Wag the Dog (1997) +348 Desperate Measures (1998) +349 Hard Rain (1998) +350 Fallen (1998) +351 Prophecy II, The (1998) +352 Spice World (1997) +353 Deep Rising (1998) +354 Wedding Singer, The (1998) +355 Sphere (1998) +356 Client, The (1994) +357 One Flew Over the Cuckoo's Nest (1975) +358 Spawn (1997) +359 Assignment, The (1997) +360 Wonderland (1997) +361 Incognito (1997) +362 Blues Brothers 2000 (1998) +363 Sudden Death (1995) +364 Ace Ventura: When Nature Calls (1995) +365 Powder (1995) +366 Dangerous Minds (1995) +367 Clueless (1995) +368 Bio-Dome (1996) +369 Black Sheep (1996) +370 Mary Reilly (1996) +371 Bridges of Madison County, The (1995) +372 Jeffrey (1995) +373 Judge Dredd (1995) +374 Mighty Morphin Power Rangers: The Movie (1995) +375 Showgirls (1995) +376 Houseguest (1994) +377 Heavyweights (1994) +378 Miracle on 34th Street (1994) +379 Tales From the Crypt Presents: Demon Knight (1995) +380 Star Trek: Generations (1994) +381 Muriel's Wedding (1994) +382 Adventures of Priscilla, Queen of the Desert, The (1994) +383 Flintstones, The (1994) +384 Naked Gun 33 1/3: The Final Insult (1994) +385 True Lies (1994) +386 Addams Family Values (1993) +387 Age of Innocence, The (1993) +388 Beverly Hills Cop III (1994) +389 Black Beauty (1994) +390 Fear of a Black Hat (1993) +391 Last Action Hero (1993) +392 Man Without a Face, The (1993) +393 Mrs. Doubtfire (1993) +394 Radioland Murders (1994) +395 Robin Hood: Men in Tights (1993) +396 Serial Mom (1994) +397 Striking Distance (1993) +398 Super Mario Bros. (1993) +399 Three Musketeers, The (1993) +400 Little Rascals, The (1994) +401 Brady Bunch Movie, The (1995) +402 Ghost (1990) +403 Batman (1989) +404 Pinocchio (1940) +405 Mission: Impossible (1996) +406 Thinner (1996) +407 Spy Hard (1996) +408 Close Shave, A (1995) +409 Jack (1996) +410 Kingpin (1996) +411 Nutty Professor, The (1996) +412 Very Brady Sequel, A (1996) +413 Tales from the Crypt Presents: Bordello of Blood (1996) +414 My Favorite Year (1982) +415 Apple Dumpling Gang, The (1975) +416 Old Yeller (1957) +417 Parent Trap, The (1961) +418 Cinderella (1950) +419 Mary Poppins (1964) +420 Alice in Wonderland (1951) +421 William Shakespeare's Romeo and Juliet (1996) +422 Aladdin and the King of Thieves (1996) +423 E.T. the Extra-Terrestrial (1982) +424 Children of the Corn: The Gathering (1996) +425 Bob Roberts (1992) +426 Transformers: The Movie, The (1986) +427 To Kill a Mockingbird (1962) +428 Harold and Maude (1971) +429 Day the Earth Stood Still, The (1951) +430 Duck Soup (1933) +431 Highlander (1986) +432 Fantasia (1940) +433 Heathers (1989) +434 Forbidden Planet (1956) +435 Butch Cassidy and the Sundance Kid (1969) +436 American Werewolf in London, An (1981) +437 Amityville 1992: It's About Time (1992) +438 Amityville 3-D (1983) +439 Amityville: A New Generation (1993) +440 Amityville II: The Possession (1982) +441 Amityville Horror, The (1979) +442 Amityville Curse, The (1990) +443 Birds, The (1963) +444 Blob, The (1958) +445 Body Snatcher, The (1945) +446 Burnt Offerings (1976) +447 Carrie (1976) +448 Omen, The (1976) +449 Star Trek: The Motion Picture (1979) +450 Star Trek V: The Final Frontier (1989) +451 Grease (1978) +452 Jaws 2 (1978) +453 Jaws 3-D (1983) +454 Bastard Out of Carolina (1996) +455 Jackie Chan's First Strike (1996) +456 Beverly Hills Ninja (1997) +457 Free Willy 3: The Rescue (1997) +458 Nixon (1995) +459 Cry, the Beloved Country (1995) +460 Crossing Guard, The (1995) +461 Smoke (1995) +462 Like Water For Chocolate (Como agua para chocolate) (1992) +463 Secret of Roan Inish, The (1994) +464 Vanya on 42nd Street (1994) +465 Jungle Book, The (1994) +466 Red Rock West (1992) +467 Bronx Tale, A (1993) +468 Rudy (1993) +469 Short Cuts (1993) +470 Tombstone (1993) +471 Courage Under Fire (1996) +472 Dragonheart (1996) +473 James and the Giant Peach (1996) +474 Dr. Strangelove or: How I Learned to Stop Worrying and Love the Bomb (1963) +475 Trainspotting (1996) +476 First Wives Club, The (1996) +477 Matilda (1996) +478 Philadelphia Story, The (1940) +479 Vertigo (1958) +480 North by Northwest (1959) +481 Apartment, The (1960) +482 Some Like It Hot (1959) +483 Casablanca (1942) +484 Maltese Falcon, The (1941) +485 My Fair Lady (1964) +486 Sabrina (1954) +487 Roman Holiday (1953) +488 Sunset Blvd. (1950) +489 Notorious (1946) +490 To Catch a Thief (1955) +491 Adventures of Robin Hood, The (1938) +492 East of Eden (1955) +493 Thin Man, The (1934) +494 His Girl Friday (1940) +495 Around the World in 80 Days (1956) +496 It's a Wonderful Life (1946) +497 Bringing Up Baby (1938) +498 African Queen, The (1951) +499 Cat on a Hot Tin Roof (1958) +500 Fly Away Home (1996) +501 Dumbo (1941) +502 Bananas (1971) +503 Candidate, The (1972) +504 Bonnie and Clyde (1967) +505 Dial M for Murder (1954) +506 Rebel Without a Cause (1955) +507 Streetcar Named Desire, A (1951) +508 People vs. Larry Flynt, The (1996) +509 My Left Foot (1989) +510 Magnificent Seven, The (1954) +511 Lawrence of Arabia (1962) +512 Wings of Desire (1987) +513 Third Man, The (1949) +514 Annie Hall (1977) +515 Boot, Das (1981) +516 Local Hero (1983) +517 Manhattan (1979) +518 Miller's Crossing (1990) +519 Treasure of the Sierra Madre, The (1948) +520 Great Escape, The (1963) +521 Deer Hunter, The (1978) +522 Down by Law (1986) +523 Cool Hand Luke (1967) +524 Great Dictator, The (1940) +525 Big Sleep, The (1946) +526 Ben-Hur (1959) +527 Gandhi (1982) +528 Killing Fields, The (1984) +529 My Life as a Dog (Mitt liv som hund) (1985) +530 Man Who Would Be King, The (1975) +531 Shine (1996) +532 Kama Sutra: A Tale of Love (1996) +533 Daytrippers, The (1996) +534 Traveller (1997) +535 Addicted to Love (1997) +536 Ponette (1996) +537 My Own Private Idaho (1991) +538 Anastasia (1997) +539 Mouse Hunt (1997) +540 Money Train (1995) +541 Mortal Kombat (1995) +542 Pocahontas (1995) +543 Misrables, Les (1995) +544 Things to Do in Denver when You're Dead (1995) +545 Vampire in Brooklyn (1995) +546 Broken Arrow (1996) +547 Young Poisoner's Handbook, The (1995) +548 NeverEnding Story III, The (1994) +549 Rob Roy (1995) +550 Die Hard: With a Vengeance (1995) +551 Lord of Illusions (1995) +552 Species (1995) +553 Walk in the Clouds, A (1995) +554 Waterworld (1995) +555 White Man's Burden (1995) +556 Wild Bill (1995) +557 Farinelli: il castrato (1994) +558 Heavenly Creatures (1994) +559 Interview with the Vampire (1994) +560 Kid in King Arthur's Court, A (1995) +561 Mary Shelley's Frankenstein (1994) +562 Quick and the Dead, The (1995) +563 Stephen King's The Langoliers (1995) +564 Tales from the Hood (1995) +565 Village of the Damned (1995) +566 Clear and Present Danger (1994) +567 Wes Craven's New Nightmare (1994) +568 Speed (1994) +569 Wolf (1994) +570 Wyatt Earp (1994) +571 Another Stakeout (1993) +572 Blown Away (1994) +573 Body Snatchers (1993) +574 Boxing Helena (1993) +575 City Slickers II: The Legend of Curly's Gold (1994) +576 Cliffhanger (1993) +577 Coneheads (1993) +578 Demolition Man (1993) +579 Fatal Instinct (1993) +580 Englishman Who Went Up a Hill, But Came Down a Mountain, The (1995) +581 Kalifornia (1993) +582 Piano, The (1993) +583 Romeo Is Bleeding (1993) +584 Secret Garden, The (1993) +585 Son in Law (1993) +586 Terminal Velocity (1994) +587 Hour of the Pig, The (1993) +588 Beauty and the Beast (1991) +589 Wild Bunch, The (1969) +590 Hellraiser: Bloodline (1996) +591 Primal Fear (1996) +592 True Crime (1995) +593 Stalingrad (1993) +594 Heavy (1995) +595 Fan, The (1996) +596 Hunchback of Notre Dame, The (1996) +597 Eraser (1996) +598 Big Squeeze, The (1996) +599 Police Story 4: Project S (Chao ji ji hua) (1993) +600 Daniel Defoe's Robinson Crusoe (1996) +601 For Whom the Bell Tolls (1943) +602 American in Paris, An (1951) +603 Rear Window (1954) +604 It Happened One Night (1934) +605 Meet Me in St. Louis (1944) +606 All About Eve (1950) +607 Rebecca (1940) +608 Spellbound (1945) +609 Father of the Bride (1950) +610 Gigi (1958) +611 Laura (1944) +612 Lost Horizon (1937) +613 My Man Godfrey (1936) +614 Giant (1956) +615 39 Steps, The (1935) +616 Night of the Living Dead (1968) +617 Blue Angel, The (Blaue Engel, Der) (1930) +618 Picnic (1955) +619 Extreme Measures (1996) +620 Chamber, The (1996) +621 Davy Crockett, King of the Wild Frontier (1955) +622 Swiss Family Robinson (1960) +623 Angels in the Outfield (1994) +624 Three Caballeros, The (1945) +625 Sword in the Stone, The (1963) +626 So Dear to My Heart (1949) +627 Robin Hood: Prince of Thieves (1991) +628 Sleepers (1996) +629 Victor/Victoria (1982) +630 Great Race, The (1965) +631 Crying Game, The (1992) +632 Sophie's Choice (1982) +633 Christmas Carol, A (1938) +634 Microcosmos: Le peuple de l'herbe (1996) +635 Fog, The (1980) +636 Escape from New York (1981) +637 Howling, The (1981) +638 Return of Martin Guerre, The (Retour de Martin Guerre, Le) (1982) +639 Tin Drum, The (Blechtrommel, Die) (1979) +640 Cook the Thief His Wife & Her Lover, The (1989) +641 Paths of Glory (1957) +642 Grifters, The (1990) +643 The Innocent (1994) +644 Thin Blue Line, The (1988) +645 Paris Is Burning (1990) +646 Once Upon a Time in the West (1969) +647 Ran (1985) +648 Quiet Man, The (1952) +649 Once Upon a Time in America (1984) +650 Seventh Seal, The (Sjunde inseglet, Det) (1957) +651 Glory (1989) +652 Rosencrantz and Guildenstern Are Dead (1990) +653 Touch of Evil (1958) +654 Chinatown (1974) +655 Stand by Me (1986) +656 M (1931) +657 Manchurian Candidate, The (1962) +658 Pump Up the Volume (1990) +659 Arsenic and Old Lace (1944) +660 Fried Green Tomatoes (1991) +661 High Noon (1952) +662 Somewhere in Time (1980) +663 Being There (1979) +664 Paris, Texas (1984) +665 Alien 3 (1992) +666 Blood For Dracula (Andy Warhol's Dracula) (1974) +667 Audrey Rose (1977) +668 Blood Beach (1981) +669 Body Parts (1991) +670 Body Snatchers (1993) +671 Bride of Frankenstein (1935) +672 Candyman (1992) +673 Cape Fear (1962) +674 Cat People (1982) +675 Nosferatu (Nosferatu, eine Symphonie des Grauens) (1922) +676 Crucible, The (1996) +677 Fire on the Mountain (1996) +678 Volcano (1997) +679 Conan the Barbarian (1981) +680 Kull the Conqueror (1997) +681 Wishmaster (1997) +682 I Know What You Did Last Summer (1997) +683 Rocket Man (1997) +684 In the Line of Fire (1993) +685 Executive Decision (1996) +686 Perfect World, A (1993) +687 McHale's Navy (1997) +688 Leave It to Beaver (1997) +689 Jackal, The (1997) +690 Seven Years in Tibet (1997) +691 Dark City (1998) +692 American President, The (1995) +693 Casino (1995) +694 Persuasion (1995) +695 Kicking and Screaming (1995) +696 City Hall (1996) +697 Basketball Diaries, The (1995) +698 Browning Version, The (1994) +699 Little Women (1994) +700 Miami Rhapsody (1995) +701 Wonderful, Horrible Life of Leni Riefenstahl, The (1993) +702 Barcelona (1994) +703 Widows' Peak (1994) +704 House of the Spirits, The (1993) +705 Singin' in the Rain (1952) +706 Bad Moon (1996) +707 Enchanted April (1991) +708 Sex, Lies, and Videotape (1989) +709 Strictly Ballroom (1992) +710 Better Off Dead... (1985) +711 Substance of Fire, The (1996) +712 Tin Men (1987) +713 Othello (1995) +714 Carrington (1995) +715 To Die For (1995) +716 Home for the Holidays (1995) +717 Juror, The (1996) +718 In the Bleak Midwinter (1995) +719 Canadian Bacon (1994) +720 First Knight (1995) +721 Mallrats (1995) +722 Nine Months (1995) +723 Boys on the Side (1995) +724 Circle of Friends (1995) +725 Exit to Eden (1994) +726 Fluke (1995) +727 Immortal Beloved (1994) +728 Junior (1994) +729 Nell (1994) +730 Queen Margot (Reine Margot, La) (1994) +731 Corrina, Corrina (1994) +732 Dave (1993) +733 Go Fish (1994) +734 Made in America (1993) +735 Philadelphia (1993) +736 Shadowlands (1993) +737 Sirens (1994) +738 Threesome (1994) +739 Pretty Woman (1990) +740 Jane Eyre (1996) +741 Last Supper, The (1995) +742 Ransom (1996) +743 Crow: City of Angels, The (1996) +744 Michael Collins (1996) +745 Ruling Class, The (1972) +746 Real Genius (1985) +747 Benny & Joon (1993) +748 Saint, The (1997) +749 MatchMaker, The (1997) +750 Amistad (1997) +751 Tomorrow Never Dies (1997) +752 Replacement Killers, The (1998) +753 Burnt By the Sun (1994) +754 Red Corner (1997) +755 Jumanji (1995) +756 Father of the Bride Part II (1995) +757 Across the Sea of Time (1995) +758 Lawnmower Man 2: Beyond Cyberspace (1996) +759 Fair Game (1995) +760 Screamers (1995) +761 Nick of Time (1995) +762 Beautiful Girls (1996) +763 Happy Gilmore (1996) +764 If Lucy Fell (1996) +765 Boomerang (1992) +766 Man of the Year (1995) +767 Addiction, The (1995) +768 Casper (1995) +769 Congo (1995) +770 Devil in a Blue Dress (1995) +771 Johnny Mnemonic (1995) +772 Kids (1995) +773 Mute Witness (1994) +774 Prophecy, The (1995) +775 Something to Talk About (1995) +776 Three Wishes (1995) +777 Castle Freak (1995) +778 Don Juan DeMarco (1995) +779 Drop Zone (1994) +780 Dumb & Dumber (1994) +781 French Kiss (1995) +782 Little Odessa (1994) +783 Milk Money (1994) +784 Beyond Bedlam (1993) +785 Only You (1994) +786 Perez Family, The (1995) +787 Roommates (1995) +788 Relative Fear (1994) +789 Swimming with Sharks (1995) +790 Tommy Boy (1995) +791 Baby-Sitters Club, The (1995) +792 Bullets Over Broadway (1994) +793 Crooklyn (1994) +794 It Could Happen to You (1994) +795 Richie Rich (1994) +796 Speechless (1994) +797 Timecop (1994) +798 Bad Company (1995) +799 Boys Life (1995) +800 In the Mouth of Madness (1995) +801 Air Up There, The (1994) +802 Hard Target (1993) +803 Heaven & Earth (1993) +804 Jimmy Hollywood (1994) +805 Manhattan Murder Mystery (1993) +806 Menace II Society (1993) +807 Poetic Justice (1993) +808 Program, The (1993) +809 Rising Sun (1993) +810 Shadow, The (1994) +811 Thirty-Two Short Films About Glenn Gould (1993) +812 Andre (1994) +813 Celluloid Closet, The (1995) +814 Great Day in Harlem, A (1994) +815 One Fine Day (1996) +816 Candyman: Farewell to the Flesh (1995) +817 Frisk (1995) +818 Girl 6 (1996) +819 Eddie (1996) +820 Space Jam (1996) +821 Mrs. Winterbourne (1996) +822 Faces (1968) +823 Mulholland Falls (1996) +824 Great White Hype, The (1996) +825 Arrival, The (1996) +826 Phantom, The (1996) +827 Daylight (1996) +828 Alaska (1996) +829 Fled (1996) +830 Power 98 (1995) +831 Escape from L.A. (1996) +832 Bogus (1996) +833 Bulletproof (1996) +834 Halloween: The Curse of Michael Myers (1995) +835 Gay Divorcee, The (1934) +836 Ninotchka (1939) +837 Meet John Doe (1941) +838 In the Line of Duty 2 (1987) +839 Loch Ness (1995) +840 Last Man Standing (1996) +841 Glimmer Man, The (1996) +842 Pollyanna (1960) +843 Shaggy Dog, The (1959) +844 Freeway (1996) +845 That Thing You Do! (1996) +846 To Gillian on Her 37th Birthday (1996) +847 Looking for Richard (1996) +848 Murder, My Sweet (1944) +849 Days of Thunder (1990) +850 Perfect Candidate, A (1996) +851 Two or Three Things I Know About Her (1966) +852 Bloody Child, The (1996) +853 Braindead (1992) +854 Bad Taste (1987) +855 Diva (1981) +856 Night on Earth (1991) +857 Paris Was a Woman (1995) +858 Amityville: Dollhouse (1996) +859 April Fool's Day (1986) +860 Believers, The (1987) +861 Nosferatu a Venezia (1986) +862 Jingle All the Way (1996) +863 Garden of Finzi-Contini, The (Giardino dei Finzi-Contini, Il) (1970) +864 My Fellow Americans (1996) +865 Ice Storm, The (1997) +866 Michael (1996) +867 Whole Wide World, The (1996) +868 Hearts and Minds (1996) +869 Fools Rush In (1997) +870 Touch (1997) +871 Vegas Vacation (1997) +872 Love Jones (1997) +873 Picture Perfect (1997) +874 Career Girls (1997) +875 She's So Lovely (1997) +876 Money Talks (1997) +877 Excess Baggage (1997) +878 That Darn Cat! (1997) +879 Peacemaker, The (1997) +880 Soul Food (1997) +881 Money Talks (1997) +882 Washington Square (1997) +883 Telling Lies in America (1997) +884 Year of the Horse (1997) +885 Phantoms (1998) +886 Life Less Ordinary, A (1997) +887 Eve's Bayou (1997) +888 One Night Stand (1997) +889 Tango Lesson, The (1997) +890 Mortal Kombat: Annihilation (1997) +891 Bent (1997) +892 Flubber (1997) +893 For Richer or Poorer (1997) +894 Home Alone 3 (1997) +895 Scream 2 (1997) +896 Sweet Hereafter, The (1997) +897 Time Tracers (1995) +898 Postman, The (1997) +899 Winter Guest, The (1997) +900 Kundun (1997) +901 Mr. Magoo (1997) +902 Big Lebowski, The (1998) +903 Afterglow (1997) +904 Ma vie en rose (My Life in Pink) (1997) +905 Great Expectations (1998) +906 Oscar & Lucinda (1997) +907 Vermin (1998) +908 Half Baked (1998) +909 Dangerous Beauty (1998) +910 Nil By Mouth (1997) +911 Twilight (1998) +912 U.S. Marshalls (1998) +913 Love and Death on Long Island (1997) +914 Wild Things (1998) +915 Primary Colors (1998) +916 Lost in Space (1998) +917 Mercury Rising (1998) +918 City of Angels (1998) +919 City of Lost Children, The (1995) +920 Two Bits (1995) +921 Farewell My Concubine (1993) +922 Dead Man (1995) +923 Raise the Red Lantern (1991) +924 White Squall (1996) +925 Unforgettable (1996) +926 Down Periscope (1996) +927 Flower of My Secret, The (Flor de mi secreto, La) (1995) +928 Craft, The (1996) +929 Harriet the Spy (1996) +930 Chain Reaction (1996) +931 Island of Dr. Moreau, The (1996) +932 First Kid (1996) +933 Funeral, The (1996) +934 Preacher's Wife, The (1996) +935 Paradise Road (1997) +936 Brassed Off (1996) +937 Thousand Acres, A (1997) +938 Smile Like Yours, A (1997) +939 Murder in the First (1995) +940 Airheads (1994) +941 With Honors (1994) +942 What's Love Got to Do with It (1993) +943 Killing Zoe (1994) +944 Renaissance Man (1994) +945 Charade (1963) +946 Fox and the Hound, The (1981) +947 Big Blue, The (Grand bleu, Le) (1988) +948 Booty Call (1997) +949 How to Make an American Quilt (1995) +950 Georgia (1995) +951 Indian in the Cupboard, The (1995) +952 Blue in the Face (1995) +953 Unstrung Heroes (1995) +954 Unzipped (1995) +955 Before Sunrise (1995) +956 Nobody's Fool (1994) +957 Pushing Hands (1992) +958 To Live (Huozhe) (1994) +959 Dazed and Confused (1993) +960 Naked (1993) +961 Orlando (1993) +962 Ruby in Paradise (1993) +963 Some Folks Call It a Sling Blade (1993) +964 Month by the Lake, A (1995) +965 Funny Face (1957) +966 Affair to Remember, An (1957) +967 Little Lord Fauntleroy (1936) +968 Inspector General, The (1949) +969 Winnie the Pooh and the Blustery Day (1968) +970 Hear My Song (1991) +971 Mediterraneo (1991) +972 Passion Fish (1992) +973 Grateful Dead (1995) +974 Eye for an Eye (1996) +975 Fear (1996) +976 Solo (1996) +977 Substitute, The (1996) +978 Heaven's Prisoners (1996) +979 Trigger Effect, The (1996) +980 Mother Night (1996) +981 Dangerous Ground (1997) +982 Maximum Risk (1996) +983 Rich Man's Wife, The (1996) +984 Shadow Conspiracy (1997) +985 Blood & Wine (1997) +986 Turbulence (1997) +987 Underworld (1997) +988 Beautician and the Beast, The (1997) +989 Cats Don't Dance (1997) +990 Anna Karenina (1997) +991 Keys to Tulsa (1997) +992 Head Above Water (1996) +993 Hercules (1997) +994 Last Time I Committed Suicide, The (1997) +995 Kiss Me, Guido (1997) +996 Big Green, The (1995) +997 Stuart Saves His Family (1995) +998 Cabin Boy (1994) +999 Clean Slate (1994) +1000 Lightning Jack (1994) +1001 Stupids, The (1996) +1002 Pest, The (1997) +1003 That Darn Cat! (1997) +1004 Geronimo: An American Legend (1993) +1005 Double vie de Vronique, La (Double Life of Veronique, The) (1991) +1006 Until the End of the World (Bis ans Ende der Welt) (1991) +1007 Waiting for Guffman (1996) +1008 I Shot Andy Warhol (1996) +1009 Stealing Beauty (1996) +1010 Basquiat (1996) +1011 2 Days in the Valley (1996) +1012 Private Parts (1997) +1013 Anaconda (1997) +1014 Romy and Michele's High School Reunion (1997) +1015 Shiloh (1997) +1016 Con Air (1997) +1017 Trees Lounge (1996) +1018 Tie Me Up! Tie Me Down! (1990) +1019 Die xue shuang xiong (Killer, The) (1989) +1020 Gaslight (1944) +1021 8 1/2 (1963) +1022 Fast, Cheap & Out of Control (1997) +1023 Fathers' Day (1997) +1024 Mrs. Dalloway (1997) +1025 Fire Down Below (1997) +1026 Lay of the Land, The (1997) +1027 Shooter, The (1995) +1028 Grumpier Old Men (1995) +1029 Jury Duty (1995) +1030 Beverly Hillbillies, The (1993) +1031 Lassie (1994) +1032 Little Big League (1994) +1033 Homeward Bound II: Lost in San Francisco (1996) +1034 Quest, The (1996) +1035 Cool Runnings (1993) +1036 Drop Dead Fred (1991) +1037 Grease 2 (1982) +1038 Switchback (1997) +1039 Hamlet (1996) +1040 Two if by Sea (1996) +1041 Forget Paris (1995) +1042 Just Cause (1995) +1043 Rent-a-Kid (1995) +1044 Paper, The (1994) +1045 Fearless (1993) +1046 Malice (1993) +1047 Multiplicity (1996) +1048 She's the One (1996) +1049 House Arrest (1996) +1050 Ghost and Mrs. Muir, The (1947) +1051 Associate, The (1996) +1052 Dracula: Dead and Loving It (1995) +1053 Now and Then (1995) +1054 Mr. Wrong (1996) +1055 Simple Twist of Fate, A (1994) +1056 Cronos (1992) +1057 Pallbearer, The (1996) +1058 War, The (1994) +1059 Don't Be a Menace to South Central While Drinking Your Juice in the Hood (1996) +1060 Adventures of Pinocchio, The (1996) +1061 Evening Star, The (1996) +1062 Four Days in September (1997) +1063 Little Princess, A (1995) +1064 Crossfire (1947) +1065 Koyaanisqatsi (1983) +1066 Balto (1995) +1067 Bottle Rocket (1996) +1068 Star Maker, The (Uomo delle stelle, L') (1995) +1069 Amateur (1994) +1070 Living in Oblivion (1995) +1071 Party Girl (1995) +1072 Pyromaniac's Love Story, A (1995) +1073 Shallow Grave (1994) +1074 Reality Bites (1994) +1075 Man of No Importance, A (1994) +1076 Pagemaster, The (1994) +1077 Love and a .45 (1994) +1078 Oliver & Company (1988) +1079 Joe's Apartment (1996) +1080 Celestial Clockwork (1994) +1081 Curdled (1996) +1082 Female Perversions (1996) +1083 Albino Alligator (1996) +1084 Anne Frank Remembered (1995) +1085 Carried Away (1996) +1086 It's My Party (1995) +1087 Bloodsport 2 (1995) +1088 Double Team (1997) +1089 Speed 2: Cruise Control (1997) +1090 Sliver (1993) +1091 Pete's Dragon (1977) +1092 Dear God (1996) +1093 Live Nude Girls (1995) +1094 Thin Line Between Love and Hate, A (1996) +1095 High School High (1996) +1096 Commandments (1997) +1097 Hate (Haine, La) (1995) +1098 Flirting With Disaster (1996) +1099 Red Firecracker, Green Firecracker (1994) +1100 What Happened Was... (1994) +1101 Six Degrees of Separation (1993) +1102 Two Much (1996) +1103 Trust (1990) +1104 C'est arriv prs de chez vous (1992) +1105 Firestorm (1998) +1106 Newton Boys, The (1998) +1107 Beyond Rangoon (1995) +1108 Feast of July (1995) +1109 Death and the Maiden (1994) +1110 Tank Girl (1995) +1111 Double Happiness (1994) +1112 Cobb (1994) +1113 Mrs. Parker and the Vicious Circle (1994) +1114 Faithful (1996) +1115 Twelfth Night (1996) +1116 Mark of Zorro, The (1940) +1117 Surviving Picasso (1996) +1118 Up in Smoke (1978) +1119 Some Kind of Wonderful (1987) +1120 I'm Not Rappaport (1996) +1121 Umbrellas of Cherbourg, The (Parapluies de Cherbourg, Les) (1964) +1122 They Made Me a Criminal (1939) +1123 Last Time I Saw Paris, The (1954) +1124 Farewell to Arms, A (1932) +1125 Innocents, The (1961) +1126 Old Man and the Sea, The (1958) +1127 Truman Show, The (1998) +1128 Heidi Fleiss: Hollywood Madam (1995) +1129 Chungking Express (1994) +1130 Jupiter's Wife (1994) +1131 Safe (1995) +1132 Feeling Minnesota (1996) +1133 Escape to Witch Mountain (1975) +1134 Get on the Bus (1996) +1135 Doors, The (1991) +1136 Ghosts of Mississippi (1996) +1137 Beautiful Thing (1996) +1138 Best Men (1997) +1139 Hackers (1995) +1140 Road to Wellville, The (1994) +1141 War Room, The (1993) +1142 When We Were Kings (1996) +1143 Hard Eight (1996) +1144 Quiet Room, The (1996) +1145 Blue Chips (1994) +1146 Calendar Girl (1993) +1147 My Family (1995) +1148 Tom & Viv (1994) +1149 Walkabout (1971) +1150 Last Dance (1996) +1151 Original Gangstas (1996) +1152 In Love and War (1996) +1153 Backbeat (1993) +1154 Alphaville (1965) +1155 Rendezvous in Paris (Rendez-vous de Paris, Les) (1995) +1156 Cyclo (1995) +1157 Relic, The (1997) +1158 Fille seule, La (A Single Girl) (1995) +1159 Stalker (1979) +1160 Love! Valour! Compassion! (1997) +1161 Palookaville (1996) +1162 Phat Beach (1996) +1163 Portrait of a Lady, The (1996) +1164 Zeus and Roxanne (1997) +1165 Big Bully (1996) +1166 Love & Human Remains (1993) +1167 Sum of Us, The (1994) +1168 Little Buddha (1993) +1169 Fresh (1994) +1170 Spanking the Monkey (1994) +1171 Wild Reeds (1994) +1172 Women, The (1939) +1173 Bliss (1997) +1174 Caught (1996) +1175 Hugo Pool (1997) +1176 Welcome To Sarajevo (1997) +1177 Dunston Checks In (1996) +1178 Major Payne (1994) +1179 Man of the House (1995) +1180 I Love Trouble (1994) +1181 Low Down Dirty Shame, A (1994) +1182 Cops and Robbersons (1994) +1183 Cowboy Way, The (1994) +1184 Endless Summer 2, The (1994) +1185 In the Army Now (1994) +1186 Inkwell, The (1994) +1187 Switchblade Sisters (1975) +1188 Young Guns II (1990) +1189 Prefontaine (1997) +1190 That Old Feeling (1997) +1191 Letter From Death Row, A (1998) +1192 Boys of St. Vincent, The (1993) +1193 Before the Rain (Pred dozhdot) (1994) +1194 Once Were Warriors (1994) +1195 Strawberry and Chocolate (Fresa y chocolate) (1993) +1196 Savage Nights (Nuits fauves, Les) (1992) +1197 Family Thing, A (1996) +1198 Purple Noon (1960) +1199 Cemetery Man (Dellamorte Dellamore) (1994) +1200 Kim (1950) +1201 Marlene Dietrich: Shadow and Light (1996) +1202 Maybe, Maybe Not (Bewegte Mann, Der) (1994) +1203 Top Hat (1935) +1204 To Be or Not to Be (1942) +1205 Secret Agent, The (1996) +1206 Amos & Andrew (1993) +1207 Jade (1995) +1208 Kiss of Death (1995) +1209 Mixed Nuts (1994) +1210 Virtuosity (1995) +1211 Blue Sky (1994) +1212 Flesh and Bone (1993) +1213 Guilty as Sin (1993) +1214 In the Realm of the Senses (Ai no corrida) (1976) +1215 Barb Wire (1996) +1216 Kissed (1996) +1217 Assassins (1995) +1218 Friday (1995) +1219 Goofy Movie, A (1995) +1220 Higher Learning (1995) +1221 When a Man Loves a Woman (1994) +1222 Judgment Night (1993) +1223 King of the Hill (1993) +1224 Scout, The (1994) +1225 Angus (1995) +1226 Night Falls on Manhattan (1997) +1227 Awfully Big Adventure, An (1995) +1228 Under Siege 2: Dark Territory (1995) +1229 Poison Ivy II (1995) +1230 Ready to Wear (Pret-A-Porter) (1994) +1231 Marked for Death (1990) +1232 Madonna: Truth or Dare (1991) +1233 Nnette et Boni (1996) +1234 Chairman of the Board (1998) +1235 Big Bang Theory, The (1994) +1236 Other Voices, Other Rooms (1997) +1237 Twisted (1996) +1238 Full Speed (1996) +1239 Cutthroat Island (1995) +1240 Ghost in the Shell (Kokaku kidotai) (1995) +1241 Van, The (1996) +1242 Old Lady Who Walked in the Sea, The (Vieille qui marchait dans la mer, La) (1991) +1243 Night Flier (1997) +1244 Metro (1997) +1245 Gridlock'd (1997) +1246 Bushwhacked (1995) +1247 Bad Girls (1994) +1248 Blink (1994) +1249 For Love or Money (1993) +1250 Best of the Best 3: No Turning Back (1995) +1251 A Chef in Love (1996) +1252 Contempt (Mpris, Le) (1963) +1253 Tie That Binds, The (1995) +1254 Gone Fishin' (1997) +1255 Broken English (1996) +1256 Designated Mourner, The (1997) +1257 Designated Mourner, The (1997) +1258 Trial and Error (1997) +1259 Pie in the Sky (1995) +1260 Total Eclipse (1995) +1261 Run of the Country, The (1995) +1262 Walking and Talking (1996) +1263 Foxfire (1996) +1264 Nothing to Lose (1994) +1265 Star Maps (1997) +1266 Bread and Chocolate (Pane e cioccolata) (1973) +1267 Clockers (1995) +1268 Bitter Moon (1992) +1269 Love in the Afternoon (1957) +1270 Life with Mikey (1993) +1271 North (1994) +1272 Talking About Sex (1994) +1273 Color of Night (1994) +1274 Robocop 3 (1993) +1275 Killer (Bulletproof Heart) (1994) +1276 Sunset Park (1996) +1277 Set It Off (1996) +1278 Selena (1997) +1279 Wild America (1997) +1280 Gang Related (1997) +1281 Manny & Lo (1996) +1282 Grass Harp, The (1995) +1283 Out to Sea (1997) +1284 Before and After (1996) +1285 Princess Caraboo (1994) +1286 Shall We Dance? (1937) +1287 Ed (1996) +1288 Denise Calls Up (1995) +1289 Jack and Sarah (1995) +1290 Country Life (1994) +1291 Celtic Pride (1996) +1292 Simple Wish, A (1997) +1293 Star Kid (1997) +1294 Ayn Rand: A Sense of Life (1997) +1295 Kicked in the Head (1997) +1296 Indian Summer (1996) +1297 Love Affair (1994) +1298 Band Wagon, The (1953) +1299 Penny Serenade (1941) +1300 'Til There Was You (1997) +1301 Stripes (1981) +1302 Late Bloomers (1996) +1303 Getaway, The (1994) +1304 New York Cop (1996) +1305 National Lampoon's Senior Trip (1995) +1306 Delta of Venus (1994) +1307 Carmen Miranda: Bananas Is My Business (1994) +1308 Babyfever (1994) +1309 Very Natural Thing, A (1974) +1310 Walk in the Sun, A (1945) +1311 Waiting to Exhale (1995) +1312 Pompatus of Love, The (1996) +1313 Palmetto (1998) +1314 Surviving the Game (1994) +1315 Inventing the Abbotts (1997) +1316 Horse Whisperer, The (1998) +1317 Journey of August King, The (1995) +1318 Catwalk (1995) +1319 Neon Bible, The (1995) +1320 Homage (1995) +1321 Open Season (1996) +1322 Metisse (Caf au Lait) (1993) +1323 Wooden Man's Bride, The (Wu Kui) (1994) +1324 Loaded (1994) +1325 August (1996) +1326 Boys (1996) +1327 Captives (1994) +1328 Of Love and Shadows (1994) +1329 Low Life, The (1994) +1330 An Unforgettable Summer (1994) +1331 Last Klezmer: Leopold Kozlowski, His Life and Music, The (1995) +1332 My Life and Times With Antonin Artaud (En compagnie d'Antonin Artaud) (1993) +1333 Midnight Dancers (Sibak) (1994) +1334 Somebody to Love (1994) +1335 American Buffalo (1996) +1336 Kazaam (1996) +1337 Larger Than Life (1996) +1338 Two Deaths (1995) +1339 Stefano Quantestorie (1993) +1340 Crude Oasis, The (1995) +1341 Hedd Wyn (1992) +1342 Convent, The (Convento, O) (1995) +1343 Lotto Land (1995) +1344 Story of Xinghua, The (1993) +1345 Day the Sun Turned Cold, The (Tianguo niezi) (1994) +1346 Dingo (1992) +1347 Ballad of Narayama, The (Narayama Bushiko) (1958) +1348 Every Other Weekend (1990) +1349 Mille bolle blu (1993) +1350 Crows and Sparrows (1949) +1351 Lover's Knot (1996) +1352 Shadow of Angels (Schatten der Engel) (1976) +1353 1-900 (1994) +1354 Venice/Venice (1992) +1355 Infinity (1996) +1356 Ed's Next Move (1996) +1357 For the Moment (1994) +1358 The Deadly Cure (1996) +1359 Boys in Venice (1996) +1360 Sexual Life of the Belgians, The (1994) +1361 Search for One-eye Jimmy, The (1996) +1362 American Strays (1996) +1363 Leopard Son, The (1996) +1364 Bird of Prey (1996) +1365 Johnny 100 Pesos (1993) +1366 JLG/JLG - autoportrait de dcembre (1994) +1367 Faust (1994) +1368 Mina Tannenbaum (1994) +1369 Forbidden Christ, The (Cristo proibito, Il) (1950) +1370 I Can't Sleep (J'ai pas sommeil) (1994) +1371 Machine, The (1994) +1372 Stranger, The (1994) +1373 Good Morning (1971) +1374 Falling in Love Again (1980) +1375 Cement Garden, The (1993) +1376 Meet Wally Sparks (1997) +1377 Hotel de Love (1996) +1378 Rhyme & Reason (1997) +1379 Love and Other Catastrophes (1996) +1380 Hollow Reed (1996) +1381 Losing Chase (1996) +1382 Bonheur, Le (1965) +1383 Second Jungle Book: Mowgli & Baloo, The (1997) +1384 Squeeze (1996) +1385 Roseanna's Grave (For Roseanna) (1997) +1386 Tetsuo II: Body Hammer (1992) +1387 Fall (1997) +1388 Gabbeh (1996) +1389 Mondo (1996) +1390 Innocent Sleep, The (1995) +1391 For Ever Mozart (1996) +1392 Locusts, The (1997) +1393 Stag (1997) +1394 Swept from the Sea (1997) +1395 Hurricane Streets (1998) +1396 Stonewall (1995) +1397 Of Human Bondage (1934) +1398 Anna (1996) +1399 Stranger in the House (1997) +1400 Picture Bride (1995) +1401 M. Butterfly (1993) +1402 Ciao, Professore! (1993) +1403 Caro Diario (Dear Diary) (1994) +1404 Withnail and I (1987) +1405 Boy's Life 2 (1997) +1406 When Night Is Falling (1995) +1407 Specialist, The (1994) +1408 Gordy (1995) +1409 Swan Princess, The (1994) +1410 Harlem (1993) +1411 Barbarella (1968) +1412 Land Before Time III: The Time of the Great Giving (1995) (V) +1413 Street Fighter (1994) +1414 Coldblooded (1995) +1415 Next Karate Kid, The (1994) +1416 No Escape (1994) +1417 Turning, The (1992) +1418 Joy Luck Club, The (1993) +1419 Highlander III: The Sorcerer (1994) +1420 Gilligan's Island: The Movie (1998) +1421 My Crazy Life (Mi vida loca) (1993) +1422 Suture (1993) +1423 Walking Dead, The (1995) +1424 I Like It Like That (1994) +1425 I'll Do Anything (1994) +1426 Grace of My Heart (1996) +1427 Drunks (1995) +1428 SubUrbia (1997) +1429 Sliding Doors (1998) +1430 Ill Gotten Gains (1997) +1431 Legal Deceit (1997) +1432 Mighty, The (1998) +1433 Men of Means (1998) +1434 Shooting Fish (1997) +1435 Steal Big, Steal Little (1995) +1436 Mr. Jones (1993) +1437 House Party 3 (1994) +1438 Panther (1995) +1439 Jason's Lyric (1994) +1440 Above the Rim (1994) +1441 Moonlight and Valentino (1995) +1442 Scarlet Letter, The (1995) +1443 8 Seconds (1994) +1444 That Darn Cat! (1965) +1445 Ladybird Ladybird (1994) +1446 Bye Bye, Love (1995) +1447 Century (1993) +1448 My Favorite Season (1993) +1449 Pather Panchali (1955) +1450 Golden Earrings (1947) +1451 Foreign Correspondent (1940) +1452 Lady of Burlesque (1943) +1453 Angel on My Shoulder (1946) +1454 Angel and the Badman (1947) +1455 Outlaw, The (1943) +1456 Beat the Devil (1954) +1457 Love Is All There Is (1996) +1458 Damsel in Distress, A (1937) +1459 Madame Butterfly (1995) +1460 Sleepover (1995) +1461 Here Comes Cookie (1935) +1462 Thieves (Voleurs, Les) (1996) +1463 Boys, Les (1997) +1464 Stars Fell on Henrietta, The (1995) +1465 Last Summer in the Hamptons (1995) +1466 Margaret's Museum (1995) +1467 Saint of Fort Washington, The (1993) +1468 Cure, The (1995) +1469 Tom and Huck (1995) +1470 Gumby: The Movie (1995) +1471 Hideaway (1995) +1472 Visitors, The (Visiteurs, Les) (1993) +1473 Little Princess, The (1939) +1474 Nina Takes a Lover (1994) +1475 Bhaji on the Beach (1993) +1476 Raw Deal (1948) +1477 Nightwatch (1997) +1478 Dead Presidents (1995) +1479 Reckless (1995) +1480 Herbie Rides Again (1974) +1481 S.F.W. (1994) +1482 Gate of Heavenly Peace, The (1995) +1483 Man in the Iron Mask, The (1998) +1484 Jerky Boys, The (1994) +1485 Colonel Chabert, Le (1994) +1486 Girl in the Cadillac (1995) +1487 Even Cowgirls Get the Blues (1993) +1488 Germinal (1993) +1489 Chasers (1994) +1490 Fausto (1993) +1491 Tough and Deadly (1995) +1492 Window to Paris (1994) +1493 Modern Affair, A (1995) +1494 Mostro, Il (1994) +1495 Flirt (1995) +1496 Carpool (1996) +1497 Line King: Al Hirschfeld, The (1996) +1498 Farmer & Chase (1995) +1499 Grosse Fatigue (1994) +1500 Santa with Muscles (1996) +1501 Prisoner of the Mountains (Kavkazsky Plennik) (1996) +1502 Naked in New York (1994) +1503 Gold Diggers: The Secret of Bear Mountain (1995) +1504 Bewegte Mann, Der (1994) +1505 Killer: A Journal of Murder (1995) +1506 Nelly & Monsieur Arnaud (1995) +1507 Three Lives and Only One Death (1996) +1508 Babysitter, The (1995) +1509 Getting Even with Dad (1994) +1510 Mad Dog Time (1996) +1511 Children of the Revolution (1996) +1512 World of Apu, The (Apur Sansar) (1959) +1513 Sprung (1997) +1514 Dream With the Fishes (1997) +1515 Wings of Courage (1995) +1516 Wedding Gift, The (1994) +1517 Race the Sun (1996) +1518 Losing Isaiah (1995) +1519 New Jersey Drive (1995) +1520 Fear, The (1995) +1521 Mr. Wonderful (1993) +1522 Trial by Jury (1994) +1523 Good Man in Africa, A (1994) +1524 Kaspar Hauser (1993) +1525 Object of My Affection, The (1998) +1526 Witness (1985) +1527 Senseless (1998) +1528 Nowhere (1997) +1529 Underground (1995) +1530 Jefferson in Paris (1995) +1531 Far From Home: The Adventures of Yellow Dog (1995) +1532 Foreign Student (1994) +1533 I Don't Want to Talk About It (De eso no se habla) (1993) +1534 Twin Town (1997) +1535 Enfer, L' (1994) +1536 Aiqing wansui (1994) +1537 Cosi (1996) +1538 All Over Me (1997) +1539 Being Human (1993) +1540 Amazing Panda Adventure, The (1995) +1541 Beans of Egypt, Maine, The (1994) +1542 Scarlet Letter, The (1926) +1543 Johns (1996) +1544 It Takes Two (1995) +1545 Frankie Starlight (1995) +1546 Shadows (Cienie) (1988) +1547 Show, The (1995) +1548 The Courtyard (1995) +1549 Dream Man (1995) +1550 Destiny Turns on the Radio (1995) +1551 Glass Shield, The (1994) +1552 Hunted, The (1995) +1553 Underneath, The (1995) +1554 Safe Passage (1994) +1555 Secret Adventures of Tom Thumb, The (1993) +1556 Condition Red (1995) +1557 Yankee Zulu (1994) +1558 Aparajito (1956) +1559 Hostile Intentions (1994) +1560 Clean Slate (Coup de Torchon) (1981) +1561 Tigrero: A Film That Was Never Made (1994) +1562 Eye of Vichy, The (Oeil de Vichy, L') (1993) +1563 Promise, The (Versprechen, Das) (1994) +1564 To Cross the Rubicon (1991) +1565 Daens (1992) +1566 Man from Down Under, The (1943) +1567 Careful (1992) +1568 Vermont Is For Lovers (1992) +1569 Vie est belle, La (Life is Rosey) (1987) +1570 Quartier Mozart (1992) +1571 Touki Bouki (Journey of the Hyena) (1973) +1572 Wend Kuuni (God's Gift) (1982) +1573 Spirits of the Dead (Tre passi nel delirio) (1968) +1574 Pharaoh's Army (1995) +1575 I, Worst of All (Yo, la peor de todas) (1990) +1576 Hungarian Fairy Tale, A (1987) +1577 Death in the Garden (Mort en ce jardin, La) (1956) +1578 Collectionneuse, La (1967) +1579 Baton Rouge (1988) +1580 Liebelei (1933) +1581 Woman in Question, The (1950) +1582 T-Men (1947) +1583 Invitation, The (Zaproszenie) (1986) +1584 Symphonie pastorale, La (1946) +1585 American Dream (1990) +1586 Lashou shentan (1992) +1587 Terror in a Texas Town (1958) +1588 Salut cousin! (1996) +1589 Schizopolis (1996) +1590 To Have, or Not (1995) +1591 Duoluo tianshi (1995) +1592 Magic Hour, The (1998) +1593 Death in Brunswick (1991) +1594 Everest (1998) +1595 Shopping (1994) +1596 Nemesis 2: Nebula (1995) +1597 Romper Stomper (1992) +1598 City of Industry (1997) +1599 Someone Else's America (1995) +1600 Guantanamera (1994) +1601 Office Killer (1997) +1602 Price Above Rubies, A (1998) +1603 Angela (1995) +1604 He Walked by Night (1948) +1605 Love Serenade (1996) +1606 Deceiver (1997) +1607 Hurricane Streets (1998) +1608 Buddy (1997) +1609 B*A*P*S (1997) +1610 Truth or Consequences, N.M. (1997) +1611 Intimate Relations (1996) +1612 Leading Man, The (1996) +1613 Tokyo Fist (1995) +1614 Reluctant Debutante, The (1958) +1615 Warriors of Virtue (1997) +1616 Desert Winds (1995) +1617 Hugo Pool (1997) +1618 King of New York (1990) +1619 All Things Fair (1996) +1620 Sixth Man, The (1997) +1621 Butterfly Kiss (1995) +1622 Paris, France (1993) +1623 Crmonie, La (1995) +1624 Hush (1998) +1625 Nightwatch (1997) +1626 Nobody Loves Me (Keiner liebt mich) (1994) +1627 Wife, The (1995) +1628 Lamerica (1994) +1629 Nico Icon (1995) +1630 Silence of the Palace, The (Saimt el Qusur) (1994) +1631 Slingshot, The (1993) +1632 Land and Freedom (Tierra y libertad) (1995) +1633 kldum klaka (Cold Fever) (1994) +1634 Etz Hadomim Tafus (Under the Domin Tree) (1994) +1635 Two Friends (1986) +1636 Brothers in Trouble (1995) +1637 Girls Town (1996) +1638 Normal Life (1996) +1639 Bitter Sugar (Azucar Amargo) (1996) +1640 Eighth Day, The (1996) +1641 Dadetown (1995) +1642 Some Mother's Son (1996) +1643 Angel Baby (1995) +1644 Sudden Manhattan (1996) +1645 Butcher Boy, The (1998) +1646 Men With Guns (1997) +1647 Hana-bi (1997) +1648 Niagara, Niagara (1997) +1649 Big One, The (1997) +1650 Butcher Boy, The (1998) +1651 Spanish Prisoner, The (1997) +1652 Temptress Moon (Feng Yue) (1996) +1653 Entertaining Angels: The Dorothy Day Story (1996) +1654 Chairman of the Board (1998) +1655 Favor, The (1994) +1656 Little City (1998) +1657 Target (1995) +1658 Substance of Fire, The (1996) +1659 Getting Away With Murder (1996) +1660 Small Faces (1995) +1661 New Age, The (1994) +1662 Rough Magic (1995) +1663 Nothing Personal (1995) +1664 8 Heads in a Duffel Bag (1997) +1665 Brother's Kiss, A (1997) +1666 Ripe (1996) +1667 Next Step, The (1995) +1668 Wedding Bell Blues (1996) +1669 MURDER and murder (1996) +1670 Tainted (1998) +1671 Further Gesture, A (1996) +1672 Kika (1993) +1673 Mirage (1995) +1674 Mamma Roma (1962) +1675 Sunchaser, The (1996) +1676 War at Home, The (1996) +1677 Sweet Nothing (1995) +1678 Mat' i syn (1997) +1679 B. Monkey (1998) +1680 Sliding Doors (1998) +1681 You So Crazy (1994) +1682 Scream of Stone (Schrei aus Stein) (1991) diff --git a/Exercise8/Figures/gaussian_fit.png b/Exercise8/Figures/gaussian_fit.png new file mode 100755 index 0000000..da08dda Binary files /dev/null and b/Exercise8/Figures/gaussian_fit.png differ diff --git a/Exercise8/exercise8.ipynb b/Exercise8/exercise8.ipynb new file mode 100755 index 0000000..3e4aa60 --- /dev/null +++ b/Exercise8/exercise8.ipynb @@ -0,0 +1,1028 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Programming Exercise 8:\n", + "# Anomaly Detection and Recommender Systems\n", + "\n", + "\n", + "## Introduction \n", + "\n", + "In this exercise, you will implement the anomaly detection algorithm and\n", + "apply it to detect failing servers on a network. In the second part, you will\n", + "use collaborative filtering to build a recommender system for movies. Before\n", + "starting on the programming exercise, we strongly recommend watching the\n", + "video lectures and completing the review questions for the associated topics.\n", + "\n", + "All the information you need for solving this assignment is in this notebook, and all the code you will be implementing will take place within this notebook. The assignment can be promptly submitted to the coursera grader directly from this notebook (code and instructions are included below).\n", + "\n", + "Before we begin with the exercises, we need to import all libraries required for this programming exercise. Throughout the course, we will be using [`numpy`](http://www.numpy.org/) for all arrays and matrix operations, [`matplotlib`](https://matplotlib.org/) for plotting, and [`scipy`](https://docs.scipy.org/doc/scipy/reference/) for scientific and numerical computation functions and tools. You can find instructions on how to install required libraries in the README file in the [github repository](https://github.com/dibgerge/ml-coursera-python-assignments)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# used for manipulating directory paths\n", + "import os\n", + "\n", + "# Scientific and vector computation for python\n", + "import numpy as np\n", + "\n", + "# Plotting library\n", + "from matplotlib import pyplot\n", + "import matplotlib as mpl\n", + "\n", + "# Optimization module in scipy\n", + "from scipy import optimize\n", + "\n", + "# will be used to load MATLAB mat datafile format\n", + "from scipy.io import loadmat\n", + "\n", + "# library written for this exercise providing additional functions for assignment submission, and others\n", + "import utils\n", + "\n", + "# define the submission/grader object for this exercise\n", + "grader = utils.Grader()\n", + "\n", + "# tells matplotlib to embed plots within the notebook\n", + "%matplotlib inline" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Submission and Grading\n", + "\n", + "\n", + "After completing each part of the assignment, be sure to submit your solutions to the grader. The following is a breakdown of how each part of this exercise is scored.\n", + "\n", + "\n", + "| Section | Part | Submitted Function | Points |\n", + "| :- |:- |:- | :-: |\n", + "| 1 | [Estimate Gaussian Parameters](#section1) | [`estimateGaussian`](#estimateGaussian) | 15 |\n", + "| 2 | [Select Threshold](#section2) | [`selectThreshold`](#selectThreshold) | 15 |\n", + "| 3 | [Collaborative Filtering Cost](#section3) | [`cofiCostFunc`](#cofiCostFunc) | 20 |\n", + "| 4 | [Collaborative Filtering Gradient](#section4) | [`cofiCostFunc`](#cofiCostFunc) | 30 |\n", + "| 5 | [Regularized Cost](#section5) | [`cofiCostFunc`](#cofiCostFunc) | 10 |\n", + "| 6 | [Gradient with regularization](#section6) | [`cofiCostFunc`](#cofiCostFunc) | 10 |\n", + "| | Total Points | |100 |\n", + "\n", + "\n", + "\n", + "You are allowed to submit your solutions multiple times, and we will take only the highest score into consideration.\n", + "\n", + "
\n", + "At the end of each section in this notebook, we have a cell which contains code for submitting the solutions thus far to the grader. Execute the cell to see your score up to the current section. For all your work to be submitted properly, you must execute those cells at least once.\n", + "
" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 1 Anomaly Detection \n", + "\n", + "In this exercise, you will implement an anomaly detection algorithm to detect anomalous behavior in server computers. The features measure the throughput (mb/s) and latency (ms) of response of each server. While your servers were operating, you collected $m = 307$ examples of how they were behaving, and thus have an unlabeled dataset $\\{x^{(1)}, \\dots, x^{(m)}\\}$. You suspect that the vast majority of these examples are “normal” (non-anomalous) examples of the servers operating normally, but there might also be some examples of servers acting anomalously within this dataset.\n", + "\n", + "You will use a Gaussian model to detect anomalous examples in your dataset. You will first start on a 2D dataset that will allow you to visualize what the algorithm is doing. On that dataset you will fit a Gaussian distribution and then find values that have very low probability and hence can be considered anomalies. After that, you will apply the anomaly detection algorithm to a larger dataset with many dimensions.\n", + "\n", + "We start this exercise by using a small dataset that is easy to visualize. Our example case consists of 2 network server statistics across several machines: the latency and throughput of each machine. " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# The following command loads the dataset.\n", + "data = loadmat(os.path.join('Data', 'ex8data1.mat'))\n", + "X, Xval, yval = data['X'], data['Xval'], data['yval'][:, 0]\n", + "\n", + "# Visualize the example dataset\n", + "pyplot.plot(X[:, 0], X[:, 1], 'bx', mew=2, mec='k', ms=6)\n", + "pyplot.axis([0, 30, 0, 30])\n", + "pyplot.xlabel('Latency (ms)')\n", + "pyplot.ylabel('Throughput (mb/s)')\n", + "pass" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 1.1 Gaussian distribution\n", + "\n", + "To perform anomaly detection, you will first need to fit a model to the data's distribution. Given a training set $\\{x^{(1)}, \\dots, x^{(m)} \\}$ (where $x^{(i)} \\in \\mathbb{R}^n$ ), you want to estimate the Gaussian distribution for each of the features $x_i$ . For each feature $i = 1 \\dots n$, you need to find parameters $\\mu_i$ and $\\sigma_i^2$ that fit the data in the $i^{th}$ dimension $\\{ x_i^{(1)}, \\dots, x_i^{(m)} \\}$ (the $i^{th}$ dimension of each example).\n", + "\n", + "The Gaussian distribution is given by\n", + "\n", + "$$ p\\left( x; \\mu, \\sigma^2 \\right) = \\frac{1}{\\sqrt{2\\pi\\sigma^2}} e^{-\\frac{\\left(x-\\mu\\right)^2}{2\\sigma^2}},$$\n", + "where $\\mu$ is the mean and $\\sigma^2$ is the variance.\n", + "\n", + "\n", + "### 1.2 Estimating parameters for a Gaussian \n", + "\n", + "You can estimate the parameters $\\left( \\mu_i, \\sigma_i^2 \\right)$, of the $i^{th}$ feature by using the following equations. To estimate the mean, you will use: \n", + "\n", + "$$ \\mu_i = \\frac{1}{m} \\sum_{j=1}^m x_i^{(j)},$$\n", + "\n", + "and for the variance you will use:\n", + "\n", + "$$ \\sigma_i^2 = \\frac{1}{m} \\sum_{j=1}^m \\left( x_i^{(j)} - \\mu_i \\right)^2.$$\n", + "\n", + "Your task is to complete the code in the function `estimateGaussian`. This function takes as input the data matrix `X` and should output an n-dimension vector `mu` that holds the mean for each of the $n$ features and another n-dimension vector `sigma2` that holds the variances of each of the features. You can implement this\n", + "using a for-loop over every feature and every training example (though a vectorized implementation might be more efficient; feel free to use a vectorized implementation if you prefer). \n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def estimateGaussian(X):\n", + " \"\"\"\n", + " This function estimates the parameters of a Gaussian distribution\n", + " using a provided dataset.\n", + " \n", + " Parameters\n", + " ----------\n", + " X : array_like\n", + " The dataset of shape (m x n) with each n-dimensional \n", + " data point in one row, and each total of m data points.\n", + " \n", + " Returns\n", + " -------\n", + " mu : array_like \n", + " A vector of shape (n,) containing the means of each dimension.\n", + " \n", + " sigma2 : array_like\n", + " A vector of shape (n,) containing the computed\n", + " variances of each dimension.\n", + " \n", + " Instructions\n", + " ------------\n", + " Compute the mean of the data and the variances\n", + " In particular, mu[i] should contain the mean of\n", + " the data for the i-th feature and sigma2[i]\n", + " should contain variance of the i-th feature.\n", + " \"\"\"\n", + " # Useful variables\n", + " m, n = X.shape\n", + "\n", + " # You should return these values correctly\n", + " mu = np.zeros(n)\n", + " sigma2 = np.zeros(n)\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + " \n", + " # =============================================================\n", + " return mu, sigma2" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Once you have completed the code in `estimateGaussian`, the next cell will visualize the contours of the fitted Gaussian distribution. You should get a plot similar to the figure below.\n", + "\n", + "![](Figures/gaussian_fit.png)\n", + "\n", + "From your plot, you can see that most of the examples are in the region with the highest probability, while\n", + "the anomalous examples are in the regions with lower probabilities.\n", + "\n", + "To do the visualization of the Gaussian fit, we first estimate the parameters of our assumed Gaussian distribution, then compute the probabilities for each of the points and then visualize both the overall distribution and where each of the points falls in terms of that distribution." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Estimate my and sigma2\n", + "mu, sigma2 = estimateGaussian(X)\n", + "\n", + "# Returns the density of the multivariate normal at each data point (row) \n", + "# of X\n", + "p = utils.multivariateGaussian(X, mu, sigma2)\n", + "\n", + "# Visualize the fit\n", + "utils.visualizeFit(X, mu, sigma2)\n", + "pyplot.xlabel('Latency (ms)')\n", + "pyplot.ylabel('Throughput (mb/s)')\n", + "pyplot.tight_layout()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[1] = estimateGaussian\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "### 1.3 Selecting the threshold, $\\varepsilon$\n", + "\n", + "Now that you have estimated the Gaussian parameters, you can investigate which examples have a very high probability given this distribution and which examples have a very low probability. The low probability examples are more likely to be the anomalies in our dataset. One way to determine which examples are anomalies is to select a threshold based on a cross validation set. In this part of the exercise, you will implement an algorithm to select the threshold $\\varepsilon$ using the $F_1$ score on a cross validation set.\n", + "\n", + "\n", + "You should now complete the code for the function `selectThreshold`. For this, we will use a cross validation set $\\{ (x_{cv}^{(1)}, y_{cv}^{(1)}), \\dots, (x_{cv}^{(m_{cv})}, y_{cv}^{(m_{cv})})\\}$, where the label $y = 1$ corresponds to an anomalous example, and $y = 0$ corresponds to a normal example. For each cross validation example, we will compute $p\\left( x_{cv}^{(i)}\\right)$. The vector of all of these probabilities $p\\left( x_{cv}^{(1)}\\right), \\dots, p\\left( x_{cv}^{(m_{cv})}\\right)$ is passed to `selectThreshold` in the vector `pval`. The corresponding labels $y_{cv}^{(1)} , \\dots , y_{cv}^{(m_{cv})}$ are passed to the same function in the vector `yval`.\n", + "\n", + "The function `selectThreshold` should return two values; the first is the selected threshold $\\varepsilon$. If an example $x$ has a low probability $p(x) < \\varepsilon$, then it is considered to be an anomaly. The function should also return the $F_1$ score, which tells you how well you are doing on finding the ground truth\n", + "anomalies given a certain threshold. For many different values of $\\varepsilon$, you will compute the resulting $F_1$ score by computing how many examples the current threshold classifies correctly and incorrectly.\n", + "\n", + "The $F_1$ score is computed using precision ($prec$) and recall ($rec$):\n", + "\n", + "$$ F_1 = \\frac{2 \\cdot prec \\cdot rec}{prec + rec}, $$\n", + "\n", + "You compute precision and recall by: \n", + "\n", + "$$ prec = \\frac{tp}{tp + fp} $$ \n", + "\n", + "$$ rec = \\frac{tp}{tp + fn} $$\n", + "\n", + "where: \n", + "\n", + "- $tp$ is the number of true positives: the ground truth label says it’s an anomaly and our algorithm correctly classified it as an anomaly.\n", + "\n", + "- $fp$ is the number of false positives: the ground truth label says it’s not an anomaly, but our algorithm incorrectly classified it as an anomaly.\n", + "- $fn$ is the number of false negatives: the ground truth label says it’s an anomaly, but our algorithm incorrectly classified it as not being anomalous.\n", + "\n", + "In the provided code `selectThreshold`, there is already a loop that will try many different values of $\\varepsilon$ and select the best $\\varepsilon$ based on the $F_1$ score. You should now complete the code in `selectThreshold`. You can implement the computation of the $F_1$ score using a for-loop over all the cross\n", + "validation examples (to compute the values $tp$, $fp$, $fn$). You should see a value for `epsilon` of about 8.99e-05.\n", + "\n", + "
\n", + "**Implementation Note:** In order to compute $tp$, $fp$ and $fn$, you may be able to use a vectorized implementation rather than loop over all the examples. This can be implemented by numpy's equality test\n", + "between a vector and a single number. If you have several binary values in an n-dimensional binary vector $v \\in \\{0, 1\\}^n$, you can find out how many values in this vector are 0 by using: np.sum(v == 0). You can also\n", + "apply a logical and operator to such binary vectors. For instance, let `cvPredictions` be a binary vector of size equal to the number of cross validation set, where the $i^{th}$ element is 1 if your algorithm considers\n", + "$x_{cv}^{(i)}$ an anomaly, and 0 otherwise. You can then, for example, compute the number of false positives using: `fp = np.sum((cvPredictions == 1) & (yval == 0))`.\n", + "
\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def selectThreshold(yval, pval):\n", + " \"\"\"\n", + " Find the best threshold (epsilon) to use for selecting outliers based\n", + " on the results from a validation set and the ground truth.\n", + " \n", + " Parameters\n", + " ----------\n", + " yval : array_like\n", + " The validation dataset of shape (m x n) where m is the number \n", + " of examples an n is the number of dimensions(features).\n", + " \n", + " pval : array_like\n", + " The ground truth labels of shape (m, ).\n", + " \n", + " Returns\n", + " -------\n", + " bestEpsilon : array_like\n", + " A vector of shape (n,) corresponding to the threshold value.\n", + " \n", + " bestF1 : float\n", + " The value for the best F1 score.\n", + " \n", + " Instructions\n", + " ------------\n", + " Compute the F1 score of choosing epsilon as the threshold and place the\n", + " value in F1. The code at the end of the loop will compare the\n", + " F1 score for this choice of epsilon and set it to be the best epsilon if\n", + " it is better than the current choice of epsilon.\n", + " \n", + " Notes\n", + " -----\n", + " You can use predictions = (pval < epsilon) to get a binary vector\n", + " of 0's and 1's of the outlier predictions\n", + " \"\"\"\n", + " bestEpsilon = 0\n", + " bestF1 = 0\n", + " F1 = 0\n", + " \n", + " for epsilon in np.linspace(1.01*min(pval), max(pval), 1000):\n", + " # ====================== YOUR CODE HERE =======================\n", + "\n", + " \n", + " \n", + "\n", + " # =============================================================\n", + " if F1 > bestF1:\n", + " bestF1 = F1\n", + " bestEpsilon = epsilon\n", + "\n", + " return bestEpsilon, bestF1" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Once you have completed the code in `selectThreshold`, the next cell will run your anomaly detection code and circle the anomalies in the plot." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "pval = utils.multivariateGaussian(Xval, mu, sigma2)\n", + "\n", + "epsilon, F1 = selectThreshold(yval, pval)\n", + "print('Best epsilon found using cross-validation: %.2e' % epsilon)\n", + "print('Best F1 on Cross Validation Set: %f' % F1)\n", + "print(' (you should see a value epsilon of about 8.99e-05)')\n", + "print(' (you should see a Best F1 value of 0.875000)')\n", + "\n", + "# Find the outliers in the training set and plot the\n", + "outliers = p < epsilon\n", + "\n", + "# Visualize the fit\n", + "utils.visualizeFit(X, mu, sigma2)\n", + "pyplot.xlabel('Latency (ms)')\n", + "pyplot.ylabel('Throughput (mb/s)')\n", + "pyplot.tight_layout()\n", + "\n", + "# Draw a red circle around those outliers\n", + "pyplot.plot(X[outliers, 0], X[outliers, 1], 'ro', ms=10, mfc='None', mew=2)\n", + "pass" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[2] = selectThreshold\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 1.4 High dimensional dataset\n", + "\n", + "The next cell will run the anomaly detection algorithm you implemented on a more realistic and much harder dataset. In this dataset, each example is described by 11 features, capturing many more properties of your compute servers, but only some features indicate whether a point is an outlier. The script will use your code to estimate the Gaussian parameters ($\\mu_i$ and $\\sigma_i^2$), evaluate the probabilities for both the training data `X` from which you estimated the Gaussian parameters, and do so for the the cross-validation set `Xval`. Finally, it will use `selectThreshold` to find the best threshold $\\varepsilon$. You should see a value epsilon of about 1.38e-18, and 117 anomalies found." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Loads the second dataset. You should now have the\n", + "# variables X, Xval, yval in your environment\n", + "data = loadmat(os.path.join('Data', 'ex8data2.mat'))\n", + "X, Xval, yval = data['X'], data['Xval'], data['yval'][:, 0]\n", + "\n", + "# Apply the same steps to the larger dataset\n", + "mu, sigma2 = estimateGaussian(X)\n", + "\n", + "# Training set \n", + "p = utils.multivariateGaussian(X, mu, sigma2)\n", + "\n", + "# Cross-validation set\n", + "pval = utils.multivariateGaussian(Xval, mu, sigma2)\n", + "\n", + "# Find the best threshold\n", + "epsilon, F1 = selectThreshold(yval, pval)\n", + "\n", + "print('Best epsilon found using cross-validation: %.2e' % epsilon)\n", + "print('Best F1 on Cross Validation Set : %f\\n' % F1)\n", + "print(' (you should see a value epsilon of about 1.38e-18)')\n", + "print(' (you should see a Best F1 value of 0.615385)')\n", + "print('\\n# Outliers found: %d' % np.sum(p < epsilon))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## 2 Recommender Systems\n", + "\n", + "In this part of the exercise, you will implement the collaborative filtering learning algorithm and apply it to a dataset of movie ratings ([MovieLens 100k Dataset](https://grouplens.org/datasets/movielens/) from GroupLens Research). This dataset consists of ratings on a scale of 1 to 5. The dataset has $n_u = 943$ users, and $n_m = 1682$ movies. \n", + "\n", + "In the next parts of this exercise, you will implement the function `cofiCostFunc` that computes the collaborative filtering objective function and gradient. After implementing the cost function and gradient, you will use `scipy.optimize.minimize` to learn the parameters for collaborative filtering.\n", + "\n", + "### 2.1 Movie ratings dataset\n", + "\n", + "The next cell will load the dataset `ex8_movies.mat`, providing the variables `Y` and `R`.\n", + "The matrix `Y` (a `num_movies` $\\times$ `num_users` matrix) stores the ratings $y^{(i,j)}$ (from 1 to 5). The matrix `R` is an binary-valued indicator matrix, where $R(i, j) = 1$ if user $j$ gave a rating to movie $i$, and $R(i, j) = 0$ otherwise. The objective of collaborative filtering is to predict movie ratings for the movies that users have not yet rated, that is, the entries with $R(i, j) = 0$. This will allow us to recommend the movies with the highest predicted ratings to the user.\n", + "\n", + "To help you understand the matrix `Y`, the following cell will compute the average movie rating for the first movie (Toy Story) and print its average rating." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Load data\n", + "data = loadmat(os.path.join('Data', 'ex8_movies.mat'))\n", + "Y, R = data['Y'], data['R']\n", + "\n", + "# Y is a 1682x943 matrix, containing ratings (1-5) of \n", + "# 1682 movies on 943 users\n", + "\n", + "# R is a 1682x943 matrix, where R(i,j) = 1 \n", + "# if and only if user j gave a rating to movie i\n", + "\n", + "# From the matrix, we can compute statistics like average rating.\n", + "print('Average rating for movie 1 (Toy Story): %f / 5' %\n", + " np.mean(Y[0, R[0, :]]))\n", + "\n", + "# We can \"visualize\" the ratings matrix by plotting it with imshow\n", + "pyplot.figure(figsize=(8, 8))\n", + "pyplot.imshow(Y)\n", + "pyplot.ylabel('Movies')\n", + "pyplot.xlabel('Users')\n", + "pyplot.grid(False)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Throughout this part of the exercise, you will also be working with the matrices, `X` and `Theta`:\n", + "\n", + "$$ \\text{X} = \n", + "\\begin{bmatrix}\n", + "- \\left(x^{(1)}\\right)^T - \\\\\n", + "- \\left(x^{(2)}\\right)^T - \\\\\n", + "\\vdots \\\\\n", + "- \\left(x^{(n_m)}\\right)^T - \\\\\n", + "\\end{bmatrix}, \\quad\n", + "\\text{Theta} = \n", + "\\begin{bmatrix}\n", + "- \\left(\\theta^{(1)}\\right)^T - \\\\\n", + "- \\left(\\theta^{(2)}\\right)^T - \\\\\n", + "\\vdots \\\\\n", + "- \\left(\\theta^{(n_u)}\\right)^T - \\\\\n", + "\\end{bmatrix}.\n", + "$$\n", + "\n", + "The $i^{th}$ row of `X` corresponds to the feature vector $x^{(i)}$ for the $i^{th}$ movie, and the $j^{th}$ row of `Theta` corresponds to one parameter vector $\\theta^{(j)}$, for the $j^{th}$ user. Both $x^{(i)}$ and $\\theta^{(j)}$ are n-dimensional vectors. For the purposes of this exercise, you will use $n = 100$, and therefore, $x^{(i)} \\in \\mathbb{R}^{100}$ and $\\theta^{(j)} \\in \\mathbb{R}^{100}$. Correspondingly, `X` is a $n_m \\times 100$ matrix and `Theta` is a $n_u \\times 100$ matrix.\n", + "\n", + "\n", + "### 2.2 Collaborative filtering learning algorithm\n", + "\n", + "Now, you will start implementing the collaborative filtering learning algorithm. You will start by implementing the cost function (without regularization).\n", + "\n", + "The collaborative filtering algorithm in the setting of movie recommendations considers a set of n-dimensional parameter vectors $x^{(1)}, \\dots, x^{(n_m)}$ and $\\theta^{(1)} , \\dots, \\theta^{(n_u)}$, where the model predicts the rating for movie $i$ by user $j$ as $y^{(i,j)} = \\left( \\theta^{(j)} \\right)^T x^{(i)}$. Given a dataset that consists of a set of ratings produced by some users on some movies, you wish to learn the parameter vectors $x^{(1)}, \\dots, x^{(n_m)}, \\theta^{(1)}, \\dots, \\theta^{(n_u)}$ that produce the best fit (minimizes the squared error).\n", + "\n", + "You will complete the code in `cofiCostFunc` to compute the cost function and gradient for collaborative filtering. Note that the parameters to the function (i.e., the values that you are trying to learn) are `X` and `Theta`. In order to use an off-the-shelf minimizer such as `scipy`'s `minimize` function, the cost function has been set up to unroll the parameters into a single vector called `params`. You had previously used the same vector unrolling method in the neural networks programming exercise.\n", + "\n", + "#### 2.2.1 Collaborative filtering cost function\n", + "\n", + "The collaborative filtering cost function (without regularization) is given by\n", + "\n", + "$$\n", + "J(x^{(1)}, \\dots, x^{(n_m)}, \\theta^{(1)}, \\dots,\\theta^{(n_u)}) = \\frac{1}{2} \\sum_{(i,j):r(i,j)=1} \\left( \\left(\\theta^{(j)}\\right)^T x^{(i)} - y^{(i,j)} \\right)^2\n", + "$$\n", + "\n", + "You should now modify the function `cofiCostFunc` to return this cost in the variable `J`. Note that you should be accumulating the cost for user $j$ and movie $i$ only if `R[i,j] = 1`.\n", + "\n", + "
\n", + "**Implementation Note**: We strongly encourage you to use a vectorized implementation to compute $J$, since it will later by called many times by `scipy`'s optimization package. As usual, it might be easiest to first write a non-vectorized implementation (to make sure you have the right answer), and the modify it to become a vectorized implementation (checking that the vectorization steps do not change your algorithm’s output). To come up with a vectorized implementation, the following tip might be helpful: You can use the $R$ matrix to set selected entries to 0. For example, `R * M` will do an element-wise multiplication between `M`\n", + "and `R`; since `R` only has elements with values either 0 or 1, this has the effect of setting the elements of M to 0 only when the corresponding value in R is 0. Hence, `np.sum( R * M)` is the sum of all the elements of `M` for which the corresponding element in `R` equals 1.\n", + "
\n", + "\n", + "" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def cofiCostFunc(params, Y, R, num_users, num_movies,\n", + " num_features, lambda_=0.0):\n", + " \"\"\"\n", + " Collaborative filtering cost function.\n", + " \n", + " Parameters\n", + " ----------\n", + " params : array_like\n", + " The parameters which will be optimized. This is a one\n", + " dimensional vector of shape (num_movies x num_users, 1). It is the \n", + " concatenation of the feature vectors X and parameters Theta.\n", + " \n", + " Y : array_like\n", + " A matrix of shape (num_movies x num_users) of user ratings of movies.\n", + " \n", + " R : array_like\n", + " A (num_movies x num_users) matrix, where R[i, j] = 1 if the \n", + " i-th movie was rated by the j-th user.\n", + " \n", + " num_users : int\n", + " Total number of users.\n", + " \n", + " num_movies : int\n", + " Total number of movies.\n", + " \n", + " num_features : int\n", + " Number of features to learn.\n", + " \n", + " lambda_ : float, optional\n", + " The regularization coefficient.\n", + " \n", + " Returns\n", + " -------\n", + " J : float\n", + " The value of the cost function at the given params.\n", + " \n", + " grad : array_like\n", + " The gradient vector of the cost function at the given params.\n", + " grad has a shape (num_movies x num_users, 1)\n", + " \n", + " Instructions\n", + " ------------\n", + " Compute the cost function and gradient for collaborative filtering.\n", + " Concretely, you should first implement the cost function (without\n", + " regularization) and make sure it is matches our costs. After that,\n", + " you should implement thegradient and use the checkCostFunction routine \n", + " to check that the gradient is correct. Finally, you should implement\n", + " regularization.\n", + " \n", + " Notes\n", + " -----\n", + " - The input params will be unraveled into the two matrices:\n", + " X : (num_movies x num_features) matrix of movie features\n", + " Theta : (num_users x num_features) matrix of user features\n", + "\n", + " - You should set the following variables correctly:\n", + "\n", + " X_grad : (num_movies x num_features) matrix, containing the \n", + " partial derivatives w.r.t. to each element of X\n", + " Theta_grad : (num_users x num_features) matrix, containing the \n", + " partial derivatives w.r.t. to each element of Theta\n", + "\n", + " - The returned gradient will be the concatenation of the raveled \n", + " gradients X_grad and Theta_grad.\n", + " \"\"\"\n", + " # Unfold the U and W matrices from params\n", + " X = params[:num_movies*num_features].reshape(num_movies, num_features)\n", + " Theta = params[num_movies*num_features:].reshape(num_users, num_features)\n", + "\n", + " # You need to return the following values correctly\n", + " J = 0\n", + " X_grad = np.zeros(X.shape)\n", + " Theta_grad = np.zeros(Theta.shape)\n", + "\n", + " # ====================== YOUR CODE HERE ======================\n", + "\n", + " \n", + " \n", + " # =============================================================\n", + " \n", + " grad = np.concatenate([X_grad.ravel(), Theta_grad.ravel()])\n", + " return J, grad" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "After you have completed the function, the next cell will run your cost function. To help you debug your cost function, we have included set of weights that we trained on that. You should expect to see an output of 22.22." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Load pre-trained weights (X, Theta, num_users, num_movies, num_features)\n", + "data = loadmat(os.path.join('Data', 'ex8_movieParams.mat'))\n", + "X, Theta, num_users, num_movies, num_features = data['X'],\\\n", + " data['Theta'], data['num_users'], data['num_movies'], data['num_features']\n", + "\n", + "# Reduce the data set size so that this runs faster\n", + "num_users = 4\n", + "num_movies = 5\n", + "num_features = 3\n", + "\n", + "X = X[:num_movies, :num_features]\n", + "Theta = Theta[:num_users, :num_features]\n", + "Y = Y[:num_movies, 0:num_users]\n", + "R = R[:num_movies, 0:num_users]\n", + "\n", + "# Evaluate cost function\n", + "J, _ = cofiCostFunc(np.concatenate([X.ravel(), Theta.ravel()]),\n", + " Y, R, num_users, num_movies, num_features)\n", + " \n", + "print('Cost at loaded parameters: %.2f \\n(this value should be about 22.22)' % J)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[3] = cofiCostFunc\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "#### 2.2.2 Collaborative filtering gradient\n", + "\n", + "Now you should implement the gradient (without regularization). Specifically, you should complete the code in `cofiCostFunc` to return the variables `X_grad` and `Theta_grad`. Note that `X_grad` should be a matrix of the same size as `X` and similarly, `Theta_grad` is a matrix of the same size as\n", + "`Theta`. The gradients of the cost function is given by:\n", + "\n", + "$$ \\frac{\\partial J}{\\partial x_k^{(i)}} = \\sum_{j:r(i,j)=1} \\left( \\left(\\theta^{(j)}\\right)^T x^{(i)} - y^{(i,j)} \\right) \\theta_k^{(j)} $$\n", + "\n", + "$$ \\frac{\\partial J}{\\partial \\theta_k^{(j)}} = \\sum_{i:r(i,j)=1} \\left( \\left(\\theta^{(j)}\\right)^T x^{(i)}- y^{(i,j)} \\right) x_k^{(j)} $$\n", + "\n", + "Note that the function returns the gradient for both sets of variables by unrolling them into a single vector. After you have completed the code to compute the gradients, the next cell run a gradient check\n", + "(available in `utils.checkCostFunction`) to numerically check the implementation of your gradients (this is similar to the numerical check that you used in the neural networks exercise. If your implementation is correct, you should find that the analytical and numerical gradients match up closely.\n", + "\n", + "
\n", + "**Implementation Note:** You can get full credit for this assignment without using a vectorized implementation, but your code will run much more slowly (a small number of hours), and so we recommend that you try to vectorize your implementation. To get started, you can implement the gradient with a for-loop over movies\n", + "(for computing $\\frac{\\partial J}{\\partial x^{(i)}_k}$) and a for-loop over users (for computing $\\frac{\\partial J}{\\theta_k^{(j)}}$). When you first implement the gradient, you might start with an unvectorized version, by implementing another inner for-loop that computes each element in the summation. After you have completed the gradient computation this way, you should try to vectorize your implementation (vectorize the inner for-loops), so that you are left with only two for-loops (one for looping over movies to compute $\\frac{\\partial J}{\\partial x_k^{(i)}}$ for each movie, and one for looping over users to compute $\\frac{\\partial J}{\\partial \\theta_k^{(j)}}$ for each user).\n", + "
\n", + "\n", + "
\n", + "**Implementation Tip:** To perform the vectorization, you might find this helpful: You should come up with a way to compute all the derivatives associated with $x_1^{(i)} , x_2^{(i)}, \\dots , x_n^{(i)}$ (i.e., the derivative terms associated with the feature vector $x^{(i)}$) at the same time. Let us define the derivatives for the feature vector of the $i^{th}$ movie as:\n", + "\n", + "$$ \\left(X_{\\text{grad}} \\left(i, :\\right)\\right)^T = \n", + "\\begin{bmatrix}\n", + "\\frac{\\partial J}{\\partial x_1^{(i)}} \\\\\n", + "\\frac{\\partial J}{\\partial x_2^{(i)}} \\\\\n", + "\\vdots \\\\\n", + "\\frac{\\partial J}{\\partial x_n^{(i)}}\n", + "\\end{bmatrix} = \\quad\n", + "\\sum_{j:r(i,j)=1} \\left( \\left( \\theta^{(j)} \\right)^T x^{(i)} - y^{(i,j)} \\right) \\theta^{(j)}\n", + "$$\n", + "\n", + "To vectorize the above expression, you can start by indexing into `Theta` and `Y` to select only the elements of interests (that is, those with `r[i, j] = 1`). Intuitively, when you consider the features for the $i^{th}$ movie, you only need to be concerned about the users who had given ratings to the movie, and this allows you to remove all the other users from `Theta` and `Y`.

\n", + "\n", + "\n", + "Concretely, you can set `idx = np.where(R[i, :] == 1)[0]` to be a list of all the users that have rated movie $i$. This will allow you to create the temporary matrices `Theta_temp = Theta[idx, :]` and `Y_temp = Y[i, idx]` that index into `Theta` and `Y` to give you only the set of users which have rated the $i^{th}$ movie. This will allow you to write the derivatives as:
\n", + "\n", + "`X_grad[i, :] = np.dot(np.dot(X[i, :], Theta_temp.T) - Y_temp, Theta_temp)`\n", + "\n", + "

\n", + "Note that the vectorized computation above returns a row-vector instead. After you have vectorized the computations of the derivatives with respect to $x^{(i)}$, you should use a similar method to vectorize the derivatives with respect to $θ^{(j)}$ as well.\n", + "
\n", + "\n", + "[Click here to go back to the function `cofiCostFunc` to update it](#cofiCostFunc). \n", + "\n", + " Do not forget to re-execute the cell containg the function `cofiCostFunc` so that it is updated with your implementation of the gradient computation." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Check gradients by running checkcostFunction\n", + "utils.checkCostFunction(cofiCostFunc)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[4] = cofiCostFunc\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "#### 2.2.3 Regularized cost function\n", + "\n", + "The cost function for collaborative filtering with regularization is given by\n", + "\n", + "$$ J(x^{(1)}, \\dots, x^{(n_m)}, \\theta^{(1)}, \\dots, \\theta^{(n_u)}) = \\frac{1}{2} \\sum_{(i,j):r(i,j)=1} \\left( \\left( \\theta^{(j)} \\right)^T x^{(i)} - y^{(i,j)} \\right)^2 + \\left( \\frac{\\lambda}{2} \\sum_{j=1}^{n_u} \\sum_{k=1}^{n} \\left( \\theta_k^{(j)} \\right)^2 \\right) + \\left( \\frac{\\lambda}{2} \\sum_{i=1}^{n_m} \\sum_{k=1}^n \\left(x_k^{(i)} \\right)^2 \\right) $$\n", + "\n", + "You should now add regularization to your original computations of the cost function, $J$. After you are done, the next cell will run your regularized cost function, and you should expect to see a cost of about 31.34.\n", + "\n", + "[Click here to go back to the function `cofiCostFunc` to update it](#cofiCostFunc)\n", + " Do not forget to re-execute the cell containing the function `cofiCostFunc` so that it is updated with your implementation of regularized cost function." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Evaluate cost function\n", + "J, _ = cofiCostFunc(np.concatenate([X.ravel(), Theta.ravel()]),\n", + " Y, R, num_users, num_movies, num_features, 1.5)\n", + " \n", + "print('Cost at loaded parameters (lambda = 1.5): %.2f' % J)\n", + "print(' (this value should be about 31.34)')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[5] = cofiCostFunc\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "#### 2.2.4 Regularized gradient\n", + "\n", + "Now that you have implemented the regularized cost function, you should proceed to implement regularization for the gradient. You should add to your implementation in `cofiCostFunc` to return the regularized gradient\n", + "by adding the contributions from the regularization terms. Note that the gradients for the regularized cost function is given by:\n", + "\n", + "$$ \\frac{\\partial J}{\\partial x_k^{(i)}} = \\sum_{j:r(i,j)=1} \\left( \\left(\\theta^{(j)}\\right)^T x^{(i)} - y^{(i,j)} \\right) \\theta_k^{(j)} + \\lambda x_k^{(i)} $$\n", + "\n", + "$$ \\frac{\\partial J}{\\partial \\theta_k^{(j)}} = \\sum_{i:r(i,j)=1} \\left( \\left(\\theta^{(j)}\\right)^T x^{(i)}- y^{(i,j)} \\right) x_k^{(j)} + \\lambda \\theta_k^{(j)} $$\n", + "\n", + "This means that you just need to add $\\lambda x^{(i)}$ to the `X_grad[i,:]` variable described earlier, and add $\\lambda \\theta^{(j)}$ to the `Theta_grad[j, :]` variable described earlier.\n", + "\n", + "[Click here to go back to the function `cofiCostFunc` to update it](#cofiCostFunc)\n", + " Do not forget to re-execute the cell containing the function `cofiCostFunc` so that it is updated with your implementation of the gradient for the regularized cost function.\n", + "\n", + "After you have completed the code to compute the gradients, the following cell will run another gradient check (`utils.checkCostFunction`) to numerically check the implementation of your gradients." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Check gradients by running checkCostFunction\n", + "utils.checkCostFunction(cofiCostFunc, 1.5)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*You should now submit your solutions.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "grader[6] = cofiCostFunc\n", + "grader.grade()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2.3 Learning movie recommendations \n", + "\n", + "After you have finished implementing the collaborative filtering cost function and gradient, you can now start training your algorithm to make movie recommendations for yourself. In the next cell, you can enter your own movie preferences, so that later when the algorithm runs, you can get your own movie recommendations! We have filled out some values according to our own preferences, but you should change this according to your own tastes. The list of all movies and their number in the dataset can be found listed in the file `Data/movie_idx.txt`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Before we will train the collaborative filtering model, we will first\n", + "# add ratings that correspond to a new user that we just observed. This\n", + "# part of the code will also allow you to put in your own ratings for the\n", + "# movies in our dataset!\n", + "movieList = utils.loadMovieList()\n", + "n_m = len(movieList)\n", + "\n", + "# Initialize my ratings\n", + "my_ratings = np.zeros(n_m)\n", + "\n", + "# Check the file movie_idx.txt for id of each movie in our dataset\n", + "# For example, Toy Story (1995) has ID 1, so to rate it \"4\", you can set\n", + "# Note that the index here is ID-1, since we start index from 0.\n", + "my_ratings[0] = 4\n", + "\n", + "# Or suppose did not enjoy Silence of the Lambs (1991), you can set\n", + "my_ratings[97] = 2\n", + "\n", + "# We have selected a few movies we liked / did not like and the ratings we\n", + "# gave are as follows:\n", + "my_ratings[6] = 3\n", + "my_ratings[11]= 5\n", + "my_ratings[53] = 4\n", + "my_ratings[63] = 5\n", + "my_ratings[65] = 3\n", + "my_ratings[68] = 5\n", + "my_ratings[182] = 4\n", + "my_ratings[225] = 5\n", + "my_ratings[354] = 5\n", + "\n", + "print('New user ratings:')\n", + "print('-----------------')\n", + "for i in range(len(my_ratings)):\n", + " if my_ratings[i] > 0:\n", + " print('Rated %d stars: %s' % (my_ratings[i], movieList[i]))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "#### 2.3.1 Recommendations\n", + "\n", + "After the additional ratings have been added to the dataset, the script\n", + "will proceed to train the collaborative filtering model. This will learn the\n", + "parameters X and Theta. To predict the rating of movie i for user j, you need to compute (θ (j) ) T x (i) . The next part of the script computes the ratings for\n", + "all the movies and users and displays the movies that it recommends (Figure\n", + "4), according to ratings that were entered earlier in the script. Note that\n", + "you might obtain a different set of the predictions due to different random\n", + "initializations." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Now, you will train the collaborative filtering model on a movie rating \n", + "# dataset of 1682 movies and 943 users\n", + "\n", + "# Load data\n", + "data = loadmat(os.path.join('Data', 'ex8_movies.mat'))\n", + "Y, R = data['Y'], data['R']\n", + "\n", + "# Y is a 1682x943 matrix, containing ratings (1-5) of 1682 movies by \n", + "# 943 users\n", + "\n", + "# R is a 1682x943 matrix, where R(i,j) = 1 if and only if user j gave a\n", + "# rating to movie i\n", + "\n", + "# Add our own ratings to the data matrix\n", + "Y = np.hstack([my_ratings[:, None], Y])\n", + "R = np.hstack([(my_ratings > 0)[:, None], R])\n", + "\n", + "# Normalize Ratings\n", + "Ynorm, Ymean = utils.normalizeRatings(Y, R)\n", + "\n", + "# Useful Values\n", + "num_movies, num_users = Y.shape\n", + "num_features = 10\n", + "\n", + "# Set Initial Parameters (Theta, X)\n", + "X = np.random.randn(num_movies, num_features)\n", + "Theta = np.random.randn(num_users, num_features)\n", + "\n", + "initial_parameters = np.concatenate([X.ravel(), Theta.ravel()])\n", + "\n", + "# Set options for scipy.optimize.minimize\n", + "options = {'maxiter': 100}\n", + "\n", + "# Set Regularization\n", + "lambda_ = 10\n", + "res = optimize.minimize(lambda x: cofiCostFunc(x, Ynorm, R, num_users,\n", + " num_movies, num_features, lambda_),\n", + " initial_parameters,\n", + " method='TNC',\n", + " jac=True,\n", + " options=options)\n", + "theta = res.x\n", + "\n", + "# Unfold the returned theta back into U and W\n", + "X = theta[:num_movies*num_features].reshape(num_movies, num_features)\n", + "Theta = theta[num_movies*num_features:].reshape(num_users, num_features)\n", + "\n", + "print('Recommender system learning completed.')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "After training the model, you can now make recommendations by computing the predictions matrix." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "p = np.dot(X, Theta.T)\n", + "my_predictions = p[:, 0] + Ymean\n", + "\n", + "movieList = utils.loadMovieList()\n", + "\n", + "ix = np.argsort(my_predictions)[::-1]\n", + "\n", + "print('Top recommendations for you:')\n", + "print('----------------------------')\n", + "for i in range(10):\n", + " j = ix[i]\n", + " print('Predicting rating %.1f for movie %s' % (my_predictions[j], movieList[j]))\n", + "\n", + "print('\\nOriginal ratings provided:')\n", + "print('--------------------------')\n", + "for i in range(len(my_ratings)):\n", + " if my_ratings[i] > 0:\n", + " print('Rated %d for %s' % (my_ratings[i], movieList[i]))" + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.6.4" + } + }, + "nbformat": 4, + "nbformat_minor": 2 +} diff --git a/Exercise8/utils.py b/Exercise8/utils.py new file mode 100755 index 0000000..938a7bb --- /dev/null +++ b/Exercise8/utils.py @@ -0,0 +1,269 @@ +import numpy as np +import sys +from os.path import join +from matplotlib import pyplot + +sys.path.append('..') +from submission import SubmissionBase + + +def normalizeRatings(Y, R): + """ + Preprocess data by subtracting mean rating for every movie (every row). + + Parameters + ---------- + Y : array_like + The user ratings for all movies. A matrix of shape (num_movies x num_users). + + R : array_like + Indicator matrix for movies rated by users. A matrix of shape (num_movies x num_users). + + Returns + ------- + Ynorm : array_like + A matrix of same shape as Y, after mean normalization. + + Ymean : array_like + A vector of shape (num_movies, ) containing the mean rating for each movie. + """ + m, n = Y.shape + Ymean = np.zeros(m) + Ynorm = np.zeros(Y.shape) + + for i in range(m): + idx = R[i, :] == 1 + Ymean[i] = np.mean(Y[i, idx]) + Ynorm[i, idx] = Y[i, idx] - Ymean[i] + + return Ynorm, Ymean + + +def loadMovieList(): + """ + Reads the fixed movie list in movie_ids.txt and returns a list of movie names. + + Returns + ------- + movieNames : list + A list of strings, representing all movie names. + """ + # Read the fixed movieulary list + with open(join('Data', 'movie_ids.txt'), encoding='ISO-8859-1') as fid: + movies = fid.readlines() + + movieNames = [] + for movie in movies: + parts = movie.split() + movieNames.append(' '.join(parts[1:]).strip()) + return movieNames + + +def computeNumericalGradient(J, theta, e=1e-4): + """ + Computes the gradient using "finite differences" and gives us a numerical estimate of the + gradient. + + Parameters + ---------- + J : func + The cost function which will be used to estimate its numerical gradient. + + theta : array_like + The one dimensional unrolled network parameters. The numerical gradient is computed at + those given parameters. + + e : float (optional) + The value to use for epsilon for computing the finite difference. + + Returns + ------- + numgrad : array_like + The numerical gradient with respect to theta. Has same shape as theta. + + Notes + ----- + The following code implements numerical gradient checking, and + returns the numerical gradient. It sets `numgrad[i]` to (a numerical + approximation of) the partial derivative of J with respect to the + i-th input argument, evaluated at theta. (i.e., `numgrad[i]` should + be the (approximately) the partial derivative of J with respect + to theta[i].) + """ + numgrad = np.zeros(theta.shape) + perturb = np.diag(e * np.ones(theta.shape)) + for i in range(theta.size): + loss1, _ = J(theta - perturb[:, i]) + loss2, _ = J(theta + perturb[:, i]) + numgrad[i] = (loss2 - loss1)/(2*e) + return numgrad + + +def checkCostFunction(cofiCostFunc, lambda_=0.): + """ + Creates a collaborative filtering problem to check your cost function and gradients. + It will output the analytical gradients produced by your code and the numerical gradients + (computed using computeNumericalGradient). These two gradient computations should result + in very similar values. + + Parameters + ---------- + cofiCostFunc: func + Implementation of the cost function. + + lambda_ : float, optional + The regularization parameter. + """ + # Create small problem + X_t = np.random.rand(4, 3) + Theta_t = np.random.rand(5, 3) + + # Zap out most entries + Y = np.dot(X_t, Theta_t.T) + Y[np.random.rand(*Y.shape) > 0.5] = 0 + R = np.zeros(Y.shape) + R[Y != 0] = 1 + + # Run Gradient Checking + X = np.random.randn(*X_t.shape) + Theta = np.random.randn(*Theta_t.shape) + num_movies, num_users = Y.shape + num_features = Theta_t.shape[1] + + params = np.concatenate([X.ravel(), Theta.ravel()]) + numgrad = computeNumericalGradient( + lambda x: cofiCostFunc(x, Y, R, num_users, num_movies, num_features, lambda_), params) + + cost, grad = cofiCostFunc(params, Y, R, num_users,num_movies, num_features, lambda_) + + print(np.stack([numgrad, grad], axis=1)) + print('\nThe above two columns you get should be very similar.' + '(Left-Your Numerical Gradient, Right-Analytical Gradient)') + + diff = np.linalg.norm(numgrad-grad)/np.linalg.norm(numgrad+grad) + print('If your cost function implementation is correct, then ' + 'the relative difference will be small (less than 1e-9).') + print('\nRelative Difference: %g' % diff) + + +def multivariateGaussian(X, mu, Sigma2): + """ + Computes the probability density function of the multivariate gaussian distribution. + + Parameters + ---------- + X : array_like + The dataset of shape (m x n). Where there are m examples of n-dimensions. + + mu : array_like + A vector of shape (n,) contains the means for each dimension (feature). + + Sigma2 : array_like + Either a vector of shape (n,) containing the variances of independent features + (i.e. it is the diagonal of the correlation matrix), or the full + correlation matrix of shape (n x n) which can represent dependent features. + + Returns + ------ + p : array_like + A vector of shape (m,) which contains the computed probabilities at each of the + provided examples. + """ + k = mu.size + + # if sigma is given as a diagonal, compute the matrix + if Sigma2.ndim == 1: + Sigma2 = np.diag(Sigma2) + + X = X - mu + p = (2 * np.pi) ** (- k / 2) * np.linalg.det(Sigma2) ** (-0.5)\ + * np.exp(-0.5 * np.sum(np.dot(X, np.linalg.pinv(Sigma2)) * X, axis=1)) + return p + + +def visualizeFit(X, mu, sigma2): + """ + Visualize the dataset and its estimated distribution. + This visualization shows you the probability density function of the Gaussian distribution. + Each example has a location (x1, x2) that depends on its feature values. + + Parameters + ---------- + X : array_like + The dataset of shape (m x 2). Where there are m examples of 2-dimensions. We need at most + 2-D features to be able to visualize the distribution. + + mu : array_like + A vector of shape (n,) contains the means for each dimension (feature). + + sigma2 : array_like + Either a vector of shape (n,) containing the variances of independent features + (i.e. it is the diagonal of the correlation matrix), or the full + correlation matrix of shape (n x n) which can represent dependent features. + """ + + X1, X2 = np.meshgrid(np.arange(0, 35.5, 0.5), np.arange(0, 35.5, 0.5)) + Z = multivariateGaussian(np.stack([X1.ravel(), X2.ravel()], axis=1), mu, sigma2) + Z = Z.reshape(X1.shape) + + pyplot.plot(X[:, 0], X[:, 1], 'bx', mec='b', mew=2, ms=8) + + if np.all(abs(Z) != np.inf): + pyplot.contour(X1, X2, Z, levels=10**(np.arange(-20., 1, 3)), zorder=100) + + +class Grader(SubmissionBase): + # Random Test Cases + n_u = 3 + n_m = 4 + n = 5 + X = np.sin(np.arange(1, 1 + n_m * n)).reshape(n_m, n, order='F') + Theta = np.cos(np.arange(1, 1 + n_u * n)).reshape(n_u, n, order='F') + Y = np.sin(np.arange(1, 1 + 2 * n_m * n_u, 2)).reshape(n_m, n_u, order='F') + R = Y > 0.5 + pval = np.concatenate([abs(Y.ravel('F')), [0.001], [1]]) + Y = Y * R # set 'Y' values to 0 for movies not reviewed + + yval = np.concatenate([R.ravel('F'), [1], [0]]) + # + params = np.concatenate([X.ravel(), Theta.ravel()]) + + def __init__(self): + part_names = ['Estimate Gaussian Parameters', + 'Select Threshold', + 'Collaborative Filtering Cost', + 'Collaborative Filtering Gradient', + 'Regularized Cost', + 'Regularized Gradient'] + super().__init__('anomaly-detection-and-recommender-systems', part_names) + + def __iter__(self): + for part_id in range(1, 7): + try: + func = self.functions[part_id] + + # Each part has different expected arguments/different function + if part_id == 1: + res = np.hstack(func(self.X)).tolist() + elif part_id == 2: + res = np.hstack(func(self.yval, self.pval)).tolist() + elif part_id == 3: + J, grad = func(self.params, self.Y, self.R, self.n_u, self.n_m, self.n) + res = J + elif part_id == 4: + J, grad = func(self.params, self.Y, self.R, self.n_u, self.n_m, self.n, 0) + xgrad = grad[:self.n_m*self.n].reshape(self.n_m, self.n) + thetagrad = grad[self.n_m*self.n:].reshape(self.n_u, self.n) + res = np.hstack([xgrad.ravel('F'), thetagrad.ravel('F')]).tolist() + elif part_id == 5: + res, _ = func(self.params, self.Y, self.R, self.n_u, self.n_m, self.n, 1.5) + elif part_id == 6: + J, grad = func(self.params, self.Y, self.R, self.n_u, self.n_m, self.n, 1.5) + xgrad = grad[:self.n_m*self.n].reshape(self.n_m, self.n) + thetagrad = grad[self.n_m*self.n:].reshape(self.n_u, self.n) + res = np.hstack([xgrad.ravel('F'), thetagrad.ravel('F')]).tolist() + else: + raise KeyError + yield part_id, res + except KeyError: + yield part_id, 0 diff --git a/README.md b/README.md new file mode 100755 index 0000000..b391919 --- /dev/null +++ b/README.md @@ -0,0 +1,99 @@ +# [Coursera Machine Learning MOOC by Andrew Ng](https://www.coursera.org/learn/machine-learning) +# Python Programming Assignments + +![](machinelearning.jpg) + +This repositry contains the python versions of the programming assignments for the [Machine Learning online class](https://www.coursera.org/learn/machine-learning) taught by Professor Andrew Ng. This is perhaps the most popular introductory online machine learning class. In addition to being popular, it is also one of the best Machine learning classes any interested student can take to get started with machine learning. An unfortunate aspect of this class is that the programming assignments are in MATLAB or OCTAVE, probably because this class was made before python become the go-to language in machine learning. + +The Python machine learning ecosystem has grown exponentially in the past few years, and still gaining momentum. I suspect that many students who want to get started with their machine learning journey would like to start it with Python also. It is for those reasons I have decided to re-write all the programming assignments in Python, so students can get acquainted with its ecosystem from the start of their learning journey. + +These assignments work seamlessly with the class and do not require any of the materials published in the MATLAB assignments. Here are some new and useful features for these sets of assignments: + +- The assignments use [Jupyter Notebook](http://jupyter-notebook-beginner-guide.readthedocs.io/en/latest/what_is_jupyter.html), which provides an intuitive flow easier than the original MATLAB/OCTAVE assignments. +- The original assignment instructions have been completely re-written and the parts which used to reference MATLAB/OCTAVE functionality have been changed to reference its `python` counterpart. +- The re-written instructions are now embedded within the Jupyter Notebook along with the `python` starter code. For each assignment, all work is done solely within the notebook. +- The `python` assignments can be submitted for grading. They were tested to work perfectly well with the original Coursera grader that is currently used to grade the MATLAB/OCTAVE versions of the assignments. +- After each part of a given assignment, the Jupyter Notebook contains a cell which prompts the user for submitting the current part of the assignment for grading. + +## Downloading the Assignments + +To get started, you can start by either downloading a zip file of these assignments by clicking on the `Clone or download` button. If you have `git` installed on your system, you can clone this repository using : + + clone + +Each assignment is contained in a separate folder. For example, assignment 1 is contained within the folder `Exercise1`. Each folder contains two files: + - The assignment `jupyter` notebook, which has a `.ipynb` extension. All the code which you need to write will be written within this notebook. + - A python module `utils.py` which contains some helper functions needed for the assignment. Functions within the `utils` module are called from the python notebook. You do not need to modify or add any code to this file. + +## Requirements + +These assignments has been tested and developed using the following libraries: + + - python==3.6.4 + - numpy==1.13.3 + - scipy==1.0.0 + - matplotlib==2.1.2 + - jupyter==1.0.0 + - jupyter-client==5.0.1 + +We recommend using at least these versions of the required libraries or later. Python 2 is not supported. + +## Python Installation + +We highly recommend using anaconda for installing python. [Click here](https://www.anaconda.com/download/) to go to Anaconda's download page. Make sure to download Python 3.6 version. +If you are on a windows machine: + - Open the executable after download is complete and follow instructions. + - Once installation is complete, open `Anaconda prompt` from the start menu. This will open a terminal with python enabled. + + If you are on a linux machine: + + - Open a terminal and navigate to the directory where Anaconda was downloaded. + - Change the permission to the downloaded file so that it can be executed. So if the downloaded file name is `Anaconda3-5.1.0-Linux-x86_64.sh`, then use the following command: + + `chmod a+x Anaconda3-5.1.0-Linux-x86_64.sh` + + - Now, run the installation script using `./Anaconda3-5.1.0-Linux-x86_64.sh`, and follow installation instructions in the terminal. + + +Once you have installed python, create a new python environment will all the requirements using the following command: + + conda create -n machine_learning python=3.6 scipy=1 numpy=1.13 matplotlib=2.1 jupyter + +After the new environment is setup, activate it using (windows) + + activate machine_learning + +or if you are on a linux machine + + source activate machine_learning + +Now we have our python environment all set up, we can start working on the assignments. To do so, navigate to the directory where the assignments were installed, and launch the jupyter notebook from the terminal using the command + + jupyter notebook + +This should automatically open a tab in the default browser. To start with assignment 1, open the notebook `./Exercise1/exercise1.ipynb`. + +## Python Tutorials + +If you are new to python and to `jupyter` notebooks, no worries! There is a plethora of tutorials and documentation to get you started. Here are a few links which might be of help: + +- [Python Programming](https://pythonprogramming.net/introduction-to-python-programming/): A turorial with videos about the basics of python. + +- [Numpy and matplotlib tutorial](http://cs231n.github.io/python-numpy-tutorial/): We will be using numpy extensively for matrix and vector operations. This is great tutorial to get you started with using numpy and matplotlib for plotting. + +- [Jupyter notebook](https://medium.com/codingthesmartway-com-blog/getting-started-with-jupyter-notebook-for-python-4e7082bd5d46): Getting started with the jupyter notebook. + +- [Python introduction based on the class's MATLAB tutorial](https://github.com/mstampfer/Coursera-Stanford-ML-Python/blob/master/Coursera%20Stanford%20ML%20Python%20wiki.ipynb): This is the equivalent of class's MATLAB tutorial, in python. + + +## Caveats and tips + +- In many of the exercises, the regularization parameter $\lambda$ is denoted as the variable name `lambda_`, notice the underscore at the end of the name. This is because `lambda` is a reserved python keyword, and should never be used as a variable name. + +- In `numpy`, the function `dot` is used to perform matrix multiplication. The operation '*' only does element-by-element multiplication (unlike MATLAB). If you are using python version 3.5+, the operator '@' is the new matrix multiplication, and it is equivalent to the `dot` function. + +## Acknowledgements + +- I would like to thank professor Andrew Ng and the crew of the Stanford Machine Learning class on Coursera for such an awesome class. + +- Some of the material used, especially the code for submitting assignments for grading is based on [`mstampfer`'s](https://github.com/mstampfer/Coursera-Stanford-ML-Python) python implementation of the assignments. \ No newline at end of file diff --git a/machinelearning.jpg b/machinelearning.jpg new file mode 100755 index 0000000..ddb4335 Binary files /dev/null and b/machinelearning.jpg differ diff --git a/requirements.txt b/requirements.txt new file mode 100755 index 0000000..1054bc1 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,174 @@ +alabaster==0.7.10 +anaconda-client==1.6.3 +anaconda-navigator==1.6.2 +anaconda-project==0.6.0 +asn1crypto==0.22.0 +astroid==1.4.9 +astropy==2.0.1 +Babel==2.4.0 +backports.shutil-get-terminal-size==1.0.0 +backports.weakref==1.0rc1 +beautifulsoup4==4.6.0 +bitarray==0.8.1 +blaze==0.10.1 +bleach==1.5.0 +bokeh==0.12.13 +boto==2.46.1 +Bottleneck==1.2.1 +cffi==1.10.0 +chardet==3.0.3 +click==6.7 +cloudpickle==0.2.2 +clyent==1.2.2 +colorama==0.3.9 +conda==4.4.8 +contextlib2==0.5.5 +coverage==4.4.1 +cryptography==1.8.1 +cycler==0.10.0 +Cython==0.25.2 +cytoolz==0.8.2 +dask==0.14.3 +datashape==0.5.4 +decorator==4.1.2 +distributed==1.16.3 +docutils==0.13.1 +entrypoints==0.2.2 +et-xmlfile==1.0.1 +fastcache==1.0.2 +Flask==0.12.2 +Flask-Cors==3.0.2 +future==0.16.0 +gevent==1.2.1 +greenlet==0.4.12 +h5py==2.7.0 +HeapDict==1.0.0 +html5lib==0.9999999 +idna==2.5 +imagesize==0.7.1 +ipdb==0.10.3 +ipdbplugin==1.4.5 +ipykernel==4.6.1 +ipython==6.2.0 +ipython-genutils==0.2.0 +ipywidgets==6.0.0 +isort==4.2.5 +itsdangerous==0.24 +jdcal==1.3 +jedi==0.11.0 +Jinja2==2.9.6 +jsonschema==2.6.0 +jupyter==1.0.0 +jupyter-client==5.0.1 +jupyter-console==5.1.0 +jupyter-contrib-core==0.3.3 +jupyter-contrib-nbextensions==0.3.3 +jupyter-core==4.3.0 +jupyter-highlight-selected-word==0.1.0 +jupyter-latex-envs==1.4.1 +jupyter-nbextensions-configurator==0.3.0 +jupyterthemes==0.17.8 +Keras==2.0.8 +lazy-object-proxy==1.2.2 +lesscpy==0.12.0 +llvmlite==0.20.0 +locket==0.2.0 +lxml==4.1.1 +Mako==1.0.7 +Markdown==2.6.9 +MarkupSafe==0.23 +matplotlib==2.1.2 +mistune==0.7.4 +mpmath==0.19 +msgpack-python==0.4.8 +multipledispatch==0.4.9 +navigator-updater==0.1.0 +nbconvert==5.3.1 +nbformat==4.4.0 +networkx==1.11 +nltk==3.2.3 +nose==1.3.7 +notebook==5.0.0 +numba==0.35.0 +numexpr==2.6.2 +numpy==1.13.1 +numpydoc==0.6.0 +odo==0.5.0 +olefile==0.44 +openpyxl==2.4.7 +packaging==16.8 +pandas==0.20.3 +pandocfilters==1.4.1 +paramiko==2.1.2 +parso==0.1.0 +partd==0.3.8 +pathlib2==2.2.1 +patsy==0.4.1 +pep8==1.7.0 +pexpect==4.2.1 +pickleshare==0.7.4 +Pillow==5.0.0 +ply==3.10 +prompt-toolkit==1.0.15 +protobuf==3.4.0 +psutil==5.2.2 +ptyprocess==0.5.2 +py==1.4.33 +pyasn1==0.2.3 +pycosat==0.6.3 +pycparser==2.17 +pycrypto==2.6.1 +pycurl==7.43.0 +pyflakes==1.5.0 +Pygments==2.2.0 +pygpu==0.6.9 +pylint==1.6.4 +pyodbc==4.0.16 +pyOpenSSL==17.0.0 +pyparsing==2.2.0 +pytest==3.0.7 +python-dateutil==2.6.1 +pytz==2017.2 +PyWavelets==0.5.2 +PyYAML==3.12 +pyzmq==16.0.2 +QtAwesome==0.4.4 +qtconsole==4.3.0 +QtPy==1.2.1 +requests==2.14.2 +rope-py3k==0.9.4.post1 +scikit-image==0.13.0 +scikit-learn==0.19.0 +scipy==0.19.1 +scons==3.0.0a20170821 +seaborn==0.7.1 +simplegeneric==0.8.1 +singledispatch==3.4.0.3 +six==1.11.0 +snowballstemmer==1.2.1 +sortedcollections==0.5.3 +sortedcontainers==1.5.7 +Sphinx==1.5.6 +spyder==3.1.4 +SQLAlchemy==1.1.9 +statsmodels==0.8.0 +sympy==1.0 +tables==3.4.2 +tblib==1.3.2 +tensorflow==1.3.0 +tensorflow-tensorboard==0.1.5 +terminado==0.6 +testpath==0.3 +Theano==0.9.0 +toolz==0.8.2 +tornado==4.5.1 +traitlets==4.3.2 +unicodecsv==0.14.1 +wcwidth==0.1.7 +Werkzeug==0.12.2 +widgetsnbextension==2.0.0 +wrapt==1.10.10 +xlrd==1.0.0 +XlsxWriter==0.9.6 +xlwt==1.2.0 +zict==0.1.2 diff --git a/submission.py b/submission.py new file mode 100755 index 0000000..752d1c7 --- /dev/null +++ b/submission.py @@ -0,0 +1,105 @@ +from urllib.parse import urlencode +from urllib.request import urlopen +import pickle +import json +from collections import OrderedDict +import numpy as np +import os + + +class SubmissionBase: + + submit_url = 'https://www-origin.coursera.org/api/' \ + 'onDemandProgrammingImmediateFormSubmissions.v1' + save_file = 'token.pkl' + + def __init__(self, assignment_slug, part_names): + self.assignment_slug = assignment_slug + self.part_names = part_names + self.login = None + self.token = None + self.functions = OrderedDict() + self.args = dict() + + def grade(self): + print('\nSubmitting Solutions | Programming Exercise %s\n' % self.assignment_slug) + self.login_prompt() + + # Evaluate the different parts of exercise + parts = OrderedDict() + for part_id, result in self: + parts[str(part_id)] = {'output': sprintf('%0.5f ', result)} + result, response = self.request(parts) + response = json.loads(response) + + # if an error was returned, print it and stop + if 'errorMessage' in response: + print(response['errorMessage']) + return + + # Print the grading table + print('%43s | %9s | %-s' % ('Part Name', 'Score', 'Feedback')) + print('%43s | %9s | %-s' % ('---------', '-----', '--------')) + for part in parts: + part_feedback = response['partFeedbacks'][part] + part_evaluation = response['partEvaluations'][part] + score = '%d / %3d' % (part_evaluation['score'], part_evaluation['maxScore']) + print('%43s | %9s | %-s' % (self.part_names[int(part) - 1], score, part_feedback)) + evaluation = response['evaluation'] + total_score = '%d / %d' % (evaluation['score'], evaluation['maxScore']) + print(' --------------------------------') + print('%43s | %9s | %-s\n' % (' ', total_score, ' ')) + + def login_prompt(self): + if os.path.isfile(self.save_file): + with open(self.save_file, 'rb') as f: + login, token = pickle.load(f) + reenter = input('Use token from last successful submission (%s)? (Y/n): ' % login) + + if reenter == '' or reenter[0] == 'Y' or reenter[0] == 'y': + self.login, self.token = login, token + return + else: + os.remove(self.save_file) + + self.login = input('Login (email address): ') + self.token = input('Token: ') + + # Save the entered credentials + if not os.path.isfile(self.save_file): + with open(self.save_file, 'wb') as f: + pickle.dump((self.login, self.token), f) + + def request(self, parts): + params = { + 'assignmentSlug': self.assignment_slug, + 'secret': self.token, + 'parts': parts, + 'submitterEmail': self.login} + + params = urlencode({'jsonBody': json.dumps(params)}).encode("utf-8") + f = urlopen(self.submit_url, params) + try: + return 0, f.read() + finally: + f.close() + + def __iter__(self): + for part_id in self.functions: + yield part_id + + def __setitem__(self, key, value): + self.functions[key] = value + + +def sprintf(fmt, arg): + """ Emulates (part of) Octave sprintf function. """ + if isinstance(arg, tuple): + # for multiple return values, only use the first one + arg = arg[0] + + if isinstance(arg, (np.ndarray, list)): + # concatenates all elements, column by column + return ' '.join(fmt % e for e in np.asarray(arg).ravel('F')) + else: + return fmt % arg