Files
Solutions-Coursera-Machine-…/Exercise7/utils.py
2018-08-25 18:06:29 -07:00

237 lines
7.7 KiB
Python

import sys
import numpy as np
from matplotlib import pyplot
from matplotlib.animation import FuncAnimation
import matplotlib as mpl
sys.path.append('..')
from submission import SubmissionBase
def displayData(X, example_width=None, figsize=(10, 10)):
"""
Displays 2D data in a nice grid.
Parameters
----------
X : array_like
The input data of size (m x n) where m is the number of examples and n is the number of
features.
example_width : int, optional
THe width of each 2-D image in pixels. If not provided, the image is assumed to be square,
and the width is the floor of the square root of total number of pixels.
figsize : tuple, optional
A 2-element tuple indicating the width and height of figure in inches.
"""
# Compute rows, cols
if X.ndim == 2:
m, n = X.shape
elif X.ndim == 1:
n = X.size
m = 1
X = X[None] # Promote to a 2 dimensional array
else:
raise IndexError('Input X should be 1 or 2 dimensional.')
example_width = example_width or int(np.round(np.sqrt(n)))
example_height = int(n / example_width)
# Compute number of items to display
display_rows = int(np.floor(np.sqrt(m)))
display_cols = int(np.ceil(m / display_rows))
fig, ax_array = pyplot.subplots(display_rows, display_cols, figsize=figsize)
fig.subplots_adjust(wspace=0.025, hspace=0.025)
ax_array = [ax_array] if m == 1 else ax_array.ravel()
for i, ax in enumerate(ax_array):
ax.imshow(X[i].reshape(example_height, example_width, order='F'), cmap='gray')
ax.axis('off')
def featureNormalize(X):
"""
Normalizes the features in X returns a normalized version of X where the mean value of each
feature is 0 and the standard deviation is 1. This is often a good preprocessing step to do when
working with learning algorithms.
Parameters
----------
X : array_like
An dataset which is a (m x n) matrix, where m is the number of examples,
and n is the number of dimensions for each example.
Returns
-------
X_norm : array_like
The normalized input dataset.
mu : array_like
A vector of size n corresponding to the mean for each dimension across all examples.
sigma : array_like
A vector of size n corresponding to the standard deviations for each dimension across
all examples.
"""
mu = np.mean(X, axis=0)
X_norm = X - mu
sigma = np.std(X_norm, axis=0, ddof=1)
X_norm /= sigma
return X_norm, mu, sigma
def plotProgresskMeans(i, X, centroid_history, idx_history):
"""
A helper function that displays the progress of k-Means as it is running. It is intended for use
only with 2D data. It plots data points with colors assigned to each centroid. With the
previous centroids, it also plots a line between the previous locations and current locations
of the centroids.
Parameters
----------
i : int
Current iteration number of k-means. Used for matplotlib animation function.
X : array_like
The dataset, which is a matrix (m x n). Note since the plot only supports 2D data, n should
be equal to 2.
centroid_history : list
A list of computed centroids for all iteration.
idx_history : list
A list of computed assigned indices for all iterations.
"""
K = centroid_history[0].shape[0]
pyplot.gcf().clf()
cmap = pyplot.cm.rainbow
norm = mpl.colors.Normalize(vmin=0, vmax=2)
for k in range(K):
current = np.stack([c[k, :] for c in centroid_history[:i+1]], axis=0)
pyplot.plot(current[:, 0], current[:, 1],
'-Xk',
mec='k',
lw=2,
ms=10,
mfc=cmap(norm(k)),
mew=2)
pyplot.scatter(X[:, 0], X[:, 1],
c=idx_history[i],
cmap=cmap,
marker='o',
s=8**2,
linewidths=1,)
pyplot.grid(False)
pyplot.title('Iteration number %d' % (i+1))
def runkMeans(X, centroids, findClosestCentroids, computeCentroids,
max_iters=10, plot_progress=False):
"""
Runs the K-means algorithm.
Parameters
----------
X : array_like
The data set of size (m, n). Each row of X is a single example of n dimensions. The
data set is a total of m examples.
centroids : array_like
Initial centroid location for each clusters. This is a matrix of size (K, n). K is the total
number of clusters and n is the dimensions of each data point.
findClosestCentroids : func
A function (implemented by student) reference which computes the cluster assignment for
each example.
computeCentroids : func
A function(implemented by student) reference which computes the centroid of each cluster.
max_iters : int, optional
Specifies the total number of interactions of K-Means to execute.
plot_progress : bool, optional
A flag that indicates if the function should also plot its progress as the learning happens.
This is set to false by default.
Returns
-------
centroids : array_like
A (K x n) matrix of the computed (updated) centroids.
idx : array_like
A vector of size (m,) for cluster assignment for each example in the dataset. Each entry
in idx is within the range [0 ... K-1].
anim : FuncAnimation, optional
A matplotlib animation object which can be used to embed a video within the jupyter
notebook. This is only returned if `plot_progress` is `True`.
"""
K = centroids.shape[0]
idx = None
idx_history = []
centroid_history = []
for i in range(max_iters):
idx = findClosestCentroids(X, centroids)
if plot_progress:
idx_history.append(idx)
centroid_history.append(centroids)
centroids = computeCentroids(X, idx, K)
if plot_progress:
fig = pyplot.figure()
anim = FuncAnimation(fig, plotProgresskMeans,
frames=max_iters,
interval=500,
repeat_delay=2,
fargs=(X, centroid_history, idx_history))
return centroids, idx, anim
return centroids, idx
class Grader(SubmissionBase):
# Random Test Cases
X = np.sin(np.arange(1, 166)).reshape(15, 11, order='F')
Z = np.cos(np.arange(1, 122)).reshape(11, 11, order='F')
C = Z[:5, :]
idx = np.arange(1, 16) % 3
def __init__(self):
part_names = ['Find Closest Centroids (k-Means)',
'Compute Centroid Means (k-Means)',
'PCA',
'Project Data (PCA)',
'Recover Data (PCA)']
super().__init__('k-means-clustering-and-pca', part_names)
def __iter__(self):
for part_id in range(1, 6):
try:
func = self.functions[part_id]
# Each part has different expected arguments/different function
if part_id == 1:
res = 1 + func(self.X, self.C)
elif part_id == 2:
res = func(self.X, self.idx, 3)
elif part_id == 3:
U, S = func(self.X)
res = np.hstack([U.ravel('F'), np.diag(S).ravel('F')]).tolist()
elif part_id == 4:
res = func(self.X, self.Z, 5)
elif part_id == 5:
res = func(self.X[:, :5], self.Z, 5)
else:
raise KeyError
yield part_id, res
except KeyError:
yield part_id, 0