First commit
10
Exercise6/Data/emailSample1.txt
Executable file
@@ -0,0 +1,10 @@
|
||||
> Anyone knows how much it costs to host a web portal ?
|
||||
>
|
||||
Well, it depends on how many visitors you're expecting.
|
||||
This can be anywhere from less than 10 bucks a month to a couple of $100.
|
||||
You should checkout http://www.rackspace.com/ or perhaps Amazon EC2
|
||||
if youre running something big..
|
||||
|
||||
To unsubscribe yourself from this mailing list, send an email to:
|
||||
groupname-unsubscribe@egroups.com
|
||||
|
||||
34
Exercise6/Data/emailSample2.txt
Executable file
@@ -0,0 +1,34 @@
|
||||
Folks,
|
||||
|
||||
my first time posting - have a bit of Unix experience, but am new to Linux.
|
||||
|
||||
|
||||
Just got a new PC at home - Dell box with Windows XP. Added a second hard disk
|
||||
for Linux. Partitioned the disk and have installed Suse 7.2 from CD, which went
|
||||
fine except it didn't pick up my monitor.
|
||||
|
||||
I have a Dell branded E151FPp 15" LCD flat panel monitor and a nVidia GeForce4
|
||||
Ti4200 video card, both of which are probably too new to feature in Suse's default
|
||||
set. I downloaded a driver from the nVidia website and installed it using RPM.
|
||||
Then I ran Sax2 (as was recommended in some postings I found on the net), but
|
||||
it still doesn't feature my video card in the available list. What next?
|
||||
|
||||
Another problem. I have a Dell branded keyboard and if I hit Caps-Lock twice,
|
||||
the whole machine crashes (in Linux, not Windows) - even the on/off switch is
|
||||
inactive, leaving me to reach for the power cable instead.
|
||||
|
||||
If anyone can help me in any way with these probs., I'd be really grateful -
|
||||
I've searched the 'net but have run out of ideas.
|
||||
|
||||
Or should I be going for a different version of Linux such as RedHat? Opinions
|
||||
welcome.
|
||||
|
||||
Thanks a lot,
|
||||
Peter
|
||||
|
||||
--
|
||||
Irish Linux Users' Group: ilug@linux.ie
|
||||
http://www.linux.ie/mailman/listinfo/ilug for (un)subscription information.
|
||||
List maintainer: listmaster@linux.ie
|
||||
|
||||
|
||||
BIN
Exercise6/Data/ex6data1.mat
Executable file
BIN
Exercise6/Data/ex6data2.mat
Executable file
BIN
Exercise6/Data/ex6data3.mat
Executable file
42
Exercise6/Data/spamSample1.txt
Executable file
@@ -0,0 +1,42 @@
|
||||
Do You Want To Make $1000 Or More Per Week?
|
||||
|
||||
|
||||
|
||||
If you are a motivated and qualified individual - I
|
||||
will personally demonstrate to you a system that will
|
||||
make you $1,000 per week or more! This is NOT mlm.
|
||||
|
||||
|
||||
|
||||
Call our 24 hour pre-recorded number to get the
|
||||
details.
|
||||
|
||||
|
||||
|
||||
000-456-789
|
||||
|
||||
|
||||
|
||||
I need people who want to make serious money. Make
|
||||
the call and get the facts.
|
||||
|
||||
Invest 2 minutes in yourself now!
|
||||
|
||||
|
||||
|
||||
000-456-789
|
||||
|
||||
|
||||
|
||||
Looking forward to your call and I will introduce you
|
||||
to people like yourself who
|
||||
are currently making $10,000 plus per week!
|
||||
|
||||
|
||||
|
||||
000-456-789
|
||||
|
||||
|
||||
|
||||
3484lJGv6-241lEaN9080lRmS6-271WxHo7524qiyT5-438rjUv5615hQcf0-662eiDB9057dMtVl72
|
||||
|
||||
8
Exercise6/Data/spamSample2.txt
Executable file
@@ -0,0 +1,8 @@
|
||||
Best Buy Viagra Generic Online
|
||||
|
||||
Viagra 100mg x 60 Pills $125, Free Pills & Reorder Discount, Top Selling 100% Quality & Satisfaction guaranteed!
|
||||
|
||||
We accept VISA, Master & E-Check Payments, 90000+ Satisfied Customers!
|
||||
http://medphysitcstech.ru
|
||||
|
||||
|
||||
BIN
Exercise6/Data/spamTest.mat
Executable file
BIN
Exercise6/Data/spamTrain.mat
Executable file
1899
Exercise6/Data/vocab.txt
Executable file
BIN
Exercise6/Figures/dataset1.png
Executable file
|
After Width: | Height: | Size: 7.8 KiB |
BIN
Exercise6/Figures/dataset2.png
Normal file
|
After Width: | Height: | Size: 79 KiB |
BIN
Exercise6/Figures/dataset3.png
Normal file
|
After Width: | Height: | Size: 20 KiB |
BIN
Exercise6/Figures/email.png
Normal file
|
After Width: | Height: | Size: 88 KiB |
BIN
Exercise6/Figures/email_cleaned.png
Normal file
|
After Width: | Height: | Size: 54 KiB |
BIN
Exercise6/Figures/svm_c1.png
Normal file
|
After Width: | Height: | Size: 13 KiB |
BIN
Exercise6/Figures/svm_c100.png
Normal file
|
After Width: | Height: | Size: 12 KiB |
BIN
Exercise6/Figures/svm_dataset2.png
Normal file
|
After Width: | Height: | Size: 113 KiB |
BIN
Exercise6/Figures/svm_dataset3_best.png
Normal file
|
After Width: | Height: | Size: 39 KiB |
BIN
Exercise6/Figures/svm_predictors.png
Executable file
|
After Width: | Height: | Size: 18 KiB |
BIN
Exercise6/Figures/vocab.png
Normal file
|
After Width: | Height: | Size: 25 KiB |
BIN
Exercise6/Figures/word_indices.png
Normal file
|
After Width: | Height: | Size: 72 KiB |
1029
Exercise6/exercise6.ipynb
Executable file
718
Exercise6/utils.py
Executable file
@@ -0,0 +1,718 @@
|
||||
import sys
|
||||
|
||||
sys.path.append('..')
|
||||
from submission import SubmissionBase
|
||||
import numpy as np
|
||||
from scipy.io import loadmat
|
||||
from os.path import join
|
||||
from matplotlib import pyplot
|
||||
|
||||
|
||||
def plotData(X, y, grid=False):
|
||||
"""
|
||||
Plots the data points X and y into a new figure. Uses `+` for positive examples, and `o` for
|
||||
negative examples. `X` is assumed to be a Mx2 matrix
|
||||
|
||||
Parameters
|
||||
----------
|
||||
X : numpy ndarray
|
||||
X is assumed to be a Mx2 matrix.
|
||||
|
||||
y : numpy ndarray
|
||||
The data labels.
|
||||
|
||||
grid : bool (Optional)
|
||||
Specify whether or not to show the grid in the plot. It is False by default.
|
||||
|
||||
Notes
|
||||
-----
|
||||
This was slightly modified such that it expects y=1 or y=0.
|
||||
"""
|
||||
# Find Indices of Positive and Negative Examples
|
||||
pos = y == 1
|
||||
neg = y == 0
|
||||
|
||||
# Plot Examples
|
||||
pyplot.plot(X[pos, 0], X[pos, 1], 'X', mew=1, ms=10, mec='k')
|
||||
pyplot.plot(X[neg, 0], X[neg, 1], 'o', mew=1, mfc='y', ms=10, mec='k')
|
||||
pyplot.grid(grid)
|
||||
|
||||
|
||||
def svmTrain(X, Y, C, kernelFunction, tol=1e-3, max_passes=5, args=()):
|
||||
"""
|
||||
Trains an SVM classifier using a simplified version of the SMO algorithm.
|
||||
|
||||
Parameters
|
||||
---------
|
||||
X : numpy ndarray
|
||||
(m x n) Matrix of training examples. Each row is a training example, and the
|
||||
jth column holds the jth feature.
|
||||
|
||||
Y : numpy ndarray
|
||||
(m, ) A vector (1-D numpy array) containing 1 for positive examples and 0 for negative examples.
|
||||
|
||||
C : float
|
||||
The standard SVM regularization parameter.
|
||||
|
||||
kernelFunction : func
|
||||
A function handle which computes the kernel. The function should accept two vectors as
|
||||
inputs, and returns a scalar as output.
|
||||
|
||||
tol : float, optional
|
||||
Tolerance value used for determining equality of floating point numbers.
|
||||
|
||||
max_passes : int, optional
|
||||
Controls the number of iterations over the dataset (without changes to alpha)
|
||||
before the algorithm quits.
|
||||
|
||||
args : tuple
|
||||
Extra arguments required for the kernel function, such as the sigma parameter for a
|
||||
Gaussian kernel.
|
||||
|
||||
Returns
|
||||
-------
|
||||
model :
|
||||
The trained SVM model.
|
||||
|
||||
Notes
|
||||
-----
|
||||
This is a simplified version of the SMO algorithm for training SVMs. In practice, if
|
||||
you want to train an SVM classifier, we recommend using an optimized package such as:
|
||||
|
||||
- LIBSVM (http://www.csie.ntu.edu.tw/~cjlin/libsvm/)
|
||||
- SVMLight (http://svmlight.joachims.org/)
|
||||
- scikit-learn (http://scikit-learn.org/stable/modules/svm.html) which contains python wrappers
|
||||
for the LIBSVM library.
|
||||
"""
|
||||
# make sure data is signed int
|
||||
Y = Y.astype(int)
|
||||
# Dataset size parameters
|
||||
m, n = X.shape
|
||||
|
||||
passes = 0
|
||||
E = np.zeros(m)
|
||||
alphas = np.zeros(m)
|
||||
b = 0
|
||||
|
||||
# Map 0 to -1
|
||||
Y[Y == 0] = -1
|
||||
|
||||
# Pre-compute the Kernel Matrix since our dataset is small
|
||||
# (in practice, optimized SVM packages that handle large datasets
|
||||
# gracefully will **not** do this)
|
||||
|
||||
# We have implemented the optimized vectorized version of the Kernels here so
|
||||
# that the SVM training will run faster
|
||||
if kernelFunction.__name__ == 'linearKernel':
|
||||
# Vectorized computation for the linear kernel
|
||||
# This is equivalent to computing the kernel on every pair of examples
|
||||
K = np.dot(X, X.T)
|
||||
elif kernelFunction.__name__ == 'gaussianKernel':
|
||||
# vectorized RBF Kernel
|
||||
# This is equivalent to computing the kernel on every pair of examples
|
||||
X2 = np.sum(X**2, axis=1)
|
||||
K = X2 + X2[:, None] - 2 * np.dot(X, X.T)
|
||||
|
||||
if len(args) > 0:
|
||||
K /= 2*args[0]**2
|
||||
|
||||
K = np.exp(-K)
|
||||
else:
|
||||
K = np.zeros((m, m))
|
||||
for i in range(m):
|
||||
for j in range(i, m):
|
||||
K[i, j] = kernelFunction(X[i, :], X[j, :])
|
||||
K[j, i] = K[i, j]
|
||||
|
||||
while passes < max_passes:
|
||||
num_changed_alphas = 0
|
||||
for i in range(m):
|
||||
E[i] = b + np.sum(alphas * Y * K[:, i]) - Y[i]
|
||||
|
||||
if (Y[i]*E[i] < -tol and alphas[i] < C) or (Y[i]*E[i] > tol and alphas[i] > 0):
|
||||
# select the alpha_j randomly
|
||||
j = np.random.choice(list(range(i)) + list(range(i+1, m)), size=1)[0]
|
||||
|
||||
E[j] = b + np.sum(alphas * Y * K[:, j]) - Y[j]
|
||||
|
||||
alpha_i_old = alphas[i]
|
||||
alpha_j_old = alphas[j]
|
||||
|
||||
if Y[i] == Y[j]:
|
||||
L = max(0, alphas[j] + alphas[i] - C)
|
||||
H = min(C, alphas[j] + alphas[i])
|
||||
else:
|
||||
L = max(0, alphas[j] - alphas[i])
|
||||
H = min(C, C + alphas[j] - alphas[i])
|
||||
|
||||
if L == H:
|
||||
continue
|
||||
|
||||
eta = 2 * K[i, j] - K[i, i] - K[j, j]
|
||||
|
||||
# objective function positive definite, there will be a minimum along the direction
|
||||
# of linear equality constrain, and eta will be greater than zero
|
||||
# we are actually computing -eta here (so we skip of eta >= 0)
|
||||
if eta >= 0:
|
||||
continue
|
||||
|
||||
alphas[j] -= Y[j] * (E[i] - E[j])/eta
|
||||
alphas[j] = max(L, min(H, alphas[j]))
|
||||
|
||||
if abs(alphas[j] - alpha_j_old) < tol:
|
||||
alphas[j] = alpha_j_old
|
||||
continue
|
||||
alphas[i] += Y[i]*Y[j]*(alpha_j_old - alphas[j])
|
||||
|
||||
b1 = b - E[i] - Y[i]*(alphas[i] - alpha_i_old) * K[i, j] \
|
||||
- Y[j] * (alphas[j] - alpha_j_old) * K[i, j]
|
||||
|
||||
b2 = b - E[j] - Y[i]*(alphas[i] - alpha_i_old) * K[i, j] \
|
||||
- Y[j] * (alphas[j] - alpha_j_old) * K[j, j]
|
||||
|
||||
if 0 < alphas[i] < C:
|
||||
b = b1
|
||||
elif 0 < alphas[j] < C:
|
||||
b = b2
|
||||
else:
|
||||
b = (b1 + b2)/2
|
||||
|
||||
num_changed_alphas += 1
|
||||
if num_changed_alphas == 0:
|
||||
passes += 1
|
||||
else:
|
||||
passes = 0
|
||||
|
||||
idx = alphas > 0
|
||||
model = {'X': X[idx, :],
|
||||
'y': Y[idx],
|
||||
'kernelFunction': kernelFunction,
|
||||
'b': b,
|
||||
'args': args,
|
||||
'alphas': alphas[idx],
|
||||
'w': np.dot(alphas * Y, X)}
|
||||
return model
|
||||
|
||||
|
||||
def svmPredict(model, X):
|
||||
"""
|
||||
Returns a vector of predictions using a trained SVM model.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
model : dict
|
||||
The parameters of the trained svm model, as returned by the function svmTrain
|
||||
|
||||
X : array_like
|
||||
A (m x n) matrix where each example is a row.
|
||||
|
||||
Returns
|
||||
-------
|
||||
pred : array_like
|
||||
A (m,) sized vector of predictions {0, 1} values.
|
||||
"""
|
||||
# check if we are getting a vector. If so, then assume we only need to do predictions
|
||||
# for a single example
|
||||
if X.ndim == 1:
|
||||
X = X[np.newaxis, :]
|
||||
|
||||
m = X.shape[0]
|
||||
p = np.zeros(m)
|
||||
pred = np.zeros(m)
|
||||
|
||||
if model['kernelFunction'].__name__ == 'linearKernel':
|
||||
# we can use the weights and bias directly if working with the linear kernel
|
||||
p = np.dot(X, model['w']) + model['b']
|
||||
elif model['kernelFunction'].__name__ == 'gaussianKernel':
|
||||
# vectorized RBF Kernel
|
||||
# This is equivalent to computing the kernel on every pair of examples
|
||||
X1 = np.sum(X**2, 1)
|
||||
X2 = np.sum(model['X']**2, 1)
|
||||
K = X2 + X1[:, None] - 2 * np.dot(X, model['X'].T)
|
||||
|
||||
if len(model['args']) > 0:
|
||||
K /= 2*model['args'][0]**2
|
||||
|
||||
K = np.exp(-K)
|
||||
p = np.dot(K, model['alphas']*model['y']) + model['b']
|
||||
else:
|
||||
# other non-linear kernel
|
||||
for i in range(m):
|
||||
predictions = 0
|
||||
for j in range(model['X'].shape[0]):
|
||||
predictions += model['alphas'][j] * model['y'][j] \
|
||||
* model['kernelFunction'](X[i, :], model['X'][j, :])
|
||||
p[i] = predictions
|
||||
|
||||
pred[p >= 0] = 1
|
||||
return pred
|
||||
|
||||
|
||||
def linearKernel(x1, x2):
|
||||
"""
|
||||
Returns a linear kernel between x1 and x2.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
x1 : numpy ndarray
|
||||
A 1-D vector.
|
||||
|
||||
x2 : numpy ndarray
|
||||
A 1-D vector of same size as x1.
|
||||
|
||||
Returns
|
||||
-------
|
||||
: float
|
||||
The scalar amplitude.
|
||||
"""
|
||||
return np.dot(x1, x2)
|
||||
|
||||
|
||||
def visualizeBoundaryLinear(X, y, model):
|
||||
"""
|
||||
Plots a linear decision boundary learned by the SVM.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
X : array_like
|
||||
(m x 2) The training data with two features (to plot in a 2-D plane).
|
||||
|
||||
y : array_like
|
||||
(m, ) The data labels.
|
||||
|
||||
model : dict
|
||||
Dictionary of model variables learned by SVM.
|
||||
"""
|
||||
w, b = model['w'], model['b']
|
||||
xp = np.linspace(min(X[:, 0]), max(X[:, 0]), 100)
|
||||
yp = -(w[0] * xp + b)/w[1]
|
||||
|
||||
plotData(X, y)
|
||||
pyplot.plot(xp, yp, '-b')
|
||||
|
||||
|
||||
def visualizeBoundary(X, y, model):
|
||||
"""
|
||||
Plots a non-linear decision boundary learned by the SVM and overlays the data on it.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
X : array_like
|
||||
(m x 2) The training data with two features (to plot in a 2-D plane).
|
||||
|
||||
y : array_like
|
||||
(m, ) The data labels.
|
||||
|
||||
model : dict
|
||||
Dictionary of model variables learned by SVM.
|
||||
"""
|
||||
plotData(X, y)
|
||||
|
||||
# make classification predictions over a grid of values
|
||||
x1plot = np.linspace(min(X[:, 0]), max(X[:, 0]), 100)
|
||||
x2plot = np.linspace(min(X[:, 1]), max(X[:, 1]), 100)
|
||||
X1, X2 = np.meshgrid(x1plot, x2plot)
|
||||
|
||||
vals = np.zeros(X1.shape)
|
||||
for i in range(X1.shape[1]):
|
||||
this_X = np.stack((X1[:, i], X2[:, i]), axis=1)
|
||||
vals[:, i] = svmPredict(model, this_X)
|
||||
|
||||
pyplot.contour(X1, X2, vals, colors='y', linewidths=2)
|
||||
pyplot.pcolormesh(X1, X2, vals, cmap='YlGnBu', alpha=0.25, edgecolors='None', lw=0)
|
||||
pyplot.grid(False)
|
||||
|
||||
|
||||
def getVocabList():
|
||||
"""
|
||||
Reads the fixed vocabulary list in vocab.txt and returns a cell array of the words
|
||||
% vocabList = GETVOCABLIST() reads the fixed vocabulary list in vocab.txt
|
||||
% and returns a cell array of the words in vocabList.
|
||||
|
||||
:return:
|
||||
"""
|
||||
vocabList = np.genfromtxt(join('Data', 'vocab.txt'), dtype=object)
|
||||
return list(vocabList[:, 1].astype(str))
|
||||
|
||||
|
||||
class PorterStemmer:
|
||||
"""
|
||||
Porter Stemming Algorithm
|
||||
|
||||
This is the Porter stemming algorithm, ported to Python from the
|
||||
version coded up in ANSI C by the author. It may be be regarded
|
||||
as canonical, in that it follows the algorithm presented in
|
||||
|
||||
Porter, 1980, An algorithm for suffix stripping, Program, Vol. 14,
|
||||
no. 3, pp 130-137,
|
||||
|
||||
only differing from it at the points maked --DEPARTURE-- below.
|
||||
|
||||
See also http://www.tartarus.org/~martin/PorterStemmer
|
||||
|
||||
The algorithm as described in the paper could be exactly replicated
|
||||
by adjusting the points of DEPARTURE, but this is barely necessary,
|
||||
because (a) the points of DEPARTURE are definitely improvements, and
|
||||
(b) no encoding of the Porter stemmer I have seen is anything like
|
||||
as exact as this version, even with the points of DEPARTURE!
|
||||
|
||||
Vivake Gupta (v@nano.com)
|
||||
|
||||
Release 1: January 2001
|
||||
|
||||
Further adjustments by Santiago Bruno (bananabruno@gmail.com)
|
||||
to allow word input not restricted to one word per line, leading
|
||||
to:
|
||||
|
||||
release 2: July 2008
|
||||
"""
|
||||
def __init__(self):
|
||||
"""
|
||||
The main part of the stemming algorithm starts here.
|
||||
b is a buffer holding a word to be stemmed. The letters are in b[k0],
|
||||
b[k0+1] ... ending at b[k]. In fact k0 = 0 in this demo program. k is
|
||||
readjusted downwards as the stemming progresses. Zero termination is
|
||||
not in fact used in the algorithm.
|
||||
|
||||
Note that only lower case sequences are stemmed. Forcing to lower case
|
||||
should be done before stem(...) is called.
|
||||
"""
|
||||
self.b = "" # buffer for word to be stemmed
|
||||
self.k = 0
|
||||
self.k0 = 0
|
||||
self.j = 0 # j is a general offset into the string
|
||||
|
||||
def cons(self, i):
|
||||
"""cons(i) is TRUE <=> b[i] is a consonant."""
|
||||
if self.b[i] in 'aeiou':
|
||||
return 0
|
||||
if self.b[i] == 'y':
|
||||
if i == self.k0:
|
||||
return 1
|
||||
else:
|
||||
return not self.cons(i - 1)
|
||||
return 1
|
||||
|
||||
def m(self):
|
||||
"""
|
||||
m() measures the number of consonant sequences between k0 and j.
|
||||
if c is a consonant sequence and v a vowel sequence, and <..>
|
||||
indicates arbitrary presence,
|
||||
|
||||
<c><v> gives 0
|
||||
<c>vc<v> gives 1
|
||||
<c>vcvc<v> gives 2
|
||||
<c>vcvcvc<v> gives 3
|
||||
....
|
||||
"""
|
||||
n = 0
|
||||
i = self.k0
|
||||
while 1:
|
||||
if i > self.j:
|
||||
return n
|
||||
if not self.cons(i):
|
||||
break
|
||||
i = i + 1
|
||||
i = i + 1
|
||||
while 1:
|
||||
while 1:
|
||||
if i > self.j:
|
||||
return n
|
||||
if self.cons(i):
|
||||
break
|
||||
i = i + 1
|
||||
i = i + 1
|
||||
n = n + 1
|
||||
while 1:
|
||||
if i > self.j:
|
||||
return n
|
||||
if not self.cons(i):
|
||||
break
|
||||
i = i + 1
|
||||
i = i + 1
|
||||
|
||||
def vowelinstem(self):
|
||||
"""vowelinstem() is TRUE <=> k0,...j contains a vowel"""
|
||||
for i in range(self.k0, self.j + 1):
|
||||
if not self.cons(i):
|
||||
return 1
|
||||
return 0
|
||||
|
||||
def doublec(self, j):
|
||||
""" doublec(j) is TRUE <=> j,(j-1) contain a double consonant. """
|
||||
if j < (self.k0 + 1):
|
||||
return 0
|
||||
if self.b[j] != self.b[j-1]:
|
||||
return 0
|
||||
return self.cons(j)
|
||||
|
||||
def cvc(self, i):
|
||||
"""
|
||||
cvc(i) is TRUE <=> i-2,i-1,i has the form consonant - vowel - consonant
|
||||
and also if the second c is not w,x or y. this is used when trying to
|
||||
restore an e at the end of a short e.g.
|
||||
|
||||
cav(e), lov(e), hop(e), crim(e), but
|
||||
snow, box, tray.
|
||||
"""
|
||||
if i < (self.k0 + 2) or not self.cons(i) or self.cons(i-1) or not self.cons(i-2):
|
||||
return 0
|
||||
ch = self.b[i]
|
||||
if ch in 'wxy':
|
||||
return 0
|
||||
return 1
|
||||
|
||||
def ends(self, s):
|
||||
"""ends(s) is TRUE <=> k0,...k ends with the string s."""
|
||||
length = len(s)
|
||||
if s[length - 1] != self.b[self.k]: # tiny speed-up
|
||||
return 0
|
||||
if length > (self.k - self.k0 + 1):
|
||||
return 0
|
||||
if self.b[self.k-length+1:self.k+1] != s:
|
||||
return 0
|
||||
self.j = self.k - length
|
||||
return 1
|
||||
|
||||
def setto(self, s):
|
||||
"""setto(s) sets (j+1),...k to the characters in the string s, readjusting k."""
|
||||
length = len(s)
|
||||
self.b = self.b[:self.j+1] + s + self.b[self.j+length+1:]
|
||||
self.k = self.j + length
|
||||
|
||||
def r(self, s):
|
||||
"""r(s) is used further down."""
|
||||
if self.m() > 0:
|
||||
self.setto(s)
|
||||
|
||||
def step1ab(self):
|
||||
"""step1ab() gets rid of plurals and -ed or -ing. e.g.
|
||||
|
||||
caresses -> caress
|
||||
ponies -> poni
|
||||
ties -> ti
|
||||
caress -> caress
|
||||
cats -> cat
|
||||
|
||||
feed -> feed
|
||||
agreed -> agree
|
||||
disabled -> disable
|
||||
|
||||
matting -> mat
|
||||
mating -> mate
|
||||
meeting -> meet
|
||||
milling -> mill
|
||||
messing -> mess
|
||||
|
||||
meetings -> meet
|
||||
"""
|
||||
if self.b[self.k] == 's':
|
||||
if self.ends("sses"):
|
||||
self.k = self.k - 2
|
||||
elif self.ends("ies"):
|
||||
self.setto("i")
|
||||
elif self.b[self.k - 1] != 's':
|
||||
self.k = self.k - 1
|
||||
if self.ends("eed"):
|
||||
if self.m() > 0:
|
||||
self.k = self.k - 1
|
||||
elif (self.ends("ed") or self.ends("ing")) and self.vowelinstem():
|
||||
self.k = self.j
|
||||
if self.ends("at"):
|
||||
self.setto("ate")
|
||||
elif self.ends("bl"):
|
||||
self.setto("ble")
|
||||
elif self.ends("iz"):
|
||||
self.setto("ize")
|
||||
elif self.doublec(self.k):
|
||||
self.k = self.k - 1
|
||||
ch = self.b[self.k]
|
||||
if ch in 'lsz':
|
||||
self.k += 1
|
||||
elif self.m() == 1 and self.cvc(self.k):
|
||||
self.setto("e")
|
||||
|
||||
def step1c(self):
|
||||
"""step1c() turns terminal y to i when there is another vowel in the stem."""
|
||||
if self.ends("y") and self.vowelinstem():
|
||||
self.b = self.b[:self.k] + 'i' + self.b[self.k+1:]
|
||||
|
||||
def step2(self):
|
||||
"""step2() maps double suffices to single ones.
|
||||
so -ization ( = -ize plus -ation) maps to -ize etc. note that the
|
||||
string before the suffix must give m() > 0.
|
||||
"""
|
||||
if self.b[self.k - 1] == 'a':
|
||||
if self.ends("ational"): self.r("ate")
|
||||
elif self.ends("tional"): self.r("tion")
|
||||
elif self.b[self.k - 1] == 'c':
|
||||
if self.ends("enci"): self.r("ence")
|
||||
elif self.ends("anci"): self.r("ance")
|
||||
elif self.b[self.k - 1] == 'e':
|
||||
if self.ends("izer"): self.r("ize")
|
||||
elif self.b[self.k - 1] == 'l':
|
||||
if self.ends("bli"): self.r("ble") # --DEPARTURE--
|
||||
# To match the published algorithm, replace this phrase with
|
||||
# if self.ends("abli"): self.r("able")
|
||||
elif self.ends("alli"): self.r("al")
|
||||
elif self.ends("entli"): self.r("ent")
|
||||
elif self.ends("eli"): self.r("e")
|
||||
elif self.ends("ousli"): self.r("ous")
|
||||
elif self.b[self.k - 1] == 'o':
|
||||
if self.ends("ization"): self.r("ize")
|
||||
elif self.ends("ation"): self.r("ate")
|
||||
elif self.ends("ator"): self.r("ate")
|
||||
elif self.b[self.k - 1] == 's':
|
||||
if self.ends("alism"): self.r("al")
|
||||
elif self.ends("iveness"): self.r("ive")
|
||||
elif self.ends("fulness"): self.r("ful")
|
||||
elif self.ends("ousness"): self.r("ous")
|
||||
elif self.b[self.k - 1] == 't':
|
||||
if self.ends("aliti"): self.r("al")
|
||||
elif self.ends("iviti"): self.r("ive")
|
||||
elif self.ends("biliti"): self.r("ble")
|
||||
elif self.b[self.k - 1] == 'g': # --DEPARTURE--
|
||||
if self.ends("logi"): self.r("log")
|
||||
# To match the published algorithm, delete this phrase
|
||||
|
||||
def step3(self):
|
||||
"""step3() dels with -ic-, -full, -ness etc. similar strategy to step2."""
|
||||
if self.b[self.k] == 'e':
|
||||
if self.ends("icate"): self.r("ic")
|
||||
elif self.ends("ative"): self.r("")
|
||||
elif self.ends("alize"): self.r("al")
|
||||
elif self.b[self.k] == 'i':
|
||||
if self.ends("iciti"): self.r("ic")
|
||||
elif self.b[self.k] == 'l':
|
||||
if self.ends("ical"): self.r("ic")
|
||||
elif self.ends("ful"): self.r("")
|
||||
elif self.b[self.k] == 's':
|
||||
if self.ends("ness"): self.r("")
|
||||
|
||||
def step4(self):
|
||||
"""step4() takes off -ant, -ence etc., in context <c>vcvc<v>."""
|
||||
if self.b[self.k - 1] == 'a':
|
||||
if self.ends("al"): pass
|
||||
else: return
|
||||
elif self.b[self.k - 1] == 'c':
|
||||
if self.ends("ance"): pass
|
||||
elif self.ends("ence"): pass
|
||||
else: return
|
||||
elif self.b[self.k - 1] == 'e':
|
||||
if self.ends("er"): pass
|
||||
else: return
|
||||
elif self.b[self.k - 1] == 'i':
|
||||
if self.ends("ic"): pass
|
||||
else: return
|
||||
elif self.b[self.k - 1] == 'l':
|
||||
if self.ends("able"): pass
|
||||
elif self.ends("ible"): pass
|
||||
else: return
|
||||
elif self.b[self.k - 1] == 'n':
|
||||
if self.ends("ant"): pass
|
||||
elif self.ends("ement"): pass
|
||||
elif self.ends("ment"): pass
|
||||
elif self.ends("ent"): pass
|
||||
else: return
|
||||
elif self.b[self.k - 1] == 'o':
|
||||
if self.ends("ion") and (self.b[self.j] == 's' or self.b[self.j] == 't'): pass
|
||||
elif self.ends("ou"): pass
|
||||
# takes care of -ous
|
||||
else: return
|
||||
elif self.b[self.k - 1] == 's':
|
||||
if self.ends("ism"): pass
|
||||
else: return
|
||||
elif self.b[self.k - 1] == 't':
|
||||
if self.ends("ate"): pass
|
||||
elif self.ends("iti"): pass
|
||||
else: return
|
||||
elif self.b[self.k - 1] == 'u':
|
||||
if self.ends("ous"): pass
|
||||
else: return
|
||||
elif self.b[self.k - 1] == 'v':
|
||||
if self.ends("ive"): pass
|
||||
else: return
|
||||
elif self.b[self.k - 1] == 'z':
|
||||
if self.ends("ize"): pass
|
||||
else: return
|
||||
else:
|
||||
return
|
||||
if self.m() > 1:
|
||||
self.k = self.j
|
||||
|
||||
def step5(self):
|
||||
"""step5() removes a final -e if m() > 1, and changes -ll to -l if
|
||||
m() > 1.
|
||||
"""
|
||||
self.j = self.k
|
||||
if self.b[self.k] == 'e':
|
||||
a = self.m()
|
||||
if a > 1 or (a == 1 and not self.cvc(self.k-1)):
|
||||
self.k = self.k - 1
|
||||
if self.b[self.k] == 'l' and self.doublec(self.k) and self.m() > 1:
|
||||
self.k = self.k -1
|
||||
|
||||
def stem(self, p, i=0, j=None):
|
||||
"""In stem(p,i,j), p is a char pointer, and the string to be stemmed
|
||||
is from p[i] to p[j] inclusive. Typically i is zero and j is the
|
||||
offset to the last character of a string, (p[j+1] == '\0'). The
|
||||
stemmer adjusts the characters p[i] ... p[j] and returns the new
|
||||
end-point of the string, k. Stemming never increases word length, so
|
||||
i <= k <= j. To turn the stemmer into a module, declare 'stem' as
|
||||
extern, and delete the remainder of this file.
|
||||
"""
|
||||
# copy the parameters into statics
|
||||
self.b = p
|
||||
self.k = j or len(p) - 1
|
||||
self.k0 = i
|
||||
if self.k <= self.k0 + 1:
|
||||
return self.b # --DEPARTURE--
|
||||
|
||||
# With this line, strings of length 1 or 2 don't go through the
|
||||
# stemming process, although no mention is made of this in the
|
||||
# published algorithm. Remove the line to match the published
|
||||
# algorithm.
|
||||
|
||||
self.step1ab()
|
||||
self.step1c()
|
||||
self.step2()
|
||||
self.step3()
|
||||
self.step4()
|
||||
self.step5()
|
||||
return self.b[self.k0:self.k+1]
|
||||
|
||||
|
||||
class Grader(SubmissionBase):
|
||||
# Random Test Cases
|
||||
x1 = np.sin(np.arange(1, 11))
|
||||
x2 = np.cos(np.arange(1, 11))
|
||||
ec = 'the quick brown fox jumped over the lazy dog'
|
||||
wi = np.abs(np.round(x1 * 1863)).astype(int)
|
||||
wi = np.concatenate([wi, wi])
|
||||
|
||||
def __init__(self):
|
||||
part_names = ['Gaussian Kernel',
|
||||
'Parameters (C, sigma) for Dataset 3',
|
||||
'Email Processing',
|
||||
'Email Feature Extraction']
|
||||
super().__init__('support-vector-machines', part_names)
|
||||
|
||||
def __iter__(self):
|
||||
for part_id in range(1, 5):
|
||||
try:
|
||||
func = self.functions[part_id]
|
||||
# Each part has different expected arguments/different function
|
||||
if part_id == 1:
|
||||
res = func(self.x1, self.x2, 2)
|
||||
elif part_id == 2:
|
||||
res = np.hstack(func()).tolist()
|
||||
elif part_id == 3:
|
||||
# add one to be compatible with matlab grader
|
||||
res = [ind+1 for ind in func(self.ec, False)]
|
||||
elif part_id == 4:
|
||||
res = func(self.wi)
|
||||
else:
|
||||
raise KeyError
|
||||
yield part_id, res
|
||||
except KeyError:
|
||||
yield part_id, 0
|
||||