diff --git a/doc/modules/svm.rst b/doc/modules/svm.rst index 4ff7c8b044b..1da3298fd38 100644 --- a/doc/modules/svm.rst +++ b/doc/modules/svm.rst @@ -18,7 +18,7 @@ functional margin), since in general the larger the margin the lower the generalization error of the classifier. Classification --------------- +============== Suppose some given data points each belong to one of two classes, and the goal is to decide which class a new data point will be in. This @@ -71,27 +71,29 @@ Complete class reference: :members: Using Custom Kernels -^^^^^^^^^^^^^^^^^^^^ +-------------------- You can also use your own defined kernels by passing a function to the keyword `kernel` in the constructor. -Your kernel must take as arguments two arrays and return a -floating-point number. +Your kernel must take as arguments two matrices and return a third matrix. -The following code defines a valid kernel and creates a Support Vector -Machine with that associated kernel:: +The following code defines a linear kernel and creates a classifier +instance that will use that kernel:: >>> import numpy as np >>> from scikits.learn import svm >>> def my_kernel(x, y): - ... return np.tanh(np.dot(x, y.T) - 1) + ... return np.dot(x, y.T) ... >>> clf = svm.SVC(kernel=my_kernel) +For a complete example, see :ref:`example_svm_plot_custom_kernel.py` + Regression ----------- +========== + The method of Support Vector Classification can be extended to solve the regression problem. This method is called Support Vector Regression. @@ -114,7 +116,7 @@ Distribution estimation One-class SVM is used for out-layer detection, that is, given a set of samples, it will detect the soft boundary of that set. -.. literalinclude:: ../../examples/plot_svm_oneclass.py +.. literalinclude:: ../../examples/svm/plot_svm_oneclass.py .. image:: svm_data/oneclass.png diff --git a/doc/sphinxext/gen_rst.py b/doc/sphinxext/gen_rst.py index 12565cafdec..af1b4683579 100644 --- a/doc/sphinxext/gen_rst.py +++ b/doc/sphinxext/gen_rst.py @@ -141,6 +141,7 @@ def generate_file_rst(fname, target_dir, src_dir): last_dir = os.path.split(src_dir)[-1] # to avoid leading . in file names if last_dir == '.': last_dir = '' + else: last_dir += '_' short_fname = last_dir + fname src_file = os.path.join(src_dir, fname) example_file = os.path.join(target_dir, fname) diff --git a/examples/svm/plot_custom_kernel.py b/examples/svm/plot_custom_kernel.py new file mode 100644 index 00000000000..79fa10c4eff --- /dev/null +++ b/examples/svm/plot_custom_kernel.py @@ -0,0 +1,56 @@ +""" +====================== +SVM with custom kernel +====================== + +Simple usage of Support Vector Machines to classify a sample. It will +plot the decision surface and the support vectors. + +""" +import numpy as np +import pylab as pl +from scikits.learn import svm, datasets + +# import some data to play with +iris = datasets.load_iris() +X = iris.data[:, :2] # we only take the first two features. We could + # avoid this ugly slicing by using a two-dim dataset +Y = iris.target + + +def my_kernel(x, y): + """ + We create a custom kernel: + + (2 0) + k(x, y) = x ( ) y.T + (0 1) + """ + M = np.array([[2, 0], [0, 1.0]]) + return np.dot(np.dot(x, M), y.T) + + +h=.02 # step size in the mesh + +# we create an instance of SVM and fit out data. We do not scale our +# data since we want to plot the support vectors +clf = svm.SVC(kernel=my_kernel) +clf.fit(X, Y) + +# Plot the decision boundary. For that, we will asign a color to each +# point in the mesh [x_min, m_max]x[y_min, y_max]. +x_min, x_max = X[:,0].min()-1, X[:,0].max()+1 +y_min, y_max = X[:,1].min()-1, X[:,1].max()+1 +xx, yy = np.meshgrid(np.arange(x_min, x_max, h), np.arange(y_min, y_max, h)) +Z = clf.predict(np.c_[xx.ravel(), yy.ravel()]) + +# Put the result into a color plot +Z = Z.reshape(xx.shape) +pl.set_cmap(pl.cm.Paired) +pl.pcolormesh(xx, yy, Z) + +# Plot also the training points +pl.scatter(X[:,0], X[:,1], c=Y) +pl.title('3-Class classification using Support Vector Machine with custom kernel') +pl.axis('tight') +pl.show()