2011-12-20 20:23:03 +08:00
|
|
|
"""
|
|
|
|
|
===========================================
|
|
|
|
|
Sparse coding with a precomputed dictionary
|
|
|
|
|
===========================================
|
|
|
|
|
|
|
|
|
|
Transform a signal as a sparse combination of Ricker wavelets. This example
|
|
|
|
|
visually compares different sparse coding methods using the
|
2011-12-22 23:33:15 +08:00
|
|
|
:class:`sklearn.decomposition.SparseCoder` estimator. The Ricker (also known
|
|
|
|
|
as mexican hat or the second derivative of a gaussian) is not a particularily
|
|
|
|
|
good kernel to represent piecewise constant signals like this one. It can
|
|
|
|
|
therefore be seen how much adding different widths of atoms matters and it
|
|
|
|
|
therefore motivates learning the dictionary to best fit your type of signals.
|
|
|
|
|
|
|
|
|
|
The richer dictionary on the right is not larger in size, heavier subsampling
|
|
|
|
|
is performed in order to stay on the same order of magnitude.
|
2011-12-20 20:23:03 +08:00
|
|
|
"""
|
2011-12-20 21:17:11 +08:00
|
|
|
print __doc__
|
2011-12-20 20:23:03 +08:00
|
|
|
|
|
|
|
|
import numpy as np
|
|
|
|
|
import matplotlib.pylab as pl
|
|
|
|
|
|
|
|
|
|
from sklearn.decomposition import SparseCoder
|
|
|
|
|
|
2011-12-20 21:17:11 +08:00
|
|
|
|
2011-12-20 20:23:03 +08:00
|
|
|
def ricker_function(resolution, center, width):
|
|
|
|
|
"""Discrete sub-sampled Ricker (mexican hat) wavelet"""
|
|
|
|
|
x = np.linspace(0, resolution - 1, resolution)
|
2011-12-20 21:17:11 +08:00
|
|
|
x = (2 / ((np.sqrt(3 * width) * np.pi ** 1 / 4))) * (
|
2011-12-20 20:23:03 +08:00
|
|
|
1 - ((x - center) ** 2 / width ** 2)) * np.exp(
|
|
|
|
|
(-(x - center) ** 2) / (2 * width ** 2))
|
|
|
|
|
return x
|
|
|
|
|
|
|
|
|
|
|
2012-11-03 19:35:12 +08:00
|
|
|
def ricker_matrix(width, resolution, n_components):
|
2011-12-20 20:23:03 +08:00
|
|
|
"""Dictionary of Ricker (mexican hat) wavelets"""
|
2012-11-03 19:35:12 +08:00
|
|
|
centers = np.linspace(0, resolution - 1, n_components)
|
|
|
|
|
D = np.empty((n_components, resolution))
|
2011-12-20 20:23:03 +08:00
|
|
|
for i, center in enumerate(centers):
|
|
|
|
|
D[i] = ricker_function(resolution, center, width)
|
2011-12-20 21:17:11 +08:00
|
|
|
D /= np.sqrt(np.sum(D ** 2, axis=1))[:, np.newaxis]
|
2011-12-20 20:23:03 +08:00
|
|
|
return D
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
resolution = 1024
|
|
|
|
|
subsampling = 3 # subsampling factor
|
2011-12-22 23:33:15 +08:00
|
|
|
width = 100
|
2012-11-03 19:35:12 +08:00
|
|
|
n_components = resolution / subsampling
|
2011-12-20 20:23:03 +08:00
|
|
|
|
|
|
|
|
# Compute a wavelet dictionary
|
2012-11-03 19:35:12 +08:00
|
|
|
D_fixed = ricker_matrix(width=width, resolution=resolution, n_components=n_components)
|
2011-12-22 23:33:15 +08:00
|
|
|
D_multi = np.r_[tuple(ricker_matrix(width=w, resolution=resolution,
|
2012-11-03 19:35:12 +08:00
|
|
|
n_components=np.floor(n_components / 5))
|
2011-12-22 23:33:15 +08:00
|
|
|
for w in (10, 50, 100, 500, 1000))]
|
2011-12-20 20:23:03 +08:00
|
|
|
|
|
|
|
|
# Generate a signal
|
|
|
|
|
y = np.linspace(0, resolution - 1, resolution)
|
2011-12-21 00:09:15 +08:00
|
|
|
first_quarter = y < resolution / 4
|
2011-12-22 23:33:15 +08:00
|
|
|
y[first_quarter] = 3.
|
|
|
|
|
y[np.logical_not(first_quarter)] = -1.
|
2011-12-20 20:23:03 +08:00
|
|
|
|
|
|
|
|
# List the different sparse coding methods in the following format:
|
|
|
|
|
# (title, transform_algorithm, transform_alpha, transform_n_nozero_coefs)
|
2011-12-21 00:09:15 +08:00
|
|
|
estimators = [('OMP', 'omp', None, 15),
|
2011-12-22 23:33:15 +08:00
|
|
|
('Lasso', 'lasso_cd', 2, None),
|
2011-12-21 00:09:15 +08:00
|
|
|
]
|
2011-12-20 20:23:03 +08:00
|
|
|
|
2011-12-22 23:33:15 +08:00
|
|
|
pl.figure(figsize=(13, 6))
|
|
|
|
|
for subplot, (D, title) in enumerate(zip((D_fixed, D_multi),
|
|
|
|
|
('fixed width', 'multiple widths'))):
|
|
|
|
|
pl.subplot(1, 2, subplot + 1)
|
|
|
|
|
pl.title('Sparse coding against %s dictionary' % title)
|
|
|
|
|
pl.plot(y, ls='dotted', label='Original signal')
|
|
|
|
|
# Do a wavelet approximation
|
|
|
|
|
for title, algo, alpha, n_nonzero in estimators:
|
|
|
|
|
coder = SparseCoder(dictionary=D, transform_n_nonzero_coefs=n_nonzero,
|
|
|
|
|
transform_alpha=alpha, transform_algorithm=algo)
|
|
|
|
|
x = coder.transform(y)
|
|
|
|
|
density = len(np.flatnonzero(x))
|
|
|
|
|
x = np.ravel(np.dot(x, D))
|
|
|
|
|
squared_error = np.sum((y - x) ** 2)
|
|
|
|
|
pl.plot(x, label='%s: %s nonzero coefs,\n%.2f error' %
|
|
|
|
|
(title, density, squared_error))
|
2011-12-20 20:23:03 +08:00
|
|
|
|
2011-12-22 23:33:15 +08:00
|
|
|
# Soft thresholding debiasing
|
|
|
|
|
coder = SparseCoder(dictionary=D, transform_algorithm='threshold',
|
|
|
|
|
transform_alpha=20)
|
|
|
|
|
x = coder.transform(y)
|
|
|
|
|
_, idx = np.where(x != 0)
|
|
|
|
|
x[0, idx], _, _, _ = np.linalg.lstsq(D[idx, :].T, y)
|
|
|
|
|
x = np.ravel(np.dot(x, D))
|
|
|
|
|
squared_error = np.sum((y - x) ** 2)
|
|
|
|
|
pl.plot(x,
|
|
|
|
|
label='Thresholding w/ debiasing:\n%d nonzero coefs, %.2f error' %
|
|
|
|
|
(len(idx), squared_error))
|
|
|
|
|
pl.axis('tight')
|
|
|
|
|
pl.legend()
|
|
|
|
|
pl.subplots_adjust(.04, .07, .97, .90, .09, .2)
|
2011-12-20 20:23:03 +08:00
|
|
|
pl.show()
|