scikit-learn/examples/cluster/plot_mini_batch_kmeans.py

117 lines
3.9 KiB
Python
Raw Normal View History

"""
2013-01-08 17:45:13 +08:00
====================================================================
2013-03-26 03:36:08 +08:00
Comparison of the K-Means and MiniBatchKMeans clustering algorithms
2013-01-08 17:45:13 +08:00
====================================================================
2011-05-18 04:57:11 +08:00
We want to compare the performance of the MiniBatchKMeans and KMeans:
the MiniBatchKMeans is faster, but gives slightly different results (see
2011-05-18 04:57:11 +08:00
:ref:`mini_batch_kmeans`).
We will cluster a set of data, first with KMeans and then with
MiniBatchKMeans, and plot the results.
We will also plot the points that are labelled differently between the two
algorithms.
"""
print(__doc__)
2011-05-18 03:50:16 +08:00
import time
import numpy as np
import matplotlib.pyplot as plt
2011-05-18 03:50:16 +08:00
from sklearn.cluster import MiniBatchKMeans, KMeans
2013-08-30 21:31:09 +08:00
from sklearn.metrics.pairwise import pairwise_distances_argmin
from sklearn.datasets import make_blobs
# #############################################################################
# Generate sample data
np.random.seed(0)
batch_size = 45
centers = [[1, 1], [-1, -1], [1, -1]]
n_clusters = len(centers)
2011-12-21 03:18:08 +08:00
X, labels_true = make_blobs(n_samples=3000, centers=centers, cluster_std=0.7)
# #############################################################################
2011-05-10 06:00:11 +08:00
# Compute clustering with Means
k_means = KMeans(init='k-means++', n_clusters=3, n_init=10)
2011-05-18 03:50:16 +08:00
t0 = time.time()
2011-05-10 06:00:11 +08:00
k_means.fit(X)
2011-05-18 03:50:16 +08:00
t_batch = time.time() - t0
2011-05-10 06:00:11 +08:00
# #############################################################################
# Compute clustering with MiniBatchKMeans
mbk = MiniBatchKMeans(init='k-means++', n_clusters=3, batch_size=batch_size,
2011-12-19 23:28:41 +08:00
n_init=10, max_no_improvement=10, verbose=0)
2011-05-18 03:50:16 +08:00
t0 = time.time()
mbk.fit(X)
2011-05-18 03:50:16 +08:00
t_mini_batch = time.time() - t0
# #############################################################################
# Plot result
fig = plt.figure(figsize=(8, 3))
2011-12-20 22:34:17 +08:00
fig.subplots_adjust(left=0.02, right=0.98, bottom=0.05, top=0.9)
colors = ['#4EACC5', '#FF9C34', '#4E9A06']
# We want to have the same colors for the same cluster from the
# MiniBatchKMeans and the KMeans algorithm. Let's pair the cluster centers per
# closest one.
k_means_cluster_centers = k_means.cluster_centers_
order = pairwise_distances_argmin(k_means.cluster_centers_,
mbk.cluster_centers_)
mbk_means_cluster_centers = mbk.cluster_centers_[order]
k_means_labels = pairwise_distances_argmin(X, k_means_cluster_centers)
mbk_means_labels = pairwise_distances_argmin(X, mbk_means_cluster_centers)
2011-05-10 06:00:11 +08:00
# KMeans
ax = fig.add_subplot(1, 3, 1)
for k, col in zip(range(n_clusters), colors):
my_members = k_means_labels == k
cluster_center = k_means_cluster_centers[k]
ax.plot(X[my_members, 0], X[my_members, 1], 'w',
markerfacecolor=col, marker='.')
2011-05-10 06:00:11 +08:00
ax.plot(cluster_center[0], cluster_center[1], 'o', markerfacecolor=col,
2012-12-25 20:16:05 +08:00
markeredgecolor='k', markersize=6)
ax.set_title('KMeans')
ax.set_xticks(())
ax.set_yticks(())
plt.text(-3.5, 1.8, 'train time: %.2fs\ninertia: %f' % (
2011-12-21 03:18:08 +08:00
t_batch, k_means.inertia_))
2011-05-10 06:00:11 +08:00
# MiniBatchKMeans
ax = fig.add_subplot(1, 3, 2)
for k, col in zip(range(n_clusters), colors):
my_members = mbk_means_labels == k
cluster_center = mbk_means_cluster_centers[k]
ax.plot(X[my_members, 0], X[my_members, 1], 'w',
markerfacecolor=col, marker='.')
2011-05-10 06:00:11 +08:00
ax.plot(cluster_center[0], cluster_center[1], 'o', markerfacecolor=col,
2012-12-25 20:16:05 +08:00
markeredgecolor='k', markersize=6)
ax.set_title('MiniBatchKMeans')
ax.set_xticks(())
ax.set_yticks(())
plt.text(-3.5, 1.8, 'train time: %.2fs\ninertia: %f' %
(t_mini_batch, mbk.inertia_))
2011-05-10 06:00:11 +08:00
# Initialise the different array to all False
different = (mbk_means_labels == 4)
2011-05-10 06:00:11 +08:00
ax = fig.add_subplot(1, 3, 3)
2015-12-16 14:19:31 +08:00
for k in range(n_clusters):
different += ((k_means_labels == k) != (mbk_means_labels == k))
identic = np.logical_not(different)
ax.plot(X[identic, 0], X[identic, 1], 'w',
markerfacecolor='#bbbbbb', marker='.')
ax.plot(X[different, 0], X[different, 1], 'w',
markerfacecolor='m', marker='.')
ax.set_title('Difference')
ax.set_xticks(())
ax.set_yticks(())
2011-05-10 06:00:11 +08:00
plt.show()