scikit-learn/examples/feature_selection/plot_rfe_with_cross_validat...

Ignoring revisions in .git-blame-ignore-revs. Click here to bypass and see the normal blame view.

54 lines
1.5 KiB
Python
Raw Normal View History

2010-07-28 00:02:17 +08:00
"""
===================================================
2010-09-01 02:45:52 +08:00
Recursive feature elimination with cross-validation
===================================================
2010-07-28 00:02:17 +08:00
2011-09-04 22:15:43 +08:00
A recursive feature elimination example with automatic tuning of the
number of features selected with cross-validation.
"""
2010-09-01 02:45:52 +08:00
import matplotlib.pyplot as plt
from sklearn.svm import SVC
from sklearn.model_selection import StratifiedKFold
from sklearn.feature_selection import RFECV
2011-12-19 20:27:35 +08:00
from sklearn.datasets import make_classification
2010-07-28 00:02:17 +08:00
# Build a classification task using 3 informative features
X, y = make_classification(
n_samples=1000,
n_features=25,
n_informative=3,
2012-12-25 20:16:05 +08:00
n_redundant=2,
n_repeated=0,
n_classes=8,
n_clusters_per_class=1,
random_state=0,
)
2010-07-28 00:02:17 +08:00
# Create the RFE object and compute a cross-validated score.
svc = SVC(kernel="linear")
# The "accuracy" scoring shows the proportion of correct classifications
min_features_to_select = 1 # Minimum number of features to consider
rfecv = RFECV(
estimator=svc,
step=1,
cv=StratifiedKFold(2),
scoring="accuracy",
min_features_to_select=min_features_to_select,
)
rfecv.fit(X, y)
2010-07-28 00:02:17 +08:00
print("Optimal number of features : %d" % rfecv.n_features_)
2010-07-28 00:02:17 +08:00
# Plot number of features VS. cross-validation scores
plt.figure()
plt.xlabel("Number of features selected")
plt.ylabel("Cross validation score (accuracy)")
plt.plot(
range(min_features_to_select, len(rfecv.grid_scores_) + min_features_to_select),
rfecv.grid_scores_,
)
plt.show()