2015-10-15 19:21:19 +08:00
|
|
|
|
2013-07-26 18:07:06 +08:00
|
|
|
import numpy as np
|
|
|
|
|
from scipy import sparse
|
|
|
|
|
|
2013-11-06 16:07:07 +08:00
|
|
|
from sklearn.utils.testing import assert_equal
|
2013-07-26 18:07:06 +08:00
|
|
|
from sklearn.utils.testing import assert_array_equal
|
2017-09-18 17:55:23 +08:00
|
|
|
from sklearn.utils.testing import assert_array_almost_equal
|
2013-07-26 18:07:06 +08:00
|
|
|
from sklearn.utils.testing import assert_raises
|
2018-02-15 03:47:06 +08:00
|
|
|
from sklearn.utils.testing import ignore_warnings
|
2013-07-26 18:07:06 +08:00
|
|
|
|
2013-07-26 18:20:35 +08:00
|
|
|
from sklearn.preprocessing.imputation import Imputer
|
2013-07-26 18:07:06 +08:00
|
|
|
from sklearn.pipeline import Pipeline
|
Main Commits - Major
--------------------
* ENH Reogranize classes/fn from grid_search into search.py
* ENH Reogranize classes/fn from cross_validation into split.py
* ENH Reogranize cls/fn from cross_validation/learning_curve into validate.py
* MAINT Merge _check_cv into check_cv inside the model_selection module
* MAINT Update all the imports to point to the model_selection module
* FIX use iter_cv to iterate throught the new style/old style cv objs
* TST Add tests for the new model_selection members
* ENH Wrap the old-style cv obj/iterables instead of using iter_cv
* ENH Use scipy's binomial coefficient function comb for calucation of nCk
* ENH Few enhancements to the split module
* ENH Improve check_cv input validation and docstring
* MAINT _get_test_folds(X, y, labels) --> _get_test_folds(labels)
* TST if 1d arrays for X introduce any errors
* ENH use 1d X arrays for all tests;
* ENH X_10 --> X (global var)
Minor
-----
* ENH _PartitionIterator --> _BaseCrossValidator;
* ENH CVIterator --> CVIterableWrapper
* TST Import the old SKF locally
* FIX/TST Clean up the split module's tests.
* DOC Improve documentation of the cv parameter
* COSMIT consistently hyphenate cross-validation/cross-validator
* TST Calculate n_samples from X
* COSMIT Use separate lines for each import.
* COSMIT cross_validation_generator --> cross_validator
Commits merged manually
-----------------------
* FIX Document the random_state attribute in RandomSearchCV
* MAINT Use check_cv instead of _check_cv
* ENH refactor OVO decision function, use it in SVC for sklearn-like
decision_function shape
* FIX avoid memory cost when sampling from large parameter grids
ENH Major to Minor incremental enhancements to the model_selection
Squashed commit messages - (For reference)
Major
-----
* ENH p --> n_labels
* FIX *ShuffleSplit: all float/invalid type errors at init and int error at split
* FIX make PredefinedSplit accept test_folds in constructor; Cleanup docstrings
* ENH+TST KFold: make rng to be generated at every split call for reproducibility
* FIX/MAINT KFold: make shuffle a public attr
* FIX Make CVIterableWrapper private.
* FIX reuse len_cv instead of recalculating it
* FIX Prevent adding *SearchCV estimators from the old grid_search module
* re-FIX In all_estimators: the sorting to use only the 1st item (name)
To avoid collision between the old and the new GridSearch classes.
* FIX test_validate.py: Use 2D X (1D X is being detected as a single sample)
* MAINT validate.py --> validation.py
* MAINT make the submodules private
* MAINT Support old cv/gs/lc until 0.19
* FIX/MAINT n_splits --> get_n_splits
* FIX/TST test_logistic.py/test_ovr_multinomial_iris:
pass predefined folds as an iterable
* MAINT expose BaseCrossValidator
* Update the model_selection module with changes from master
- From #5161
- - MAINT remove redundant p variable
- - Add check for sparse prediction in cross_val_predict
- From #5201 - DOC improve random_state param doc
- From #5190 - LabelKFold and test
- From #4583 - LabelShuffleSplit and tests
- From #5300 - shuffle the `labels` not the `indxs` in LabelKFold + tests
- From #5378 - Make the GridSearchCV docs more accurate.
- From #5458 - Remove shuffle from LabelKFold
- From #5466(#4270) - Gaussian Process by Jan Metzen
- From #4826 - Move custom error / warnings into sklearn.exception
Minor
-----
* ENH Make the KFold shuffling test stronger
* FIX/DOC Use the higher level model_selection module as ref
* DOC in check_cv "y : array-like, optional"
* DOC a supervised learning problem --> supervised learning problems
* DOC cross-validators --> cross-validation strategies
* DOC Correct Olivier Grisel's name ;)
* MINOR/FIX cv_indices --> kfold
* FIX/DOC Align the 'See also' section of the new KFold, LeaveOneOut
* TST/FIX imports on separate lines
* FIX use __class__ instead of classmethod
* TST/FIX import directly from model_selection
* COSMIT Relocate the random_state documentation
* COSMIT remove pass
* MAINT Remove deprecation warnings from old tests
* FIX correct import at test_split
* FIX/MAINT Move P_sparse, X, y defns to top; rm unused W_sparse, X_sparse
* FIX random state to avoid doctest failure
* TST n_splits and split wrapping of _CVIterableWrapper
* FIX/MAINT Use multilabel indicator matrix directly
* TST/DOC clarify why we conflate classes 0 and 1
* DOC add comment that this was taken from BaseEstimator
* FIX use of labels is not needed in stratified k fold
* Fix cross_validation reference
* Fix the labels param doc
FIX/DOC/MAINT Addressing the review comments by Arnaud and Andy
COSMIT Sort the members alphabetically
COSMIT len_cv --> n_splits
COSMIT Merge 2 if; FIX Use kwargs
DOC Add my name to the authors :D
DOC make labels parameter consistent
FIX Remove hack for boolean indices; + COSMIT idx --> indices; DOC Add Returns
COSMIT preds --> predictions
DOC Add Returns and neatly arrange X, y, labels
FIX idx(s)/ind(s)--> indice(s)
COSMIT Merge if and else to elif
COSMIT n --> n_samples
COSMIT Use bincount only once
COSMIT cls --> class_i / class_i (ith class indices) -->
perm_indices_class_i
FIX/ENH/TST Addressing the final reviews
COSMIT c --> count
FIX/TST make check_cv raise ValueError for string cv value
TST nested cv (gs inside cross_val_score) works for diff cvs
FIX/ENH Raise ValueError when labels is None for label based cvs;
TST if labels is being passed correctly to the cv and that the
ValueError is being propagated to the cross_val_score/predict and grid
search
FIX pass labels to cross_val_score
FIX use make_classification
DOC Add Returns; COSMIT Remove scaffolding
TST add a test to check the _build_repr helper
REVERT the old GS/RS should also be tested by the common tests.
ENH Add a tuple of all/label based CVS
FIX raise VE even at get_n_splits if labels is None
FIX Fabian's comments
PEP8
2015-06-05 03:45:10 +08:00
|
|
|
from sklearn.model_selection import GridSearchCV
|
2013-07-26 18:07:06 +08:00
|
|
|
from sklearn import tree
|
|
|
|
|
from sklearn.random_projection import sparse_random_matrix
|
2016-10-25 21:15:58 +08:00
|
|
|
|
2013-07-26 18:07:06 +08:00
|
|
|
|
2018-02-15 03:47:06 +08:00
|
|
|
@ignore_warnings
|
2013-07-26 18:07:06 +08:00
|
|
|
def _check_statistics(X, X_true,
|
|
|
|
|
strategy, statistics, missing_values):
|
|
|
|
|
"""Utility function for testing imputation for a given strategy.
|
|
|
|
|
|
|
|
|
|
Test:
|
|
|
|
|
- along the two axes
|
|
|
|
|
- with dense and sparse arrays
|
|
|
|
|
|
|
|
|
|
Check that:
|
|
|
|
|
- the statistics (mean, median, mode) are correct
|
|
|
|
|
- the missing values are imputed correctly"""
|
|
|
|
|
|
|
|
|
|
err_msg = "Parameters: strategy = %s, missing_values = %s, " \
|
2013-07-27 21:32:24 +08:00
|
|
|
"axis = {0}, sparse = {1}" % (strategy, missing_values)
|
2013-07-26 18:07:06 +08:00
|
|
|
|
2017-09-18 17:55:23 +08:00
|
|
|
assert_ae = assert_array_equal
|
|
|
|
|
if X.dtype.kind == 'f' or X_true.dtype.kind == 'f':
|
|
|
|
|
assert_ae = assert_array_almost_equal
|
|
|
|
|
|
2013-07-26 18:07:06 +08:00
|
|
|
# Normal matrix, axis = 0
|
|
|
|
|
imputer = Imputer(missing_values, strategy=strategy, axis=0)
|
|
|
|
|
X_trans = imputer.fit(X).transform(X.copy())
|
2017-09-18 17:55:23 +08:00
|
|
|
assert_ae(imputer.statistics_, statistics,
|
|
|
|
|
err_msg=err_msg.format(0, False))
|
|
|
|
|
assert_ae(X_trans, X_true, err_msg=err_msg.format(0, False))
|
2013-07-26 18:07:06 +08:00
|
|
|
|
|
|
|
|
# Normal matrix, axis = 1
|
|
|
|
|
imputer = Imputer(missing_values, strategy=strategy, axis=1)
|
|
|
|
|
imputer.fit(X.transpose())
|
|
|
|
|
if np.isnan(statistics).any():
|
|
|
|
|
assert_raises(ValueError, imputer.transform, X.copy().transpose())
|
|
|
|
|
else:
|
|
|
|
|
X_trans = imputer.transform(X.copy().transpose())
|
2017-09-18 17:55:23 +08:00
|
|
|
assert_ae(X_trans, X_true.transpose(),
|
|
|
|
|
err_msg=err_msg.format(1, False))
|
2013-07-26 18:07:06 +08:00
|
|
|
|
|
|
|
|
# Sparse matrix, axis = 0
|
|
|
|
|
imputer = Imputer(missing_values, strategy=strategy, axis=0)
|
|
|
|
|
imputer.fit(sparse.csc_matrix(X))
|
|
|
|
|
X_trans = imputer.transform(sparse.csc_matrix(X.copy()))
|
|
|
|
|
|
|
|
|
|
if sparse.issparse(X_trans):
|
|
|
|
|
X_trans = X_trans.toarray()
|
|
|
|
|
|
2017-09-18 17:55:23 +08:00
|
|
|
assert_ae(imputer.statistics_, statistics,
|
|
|
|
|
err_msg=err_msg.format(0, True))
|
|
|
|
|
assert_ae(X_trans, X_true, err_msg=err_msg.format(0, True))
|
2013-07-26 18:07:06 +08:00
|
|
|
|
|
|
|
|
# Sparse matrix, axis = 1
|
|
|
|
|
imputer = Imputer(missing_values, strategy=strategy, axis=1)
|
|
|
|
|
imputer.fit(sparse.csc_matrix(X.transpose()))
|
|
|
|
|
if np.isnan(statistics).any():
|
|
|
|
|
assert_raises(ValueError, imputer.transform,
|
|
|
|
|
sparse.csc_matrix(X.copy().transpose()))
|
|
|
|
|
else:
|
|
|
|
|
X_trans = imputer.transform(sparse.csc_matrix(X.copy().transpose()))
|
|
|
|
|
|
|
|
|
|
if sparse.issparse(X_trans):
|
|
|
|
|
X_trans = X_trans.toarray()
|
|
|
|
|
|
2017-09-18 17:55:23 +08:00
|
|
|
assert_ae(X_trans, X_true.transpose(),
|
|
|
|
|
err_msg=err_msg.format(1, True))
|
2013-07-26 18:07:06 +08:00
|
|
|
|
|
|
|
|
|
2018-02-15 03:47:06 +08:00
|
|
|
@ignore_warnings
|
2013-11-06 12:54:29 +08:00
|
|
|
def test_imputation_shape():
|
2015-03-21 13:22:08 +08:00
|
|
|
# Verify the shapes of the imputed matrix for different strategies.
|
2013-11-06 12:54:29 +08:00
|
|
|
X = np.random.randn(10, 2)
|
|
|
|
|
X[::2] = np.nan
|
|
|
|
|
|
|
|
|
|
for strategy in ['mean', 'median', 'most_frequent']:
|
|
|
|
|
imputer = Imputer(strategy=strategy)
|
|
|
|
|
X_imputed = imputer.fit_transform(X)
|
2013-11-06 16:07:07 +08:00
|
|
|
assert_equal(X_imputed.shape, (10, 2))
|
2013-11-06 12:54:29 +08:00
|
|
|
X_imputed = imputer.fit_transform(sparse.csr_matrix(X))
|
2013-11-06 16:07:07 +08:00
|
|
|
assert_equal(X_imputed.shape, (10, 2))
|
2013-11-06 12:54:29 +08:00
|
|
|
|
|
|
|
|
|
2018-02-15 03:47:06 +08:00
|
|
|
@ignore_warnings
|
2013-07-26 18:07:06 +08:00
|
|
|
def test_imputation_mean_median_only_zero():
|
2015-03-21 13:22:08 +08:00
|
|
|
# Test imputation using the mean and median strategies, when
|
|
|
|
|
# missing_values == 0.
|
2013-07-26 18:07:06 +08:00
|
|
|
X = np.array([
|
2016-10-25 21:15:58 +08:00
|
|
|
[np.nan, 0, 0, 0, 5],
|
|
|
|
|
[np.nan, 1, 0, np.nan, 3],
|
|
|
|
|
[np.nan, 2, 0, 0, 0],
|
|
|
|
|
[np.nan, 6, 0, 5, 13],
|
2013-07-26 18:07:06 +08:00
|
|
|
])
|
|
|
|
|
|
|
|
|
|
X_imputed_mean = np.array([
|
2016-10-25 21:15:58 +08:00
|
|
|
[3, 5],
|
|
|
|
|
[1, 3],
|
|
|
|
|
[2, 7],
|
2013-07-26 18:07:06 +08:00
|
|
|
[6, 13],
|
|
|
|
|
])
|
|
|
|
|
statistics_mean = [np.nan, 3, np.nan, np.nan, 7]
|
|
|
|
|
|
2014-03-03 20:19:19 +08:00
|
|
|
# Behaviour of median with NaN is undefined, e.g. different results in
|
|
|
|
|
# np.median and np.ma.median
|
|
|
|
|
X_for_median = X[:, [0, 1, 2, 4]]
|
2013-07-26 18:07:06 +08:00
|
|
|
X_imputed_median = np.array([
|
2014-03-03 20:19:19 +08:00
|
|
|
[2, 5],
|
|
|
|
|
[1, 3],
|
|
|
|
|
[2, 5],
|
|
|
|
|
[6, 13],
|
2013-07-26 18:07:06 +08:00
|
|
|
])
|
2014-03-03 20:19:19 +08:00
|
|
|
statistics_median = [np.nan, 2, np.nan, 5]
|
2013-07-26 18:07:06 +08:00
|
|
|
|
|
|
|
|
_check_statistics(X, X_imputed_mean, "mean", statistics_mean, 0)
|
2014-03-03 20:19:19 +08:00
|
|
|
_check_statistics(X_for_median, X_imputed_median, "median",
|
|
|
|
|
statistics_median, 0)
|
2013-07-26 18:07:06 +08:00
|
|
|
|
|
|
|
|
|
2015-10-15 19:21:19 +08:00
|
|
|
def safe_median(arr, *args, **kwargs):
|
|
|
|
|
# np.median([]) raises a TypeError for numpy >= 1.10.1
|
|
|
|
|
length = arr.size if hasattr(arr, 'size') else len(arr)
|
|
|
|
|
return np.nan if length == 0 else np.median(arr, *args, **kwargs)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def safe_mean(arr, *args, **kwargs):
|
|
|
|
|
# np.mean([]) raises a RuntimeWarning for numpy >= 1.10.1
|
|
|
|
|
length = arr.size if hasattr(arr, 'size') else len(arr)
|
|
|
|
|
return np.nan if length == 0 else np.mean(arr, *args, **kwargs)
|
|
|
|
|
|
|
|
|
|
|
2018-02-15 03:47:06 +08:00
|
|
|
@ignore_warnings
|
2013-07-26 18:07:06 +08:00
|
|
|
def test_imputation_mean_median():
|
2015-03-21 13:22:08 +08:00
|
|
|
# Test imputation using the mean and median strategies, when
|
|
|
|
|
# missing_values != 0.
|
2013-07-26 18:07:06 +08:00
|
|
|
rng = np.random.RandomState(0)
|
|
|
|
|
|
|
|
|
|
dim = 10
|
|
|
|
|
dec = 10
|
|
|
|
|
shape = (dim * dim, dim + dec)
|
|
|
|
|
|
|
|
|
|
zeros = np.zeros(shape[0])
|
2016-10-25 21:15:58 +08:00
|
|
|
values = np.arange(1, shape[0] + 1)
|
2013-07-26 18:07:06 +08:00
|
|
|
values[4::2] = - values[4::2]
|
|
|
|
|
|
2015-10-15 19:21:19 +08:00
|
|
|
tests = [("mean", "NaN", lambda z, v, p: safe_mean(np.hstack((z, v)))),
|
2013-07-26 18:07:06 +08:00
|
|
|
("mean", 0, lambda z, v, p: np.mean(v)),
|
2015-10-15 19:21:19 +08:00
|
|
|
("median", "NaN", lambda z, v, p: safe_median(np.hstack((z, v)))),
|
2013-07-26 18:07:06 +08:00
|
|
|
("median", 0, lambda z, v, p: np.median(v))]
|
|
|
|
|
|
|
|
|
|
for strategy, test_missing_values, true_value_fun in tests:
|
|
|
|
|
X = np.empty(shape)
|
|
|
|
|
X_true = np.empty(shape)
|
|
|
|
|
true_statistics = np.empty(shape[1])
|
|
|
|
|
|
|
|
|
|
# Create a matrix X with columns
|
|
|
|
|
# - with only zeros,
|
|
|
|
|
# - with only missing values
|
|
|
|
|
# - with zeros, missing values and values
|
|
|
|
|
# And a matrix X_true containing all true values
|
|
|
|
|
for j in range(shape[1]):
|
|
|
|
|
nb_zeros = (j - dec + 1 > 0) * (j - dec + 1) * (j - dec + 1)
|
|
|
|
|
nb_missing_values = max(shape[0] + dec * dec
|
|
|
|
|
- (j + dec) * (j + dec), 0)
|
|
|
|
|
nb_values = shape[0] - nb_zeros - nb_missing_values
|
|
|
|
|
|
|
|
|
|
z = zeros[:nb_zeros]
|
|
|
|
|
p = np.repeat(test_missing_values, nb_missing_values)
|
|
|
|
|
v = values[rng.permutation(len(values))[:nb_values]]
|
|
|
|
|
|
|
|
|
|
true_statistics[j] = true_value_fun(z, v, p)
|
|
|
|
|
|
|
|
|
|
# Create the columns
|
|
|
|
|
X[:, j] = np.hstack((v, z, p))
|
|
|
|
|
|
|
|
|
|
if 0 == test_missing_values:
|
|
|
|
|
X_true[:, j] = np.hstack((v,
|
|
|
|
|
np.repeat(
|
|
|
|
|
true_statistics[j],
|
|
|
|
|
nb_missing_values + nb_zeros)))
|
|
|
|
|
else:
|
|
|
|
|
X_true[:, j] = np.hstack((v,
|
|
|
|
|
z,
|
|
|
|
|
np.repeat(true_statistics[j],
|
|
|
|
|
nb_missing_values)))
|
|
|
|
|
|
|
|
|
|
# Shuffle them the same way
|
|
|
|
|
np.random.RandomState(j).shuffle(X[:, j])
|
|
|
|
|
np.random.RandomState(j).shuffle(X_true[:, j])
|
|
|
|
|
|
|
|
|
|
# Mean doesn't support columns containing NaNs, median does
|
|
|
|
|
if strategy == "median":
|
|
|
|
|
cols_to_keep = ~np.isnan(X_true).any(axis=0)
|
|
|
|
|
else:
|
|
|
|
|
cols_to_keep = ~np.isnan(X_true).all(axis=0)
|
|
|
|
|
|
|
|
|
|
X_true = X_true[:, cols_to_keep]
|
|
|
|
|
|
|
|
|
|
_check_statistics(X, X_true, strategy,
|
|
|
|
|
true_statistics, test_missing_values)
|
|
|
|
|
|
|
|
|
|
|
2018-02-15 03:47:06 +08:00
|
|
|
@ignore_warnings
|
2014-03-03 20:19:19 +08:00
|
|
|
def test_imputation_median_special_cases():
|
2015-03-21 13:22:08 +08:00
|
|
|
# Test median imputation with sparse boundary cases
|
2014-03-03 20:19:19 +08:00
|
|
|
X = np.array([
|
|
|
|
|
[0, np.nan, np.nan], # odd: implicit zero
|
|
|
|
|
[5, np.nan, np.nan], # odd: explicit nonzero
|
|
|
|
|
[0, 0, np.nan], # even: average two zeros
|
|
|
|
|
[-5, 0, np.nan], # even: avg zero and neg
|
|
|
|
|
[0, 5, np.nan], # even: avg zero and pos
|
|
|
|
|
[4, 5, np.nan], # even: avg nonzeros
|
|
|
|
|
[-4, -5, np.nan], # even: avg negatives
|
|
|
|
|
[-1, 2, np.nan], # even: crossing neg and pos
|
|
|
|
|
]).transpose()
|
|
|
|
|
|
|
|
|
|
X_imputed_median = np.array([
|
|
|
|
|
[0, 0, 0],
|
|
|
|
|
[5, 5, 5],
|
|
|
|
|
[0, 0, 0],
|
|
|
|
|
[-5, 0, -2.5],
|
|
|
|
|
[0, 5, 2.5],
|
|
|
|
|
[4, 5, 4.5],
|
|
|
|
|
[-4, -5, -4.5],
|
|
|
|
|
[-1, 2, .5],
|
|
|
|
|
]).transpose()
|
|
|
|
|
statistics_median = [0, 5, 0, -2.5, 2.5, 4.5, -4.5, .5]
|
|
|
|
|
|
|
|
|
|
_check_statistics(X, X_imputed_median, "median",
|
|
|
|
|
statistics_median, 'NaN')
|
|
|
|
|
|
|
|
|
|
|
2018-02-15 03:47:06 +08:00
|
|
|
@ignore_warnings
|
2013-07-26 18:07:06 +08:00
|
|
|
def test_imputation_most_frequent():
|
2015-03-21 13:22:08 +08:00
|
|
|
# Test imputation using the most-frequent strategy.
|
2013-07-26 18:07:06 +08:00
|
|
|
X = np.array([
|
2016-10-25 21:15:58 +08:00
|
|
|
[-1, -1, 0, 5],
|
|
|
|
|
[-1, 2, -1, 3],
|
|
|
|
|
[-1, 1, 3, -1],
|
|
|
|
|
[-1, 2, 3, 7],
|
2013-07-26 18:07:06 +08:00
|
|
|
])
|
|
|
|
|
|
|
|
|
|
X_true = np.array([
|
2016-10-25 21:15:58 +08:00
|
|
|
[2, 0, 5],
|
|
|
|
|
[2, 3, 3],
|
|
|
|
|
[1, 3, 3],
|
|
|
|
|
[2, 3, 7],
|
2013-07-26 18:07:06 +08:00
|
|
|
])
|
|
|
|
|
|
|
|
|
|
# scipy.stats.mode, used in Imputer, doesn't return the first most
|
|
|
|
|
# frequent as promised in the doc but the lowest most frequent. When this
|
|
|
|
|
# test will fail after an update of scipy, Imputer will need to be updated
|
|
|
|
|
# to be consistent with the new (correct) behaviour
|
|
|
|
|
_check_statistics(X, X_true, "most_frequent", [np.nan, 2, 3, 3], -1)
|
|
|
|
|
|
|
|
|
|
|
2018-02-15 03:47:06 +08:00
|
|
|
@ignore_warnings
|
2013-07-26 18:07:06 +08:00
|
|
|
def test_imputation_pipeline_grid_search():
|
2015-03-21 13:22:08 +08:00
|
|
|
# Test imputation within a pipeline + gridsearch.
|
2013-07-26 18:07:06 +08:00
|
|
|
pipeline = Pipeline([('imputer', Imputer(missing_values=0)),
|
|
|
|
|
('tree', tree.DecisionTreeRegressor(random_state=0))])
|
|
|
|
|
|
|
|
|
|
parameters = {
|
|
|
|
|
'imputer__strategy': ["mean", "median", "most_frequent"],
|
|
|
|
|
'imputer__axis': [0, 1]
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
l = 100
|
|
|
|
|
X = sparse_random_matrix(l, l, density=0.10)
|
2014-05-21 09:44:34 +08:00
|
|
|
Y = sparse_random_matrix(l, 1, density=0.10).toarray()
|
Main Commits - Major
--------------------
* ENH Reogranize classes/fn from grid_search into search.py
* ENH Reogranize classes/fn from cross_validation into split.py
* ENH Reogranize cls/fn from cross_validation/learning_curve into validate.py
* MAINT Merge _check_cv into check_cv inside the model_selection module
* MAINT Update all the imports to point to the model_selection module
* FIX use iter_cv to iterate throught the new style/old style cv objs
* TST Add tests for the new model_selection members
* ENH Wrap the old-style cv obj/iterables instead of using iter_cv
* ENH Use scipy's binomial coefficient function comb for calucation of nCk
* ENH Few enhancements to the split module
* ENH Improve check_cv input validation and docstring
* MAINT _get_test_folds(X, y, labels) --> _get_test_folds(labels)
* TST if 1d arrays for X introduce any errors
* ENH use 1d X arrays for all tests;
* ENH X_10 --> X (global var)
Minor
-----
* ENH _PartitionIterator --> _BaseCrossValidator;
* ENH CVIterator --> CVIterableWrapper
* TST Import the old SKF locally
* FIX/TST Clean up the split module's tests.
* DOC Improve documentation of the cv parameter
* COSMIT consistently hyphenate cross-validation/cross-validator
* TST Calculate n_samples from X
* COSMIT Use separate lines for each import.
* COSMIT cross_validation_generator --> cross_validator
Commits merged manually
-----------------------
* FIX Document the random_state attribute in RandomSearchCV
* MAINT Use check_cv instead of _check_cv
* ENH refactor OVO decision function, use it in SVC for sklearn-like
decision_function shape
* FIX avoid memory cost when sampling from large parameter grids
ENH Major to Minor incremental enhancements to the model_selection
Squashed commit messages - (For reference)
Major
-----
* ENH p --> n_labels
* FIX *ShuffleSplit: all float/invalid type errors at init and int error at split
* FIX make PredefinedSplit accept test_folds in constructor; Cleanup docstrings
* ENH+TST KFold: make rng to be generated at every split call for reproducibility
* FIX/MAINT KFold: make shuffle a public attr
* FIX Make CVIterableWrapper private.
* FIX reuse len_cv instead of recalculating it
* FIX Prevent adding *SearchCV estimators from the old grid_search module
* re-FIX In all_estimators: the sorting to use only the 1st item (name)
To avoid collision between the old and the new GridSearch classes.
* FIX test_validate.py: Use 2D X (1D X is being detected as a single sample)
* MAINT validate.py --> validation.py
* MAINT make the submodules private
* MAINT Support old cv/gs/lc until 0.19
* FIX/MAINT n_splits --> get_n_splits
* FIX/TST test_logistic.py/test_ovr_multinomial_iris:
pass predefined folds as an iterable
* MAINT expose BaseCrossValidator
* Update the model_selection module with changes from master
- From #5161
- - MAINT remove redundant p variable
- - Add check for sparse prediction in cross_val_predict
- From #5201 - DOC improve random_state param doc
- From #5190 - LabelKFold and test
- From #4583 - LabelShuffleSplit and tests
- From #5300 - shuffle the `labels` not the `indxs` in LabelKFold + tests
- From #5378 - Make the GridSearchCV docs more accurate.
- From #5458 - Remove shuffle from LabelKFold
- From #5466(#4270) - Gaussian Process by Jan Metzen
- From #4826 - Move custom error / warnings into sklearn.exception
Minor
-----
* ENH Make the KFold shuffling test stronger
* FIX/DOC Use the higher level model_selection module as ref
* DOC in check_cv "y : array-like, optional"
* DOC a supervised learning problem --> supervised learning problems
* DOC cross-validators --> cross-validation strategies
* DOC Correct Olivier Grisel's name ;)
* MINOR/FIX cv_indices --> kfold
* FIX/DOC Align the 'See also' section of the new KFold, LeaveOneOut
* TST/FIX imports on separate lines
* FIX use __class__ instead of classmethod
* TST/FIX import directly from model_selection
* COSMIT Relocate the random_state documentation
* COSMIT remove pass
* MAINT Remove deprecation warnings from old tests
* FIX correct import at test_split
* FIX/MAINT Move P_sparse, X, y defns to top; rm unused W_sparse, X_sparse
* FIX random state to avoid doctest failure
* TST n_splits and split wrapping of _CVIterableWrapper
* FIX/MAINT Use multilabel indicator matrix directly
* TST/DOC clarify why we conflate classes 0 and 1
* DOC add comment that this was taken from BaseEstimator
* FIX use of labels is not needed in stratified k fold
* Fix cross_validation reference
* Fix the labels param doc
FIX/DOC/MAINT Addressing the review comments by Arnaud and Andy
COSMIT Sort the members alphabetically
COSMIT len_cv --> n_splits
COSMIT Merge 2 if; FIX Use kwargs
DOC Add my name to the authors :D
DOC make labels parameter consistent
FIX Remove hack for boolean indices; + COSMIT idx --> indices; DOC Add Returns
COSMIT preds --> predictions
DOC Add Returns and neatly arrange X, y, labels
FIX idx(s)/ind(s)--> indice(s)
COSMIT Merge if and else to elif
COSMIT n --> n_samples
COSMIT Use bincount only once
COSMIT cls --> class_i / class_i (ith class indices) -->
perm_indices_class_i
FIX/ENH/TST Addressing the final reviews
COSMIT c --> count
FIX/TST make check_cv raise ValueError for string cv value
TST nested cv (gs inside cross_val_score) works for diff cvs
FIX/ENH Raise ValueError when labels is None for label based cvs;
TST if labels is being passed correctly to the cv and that the
ValueError is being propagated to the cross_val_score/predict and grid
search
FIX pass labels to cross_val_score
FIX use make_classification
DOC Add Returns; COSMIT Remove scaffolding
TST add a test to check the _build_repr helper
REVERT the old GS/RS should also be tested by the common tests.
ENH Add a tuple of all/label based CVS
FIX raise VE even at get_n_splits if labels is None
FIX Fabian's comments
PEP8
2015-06-05 03:45:10 +08:00
|
|
|
gs = GridSearchCV(pipeline, parameters)
|
2013-07-26 18:07:06 +08:00
|
|
|
gs.fit(X, Y)
|
|
|
|
|
|
|
|
|
|
|
2018-02-15 03:47:06 +08:00
|
|
|
@ignore_warnings
|
2013-07-26 18:07:06 +08:00
|
|
|
def test_imputation_pickle():
|
2015-03-21 13:22:08 +08:00
|
|
|
# Test for pickling imputers.
|
2013-07-26 18:07:06 +08:00
|
|
|
import pickle
|
|
|
|
|
|
|
|
|
|
l = 100
|
|
|
|
|
X = sparse_random_matrix(l, l, density=0.10)
|
|
|
|
|
|
|
|
|
|
for strategy in ["mean", "median", "most_frequent"]:
|
|
|
|
|
imputer = Imputer(missing_values=0, strategy=strategy)
|
|
|
|
|
imputer.fit(X)
|
|
|
|
|
|
|
|
|
|
imputer_pickled = pickle.loads(pickle.dumps(imputer))
|
|
|
|
|
|
2017-09-18 17:55:23 +08:00
|
|
|
assert_array_almost_equal(
|
|
|
|
|
imputer.transform(X.copy()),
|
|
|
|
|
imputer_pickled.transform(X.copy()),
|
|
|
|
|
err_msg="Fail to transform the data after pickling "
|
|
|
|
|
"(strategy = %s)" % (strategy)
|
|
|
|
|
)
|
2013-07-26 18:07:06 +08:00
|
|
|
|
|
|
|
|
|
2018-02-15 03:47:06 +08:00
|
|
|
@ignore_warnings
|
2013-07-26 18:07:06 +08:00
|
|
|
def test_imputation_copy():
|
2015-03-21 13:22:08 +08:00
|
|
|
# Test imputation with copy
|
2013-12-31 23:10:02 +08:00
|
|
|
X_orig = sparse_random_matrix(5, 5, density=0.75, random_state=0)
|
|
|
|
|
|
2014-01-02 00:48:02 +08:00
|
|
|
# copy=True, dense => copy
|
2014-05-21 09:44:34 +08:00
|
|
|
X = X_orig.copy().toarray()
|
2013-12-31 23:10:02 +08:00
|
|
|
imputer = Imputer(missing_values=0, strategy="mean", copy=True)
|
|
|
|
|
Xt = imputer.fit(X).transform(X)
|
|
|
|
|
Xt[0, 0] = -1
|
2018-11-28 09:16:26 +08:00
|
|
|
assert not np.all(X == Xt)
|
2013-12-31 23:10:02 +08:00
|
|
|
|
2014-01-02 00:48:02 +08:00
|
|
|
# copy=True, sparse csr => copy
|
2013-12-31 23:10:02 +08:00
|
|
|
X = X_orig.copy()
|
2014-01-02 00:48:02 +08:00
|
|
|
imputer = Imputer(missing_values=X.data[0], strategy="mean", copy=True)
|
2013-12-31 23:10:02 +08:00
|
|
|
Xt = imputer.fit(X).transform(X)
|
2014-01-02 00:48:02 +08:00
|
|
|
Xt.data[0] = -1
|
2018-11-28 09:16:26 +08:00
|
|
|
assert not np.all(X.data == Xt.data)
|
2013-12-31 23:10:02 +08:00
|
|
|
|
2014-01-02 00:48:02 +08:00
|
|
|
# copy=False, dense => no copy
|
2014-05-21 09:44:34 +08:00
|
|
|
X = X_orig.copy().toarray()
|
2013-12-31 23:10:02 +08:00
|
|
|
imputer = Imputer(missing_values=0, strategy="mean", copy=False)
|
|
|
|
|
Xt = imputer.fit(X).transform(X)
|
|
|
|
|
Xt[0, 0] = -1
|
2017-09-18 17:55:23 +08:00
|
|
|
assert_array_almost_equal(X, Xt)
|
2013-12-31 23:10:02 +08:00
|
|
|
|
2014-01-02 00:48:02 +08:00
|
|
|
# copy=False, sparse csr, axis=1 => no copy
|
|
|
|
|
X = X_orig.copy()
|
2014-01-16 21:29:48 +08:00
|
|
|
imputer = Imputer(missing_values=X.data[0], strategy="mean",
|
|
|
|
|
copy=False, axis=1)
|
2014-01-02 00:48:02 +08:00
|
|
|
Xt = imputer.fit(X).transform(X)
|
|
|
|
|
Xt.data[0] = -1
|
2017-09-18 17:55:23 +08:00
|
|
|
assert_array_almost_equal(X.data, Xt.data)
|
2014-01-02 00:48:02 +08:00
|
|
|
|
|
|
|
|
# copy=False, sparse csc, axis=0 => no copy
|
|
|
|
|
X = X_orig.copy().tocsc()
|
2014-01-16 21:29:48 +08:00
|
|
|
imputer = Imputer(missing_values=X.data[0], strategy="mean",
|
|
|
|
|
copy=False, axis=0)
|
2014-01-02 00:48:02 +08:00
|
|
|
Xt = imputer.fit(X).transform(X)
|
|
|
|
|
Xt.data[0] = -1
|
2017-09-18 17:55:23 +08:00
|
|
|
assert_array_almost_equal(X.data, Xt.data)
|
2014-01-02 00:48:02 +08:00
|
|
|
|
|
|
|
|
# copy=False, sparse csr, axis=0 => copy
|
2013-12-31 23:10:02 +08:00
|
|
|
X = X_orig.copy()
|
2014-01-16 21:29:48 +08:00
|
|
|
imputer = Imputer(missing_values=X.data[0], strategy="mean",
|
|
|
|
|
copy=False, axis=0)
|
2014-01-02 00:48:02 +08:00
|
|
|
Xt = imputer.fit(X).transform(X)
|
|
|
|
|
Xt.data[0] = -1
|
2018-11-28 09:16:26 +08:00
|
|
|
assert not np.all(X.data == Xt.data)
|
2014-01-02 00:48:02 +08:00
|
|
|
|
|
|
|
|
# copy=False, sparse csc, axis=1 => copy
|
|
|
|
|
X = X_orig.copy().tocsc()
|
2014-01-16 21:29:48 +08:00
|
|
|
imputer = Imputer(missing_values=X.data[0], strategy="mean",
|
|
|
|
|
copy=False, axis=1)
|
2013-12-31 23:10:02 +08:00
|
|
|
Xt = imputer.fit(X).transform(X)
|
2014-01-02 00:48:02 +08:00
|
|
|
Xt.data[0] = -1
|
2018-11-28 09:16:26 +08:00
|
|
|
assert not np.all(X.data == Xt.data)
|
2014-01-02 00:48:02 +08:00
|
|
|
|
|
|
|
|
# copy=False, sparse csr, axis=1, missing_values=0 => copy
|
|
|
|
|
X = X_orig.copy()
|
2014-01-16 21:29:48 +08:00
|
|
|
imputer = Imputer(missing_values=0, strategy="mean",
|
|
|
|
|
copy=False, axis=1)
|
2014-01-02 00:48:02 +08:00
|
|
|
Xt = imputer.fit(X).transform(X)
|
2018-11-28 09:16:26 +08:00
|
|
|
assert not sparse.issparse(Xt)
|
2013-12-31 23:10:02 +08:00
|
|
|
|
|
|
|
|
# Note: If X is sparse and if missing_values=0, then a (dense) copy of X is
|
|
|
|
|
# made, even if copy=False.
|