183 lines
7.1 KiB
Cython
183 lines
7.1 KiB
Cython
# cython: cdivision=True
|
|
# cython: boundscheck=False
|
|
# cython: wraparound=False
|
|
#
|
|
# Author: Peter Prettenhofer <peter.prettenhofer@gmail.com>
|
|
#
|
|
# Licence: BSD 3 clause
|
|
|
|
cimport cython
|
|
from libc.limits cimport INT_MAX
|
|
cimport numpy as np
|
|
import numpy as np
|
|
|
|
np.import_array()
|
|
|
|
|
|
cdef class SequentialDataset:
|
|
"""Base class for datasets with sequential data access. """
|
|
|
|
cdef void next(self, double **x_data_ptr, int **x_ind_ptr,
|
|
int *nnz, double *y, double *sample_weight) nogil:
|
|
"""Get the next example ``x`` from the dataset.
|
|
|
|
Parameters
|
|
----------
|
|
x_data_ptr : double**
|
|
A pointer to the double array which holds the feature
|
|
values of the next example.
|
|
x_ind_ptr : np.intc**
|
|
A pointer to the int array which holds the feature
|
|
indices of the next example.
|
|
nnz : int*
|
|
A pointer to an int holding the number of non-zero
|
|
values of the next example.
|
|
y : double*
|
|
The target value of the next example.
|
|
sample_weight : double*
|
|
The weight of the next example.
|
|
"""
|
|
with gil:
|
|
raise NotImplementedError()
|
|
|
|
cdef void shuffle(self, seed):
|
|
"""Permutes the ordering of examples. """
|
|
raise NotImplementedError()
|
|
|
|
|
|
cdef class ArrayDataset(SequentialDataset):
|
|
"""Dataset backed by a two-dimensional numpy array.
|
|
|
|
The dtype of the numpy array is expected to be ``np.float64`` (double)
|
|
and C-style memory layout.
|
|
"""
|
|
|
|
def __cinit__(self, np.ndarray[double, ndim=2, mode='c'] X,
|
|
np.ndarray[double, ndim=1, mode='c'] Y,
|
|
np.ndarray[double, ndim=1, mode='c'] sample_weights):
|
|
"""A ``SequentialDataset`` backed by a two-dimensional numpy array.
|
|
|
|
Parameters
|
|
----------
|
|
X : ndarray, dtype=double, ndim=2, mode='c'
|
|
The samples; a two-dimensional c-continuous numpy array of
|
|
dtype double.
|
|
Y : ndarray, dtype=double, ndim=1, mode='c'
|
|
The target values; a one-dimensional c-continuous numpy array of
|
|
dtype double.
|
|
sample_weights : ndarray, dtype=double, ndim=1, mode='c'
|
|
The weight of each sample; a one-dimensional c-continuous numpy
|
|
array of dtype double.
|
|
"""
|
|
if X.shape[0] > INT_MAX or X.shape[1] > INT_MAX:
|
|
raise ValueError("More than %d samples or features not supported;"
|
|
" got (%d, %d)."
|
|
% (INT_MAX, X.shape[0], X.shape[1]))
|
|
|
|
self.n_samples = X.shape[0]
|
|
self.n_features = X.shape[1]
|
|
cdef np.ndarray[int, ndim=1,
|
|
mode='c'] feature_indices = np.arange(0, self.n_features,
|
|
dtype=np.intc)
|
|
self.feature_indices = feature_indices
|
|
self.feature_indices_ptr = <int *> feature_indices.data
|
|
self.current_index = -1
|
|
self.stride = X.strides[0] / X.itemsize
|
|
self.X_data_ptr = <double *>X.data
|
|
self.Y_data_ptr = <double *>Y.data
|
|
self.sample_weight_data = <double *>sample_weights.data
|
|
|
|
# Use index array for fast shuffling
|
|
cdef np.ndarray[int, ndim=1,
|
|
mode='c'] index = np.arange(0, self.n_samples,
|
|
dtype=np.intc)
|
|
self.index = index
|
|
self.index_data_ptr = <int *> index.data
|
|
|
|
cdef void next(self, double **x_data_ptr, int **x_ind_ptr,
|
|
int *nnz, double *y, double *sample_weight) nogil:
|
|
cdef int current_index = self.current_index
|
|
if current_index >= (self.n_samples - 1):
|
|
current_index = -1
|
|
|
|
current_index += 1
|
|
cdef int sample_idx = self.index_data_ptr[current_index]
|
|
cdef int offset = sample_idx * self.stride
|
|
|
|
y[0] = self.Y_data_ptr[sample_idx]
|
|
x_data_ptr[0] = self.X_data_ptr + offset
|
|
x_ind_ptr[0] = self.feature_indices_ptr
|
|
nnz[0] = self.n_features
|
|
sample_weight[0] = self.sample_weight_data[sample_idx]
|
|
|
|
self.current_index = current_index
|
|
|
|
cdef void shuffle(self, seed):
|
|
np.random.RandomState(seed).shuffle(self.index)
|
|
|
|
|
|
cdef class CSRDataset(SequentialDataset):
|
|
"""A ``SequentialDataset`` backed by a scipy sparse CSR matrix. """
|
|
|
|
def __cinit__(self, np.ndarray[double, ndim=1, mode='c'] X_data,
|
|
np.ndarray[int, ndim=1, mode='c'] X_indptr,
|
|
np.ndarray[int, ndim=1, mode='c'] X_indices,
|
|
np.ndarray[double, ndim=1, mode='c'] Y,
|
|
np.ndarray[double, ndim=1, mode='c'] sample_weight):
|
|
"""Dataset backed by a scipy sparse CSR matrix.
|
|
|
|
The feature indices of ``x`` are given by x_ind_ptr[0:nnz].
|
|
The corresponding feature values are given by
|
|
x_data_ptr[0:nnz].
|
|
|
|
Parameters
|
|
----------
|
|
X_data : ndarray, dtype=double, ndim=1, mode='c'
|
|
The data array of the CSR matrix; a one-dimensional c-continuous
|
|
numpy array of dtype double.
|
|
X_indptr : ndarray, dtype=np.intc, ndim=1, mode='c'
|
|
The index pointer array of the CSR matrix; a one-dimensional
|
|
c-continuous numpy array of dtype np.intc.
|
|
X_indices : ndarray, dtype=np.intc, ndim=1, mode='c'
|
|
The column indices array of the CSR matrix; a one-dimensional
|
|
c-continuous numpy array of dtype np.intc.
|
|
Y : ndarray, dtype=double, ndim=1, mode='c'
|
|
The target values; a one-dimensional c-continuous numpy array of
|
|
dtype double.
|
|
sample_weights : ndarray, dtype=double, ndim=1, mode='c'
|
|
The weight of each sample; a one-dimensional c-continuous numpy
|
|
array of dtype double.
|
|
"""
|
|
self.n_samples = Y.shape[0]
|
|
self.current_index = -1
|
|
self.X_data_ptr = <double *>X_data.data
|
|
self.X_indptr_ptr = <int *>X_indptr.data
|
|
self.X_indices_ptr = <int *>X_indices.data
|
|
self.Y_data_ptr = <double *>Y.data
|
|
self.sample_weight_data = <double *>sample_weight.data
|
|
# Use index array for fast shuffling
|
|
cdef np.ndarray[int, ndim=1, mode='c'] idx = np.arange(self.n_samples,
|
|
dtype=np.intc)
|
|
self.index = idx
|
|
self.index_data_ptr = <int *>idx.data
|
|
|
|
cdef void next(self, double **x_data_ptr, int **x_ind_ptr,
|
|
int *nnz, double *y, double *sample_weight) nogil:
|
|
cdef int current_index = self.current_index
|
|
if current_index >= (self.n_samples - 1):
|
|
current_index = -1
|
|
|
|
current_index += 1
|
|
cdef int sample_idx = self.index_data_ptr[current_index]
|
|
cdef int offset = self.X_indptr_ptr[sample_idx]
|
|
y[0] = self.Y_data_ptr[sample_idx]
|
|
x_data_ptr[0] = self.X_data_ptr + offset
|
|
x_ind_ptr[0] = self.X_indices_ptr + offset
|
|
nnz[0] = self.X_indptr_ptr[sample_idx + 1] - offset
|
|
sample_weight[0] = self.sample_weight_data[sample_idx]
|
|
|
|
self.current_index = current_index
|
|
|
|
cdef void shuffle(self, seed):
|
|
np.random.RandomState(seed).shuffle(self.index)
|