SHOGUN v0.9.0
|
Template class SparseFeatures implements sparse matrices.
Features are an array of TSparse, sorted w.r.t. vec_index (increasing) and withing same vec_index w.r.t. feat_index (increasing);
Sparse feature vectors can be accessed via get_sparse_feature_vector() and should be freed (this operation is a NOP in most cases) via free_sparse_feature_vector().
As this is a template class it can directly be used for different data types like sparse matrices of real valued, integer, byte etc type.
在文件SparseFeatures.h第55行定义。
组合类型 | |
struct | sparse_feature_iterator |
公有成员 | |
CSparseFeatures (int32_t size=0) | |
CSparseFeatures (TSparse< ST > *src, int32_t num_feat, int32_t num_vec, bool copy=false) | |
CSparseFeatures (ST *src, int32_t num_feat, int32_t num_vec) | |
CSparseFeatures (const CSparseFeatures &orig) | |
CSparseFeatures (CFile *loader) | |
virtual | ~CSparseFeatures () |
void | free_sparse_feature_matrix () |
void | free_sparse_features () |
virtual CFeatures * | duplicate () const |
ST | get_feature (int32_t num, int32_t index) |
ST * | get_full_feature_vector (int32_t num, int32_t &len) |
void | get_full_feature_vector (ST **dst, int32_t *len, int32_t num) |
virtual int32_t | get_nnz_features_for_vector (int32_t num) |
TSparseEntry< ST > * | get_sparse_feature_vector (int32_t num, int32_t &len, bool &vfree) |
ST | dense_dot (ST alpha, int32_t num, ST *vec, int32_t dim, ST b) |
void | add_to_dense_vec (float64_t alpha, int32_t num, float64_t *vec, int32_t dim, bool abs_val=false) |
void | free_sparse_feature_vector (TSparseEntry< ST > *feat_vec, int32_t num, bool free) |
TSparse< ST > * | get_sparse_feature_matrix (int32_t &num_feat, int32_t &num_vec) |
void | get_sparse_feature_matrix (TSparse< ST > **dst, int32_t *num_feat, int32_t *num_vec, int64_t *nnz) |
void | clean_tsparse (TSparse< ST > *sfm, int32_t num_vec) |
CSparseFeatures< ST > * | get_transposed () |
TSparse< ST > * | get_transposed (int32_t &num_feat, int32_t &num_vec) |
virtual void | set_sparse_feature_matrix (TSparse< ST > *src, int32_t num_feat, int32_t num_vec) |
ST * | get_full_feature_matrix (int32_t &num_feat, int32_t &num_vec) |
void | get_full_feature_matrix (ST **dst, int32_t *num_feat, int32_t *num_vec) |
virtual bool | set_full_feature_matrix (ST *src, int32_t num_feat, int32_t num_vec) |
virtual bool | apply_preproc (bool force_preprocessing=false) |
virtual int32_t | get_size () |
bool | obtain_from_simple (CSimpleFeatures< ST > *sf) |
virtual int32_t | get_num_vectors () |
int32_t | get_num_features () |
int32_t | set_num_features (int32_t num) |
virtual EFeatureClass | get_feature_class () |
virtual EFeatureType | get_feature_type () |
void | free_feature_vector (TSparseEntry< ST > *feat_vec, int32_t num, bool free) |
int64_t | get_num_nonzero_entries () |
float64_t * | compute_squared (float64_t *sq) |
float64_t | compute_squared_norm (CSparseFeatures< float64_t > *lhs, float64_t *sq_lhs, int32_t idx_a, CSparseFeatures< float64_t > *rhs, float64_t *sq_rhs, int32_t idx_b) |
void | load (CFile *loader) |
void | save (CFile *writer) |
CLabels * | load_svmlight_file (char *fname, bool do_sort_features=true) |
void | sort_features () |
bool | write_svmlight_file (char *fname, CLabels *label) |
virtual int32_t | get_dim_feature_space () |
virtual float64_t | dot (int32_t vec_idx1, CDotFeatures *df, int32_t vec_idx2) |
virtual float64_t | dense_dot (int32_t vec_idx1, const float64_t *vec2, int32_t vec2_len) |
virtual void * | get_feature_iterator (int32_t vector_index) |
virtual bool | get_next_feature (int32_t &index, float64_t &value, void *iterator) |
virtual void | free_feature_iterator (void *iterator) |
virtual const char * | get_name () const |
静态公有成员 | |
static ST | sparse_dot (ST alpha, TSparseEntry< ST > *avec, int32_t alen, TSparseEntry< ST > *bvec, int32_t blen) |
保护成员 | |
virtual TSparseEntry< ST > * | compute_sparse_feature_vector (int32_t num, int32_t &len, TSparseEntry< ST > *target=NULL) |
保护属性 | |
int32_t | num_vectors |
total number of vectors | |
int32_t | num_features |
total number of features | |
TSparse< ST > * | sparse_feature_matrix |
array of sparse vectors of size num_vectors | |
CCache< TSparseEntry< ST > > * | feature_cache |
CSparseFeatures | ( | int32_t | size = 0 | ) |
CSparseFeatures | ( | TSparse< ST > * | src, |
int32_t | num_feat, | ||
int32_t | num_vec, | ||
bool | copy = false |
||
) |
convenience constructor that creates sparse features from the ones passed as argument
src | dense feature matrix |
num_feat | number of features |
num_vec | number of vectors |
copy | true to copy feature matrix |
在文件SparseFeatures.h第85行定义。
CSparseFeatures | ( | ST * | src, |
int32_t | num_feat, | ||
int32_t | num_vec | ||
) |
convenience constructor that creates sparse features from dense features
src | dense feature matrix |
num_feat | number of features |
num_vec | number of vectors |
在文件SparseFeatures.h第113行定义。
CSparseFeatures | ( | const CSparseFeatures< ST > & | orig | ) |
copy constructor
在文件SparseFeatures.h第123行定义。
CSparseFeatures | ( | CFile * | loader | ) |
constructor loading features from file
loader | File object to load data from |
在文件SparseFeatures.h第149行定义。
virtual ~CSparseFeatures | ( | ) | [virtual] |
default destructor
在文件SparseFeatures.h第159行定义。
void add_to_dense_vec | ( | float64_t | alpha, |
int32_t | num, | ||
float64_t * | vec, | ||
int32_t | dim, | ||
bool | abs_val = false |
||
) | [virtual] |
add a sparse feature vector onto a dense one dense+=alpha*sparse
alpha | scalar to multiply with |
num | index of feature vector |
vec | dense vector |
dim | length of the dense vector |
abs_val | if true, do dense+=alpha*abs(sparse) |
实现了CDotFeatures。
在文件SparseFeatures.h第471行定义。
virtual bool apply_preproc | ( | bool | force_preprocessing = false | ) | [virtual] |
apply preprocessor
force_preprocessing | if preprocssing shall be forced |
在文件SparseFeatures.h第823行定义。
void clean_tsparse | ( | TSparse< ST > * | sfm, |
int32_t | num_vec | ||
) |
clean TSparse
sfm | sparse feature matrix |
num_vec | number of vectors in matrix |
在文件SparseFeatures.h第553行定义。
virtual TSparseEntry<ST>* compute_sparse_feature_vector | ( | int32_t | num, |
int32_t & | len, | ||
TSparseEntry< ST > * | target = NULL |
||
) | [protected, virtual] |
compute feature vector for sample num if target is set the vector is written to target len is returned by reference
NOT IMPLEMENTED!
num | num |
len | len |
target | target |
在文件SparseFeatures.h第1476行定义。
compute a^2 on all feature vectors
sq | the square for each vector is stored in here |
在文件SparseFeatures.h第946行定义。
float64_t compute_squared_norm | ( | CSparseFeatures< float64_t > * | lhs, |
float64_t * | sq_lhs, | ||
int32_t | idx_a, | ||
CSparseFeatures< float64_t > * | rhs, | ||
float64_t * | sq_rhs, | ||
int32_t | idx_b | ||
) |
compute (a-b)^2 (== a^2+b^2-2ab) usually called by kernels'/distances' compute functions works on two feature vectors, although it is a member of a single feature: can either be called by lhs or rhs.
lhs | left-hand side features |
sq_lhs | squared values of left-hand side |
idx_a | index of left-hand side's vector to compute |
rhs | right-hand side features |
sq_rhs | squared values of right-hand side |
idx_b | index of right-hand side's vector to compute |
在文件SparseFeatures.h第979行定义。
virtual float64_t dense_dot | ( | int32_t | vec_idx1, |
const float64_t * | vec2, | ||
int32_t | vec2_len | ||
) | [virtual] |
compute dot product between vector1 and a dense vector
vec_idx1 | index of first vector |
vec2 | pointer to real valued vector |
vec2_len | length of real valued vector |
实现了CDotFeatures。
在文件SparseFeatures.h第1347行定义。
ST dense_dot | ( | ST | alpha, |
int32_t | num, | ||
ST * | vec, | ||
int32_t | dim, | ||
ST | b | ||
) |
compute the dot product between dense weights and a sparse feature vector alpha * sparse^T * w + b
alpha | scalar to multiply with |
num | index of feature vector |
vec | dense vector to compute dot product with |
dim | length of the dense vector |
b | bias |
在文件SparseFeatures.h第442行定义。
virtual float64_t dot | ( | int32_t | vec_idx1, |
CDotFeatures * | df, | ||
int32_t | vec_idx2 | ||
) | [virtual] |
compute dot product between vector1 and vector2, appointed by their indices
vec_idx1 | index of first vector |
df | DotFeatures (of same kind) to compute dot product with |
vec_idx2 | index of second vector |
实现了CDotFeatures。
在文件SparseFeatures.h第1321行定义。
virtual CFeatures* duplicate | ( | ) | const [virtual] |
virtual void free_feature_iterator | ( | void * | iterator | ) | [virtual] |
clean up iterator call this function with the iterator returned by get_first_feature
iterator | as returned by get_first_feature |
实现了CDotFeatures。
在文件SparseFeatures.h第1452行定义。
void free_feature_vector | ( | TSparseEntry< ST > * | feat_vec, |
int32_t | num, | ||
bool | free | ||
) |
free feature vector
feat_vec | feature vector to free |
num | index of vector in cache |
free | if vector really should be deleted |
在文件SparseFeatures.h第919行定义。
void free_sparse_feature_matrix | ( | ) |
free sparse feature matrix
在文件SparseFeatures.h第167行定义。
void free_sparse_feature_vector | ( | TSparseEntry< ST > * | feat_vec, |
int32_t | num, | ||
bool | free | ||
) |
free sparse feature vector
feat_vec | feature vector to free |
num | index of this vector in the cache |
free | if vector should be really deleted |
在文件SparseFeatures.h第507行定义。
void free_sparse_features | ( | ) |
free sparse feature matrix and cache
在文件SparseFeatures.h第178行定义。
virtual int32_t get_dim_feature_space | ( | ) | [virtual] |
obtain the dimensionality of the feature space
(not mix this up with the dimensionality of the input space, usually obtained via get_num_features())
实现了CDotFeatures。
在文件SparseFeatures.h第1309行定义。
ST get_feature | ( | int32_t | num, |
int32_t | index | ||
) |
get a single feature
num | number of feature vector to retrieve |
index | index of feature in this vector |
在文件SparseFeatures.h第201行定义。
virtual EFeatureClass get_feature_class | ( | ) | [virtual] |
virtual void* get_feature_iterator | ( | int32_t | vector_index | ) | [virtual] |
iterate over the non-zero features
call get_feature_iterator first, followed by get_next_feature and free_feature_iterator to cleanup
vector_index | the index of the vector over whose components to iterate over |
实现了CDotFeatures。
在文件SparseFeatures.h第1404行定义。
virtual EFeatureType get_feature_type | ( | ) | [virtual] |
ST* get_full_feature_matrix | ( | int32_t & | num_feat, |
int32_t & | num_vec | ||
) |
gets a copy of a full feature matrix num_feat,num_vectors are returned by reference
num_feat | number of features in matrix |
num_vec | number of vectors in matrix |
在文件SparseFeatures.h第664行定义。
void get_full_feature_matrix | ( | ST ** | dst, |
int32_t * | num_feat, | ||
int32_t * | num_vec | ||
) |
gets a copy of a full feature matrix (swig compatible) num_feat,num_vectors are returned by reference
dst | full feature matrix |
num_feat | number of features in matrix |
num_vec | number of vectors in matrix |
在文件SparseFeatures.h第699行定义。
ST* get_full_feature_vector | ( | int32_t | num, |
int32_t & | len | ||
) |
converts a sparse feature vector into a dense one preprocessed compute_feature_vector caller cleans up
num | index of feature vector |
len | length is returned by reference |
在文件SparseFeatures.h第233行定义。
void get_full_feature_vector | ( | ST ** | dst, |
int32_t * | len, | ||
int32_t | num | ||
) |
get the fully expanded dense feature vector num
dst | feature vector |
len | length is returned by reference |
num | index of feature vector |
在文件SparseFeatures.h第265行定义。
virtual const char* get_name | ( | void | ) | const [virtual] |
virtual bool get_next_feature | ( | int32_t & | index, |
float64_t & | value, | ||
void * | iterator | ||
) | [virtual] |
iterate over the non-zero features
call this function with the iterator returned by get_first_feature and call free_feature_iterator to cleanup
index | is returned by reference (-1 when not available) |
value | is returned by reference |
iterator | as returned by get_first_feature |
实现了CDotFeatures。
在文件SparseFeatures.h第1433行定义。
virtual int32_t get_nnz_features_for_vector | ( | int32_t | num | ) | [virtual] |
get number of non-zero features in vector
num | which vector |
实现了CDotFeatures。
在文件SparseFeatures.h第297行定义。
int32_t get_num_features | ( | ) |
int64_t get_num_nonzero_entries | ( | ) |
get number of non-zero entries in sparse feature matrix
在文件SparseFeatures.h第932行定义。
virtual int32_t get_num_vectors | ( | ) | [virtual] |
get number of feature vectors
实现了CFeatures。
在文件SparseFeatures.h第874行定义。
virtual int32_t get_size | ( | ) | [virtual] |
get memory footprint of one feature
实现了CFeatures。
在文件SparseFeatures.h第853行定义。
TSparse<ST>* get_sparse_feature_matrix | ( | int32_t & | num_feat, |
int32_t & | num_vec | ||
) |
get the pointer to the sparse feature matrix num_feat,num_vectors are returned by reference
num_feat | number of features in matrix |
num_vec | number of vectors in matrix |
在文件SparseFeatures.h第523行定义。
void get_sparse_feature_matrix | ( | TSparse< ST > ** | dst, |
int32_t * | num_feat, | ||
int32_t * | num_vec, | ||
int64_t * | nnz | ||
) |
get the pointer to the sparse feature matrix (swig compatible) num_feat,num_vectors are returned by reference
dst | feature matrix |
num_feat | number of features in matrix |
num_vec | number of vectors in matrix |
nnz | number of nonzero elements |
在文件SparseFeatures.h第539行定义。
TSparseEntry<ST>* get_sparse_feature_vector | ( | int32_t | num, |
int32_t & | len, | ||
bool & | vfree | ||
) |
get sparse feature vector for sample num from the matrix as it is if matrix is initialized, else return preprocessed compute_feature_vector
num | index of feature vector |
len | number of sparse entries is returned by reference |
vfree | whether returned vector must be freed by caller via free_sparse_feature_vector |
在文件SparseFeatures.h第316行定义。
TSparse<ST>* get_transposed | ( | int32_t & | num_feat, |
int32_t & | num_vec | ||
) |
compute and return the transpose of the sparse feature matrix which will be prepocessed. num_feat, num_vectors are returned by reference caller has to clean up
num_feat | number of features in matrix |
num_vec | number of vectors in matrix |
在文件SparseFeatures.h第585行定义。
CSparseFeatures<ST>* get_transposed | ( | ) |
void load | ( | CFile * | loader | ) | [virtual] |
CLabels* load_svmlight_file | ( | char * | fname, |
bool | do_sort_features = true |
||
) |
load features from file
fname | filename to load from |
do_sort_features | if true features will be sorted to ensure they are in ascending order |
在文件SparseFeatures.h第1054行定义。
bool obtain_from_simple | ( | CSimpleFeatures< ST > * | sf | ) |
obtain sparse features from simple features
sf | simple features |
在文件SparseFeatures.h第860行定义。
void save | ( | CFile * | writer | ) | [virtual] |
virtual bool set_full_feature_matrix | ( | ST * | src, |
int32_t | num_feat, | ||
int32_t | num_vec | ||
) | [virtual] |
creates a sparse feature matrix from a full dense feature matrix necessary to set feature_matrix, num_features and num_vectors where num_features is the column offset, and columns are linear in memory see above for definition of sparse_feature_matrix
src | full feature matrix |
num_feat | number of features in matrix |
num_vec | number of vectors in matrix |
在文件SparseFeatures.h第734行定义。
int32_t set_num_features | ( | int32_t | num | ) |
set number of features
Sometimes when loading sparse features not all possible dimensions are used. This may pose a problem to classifiers when being applied to higher dimensional test-data. This function allows to artificially explode the feature space
num | the number of features, must be larger than the current number of features |
在文件SparseFeatures.h第893行定义。
virtual void set_sparse_feature_matrix | ( | TSparse< ST > * | src, |
int32_t | num_feat, | ||
int32_t | num_vec | ||
) | [virtual] |
set feature matrix necessary to set feature_matrix, num_features, num_vectors, where num_features is the column offset, and columns are linear in memory see below for definition of feature_matrix
src | new sparse feature matrix |
num_feat | number of features in matrix |
num_vec | number of vectors in matrix |
在文件SparseFeatures.h第648行定义。
void sort_features | ( | ) |
ensure that features occur in ascending order, only call when no preprocessors are attached
在文件SparseFeatures.h第1221行定义。
static ST sparse_dot | ( | ST | alpha, |
TSparseEntry< ST > * | avec, | ||
int32_t | alen, | ||
TSparseEntry< ST > * | bvec, | ||
int32_t | blen | ||
) | [static] |
compute the dot product between two sparse feature vectors alpha * vec^T * vec
alpha | scalar to multiply with |
avec | first sparse feature vector |
alen | avec's length |
bvec | second sparse feature vector |
blen | bvec's length |
在文件SparseFeatures.h第384行定义。
bool write_svmlight_file | ( | char * | fname, |
CLabels * | label | ||
) |
write features to file using svm light format
fname | filename to write to |
label | Label object (number of labels must correspond to number of features) |
在文件SparseFeatures.h第1269行定义。
CCache< TSparseEntry<ST> >* feature_cache [protected] |
feature cache
在文件SparseFeatures.h第1494行定义。
int32_t num_features [protected] |
total number of features
在文件SparseFeatures.h第1488行定义。
int32_t num_vectors [protected] |
total number of vectors
在文件SparseFeatures.h第1485行定义。
TSparse<ST>* sparse_feature_matrix [protected] |
array of sparse vectors of size num_vectors
在文件SparseFeatures.h第1491行定义。