6#ifndef DEEPLIMA_LIBS_TASKS_NER_TRAIN_WORD_SEQ_VECTORIZER_H
7#define DEEPLIMA_LIBS_TASKS_NER_TRAIN_WORD_SEQ_VECTORIZER_H
22template <
class DataSet,
class StrFeatExtractor,
class UIntFeatExtractor,
class MatrixInt,
class MatrixFloat>
44 const std::vector<embeddable_feature_descr_t>& embeddable_features)
53 const std::vector<embeddable_feature_descr_t>& embeddable_features,
54 const StrFeatExtractor& str_feat_extractor)
55 :
Parent(features, str_feat_extractor),
64 std::vector<deeplima::nets::embd_descr_t> d;
67 assert(feat_descr.m_dim > 0);
68 assert(!feat_descr.m_name.empty());
69 d.emplace_back(feat_descr.m_name, feat_descr.m_dim);
74 typedef std::pair<std::shared_ptr<MatrixInt>, std::shared_ptr<MatrixFloat>>
vectorization_t;
80 typename DataSet::const_iterator i = src.begin();
81 while (src.end() != i)
110 std::shared_ptr<MatrixInt> embeddable_features = dst.first;
111 std::shared_ptr<MatrixFloat> frozen_features = dst.second;
113 typename DataSet::const_iterator it = src.begin();
114 uint64_t current_timepoint = start;
115 while (src.end() != it)
117 while(!(*it).is_word() && src.end() != it)
121 if (src.end() == it)
break;
126 if (current_timepoint == std::numeric_limits<uint64_t>::max())
128 throw std::overflow_error(
"Too much words in the dataset.");
136 uint64_t timepoint,
const typename DataSet::token_t& token)
const
143 switch (feat_descr.
m_type)
146 throw std::runtime_error(
"Unsupported");
149 throw std::runtime_error(
"Unsupported");
155 std::shared_ptr<StringDict> dict
156 = std::dynamic_pointer_cast<StringDict, DictBase>(feat_descr.
m_dict);
157 uint64_t idx = dict->get_idx(feat_val);
158 embeddable_features.set(timepoint, i, idx);
162 throw std::runtime_error(
"Unknown argument type");
std::pair< std::shared_ptr< MatrixInt >, std::shared_ptr< MatrixFloat > > vectorization_t
const std::vector< deeplima::nets::embd_descr_t > get_embd_descr() const
WordSeqVectorizerImpl(const std::vector< typename Parent::feature_descr_t > &features, const std::vector< embeddable_feature_descr_t > &embeddable_features)
WordSeqVectorizerImpl(const std::vector< typename Parent::feature_descr_t > &features, const std::vector< embeddable_feature_descr_t > &embeddable_features, const StrFeatExtractor &str_feat_extractor)
std::vector< embeddable_feature_descr_t > m_embeddable_features
void process(const DataSet &src, vectorization_t dst, uint64_t start) const
void vectorize_timepoint(MatrixFloat &frozen_features, MatrixInt &embeddable_features, uint64_t timepoint, const typename DataSet::token_t &token) const
vectorizers::WordSeqEmbdVectorizer< DataSet, StrFeatExtractor, UIntFeatExtractor, MatrixFloat > Parent
vectorization_t process(const DataSet &src)
vectorization_t init_dst(uint64_t len) const
void vectorize_timepoint(MatrixFloat &target, uint64_t timepoint, const typename DataSet::token_t &token) const
const StrFeatExtractor m_str_feat_extractor
std::shared_ptr< DictBase > m_dict
embeddable_feature_descr_t(typename Parent::feature_type_t type, const std::string &name, int dim, std::shared_ptr< DictBase > dict)