LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
lemmatization_impl.cpp
Go to the documentation of this file.
1// Copyright 2002-2023 CEA LIST
2// SPDX-FileCopyrightText: 2023 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
7
9
10
12 : m_beam_size(5),
13 m_upos_idx(std::numeric_limits<size_t>::max())
14{}
15
17 size_t /*threads*/,
18 size_t /*buffer_size_per_thread*/
19 )
20 : m_beam_size(5),
21 m_upos_idx(std::numeric_limits<size_t>::max())
22{
23}
24
25
26void LemmatizationImpl::load(const std::string& fn, const PathResolver& /*path_resolver*/)
27{
28 try
29 {
31 }
32 catch (const std::runtime_error& e)
33 {
34 std::cerr << "LemmatizationImpl failed to load " << fn << ": " << e.what();
35 throw;
36 }
37}
38
39void LemmatizationImpl::init(size_t max_input_word_len,
40 const std::vector<std::string>& class_names,
41 const std::vector<std::vector<std::string>>& class_values)
42 {
43 m_upos_idx = std::numeric_limits<size_t>::max();
44 m_max_input_word_len = max_input_word_len;
45 auto uint_dicts = RnnSeq2Seq::get_input_uint_dicts();
46 decltype(uint_dicts) enc_uint_dict;
47 enc_uint_dict.push_back(uint_dicts[0]);
48 m_vectorizer.init(enc_uint_dict, max_input_word_len, 1);
49
50 const auto& str_dicts = RnnSeq2Seq::get_input_str_dicts();
51 const auto& lang_morph_model = RnnSeq2Seq::get_morph_model();
52 m_fixed_upos = std::vector<bool>(32, false); // TODO: find the number of possible UPOS values
53 for (auto& idx : RnnSeq2Seq::get_fixed_upos())
54 {
55 m_fixed_upos[idx] = true;
56 }
57
58 assert(class_names.size() == class_values.size());
59 EmbdUInt64FloatHolder enc_feats_dict;
60 for (size_t feat_idx = 0; feat_idx < lang_morph_model.get_feats_count(); ++feat_idx)
61 {
62 std::vector<uint64_t> v;
63 auto cls_idx = std::numeric_limits<size_t>::max();
64
65 const auto& feat_name = lang_morph_model.get_feat_name(feat_idx);
66 auto it = std::find(class_names.begin(), class_names.end(), feat_name);
67
68 if (class_names.end() != it)
69 {
70 const auto& feat_vec = lang_morph_model.get_feat_vec_ref(feat_idx);
71 auto it_diff = it - class_names.begin();
72 if (it_diff != std::numeric_limits<ptrdiff_t>::max())
73 cls_idx = it_diff;
74 assert(cls_idx != std::numeric_limits<uint64_t>::max());
75 assert(cls_idx != std::numeric_limits<size_t>::max());
76
77 v.resize(class_values[cls_idx].size(), 0);
78 // assert(class_values[cls_idx].size() == feat_vec.size());
79 for (size_t j = 0; j < feat_vec.size(); ++j)
80 {
81 // TODO: rewrite this
82 for (size_t k = 0; k < class_values[cls_idx].size(); ++k)
83 {
84 if (feat_vec[j] == class_values[cls_idx][k]
85 || ("_" == feat_vec[j] && "-" == class_values[cls_idx][k]))
86 {
87 v[k] = j;
88 break;
89 }
90 }
91 }
92 }
93 else
94 {
95 std::cerr << "Warning: classifier doesn't provide required feature: \""
96 << feat_name << "\"" << std::endl;
97 }
98
99 auto dd = std::make_shared<Dict<uint64_t>>(v);
101
102 d.init(dd, str_dicts[feat_idx].get_tensor().transpose());
103 enc_feats_dict.push_back(d);
104 m_feat2cls.push_back(cls_idx);
105 if (cls_idx != std::numeric_limits<size_t>::max() && cls_idx < class_names.size() && class_names[cls_idx] == "upos")
106 {
107 m_upos_idx = cls_idx;
108 }
109 }
110 m_feat_vectorizer.init(enc_feats_dict, 1, enc_feats_dict.size());
111
112 if (m_upos_idx == std::numeric_limits<size_t>::max())
113 {
114 throw std::logic_error("Underlying classifier doesn't provide UPOS.");
115 }
116
117 RnnSeq2Seq::init_new_worker(max_input_word_len);
118}
119
120bool LemmatizationImpl::is_fixed(std::shared_ptr< StdMatrix<uint8_t> > classes, size_t idx)
121{
122 assert(m_upos_idx != std::numeric_limits<size_t>::max());
123 auto upos = classes->get(idx, m_upos_idx);
124 return m_fixed_upos[upos];
125}
126
128{
129 const auto& lang_morph_model = RnnSeq2Seq::get_morph_model();
130 std::vector<size_t> feats(lang_morph_model.get_feats_count());
131
132 const auto& feat2cls = m_feat2cls;
133 const morph_model::morph_feats_t v = lang_morph_model.convert([idx, &feat2cls, &classes](size_t feat_idx) {
134 if (feat_idx == std::numeric_limits<size_t>::max())
135 {
136 throw std::runtime_error(std::string("get_morph_feats wrong feat_idx: max size_t"));
137 }
138 else if (feat_idx >= feat2cls.size())
139 {
140 throw std::runtime_error(std::string("get_morph_feats wrong feat_idx: larger than feat2cls size"));
141 }
142 assert(feat_idx < feat2cls.size() && feat_idx != std::numeric_limits<size_t>::max());
143 auto class_idx = feat2cls[feat_idx];
144 if (class_idx == std::numeric_limits<size_t>::max())
145 return std::numeric_limits<uint64_t>::max();
146 assert(class_idx != std::numeric_limits<uint64_t>::max());
147 return classes->get(idx, class_idx);
148 });
149
150 return v;
151}
152
153void LemmatizationImpl::predict(const std::u32string& form,
154 std::shared_ptr< StdMatrix<uint8_t> > classes, size_t idx,
155 std::u32string& target)
156{
157 // An empty form has nothing to lemmatize. It must never reach the seq2seq
158 // encoder: the BiLSTM cannot run on a zero-length sequence (with input_len == 0
159 // it indexes column -1 of its workbench → SIGSEGV). Return an empty lemma.
160 if (form.empty())
161 {
162 target.clear();
163 return;
164 }
165
166 // The vectorizer and the encoder workbench were sized for at most
167 // m_max_input_word_len characters (see init). A longer form would overrun both
168 // the vectorizer storage and the encoder output matrix, so clamp it.
169 const size_t form_len = (m_max_input_word_len > 0 && form.size() > m_max_input_word_len)
171 : form.size();
172
173 // 1. vectorize
174 for (size_t i = 0; i < form_len; ++i)
175 {
176 m_vectorizer.set(i, 0, form[i]);
177 }
178
179 for (size_t i = 0; i < m_feat2cls.size(); ++i)
180 {
181 if (m_feat2cls[i] != std::numeric_limits<size_t>::max())
182 {
183 m_feat_vectorizer.set(0, i, classes->get(idx, m_feat2cls[i]));
184 }
185 else
186 {
187 m_feat_vectorizer.set(0, i, 0);
188 }
189 }
190
191 // 2. run prediction
192 std::vector<uint32_t> output(0, form_len * 2);
194 static_cast<const EmbdVectorizer>(m_vectorizer).get_tensor(),
195 static_cast<const EmbdVectorizer>(m_feat_vectorizer).get_tensor(),
196 form_len,
197 form_len * 2,
199 output,
200 { "output" });
201
202 // 3. generate output
203 target.clear();
204 output.push_back(0);
205 target = std::u32string((char32_t*)(output.data()));
206}
207
208
209
210} // namespace impl
211 // namespace lemmatization
212 // namespace deeplima
213
void init(const DH &dicts, int64_t max_time, uint64_t max_feat)
void set(uint64_t time, uint64_t feat, typename D::value_t value)
void init(std::shared_ptr< Dict< T > > dict, const M &tensor, bool transpose=true)
Definition embd_dict.h:34
void load(const std::string &fn)
virtual void predict(size_t worker_id, const Eigen::MatrixXf &inputs, int64_t input_begin, int64_t input_end, int64_t output_begin, int64_t output_end, std::shared_ptr< StdMatrix< uint8_t > > &output, const std::vector< std::string > &outputs_names)=0
const uint_dicts_holder_t & get_input_uint_dicts() const
virtual size_t init_new_worker(size_t input_len, bool precomputed_input=false)
const str_dicts_holder_t & get_input_str_dicts() const
bool is_fixed(std::shared_ptr< StdMatrix< uint8_t > > classes, size_t idx)
size_t m_upos_idx
index in the features matrix of the upos line
std::vector< bool > m_fixed_upos
index i is true if upos whose index is i is considered as fixed
size_t m_max_input_word_len
max word length the encoder workbench / vectorizer were sized for (see init)
void init(size_t max_input_word_len, const std::vector< std::string > &class_names, const std::vector< std::vector< std::string > > &class_values)
void predict(const std::u32string &form, std::shared_ptr< StdMatrix< uint8_t > > classes, size_t idx, std::u32string &target)
morph_model::morph_feats_t get_morph_feats(std::shared_ptr< StdMatrix< uint8_t > > classes, size_t idx) const
virtual void load(const std::string &fn, const PathResolver &)
Encoding on one 64 bits integer of the set of morphological features for one token.
Definition morph_model.h:36
STL namespace.