LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
token_sequence_analyzer.h
Go to the documentation of this file.
1// Copyright 2021 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6#ifndef DEEPLIMA_TOKEN_SEQUENCE_ANALYZER
7#define DEEPLIMA_TOKEN_SEQUENCE_ANALYZER
8
9#include <iostream>
10#include <memory>
11#include <string>
12#include <unordered_map>
13#include <unordered_set>
14
15#include <unicode/unistr.h>
16#include <unicode/ustream.h>
17#include "unicode/utypes.h"
18#include <iostream>
19
20// #include "segmentation.h"
22#include "utils/str_index.h"
23#include "token_type.h"
24#include "ner.h"
25#include "conllu/line.h"
30#include "deeplima/token_type.h"
31
32
33template<> struct std::hash<deeplima::morph_model::morph_feats_t> {
34 std::size_t operator()(deeplima::morph_model::morph_feats_t const& s) const noexcept {
35 return s.hash();
36 }
37};
38
39namespace deeplima
40{
41
54{
55public:
56 typedef std::function < void (std::shared_ptr< StringIndex > stridx,
57 const token_buffer_t<>& tokens,
58 const std::vector<StringIndex::idx_t>& lemmata,
59 std::shared_ptr< StdMatrix<uint8_t> > classes,
60 size_t begin,
61 size_t end) > output_callback_t;
62
63 virtual void register_handler(const output_callback_t fn) = 0;
64
65 virtual const std::vector<std::vector<std::string>>& get_classes() const = 0;
66
67 virtual const std::vector<std::string>& get_class_names() const = 0;
68
69 virtual std::shared_ptr<StringIndex> get_stridx() const = 0;
70
71 virtual void finalize() = 0;
72
73 virtual void operator()(const std::vector<segmentation::token_pos>& tokens, uint32_t len) = 0;
74};
75
76template <typename TaggingAuxScalar=float>
78{
79public:
81 {
82 const StringIndex& m_stridx;
83 const token_buffer_t<>& m_buffer;
84 const std::vector<StringIndex::idx_t> m_lemm_buffer;
85 std::shared_ptr< StdMatrix<uint8_t> > m_classes;
86 size_t m_current;
87 size_t m_offset;
88 size_t m_end;
89
90 public:
91 TokenIterator(const StringIndex& stridx, const token_buffer_t<>& buffer,
92 const std::vector<StringIndex::idx_t>& lemm_buffer,
93 std::shared_ptr< StdMatrix<uint8_t> > classes, size_t offset, size_t end)
94 : m_stridx(stridx), m_buffer(buffer), m_lemm_buffer(lemm_buffer), m_classes(classes),
95 m_current(0), m_offset(offset), m_end(end - offset)
96 {
97 assert(end >= offset + 1);
98 }
99
100 inline bool end() const
101 {
102 return m_current >= m_end;
103 }
104
105 inline token_flags_t flags() const
106 {
107 assert(! end());
108 return m_buffer[m_current].m_flags;
109 }
110
111 inline uint16_t token_offset() const
112 {
113 return m_buffer[m_current].m_offset;
114 }
115
116 inline uint16_t token_len() const
117 {
118 return m_buffer[m_current].m_len;
119 }
120
121 // MWT: number of sub-words if this token starts an expanded multiword token
122 // (0 otherwise), and the surface form for the "N-M surface" range line.
123 inline uint8_t mwt_len() const
124 {
125 return m_buffer[m_current].m_mwt_len;
126 }
127
128 inline const char* mwt_surface() const
129 {
130 return m_stridx.get_str(m_buffer[m_current].m_mwt_surface_idx).c_str();
131 }
132
133 // Raw StringIndex id of the surface form; the dependency parser shares this
134 // StringIndex, so the id can be carried over directly when copying tokens.
135 inline uint32_t mwt_surface_idx() const
136 {
137 return m_buffer[m_current].m_mwt_surface_idx;
138 }
139
140 inline uint32_t form_idx() const
141 {
142 return m_buffer[m_current].m_form_idx;
143 }
144
145 inline uint32_t lemma_idx() const
146 {
147 return m_lemm_buffer[m_current];
148 }
149
150 inline const char* form() const
151 {
152 assert(! end());
153 const std::string& f = m_stridx.get_str(m_buffer[m_current].m_form_idx);
154 return f.c_str();
155 }
156
157 inline const char* lemma() const
158 {
159 assert(! end());
160 const std::string& f = m_stridx.get_str(m_lemm_buffer[m_current]);
161 return f.c_str();
162 }
163
164 inline uint32_t head() const
165 {
166 throw std::runtime_error("TokenSequenceAnalyzer<T> does not implement head");
167 return 0;
168 }
169
170 // The tagger iterator carries no dependency information; provided only so the
171 // CoNLL-U dumper template can be instantiated. Never reached with hasDeps=true.
172 inline const char* deprel() const
173 {
174 return "dep";
175 }
176
177 inline void next()
178 {
179 m_current++;
180 }
181
182 inline void reset(size_t position = 0)
183 {
184 // std::cerr << "TokenSequenceAnalyzer::reset" << std::endl;
185 m_current = position;
186 }
187
188 inline size_t position() const
189 {
190 return m_current;
191 }
192
193 inline uint8_t token_class(size_t cls_idx) const
194 {
195 // std::cerr << "time: " << m_current + m_offset << "\n";
196 // std::cerr << "cls_idx: " << cls_idx << "\n";
197 uint8_t val = m_classes->get(m_current + m_offset, cls_idx);
198 return val;
199 }
200 };
201
203
204protected:
205
207 {
209
210 protected:
213
215 {
216 m_ptoken = p;
217 }
218 public:
219
221 : m_stridx(stridx),
222 m_ptoken(nullptr)
223 { }
224
225 inline token_flags_t flags() const
226 {
227 assert(nullptr != m_ptoken);
228 return m_ptoken->m_flags;
229 }
230
231 inline bool eos() const
232 {
233 assert(nullptr != m_ptoken);
235 }
236
237 inline const std::string& form() const
238 {
239 assert(nullptr != m_ptoken);
240 const std::string& f = m_stridx.get_str(m_ptoken->m_form_idx);
241 return f;
242 }
243 };
244
246 {
247 const token_buffer_t<>& m_data;
248 mutable enriched_token_t m_token; // WARNING: only one iterator is supported
249
250 public:
252
254 : m_data(data), m_token(stridx) { }
255
257 {
258 return m_data.size();
259 }
260
261 inline const enriched_token_t& operator[](size_t idx) const
262 {
263 m_token.set_token(m_data.data() + idx);
264 return m_token;
265 }
266 };
267
268 //typedef tagging::impl::FeaturesVectorizer<
269 // enriched_token_buffer_t,
270 // typename enriched_token_buffer_t::token_t> FeaturesVectorizer;
271
272 //typedef tagging::impl::FeaturesVectorizerWithCache<
273 // enriched_token_buffer_t,
274 // typename enriched_token_buffer_t::token_t> FeaturesVectorizer;
275
276 // typedef tagging::impl::EntityTaggingClassifier<TaggingAuxScalar> Classifier;
277
278 // typedef tagging::impl::FeaturesVectorizerWithPrecomputing<
279 // Classifier,
280 // enriched_token_buffer_t,
281 // typename enriched_token_buffer_t::token_t> FeaturesVectorizer;
282
283 // typedef tagging::impl::TaggingImpl< Classifier,
284 // FeaturesVectorizer,
285 // Matrix > EntityTaggingModule;
286
287 // typedef DictEmbdVectorizer<EmbdUInt64FloatHolder, EmbdUInt64Float,
288 // eigen_wrp::EigenMatrixXf> EmbdVectorizer;
289 // typedef lemmatization::impl::LemmatizationImpl< RnnSeq2Seq, EmbdVectorizer,
290 // Matrix> LemmatizationModule;
291
292public:
294 m_buffer_size(0),
297 m_stridx_ptr(std::make_shared<StringIndex>()),
299 m_classes(std::make_shared<StdMatrix<uint8_t>>())
300 {
301 }
302
303 TokenSequenceAnalyzer(const std::string& model_fn,
304 const std::string& lemm_model_fn,
305 const std::string& lemm_dict_fn,
306 const std::string& fixed_ini_fn,
307 const std::string& lower_ini_fn,
308 const std::string& fixed_lemm_fn,
309 const PathResolver& path_resolver,
310 size_t buffer_size,
311 size_t num_buffers)
312 : m_buffer_size(buffer_size),
315 m_stridx_ptr(std::make_shared<StringIndex>()),
317 m_cls(),
318 m_classes(std::make_shared<StdMatrix<uint8_t>>())
319{
320 // std::cerr << "TokenSequenceAnalyzer::TokenSequenceAnalyzer " << model_fn << ", "
321 // << lemm_model_fn << ", " << lemm_dict_fn << ", "
322 // << fixed_ini_fn << ", " << lower_ini_fn << ", "
323 // << fixed_lemm_fn
324 // << std::endl;
325 assert(m_buffer_size > 0);
326 assert(num_buffers > 0);
327 m_buffers.resize(num_buffers);
328 m_lemm_buffers.resize(num_buffers);
329 for ( token_buffer_t<>& b : m_buffers ) b.resize(m_buffer_size);
330 for ( auto& b : m_lemm_buffers ) b.resize(m_buffer_size);
331
332 m_cls.load(model_fn, path_resolver);
333 m_cls.init(1, num_buffers, buffer_size, m_stridx);
334
336 for ( auto& b : m_lemm_buffers )
337 std::fill(b.begin(), b.end(), m_unk_idx);
338
339 {
341 for (uint32_t i = 0; i < buff.size(); ++i)
342 {
343 buff[i].m_form_idx = i;
344 }
345
346 m_cls.precompute_inputs(tagging::impl::enriched_token_buffer_t(buff, m_stridx));
347 }
348
349 if (lemm_model_fn.size() > 0)
350 {
351 try
352 {
353 m_lemm.load(lemm_model_fn, path_resolver);
354 }
355 catch (const std::runtime_error& e)
356 {
357 std::cerr << "TokenSequenceAnalyzer failed to load lemmatization model " << lemm_model_fn << std::endl;
358 throw;
359 }
360 // TODO replace value 128 below (max_input_word_len) by an optimizable parameter
361 m_lemm.init(128, m_cls.get_output_str_dicts_names(), m_cls.get_output_str_dicts());
362
363 m_cls.register_handler([this](
364 std::shared_ptr< StdMatrix<uint8_t> > classes,
365 size_t begin, size_t end, size_t slot_idx){
366 // std::cerr << "handler called: " << slot_idx << std::endl;
367
368 lemmatize(m_buffers[slot_idx], m_lemm_buffers[slot_idx], classes, begin, end);
369
371 m_buffers[slot_idx],
372 m_lemm_buffers[slot_idx],
373 classes,
374 begin,
375 end);
376
377 m_buffers[slot_idx].unlock();
378 });
379 }
380 else
381 {
382 m_cls.register_handler([this](
383 std::shared_ptr< StdMatrix<uint8_t> > classes,
384 size_t begin, size_t end, size_t slot_idx)
385 {
386 // std::cerr << "handler called: " << slot_idx << std::endl;
387 m_classes = classes;
389 m_buffers[slot_idx],
390 m_lemm_buffers[slot_idx],
391 m_classes,
392 begin,
393 end);
394
395 m_buffers[slot_idx].unlock();
396 });
397 }
398
399 if (lemm_dict_fn.size() > 0)
400 {
401 load_lemm_cache(lemm_dict_fn);
402 }
403 if (fixed_ini_fn.size() > 0)
404 {
405 m_fixed_ini_cache = load_pos_cache(fixed_ini_fn);
406 }
407 if (lower_ini_fn.size() > 0)
408 {
409 m_lower_ini_cache = load_pos_cache(lower_ini_fn);
410 }
411 if (fixed_lemm_fn.size() > 0)
412 {
413 m_fixed_lemm_cache = load_pos_cache(fixed_lemm_fn);
414 }
415 }
416
417 void get_classes_from_fn(const std::string& fn,
418 std::vector<std::string>& classes_names,
419 std::vector<std::vector<std::string>>& classes)
420 {
421 m_cls.get_classes_from_fn(fn, classes_names, classes);
422 }
423
425 {
426 }
427
428 virtual void register_handler(const output_callback_t fn) override
429 {
430 // std::cerr << "TokenSequenceAnalyzer::register_handler" << std::endl;
432 }
433
434 virtual const std::vector<std::vector<std::string>>& get_classes() const override
435 {
436 return m_cls.get_output_str_dicts();
437 }
438
439 virtual const std::vector<std::string>& get_class_names() const override
440 {
441 return m_cls.get_output_str_dicts_names();
442 }
443
444 virtual std::shared_ptr<StringIndex> get_stridx() const override
445 {
446 return m_stridx_ptr;
447 }
448
449 /*
450 * The logic is contiguous and the final stop and release of locks are
451 * handled by finalize() and no_more_data().
452 * Contiguous means that only the 100% full buffer triggers processing.
453 * Partially full buffers will wait. Finalization triggers the dispatch of
454 * the remaining data to the pipeline.
455 */
456 virtual void finalize() override
457 {
458 // std::cerr << "TokenSequenceAnalyzer::finalize" << std::endl;
459 if (m_current_timepoint > 0)
460 {
462 {
464 }
465 else
466 {
467 m_cls.no_more_data(m_current_buffer);
468 }
469 }
470
471 m_cls.send_all_results();
474
475 m_cls.reset();
476 }
477
478 virtual void operator()(const std::vector<deeplima::segmentation::token_pos>& tokens, uint32_t len) override
479 {
481 {
483 }
484
485 for (size_t i = 0; i < len; i++)
486 {
488 assert(m_current_buffer < m_buffers.size());
489
491 const segmentation::token_pos& src = tokens[i];
492 token.m_offset = src.m_offset;
493 token.m_len = src.m_len;
494 token.m_form_idx = m_stridx.get_idx(src.m_pch, src.m_len);
495 token.m_flags = token_flags_t(src.m_flags);
496 // Carry MWT metadata: on the first sub-word of an expanded surface token,
497 // intern the surface form so the dumper can emit the "N-M surface" line.
498 token.m_mwt_len = src.m_mwt_len;
499 token.m_mwt_surface_idx =
500 (src.m_mwt_len > 0 && nullptr != src.m_mwt_surface_pch)
502 : 0;
503
506 {
508 if (i < len - 1)
509 {
510 // if we can't wait until next call
512 }
513 }
514 }
515 }
516
517protected:
518
520 {
521 // std::cerr << "acquire_buffer" << std::endl;
522 size_t next_buffer_idx = (m_current_buffer + 1 < m_buffers.size()) ? (m_current_buffer + 1) : 0;
523 const token_buffer_t<>& next_buffer = m_buffers[next_buffer_idx];
524//
525 // wait for buffer
526 while (next_buffer.locked())
527 {
528 m_cls.send_next_results();
529 }
530 assert(!next_buffer.locked());
531
532 m_current_buffer = next_buffer_idx;
534 }
535
536 void start_analysis(size_t buffer_idx, int count = -1)
537 {
538 // std::cerr << "TokenSequenceAnalyzer::start_analysis " << buffer_idx << ", " << count << std::endl;
539 assert(!m_buffers[buffer_idx].locked());
540 m_buffers[buffer_idx].lock();
541
542 const token_buffer_t<>& current_buffer = m_buffers[buffer_idx];
543 m_cls.handle_token_buffer(buffer_idx, tagging::impl::enriched_token_buffer_t(current_buffer, m_stridx), count);
544 }
545
546 void process_buffer(size_t buffer_idx)
547 {
548 const token_buffer_t<>& current_buffer = m_buffers[buffer_idx];
549 m_cls.handle_token_buffer(buffer_idx, tagging::impl::enriched_token_buffer_t(current_buffer, m_stridx));
550 }
551
552 typedef std::pair<StringIndex::idx_t, morph_model::morph_feats_t> lemm_cache_key_t;
554 {
555 std::size_t operator() (const lemm_cache_key_t &arg) const {
556 return std::hash<StringIndex::idx_t>()(arg.first) ^ arg.second.hash();
557 }
558 };
559
560 void load_lemm_cache(const std::string& fn)
561 {
563 std::ifstream f(fn, std::ios::in);
564 if (!f) {
565 std::cerr << "load_lemm_cache failed to open file " << fn << ".\n";
566 throw std::runtime_error(std::string("load_lemm_cache failed to open file ") + fn);
567 }
568 std::string line;
569 while (std::getline(f, line))
570 {
571 if (line.size() == 0)
572 {
573 continue;
574 }
575 const std::vector<std::string> v = utils::split(line, '\t');
576 if (v.size() != 3)
577 {
578 throw std::runtime_error(std::string("Can't decode dict line \"") + line + "\"");
579 }
580
581 const std::vector<std::string> upos_feats = utils::split(v[1], ' ');
582 if (upos_feats.size() != 2)
583 {
584 throw std::runtime_error(std::string("Can't decode upos and feats in dict line \"") + line + "\"");
585 }
586
587 std::map<std::string, std::set<std::string>> feats;
588 if (!deeplima::CoNLLU::CoNLLULine::parse_feats(upos_feats[1], feats))
589 {
590 throw std::runtime_error(std::string("Can't parse feats in dict line \"") + line + "\"");
591 }
592
593 morph_model::morph_feats_t encoded_feats = mm.convert(upos_feats[0], feats);
594
595 const StringIndex::idx_t form_idx = m_stridx.get_idx(v[0]);
596
597 const lemm_cache_key_t form_key(form_idx, encoded_feats);
598
599 const auto it = m_lemm_cache.find(form_key);
600 if (m_lemm_cache.end() != it)
601 {
602 throw std::runtime_error(std::string("Duplicate keys in dict: \"") + line + "\"");
603 }
604
605 m_lemm_cache[form_key] = m_stridx.get_idx(v[2]);
606 }
607
608 }
609
610 // struct fixed_lemm_cache_key_hash
611 // {
612 // std::size_t operator() (const morph_model::morph_feats_t &arg) const {
613 // return arg.hash();
614 // }
615 // };
616
617 std::unordered_set<morph_model::morph_feats_t> load_pos_cache(const std::string& fn)
618 {
619 std::unordered_set<morph_model::morph_feats_t> result;
621 std::ifstream f(fn, std::ios::in);
622 if (!f) {
623 std::cerr << "load_pos_cache failed to open file " << fn << ".\n";
624 throw std::runtime_error(std::string("load_pos_cache failed to open file ") + fn);
625 }
626 std::string line;
627 while (std::getline(f, line))
628 {
629 if (line.size() == 0 || line[0] == '#')
630 {
631 continue;
632 }
633
634 std::map<std::string, std::set<std::string>> feats;
635 morph_model::morph_feats_t encoded_feats = mm.convert(line, feats);
636 // std::cerr << "load_pos_cache add " << line << " " << encoded_feats.toBaseType() << std::endl;
637 result.insert(encoded_feats);
638 }
639 return result;
640 }
641
642
646 inline static std::u32string to_lower(const std::u32string& src)
647 {
648 std::u32string copy = src;
649 std::transform(copy.begin(), copy.end(), copy.begin(),
650 [](unsigned char c){ return std::tolower(c); });
651 return copy;
652 }
653
658 // inline std::u32string to_lower(const std::u32string& utf32String)
659 // {
660 // // Convert the UTF-32 string to a UnicodeString
661 // icu::UnicodeString unicodeString = icu::UnicodeString::fromUTF32(
662 // (const UChar32*)(utf32String.c_str()), utf32String.size());
663 //
664 // // Convert to lowercase
665 // unicodeString.toLower();
666 // std::cerr << unicodeString.toUTF8() << std::endl;
667 // // Convert back to UTF-32 string
668 // std::u32string lowercaseString;
669 // lowercaseString.resize(unicodeString.length());
670 // UErrorCode errorCode ;
671 // unicodeString.toUTF32((UChar32*)(lowercaseString.c_str()),
672 // unicodeString.length(), errorCode);
673 // return lowercaseString;
674 //
675 // }
676
677 void lemmatize(const token_buffer_t<>& buffer,
678 std::vector<StringIndex::idx_t>& lemm_buffer,
679 std::shared_ptr< StdMatrix<uint8_t> > classes,
680 size_t offset, size_t end)
681 {
682 std::u32string target;
683 const auto& lang_morph_model = m_lemm.get_morph_model();
684 for (size_t i = 0; i < end - offset; ++i)
685 {
686 bool sentence_begin = (i==0 || buffer[i-1].eos());
687 if (m_lemm.is_fixed(classes, i + offset))
688 {
689 // std::cerr << "lemmatize " << m_stridx.get_str(buffer[i].m_form_idx)
690 // << ": use buffer" << std::endl;
691 lemm_buffer[i] = buffer[i].m_form_idx;
692 }
693 else
694 {
695 const auto& morph_feats_i = m_lemm.get_morph_feats(classes, i + offset);
696
697 auto upos = morph_model::morph_feats_t(lang_morph_model.decode_upos(morph_feats_i));
698
699 if (sentence_begin && m_fixed_ini_cache.end() != m_fixed_ini_cache.find(upos))
700 {
701 // std::cerr << "lemmatize " << m_stridx.get_str(buffer[i].m_form_idx)
702 // << ": ini fixed POS" << std::endl;
703 lemm_buffer[i] = buffer[i].m_form_idx;
704 }
705 else if (sentence_begin && m_lower_ini_cache.end() != m_lower_ini_cache.find(upos))
706 {
707 // std::cerr << "lemmatize " << m_stridx.get_str(buffer[i].m_form_idx)
708 // << ": ini lowercase POS" << std::endl;
709 target = to_lower(m_stridx.get_ustr(buffer[i].m_form_idx));
710 lemm_buffer[i] = m_stridx.get_idx(target);
711 }
712 else if (m_fixed_lemm_cache.end() != m_fixed_lemm_cache.find(upos))
713 {
714 // std::cerr << "lemmatize " << m_stridx.get_str(buffer[i].m_form_idx)
715 // << ": fixed POS" << std::endl;
716 lemm_buffer[i] = buffer[i].m_form_idx;
717 }
718 else
719 {
720 // std::cerr << "lemmatize " << m_stridx.get_str(buffer[i].m_form_idx)
721 // << ": use model" << std::endl;
722 const lemm_cache_key_t form_key(buffer[i].m_form_idx, m_lemm.get_morph_feats(classes, i + offset) );
723 const auto it = m_lemm_cache.find(form_key);
724 if (m_lemm_cache.end() == it)
725 {
726 const std::u32string& f = m_stridx.get_ustr(buffer[i].m_form_idx);
727 m_lemm.predict(f, classes, i + offset, target);
728
729 // Never emit an empty lemma (invalid CoNLL-U): if the model produced
730 // nothing (e.g. the form reaching it was empty), keep the surface form.
731 lemm_buffer[i] = target.empty() ? buffer[i].m_form_idx
732 : m_stridx.get_idx(target);
733
734 // add form to cache
735 m_lemm_cache[form_key] = lemm_buffer[i];
736 }
737 else
738 {
739 lemm_buffer[i] = it->second;
740 }
741 }
742 }
743 }
744 }
745
749
750 std::shared_ptr<StringIndex> m_stridx_ptr;
753 std::vector<token_buffer_t<>> m_buffers;
754 std::vector<std::vector<StringIndex::idx_t>> m_lemm_buffers; // access control is sync with m_buffers
755
758 std::shared_ptr< StdMatrix<uint8_t> > m_classes;
759
760 std::unordered_map<lemm_cache_key_t, StringIndex::idx_t, lemm_cache_key_hash> m_lemm_cache;
761
762 // The three caches below has been added to allow to force copying the
763 // (possibly lowercased) token as lemma, for a fixed list of pos tags for a given language
764 // for sentence initial (fixed ini and lower ini) and for other tokens (lower lemm)
765 std::unordered_set<morph_model::morph_feats_t> m_fixed_ini_cache;
766 std::unordered_set<morph_model::morph_feats_t> m_lower_ini_cache;
767 std::unordered_set<morph_model::morph_feats_t> m_fixed_lemm_cache;
768
770};
771
772} // namespace deeplima
773
774#endif
static bool parse_feats(const std::string &s, std::map< std::string, std::set< std::string > > &_feats)
Definition line.cpp:142
Handle multithread processing of token sequence.
std::function< void(std::shared_ptr< StringIndex > stridx, const token_buffer_t<> &tokens, const std::vector< StringIndex::idx_t > &lemmata, std::shared_ptr< StdMatrix< uint8_t > > classes, size_t begin, size_t end) > output_callback_t
virtual void register_handler(const output_callback_t fn)=0
virtual std::shared_ptr< StringIndex > get_stridx() const =0
virtual const std::vector< std::vector< std::string > > & get_classes() const =0
virtual void operator()(const std::vector< segmentation::token_pos > &tokens, uint32_t len)=0
virtual const std::vector< std::string > & get_class_names() const =0
const S & get_str(const idx_t idx) const
Definition str_index.h:59
idx_t get_idx(const char *p, size_t len)
Definition str_index.h:24
const U & get_ustr(const idx_t idx)
accessor with side effect.
Definition str_index.h:76
size_t size() const
Definition str_index.h:106
TokenIterator(const StringIndex &stridx, const token_buffer_t<> &buffer, const std::vector< StringIndex::idx_t > &lemm_buffer, std::shared_ptr< StdMatrix< uint8_t > > classes, size_t offset, size_t end)
enriched_token_buffer_t(const token_buffer_t<> &data, const StringIndex &stridx)
void set_token(const token_buffer_t<>::token_t *p)
void start_analysis(size_t buffer_idx, int count=-1)
std::unordered_set< morph_model::morph_feats_t > m_fixed_lemm_cache
std::vector< token_buffer_t<> > m_buffers
std::unordered_map< lemm_cache_key_t, StringIndex::idx_t, lemm_cache_key_hash > m_lemm_cache
std::unordered_set< morph_model::morph_feats_t > m_lower_ini_cache
TokenSequenceAnalyzer(const std::string &model_fn, const std::string &lemm_model_fn, const std::string &lemm_dict_fn, const std::string &fixed_ini_fn, const std::string &lower_ini_fn, const std::string &fixed_lemm_fn, const PathResolver &path_resolver, size_t buffer_size, size_t num_buffers)
std::pair< StringIndex::idx_t, morph_model::morph_feats_t > lemm_cache_key_t
lemmatization::impl::LemmatizationImpl m_lemm
virtual const std::vector< std::vector< std::string > > & get_classes() const override
tagging::impl::TaggingImpl< TaggingAuxScalar > m_cls
static std::u32string to_lower(const std::u32string &src)
This well lower onlu Latin1 characters.
std::vector< std::vector< StringIndex::idx_t > > m_lemm_buffers
virtual void register_handler(const output_callback_t fn) override
virtual const std::vector< std::string > & get_class_names() const override
std::unordered_set< morph_model::morph_feats_t > load_pos_cache(const std::string &fn)
void get_classes_from_fn(const std::string &fn, std::vector< std::string > &classes_names, std::vector< std::vector< std::string > > &classes)
std::shared_ptr< StringIndex > m_stridx_ptr
virtual void operator()(const std::vector< deeplima::segmentation::token_pos > &tokens, uint32_t len) override
void load_lemm_cache(const std::string &fn)
std::shared_ptr< StdMatrix< uint8_t > > m_classes
virtual std::shared_ptr< StringIndex > get_stridx() const override
void lemmatize(const token_buffer_t<> &buffer, std::vector< StringIndex::idx_t > &lemm_buffer, std::shared_ptr< StdMatrix< uint8_t > > classes, size_t offset, size_t end)
TODO correct this function.
std::unordered_set< morph_model::morph_feats_t > m_fixed_ini_cache
bool is_fixed(std::shared_ptr< StdMatrix< uint8_t > > classes, size_t idx)
void init(size_t max_input_word_len, const std::vector< std::string > &class_names, const std::vector< std::vector< std::string > > &class_values)
void predict(const std::u32string &form, std::shared_ptr< StdMatrix< uint8_t > > classes, size_t idx, std::u32string &target)
morph_model::morph_feats_t get_morph_feats(std::shared_ptr< StdMatrix< uint8_t > > classes, size_t idx) const
virtual void load(const std::string &fn, const PathResolver &)
Encoding on one 64 bits integer of the set of morphological features for one token.
Definition morph_model.h:36
Helper class for morphology data (upos, features) binarization.
Definition morph_model.h:93
morph_feats_t convert(const std::string &upos, const std::map< std::string, std::set< std::string > > &feats) const
Class implementing the tagger, used as member in TokenSequenceAnalyzer, the main tagger class Son of ...
std::vector< std::string > split(const std::string &str, char delim)
@ sentence_brk
Definition token_type.h:21
STL namespace.
std::size_t operator()(const lemm_cache_key_t &arg) const
token_flags_t m_flags
Definition token_type.h:44
std::size_t operator()(deeplima::morph_model::morph_feats_t const &s) const noexcept