LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
RnnNER.cpp
Go to the documentation of this file.
1// Copyright 2002-2022 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6#include <QtCore/QTemporaryFile>
7#include <QtCore/QRegularExpression>
8#include <QDir>
9
10
16
29
30
31
32#include "RnnNER.h"
34#include "deeplima/token_type.h"
37
38#define DEBUG_THIS_FILE true
39
41using namespace Lima::Common::PropertyCode;
42using namespace Lima::Common::MediaticData;
43using namespace Lima::Common::Misc;
45using namespace Lima::Common::AnnotationGraphs;
46using namespace deeplima;
47
48
50{
51
52#if defined(DEBUG_LP) && defined(DEBUG_THIS_FILE)
53#define LOG_MESSAGE(stream, msg) stream << msg;
54#define LOG_MESSAGE_WITH_PROLOG(stream, msg) PTLOGINIT; LOG_MESSAGE(stream, msg);
55#else
56 #define LOG_MESSAGE(stream, msg) ;
57#define LOG_MESSAGE_WITH_PROLOG(stream, msg) ;
58#endif
59
60static SimpleFactory<MediaProcessUnit, RnnNER> rnnnerFactory(RNNNER_CLASSID); // clazy:exclude=non-pod-global-static
61
63
65{
66public:
68 ~RnnNERPrivate() = default;
69 void init(GroupConfigurationStructure& unitConfiguration);
70 void tagger(std::vector<segmentation::token_pos>& buffer);
75 const std::set<LinguisticCode>& micros,
77
78 std::map<QString,QString> loadFileTags(const QString& filepath);
79
81
83
84 MediaId m_language;
86 QString m_data;
87 std::shared_ptr< TokenSequenceAnalyzer<> > m_tag;
88 std::function<void()> m_load_fn;
91 std::vector<std::string> m_tags;
93 std::map<QString, QString> m_typeMap;
95};
96
98 ConfigurationHelper("RnnNERPrivate", THIS_FILE_LOGGING_CATEGORY()),
99 m_stringsPool(nullptr), m_tag(nullptr), m_stridx(), m_tags(), m_microAccessor(nullptr), m_loaded(false)
100{
101}
102
104
105}
106
108{
109 delete m_d;
110}
111
112void RnnNER::init(GroupConfigurationStructure &unitConfiguration, Manager *manager)
113{
114 LOG_MESSAGE_WITH_PROLOG(LDEBUG, "RnnNER::init");
115
116 m_d->m_language = manager->getInitializationParameters().media;
117 m_d->m_stringsPool = &MediaticData::changeable().stringsPool(m_d->m_language);
118
119 m_d->init(unitConfiguration);
120}
121
123{
125 TimeUtilsController RnnNERProcessTime("RnnNER");
126 LOG_MESSAGE_WITH_PROLOG(LDEBUG, "start RnnNER");
127 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"));
128 if (anagraph == nullptr)
129 {
130 PTLOGINIT;
131 LERROR << "Can't Process RnnNER: missing data 'AnalysisGraph'";
132 return MISSING_DATA;
133 }
134 auto srcgraph = anagraph->getGraph();
135 auto endVx = anagraph->lastVertex();
140 auto posgraph = std::make_shared<LinguisticAnalysisStructure::AnalysisGraph>(
141 "PosGraph", m_d->m_language, false, true);
142 analysis.setData("PosGraph", posgraph);
143
145 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.getData("AnnotationData"));
146 if (annotationData==nullptr)
147 {
148 annotationData = std::make_shared<AnnotationData>();
152 if (std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph")) != nullptr)
153 {
154 std::dynamic_pointer_cast<AnalysisGraph>(
155 analysis.getData("AnalysisGraph"))->populateAnnotationGraph(
156 annotationData.get(),
157 "AnalysisGraph");
158 }
159 analysis.setData("AnnotationData", annotationData);
160 }
161 if (num_vertices(*srcgraph)<=2)
162 {
163 return SUCCESS_ID;
164 }
165
166 VertexTokenPropertyMap vTokens = get(vertex_token, *srcgraph);
167 auto currentVx = anagraph->firstVertex();
168 auto resultgraph = posgraph->getGraph();
169 remove_edge(posgraph->firstVertex(), posgraph->lastVertex(), *resultgraph);
170
171 std::vector<segmentation::token_pos> buffer;
172 std::vector< LinguisticGraphVertex > anaVertices;
173 std::vector<std::string> v;
174
175 while(currentVx != endVx)
176 {
177 if (currentVx != 0 && vTokens[currentVx] != nullptr)
178 {
179 const auto& src = vTokens[currentVx];
180 v.push_back(src->stringForm().toStdString());
181 buffer.emplace_back();
182 anaVertices.push_back(currentVx);
183 }
184 LinguisticGraphOutEdgeIt it, it_end;
185 boost::tie(it, it_end) = boost::out_edges(currentVx, *srcgraph);
186 if (it != it_end)
187 {
188 currentVx = boost::target(*it, *srcgraph);
189 }
190 else
191 {
192 currentVx = endVx;
193 }
194 }
198 for(unsigned long k = 0; k < anaVertices.size(); k++)
199 {
200 currentVx = anaVertices[k];
201 if (currentVx != 0 && vTokens[currentVx] != nullptr)
202 {
203 const auto& src = vTokens[currentVx];
204 auto& token = buffer[k];
205 token.m_offset = src->position();
206 token.m_len = src->length();
207 token.m_pch = v[k].c_str();
208 token.m_flags = token_flags_t(src->status().getStatus()
209 & StatusType::T_SENTENCE_BRK);
210 }
211 }
212 m_d->tagger(buffer);
213 LOG_MESSAGE(LDEBUG, "tag size: " << m_d->m_tags.size());
214 std::vector<LinguisticGraphVertex>::size_type anaVerticesIndex = 0;
215 // LinguisticGraphVertex previousPosVertex = posgraph->firstVertex();
216 /*
217 * Here we add the part of speech data to the tokens
218 * Adding link beetween the node in the analysis graph and the pos graph.
219 */
220 std::shared_ptr<Automaton::RecognizerMatch> entityFound;
221 QString prev_tag = "O";
222 bool isTagged = false;
223 while (anaVerticesIndex < anaVertices.size())
224 {
225 auto anaVertex = anaVertices[anaVerticesIndex];
226/* auto newVx = boost::add_vertex(*resultgraph);
227 auto agv = annotationData->createAnnotationVertex();
228 annotationData->addMatching("PosGraph", newVx, "annot", agv);
229 annotationData->addMatching("AnalysisGraph", anaVertex, "PosGraph", newVx);
230 annotationData->annotate(agv, QString::fromUtf8("PosGraph"), newVx);*/
231 auto morphoData = get(vertex_data,*srcgraph,anaVertex);
232 // auto srcToken = get(vertex_token,*srcgraph,anaVertex);
233
234 if (morphoData != nullptr)
235 {
236 auto entityTag = QString::fromUtf8(m_d->m_tags[anaVerticesIndex].c_str());
237 if((anaVerticesIndex>0 && entityTag != prev_tag)
238 || (entityTag[0] != 'B' && prev_tag != "O") )
239 {
240 LinguisticGraphVertex newVertex = anagraph->firstVertex();
241 if (entityFound->size() == 1)
242 {
243 auto matches = annotationData->matches(anagraph->getGraphId(),
244 entityFound->getBegin(), "annot");
245 for (auto it = matches.cbegin(); it != matches.cend(); it++)
246 {
247 if (annotationData->hasAnnotation(*it, QString::fromUtf8("SpecificEntity"))
248 && annotationData->annotation(*it, QString::fromUtf8("SpecificEntity"))
249 .pointerValue<SpecificEntities::SpecificEntityAnnotation>()->getType() == entityFound->getType() )
250 {
251 entityFound = nullptr;
252 isTagged = true;
253 }
254 }
255 }
256 if(isTagged)
257 {
258 if (annotationData->dumpFunction("SpecificEntity") == nullptr)
259 {
260 annotationData->dumpFunction("SpecificEntity",
262 }
263
264 auto lingGraph = const_cast<LinguisticGraph*>(anagraph->getGraph());
265 auto tokenMap = get(vertex_token, *lingGraph);
266 auto dataMap = get(vertex_data, *lingGraph);
267
269 auto head = annot.getHead();
270 auto dataHead = dataMap[head];
271
272 // Prepare a new Token and a new MorphoSyntacticData for the new Vertex built basing on entity's head from specificentityannotation data
273 auto seFlex = annot.getString();
274 auto seLemma = annot.getNormalizedString();
275 //No features with this method
276 auto seNorm = annot.getNormalizedForm();
277
278 // creata a new MorphoSyntacticData
279 auto newMorphData = new MorphoSyntacticData();
280
281 // all linguisticElements of this morphosyntacticData share common SE information
283 elem.inflectedForm = seFlex; // StringsPoolIndex
284 elem.lemma = seLemma; // StringsPoolIndex
285 elem.normalizedForm = seNorm; // StringsPoolIndex
286 elem.type = SPECIFIC_ENTITY; // MorphoSyntacticType
287
288 auto seType = entityFound->getType();
289 const auto& resourceName = Common::MediaticData::MediaticData::single().getEntityGroupName(seType.getGroupId())+"Micros";
290 auto res = LinguisticResources::single().getResource(m_d->m_language,
291 resourceName.toStdString());
292 if (res != nullptr)
293 {
294 auto entityMicros = std::dynamic_pointer_cast<SpecificEntities::SpecificEntitiesMicros>(res);
295 auto micros = entityMicros->getMicros(seType);
296 //create a set of linguisticElements. Each LinguisticElement is linked to a
297 //LinguisticCode from the entity
298 m_d->addMicrosToMorphoSyntacticData(newMorphData, dataHead, *micros, elem);
299 }
300
301 const auto& sp = *m_d->m_stringsPool; //match id to string
302 auto newToken = new Token(
303 seFlex,
304 sp[seFlex],
305 entityFound->positionBegin(),
306 entityFound->length());
307 auto tStatus = tokenMap[head]->status();
308
309 auto syntacticData = std::dynamic_pointer_cast<SyntacticAnalysis::SyntacticData>(analysis.getData("SyntacticData"));
310 //LinguisticGraphVertex newVertex;
311 DependencyGraphVertex newDepVertex = 0;
312 if (syntacticData != nullptr)
313 {
314 boost::tie (newVertex, newDepVertex) = syntacticData->addVertex();
315 }
316 else
317 {
318 newVertex = add_vertex(*lingGraph);
319 }
320
321 // Update AnnotationGraph : create a new vertex and annotation
322 auto agv = annotationData->createAnnotationVertex();
323 annotationData->addMatching(anagraph->getGraphId(), newVertex, "annot", agv);
324 annotationData->annotate(agv, QString::fromStdString(anagraph->getGraphId()), newVertex);
325 tokenMap[newVertex] = newToken;
326 dataMap[newVertex] = newMorphData;
327 GenericAnnotation ga(annot);
328 annotationData->annotate(agv, QString::fromUtf8("SpecificEntity"), ga);
329
330 LinguisticGraphInEdgeIt inEdgeIt, inEdgeItEnd;
331 boost::tie(inEdgeIt,inEdgeItEnd) = boost::in_edges(head, *lingGraph);
332 bool success;
334 if(inEdgeIt != inEdgeItEnd)
335 {
336 auto previous = boost::source(*inEdgeIt, *lingGraph);
337 boost::remove_edge(head, previous, *lingGraph);
338 boost::tie(e, success) = boost::add_edge(previous, newVertex, *lingGraph);
339
340 m_d->clearUnreachableVertices(anagraph.get(), previous);
341 m_d->clearUnreachableVertices(anagraph.get(), head);
342 }
343
344 inEdgeIt++;
345 //It is supposed that only one path in the graph exists before this module. Necessarily, only one edge in and out have to be updated
346 if(inEdgeIt != inEdgeItEnd)
347 {
348 LimaException("Error: graph contains multiple paths");
349 }
350
351 LinguisticGraphOutEdgeIt outEdgeIt, outEdgeItEnd;
352 boost::tie(outEdgeIt, outEdgeItEnd) = boost::out_edges(entityFound->getEnd(), *lingGraph);
353 if(outEdgeIt != outEdgeItEnd)
354 {
355 auto next = boost::target(*outEdgeIt, *lingGraph);
356 boost::remove_edge(entityFound->getEnd(), next, *lingGraph);
357 boost::tie(e, success) = boost::add_edge(newVertex, next, *lingGraph);
358
359
360 m_d->clearUnreachableVertices(anagraph.get(), entityFound->getEnd());
361 m_d->clearUnreachableVertices(anagraph.get(), next);
362 }
363
364 outEdgeIt++;
365 if(outEdgeIt != outEdgeItEnd)
366 {
367 LimaException("Error: graph contains multiple paths");
368 }
369
370 // Finalysing cleaning
371 // TO DO : Check if it is useful
372 auto entityFoundIt = entityFound->cbegin();
373 auto entityFoundItEnd = entityFound->cend();
374 for (; entityFoundIt != entityFoundItEnd; entityFoundIt++)
375 {
376 m_d->clearUnreachableVertices(anagraph.get(), (*entityFoundIt).getVertex());
377 }
378 }
379 }
380 if(entityTag == "O")
381 {
382 prev_tag = "O";
383 }
384 else
385 {
386 if(entityFound == nullptr)
387 {
388 entityFound = std::make_shared<Automaton::RecognizerMatch>(anagraph.get(), anaVertex, true);
389 auto entityModex = m_d->m_typeMap[entityTag];
390 entityModex.remove(QRegularExpression("^[BI]-"));
391 EntityType seType;
392 try
393 {
395 }
396 catch (LimaException& e)
397 {
398 PTLOGINIT;
399 LIMA_EXCEPTION( "Lima exception while getting entity type "
400 << entityModex << ": " << e.what());
401 }
402 entityFound->setType(seType);
403 }
404 else
405 {
406 entityFound->addFrontVertex(anaVertex);
407 }
408 prev_tag = QString::fromUtf8(m_d->m_tags[anaVerticesIndex].c_str());
409
410 }
411
412
413 }
414 isTagged = false;
415 anaVerticesIndex++;
416 }
417 LOG_MESSAGE(LDEBUG, "RnnNER NER done.");
419 return SUCCESS_ID;
420}
421
423{
424 m_data = QString::fromStdString(getStringParameter(unitConfiguration, "data", 0, "SentenceBoundaries"));
425 auto model_prefix = QString::fromStdString(getStringParameter(
426 unitConfiguration, "model_prefix", ConfigurationHelper::REQUIRED | ConfigurationHelper::NOT_EMPTY));
427
428 LOG_MESSAGE_WITH_PROLOG(LDEBUG, "RnnNERPrivate::init" << model_prefix);
429
430 QString lang_str = MediaticData::single().media(m_language).c_str();
431 QString resources_path = MediaticData::single().getResourcesPath().c_str();
432 QString model_name = model_prefix;
433 std::string udlang;
434 MediaticData::single().getOptionValue("udlang", udlang);
435
436 if (!fix_lang_codes(lang_str, udlang))
437 {
439 "RnnNERPrivate::init: Can't parse language id " << udlang.c_str(),
441 }
442
443 model_name.replace(QString("$udlang"), QString(udlang.c_str()));
444
445 auto model_file_name = findFileInPaths(resources_path,
446 QString::fromUtf8("/RnnNER/%1/%2.pt")
447 .arg(lang_str, model_name));
448
449 if (model_file_name.isEmpty())
450 {
451 throw InvalidConfiguration("RnnNERPrivate::init: tagger model file not found.");
452 }
453 auto config_file_name = findFileInPaths(resources_path,
454 QString::fromUtf8("/RnnNER/%1/ner.cfg")
455 .arg(lang_str));
456 if (config_file_name.isEmpty())
457 {
458 throw InvalidConfiguration("RnnNERPrivate::init: ner config file not found.");
459 }
460
461 m_typeMap = loadFileTags(config_file_name);
462
463 m_load_fn = [this, model_file_name]()
464 {
465 if (m_loaded)
466 {
467 return;
468 }
469 m_tag = std::make_shared< TokenSequenceAnalyzer<> >(model_file_name.toStdString(),
470 "", "", "", "", "",
471 m_pResolver, 1024, 8);
472
473
474 m_loaded = true;
475 };
476
477 if (!isInitLazy())
478 {
479 m_load_fn();
480 }
481 LOG_MESSAGE(LDEBUG, "classes name: " << m_tag->get_class_names());
482 LOG_MESSAGE(LDEBUG, "classes: " << m_tag->get_classes());
483 for (size_t i = 0; i < m_tag->get_classes().size(); ++i)
484 {
485 m_dumper.set_classes(i, m_tag->get_class_names()[i], m_tag->get_classes()[i]);
486 }
487 m_microAccessor=&(static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager().getPropertyAccessor("MICRO"));
488
489}
490
491void RnnNERPrivate::tagger(std::vector<segmentation::token_pos>& buffer)
492{
493 m_tag->register_handler([this](std::shared_ptr< StringIndex > stridx,
494 const token_buffer_t<>& tokens,
495 const std::vector<StringIndex::idx_t>& lemmata,
496 std::shared_ptr< StdMatrix<uint8_t> > classes,
497 size_t begin,
498 size_t end){
499 TokenSequenceAnalyzer<>::TokenIterator ti(*stridx, tokens, lemmata, classes, begin, end);
500 insertTags(ti);
501 });
502 LOG_MESSAGE_WITH_PROLOG(LDEBUG,buffer[0].m_pch);
503 (*m_tag)(buffer, buffer.size());
504 m_tag->finalize();
505}
506
508{
509 auto classes = m_dumper.getMClasses();
510 while(!ti.end())
511 {
512 LOG_MESSAGE_WITH_PROLOG(LDEBUG, "index: " << ti.token_class(0));
513 m_tags.push_back(classes[0][ti.token_class(0)]);
514 ti.next();
515 }
516}
517
521 const std::set<LinguisticCode>& micros,
523{
524 // try to filter existing microcategories
525 for (auto it = oldMorphData->cbegin(), it_end = oldMorphData->cend(); it != it_end; it++)
526 {
527 if (micros.find(m_microAccessor->readValue((*it).properties)) != micros.end())
528 {
529 elem.properties = (*it).properties;
530 newMorphData->push_back(elem);
531 }
532 }
533 // if no categories kept : assign all micros to keep
534 if (newMorphData->empty())
535 {
536 for (auto it = micros.cbegin(), it_end = micros.cend(); it!=it_end; it++)
537 {
538 elem.properties = *it;
539 newMorphData->push_back(elem);
540 }
541 }
542}
543
545 AnalysisGraph* anagraph,
546 LinguisticGraphVertex from) const
547{
548 auto& g = *(anagraph->getGraph());
549
550 std::queue<LinguisticGraphVertex> verticesToCheck;
551 verticesToCheck.push( from );
552 while (! verticesToCheck.empty() )
553 {
554
555 auto v = verticesToCheck.front();
556 verticesToCheck.pop();
557 bool toClear = false;
558 if (out_degree(v, g) == 0 && v != anagraph->lastVertex())
559 {
560 toClear = true;
561 LinguisticGraphInEdgeIt it, it_end;
562 boost::tie(it,it_end) = in_edges(v,g);
563 for (; it != it_end; it++)
564 {
565 verticesToCheck.push(source(*it,g));
566 }
567 }
568 if (in_degree(v, g) == 0 && v != anagraph->firstVertex())
569 {
570 toClear = true;
571 LinguisticGraphOutEdgeIt it, it_end;
572 boost::tie(it, it_end) = out_edges(v, g);
573 for (; it != it_end; it++)
574 {
575 verticesToCheck.push(target(*it, g));
576 }
577 }
578 if (toClear)
579 {
580 boost::clear_vertex(v, g);
581 }
582 }
583}
584
585std::map<QString,QString> RnnNERPrivate::loadFileTags(const QString& filepath)
586{
587 QFile file(filepath);
588 if (!file.open(QIODevice::ReadOnly | QIODevice::Text))
589 {
590 throw Lima::BadFileException("The file "+filepath.toStdString()+" doesn't exist.");
591 }
592 if(file.size()==0)
593 {
594 std::cout<<"The file is empty.";
595 return {};
596 }
597 QTextStream in(&file);
598 std::map<QString,QString> d;
599 unsigned int i=0;
600 while (!in.atEnd())
601 {
602 QString line = in.readLine();
603 line=line.simplified();
604 i=line.indexOf(":");
605 QString type = line.left(i);
606 QString modex = line.mid(i+1);
607 d[type]=modex;
608 }
609 return d;
610}
611
612}
This file is the main header file for the data related to annotation graphs.
#define CONFIGURATIONHELPER_LOGGING_INIT(X)
#define LOG_MESSAGE_WITH_PROLOG(stream, msg)
#define LOG_MESSAGE(stream, msg)
DependencyGraph::vertex_descriptor DependencyGraphVertex
#define LIMA_EXCEPTION_SELECT_LOGINIT(X, Y, Z)
This macro writes the message Y to the error stream configured by its first parameter X,...
Definition LimaCommon.h:332
#define LIMA_EXCEPTION(X)
This macro writes the message X to a previously configured error stream before throwing a LimaExcepti...
Definition LimaCommon.h:293
#define LDEBUG
Definition LimaCommon.h:157
#define LERROR
Definition LimaCommon.h:161
A graph structure for linguistic analysis.
LinguisticGraph::in_edge_iterator LinguisticGraphInEdgeIt
boost::graph_traits< LinguisticGraph >::edge_descriptor LinguisticGraphEdge
typedefs to simplify the access to various graphs elements
boost::property_map< LinguisticGraph, vertex_token_t >::type VertexTokenPropertyMap
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define PTLOGINIT
#define RNNNER_CLASSID
Definition RnnNER.h:15
Defines a Factory to create Object of type Base.
Data used for the syntactic analyzis of texts.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
This class allows to convert any object into an annotation by inheritance.
Holds linguistic data for one language.
const LimaString & getEntityGroupName(EntityGroupId id) const
const MediaData & mediaData(MediaId media) const
EntityType getEntityType(const LimaString &entityName) const
entity types manager
Provide function to read write and check a property.
LinguisticCode readValue(const LinguisticCode &code) const
read a property in a coded int.
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Use this exception to signal an error in one of the configuration files.
Definition LimaCommon.h:345
The main LIMA exception class.
Definition LimaCommon.h:262
virtual const char * what() const override
Definition LimaCommon.h:280
void getStringParameter(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, const std::string &name, std::string &value, int flags=Flags::REQUIRED, std::string default_value="")
void init(GroupConfigurationStructure &unitConfiguration)
Definition RnnNER.cpp:422
std::map< QString, QString > loadFileTags(const QString &filepath)
Definition RnnNER.cpp:585
void clearUnreachableVertices(AnalysisGraph *anagraph, LinguisticGraphVertex from) const
Definition RnnNER.cpp:544
dumper::AnalysisToConllU< TokenSequenceAnalyzer<>::TokenIterator > m_dumper
Definition RnnNER.cpp:82
std::shared_ptr< TokenSequenceAnalyzer<> > m_tag
Definition RnnNER.cpp:87
void tagger(std::vector< segmentation::token_pos > &buffer)
Definition RnnNER.cpp:491
void addMicrosToMorphoSyntacticData(LinguisticAnalysisStructure::MorphoSyntacticData *newMorphData, const LinguisticAnalysisStructure::MorphoSyntacticData *oldMorphData, const std::set< LinguisticCode > &micros, LinguisticAnalysisStructure::LinguisticElement &elem) const
Definition RnnNER.cpp:518
void insertTags(TokenSequenceAnalyzer<>::TokenIterator &ti)
Definition RnnNER.cpp:507
const Common::PropertyCode::PropertyAccessor * m_microAccessor
Definition RnnNER.cpp:92
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
Definition RnnNER.cpp:122
void init(Lima::Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
Definition RnnNER.cpp:112
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
Definition of a function suitable to be used as a dumper for specific entities annotations of an anno...
A representation of a specific entity to store in the annotation graph.
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
This file contains a class to control log of informations about time, such as logging cumulated time ...
static void logElapsedTime(const std::string &mess, const std::string &taskCategory=std::string(""))
log the number of microseconds since last UpdateCurrentTime
static void updateCurrentTime(const std::string &taskCategory=std::string(""))
store current time for new elapsed time computation
void set_classes(size_t idx, const std::string &class_name, const std::vector< std::string > &data)
const std::vector< std::vector< std::string > > & getMClasses() const
QString findFileInPaths(const QString &paths, const QString &fileName, const QChar &separator)
Find the given file in the given paths.
static SimpleFactory< MediaProcessUnit, RnnNER > rnnnerFactory(RNNNER_CLASSID)
bool fix_lang_codes(QString &lang_str, std::string &udlang)
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
@ MISSING_DATA
Definition LimaCommon.h:243