LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
WordSenseDisambiguator.cpp
Go to the documentation of this file.
1// Copyright 2002-2019 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
17#include <boost/algorithm/string.hpp>
18//#include "boost/graph/adjacency_list.hpp"
19
30#include <stack>
31#include <iostream>
32#include <math.h>
33#include <fstream>
34
35
36
37
38
39using namespace std;
40//using namespace boost;
42
43using namespace Lima::Common::Misc;
44using namespace Lima::Common::MediaticData;
45using namespace Lima::Common::AnnotationGraphs;
48
49
50namespace Lima
51{
52namespace LinguisticProcessing
53{
54namespace WordSenseDisambiguation
55{
57
58
61 Manager* manager)
62
63{
64 LOGINIT("WordSenseDisambiguator");
66
67 try
68 {
69 string mode = unitConfiguration.getParamsValueAtKey("mode");
70 LDEBUG << "Reading mode : " << mode;
71 if (mode.compare("b_most_frequent")==0)
72 {
74 }
75 else if (mode.compare("b_Romanseval_most_frequent")==0)
76 {
78 }
79 else if (mode.compare("b_Jaws_most_frequent")==0)
80 {
82 }
83 else if (mode.compare("s_Wsi_mrd")==0)
84 {
86 }
87 else if (mode.compare("s_Wsi_Dempster_Schaffer")==0)
88 {
90 }
91 else
92 {
94 }
95 }
97 {
98 LERROR << "No 'mode' defined in "<<unitConfiguration.getName()<<" configuration group for language " << (int)m_language;
100 LERROR << "Mode is set to UNKNOWN by default.";
101 }
102 LDEBUG << "Mode is "<< m_mode << " - " << mode();
103 try
104 {
105 string mapping = unitConfiguration.getParamsValueAtKey("mapping");
106 if (mapping.compare("m_Romanseval_senses"))
107 {
109 }
110 else if (mapping.compare("m_Jaws_senses"))
111 {
113 }
114 else
115 {
117 }
118 try
119 {
120 string mappingFile = unitConfiguration.getParamsValueAtKey("mappingFile");
121 loadMapping(mappingFile);
122 }
124 {
125 LERROR << "No 'mappingFile' defined in "<<unitConfiguration.getName()<<" configuration group for language " << (int)m_language;
126 LERROR << "Mapping will not be performed.";
127 }
128 }
130 {
131 LERROR << "No 'mapping' defined in "<<unitConfiguration.getName()<<" configuration group for language " << (int)m_language;
133 LERROR << "Scope is set to UNKNOWN by default.";
134 }
135
136 string dictionaryPath = "";
137 try {
138 dictionaryPath=unitConfiguration.getParamsValueAtKey("dictionaryFile");
139 }
141 {
142 LERROR << "No 'dictionaryFile' defined in "<<unitConfiguration.getName()<<" configuration group for language " << (int)m_language;
143 dictionaryPath = "words.ids";
144 LERROR << "DictionaryFile is set to 'words.ids' by default.";
145 }
146 initDictionaries(dictionaryPath);
147
148
149
150 try {
151 m_sensesPath=unitConfiguration.getParamsValueAtKey("sensesPath");
152 }
154 {
155 LERROR << "No 'sensesPath' defined in "<<unitConfiguration.getName()<<" configuration group for language " << (int)m_language;
156 m_sensesPath = "clusterDir";
157 LERROR << "SensesPath is set to 'clusterDir' by default.";
158 }
159
160 LDEBUG << "SensesPath config ok " ;
161
162
163 if (mode()== S_WSI_MRD || mode()== S_WSI_DS) {
164 try {
165 deque<string> tmpDeque = unitConfiguration.getListsValueAtKey("NounContextList");
166 for (deque<string>::const_iterator it = tmpDeque.begin(); it!=tmpDeque.end(); it++)
167 {
168 m_contextList["N"].insert(tmpDeque.begin(), tmpDeque.end());
169 }
170 tmpDeque.clear();
171 }
173 {
174 LWARN << "No 'NounContextList' defined in "<<unitConfiguration.getName()<<" configuration group for language " << (int)m_language;
175 LWARN << "Default list for NounContext is set to : SUJ_V, COD_V, COMPDUNOM, COMPDUNOM.reverse, ADJPRENSUB.reverse, SUBADJPOST.rverse, window5" ;
176 }
177
178
179 LDEBUG << "ContextLists config ok " ;
180
181 try {
182 m_knnDir=unitConfiguration.getParamsValueAtKey("knnDir");
183 }
185 {
186 LERROR << "No 'knnDir' defined in "<<unitConfiguration.getName()<<" configuration group for language " << (int)m_language;
187 m_knnDir = "knnall";
188 LERROR << "KnnDir is set to 'knnall' by default.";
189 }
190
191 LDEBUG << "KnnDir config ok " ;
192
193 try
194 {
195 const map<string,string>& knnSearchConfig=unitConfiguration.getMapAtKey("knnsearchConfig");
196 m_searcher = new KnnSearcher(knnSearchConfig);
197 }
199 {
200 LERROR << "No 'knnsearchConfig' defined in "<<unitConfiguration.getName()<<" configuration group for language " << (int)m_language;
201 }
202
203 LDEBUG << "KnnSearchConfig config ok " ;
204 } // end mode == S_WSI_XX
205
206 cerr << m_language << endl;
207 const Common::PropertyCode::PropertyManager& macroManager=static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager().getPropertyManager("MACRO");
208
209 m_macroAccessor=&macroManager.getPropertyAccessor();
210
211 m_L_NC = macroManager.getPropertyValue("NC");
212 m_L_NP = macroManager.getPropertyValue("NP");
213 m_L_V = macroManager.getPropertyValue("V");
214 m_L_ADJ = macroManager.getPropertyValue("ADJ");
215 m_L_ADV = macroManager.getPropertyValue("ADV");
216}
217
218
219void WordSenseDisambiguator::initDictionaries(const string& dictionaryPath)
220{
221 LOGINIT("WordSenseDisambiguator");
222 LINFO << "Loading dictionaries from " << dictionaryPath << ".";
223
224 ifstream is(dictionaryPath.c_str(), std::ifstream::binary);
225 if ( !is.good() ) {
226 LERROR << "File " << dictionaryPath << " not read" ;
227 if ( is.eof() ) {
228 LERROR << "(reason is eof)" ;
229 } else if ( is.fail() ) {
230 LERROR << "(reason is fail)" ;
231 } else if ( is.bad() ) {
232 LERROR << "(reason is bad)" ;
233 } else {
234 LERROR << "(reason unknown)" ;
235 }
236 LERROR << ". ";
237 return;
238 }
239 string s;
240 while (getline(is, s)) {
241 vector<string> strs;
242 boost::split( strs, s, boost::is_any_of(" ") );
243 stringstream ss;
244 ss << strs.at(1);
245 uint64_t id;
246 ss>> id;
247 m_lemma2Index[strs.at(0)] = id;
248 m_index2Lemma[id] = strs.at(0);
249 }
250 is.close();
251 LINFO << "Dictionaries loaded from " << dictionaryPath << ".";
252
253}
254
255
256
257void WordSenseDisambiguator::loadMapping(const string& mappingPath)
258{
259 LOGINIT("WordSenseDisambiguator");
260 ifstream is(mappingPath.c_str(), std::ifstream::binary);
261 if ( !is.good() ) {
262 LERROR << "File " << mappingPath << " not read" ;
263 if ( is.eof() ) {
264 LERROR << "(reason is eof)" ;
265 } else if ( is.fail() ) {
266 LERROR << "(reason is fail)" ;
267 } else if ( is.bad() ) {
268 LERROR << "(reason is bad)" ;
269 } else {
270 LERROR << "(reason unknown)" ;
271 }
272 LERROR << ". ";
273 return;
274 }
275 string s;
276 while (getline(is, s)) {
277
278 }
279 is.close();
280 LINFO << "Mapping loaded from " << mappingPath << ".";
281
282}
283
290 AnalysisContent& analysis) const
291{
292 LOGINIT("WordSenseDisambiguator");
294 LINFO << "start WordSenseDisambiguator";
295
296
297
298
299 // create syntacticData
300 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("PosGraph"));
301 if (anagraph==0)
302 {
303 LERROR << "no AnalysisGraph ! abort";
304 return MISSING_DATA;
305 }
306 auto sb = std::dynamic_pointer_cast<SegmentationData>(analysis.getData("SentenceBoundaries"));
307 if (sb==0)
308 {
309 LERROR << "no sentence bounds ! abort";
310 return MISSING_DATA;
311 }
312 if (sb->getGraphId() != "PosGraph") {
313 LERROR << "SentenceBounds computed on graph '" << sb->getGraphId() << "'. WordSenseDisambiguator needs " <<
314 "sentence bounds on PosGraph";
316 }
317 //auto syntacticData = std::dynamic_pointer_cast<SyntacticData>(analysis.getData("SyntacticData"));
318 if (sb==0)
319 {
320 LERROR << "no syntactic data ! abort";
321 return MISSING_DATA;
322 }
323
324
326 auto annotationData = std::dynamic_pointer_cast<AnnotationData>(analysis.getData("AnnotationData"));
327 if (annotationData==0)
328 {
329 annotationData = std::make_shared<AnnotationData>();
333 if (std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph")) != 0)
334 {
335 std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"))->populateAnnotationGraph(annotationData.get(), "AnalysisGraph");
336 }
337 if (std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("PosGraph")) != 0)
338 {
339
340 std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("PosGraph"))->populateAnnotationGraph(annotationData.get(), "PosGraph");
341 }
342
343 analysis.setData("AnnotationData",annotationData);
344 }
345
346
351 if (annotationData->dumpFunction("WordSense") == 0)
352 {
353 annotationData->dumpFunction("WordSense", new DumpWordSense());
354 }
355
357
358 LinguisticGraph* graph=anagraph->getGraph();
359 set< LinguisticGraphVertex > alreadyProcessedVertices;
360
361
362
363 map<string, WordUnit> referenceWords;
364 vector<TargetWordWithContext> targetWords;
365
366
367 //LinguisticGraphVertex beginSentence=sb->getStartVertex();
368 // for each sentence
369 // ??OME2 for (SegmentationData::const_iterator boundItr=sb->begin();
370 // boundItr!=sb->end();
371 for (std::vector<Segment>::const_iterator boundItr=(sb->getSegments()).begin();
372 boundItr!=(sb->getSegments()).end();
373 boundItr++)
374 {
375 LinguisticGraphVertex beginSentence=boundItr->getFirstVertex();
376 LinguisticGraphVertex endSentence=boundItr->getLastVertex();
377 LINFO << "analyze sentence from vertex " << beginSentence << " to vertex " << endSentence;
378
379 // Parse Sentence
380 // for each word in the sentence
381 LinguisticGraphVertex v=beginSentence;
382 vector<set<uint64_t> >lemmasBuffer;
383 while (v!=endSentence)
384 {
385 //LDEBUG << "Processing vertex : " << v ;
386 /*
387 if (alreadyProcessedVertices.find(v) != alreadyProcessedVertices.end())
388 {
389 LinguisticGraphOutEdgeIt ite, ite_end;
390 boost::tie(ite, ite_end)=boost::out_edges(v, *graph);
391 v=target(*ite, *graph);
392 continue;
393 }*/
394
395 // if v is empty, continue
396 MorphoSyntacticData* data = get(vertex_data,*graph,v);
397 if (data == 0
398 || data->empty() )
399 {
400 LinguisticGraphOutEdgeIt ite, ite_end;
401 boost::tie(ite, ite_end)=boost::out_edges(v, *graph);
402 v=target(*ite, *graph);
403 continue;
404 }
405 set<string> lemmas;
406 getLemmas(data, stringspool, lemmas);
407 set<uint64_t> lemmasIds;
408 for (set<string>::const_iterator itLemmas = lemmas.begin(); itLemmas != lemmas.end(); itLemmas++)
409 {
410 // Store preview window contexts
411 LDEBUG << "Storing preview window contexts... ";
412 if ( data->firstValue(*m_macroAccessor) == m_L_NC
413 || data->firstValue(*m_macroAccessor) == m_L_NP
414 || data->firstValue(*m_macroAccessor) == m_L_V
417 )
418 {
419 //LDEBUG << "lemma : "<< *itLemmas ;
420 lemmasIds.insert(lemma2Index(*itLemmas));
421 }
422
423 // Load matching Reference Words
424 if(data->firstValue(*m_macroAccessor) != m_L_NC
425 && data->firstValue(*m_macroAccessor) != m_L_NP)
426 {
427 // Don't process nouns
428 continue;
429 }
430
432 /*
433 LDEBUG << "test size : " << wu.wordSensesUnits().size();
434 if (wu.wordSensesUnits().size()>0) {
435 LDEBUG << "test begint it : " << wu.wordSensesUnits().begin()->senseId();
436 LDEBUG << "test begint it : " << (++wu.wordSensesUnits().begin())->senseId();
437 }
438 for(set<WordSenseUnit>::iterator itSenses = wu.wordSensesUnits().begin();
439 itSenses!= wu.wordSensesUnits().end();
440 itSenses++ ) {
441
442 LDEBUG << "Checking 4 Sense id "<< *itLemmas << " : " << itSenses->senseId()
443 << " : " << itSenses->senseTag()
444 << " : " << itSenses->parentLemmaId();
445 }*/
446
447 referenceWords[*itLemmas] = wu;
448
449
450 LDEBUG << "Added wordunit : " << *itLemmas << " : " << wu ;
451
452
453
454 }
455 lemmasBuffer.push_back(lemmasIds);
456
457 // Load matching Target Words
458 //LDEBUG << "Loading matching target words... ";
459 map<string, set<uint64_t> >context = map<string, set<uint64_t> >();
460
461 // Get syntactic contexts
462 //int nbContxts = getContext(syntacticData, v, graph, stringspool, context);
463 //LDEBUG << "Nb contexts : " << nbContxts << " vs. " << context.size();
464
465 // Add prestored preview window context
466 addPreviewWindowContext(lemmasBuffer, context);
467
468
469 // Store context into targetWords
470 /* print debug */
471 /*
472 LINFO << "Storing current context into targetWords... ";
473 for (set<uint64_t>::iterator itLemmaId = lemmasIds.begin();
474 itLemmaId != lemmasIds.end();
475 itLemmaId++)
476 {
477 cerr << "Adding "<< *itLemmaId << endl;
478 for (SemanticContext::iterator itCtxt = context.begin();
479 itCtxt!= context.end();
480 itCtxt++)
481 {
482 LDEBUG << itCtxt->first << " : " ;
483 for (set<uint64_t>::iterator itValues = itCtxt->second.begin();
484 itValues!= itCtxt->second.end();
485 itValues++)
486 {
487 LDEBUG << *itValues ;
488 }
489 }
490 }
491 */
492 /* end print debug */
493
494 targetWords.push_back(TargetWordWithContext(lemmas, v, context));
495
496
497 // Add postview window contexts
498 //LDEBUG << "Adding postview window contexts... ";
499 addPostviewWindowContext(lemmasIds, targetWords);
500
501
502 // Prepare Next
503
504 alreadyProcessedVertices.insert(v);
505 LinguisticGraphOutEdgeIt ite, ite_end;
506 boost::tie(ite, ite_end)=boost::out_edges(v, *graph);
507 v=target(*ite, *graph);
508 } //end for each word in sentence
509
510
511 // Disambiguate Target Words (separate step due to window contexts addition)
512 for (vector<TargetWordWithContext>::const_iterator itTargets = targetWords.begin();
513 itTargets != targetWords.end();
514 itTargets++)
515 {
516 for (set<string>::const_iterator itLemmas = itTargets->lemmas.begin();
517 itLemmas!= itTargets->lemmas.end();
518 itLemmas++)
519 {
520 //* debug printing
521 LDEBUG << "Context of " << *itLemmas << " : ";
522 for (SemanticContext::const_iterator itContext = itTargets->context.begin();
523 itContext != itTargets->context.end();
524 itContext++)
525 {
526 LDEBUG << "Rel " << itContext->first << " : ";
527 for (set<uint64_t>::iterator itContextValue = itContext->second.begin();
528 itContextValue != itContext->second.end();
529 itContextValue++)
530 {
531 if(m_index2Lemma.find(*itContextValue)!=m_index2Lemma.end())
532 {
533 LDEBUG << "Contextual value : " << index2Lemma(*itContextValue) ;
534 }
535 }
536
537 }
538 //* end debug printing
539
540 // Instanciate annotation
541 WordSenseAnnotation wsa (mode(), mapping(), itTargets->vertex);
542
543 // Disambiguate
544 LDEBUG << *itLemmas ;
545 if (referenceWords.find(*itLemmas) != referenceWords.end())
546 {
547 LDEBUG << "Reference word found for " << *itLemmas << " and mode is " << mode();
548 bool disambOk=false;
549 switch (mode())
550 {
551 case B_MOST_FREQUENT:
554 LDEBUG << "Disambiguation processing : MOST_FREQUENT";
555 disambOk=wsa.disambiguate(referenceWords[*itLemmas]);
556 break;
557 case S_WSI_MRD:
558 LDEBUG << "Disambiguation processing : WSI_MRD";
559 try
560 {
561 disambOk=wsa.disambiguate(m_searcher, referenceWords[*itLemmas], itTargets->context, 0.95, 'A');
562 }
563 catch (std::exception &e)
564 {
565 LDEBUG << "LPException";
566 LDEBUG << e.what();
567 }
568 break;
569 default:
570 LWARN << "No Disambiguation processing. Bad configuration";
571 break;
572 }
573 if (disambOk)
574 {
575 LINFO << "write word sense annotations for "<< *itLemmas <<" on graph";
576 wsa.writeAnnotation(annotationData.get());
577 }
578 else
579 {
580 LWARN << *itLemmas << " was not disambiguated (still ambiguous).";
581 }
582 }
583 else
584 {
585 LWARN << *itLemmas << " was not disambiguated (no referenceWord).";
586 }
587 } // end ambiguous lemmas
588 } // end ambiguous words
589
590 // Release target words
591 targetWords.clear();
592 beginSentence=endSentence;
593 } // end for each sentence in Doc
594
595 // Release Reference Words
596 referenceWords.clear();
597
598 TimeUtils::logElapsedTime("WordSense");
599 return SUCCESS_ID;
600}
601
602int WordSenseDisambiguator::addPostviewWindowContext(const set<uint64_t>& lemmasIds,
603 vector<TargetWordWithContext>& targetWordsWithContext) const
604{
605 int cntPostContext = 0;
606 int maxPostContext = 20;
607 if (lemmasIds.size()>0)
608 {
609 for (vector<TargetWordWithContext>::reverse_iterator itStoredContext = targetWordsWithContext.rbegin()+1;
610 itStoredContext != targetWordsWithContext.rend();
611 itStoredContext++)
612 {
613 if (cntPostContext >= maxPostContext)
614 {
615 break;
616 }
617 if (cntPostContext < 5)
618 {
619 if (contextList("N").find("window5") != contextList("N").end())
620 {
621 itStoredContext->context["window5"].insert(lemmasIds.begin(), lemmasIds.end());
622 }
623 if (cntPostContext < 10)
624 {
625 if (contextList("N").find("window10") != contextList("N").end())
626 {
627 itStoredContext->context["window10"].insert(lemmasIds.begin(), lemmasIds.end());
628 }
629 if (cntPostContext < 20)
630 {
631 if (contextList("N").find("window20") != contextList("N").end())
632 {
633 itStoredContext->context["window20"].insert(lemmasIds.begin(), lemmasIds.end());
634 }
635 }
636 }
637 }
638 cntPostContext++;
639 }
640 }
641 return targetWordsWithContext.size();
642}
643
644
645int WordSenseDisambiguator::addPreviewWindowContext(vector<set<uint64_t> >& previewWindow, map<string, set<uint64_t> >& context) const
646{
647 int cnt = 0;
648 int max = 20;
649 cerr << "previewWindow.size() " << previewWindow.size() << endl;
650 for (vector<set<uint64_t> >::reverse_iterator itWindow = previewWindow.rbegin()+1;
651 itWindow != previewWindow.rend();
652 itWindow++)
653 {
654 if (cnt>=max)
655 {
656 break;
657 }
658 if (cnt < 5)
659 {
660 if (contextList("N").find("window5") != contextList("N").end())
661 {
662 context["window5"].insert(itWindow->begin(), itWindow->end());
663 }
664 if (cnt < 10)
665 {
666 if (contextList("N").find("window10") != contextList("N").end())
667 {
668 context["window10"].insert(itWindow->begin(), itWindow->end());
669 }
670 if (cnt < 20)
671 {
672 if (contextList("N").find("window20") != contextList("N").end())
673 {
674 context["window20"].insert(itWindow->begin(), itWindow->end());
675 }
676 }
677 }
678 }
679 cnt++;
680 }
681 return context.size();
682}
683
686 LinguisticGraph* graph,
687 const FsaStringsPool& stringspool,
688 map<string, set<uint64_t> >& context) const
689{
690 // get Dependency graph and relations
691 EdgeDepRelTypePropertyMap map = get(edge_deprel_type, *(syntacticData-> dependencyGraph()));
692 DependencyGraphVertex dv = syntacticData-> depVertexForTokenVertex(v);
693
694 // out edges
695 DependencyGraphOutEdgeIt it_out, it_out_end;
696 boost::tie(it_out, it_out_end) = out_edges(dv, *(syntacticData-> dependencyGraph()));
697 for (; it_out != it_out_end; it_out++)
698 {
699 LinguisticGraphVertex targetV = target(*it_out, *(syntacticData-> dependencyGraph()));
700 set<string> targetLemmas;
701 MorphoSyntacticData* targetData = get(vertex_data,*graph,targetV);
702 getLemmas(targetData, stringspool, targetLemmas);
703 for (set<string>::iterator itLemmas = targetLemmas.begin(); itLemmas != targetLemmas.end(); itLemmas++)
704 {
705 string relation = static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getSyntacticRelationName(map[*it_out]);
706 if (contextList("N").find(relation) != contextList("N").end())
707 {
708 context[relation].insert(lemma2Index(*itLemmas));
709 }
710 }
711 }
712
713 // in edges
714 DependencyGraphInEdgeIt it_in, it_in_end;
715 boost::tie(it_in, it_in_end) = in_edges(dv, *(syntacticData-> dependencyGraph()));
716 for (; it_in != it_in_end; it_in++)
717 {
718 LinguisticGraphVertex sourceV = source(*it_in, *(syntacticData-> dependencyGraph()));
719 set<string> sourceLemmas;
720 MorphoSyntacticData* sourceData = get(vertex_data,*graph,sourceV);
721 getLemmas(sourceData, stringspool, sourceLemmas);
722 for (set<string>::iterator itLemmas = sourceLemmas.begin(); itLemmas != sourceLemmas.end(); itLemmas++)
723 {
724 string relation = static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getSyntacticRelationName(map[*it_in])+".reverse";
725 if (contextList("N").find(relation) != contextList("N").end())
726 {
727 context[relation].insert(lemma2Index(*itLemmas));
728 }
729 }
730 }
731 return context.size();
732}
733
734
736 const FsaStringsPool& stringspool,
737 set<string>& lemmas) const
738{
739 std::set<StringsPoolIndex> forms=data->allLemma();
740 for (std::set<StringsPoolIndex>::const_iterator formItr=forms.begin();
741 formItr!=forms.end();
742 formItr++)
743 {
744 lemmas.insert(Common::Misc::limastring2utf8stdstring(stringspool[*formItr]));
745 }
746 return lemmas.size();
747}
748
749
750} // closing namespace WordSenseDisambiguation
751} // closing namespace LinguisticProcessing
752} // closing namespace Lima
This file is the main header file for the data related to annotation graphs.
A graph that stores the relations of syntactic dependency between the elements of a DependencyGraph.
DependencyGraph::in_edge_iterator DependencyGraphInEdgeIt
DependencyGraph::out_edge_iterator DependencyGraphOutEdgeIt
boost::property_map< DependencyGraph, edge_deprel_type_t >::type EdgeDepRelTypePropertyMap
DependencyGraph::vertex_descriptor DependencyGraphVertex
@ edge_deprel_type
#define LWARN
Definition LimaCommon.h:160
#define LOGINIT(X)
Definition LimaCommon.h:187
#define LDEBUG
Definition LimaCommon.h:157
#define LINFO
Definition LimaCommon.h:158
#define LERROR
Definition LimaCommon.h:161
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
Defines a Factory to create Object of type Base.
Data used for the syntactic analyzis of texts.
#define WORDSENSEDISAMBIGUATIONPU_CLASSID
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Holds linguistic data for one language.
const FsaStringsPool & stringsPool(MediaId med) const
const MediaData & mediaData(MediaId media) const
Provide tools to manage a specific property.
const PropertyAccessor & getPropertyAccessor() const
give the corresponding PropertyAccessor
LinguisticCode getPropertyValue(const std::string &symbolicValue) const
Get the coded property value from the symbolic value.
std::deque< std::string > & getListsValueAtKey(const std::string &key)
std::map< std::string, std::string > & getMapAtKey(const std::string &key)
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
LinguisticCode firstValue(const Common::PropertyCode::PropertyAccessor &propertyAccessor) const
Return the first non empty value for the given accessor.
This class points to a graph, its dependency graph and the structure that holds the maping between th...
Definition of a function suitable to be used as a dumper for WordSense Annotations of an Annotation g...
bool disambiguate(const WordUnit &wu)
main functions of the global algorithm (called by WordSenseDisambiguator)
AnnotationGraphVertex writeAnnotation(Common::AnnotationGraphs::AnnotationData *ad) const
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
int getLemmas(LinguisticAnalysisStructure::MorphoSyntacticData *data, const FsaStringsPool &stringspool, std::set< std::string > &lemmas) const
int addPreviewWindowContext(std::vector< std::set< uint64_t > > &previewWindow, std::map< std::string, std::set< uint64_t > > &context) const
int getContext(SyntacticAnalysis::SyntacticData *syntacticData, LinguisticGraphVertex &v, LinguisticGraph *graph, const FsaStringsPool &stringspool, std::map< std::string, std::set< uint64_t > > &context) const
int addPostviewWindowContext(const std::set< uint64_t > &lemmasIds, std::vector< TargetWordWithContext > &targetWordsWithContext) const
const std::map< std::string, std::set< std::string > > & contextList() const
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
static void logElapsedTime(const std::string &mess, const std::string &taskCategory=std::string(""))
log the number of microseconds since last UpdateCurrentTime
static void updateCurrentTime(const std::string &taskCategory=std::string(""))
store current time for new elapsed time computation
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
SimpleFactory< MediaProcessUnit, WordSenseDisambiguator > wordSenseDisambiguationFactory(WORDSENSEDISAMBIGUATIONPU_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
@ INVALID_CONFIGURATION
Definition LimaCommon.h:242
@ MISSING_DATA
Definition LimaCommon.h:243
STL namespace.