17#include <boost/algorithm/string.hpp>
52namespace LinguisticProcessing
54namespace WordSenseDisambiguation
64 LOGINIT(
"WordSenseDisambiguator");
71 if (
mode.compare(
"b_most_frequent")==0)
75 else if (
mode.compare(
"b_Romanseval_most_frequent")==0)
79 else if (
mode.compare(
"b_Jaws_most_frequent")==0)
83 else if (
mode.compare(
"s_Wsi_mrd")==0)
87 else if (
mode.compare(
"s_Wsi_Dempster_Schaffer")==0)
98 LERROR <<
"No 'mode' defined in "<<unitConfiguration.
getName()<<
" configuration group for language " << (int)
m_language;
100 LERROR <<
"Mode is set to UNKNOWN by default.";
106 if (
mapping.compare(
"m_Romanseval_senses"))
110 else if (
mapping.compare(
"m_Jaws_senses"))
125 LERROR <<
"No 'mappingFile' defined in "<<unitConfiguration.
getName()<<
" configuration group for language " << (int)
m_language;
126 LERROR <<
"Mapping will not be performed.";
131 LERROR <<
"No 'mapping' defined in "<<unitConfiguration.
getName()<<
" configuration group for language " << (int)
m_language;
133 LERROR <<
"Scope is set to UNKNOWN by default.";
136 string dictionaryPath =
"";
142 LERROR <<
"No 'dictionaryFile' defined in "<<unitConfiguration.
getName()<<
" configuration group for language " << (int)
m_language;
143 dictionaryPath =
"words.ids";
144 LERROR <<
"DictionaryFile is set to 'words.ids' by default.";
155 LERROR <<
"No 'sensesPath' defined in "<<unitConfiguration.
getName()<<
" configuration group for language " << (int)
m_language;
157 LERROR <<
"SensesPath is set to 'clusterDir' by default.";
160 LDEBUG <<
"SensesPath config ok " ;
166 for (deque<string>::const_iterator it = tmpDeque.begin(); it!=tmpDeque.end(); it++)
174 LWARN <<
"No 'NounContextList' defined in "<<unitConfiguration.
getName()<<
" configuration group for language " << (int)
m_language;
175 LWARN <<
"Default list for NounContext is set to : SUJ_V, COD_V, COMPDUNOM, COMPDUNOM.reverse, ADJPRENSUB.reverse, SUBADJPOST.rverse, window5" ;
179 LDEBUG <<
"ContextLists config ok " ;
186 LERROR <<
"No 'knnDir' defined in "<<unitConfiguration.
getName()<<
" configuration group for language " << (int)
m_language;
188 LERROR <<
"KnnDir is set to 'knnall' by default.";
191 LDEBUG <<
"KnnDir config ok " ;
195 const map<string,string>& knnSearchConfig=unitConfiguration.
getMapAtKey(
"knnsearchConfig");
200 LERROR <<
"No 'knnsearchConfig' defined in "<<unitConfiguration.
getName()<<
" configuration group for language " << (int)
m_language;
203 LDEBUG <<
"KnnSearchConfig config ok " ;
221 LOGINIT(
"WordSenseDisambiguator");
222 LINFO <<
"Loading dictionaries from " << dictionaryPath <<
".";
224 ifstream is(dictionaryPath.c_str(), std::ifstream::binary);
226 LERROR <<
"File " << dictionaryPath <<
" not read" ;
228 LERROR <<
"(reason is eof)" ;
229 }
else if ( is.fail() ) {
230 LERROR <<
"(reason is fail)" ;
231 }
else if ( is.bad() ) {
232 LERROR <<
"(reason is bad)" ;
234 LERROR <<
"(reason unknown)" ;
240 while (getline(is, s)) {
242 boost::split( strs, s, boost::is_any_of(
" ") );
251 LINFO <<
"Dictionaries loaded from " << dictionaryPath <<
".";
259 LOGINIT(
"WordSenseDisambiguator");
260 ifstream is(mappingPath.c_str(), std::ifstream::binary);
262 LERROR <<
"File " << mappingPath <<
" not read" ;
264 LERROR <<
"(reason is eof)" ;
265 }
else if ( is.fail() ) {
266 LERROR <<
"(reason is fail)" ;
267 }
else if ( is.bad() ) {
268 LERROR <<
"(reason is bad)" ;
270 LERROR <<
"(reason unknown)" ;
276 while (getline(is, s)) {
280 LINFO <<
"Mapping loaded from " << mappingPath <<
".";
292 LOGINIT(
"WordSenseDisambiguator");
294 LINFO <<
"start WordSenseDisambiguator";
300 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"PosGraph"));
303 LERROR <<
"no AnalysisGraph ! abort";
306 auto sb = std::dynamic_pointer_cast<SegmentationData>(analysis.
getData(
"SentenceBoundaries"));
309 LERROR <<
"no sentence bounds ! abort";
312 if (sb->getGraphId() !=
"PosGraph") {
313 LERROR <<
"SentenceBounds computed on graph '" << sb->getGraphId() <<
"'. WordSenseDisambiguator needs " <<
314 "sentence bounds on PosGraph";
320 LERROR <<
"no syntactic data ! abort";
326 auto annotationData = std::dynamic_pointer_cast<AnnotationData>(analysis.
getData(
"AnnotationData"));
327 if (annotationData==0)
329 annotationData = std::make_shared<AnnotationData>();
333 if (std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"AnalysisGraph")) != 0)
335 std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"AnalysisGraph"))->populateAnnotationGraph(annotationData.get(),
"AnalysisGraph");
337 if (std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"PosGraph")) != 0)
340 std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"PosGraph"))->populateAnnotationGraph(annotationData.get(),
"PosGraph");
343 analysis.
setData(
"AnnotationData",annotationData);
351 if (annotationData->dumpFunction(
"WordSense") == 0)
353 annotationData->dumpFunction(
"WordSense",
new DumpWordSense());
359 set< LinguisticGraphVertex > alreadyProcessedVertices;
363 map<string, WordUnit> referenceWords;
364 vector<TargetWordWithContext> targetWords;
371 for (std::vector<Segment>::const_iterator boundItr=(sb->getSegments()).begin();
372 boundItr!=(sb->getSegments()).end();
377 LINFO <<
"analyze sentence from vertex " << beginSentence <<
" to vertex " << endSentence;
382 vector<set<uint64_t> >lemmasBuffer;
383 while (v!=endSentence)
401 boost::tie(ite, ite_end)=boost::out_edges(v, *graph);
402 v=target(*ite, *graph);
407 set<uint64_t> lemmasIds;
408 for (set<string>::const_iterator itLemmas = lemmas.begin(); itLemmas != lemmas.end(); itLemmas++)
411 LDEBUG <<
"Storing preview window contexts... ";
447 referenceWords[*itLemmas] = wu;
450 LDEBUG <<
"Added wordunit : " << *itLemmas <<
" : " << wu ;
455 lemmasBuffer.push_back(lemmasIds);
459 map<string, set<uint64_t> >context = map<string, set<uint64_t> >();
504 alreadyProcessedVertices.insert(v);
506 boost::tie(ite, ite_end)=boost::out_edges(v, *graph);
507 v=target(*ite, *graph);
512 for (vector<TargetWordWithContext>::const_iterator itTargets = targetWords.begin();
513 itTargets != targetWords.end();
516 for (set<string>::const_iterator itLemmas = itTargets->lemmas.begin();
517 itLemmas!= itTargets->lemmas.end();
521 LDEBUG <<
"Context of " << *itLemmas <<
" : ";
522 for (SemanticContext::const_iterator itContext = itTargets->context.begin();
523 itContext != itTargets->context.end();
526 LDEBUG <<
"Rel " << itContext->first <<
" : ";
527 for (set<uint64_t>::iterator itContextValue = itContext->second.begin();
528 itContextValue != itContext->second.end();
545 if (referenceWords.find(*itLemmas) != referenceWords.end())
547 LDEBUG <<
"Reference word found for " << *itLemmas <<
" and mode is " <<
mode();
554 LDEBUG <<
"Disambiguation processing : MOST_FREQUENT";
558 LDEBUG <<
"Disambiguation processing : WSI_MRD";
563 catch (std::exception &e)
570 LWARN <<
"No Disambiguation processing. Bad configuration";
575 LINFO <<
"write word sense annotations for "<< *itLemmas <<
" on graph";
580 LWARN << *itLemmas <<
" was not disambiguated (still ambiguous).";
585 LWARN << *itLemmas <<
" was not disambiguated (no referenceWord).";
592 beginSentence=endSentence;
596 referenceWords.clear();
603 vector<TargetWordWithContext>& targetWordsWithContext)
const
605 int cntPostContext = 0;
606 int maxPostContext = 20;
607 if (lemmasIds.size()>0)
609 for (vector<TargetWordWithContext>::reverse_iterator itStoredContext = targetWordsWithContext.rbegin()+1;
610 itStoredContext != targetWordsWithContext.rend();
613 if (cntPostContext >= maxPostContext)
617 if (cntPostContext < 5)
621 itStoredContext->context[
"window5"].insert(lemmasIds.begin(), lemmasIds.end());
623 if (cntPostContext < 10)
627 itStoredContext->context[
"window10"].insert(lemmasIds.begin(), lemmasIds.end());
629 if (cntPostContext < 20)
633 itStoredContext->context[
"window20"].insert(lemmasIds.begin(), lemmasIds.end());
641 return targetWordsWithContext.size();
649 cerr <<
"previewWindow.size() " << previewWindow.size() << endl;
650 for (vector<set<uint64_t> >::reverse_iterator itWindow = previewWindow.rbegin()+1;
651 itWindow != previewWindow.rend();
662 context[
"window5"].insert(itWindow->begin(), itWindow->end());
668 context[
"window10"].insert(itWindow->begin(), itWindow->end());
674 context[
"window20"].insert(itWindow->begin(), itWindow->end());
681 return context.size();
688 map<
string, set<uint64_t> >& context)
const
696 boost::tie(it_out, it_out_end) = out_edges(dv, *(syntacticData-> dependencyGraph()));
697 for (; it_out != it_out_end; it_out++)
700 set<string> targetLemmas;
702 getLemmas(targetData, stringspool, targetLemmas);
703 for (set<string>::iterator itLemmas = targetLemmas.begin(); itLemmas != targetLemmas.end(); itLemmas++)
715 boost::tie(it_in, it_in_end) = in_edges(dv, *(syntacticData-> dependencyGraph()));
716 for (; it_in != it_in_end; it_in++)
719 set<string> sourceLemmas;
721 getLemmas(sourceData, stringspool, sourceLemmas);
722 for (set<string>::iterator itLemmas = sourceLemmas.begin(); itLemmas != sourceLemmas.end(); itLemmas++)
731 return context.size();
737 set<string>& lemmas)
const
739 std::set<StringsPoolIndex> forms=data->
allLemma();
740 for (std::set<StringsPoolIndex>::const_iterator formItr=forms.begin();
741 formItr!=forms.end();
746 return lemmas.size();
This file is the main header file for the data related to annotation graphs.
A graph that stores the relations of syntactic dependency between the elements of a DependencyGraph.
DependencyGraph::in_edge_iterator DependencyGraphInEdgeIt
DependencyGraph::out_edge_iterator DependencyGraphOutEdgeIt
boost::property_map< DependencyGraph, edge_deprel_type_t >::type EdgeDepRelTypePropertyMap
DependencyGraph::vertex_descriptor DependencyGraphVertex
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
Defines a Factory to create Object of type Base.
Data used for the syntactic analyzis of texts.
#define WORDSENSEDISAMBIGUATIONPU_CLASSID
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Provide tools to manage a specific property.
const PropertyAccessor & getPropertyAccessor() const
give the corresponding PropertyAccessor
LinguisticCode getPropertyValue(const std::string &symbolicValue) const
Get the coded property value from the symbolic value.
std::string & getParamsValueAtKey(const std::string &key)
std::deque< std::string > & getListsValueAtKey(const std::string &key)
std::map< std::string, std::string > & getMapAtKey(const std::string &key)
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Holds morphosyntactic informations.
LinguisticCode firstValue(const Common::PropertyCode::PropertyAccessor &propertyAccessor) const
Return the first non empty value for the given accessor.
std::set< StringsPoolIndex > allLemma() const
This class points to a graph, its dependency graph and the structure that holds the maping between th...
Definition of a function suitable to be used as a dumper for WordSense Annotations of an Annotation g...
bool disambiguate(const WordUnit &wu)
main functions of the global algorithm (called by WordSenseDisambiguator)
AnnotationGraphVertex writeAnnotation(Common::AnnotationGraphs::AnnotationData *ad) const
Lemma2Index lemma2Index() const
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
int getLemmas(LinguisticAnalysisStructure::MorphoSyntacticData *data, const FsaStringsPool &stringspool, std::set< std::string > &lemmas) const
Index2Lemma m_index2Lemma
void loadMapping(const std::string &mappingPath)
const Common::PropertyCode::PropertyAccessor * m_macroAccessor
Index2Lemma index2Lemma() const
Lemma2Index m_lemma2Index
LimaStatusCode process(AnalysisContent &analysis) const override
void initDictionaries(const std::string &dictionaryPath)
int addPreviewWindowContext(std::vector< std::set< uint64_t > > &previewWindow, std::map< std::string, std::set< uint64_t > > &context) const
int getContext(SyntacticAnalysis::SyntacticData *syntacticData, LinguisticGraphVertex &v, LinguisticGraph *graph, const FsaStringsPool &stringspool, std::map< std::string, std::set< uint64_t > > &context) const
int addPostviewWindowContext(const std::set< uint64_t > &lemmasIds, std::vector< TargetWordWithContext > &targetWordsWithContext) const
std::map< std::string, std::set< std::string > > m_contextList
const std::map< std::string, std::set< std::string > > & contextList() const
static const MediaticData & single()
const singleton accessor
static void logElapsedTime(const std::string &mess, const std::string &taskCategory=std::string(""))
log the number of microseconds since last UpdateCurrentTime
static void updateCurrentTime(const std::string &taskCategory=std::string(""))
store current time for new elapsed time computation
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
SimpleFactory< MediaProcessUnit, WordSenseDisambiguator > wordSenseDisambiguationFactory(WORDSENSEDISAMBIGUATIONPU_CLASSID)
@ B_ROMANSEVAL_MOST_FREQUENT