LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
WordSenseDisambiguator.h
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
17#ifndef LIMA_WORDSENSEDISAMBIGUATION_WORDSENSEDISAMBIGUATOR_H
18#define LIMA_WORDSENSEDISAMBIGUATION_WORDSENSEDISAMBIGUATOR_H
19
27#include "WordSenseAnnotation.h"
28
29
30#include <map>
31#include <set>
32#include <string>
33#include "WordUnit.h"
34#include "WordSenseAnnotation.h"
35#include "CommonTypedefs.h"
36
37namespace Lima
38{
39namespace Common {
40namespace AnnotationGraphs {
41 class AnnotationData;
42}
43}
44namespace LinguisticProcessing
45{
46namespace SyntacticAnalysis {
47class SyntacticData;
48}
49namespace LinguisticAnalysisStructure {
50class AnalysisGraph;
51}
52namespace WordSenseDisambiguation
53{
54class WordSenseAnnotation;
55
56#define WORDSENSEDISAMBIGUATIONPU_CLASSID "WordSenseDisambiguation"
57
58
59
60typedef std::map<int, std::string> SensesMapping;
61
62
66{
67public:
69
70 ~WordSenseDisambiguator() { delete m_searcher; }
71
72 void init(
74 Manager* manager) override;
75
76 LimaStatusCode process(AnalysisContent& analysis) const override;
77
78 // inline
79 MediaId language() const;
80 const std::string& knnDir() const ;
81 Mode mode() const ;
82 Mapping mapping() const ;
83 bool resolve() const;
84 const Common::PropertyCode::PropertyAccessor* macroAccessor() const;
86
87 Lemma2Index lemma2Index() const;
88 Index2Lemma index2Lemma() const;
89 uint64_t lemma2Index(std::string lemma) const;
90 std::string index2Lemma(uint64_t index) const;
91 const std::map<std::string, std::set<std::string> >& contextList() const ;
92 std::set<std::string> contextList(std::string relation) const ;
93 //end inline
94
95private:
96
97
98protected:
104
106 MediaId m_language;
107 std::string m_knnDir;
108 std::map<std::string, std::set<std::string> >m_contextList;
115 std::string m_sensesPath;
116
117 void initDictionaries(const std::string& dictionaryPath);
118 void loadMapping(const std::string& mappingPath);
120 const FsaStringsPool& stringspool,
121 std::set<std::string>& lemmas) const ;
122 int getContext(SyntacticAnalysis::SyntacticData* syntacticData,
124 LinguisticGraph* graph,
125 const FsaStringsPool& stringspool,
126 std::map<std::string, std::set<uint64_t> >& context) const ;
127 int addPreviewWindowContext(std::vector<std::set<uint64_t> >& previewWindow,
128 std::map<std::string, std::set<uint64_t> >& context) const ;
129 int addPostviewWindowContext(const std::set<uint64_t>& lemmasIds,
130 std::vector<TargetWordWithContext>& targetWordsWithContext) const ;
131
132};
133
134
136{
137 return m_language;
138}
139inline const std::map<std::string, std::set<std::string> >& WordSenseDisambiguator::contextList() const
140{
141 return m_contextList;
142}
143inline const std::string& WordSenseDisambiguator::knnDir() const
144{
145 return m_knnDir;
146}
148{
149 return m_mode;
150}
152{
153 return m_mappingMode;
154}
159
160inline
169inline
170 uint64_t WordSenseDisambiguator::lemma2Index(std::string lemma) const
171{
172 if (m_lemma2Index.find(lemma)!=m_lemma2Index.end())
173 {
174 return m_lemma2Index.find(lemma)->second;
175 }
176 return 0;
177}
178inline
179 std::string WordSenseDisambiguator::index2Lemma(uint64_t index) const
180{
181 if (m_index2Lemma.find(index)!=m_index2Lemma.end())
182 {
183 return m_index2Lemma.find(index)->second;
184 }
185 return "";
186}
187
188inline std::set<std::string> WordSenseDisambiguator::contextList(std::string relation) const
189{
190 if (m_contextList.find(relation) != m_contextList.end())
191 {
192 return m_contextList.find(relation)->second;
193 }
194 return std::set<std::string>() ;
195}
196
197} // closing namespace WordSenseDisambiguation
198} // closing namespace LinguisticProcessing
199} // closing namespace Lima
200
201#endif // LIMA_WORDSENSEDISAMBIGUATION_WORDSENSEDISAMBIGUATOR_H
A graph structure for linguistic analysis.
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
Data used for the syntactic analyzis of texts.
#define LIMA_WORDSENSEANALYSIS_EXPORT
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
Provide function to read write and check a property.
This class points to a graph, its dependency graph and the structure that holds the maping between th...
const Common::PropertyCode::PropertyAccessor * macroAccessor() const
const Common::PropertyCode::PropertyAccessor * microAccessor() const
const std::map< std::string, std::set< std::string > > & contextList() const
std::map< std::string, uint64_t > Lemma2Index
Definition WordUnit.h:26
std::map< uint64_t, std::string > Index2Lemma
Definition WordUnit.h:25
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236