LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
WordSenseAnnotation.cpp
Go to the documentation of this file.
1// Copyright 2002-2019 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
7
8#include "KnnSearcher.h"
15
16
17//#include "common/time/traceUtils.h"
18
19#include <iostream>
20
21
22//using namespace boost;
23using namespace std;
25
26using namespace Lima::Common::Misc;
27using namespace Lima::Common::MediaticData;
28using namespace Lima::Common::AnnotationGraphs;
31
32
33namespace Lima
34{
35namespace LinguisticProcessing
36{
37namespace WordSenseDisambiguation
38{
39
40
42{
43
45 try
46 {
48 return SUCCESS_ID;
49 }
50 catch (const boost::bad_any_cast& e)
51 {
52 LERROR << "This annotation is not a WordSenseAnnotation ; nothing dumped";
53 return UNKNOWN_ERROR;
54 }
55}
56
61{
62 return confidence() != -1 ;
63}
64
65
66
67
71{
72 LOGINIT("WordSenseDisambiguator");
73 if (wu.nbSenses()==0 || wu.wordSensesUnits().size()==0)
74 {
75 return false;
76 }
77 switch(m_mode)
78 {
79 case B_MOST_FREQUENT:
82 for (set<WordSenseUnit>::iterator itSenses = wu.wordSensesUnits().begin();
83 itSenses != wu.wordSensesUnits().end();
84 itSenses++)
85 {
86 if (itSenses->senseId()==0 && m_mode == itSenses->mode())
87 {
88 m_confidence = 0;
89 retrieveSense(itSenses);
90 break;
91 }
92 }
93 break;
94 case S_WSI_MRD:
95 LERROR << "Wrong usage";
96 // fall through
97 default:
98 LWARN << "No Disambiguation processing. Bad configuration";
99 return false;
100 }
101
102 if (m_confidence == -1)
103 {
104 return false;
105 }
106 return true;
107}
108
109
111 const WordUnit& wu,
112 const SemanticContext& context,
113 double thres,
114 char method)
115{
116 LOGINIT("WordSenseDisambiguator");
117 LDEBUG << "Begin disambiguation of "<< wu.lemma();
118 if (wu.nbSenses()==0 || wu.wordSensesUnits().size()==0)
119 {
120 return false;
121 }
122 switch(m_mode)
123 {
124 case S_WSI_MRD:
125 classifKnnMRD(searcher, wu, context, thres, method);
126 break;
127 case B_MOST_FREQUENT:
130 LERROR << "Wrong usage";
131 // fall through
132 default:
133 LWARN << "No Disambiguation processing. Bad configuration";
134 return false;
135 }
136
137 if (m_confidence == -1)
138 {
139 return false;
140 }
141 return true;
142}
143
145 const WordUnit& wu,
146 const SemanticContext& context,
147 double thres,
148 char method)
149{
150 LOGINIT("WordSenseDisambiguator");
151 float confidence = -1;
152 map<int, map<string, double> > cvListRep;
153 // Compute needed knn dists
154 for (SemanticContext::const_iterator itContext = context.begin();
155 itContext!= context.end();
156 itContext++) {
157
158 LDEBUG << " Computing knn dists for " << itContext->first ;
159 NNList knns;
160 searcher->getKNN(wu.lemmaId(), itContext, knns) ;
161
162 for (NNList::iterator itNN = knns.begin();
163 itNN!= knns.end();
164 itNN++)
165 {
166 LDEBUG << itNN->first << ":" << itNN->second;
167 }
168 if (knns.size()==0 || knns.rbegin()->second==0) {
169
170 LWARN << wu.lemmaId() << " : " << wu.lemma() << " has a corrupted sphere. Process may return biased results ";
171
172 for (NNList::iterator itNN = knns.begin();
173 itNN!= knns.end();
174 itNN++) {
175 LDEBUG << itNN->first << ":" << itNN->second;
176 }
177
178 continue;
179 }
180 computeConfidenceVector(wu, knns, itContext, cvListRep);
181 }
182
183 // Classsify
184
185 std::set<WordSenseUnit>::iterator bestSense = classify(wu, context, cvListRep, thres, method);
186 retrieveSense(bestSense);
187 return confidence;
188}
189
190/* internal computation functions called by classifKnnMRD */
191
192
193std::set<WordSenseUnit>::iterator WordSenseAnnotation::classify(const WordUnit& wu,
194 const SemanticContext& context,
195 const map<int, map<string, double> >& cvListRep,
196 const double thres,
197 const char method )
198{
199 LOGINIT("WordSenseDisambiguator");
200 LDEBUG << "Classifying : " << wu.lemma() << " nbSenses : " << wu.wordSensesUnits().size();
201 // prepare confidence vectors
202 map<int, double> assignedClasses;
203 map<int, double> scoresByClass;
204 double max = 0;
205 set<WordSenseUnit>::iterator bestSense;
206#ifndef WIN32
207 for(set<WordSenseUnit>::iterator itSenses = wu.wordSensesUnits().begin();
208 itSenses!= wu.wordSensesUnits().end();
209 itSenses++ ) {
210 LDEBUG << "cluster : " << itSenses->senseId();
211 double checkSum = 0;
212 double w,
213 val_cluster = 0.;
214 double entropySum=sumLogCvList(cvListRep);
215
216
217 /* print debug */
218 for (map<int, map<string, double> >::const_iterator itdeb = cvListRep.begin();
219 itdeb!= cvListRep.end();
220 itdeb++) {
221 LDEBUG << "cvlistrepsize : " << itdeb->first << " : " << itdeb->second.size();
222 for (map<string, double>::const_iterator itdeb2 = itdeb->second.begin();
223 itdeb2 !=itdeb->second.end();
224 itdeb2++) {
225 LDEBUG << itdeb2->first << " -> " << itdeb2->second;
226 }
227 }
228 /* end print debug */
229 for (SemanticContext::const_iterator itContext = context.begin();
230 itContext!= context.begin();
231 itContext++) {
232 if (cvListRep.find(itSenses->senseId())!=cvListRep.end()
233 && cvListRep.find(itSenses->senseId())->second.find(itContext->first)
234 ==cvListRep.find(itSenses->senseId())->second.end() ) {
235 w=0;
236 LDEBUG << itContext->first << " not found for cluster "<< itSenses->senseId();
237 } else {
238 w = sumLogCv(cvListRep, itContext->first) / entropySum;
239 LDEBUG << itContext->first << " "<< itSenses->senseId() << " w= " << sumLogCv(cvListRep, itContext->first) << "/" << entropySum << " = " << w ;
240 }
241 val_cluster += w * cvCluster(cvListRep, itContext->first, itSenses->senseId());
242 checkSum+=w;
243 }
244 switch (method) {
245 case 'B':
246 val_cluster/=itSenses->senseMembersIds().size();
247 break;
248 case 'C' :
249 val_cluster/=log(1+itSenses->senseMembersIds().size());
250 break;
251 case 'D' :
252 val_cluster/=log(10+itSenses->senseMembersIds().size());
253 break;
254 default:
255 break;
256 }
257 if(val_cluster>max) {
258 max=val_cluster;
259 bestSense=itSenses;
260 }
261 scoresByClass[itSenses->senseId()]=val_cluster;
262 LDEBUG << "SCORES : " << itSenses->senseId() << " -> " << val_cluster;
263 }
264
265 // cerr << "THRESCUT2 : "<< thresCut << endl;
266 for(map<int, double >::iterator itSenses = scoresByClass.begin();
267 itSenses!= scoresByClass.end();
268 itSenses++ ) {
269 LDEBUG << wu.lemma() << " : BEFORE ASSIGN : " << scoresByClass[itSenses->first] << " - " << thres*max ;
270 if (scoresByClass[itSenses->first]>=(thres*max)
271 // && scoresByClass[itcl->first]>= thresCut
272 && max!=0) {
273 assignedClasses[itSenses->first]=scoresByClass[itSenses->first];
274 }
275 }
276#endif
277 return bestSense;
278}
279
280
282 const NNList& knns,
283 const SemanticContext::const_iterator itContext,
284
285 map<int, map<string, double> >& cvListRep)
286{
287#ifndef WIN32
288 LOGINIT("WordSenseDisambiguator");
289 map<int, double> simByCluster;
290 int maxDist = 0;
291 if (knns.size()>0)
292 {
293 maxDist = knns.rbegin()->second;
294 }
295 // somme des distances des ppv du mot dans 'id_relation' qui ont été attribués au cluster 'id_cluster'
296 for(set<WordSenseUnit>::iterator itSenses = wu.wordSensesUnits().begin();
297 itSenses!= wu.wordSensesUnits().end();
298 itSenses++ )
299 {
300 double dist;
301 double sim_cluster=0;
302 LDEBUG << itSenses->senseTag() << " : " << itSenses->senseMembersIds().size() ;
303 for(set<uint64_t>::iterator itMembers =itSenses->senseMembersIds().begin() ;
304 itMembers!=itSenses->senseMembersIds().end() ;
305 itMembers++ ) {
306 LDEBUG << "itMembers "<< *itMembers;
307 if (knns.find(*itMembers)!=knns.end()) {
308 LDEBUG << "Dist "<< knns.find(*itMembers)->second;
309 dist = knns.find(*itMembers)->second;
310 double distCos = 1-cos(dist*M_PI/16384);
311 if (distCos!=0.) {
312 //sim_cluster += (1 / (dist*dist));
313 sim_cluster += (1 / (distCos*distCos));
314 }
315 }
316 }
317
318
319 simByCluster[itSenses->senseId()]=sim_cluster*maxDist*maxDist;
320 LDEBUG << sim_cluster << "*" << maxDist <<" = " <<sim_cluster*maxDist*maxDist;
321 LDEBUG << wu.lemmaId() << " : "<< wu.lemma() << " : similarities : " << itContext->first << " " << itSenses->senseId() << ":" << itSenses->senseTag() << " -> " << simByCluster[itSenses->senseId()];
322 }
323 //somme des dist totales des clusters de la représentation
324 double sim_clusters_rep = 0;
325 for (map<int, double>::iterator itdist = simByCluster.begin(); itdist != simByCluster.end(); itdist++) {
326 sim_clusters_rep+=itdist->second;
327 }
328 // compute cv
329 if( sim_clusters_rep != 0. ){
330 for(set<WordSenseUnit>::iterator itSenses = wu.wordSensesUnits().begin();
331 itSenses!= wu.wordSensesUnits().end();
332 itSenses++ ) {
333 LDEBUG << ": Key : " << wu.lemma() ;
334 cvListRep[itSenses->senseId()][itContext->first]=( simByCluster[itSenses->senseId()] / sim_clusters_rep );
335 /*TODECIDE
336 if (itSenses->senseMembersIds().find(wu.lemmaId())!=itSenses->senseMembersIds().end()) {
337 cvListRep[itSenses->senseId()][*itContext]=( simByCluster[itSenses->senseId()] / sim_clusters_rep );
338 // / (itcl->second.size()-1);
339 } else {
340 cvListRep[itSenses->senseId][*itContext]=( simByCluster[itSenses->senseId] / sim_clusters_rep );
341 // / (itcl->second.size());
342 }
343 // *10/ (10+ log(itcl->second.size()));
344 */
345 stringstream sscvlist ;
346 sscvlist << " cvlistrep "<< itSenses->senseId() << " "<< itContext->first <<" : " << simByCluster[itSenses->senseId()] <<"/"<< sim_clusters_rep << "=" << ( simByCluster[itSenses->senseId()] / sim_clusters_rep )<< endl;
347 LDEBUG << sscvlist.str();
348 }
349 } else {
350 // if no intersection between knn of relation and verbes in classes
351 LWARN << "No intersection between " << itContext->first
352 << " and elements in classes for " << wu.lemmaId();
353 }
354 return simByCluster.size();
355#else
356 return 0;
357#endif
358}
359
360
361
362
363
364
365
366
367double WordSenseAnnotation::cvCluster(const map<int, map<string, double> >& cvListRep,
368 const string& relation,
369 int id_cluster)
370{
371 if (cvListRep.find(id_cluster) != cvListRep.end()
372 && cvListRep.find(id_cluster)->second.find(relation)
373 !=cvListRep.find(id_cluster)->second.end())
374 {
375 return cvListRep.find(id_cluster)->second.find(relation)->second;
376 }
377 return 0;
378}
379
380double WordSenseAnnotation::sumCvCluster(map<int, map<string, double> >& cvListRep, string relation)
381{
382 double sum_cv_rep = 0.;
383 for(map<int, map<string, double> >::iterator itcv = cvListRep.begin(); itcv !=cvListRep.end(); itcv++)
384 {
385 if (itcv->second.find(relation)!=itcv->second.end())
386 {
387 sum_cv_rep += itcv->second[relation];
388 }
389 }
390 return sum_cv_rep;
391}
392
393//(F)
394double WordSenseAnnotation::sumLogCv(const map<int, map<string, double> >& cvListRep,
395 const string& relation)
396{
397 double sum = 0.;
398 for(map<int, map<string, double> >::const_iterator itcv = cvListRep.begin();
399 itcv !=cvListRep.end();
400 itcv++)
401 {
402 if (itcv->second.find(relation)!=itcv->second.end()
403 && itcv->second.find(relation)->second!= 0 )
404 {
405 //stringstream ssquicksum;
406 sum += itcv->second.find(relation)->second * log( itcv->second.find(relation)->second );
407 /*
408 ssquicksum << "QuickSum : " << itcv->second.find(relation)->second <<"*"<< log( itcv->second.find(relation)->second )
409 << " = " << itcv->second.find(relation)->second * log( itcv->second.find(relation)->second ) << endl
410 << "from : " << itcv->second.find(relation)->second << " (" << relation << ")" << endl;
411
412 cout << ssquicksum.str();
413 */
414 }
415 }
416 return (sum+1);
417}
418
419//(G)
420double WordSenseAnnotation::sumLogCvList(const map<int, map<string, double> >& cvListRep)
421{
422 double sum_log = 0.;
423 if (cvListRep.size()==0)
424 {
425 return sum_log;
426 }
427 for(map<string, double>::const_iterator itrel = cvListRep.begin()->second.begin();
428 itrel!= cvListRep.begin()->second.end();
429 itrel++) {
430 sum_log += (double)sumLogCv(cvListRep, itrel->first);
431 }
432
433 return sum_log;
434}
435
436/* end internal computation functions */
437
438void WordSenseAnnotation::retrieveSense(set<WordSenseUnit>::iterator itSenses)
439{
440 cerr << itSenses->senseId() << " : " << itSenses->senseTag() << endl;
441 m_wsu=*itSenses;
442 m_senseTag = itSenses->senseTag();
443 cerr << m_wsu.senseId() << " : " << m_wsu.senseTag() << endl;
444 cerr << "TEST1 : " << senseId() << " : " << senseTag() << endl;
445}
446
447
450 ) const
451{
452// LOGINIT("WordSenseDisambiguator");
453
454 // can have multiple annotations in case of mapping
456 GenericAnnotation ga(*this);
457
459 ad->annotate(morphVertex(), utf8stdstring2limastring("WordSense"), ga);
460
461 return AnnotationGraphVertex(); //unused;
462}
463
464
465void WordSenseAnnotation::outputXml(std::ostream& xmlStream,const LinguisticGraph& g) const
466{
467// LOGINIT("WordSenseDisambiguator");
468
469 xmlStream << "<WORDSENSE SENSEID=\"" << senseId() << "\" SENSETAG=\"" << senseTag() << "\" MODE=\"" << mode() << "\" CONFIDENCE=\"" << confidence() << "\">";
470
471
472 Token* token = get(vertex_token, g, morphVertex());
473 if (token != 0)
474 {
475 xmlStream << limastring2utf8stdstring(token->stringForm());
476 }
477 xmlStream << "</WORDSENSE>";
478}
479
480std::ostream& operator << (std::ostream& os, const WordSenseAnnotation& wsa)
481{
482 os << wsa.senseId() << "(" << wsa.mode() << "):" << wsa.senseTag() << " - " << wsa.morphVertex();
483 return os;
484}
485} // closing namespace WordSenseDisambiguation
486} // closing namespace LinguisticProcessing
487} // closing namespace Lima
This file is the main header file for the data related to annotation graphs.
#define LWARN
Definition LimaCommon.h:160
#define LOGINIT(X)
Definition LimaCommon.h:187
#define LDEBUG
Definition LimaCommon.h:157
#define LERROR
Definition LimaCommon.h:161
#define PROCESSORSLOGINIT
Definition LimaCommon.h:211
@ vertex_token
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
Data used for the syntactic analyzis of texts.
Holds an annotation graph and gives an API to manipulate it.
This class allows to convert any object into an annotation by inheritance.
virtual int dump(std::ostream &os, Common::AnnotationGraphs::GenericAnnotation &ga) const override
int getKNN(uint64_t lemmaId, SemanticContext::const_iterator itContext, NNList &knns)
bool disambiguate(const WordUnit &wu)
main functions of the global algorithm (called by WordSenseDisambiguator)
AnnotationGraphVertex writeAnnotation(Common::AnnotationGraphs::AnnotationData *ad) const
double sumCvCluster(std::map< int, std::map< std::string, double > > &cvListRep, std::string relation)
sum_cv_cluster_rep: @cv_list_rep: la matrice 'cv_list_rep' (voir fonction précédente) @nb_clusters: n...
double sumLogCvList(const std::map< int, std::map< std::string, double > > &cvListRep)
double sumLogCv(const std::map< int, std::map< std::string, double > > &cvListRep, const std::string &relation)
float classifKnnMRD(KnnSearcher *searcher, const WordUnit &wu, const SemanticContext &context, double thres, char method)
int computeConfidenceVector(const WordUnit &wu, const NNList &knns, const SemanticContext::const_iterator itContext, std::map< int, std::map< std::string, double > > &cvListRep)
std::set< WordSenseUnit >::iterator classify(const WordUnit &wu, const SemanticContext &context, const std::map< int, std::map< std::string, double > > &cvListRep, const double thres, const char method)
double cvCluster(const std::map< int, std::map< std::string, double > > &cvListRep, const std::string &relation, int id_cluster)
cv_cluster_rep: @cv_list_rep: la matrice 'cv_list_rep' (voir fonction précédente) @nb_clusters: nombr...
virtual void outputXml(std::ostream &xmlStream, const LinguisticGraph &g) const
const std::set< WordSenseUnit > & wordSensesUnits() const
Definition WordUnit.h:83
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
void annotate(AnnotationGraphVertex v, const LimaString &annot, uint64_t value)
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
LimaString utf8stdstring2limastring(const std::string &src)
std::map< std::string, std::set< uint64_t > > SemanticContext
std::ostream & operator<<(std::ostream &os, const WordSenseAnnotation &wsa)
NAUTITIA.
@ UNKNOWN_ERROR
Definition LimaCommon.h:238
@ SUCCESS_ID
Definition LimaCommon.h:237
STL namespace.