35namespace LinguisticProcessing
37namespace WordSenseDisambiguation
50 catch (
const boost::bad_any_cast& e)
52 LERROR <<
"This annotation is not a WordSenseAnnotation ; nothing dumped";
72 LOGINIT(
"WordSenseDisambiguator");
82 for (set<WordSenseUnit>::iterator itSenses = wu.
wordSensesUnits().begin();
86 if (itSenses->senseId()==0 &&
m_mode == itSenses->mode())
98 LWARN <<
"No Disambiguation processing. Bad configuration";
116 LOGINIT(
"WordSenseDisambiguator");
133 LWARN <<
"No Disambiguation processing. Bad configuration";
150 LOGINIT(
"WordSenseDisambiguator");
152 map<int, map<string, double> > cvListRep;
154 for (SemanticContext::const_iterator itContext = context.begin();
155 itContext!= context.end();
158 LDEBUG <<
" Computing knn dists for " << itContext->first ;
162 for (NNList::iterator itNN = knns.begin();
166 LDEBUG << itNN->first <<
":" << itNN->second;
168 if (knns.size()==0 || knns.rbegin()->second==0) {
170 LWARN << wu.
lemmaId() <<
" : " << wu.
lemma() <<
" has a corrupted sphere. Process may return biased results ";
172 for (NNList::iterator itNN = knns.begin();
175 LDEBUG << itNN->first <<
":" << itNN->second;
185 std::set<WordSenseUnit>::iterator bestSense =
classify(wu, context, cvListRep, thres, method);
195 const map<
int, map<string, double> >& cvListRep,
199 LOGINIT(
"WordSenseDisambiguator");
202 map<int, double> assignedClasses;
203 map<int, double> scoresByClass;
205 set<WordSenseUnit>::iterator bestSense;
207 for(set<WordSenseUnit>::iterator itSenses = wu.
wordSensesUnits().begin();
210 LDEBUG <<
"cluster : " << itSenses->senseId();
218 for (map<
int, map<string, double> >::const_iterator itdeb = cvListRep.begin();
219 itdeb!= cvListRep.end();
221 LDEBUG <<
"cvlistrepsize : " << itdeb->first <<
" : " << itdeb->second.size();
222 for (map<string, double>::const_iterator itdeb2 = itdeb->second.begin();
223 itdeb2 !=itdeb->second.end();
225 LDEBUG << itdeb2->first <<
" -> " << itdeb2->second;
229 for (SemanticContext::const_iterator itContext = context.begin();
230 itContext!= context.begin();
232 if (cvListRep.find(itSenses->senseId())!=cvListRep.end()
233 && cvListRep.find(itSenses->senseId())->second.find(itContext->first)
234 ==cvListRep.find(itSenses->senseId())->second.end() ) {
236 LDEBUG << itContext->first <<
" not found for cluster "<< itSenses->senseId();
238 w =
sumLogCv(cvListRep, itContext->first) / entropySum;
239 LDEBUG << itContext->first <<
" "<< itSenses->senseId() <<
" w= " <<
sumLogCv(cvListRep, itContext->first) <<
"/" << entropySum <<
" = " << w ;
241 val_cluster += w *
cvCluster(cvListRep, itContext->first, itSenses->senseId());
246 val_cluster/=itSenses->senseMembersIds().size();
249 val_cluster/=log(1+itSenses->senseMembersIds().size());
252 val_cluster/=log(10+itSenses->senseMembersIds().size());
257 if(val_cluster>max) {
261 scoresByClass[itSenses->senseId()]=val_cluster;
262 LDEBUG <<
"SCORES : " << itSenses->senseId() <<
" -> " << val_cluster;
266 for(map<int, double >::iterator itSenses = scoresByClass.begin();
267 itSenses!= scoresByClass.end();
269 LDEBUG << wu.
lemma() <<
" : BEFORE ASSIGN : " << scoresByClass[itSenses->first] <<
" - " << thres*max ;
270 if (scoresByClass[itSenses->first]>=(thres*max)
273 assignedClasses[itSenses->first]=scoresByClass[itSenses->first];
283 const SemanticContext::const_iterator itContext,
285 map<
int, map<string, double> >& cvListRep)
288 LOGINIT(
"WordSenseDisambiguator");
289 map<int, double> simByCluster;
293 maxDist = knns.rbegin()->second;
296 for(set<WordSenseUnit>::iterator itSenses = wu.
wordSensesUnits().begin();
301 double sim_cluster=0;
302 LDEBUG << itSenses->senseTag() <<
" : " << itSenses->senseMembersIds().size() ;
303 for(set<uint64_t>::iterator itMembers =itSenses->senseMembersIds().begin() ;
304 itMembers!=itSenses->senseMembersIds().end() ;
306 LDEBUG <<
"itMembers "<< *itMembers;
307 if (knns.find(*itMembers)!=knns.end()) {
308 LDEBUG <<
"Dist "<< knns.find(*itMembers)->second;
309 dist = knns.find(*itMembers)->second;
310 double distCos = 1-cos(dist*M_PI/16384);
313 sim_cluster += (1 / (distCos*distCos));
319 simByCluster[itSenses->senseId()]=sim_cluster*maxDist*maxDist;
320 LDEBUG << sim_cluster <<
"*" << maxDist <<
" = " <<sim_cluster*maxDist*maxDist;
321 LDEBUG << wu.
lemmaId() <<
" : "<< wu.
lemma() <<
" : similarities : " << itContext->first <<
" " << itSenses->senseId() <<
":" << itSenses->senseTag() <<
" -> " << simByCluster[itSenses->senseId()];
324 double sim_clusters_rep = 0;
325 for (map<int, double>::iterator itdist = simByCluster.begin(); itdist != simByCluster.end(); itdist++) {
326 sim_clusters_rep+=itdist->second;
329 if( sim_clusters_rep != 0. ){
330 for(set<WordSenseUnit>::iterator itSenses = wu.
wordSensesUnits().begin();
334 cvListRep[itSenses->senseId()][itContext->first]=( simByCluster[itSenses->senseId()] / sim_clusters_rep );
345 stringstream sscvlist ;
346 sscvlist <<
" cvlistrep "<< itSenses->senseId() <<
" "<< itContext->first <<
" : " << simByCluster[itSenses->senseId()] <<
"/"<< sim_clusters_rep <<
"=" << ( simByCluster[itSenses->senseId()] / sim_clusters_rep )<< endl;
351 LWARN <<
"No intersection between " << itContext->first
352 <<
" and elements in classes for " << wu.
lemmaId();
354 return simByCluster.size();
368 const string& relation,
371 if (cvListRep.find(id_cluster) != cvListRep.end()
372 && cvListRep.find(id_cluster)->second.find(relation)
373 !=cvListRep.find(id_cluster)->second.end())
375 return cvListRep.find(id_cluster)->second.find(relation)->second;
382 double sum_cv_rep = 0.;
383 for(map<
int, map<string, double> >::iterator itcv = cvListRep.begin(); itcv !=cvListRep.end(); itcv++)
385 if (itcv->second.find(relation)!=itcv->second.end())
387 sum_cv_rep += itcv->second[relation];
395 const string& relation)
398 for(map<
int, map<string, double> >::const_iterator itcv = cvListRep.begin();
399 itcv !=cvListRep.end();
402 if (itcv->second.find(relation)!=itcv->second.end()
403 && itcv->second.find(relation)->second!= 0 )
406 sum += itcv->second.find(relation)->second * log( itcv->second.find(relation)->second );
423 if (cvListRep.size()==0)
427 for(map<string, double>::const_iterator itrel = cvListRep.begin()->second.begin();
428 itrel!= cvListRep.begin()->second.end();
430 sum_log += (double)
sumLogCv(cvListRep, itrel->first);
440 cerr << itSenses->senseId() <<
" : " << itSenses->senseTag() << endl;
469 xmlStream <<
"<WORDSENSE SENSEID=\"" <<
senseId() <<
"\" SENSETAG=\"" <<
senseTag() <<
"\" MODE=\"" <<
mode() <<
"\" CONFIDENCE=\"" <<
confidence() <<
"\">";
477 xmlStream <<
"</WORDSENSE>";
This file is the main header file for the data related to annotation graphs.
#define PROCESSORSLOGINIT
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
Data used for the syntactic analyzis of texts.
Holds an annotation graph and gives an API to manipulate it.
This class allows to convert any object into an annotation by inheritance.
holds surface data of a token
const LimaString & stringForm() const
virtual int dump(std::ostream &os, Common::AnnotationGraphs::GenericAnnotation &ga) const override
int getKNN(uint64_t lemmaId, SemanticContext::const_iterator itContext, NNList &knns)
bool disambiguate(const WordUnit &wu)
main functions of the global algorithm (called by WordSenseDisambiguator)
AnnotationGraphVertex writeAnnotation(Common::AnnotationGraphs::AnnotationData *ad) const
double sumCvCluster(std::map< int, std::map< std::string, double > > &cvListRep, std::string relation)
sum_cv_cluster_rep: @cv_list_rep: la matrice 'cv_list_rep' (voir fonction précédente) @nb_clusters: n...
void retrieveSense(std::set< WordSenseUnit >::iterator itSenses)
double sumLogCvList(const std::map< int, std::map< std::string, double > > &cvListRep)
double sumLogCv(const std::map< int, std::map< std::string, double > > &cvListRep, const std::string &relation)
float classifKnnMRD(KnnSearcher *searcher, const WordUnit &wu, const SemanticContext &context, double thres, char method)
int computeConfidenceVector(const WordUnit &wu, const NNList &knns, const SemanticContext::const_iterator itContext, std::map< int, std::map< std::string, double > > &cvListRep)
bool isDisambiguated() const
general test functions
std::set< WordSenseUnit >::iterator classify(const WordUnit &wu, const SemanticContext &context, const std::map< int, std::map< std::string, double > > &cvListRep, const double thres, const char method)
LinguisticGraphVertex morphVertex()
double cvCluster(const std::map< int, std::map< std::string, double > > &cvListRep, const std::string &relation, int id_cluster)
cv_cluster_rep: @cv_list_rep: la matrice 'cv_list_rep' (voir fonction précédente) @nb_clusters: nombr...
virtual void outputXml(std::ostream &xmlStream, const LinguisticGraph &g) const
const std::string & lemma() const
const std::set< WordSenseUnit > & wordSensesUnits() const
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
void annotate(AnnotationGraphVertex v, const LimaString &annot, uint64_t value)
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
LimaString utf8stdstring2limastring(const std::string &src)
@ B_ROMANSEVAL_MOST_FREQUENT
std::map< std::string, std::set< uint64_t > > SemanticContext
std::ostream & operator<<(std::ostream &os, const WordSenseAnnotation &wsa)
std::map< uint64_t, float > NNList