LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
corefSolver.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6#include "corefSolver.h"
8
19#include <stack>
20#include <iostream>
21#include <math.h>
22
23using namespace std;
24//using namespace boost;
25using namespace Lima::Common::Misc;
26using namespace Lima::Common::MediaticData;
27using namespace Lima::Common::AnnotationGraphs;
30
31
32namespace Lima
33{
34namespace LinguisticProcessing
35{
36namespace Coreferences
37{
39
40
41
43 m_language()
44{}
45
46
49 Manager* manager)
50
51{
54 try {
55 m_scope=atoi(unitConfiguration.getParamsValueAtKey("scope").c_str());
56 }
58 {
59 LERROR << "No 'scope' defined in "<<unitConfiguration.getName()<<" configuration group for language " << (int)m_language;
60 m_scope = 3;
61 LERROR << "Scope is set to 3 by default.";
62 }
63 try {
64 m_threshold=atoi(unitConfiguration.getParamsValueAtKey("threshold").c_str());
65 }
67 {
68 LERROR << "No 'threshold' defined in "<<unitConfiguration.getName()<<" configuration group for language " << (int)m_language;
69 m_threshold = 70;
70 LERROR << "Threshold is set to 130 by default.";
71 }
72 try {
73 m_resolveDefinites=atoi(unitConfiguration.getParamsValueAtKey("Resolve Definites").c_str());
74 m_resolveN3PPronouns=atoi(unitConfiguration.getParamsValueAtKey("Resolve non third person pronouns").c_str());
75 }
77 {
78 LERROR << "Please define 'Resolve Definites' and 'Resolve non third person pronouns' in "<<unitConfiguration.getName()<<" configuration group for language " << (int)m_language;
81 LERROR << "Resolve Definites is set to true (1) by default.";
82 LERROR << "Resolve non third person pronouns is set to false (0) by default.";
83 }
84 cerr << m_language << endl;
85 const Common::PropertyCode::PropertyManager& macroManager=static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager().getPropertyManager("MACRO");
86 const Common::PropertyCode::PropertyManager& microManager=static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager().getPropertyManager("MICRO");
88 m_microAccessor=&static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager().getPropertyAccessor("MICRO");
89 m_genderAccessor=&(static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager().getPropertyAccessor("GENDER"));
90 m_personAccessor=&(static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager().getPropertyAccessor("PERSON"));
91 m_numberAccessor=&(static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager().getPropertyAccessor("NUMBER"));
94
95 std::map<string, string> tmpMap = unitConfiguration.getMapAtKey("MacroCategories");
96 for (map<string, string>::const_iterator it = tmpMap.begin(); it!=tmpMap.end(); it++)
97 m_tagLocalDef.insert(make_pair((*it).first,macroManager.getPropertyValue((*it).second)));
98 try {
99 deque<string> tmpDeque = unitConfiguration.getListsValueAtKey("LexicalAnaphora");
100 for (deque<string>::const_iterator it = tmpDeque.begin(); it!=tmpDeque.end(); it++)
101 {
102 m_reflexiveReciprocal.insert(microManager.getPropertyValue((*it)));
103 }
104 tmpDeque.clear();
105 tmpDeque =
106 unitConfiguration.getListsValueAtKey("UndefinitePronouns");
107 for (deque<string>::const_iterator it = tmpDeque.begin(); it!=tmpDeque.end(); it++)
108 {
109 m_undefPronouns.insert(microManager.getPropertyValue((*it)));
110 }
111 tmpDeque.clear();
112 tmpDeque =
113 unitConfiguration.getListsValueAtKey("PossessivePronouns");
114 for (deque<string>::const_iterator it = tmpDeque.begin(); it!=tmpDeque.end(); it++)
115 {
116 m_undefPronouns.insert(microManager.getPropertyValue((*it)));
117 }
118 m_conjCoord = microManager.getPropertyValue("CC");
119 }
121 {
122 LERROR << "One of the tags list is not defined in "<<unitConfiguration.getName()<<" configuration group for language " << (int)m_language;
123 LERROR << "Please check all of the following categories:";
124 LERROR << "'LexicalAnaphora', 'UndefinitePronouns', 'PossessivePronouns'.";
125 }
126
127 try {
128 m_relLocalDef.insert(make_pair("PrepRelation",unitConfiguration.getListsValueAtKey("PrepRelation")));
129 m_relLocalDef.insert(make_pair("PleonasticRelation",unitConfiguration.getListsValueAtKey("PleonasticRelation")));
130 m_relLocalDef.insert(make_pair("DefiniteRelation",unitConfiguration.getListsValueAtKey("DefiniteRelation")));
131 m_relLocalDef.insert(make_pair("SubjectRelation",unitConfiguration.getListsValueAtKey("SubjectRelation")));
132 m_relLocalDef.insert(make_pair("AttributeRelation",unitConfiguration.getListsValueAtKey("AttributeRelation")));
133 m_relLocalDef.insert(make_pair("CODRelation",unitConfiguration.getListsValueAtKey("CODRelation")));
134 m_relLocalDef.insert(make_pair("COIRelation",unitConfiguration.getListsValueAtKey("COIRelation")));
135 m_relLocalDef.insert(make_pair("AdjunctRelation",unitConfiguration.getListsValueAtKey("AdjunctRelation")));
136 m_relLocalDef.insert(make_pair("AgentRelation",unitConfiguration.getListsValueAtKey("AgentRelation")));
137 m_relLocalDef.insert(make_pair("NPDeterminerRelation",unitConfiguration.getListsValueAtKey("NPDeterminerRelation")));
138 }
140 {
141 LERROR << "One of the macro relation is not defined in "<<unitConfiguration.getName()<<" configuration group for language " << (int)m_language;
142 LERROR << "Please check all of the following relations:";
143 LERROR << "'PrepRelation', 'PleonasticRelation', 'DefiniteRelation', 'SubjectRelation', 'AttributeRelation', 'CODRelation', 'COIRelation', 'AdjunctRelation', 'AgentRelation', 'NPDeterminerRelation'.";
144 }
145
146
147 // Salience factors
148 try
149 {
150 const std::map<std::string,std::string>& salience=unitConfiguration.getMapAtKey("SalienceFactors");
151 for (std::map<std::string,std::string>::const_iterator it=salience.begin();
152 it!=salience.end();
153 it++)
154 {
155 m_salienceWeights[it->first]=QString::fromUtf8(it->second.c_str()).toDouble();
156 }
157 }
159 {
160 LERROR << "No map 'SalienceFactors' in "<<unitConfiguration.getName()<<" configuration group for language " << (int)m_language;
161// throw InvalidConfiguration();
162 }
163
164 // Slot Values
165 try
166 {
167 const std::map<std::string,std::string>& slotValuesStr=unitConfiguration.getMapAtKey("SlotValues");
168 for (std::map<std::string,std::string>::const_iterator it=slotValuesStr.begin();
169 it!=slotValuesStr.end();
170 it++)
171 {
172 m_slotValues[it->first]=atoi((it->second).c_str());
173 }
174 }
176 {
177 LERROR << "No map 'SlotValues' in "<<unitConfiguration.getName()<<" configuration group for language " << (int)m_language;
178// throw InvalidConfiguration();
179 }
180
181}
182
189 AnalysisContent& analysis) const
190{
191#ifdef DEBUG_LP
194 LINFO << "start CorefSolver";
195#endif
196 // create syntacticData
197 auto posgraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("PosGraph"));
198 if (posgraph==0)
199 {
201 LERROR << "no PosGraph ! abort";
202 return MISSING_DATA;
203 }
204 auto sb = std::dynamic_pointer_cast<SegmentationData>(analysis.getData("SentenceBoundaries"));
205 if (sb==0)
206 {
208 LERROR << "no sentence bounds ! abort";
209 return MISSING_DATA;
210 }
211 if (sb->getGraphId() != "PosGraph")
212 {
214 LERROR << "SentenceBounds computed on graph '" << sb->getGraphId() << "'. CorefSolver needs " <<
215 "sentence bounds on PosGraph";
217 }
218 auto syntacticData = std::dynamic_pointer_cast<SyntacticData>(analysis.getData("SyntacticData"));
219 if (sb==0)
220 {
222 LERROR << "no syntactic data ! abort";
223 return MISSING_DATA;
224 }
225
226
228 auto annotationData = std::dynamic_pointer_cast<AnnotationData>(analysis.getData("AnnotationData"));
229 if (annotationData == 0)
230 {
231 annotationData = std::make_shared<AnnotationData>();
235 if (std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph")) != 0)
236 {
237 std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"))->populateAnnotationGraph(
238 annotationData.get(), "AnalysisGraph");
239 }
240 if (std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("PosGraph")) != 0)
241 {
242
243 std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("PosGraph"))->populateAnnotationGraph(
244 annotationData.get(), "PosGraph");
245 }
246
247 analysis.setData("AnnotationData",annotationData);
248 }
249
250
252// trop long !!
253// LinguisticGraphVertexIt itg, itg_end;
254// boost::tie(itg, itg_end) = vertices(*anagraph->getGraph());
255// for (; itg != itg_end; itg++)
256// {
257// LinguisticGraph graph =*anagraph->getGraph();
258// LinguisticAnalysisStructure::Token* token=get(vertex_token, graph, *itg);
259// if (token != 0)
260// {
261// LimaString mot = token->stringForm();
262// annotationData->annotate(*itg, Common::Misc::utf8stdstring2limastring("Mot"),mot);
263// }
264// }
265
266
267
272 if (annotationData->dumpFunction("Coreferent") == 0)
273 {
274 annotationData->dumpFunction("Coreferent", new DumpCoreferent(annotationData.get()));
275 }
276
277
278 /* npVertices and npCandidates are respectively deques of vectors and sets.
279 The number of elements of these deques matches the size of the scope (ie. number of sentences in which the antecedents can be found) entered as a parameter.
280 Thus each of the internal vectors (resp. sets) corresponds to one and only one sentence. */
281 std::deque<std::vector<CoreferentAnnotation> > npVertices;
282 std::deque<Vertices>* npCandidates = new std::deque<Vertices>();
283
284
285 LinguisticGraph* graph=posgraph->getGraph();
286 uint64_t nextReferentId = 1;
287 std::set< LinguisticGraphVertex > alreadyProcessedVertices;
288
289 //LinguisticGraphVertex beginSentence=sb->getStartVertex();
290 // for each sentence
291 // ??OME2 for (SegmentationData::const_iterator boundItr=sb->begin();
292 // boundItr!=sb->end();
293 for (std::vector<Segment>::const_iterator boundItr=(sb->getSegments()).begin();
294 boundItr!=(sb->getSegments()).end();
295 boundItr++)
296 {
297 LinguisticGraphVertex beginSentence=boundItr->getFirstVertex();
298 LinguisticGraphVertex endSentence=boundItr->getLastVertex();
299
300 // non pleonastic pronouns (or definite NP)
301 Vertices* npAnaphora = new Vertices();
302 // Ruled Out Binding by syntacticFilter
303 VerticesRelation* roBinding = new VerticesRelation();
304 // lexical Anaphora Binding
306 // potentialBinding
308
309 // add entry in npVertices and npCandidates
310 npVertices.push_front(std::vector<CoreferentAnnotation>());
311 npCandidates->push_front(Vertices());
312 // if size>scope, remove last
313 if (npVertices.size()>m_scope)
314 npVertices.pop_back();
315 if (npCandidates->size()>m_scope)
316 npCandidates->pop_back();
317 //degradate the existing salience factors
318 for (std::deque<Vertices>::iterator itm = npCandidates->begin();
319 itm != npCandidates->end( );
320 itm++)
321 {
322 for (Vertices::iterator candidateItr = (*itm).begin( );
323 candidateItr != (*itm).end( );
324 candidateItr++ )
325 {
326 (*candidateItr)->salience((*candidateItr)->salience()/2);
327 }
328 }
329
330#ifdef DEBUG_LP
331 LDEBUG << "analyze sentence from vertex " << beginSentence << " to vertex " << endSentence;
332#endif
333
334
335 // for each word in the sentence
336 LinguisticGraphVertex v=beginSentence;
337 LinguisticGraphVertex savedSb=beginSentence;
338 while (v!=endSentence)
339 {
340 if (alreadyProcessedVertices.find(v) != alreadyProcessedVertices.end())
341 continue;
342 //else alreadyProcessedVertices.insert(v);
343 CoreferentAnnotation ca(nextReferentId,v);
344 if (ca.isIncludedInNounPhrase(graph, m_language, posgraph.get(), analysis, m_inNpCategs, m_microAccessor) )
345 {
346 (*npVertices.begin()).push_back(ca);
347 alreadyProcessedVertices.insert(v);
348 nextReferentId++;
349 }
350 beginSentence=endSentence;
351 LinguisticGraphOutEdgeIt ite, ite_end;
352 boost::tie(ite, ite_end)=boost::out_edges(v, *graph);
353 v=target(*ite, *graph);
354
355 }/*
356TimeUtils::logElapsedTime("retrieve nps");
357TimeUtils::updateCurrentTime();*/
358
359#ifdef DEBUG_LP
360 LDEBUG << "classify NPs and calculate salience weights";
361#endif
362 for(vector<CoreferentAnnotation>::iterator itca = (*npVertices.begin()).begin();
363 itca!=(*npVertices.begin()).end();
364 itca++)
365 {
366 (*itca).newerRef(&(*itca));
367 int classif = (*itca).classify(graph, syntacticData.get(), m_macroAccessor, m_microAccessor, m_tagLocalDef,
370 posgraph.get(), m_language);
371 if (classif == 1 || classif == 11)
372 {
373 npAnaphora->insert(&(*itca));
374 }
375 if (classif == 10 || classif == 11)
376 {
377 (*npCandidates->begin()).insert(&(*itca));
378 }
379
380 (*itca).salience((*itca).salienceWeighting(m_salienceWeights, syntacticData.get(), m_macroAccessor,posgraph.get(),
381 m_language, m_tagLocalDef, m_relLocalDef, analysis, savedSb, endSentence));
382 }
383
384// TimeUtils::logElapsedTime("classify");
385// TimeUtils::updateCurrentTime();
386
387#ifdef DEBUG_LP
388 LDEBUG<< "Anaphora:";
389 for (Vertices::iterator anaphItr=npAnaphora->begin();
390 anaphItr!=npAnaphora->end();
391 anaphItr++)
392 LDEBUG<<(*anaphItr)->morphVertex()<< " - " << limastring2utf8stdstring(get(vertex_token, *graph, (*anaphItr)->morphVertex())->stringForm()) <<(*anaphItr)->newerRef()->morphVertex();
393 LDEBUG<< " - ";
394
395 LDEBUG<< "Candidates:";
396#endif
397 for (std::deque<Vertices>::iterator itm = npCandidates->begin( );
398 itm != npCandidates->end( );
399 itm++ )
400#ifdef DEBUG_LP
401 for (Vertices::iterator itc = (*itm).begin( );
402 itc != (*itm).end( );
403 itc++ )
404 LDEBUG<< " - " << limastring2utf8stdstring(get(vertex_token, *graph, (*itc)->morphVertex())->stringForm()) << ":" << (*itc)->salience();
405 LDEBUG<< " - ";
406
407 LDEBUG << "initialize syntactic filter";
408#endif
409 initSyntacticFilter(analysis, posgraph.get(), syntacticData.get(), npAnaphora, npCandidates, roBinding);
410
411// TimeUtils::logElapsedTime("initialize syntactic filter");
412// TimeUtils::updateCurrentTime();
413
414#ifdef DEBUG_LP
415 LDEBUG<< "RuledOutBinding:";
416 for (VerticesRelation::iterator itro = roBinding->begin( );
417 itro != roBinding->end( );
418 itro++ )
419 {
420 LDEBUG<< " Anaphora: " << limastring2utf8stdstring(get(vertex_token, *graph, (*itro).first->morphVertex())->stringForm());
421 for (std::set<CoreferentAnnotation*>::iterator its = (*itro).second.begin( );
422 its != (*itro).second.end( );
423 its++ )
424 LDEBUG << " - " << limastring2utf8stdstring(get(vertex_token, *graph, (*its)->morphVertex())->stringForm()) << " - ";
425 }
426 LDEBUG;
427
428 LDEBUG << "initialize lexical anaphora binding";
429#endif
430 bindingLexicalAnaphora(analysis,posgraph.get(), syntacticData.get(), npAnaphora, npCandidates, lexAnaBinding);
431
432// TimeUtils::logElapsedTime("initialize lex ana binding");
433// TimeUtils::updateCurrentTime();
434
435#ifdef DEBUG_LP
436 LDEBUG<< "Lexical Anaphora Binding:";
437 for (WeightedVerticesRelation::iterator itp = lexAnaBinding->begin( );
438 itp != lexAnaBinding->end( );
439 itp++ )
440 {
441 LDEBUG<< "Anaphora: " << limastring2utf8stdstring(get(vertex_token, *graph, (*itp).first->morphVertex())->stringForm());
442 for (std::map<CoreferentAnnotation*, float>::iterator its = (*itp).second.begin( );
443 its != (*itp).second.end( );
444 its++ )
445 LDEBUG <<(*its).first->morphVertex()<< " - " << limastring2utf8stdstring(get(vertex_token, *graph, (*its).first->morphVertex())->stringForm()) << " : " << (*its).second;
446 }
447
448 LDEBUG << "resolve lexical anaphora";
449#endif
450 getBest(syntacticData.get(), posgraph.get(), lexAnaBinding, /*results,*/ npCandidates, annotationData.get());
451
452
453// TimeUtils::logElapsedTime("resolve lexical anaphora");
454// TimeUtils::updateCurrentTime();
455
456#ifdef DEBUG_LP
457 LDEBUG << "initialize potential binding";
458#endif
459 bindingPotentialCandidates(posgraph.get(), npAnaphora, npCandidates, pBinding);
460
461#ifdef DEBUG_LP
462 LDEBUG << "adjust local saliences";
463#endif
464 adjustSaliences(syntacticData.get(), npCandidates, pBinding, endSentence, posgraph.get(), analysis);
465
466#ifdef DEBUG_LP
467 LDEBUG<< "Potential Binding:";
468 for (WeightedVerticesRelation::iterator itp = pBinding->begin( );
469 itp != pBinding->end( );
470 itp++ )
471 {
472 LDEBUG<< "Anaphora: " << limastring2utf8stdstring(get(vertex_token, *graph, (*itp).first->morphVertex())->stringForm());
473 for (std::map<CoreferentAnnotation*, float>::iterator its = (*itp).second.begin( );
474 its != (*itp).second.end( );
475 its++ )
476 {
477 LDEBUG <<(*its).first->morphVertex()<< " - "
478 << limastring2utf8stdstring(get(vertex_token, *graph, (*its).first->morphVertex())->stringForm())
479 << " : " << (*its).second;
480 }
481 }
482 LDEBUG;
483
484
485
486 LDEBUG << "apply threshold filter";
487#endif
488 applyThresholdFilter(pBinding);
489
490#ifdef DEBUG_LP
491 LDEBUG << "apply circular filter";
492#endif
493 applyCircularFilter(pBinding);
494
495#ifdef DEBUG_LP
496 LDEBUG << "apply morphosyntactic filter";
497#endif
498 applyMorphoSyntacticFilter(pBinding,roBinding);
499
500/*TimeUtils::logElapsedTime("apply filters");
501TimeUtils::updateCurrentTime();*/
502#ifdef DEBUG_LP
503 LDEBUG << "resolve binding";
504#endif
505 getBest(syntacticData.get(), posgraph.get(), pBinding,/*results,*/npCandidates, annotationData.get());
506
507/*TimeUtils::logElapsedTime("resolve binding");
508TimeUtils::updateCurrentTime();*/
509 delete npAnaphora;
510 delete roBinding;
511 delete pBinding;
512 delete lexAnaBinding;
513 }
514
515#ifdef DEBUG_LP
516 LDEBUG << "write coreferent annotations on graph";
517#endif
518
519 delete npCandidates;
520// delete results;
521 TimeUtils::logElapsedTime("Coreferences");
522 return SUCCESS_ID;
523}
524
525
526void CorefSolver::initSyntacticFilter(
527 AnalysisContent& ac,
528 AnalysisGraph* anagraph,
529 SyntacticData* syntacticData,
530 Vertices* npAnaphora,
531 std::deque<CoreferentAnnotation::Vertices>* npCandidates,
532 VerticesRelation* roBinding) const
533{
534// COREFSOLVERLOGINIT;
535/*TimeUtils::logElapsedTime("init synt filter");
536TimeUtils::updateCurrentTime();*/
537 LinguisticGraph* graph = anagraph->getGraph();
538 // for every anaphora
539 for (Vertices::iterator anaphItr=npAnaphora->begin();
540 anaphItr!=npAnaphora->end();
541 anaphItr++)
542 {
543 // for every candidate
544 std::set<CoreferentAnnotation*> tmpCandidate;
545 for (std::deque<Vertices>::iterator itm = npCandidates->begin();
546 itm != npCandidates->end( );
547 itm++)
548 for (Vertices::iterator candidateItr = (*itm).begin( );
549 candidateItr != (*itm).end( );
550 candidateItr++ )
551 {
552 // if to be ruled out, insert in roBinding
553 if (
554 // P is diferent from N.
555 /*(*anaphItr)->morphVertex() != (*candidateItr)->morphVertex() && */
556 // P and N have incompatible agreement features.
557 !(*anaphItr)->isAgreementCompatibleWith(**candidateItr, syntacticData, m_genderAccessor, m_personAccessor,
558 m_numberAccessor, m_language, anagraph, ac) ||
559 // P & N are in the same sentence AND ...
560 (
561 (*npCandidates->begin()).find(*candidateItr)!=(*npCandidates->begin()).end()
562 &&
563 // P is in the argument domain of N.
564 // do not rule out if P is SujInv et N est SUJ_V
565 (
566 (*anaphItr)->isInTheArgumentDomainOf(**candidateItr, syntacticData, m_language, anagraph, ac,
568 new set<LinguisticGraphVertex>())
569 &&
570 !(
571 GovernorOf(m_language,utf8stdstring2limastring("SujInv"))(*anagraph,(*anaphItr)->morphVertex(),ac)
572 &&
573 GovernorOf(m_language,utf8stdstring2limastring("SUJ_V"))(*anagraph,(*candidateItr)->morphVertex(),ac)
574 )
575// && limastring2utf8stdstring(get(vertex_token, *graph, (*anaphItr)->morphVertex())->stringForm())!="en"
576 )
577 )||
578 // P is in the adjunct domain of N.
579 ((*anaphItr)->isInTheAdjunctDomainOf(**candidateItr, syntacticData, graph, m_macroAccessor, m_tagLocalDef,
580 m_relLocalDef, m_language) /*&& limastring2utf8stdstring(get(vertex_token, *graph, (*anaphItr)->morphVertex())->stringForm())!="en" no change*/)||
581 // P is an argument of a head H, N is not a pronoun, and N is contained in H.
582// (*anaphItr)->sf4(**candidateItr, syntacticData, m_macroAccessor, L_PRON(), m_language, anagraph, ac) ||
583 // P is in the NP domain of N.
584 (*anaphItr)->isInTheNpDomainOf(**candidateItr, syntacticData, m_macroAccessor, m_tagLocalDef, m_relLocalDef,
585 m_language, anagraph, ac) ||
586 // P is a determiner of a noun Q, and N is contained in Q.
587 (*anaphItr)->sf6(**candidateItr, syntacticData, m_relLocalDef, m_language, anagraph, ac)
588 )
589 {
590 tmpCandidate.insert(*candidateItr);
591 }
592 }
593 roBinding->insert(make_pair(*anaphItr, tmpCandidate));
594 tmpCandidate.clear();
595 }
596}
597
598 void CorefSolver::bindingLexicalAnaphora(
599 AnalysisContent& ac,
600 AnalysisGraph* anagraph,
601 SyntacticData* syntacticData,
602 Vertices* npAnaphora,
603 std::deque<CoreferentAnnotation::Vertices>* npCandidates,
604 WeightedVerticesRelation* lexAnaBinding) const
605{
606// COREFSOLVERLOGINIT;
607 /*TimeUtils::logElapsedTime("init blA");
608 TimeUtils::updateCurrentTime();
609 */LinguisticGraph* graph = anagraph->getGraph();
610 // for every anaphora
611 for (Vertices::iterator anaphItr=npAnaphora->begin();
612 anaphItr!=npAnaphora->end();
613 anaphItr++)
614 {
615 // if not reciprocal or reflexive, ignore
616 if (!(*anaphItr)->isTaggedAsOneOfThese(graph,m_reflexiveReciprocal,m_microAccessor))
617 continue;
618 // for every candidate
619 std::map<CoreferentAnnotation*, float> tmpCandidate;
620 for (std::deque<Vertices>::iterator itm = npCandidates->begin();
621 itm != npCandidates->end( );
622 itm++)
623 for (Vertices::iterator candidateItr = (*itm).begin( );
624 candidateItr != (*itm).end( );
625 candidateItr++ )
626 {
627 // if to be binded, insert in lexAnaBinding
628 if (
629 // A is diferent from N.
630 (*anaphItr)->morphVertex() != (*candidateItr)->morphVertex() &&
631 // A and N do not have any incompatible agreement features
632 (*anaphItr)->isAgreementCompatibleWith(**candidateItr, syntacticData, m_genderAccessor, m_personAccessor,
633 m_numberAccessor, m_language, anagraph, ac) && (
634 // A is SUBSUBJUX of N
635 !SecondUngovernedBy(m_language, utf8stdstring2limastring("SUBSUBJUX"))(*anagraph,(*anaphItr)->morphVertex(),
636 (*candidateItr)->morphVertex(),ac) ||
637 // A is in the argument domain of N,
638 // and N fills a higher argument slot than A.
639 (*anaphItr)->isInTheArgumentDomainOf2(**candidateItr, syntacticData,m_language,anagraph,ac,m_macroAccessor,
641 new set<LinguisticGraphVertex>()) ||
642 // A is in the adjunct domain of N.
643 (*anaphItr)->isInTheAdjunctDomainOf(**candidateItr,syntacticData, anagraph->getGraph(), m_macroAccessor,
645 // A is in the NP domain of N.
646 (*anaphItr)->isInTheNpDomainOf(**candidateItr, syntacticData, m_macroAccessor, m_tagLocalDef, m_relLocalDef,
647 m_language, anagraph, ac) ||
648 // N is an argument of a verb V,
649 // there is an NP Q in the argument domain of N such that Q has no noun determiner,
650 // and
651 // (i) A is an argument of Q.
652 // or (ii) A is an argument of a preposition PREP and PREP is an adjunct of Q.
653 (*anaphItr)->aba4(**candidateItr,syntacticData,m_tagLocalDef,m_relLocalDef,m_macroAccessor,m_microAccessor,
654 m_language,m_inNpCategs,anagraph,ac,m_conjCoord,npCandidates) ||
655 // A is a determiner of a noun Q,
656 // and
657 // (i) Q is in the argument domain of N,
658 // and N fills a higher argument slot than Q.
659 // or (ii) Q is in the adjunct domain of N.
660 (*anaphItr)->aba5(**candidateItr,syntacticData,m_tagLocalDef,m_relLocalDef,m_macroAccessor,m_microAccessor,
662 ))
663 {
664 tmpCandidate.insert(make_pair(*candidateItr, (*candidateItr)->salience()));
665 }
666 }
667 lexAnaBinding->insert(make_pair(*anaphItr, tmpCandidate));
668 }
669}
670
671
672 void CorefSolver::bindingPotentialCandidates(
673 AnalysisGraph* anagraph,
674 Vertices* npAnaphora,
675 std::deque<Vertices>* npCandidates,
676 WeightedVerticesRelation* pBinding) const
677{
678// COREFSOLVERLOGINIT;
679 LinguisticGraph* graph = anagraph->getGraph();
680 applyEquivalentClassFilter(graph,npCandidates);
681 // for every anaphora
682 for (Vertices::iterator anaphItr=npAnaphora->begin();
683 anaphItr!=npAnaphora->end();
684 anaphItr++)
685 {
686 // if reciprocal or reflexive, ignore
687 if ((*anaphItr)->isTaggedAsOneOfThese(graph,m_reflexiveReciprocal,m_microAccessor))
688 continue;
689 // for every candidate
690 for (std::deque<Vertices>::iterator itm = npCandidates->begin();
691 itm != npCandidates->end( );
692 itm++)
693 {
694 for (Vertices::iterator candidateItr = (*itm).begin( );
695 candidateItr != (*itm).end( );
696 candidateItr++ )
697 {
698 WeightedVerticesRelation::iterator tmpBinding = pBinding->find(*anaphItr);
699 if (tmpBinding==pBinding->end())
700 {
701 std::map <CoreferentAnnotation*,float> tmpMap;
702 tmpMap.insert(make_pair(*candidateItr,(*candidateItr)->salience()));
703 pBinding->insert(make_pair(*anaphItr,tmpMap));
704 }
705 else
706 {
707 (*tmpBinding).second.insert(make_pair(*candidateItr,(*candidateItr)->salience()));
708 }
709 }
710 }
711 }
712}
713
714void CorefSolver::getBest(
718// std::vector<std::pair<CoreferentAnnotation, CoreferentAnnotation> >* results,
719 std::deque<CoreferentAnnotation::Vertices>* npCandidates,
721{
722#ifdef DEBUG_LP
724 LDEBUG << "CorefSolver::getBest binding size = " << binding->size();
725#endif
726 const LinguisticGraph* graph = anagraph->getGraph();
727 for (WeightedVerticesRelation::iterator itp = binding->begin( );
728 itp != binding->end( );
729 itp++ )
730 {
731#ifdef DEBUG_LP
732 LDEBUG << " outer for loop on " << (*itp).first->morphVertex();
733#endif
734 bool result = false;
735 bool erase = false;
736 CoreferentAnnotation* ca2erase = 0;
737 std::pair<CoreferentAnnotation, CoreferentAnnotation> tmpPair;
738 for (std::map<CoreferentAnnotation*, float>::iterator its = (*itp).second.begin( );
739 its != (*itp).second.end( );
740 its++ )
741 {
742#ifdef DEBUG_LP
743 LDEBUG << " looking at " << (*itp).first->morphVertex() << " ("<<(*itp).first->bindingSalience()<<") -> " << (*its).first->morphVertex() << " ("<<(*its).second<<")";
744#endif
745 if ((*its).second>=(*itp).first->bindingSalience())
746 {
747 (*itp).first->bindingSalience((*its).second);
748 tmpPair = make_pair(*(*itp).first, *(*its).first);
749 ca2erase = (*its).first;
750 // si id!=ref
751 // et si id n'est pas anaphore lexicale
752 // et si id n'est pas un défini
753 // et si id != "y"
754 // effacer ancienne référence
755
756 if ((*itp).first!=(*its).first && (!((*its).first)->isTaggedAsOneOfThese(graph,m_reflexiveReciprocal,m_microAccessor)) && (*itp).first->referentType(sd,anagraph->getGraph(),m_macroAccessor,m_microAccessor,m_tagLocalDef,m_relLocalDef,m_definiteCategs,m_reflexiveReciprocal,m_undefPronouns,m_possPronouns,m_personAccessor,anagraph, m_language)!="def" &&
757 limastring2utf8stdstring(get(vertex_token, *graph, ((*itp).first)->morphVertex())->stringForm())!="y" )
758 erase = true;
759 result = true;
760 }
761// used by First_NP baseline
762// tmpPair = make_pair(*(*itp).first, *(*its).first);
763// break;
764 }
765 if (result)
766 {
767// results->push_back(tmpPair);
768 tmpPair.first.writeAnnotation(ad, tmpPair.second);
769 if (erase)
770 {
771 // mark newer Ref in npCandidates list so that former ref will be removed at next sentence
772 for (std::deque<Vertices>::iterator itm = npCandidates->begin();
773 itm != npCandidates->end( );
774 itm++)
775 {
776 if ((*itm).find(ca2erase)!=(*itm).end())
777 {
778 (*(*itm).find(ca2erase))->newerRef((*itp).first);
779 }
780 }
781 // erase the winning candidate from the current list of candidates. last equivalent will be this pronoun from now on.
782 for (WeightedVerticesRelation::iterator itp2 = binding->begin( );
783 itp2 != binding->end( );
784 itp2++ )
785 {
786#ifdef DEBUG_LP
787 LDEBUG << " erasing " << (*itp2).first->morphVertex() << " -> " << ca2erase->morphVertex();
788#endif
789 (*itp2).second.erase(ca2erase);
790 }
791 }
792 }
793 }
794}
795
796
797 void CorefSolver::applyEquivalentClassFilter(
798 const LinguisticGraph* graph,
799 std::deque<CoreferentAnnotation::Vertices>* npCandidates) const
800{
801// COREFSOLVERLOGINIT;
802 std::set<CoreferentAnnotation*> candidates2erase;
803 for (std::deque<Vertices>::iterator itm = npCandidates->begin();
804 itm != npCandidates->end( );
805 itm++)
806 {
807 for (Vertices::iterator candidateItr = (*itm).begin( );
808 candidateItr != (*itm).end( );
809 candidateItr++ )
810 {
811 if (!(*candidateItr)->newerRef()->isTaggedAsOneOfThese(graph,m_reflexiveReciprocal,m_microAccessor) && (*candidateItr)->hasNewerRef())
812 {
813// cerr << (*candidateItr)->morphVertex() << " will be erased"<< endl;
814 candidates2erase.insert(*candidateItr);
815 }
816 }
817 }
818 for (std::set<CoreferentAnnotation*>::iterator itE = candidates2erase.begin();
819 itE != candidates2erase.end( );
820 itE++)
821 for (std::deque<Vertices>::iterator itm = npCandidates->begin();
822 itm != npCandidates->end( );
823 itm++)
824 {
825 // keep best salience
826 for (std::set<CoreferentAnnotation*>::iterator itm2 = (*itm).begin();
827 itm2 != (*itm).end( );
828 itm2++)
829 {
830 if ((*itm2)==(*itE)->newerRef() && (*itE)->salience()>(*itm2)->salience())
831 {
832 (*itm2)->salience((*itE)->salience());
833 continue;
834 }
835 }
836 // remove **itE from the candidates list
837 (*itm).erase(*itE);
838 }
839}
840
841void CorefSolver::adjustSaliences(
843 std::deque<Vertices>* npCandidates,
845 LinguisticGraphVertex /*endSentence*/,
847 AnalysisContent& /*ac*/) const
848{
849// COREFSOLVERLOGINIT;
850 float cataphoraWeight = m_salienceWeights.find("Cataphora")->second;
851 float sameSlotWeight = m_salienceWeights.find("SameSlot")->second;
852 float itselfWeight = m_salienceWeights.find("Itself")->second;
853// float slightDiff = m_salienceWeights.find("SlightDifference")->second;
854
855 for (WeightedVerticesRelation::iterator itp = pBinding->begin( );
856 itp != pBinding->end( );
857 itp++ )
858 {
859 for (std::map<CoreferentAnnotation*, float>::iterator its = (*itp).second.begin( );
860 its != (*itp).second.end( );
861 its++ )
862 {
863
864 // if (NP_CANDIDATE in the same sentence as PRON or after)
865 // && NP_CANDIDATE is after PRON
866 // <=>
867 // NP_candidate is in last part of npCandidates
868 // && vertexOf(PRON) >= vertexOf(NP_Candidate)
869 if ((*npCandidates->begin()).find((*its).first) != (*npCandidates->begin()).end() && (*itp).first->morphVertex()>=(*its).first->morphVertex())
870 {
871 (*its).second += cataphoraWeight;
872 cataphoraWeight-=10;
873 }
874// bool res = false;
875 // for each "macro" dependency relation
876 for (std::map< std::string,std::deque<std::string> >::const_iterator itDr = m_relLocalDef.begin();
877 itDr != m_relLocalDef.end();
878 itDr++)
879 {
880 const std::deque<std::string> macroDependencyRel = (*itDr).second;
881 // if PRON and NP fills the same "macro" dependency relation
882 if ((*itp).first->isFunctionMasterOf(sd,macroDependencyRel,m_language)
883 && (*its).first->isFunctionMasterOf(sd,macroDependencyRel,m_language))
884 {
885 (*its).second += sameSlotWeight;
886 }
887 }
888// // if PRON and NP fills each a different one of these slots : COI && COMPDUNOM
889// if(m_relLocalDef.find("COIRelation")!=m_relLocalDef.end()
890// &&
891// (
892// (*itp).first->isFunctionMasterOf(sd,(*m_relLocalDef.find("COIRelation")).second)
893// &&
894// (GovernorOf(m_language,utf8stdstring2limastring("COMPDUNOM"))(*anagraph,(*its).first->morphVertex(),ac)/*||GovernorOf(m_language,utf8stdstring2limastring("COD_V"))(*anagraph,(*its).first->morphVertex(),ac)*/)
895// ))
896// {
897// // (*its).second += sameSlotWeight+slightDiff;
898// }
899 // decrease potential if CANDIDATE=PRON
900 if ((*itp).first==(*its).first)
901 {
902 (*its).second += itselfWeight;
903 }
904 }
905 }
906}
907
911void CorefSolver::applyCircularFilter(CoreferentAnnotation::WeightedVerticesRelation* pBinding) const
912{
913#ifdef DEBUG_LP
915 LDEBUG << "CorefSolver::applyCircularFilter binding size = " << pBinding->size();
916#endif
917 bool crossReferenceFound = false;
918 do
919 {
920 crossReferenceFound = false;
921 for (WeightedVerticesRelation::iterator itp = pBinding->begin( ); itp != pBinding->end( );
922 itp++ )
923 {
924 CoreferentAnnotation* source = (*itp).first;
925 source->morphVertex();
926 for (std::map<CoreferentAnnotation*, float>::iterator its = (*itp).second.begin( ); its != (*itp).second.end( ); its++ )
927 {
928 CoreferentAnnotation* target = (*its).first;
929 // current target of current source points also to references
930 if (source != target && pBinding->find(target) != pBinding->end())
931 {
932#ifdef DEBUG_LP
933 LDEBUG << "Check cross reference between " << source->morphVertex() <<" and " << target->morphVertex();
934#endif
935 WeightedVerticesRelation::iterator targetIt = pBinding->find(target);
936 // current source and target are effectively cross-referencing each other
937 if ( (*targetIt).second.find(source) != (*targetIt).second.end() )
938 {
939 crossReferenceFound = true;
940 float sourceToTargetWeight = (*its).second;
941 float targetToSourceWeight = (*(*targetIt).second.find(source)).second;
942#ifdef DEBUG_LP
943 LDEBUG << "Cross reference found between " << source->morphVertex() << " ("<<sourceToTargetWeight<<") and " << target->morphVertex() << " ("<<targetToSourceWeight<<")";
944#endif
945 // source to target is better
946 if (sourceToTargetWeight >= targetToSourceWeight)
947 {
948#ifdef DEBUG_LP
949 LDEBUG << " erasing " << (*targetIt).first->morphVertex() << " -> " << (*(*targetIt).second.find(source)).first->morphVertex();
950#endif
951 (*targetIt).second.erase( (*targetIt).second.find(source) ) ;
952 }
953 // target to source is better
954 else
955 {
956#ifdef DEBUG_LP
957 LDEBUG << " erasing " << (*itp).first->morphVertex() << " -> " << (*its).first->morphVertex();
958#endif
959 (*itp).second.erase(its);
960 }
961 break;
962 }
963 }
964 }
965 // go out of outer for loop if cross reference found as iterators are invalidated
966 if (crossReferenceFound) break;
967 }
968 } while (crossReferenceFound);
969#ifdef DEBUG_LP
970 LDEBUG << "No more cross references";
971#endif
972}
973
974 void CorefSolver::applyThresholdFilter(CoreferentAnnotation::WeightedVerticesRelation* pBinding) const
975{
976#ifdef DEBUG_LP
978 LDEBUG << "CorefSolver::applyThresholdFilter " << m_threshold;
979#endif
980 for (WeightedVerticesRelation::iterator itp = pBinding->begin( );
981 itp != pBinding->end( );
982 itp++ )
983 {
984 for (std::map<CoreferentAnnotation*, float>::iterator its = (*itp).second.begin( );
985 its != (*itp).second.end( );
986 its++ )
987 {
988#ifdef DEBUG_LP
989 LDEBUG << " threshold bewteen " << (*itp).first->morphVertex() << " and " << (*its).first->morphVertex() << " ; value: " << (*its).second;
990#endif
991 if ((*its).second<m_threshold)
992 {
993#ifdef DEBUG_LP
994 LDEBUG << " REMOVING ";
995#endif
996 (*itp).second.erase(its);
997 // its is invalidated; reinitialize it
998 its = (*itp).second.begin( );
999 }
1000 }
1001 }
1002}
1003
1004void CorefSolver::applyMorphoSyntacticFilter(
1007{
1008#ifdef DEBUG_LP
1010 LDEBUG << "CorefSolver::applyMorphoSyntacticFilter binding size = " << pBinding->size();
1011#endif
1012 for (WeightedVerticesRelation::iterator itp = pBinding->begin( );
1013 itp != pBinding->end( );
1014 itp++ )
1015 {
1016 std::set<CoreferentAnnotation*> roSet = (*roBinding->find((*itp).first)).second;
1017 for (std::set<CoreferentAnnotation*>::iterator itro = roSet.begin( );
1018 itro != roSet.end( );
1019 itro++ )
1020 {
1021#ifdef DEBUG_LP
1022 LDEBUG << " erasing " << (*itp).first->morphVertex() << " -> " << (*itro)->morphVertex();
1023#endif
1024 (*itp).second.erase(*itro);
1025 }
1026 }
1027}
1028
1029
1030} // closing namespace Coreferences
1031} // closing namespace LinguisticProcessing
1032} // closing namespace Lima
This file is the main header file for the data related to annotation graphs.
#define LDEBUG
Definition LimaCommon.h:157
#define LINFO
Definition LimaCommon.h:158
#define LERROR
Definition LimaCommon.h:161
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define COREFSOLVERLOGINIT
Defines a Factory to create Object of type Base.
Data used for the simplification of sentences allowing easier heterosyntagmatic analysis.
Data used for the syntactic analyzis of texts.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Holds an annotation graph and gives an API to manipulate it.
Holds linguistic data for one language.
const MediaData & mediaData(MediaId media) const
Provide tools to manage a specific property.
const PropertyAccessor & getPropertyAccessor() const
give the corresponding PropertyAccessor
LinguisticCode getPropertyValue(const std::string &symbolicValue) const
Get the coded property value from the symbolic value.
std::deque< std::string > & getListsValueAtKey(const std::string &key)
std::map< std::string, std::string > & getMapAtKey(const std::string &key)
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
const Common::PropertyCode::PropertyAccessor * m_numberAccessor
const Common::PropertyCode::PropertyAccessor * m_macroAccessor
const Common::PropertyCode::PropertyAccessor * m_personAccessor
std::map< std::string, LinguisticCode > m_tagLocalDef
std::map< std::string, std::deque< std::string > > m_relLocalDef
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
const Common::PropertyCode::PropertyAccessor * m_genderAccessor
LimaStatusCode process(AnalysisContent &analysis) const override
const Common::PropertyCode::PropertyAccessor * m_microAccessor
std::map< CoreferentAnnotation *, std::set< CoreferentAnnotation * > > VerticesRelation
bool isIncludedInNounPhrase(const LinguisticGraph *g, MediaId language, const LinguisticAnalysisStructure::AnalysisGraph *anagraph, AnalysisContent &ac, const std::set< LinguisticCode > &inNpCategs, const Common::PropertyCode::PropertyAccessor *microAccessor) const
general test functions
std::map< CoreferentAnnotation *, std::map< CoreferentAnnotation *, float > > WeightedVerticesRelation
Definition of a function suitable to be used as a dumper for Coreferent Annotations of an Annotation ...
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
This constraint tests if its argument is the governor of a relation of the given type.
This constraint tests if the second argument is not governed by the first one with a relation of the ...
This class points to a graph, its dependency graph and the structure that holds the maping between th...
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
static void logElapsedTime(const std::string &mess, const std::string &taskCategory=std::string(""))
log the number of microseconds since last UpdateCurrentTime
static void updateCurrentTime(const std::string &taskCategory=std::string(""))
store current time for new elapsed time computation
#define COREFSOLVINGPU_CLASSID
Definition corefSolver.h:50
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
LimaString utf8stdstring2limastring(const std::string &src)
std::set< CoreferentAnnotation * > Vertices
Definition corefSolver.h:52
std::map< CoreferentAnnotation *, std::set< CoreferentAnnotation * > > VerticesRelation
Definition corefSolver.h:53
std::map< CoreferentAnnotation *, std::map< CoreferentAnnotation *, float > > WeightedVerticesRelation
Definition corefSolver.h:54
SimpleFactory< MediaProcessUnit, CorefSolver > coreferencesSolvingProcessUnit(COREFSOLVINGPU_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
@ INVALID_CONFIGURATION
Definition LimaCommon.h:242
@ MISSING_DATA
Definition LimaCommon.h:243
STL namespace.