LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
AbstractIEDumper.cpp
Go to the documentation of this file.
1// Copyright (C) 2016 by CEA - LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6#include "AbstractIEDumper.h"
8
21
22#include <boost/algorithm/string.hpp>
23#include <fstream>
24#include <queue>
25#include <tuple>
26
27using namespace std;
28using namespace Lima::Common::AnnotationGraphs;
29using namespace Lima::Common::MediaticData;
34using namespace boost;
35
36namespace Lima {
37namespace LinguisticProcessing {
38namespace AnalysisDumpers {
39
40// a class to hold information about current state of the output,
41// when the output is used in append mode (e.g. when the analysis is done
42// separately on each paragraph): store the number of elements printed to get ids right
43
49
51 nbEntities(0),
52 nbAttributes(0),
53 nbRelations(0),
54 nbEvents(0)
55 {}
56
58 nbEntities(0),
59 nbAttributes(0),
60 nbRelations(0),
61 nbEvents(0)
62 {
63 if (metadata->hasMetaData("IEDumper.nbEntities"))
64 nbEntities=std::stoi(metadata->getMetaData("IEDumper.nbEntities"));
65 if (metadata->hasMetaData("IEDumper.nbAttributes"))
66 nbAttributes=std::stoi(metadata->getMetaData("IEDumper.nbAttributes"));
67 if (metadata->hasMetaData("IEDumper.nbRelations"))
68 nbRelations=std::stoi(metadata->getMetaData("IEDumper.nbRelations"));
69 if (metadata->hasMetaData("IEDumper.nbEvents"))
70 nbEvents=std::stoi(metadata->getMetaData("IEDumper.nbEvents"));
71
72 //std::cerr << *this << std::endl;
73 }
74
75 void save(LinguisticMetaData* metadata) {
76 metadata->setMetaData("IEDumper.nbEntities",std::to_string(nbEntities));
77 metadata->setMetaData("IEDumper.nbAttributes",std::to_string(nbAttributes));
78 metadata->setMetaData("IEDumper.nbRelations",std::to_string(nbRelations));
79 metadata->setMetaData("IEDumper.nbEvents",std::to_string(nbEvents));
80 }
81
82 friend std::ostream& operator<<(std::ostream& os, IEDumperMetaData& m) {
83 os << "DumperMetaData: nbEntities="<< m.nbEntities << ",nbRelations=" << m.nbRelations << ",nbEvents="<< m.nbEvents;
84 return os;
85 }
86
87};
88
91m_language(0),
92m_graph("PosGraph"),
93m_followGraph(false),
94m_domains(),
95m_ignore(),
96m_attributes(),
97m_all_attributes(false),
98m_templateDefinitions(),
99m_templateNames(),
100m_offsetMapping(nullptr),
101m_outputGroups(false),
102m_posOffset(-1)
103{
105 LDEBUG << "AbstractIEDumper::AbstractIEDumper()";
106}
107
110
113 Manager* manager)
114
115{
117 LDEBUG << "AbstractIEDumper::init";
118 AbstractTextualAnalysisDumper::init(unitConfiguration,manager);
119 std::deque<std::string> eventTemplates;
120
122
123 try
124 {
125 m_graph=unitConfiguration.getParamsValueAtKey("graph");
126 }
128 {
130 LWARN << "No 'graph' parameter in unit configuration '"
131 << unitConfiguration.getName() << "' ; using PosGraph";
132 m_graph=string("PosGraph");
133 }
134
135 try
136 {
137 // single domain
138 string val=unitConfiguration.getParamsValueAtKey("domain");
139 if (! val.empty()) {
140 m_domains.insert(val);
141 }
142 }
144 {
145 m_domains=set<string>();
147 LDEBUG << "no parameter 'domain' in AbstractIEDumper: entities from all domains will be printed";
148 } // if empty set, all domains are printed
149
150 try
151 {
152 deque< string > vals=unitConfiguration.getListsValueAtKey("domains");
153 for (const auto& v: vals) {
154 m_domains.insert(v);
155 }
156 }
158 {
159 LDEBUG << "no list 'domains' in AbstractIEDumper";
160 } // if empty set, all domains are printed
161
162 try
163 {
164 deque< string > vals=unitConfiguration.getListsValueAtKey("ignore");
165 for (const auto& v: vals) {
166 m_ignore.insert(v);
167 }
168 }
170 {
171 LDEBUG << "no list 'ignore' in AbstractIEDumper: all entity types of authorized domains are printed";
172 }
173
174 try
175 {
176 string val=unitConfiguration.getParamsValueAtKey("outputAllAttributes");
177 if (val=="1" || val=="true" || val=="yes") {
178 m_all_attributes=true;
179 }
180 }
181 catch (Common::XMLConfigurationFiles::NoSuchParam& ) {}// keep default value (false)
182
183 if(!m_all_attributes){
184 try
185 {
186 deque< string > vals=unitConfiguration.getListsValueAtKey("attributes");
187 m_attributes=vals;
188 }
190 {
191 m_attributes=std::deque<std::string>();
193 LDEBUG << "no list 'attributes' in AbstractIEDumper: no attributes will be printed";
194 } // no attributes are printed
195 }
196
197 try
198 {
199 string val=unitConfiguration.getParamsValueAtKey("outputGroups");
200 if (val=="1" || val=="true" || val=="yes") {
201 m_outputGroups=true;
202 }
203 }
204 catch (Common::XMLConfigurationFiles::NoSuchParam& ) {} // keep default value (false)
205
206
207 try
208 {
209 eventTemplates=unitConfiguration.getListsValueAtKey("eventTemplates");
210 //std::cout << "eventTemplates.size()=" << eventTemplates.size() << "\n";
211
212 MediaId language=manager->getInitializationParameters().media;
213 for(std::deque<std::string>::const_iterator it=eventTemplates.begin();it!=eventTemplates.end();it++)
214 {
215 std::string templateResource=*it;
216 //std::cout << "templateResource=" << templateResource << "\n";
217 auto res = LinguisticResources::single().getResource(language,templateResource);
218 if (res)
219 {
220 //std::cout << " La ressource est lue \n";
221 auto templateDefinitions = std::dynamic_pointer_cast<EventTemplateDefinitionResource>(res);
222 m_templateDefinitions.insert(std::make_pair(templateResource,templateDefinitions));
223 }
224 }
225 for (const auto& t:m_templateDefinitions) {
226 m_templateNames.insert(t.second->getMention());
227 }
228 }
229 catch (Common::XMLConfigurationFiles::NoSuchParam& ) {} // do nothing: optional
230
231 try {
232 std::string str=unitConfiguration.getParamsValueAtKey("followGraph");
233 if (str=="1" || str=="true" || str=="yes") {
234 m_followGraph=true;
235 }
236 else {
237 m_followGraph=false;
238 }
239 }
240 catch (Common::XMLConfigurationFiles::NoSuchParam& ) {} // keep default value
241
242 try {
243 std::string str=unitConfiguration.getParamsValueAtKey("posOffset");
244 m_posOffset=std::stoi(str);
245 }
246 catch (Common::XMLConfigurationFiles::NoSuchParam& ) {} // keep default value
247
248}
249
251 AnalysisContent& analysis) const
252{
254 LDEBUG << "AbstractIEDumper::process";
257 std::map<std::tuple<std::uint64_t,std::uint64_t,std::string>, std::size_t > mapEntities;
259 std::map <std::tuple <std::size_t, std::string , std::string >, std::size_t > mapAttributes;
260
261 std::string sourceFile;
262
263 auto originalText = std::dynamic_pointer_cast<LimaStringText>(analysis.getData("Text"));
264
265 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.getData("LinguisticMetaData"));
266 if (metadata == 0) {
268 LERROR << "no LinguisticMetaData ! abort";
269 return MISSING_DATA;
270 }
271
272 sourceFile=metadata->getMetaData("FileName");
273
274 IEDumperMetaData* dumperMetadata=new IEDumperMetaData(metadata.get());
275
276 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.getData("AnnotationData"));
277 if (annotationData == 0) {
279 LERROR << "no annotationData ! abort";
280 return MISSING_DATA;
281 }
282
283
284 auto graphp = std::dynamic_pointer_cast<LinguisticAnalysisStructure::AnalysisGraph>(analysis.getData(m_graph));
285 if (graphp == 0) {
287 LERROR << "no graph "<< m_graph <<" ! abort";
288 return MISSING_DATA;
289 }
290
291 auto eventData = std::dynamic_pointer_cast<EventTemplateData>(analysis.getData("EventTemplateData"));
292 if (eventData==0) {
294 LDEBUG << "No data 'EventTemplateData'";
295 }
296
297 auto offsetData = analysis.getData("OffsetMapping");
298 if (offsetData!=0) {
299 m_offsetMapping = std::dynamic_pointer_cast<OffsetMapping>(offsetData).get();
300 }
301
302 const LinguisticAnalysisStructure::AnalysisGraph& graph = *graphp;
303 LinguisticGraph* lingGraph = const_cast<LinguisticGraph*>(graph.getGraph());
304 VertexTokenPropertyMap tokenMap = get(vertex_token, *lingGraph);
305
306
307
308 auto dstream = initialize(analysis);
309 ostream& out=dstream->out();
310
311 uint64_t offset(0);
312 try {
313 offset=QString::fromStdString(metadata->getMetaData("StartOffset")).toUInt();
314 }
316 // do nothing: not set in analyzeText (only in analyzeXmlDocuments)
317 }
318
319 /*uint64_t offsetIndexingNode(0);
320 try {
321 offsetIndexingNode=atoi(metadata->getMetaData("StartOffsetIndexingNode").c_str());
322 }
323 catch (LinguisticProcessingException& ) {
324 // do nothing: not set in analyzeText (only in analyzeXmlDocuments)
325 }*/
326
327 std::string docId("");
328 try {
329 docId=metadata->getMetaData("DocId");
330 }
332 // do nothing: not set in analyzeText (only in analyzeXmlDocuments)
333 }
334 outputGlobalHeader(out,sourceFile,*originalText);
335
337
338 if (m_followGraph) {
339 // instead of looking to all annotations, follow the graph (in
340 // morphological graph, some vertices are not related to main graph:
341 // idiomatic expressions parts and named entity parts)
342 // -> this will not include nested entities
343
344 auto tokenList = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData(m_graph));
345 if (tokenList==0) {
346 LERROR << "graph " << m_graph << " has not been produced: check pipeline";
347 return MISSING_DATA;
348 }
349 LinguisticGraph* graph=tokenList->getGraph();
350 //const FsaStringsPool& sp=Common::MediaticData::MediaticData::single().stringsPool(m_language);
351
352 std::queue<LinguisticGraphVertex> toVisit;
353 std::set<LinguisticGraphVertex> visited;
354 toVisit.push(tokenList->firstVertex());
355
356 LinguisticGraphOutEdgeIt outItr,outItrEnd;
357 while (!toVisit.empty()) {
358 LinguisticGraphVertex v=toVisit.front();
359 toVisit.pop();
360 if (v == tokenList->lastVertex()) {
361 continue;
362 }
363
364 for (boost::tie(outItr,outItrEnd)=out_edges(v,*graph); outItr!=outItrEnd; outItr++)
365 {
366 LinguisticGraphVertex next=target(*outItr,*graph);
367 if (visited.find(next)==visited.end())
368 {
369 visited.insert(next);
370 toVisit.push(next);
371 }
372 }
373 const SpecificEntityAnnotation* annot=getSpecificEntityAnnotation(v,annotationData.get());
374 if (annot != 0) {
375 outputEntity(out,v,annot,tokenMap,offset,*originalText,mapEntities,mapAttributes,dumperMetadata);
376 }
377 }
378 }
379 else {
380 // take all annotations
381 AnnotationGraphVertexIt itv, itv_end;
382 boost::tie(itv, itv_end) = vertices(annotationData->getGraph());
383 for (; itv != itv_end; itv++)
384 {
385 // LDEBUG << "AbstractIEDumper on annotation vertex " << *itv;
386 if (annotationData->hasAnnotation(*itv,QString::fromUtf8("SpecificEntity")))
387 {
388 // LDEBUG << " it has SpecificEntityAnnotation";
389 const SpecificEntityAnnotation* annot = 0;
390 try
391 {
392 annot = annotationData->annotation(*itv,QString::fromUtf8("SpecificEntity"))
393 .pointerValue<SpecificEntityAnnotation>();
394 }
395 catch (const boost::bad_any_cast& )
396 {
398 LERROR << "This annotation is not a SpecificEntity; SE not logged";
399 continue;
400 }
401
402 // recuperer l'id du vertex morph cree
404 if (!annotationData->hasIntAnnotation(*itv,QString::fromStdString(m_graph)))
405 {
406 // DUMPERLOGINIT;
407 // LDEBUG << *itv << " has no " << m_graph << " annotation. Skeeping it.";
408 continue;
409 }
410 v = annotationData->intAnnotation(*itv,QString::fromStdString(m_graph));
411 outputEntity(out,v,annot,tokenMap,offset,*originalText,mapEntities,mapAttributes,dumperMetadata);
412 }
413 }
414 }
415 // update number of entities/attributes
416 dumperMetadata->nbEntities+=mapEntities.size();
417 dumperMetadata->nbAttributes+=mapAttributes.size();
418 //std::cerr << *dumperMetadata << std::endl;
420
422 if (eventData!=0) {
423 uint nbEvents=outputEventData(out,eventData.get(),annotationData.get(),tokenMap,mapEntities,offset,
424 *originalText, dumperMetadata);
425 dumperMetadata->nbEvents+=nbEvents;
426 }
428
430 uint nbRelations=outputSemanticRelations( out, annotationData.get(), tokenMap, mapEntities, offset, dumperMetadata);
431 dumperMetadata->nbRelations+=nbRelations;
433
435
436 // save current state for future output
437 dumperMetadata->save(metadata.get());
438
439 TimeUtils::logElapsedTime("AbstractIEDumper");
440 return SUCCESS_ID;
441
442}
443
444// due to text expansion, the same event may have been recognized in several parts of the text: after
445// changing back the positions, this creates duplicate events: need to remove those duplicates
446// use an internal class to store event infos, with a comparison operator that allows to find duplicates
448public:
451 mutable unsigned int eventMentionId; // mutable because does not count for set<> in comparison operator
452 std::string eventMentionType;
453 std::vector<unsigned int> eventRoleId;
454 std::vector<std::string> eventRoleType;
455 // infos on the mention to add entity if needed
456 mutable std::string eventMentionString; // mutable because does not count for set<> in comparison operator
459
460 bool operator<(const EventInfos& other) const {
461 // use type and string of event mention for comparison
462 // + all events roles (ids of entities)
463 // do not use entityMentionId because it is not fixed (id of existing entity or new entity to be added)
464 // or entityMentionPosition because it is the position before changing back to original offsets
466 return (eventMentionType<other.eventMentionType);
467 }
468 else if (eventMentionString!=other.eventMentionString) {
470 }
471 else {
472 return (eventRoleId<other.eventRoleId);
473 }
474 }
475
476 bool isIncluded(const EventInfos& other) const {
478 return false;
479 }
481 return false;
482 }
483 for (const auto& role: eventRoleId) {
484 if (std::find(other.eventRoleId.begin(), other.eventRoleId.end(), role) == other.eventRoleId.end()) {
485 return false;
486 }
487 }
488 return true;
489 }
490
491 // for debug
492 friend std::ostream& operator<<(std::ostream& os, const EventInfos e) {
493 os << "id=" << e.eventMentionId << "/type=" << e.eventMentionType
494 << "/mention=" << e.eventMentionString << "/roles=";
495 for (unsigned int i(0); i<e.eventRoleId.size(); i++) {
496 os << "[" << e.eventRoleType[i] << ":" << e.eventRoleId[i] << "]";
497 }
498 return os;
499 }
500 friend QDebug& operator<<(QDebug& os, const EventInfos e) {
501 ostringstream oss; oss << e; os << oss.str(); return os;
502 }
503
504};
505
507 const EventAnalysis::EventTemplateData* eventData,
508 const Common::AnnotationGraphs::AnnotationData* /*annotationData*/,
509 const VertexTokenPropertyMap& tokenMap,
510 std::map<std::tuple<std::uint64_t,std::uint64_t,std::string>,std::size_t > mapEntities,
511 uint64_t offset,
512 LimaStringText originalText,
513 IEDumperMetaData* metadata
514 ) const
515{
516 // output of events is based on the idea of elements of information to extraction: if the elements
517 // coming from different parts of the text are the same, printing them several times is not relevant
518 // (differs from an annotation task, maybe @todo have a parameter to handle this difference)
519 // Here, the rules for the output of events are:
520 // - equal events are kept only once
521 // - if an event is included in another event, it is not kept
522 // from these rules: most single event mentions (without roles) are not kept: for each different mention, one
523 // could be kept if there is no other event with roles that has the same mention
524
526 //return the number of events
527 // use a set of EventInfos to remove duplicates
528 set<EventInfos> events;
529 // use a bufferEvents to possibly have all event mentions as entities before all events
530 stringstream bufferEvents;
531 for (std::vector<EventTemplate>::const_iterator it= eventData->begin(); it!= eventData->end();it++)
532 {
533 const TemplateElements& templateElements=(*it).getTemplateElements();
534 if (! templateElements.empty()) {
535
536 const std::string& templateName=it->getType();
537 LDEBUG << "templateName:" << templateName;
538
539 std::string templateMention="";
540 for(const auto& t:m_templateDefinitions)
541 {
542 if (t.second->getName()==templateName) templateMention=t.second->getMention();
543 }
544 LDEBUG << "templateMention" << templateMention;
545
546 EventInfos eventInfos;
547
548 for(map<string,EventTemplateElement>::const_iterator it1= templateElements.begin(); it1!= templateElements.end();it1++)
549 {
550 const string typeName=it1->first;
551 LDEBUG << "typeName='" << typeName << "',templateMention='" << templateMention << "'";
552
553 LinguisticGraphVertex v=(*it1).second.getVertex();
554 LinguisticAnalysisStructure::Token* vToken = tokenMap[v];
555 std::uint64_t position= vToken->position();
556 std::uint64_t length= vToken->length();
557 string stringForm = originalText.mid( position+m_posOffset,length).toUtf8().data();
558
559 if (addEventMentionAsEntity() && templateMention.compare(typeName)==0)
560 {
561 // specific treatment for event mention: add a new entity with the corresponding type
562 LDEBUG << "add event mention as entity typeName=" << typeName;
563 eventInfos.eventMentionString=stringForm;
564 eventInfos.eventMentionPosition=position;
565 eventInfos.eventMentionLength=length;
566 eventInfos.eventMentionType=typeName;
567 }
568 else {
569
570 // entities in map are stored with original positions
571 adjustPosition(position,offset);
572
573 // get corresponding entity
574 string entityType=Common::Misc::limastring2utf8stdstring(Common::MediaticData::MediaticData::single().getEntityName((*it1).second.getType()));
575 const auto itE=mapEntities.find(make_tuple(position,length,entityType));
576 uint64_t id(0);
577 if (itE!=mapEntities.end()) {
578 id=(*itE).second;
579 LDEBUG << "AbstractIEDumper: found index" << id << "for entity (" << position << "," << length << "," << entityType << ")";
580 }
581 else {
582 LERROR << "AbstractIEDumper: cannot find index of entity (" << position << "," << length << "," << entityType << ")";
583 // push something to ensure both vectors are aligned (eventRoleType and eventRoleId)
584 }
585
586 if (templateMention.compare(typeName)==0) {
587 eventInfos.eventMentionType=typeName;
588 eventInfos.eventMentionId=id;
589 // save string because used in comparison operator (for set)
590 eventInfos.eventMentionString=stringForm;
591 }
592 else {
593 eventInfos.eventRoleType.push_back(typeName);
594 eventInfos.eventRoleId.push_back(id);
595 }
596 }
597 }
598 LDEBUG << "Add event infos" << eventInfos;
599 events.insert(eventInfos);
600 LDEBUG << "=>" << events.size() << "events";
601 // use a bufferEvents to have all event mentions as entities before all events
602 }
603 }
604 // output event mentions as entities
606 unsigned int idEntity=mapEntities.size()+1+metadata->nbEntities;
607 for (auto& e: events) {
608 e.eventMentionId=idEntity; // new entity id
609 std::vector<pair<uint64_t,uint64_t> > positions;
610 LimaString eventMentionString=QString::fromStdString(e.eventMentionString);
611 computePositions(positions,eventMentionString,e.eventMentionPosition,e.eventMentionLength);
612 // if (m_outputGroups && !m_domain.empty()) {
613 // outputEntityString(out, e.eventMentionId, m_domain+"."+e.eventMentionType, eventMentionString.toUtf8().data(), positions, Automaton::EntityFeatures(), true);
614 // }
615 // else {
616 outputEntityString(out, e.eventMentionId, e.eventMentionType, eventMentionString.toUtf8().data(), positions, Automaton::EntityFeatures(), true);
617 // }
618 idEntity++;
619 }
620 }
621
622 // filter events: remove events that are included in other events (possible because of text expansion)
623 for (set<EventInfos>::iterator it=events.begin();it!=events.end();) {
624 bool isIncluded(false);
625 for (set<EventInfos>::iterator it2=std::next(it); it2!=events.end(); it2++) {
626 if ((*it).isIncluded(*it2)) {
627 LDEBUG << "=> filter event" << *it << "(included in" << *it2 << ")";
628 isIncluded=true;
629 it=events.erase(it);
630 break;
631 }
632 }
633 if (! isIncluded) {
634 it++;
635 }
636 }
637
638 unsigned int eventId(1+metadata->nbEvents);
639 for (const auto& e: events) {
640 LDEBUG << "=> output event"<< eventId << "/" << events.size();
641 outputEventString(out, eventId, e.eventMentionId, e.eventMentionType, e.eventRoleId, e.eventRoleType);
642 eventId++;
643 }
644 return events.size();
645}
646
648computePositions(std::vector<pair<uint64_t,uint64_t> >& positions,
649 LimaString& stringForm,
650 uint64_t pos,
651 uint64_t len) const
652{
653 // if string contains \n, have to set several position intervals around these characters (brat-style)
654 positions.push_back(make_pair(pos+m_posOffset,pos+len+m_posOffset));
655 string::size_type prev(0),i=stringForm.indexOf("\n");
656 vector<unsigned int> toErase;
657 while (i!=string::npos) {
658// LimaString debugString(stringForm);
659// debugString.replace(' ','_');
660// debugString.replace('\n','N');
661// cout << "--prev=" << prev << ",i=" << i << " [" << debugString.toUtf8().data() << "] positions=";
662// for (const auto& p:positions) {cout << "(" << p.first << "," << p.second << ") "; }; cout << "--" << endl;
663 // replace by space
664 //stringForm[i]=' ';
665 stringForm.replace(i,1,' ');
666 if (prev!=0 && i==prev+1) { // case of several consecutive \n
667 positions.back().first++;
668 toErase.push_back(i); // erase it on output string, but do it later to avoid changing positions in original string
669 }
670 else {
671 // add an interval
672 unsigned int nbCharsIgnored=1; // the carriage return
673 // ignore spaces before
674 /*while (stringForm.size()>0 && stringForm[i-1]==' ') {
675 // erase space in the string
676 stringForm.erase(i-1,1);
677 i--;
678 nbCharsIgnored++;
679 }*/
680 // ignore spaces after the \n
681 /*while (i< stringForm.size() && stringForm[i+1]==' ') {
682 i++;
683 nbCharsIgnored++;
684 }*/
685 uint64_t previousEnd=positions.back().second;
686 if (prev==0) {
687 positions.back().second=positions.back().first+i;
688 }
689 else {
690 positions.back().second=positions.back().first+(i-prev-1);
691 }
692 positions.push_back(make_pair(positions.back().second+nbCharsIgnored,previousEnd));
693// cout << "-->positions="; for (const auto& p:positions) {cout << "(" << p.first << "," << p.second << ") "; }; cout << endl;
694 }
695 // find next
696 prev=i;
697 i=stringForm.indexOf("\n",prev+1);
698 }
699 // change string form to remove consecutive spaces due to \n
700 unsigned int shift=0;
701 for (const auto i: toErase) {
702 stringForm.remove(i-shift,1); // must change positions to follow string with removed chars
703 shift++;
704 }
705}
706
708outputEntity(std::ostream& out,
710 const SpecificEntityAnnotation* annot,
711 const VertexTokenPropertyMap& tokenMap,
712 uint64_t offset,
713 LimaStringText originalText,
714 std::map<std::tuple<std::uint64_t,std::uint64_t,std::string>,std::size_t>& mapEntities,
715 std::map <std::tuple <std::size_t, std::string , std::string >, std::size_t >& mapAttributes,
716 IEDumperMetaData* metadata
717 ) const
718{
719 LinguisticAnalysisStructure::Token* vToken = tokenMap[v];
720 // LDEBUG << "AbstractIEDumper tokenMap[" << v << "] = " << vToken;
721 if (vToken == 0)
722 {
724 LERROR << "Vertex " << v << " has no entry in the analysis graph token map. This should not happen !!";
725 }
726 else
727 {
728 std::uint64_t pos=annot->getPosition();
729 std::uint64_t len=annot->getLength();
730 //string stringForm=originalText.mid( pos-1,len).toUtf8().data();
731 auto stringForm=originalText.mid( pos+m_posOffset,len);
732 std::vector<pair<uint64_t,uint64_t> > positions;
733// cerr << "computePositions("<< stringForm.toUtf8().data() << ") " << pos << ":" << len << " -> ";
734 computePositions(positions,stringForm,pos,len);
735// for (const auto& p: positions) { cerr << p.first << ":" << p.second << " "; }; cerr << endl;
736
737 // get entity type and group type
738 EntityType eType=annot->getType();
739 std::string domainType=MediaticData::single().getEntityGroupName(eType.getGroupId()).toUtf8().data();
740 std::string entityName= MediaticData::single().getEntityName(eType).toUtf8().data();
741 std::string entityType(entityName);
742 std::size_t posG=entityType.find(".");
743 if (posG!=std::string::npos && ! m_outputGroups){
744 entityType = entityType.substr(posG+1);
745 }
746 if (m_ignore.find(entityType)!=m_ignore.end()) {
747 //LDEBUG << "AbstractIEDumper: ignored entity type" << entityType;
748 return false;
749 }
750
751 // back to the original offset if text has been expanded
752 // do this before inserting in mapEntities to find real duplicates
753 adjustPosition(pos,offset);
754 adjustPositions(positions,offset);
755 // if non-contiguous position (especially after adjustment to original positions), need to take real length to check for duplicates
756 if (positions.size()>0) {
757 len=positions.back().second-pos+1;
758 }
759
760 auto entity=std::make_tuple(pos,len,entityName);
761
762 if (mapEntities.find(entity)!=mapEntities.end())
763 {
764 // entity already exists
766 LWARN << "AbstractIEDumper: duplicate entity: " << pos << "," << len << " " << entityName;
767 return false;
768 }
769
770 std::size_t index = mapEntities.size()+1 + metadata->nbEntities;
771
772 if (! m_domains.empty() && m_domains.find(domainType)==m_domains.end())
773 {
774 // entity is not in considered domain: do not print it (nor store it in the map)
776 LDEBUG << "AbstractIEDumper: ignore entity of domain" << domainType;
777 return false;
778 }
779
780 //std::cout << "m_domain " << m_domain << " m_domain.size " << m_domain.size() << std::endl;
781 //std::cout << "domainType " << domainType << " domainType.size " << domainType.size() << std::endl;
782 //std::cout << "Entity is printed " << entityType << "domain=" << domainType << " / targetDomain=" << m_domain << std::endl;
783 const Automaton::EntityFeatures& entityFeatures = annot->getFeatures();
784 bool forceNoNorm=false;
785 if (m_templateNames.find(entityType)!=m_templateNames.end()) {
786 // if the entity is a template mention, must not have a normalization (according to Brat)
787 forceNoNorm=true;
788 }
789 outputEntityString(out, index, entityType, stringForm.toUtf8().data(), positions, entityFeatures,forceNoNorm);
790
791 // identify attributes as entityFeatures having a "POSITION" (or parts of a numex) and being in the declared list of attributes to display
792 for(Automaton::EntityFeatures::const_iterator featureItr=entityFeatures.cbegin(),features_end=entityFeatures.cend();
793 featureItr!=features_end; featureItr++) {
794 const std::string featName = featureItr->getName();
795 if( (featureItr->getPosition() != UNDEFPOSITION) ||
796 (entityName=="Numex.NUMBER" && featName == "numvalue") ||
797 (entityName=="Numex.NUMEX" && featName == "numvalue") ||
798 (entityName=="Numex.NUMEX" && featName == "unit") ) {
799 if( m_all_attributes || std::find(m_attributes.begin(), m_attributes.end(), featName) != m_attributes.end() ){
800 std::size_t feat_index = mapAttributes.size()+1 +metadata->nbAttributes;
801 std::string featValue = featureItr->getValueString();
802 auto feature = std::make_tuple(index,featName,featValue);
803 mapAttributes.insert( std::make_pair(feature, feat_index ) );
804 }
805 }
806 }
807 outputAttributesString(out, index, mapAttributes);
808
810 LDEBUG << "AbstractIEDumper: add entity (" << get<0>(entity) << "," << get<1>(entity) << "," << get<2>(entity) << "): index=" << index;
811
812 mapEntities.insert(std::make_pair(entity,index));
813 return true;
814 }
815 return false;
816}
817
820 const Common::AnnotationGraphs::AnnotationData* annotationData) const
821{
822
823 const SpecificEntityAnnotation* se=0;
824
825 // check only entity found in current graph
826
827 std::set< AnnotationGraphVertex > matches = annotationData->matches(m_graph,v,"annot");
828 for (std::set< AnnotationGraphVertex >::const_iterator it = matches.begin();
829 it != matches.end(); it++)
830 {
832 if (annotationData->hasAnnotation(vx, QString::fromUtf8("SpecificEntity")))
833 {
834 //BoWToken* se = createSpecificEntity(v,*it, annotationData, anagraph, posgraph, offsetBegin);
835 se = annotationData->annotation(vx, QString::fromUtf8("SpecificEntity")).
836 pointerValue<SpecificEntityAnnotation>();
837 if (se!=0) {
838 return se;
839 }
840 }
841 }
842
843 // special case: if specified graph is posgraph, allows to search in analysis graph
844 // (entities found before pos-tagging)
845 if (m_graph=="PosGraph") {
846 std::set< AnnotationGraphVertex > anaVertices = annotationData->matches("PosGraph",v,"AnalysisGraph");
847
848 // note: anaVertices size should be 0 or 1
849 for (const auto& anaVertex : anaVertices) {
850
851 std::set< AnnotationGraphVertex > matches = annotationData->matches("AnalysisGraph",anaVertex,"annot");
852
853 for (const auto& vx: matches)
854 {
855 if (annotationData->hasAnnotation(vx, QString::fromUtf8("SpecificEntity")))
856 {
857 se = annotationData->annotation(vx, QString::fromUtf8("SpecificEntity")).
858 pointerValue<SpecificEntityAnnotation>();
859 if (se!=0) {
860 return se;
861 }
862 }
863 }
864 }
865 }
866
867 return se;
868
869}
870
872outputSemanticRelationArg(const std::string& /*vertexRole*/,
873 const AnnotationGraphVertex& vertex,
874 const VertexTokenPropertyMap& tokenMap,
875 std::map<std::tuple<std::uint64_t,std::uint64_t,std::string>,std::size_t > mapEntities,
876 const Common::AnnotationGraphs::AnnotationData* annotationData,
877 uint64_t offset) const
878{
879 ostringstream oss;
880
881 // get id of the corresponding vertex in analysis graph
883 if (!annotationData->hasIntAnnotation(vertex,QString::fromStdString(m_graph)))
884 {
885 // DUMPERLOGINIT;
886 // LDEBUG << *itv << " has no " << m_graph << " annotation. Skeeping it.";
887 return "";
888 }
889 v = annotationData->intAnnotation(vertex,QString::fromStdString(m_graph));
890 LinguisticAnalysisStructure::Token* vToken = tokenMap[v];
891 // LDEBUG << "SemanticRelationsXmlLogger tokenMap[" << v << "] = " << vToken;
892 if (vToken == 0)
893 {
894 return "";
895 }
896
897 // get annotation : element in relation can be an entity => get entity type
898 // otherwise, its type is "token"
899 //EntityT type("token");
900
901 //std::set< uint32_t > matches = annotationData->matches(m_graph,v,"annot");
902 std::set< AnnotationGraphVertex > matches = annotationData->matches(m_graph,v,"annot");
903 for (std::set< AnnotationGraphVertex >::const_iterator it = matches.begin();
904 it != matches.end(); it++)
905 {
906 if (annotationData->hasAnnotation(*it,QString::fromUtf8("SpecificEntity"))) {
907 const SpecificEntityAnnotation* annot = 0;
908 try {
909 annot = annotationData->annotation(*it,QString::fromUtf8("SpecificEntity"))
910 .pointerValue<SpecificEntityAnnotation>();
911 }
912 catch (const boost::bad_any_cast& e) {
913
914 continue;
915 }
917 std::uint64_t position= offset+vToken->position() ;
918 std::uint64_t length= vToken->length();
919 for(std::map<std::tuple<std::uint64_t,std::uint64_t,std::string>,std::size_t > ::const_iterator itE=mapEntities.begin();itE!=mapEntities.end();itE++)
920 {
921 std::tuple<std::uint64_t,std::uint64_t,std::string> entityIndex= itE->first;
922 if (std::get<0>(entityIndex)==position &&
923 std::get<1>(entityIndex)==length &&
924 typeName.compare(LimaString(std::get<2>(entityIndex).c_str()))==0) oss << itE->second;
925
926 }
927 break;
928 }
929 }
930
931
932 return oss.str();
933}
934
936outputSemanticRelations(std::ostream& out,
937 const Common::AnnotationGraphs::AnnotationData* annotationData,
938 const VertexTokenPropertyMap& tokenMap,
939 std::map<std::tuple<std::uint64_t,std::uint64_t,std::string>,std::size_t > mapEntities,
940 uint64_t offset,
941 IEDumperMetaData* metadata
942 ) const
943{
944 // return the number of relations produced in ouptut
945 uint nbRelations(0);
946
947 AnnotationGraphEdgeIt it,it_end;
948 std::uint64_t index=1+metadata->nbRelations;
949 const AnnotationGraph& annotGraph=annotationData->getGraph();
950 boost::tie(it, it_end) = edges(annotGraph);
951 for (; it != it_end; it++) {
952
953 if (annotationData->hasAnnotation(*it,QString::fromUtf8("SemanticRelation")))
954 {
955
956 const SemanticRelationAnnotation* annot = 0;
957 try
958 {
959 annot = annotationData->annotation(*it,QString::fromUtf8("SemanticRelation"))
961 }
962 catch (const boost::bad_any_cast& e)
963 {
964
965 continue;
966 }
967
968 std::string annotType( annot->type() );
969 std::size_t posG=annotType.find(".");
970 if (posG!=std::string::npos && ! m_outputGroups){
971 annotType = annotType.substr(posG+1);
972 }
973 //output
975 LDEBUG << "AbstractIEDumper: add relation (type" << annotType << ", source" << source(*it,annotGraph) << ", target " << target(*it,annotGraph) << "): index=" << index;
976 index += outputRelationString(out, index, annotType,
977 outputSemanticRelationArg("source",source(*it,annotGraph),tokenMap,mapEntities,annotationData,offset),
978 outputSemanticRelationArg("target",target(*it,annotGraph),tokenMap,mapEntities,annotationData,offset));
979 nbRelations++;
980 }
981 }
982 return nbRelations;
983}
984
987 const Common::AnnotationGraphs::AnnotationData* annotationData) const
988{
989
990 const SemanticRelationAnnotation* sr=0;
991
992 // check only entity found in current graph (not previous graph such as AnalysisGraph)
993
994 std::set< AnnotationGraphVertex > matches = annotationData->matches(m_graph,v,"annot");
995 for (std::set< AnnotationGraphVertex >::const_iterator it = matches.begin();
996 it != matches.end(); it++)
997 {
999 if (annotationData->hasAnnotation(vx, QString::fromUtf8("SemanticRelation")))
1000 {
1001 //BoWToken* se = createSpecificEntity(v,*it, annotationData, anagraph, posgraph, offsetBegin);
1002 sr = annotationData->annotation(vx, QString::fromUtf8("SemanticRelation")).
1003 pointerValue<SemanticRelationAnnotation>();
1004 if (sr!=0) {
1005 return sr;
1006 }
1007 }
1008 }
1009 return sr;
1010
1011}
1012
1013void AbstractIEDumper::adjustPosition(std::uint64_t& position, uint64_t offset) const
1014{
1015 //DUMPERLOGINIT;
1016// std::uint64_t prevPos(position);
1017 if (m_offsetMapping!=0) {
1018 position=m_offsetMapping->getOriginalOffset(position)+offset;
1019 }
1020 if (offset) {
1021 position+=offset;
1022 }
1023
1024 //LDEBUG << "AbstractIEDumper::adjustPosition" << prevPos << "->" << position;
1025}
1026
1027void AbstractIEDumper::adjustPositions(std::vector<pair<uint64_t,uint64_t> >& positions, uint64_t offset) const
1028{
1030 ostringstream prevPos;
1031 for (const auto& p: positions) { prevPos << "(" << p.first << "," << p.second << ") "; }
1032 if (m_offsetMapping) {
1033 for (auto& p: positions) {
1034 p.first=m_offsetMapping->getOriginalOffset(p.first);
1035 p.second=m_offsetMapping->getOriginalOffset(p.second);
1036 }
1037 }
1038 if (offset) {
1039 for (auto& p: positions) {
1040 p.first+=offset;
1041 p.second+=offset;
1042 }
1043 }
1044
1045 ostringstream newPos;
1046 for (const auto& p: positions) { newPos << "(" << p.first << "," << p.second << ") "; }
1047 LDEBUG << "AbstractIEDumper::adjustPositions" << prevPos.str() << "->" << newPos.str();
1048}
1049
1050
1051} // AnalysisDumpers
1052} // LinguisticProcessing
1053} // Lima
This file is the main header file for the data related to annotation graphs.
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, AGVertexProperties, AGEdgeProperties > AnnotationGraph
The graph class.
#define UNDEFPOSITION
#define LWARN
Definition LimaCommon.h:160
#define LDEBUG
Definition LimaCommon.h:157
#define LERROR
Definition LimaCommon.h:161
boost::property_map< LinguisticGraph, vertex_token_t >::type VertexTokenPropertyMap
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define DUMPERLOGINIT
Defines a Factory to create Object of type Base.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
Holds an annotation graph and gives an API to manipulate it.
uint64_t intAnnotation(AnnotationGraphVertex v1, AnnotationGraphVertex v2, const LimaString &annot) const
std::set< AnnotationGraphVertex > matches(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second) const
Gets the set of vertices matched in the second graph by the given vertex of the first graph.
const GenericAnnotation & annotation(AnnotationGraphVertex v1, AnnotationGraphVertex v2, const LimaString &annot) const
LimaString getEntityName(const EntityType &type) const
std::deque< std::string > & getListsValueAtKey(const std::string &key)
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
std::shared_ptr< DumperStream > initialize(AnalysisContent &analysis) const
virtual void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
std::string outputSemanticRelationArg(const std::string &vertexRole, const AnnotationGraphVertex &vertex, const VertexTokenPropertyMap &tokenMap, std::map< std::tuple< std::uint64_t, std::uint64_t, std::string >, std::size_t > mapEntities, const Common::AnnotationGraphs::AnnotationData *annotationData, uint64_t offset) const
virtual void outputGlobalHeader(std::ostream &, const std::string &="", const LimaStringText &=LimaStringText("")) const
specific output functions to deal with actual output format these functions are called inside the pro...
virtual void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
uint outputEventData(std::ostream &out, const EventAnalysis::EventTemplateData *eventData, const Common::AnnotationGraphs::AnnotationData *annotationData, const VertexTokenPropertyMap &tokenMap, std::map< std::tuple< std::uint64_t, std::uint64_t, std::string >, std::size_t > mapEntities, uint64_t offset, LimaStringText originalText, IEDumperMetaData *metadata) const
const SpecificEntities::SpecificEntityAnnotation * getSpecificEntityAnnotation(LinguisticGraphVertex v, const Common::AnnotationGraphs::AnnotationData *annotationData) const
virtual unsigned int outputRelationString(std::ostream &out, unsigned int relationId, const std::string &relationType, const std::string &sourceArgString, const std::string &targetArgString) const =0
virtual void outputEventString(std::ostream &out, unsigned int eventId, unsigned int eventMentionId, const std::string &eventMentionType, const std::vector< unsigned int > &eventRoleId, const std::vector< std::string > &eventRoleType) const =0
bool outputEntity(std::ostream &out, LinguisticGraphVertex v, const SpecificEntities::SpecificEntityAnnotation *annot, const VertexTokenPropertyMap &tokenMap, uint64_t offset, LimaStringText originalText, std::map< std::tuple< uint64_t, uint64_t, std::string >, std::size_t > &mapEntities, std::map< std::tuple< std::size_t, std::string, std::string >, std::size_t > &mapAttributes, IEDumperMetaData *metadata) const
virtual void outputEntityString(std::ostream &out, unsigned int entityId, const std::string &entityType, const std::string &entityString, const std::vector< std::pair< uint64_t, uint64_t > > &positions, const Automaton::EntityFeatures &entityFeatures, bool noNorm=false) const =0
uint outputSemanticRelations(std::ostream &out, const Common::AnnotationGraphs::AnnotationData *annotationData, const VertexTokenPropertyMap &tokenMap, std::map< std::tuple< std::uint64_t, std::uint64_t, std::string >, std::size_t > mapEntities, uint64_t offset, IEDumperMetaData *metadata) const
void adjustPositions(std::vector< std::pair< uint64_t, uint64_t > > &positions, uint64_t offset=0) const
virtual void outputAttributesString(std::ostream &out, unsigned int entityId, std::map< std::tuple< std::size_t, std::string, std::string >, std::size_t > &mapAttributes) const =0
virtual LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
void computePositions(std::vector< std::pair< uint64_t, uint64_t > > &positions, LimaString &stringForm, uint64_t pos, uint64_t len) const
std::map< std::string, std::shared_ptr< EventAnalysis::EventTemplateDefinitionResource > > m_templateDefinitions
const SemanticAnalysis::SemanticRelationAnnotation * getSemanticRelationAnnotation(LinguisticGraphVertex v, const Common::AnnotationGraphs::AnnotationData *annotationData) const
void adjustPosition(std::uint64_t &position, uint64_t offset=0) const
unsigned int getOriginalOffset(unsigned int newOffset) const
a list of generic features: each feature is unique (only one feature for a name)
define the limaStringText resource which is the text in limaString formal
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
contains metadata to be kept during the analysis: global generic metadata: file name,...
bool hasMetaData(const std::string &id) const
void setMetaData(const std::string &id, const std::string &value)
const std::string & getMetaData(const std::string &id) const
An annotation between two annotation graph vertices denoting the semantic relation(s) holding between...
A representation of a specific entity to store in the annotation graph.
static const LinguisticResources & single()
const singleton accessor
Definition Singleton.h:51
static void logElapsedTime(const std::string &mess, const std::string &taskCategory=std::string(""))
log the number of microseconds since last UpdateCurrentTime
static void updateCurrentTime(const std::string &taskCategory=std::string(""))
store current time for new elapsed time computation
AnnotationGraph::vertex_iterator AnnotationGraphVertexIt
AnnotationGraph::edge_iterator AnnotationGraphEdgeIt
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
bool hasIntAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
bool hasAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
std::multimap< std::string, EventTemplateElement > TemplateElements
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
@ MISSING_DATA
Definition LimaCommon.h:243
QString LimaString
Definition LimaString.h:33
STL namespace.
friend std::ostream & operator<<(std::ostream &os, IEDumperMetaData &m)
friend std::ostream & operator<<(std::ostream &os, const EventInfos e)
friend QDebug & operator<<(QDebug &os, const EventInfos e)