LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
GenericXmlDumper.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/***************************************************************************
7 * Copyright (C) 2010 by CEA LIST *
8 * *
9 ***************************************************************************/
10#include "GenericXmlDumper.h"
11#include "TextDumper.h" // for lTokenPosition comparison function to order tokens
12
13// #include "linguisticProcessing/core/LinguisticProcessors/HandlerStreamBuf.h"
30
31#include <boost/graph/properties.hpp>
32
33#include <fstream>
34#include <deque>
35#include <queue>
36#include <iostream>
37
38using namespace std;
39//using namespace boost;
40using namespace boost::tuples;
42using namespace Lima::Common::MediaticData;
43using namespace Lima::Common::AnnotationGraphs;
44using namespace Lima::Common::BagOfWords;
48
49namespace Lima {
50namespace LinguisticProcessing {
51namespace AnalysisDumpers {
52
54
57m_graph("PosGraph"),
58m_features(),
59m_bowFeatures(),
60m_featureNames(),
61m_featureTags(),
62m_defaultFeatures(),
63m_outputWords(true),
64m_outputSentenceBoundaries(false),
65m_outputSpecificEntities(false),
66m_outputSpecificEntityParts(false),
67m_outputCompounds(false),
68m_outputCompoundParts(false),
69m_outputAllCompounds(false),
70m_wordTag("w"),
71m_sentenceBoundaryTag("s"),
72m_specificEntityTag("e"),
73m_compoundTag("c"),
74m_bowGenerator(0)
75{
76 // default features
77 m_defaultFeatures["p"]="position";
78 m_defaultFeatures["inf"]="word";
79 m_defaultFeatures["pos"]="property:MICRO";
80 m_defaultFeatures["lemma"]="lemma";
81}
82
83//@todo : copy constructor: deal with copy of pointers
84
93
95 for (WordFeatures::const_iterator it=m_features.begin(),it_end=m_features.end();it!=it_end;it++) {
96 if (*it!=0) {
97 delete *it;
98 }
99 }
100 m_features.clear();
101 m_bowFeatures.clear();
102 m_featureTags.clear();
103}
104
107 Manager* manager)
108
109{
111 AbstractTextualAnalysisDumper::init(unitConfiguration,manager);
112
113 try
114 {
115 m_graph=unitConfiguration.getParamsValueAtKey("graph");
116 }
117 catch (NoSuchParam& ) {} // keep default value
118
121 try {
122 // if some features are specified, all must be specified: do not keep any default ones
123 map<string,string> featuresMap=unitConfiguration.getMapAtKey("features");
124 try {
125 // order can be specified (not in map: map is unordered)
126 deque<string> featureOrder=unitConfiguration.getListsValueAtKey("featureOrder");
127 initializeFeatures(featuresMap,featureOrder);
128 }
129 catch (NoSuchList& ) { // no order specified: keep order from map
130 initializeFeatures(featuresMap);
131 }
132 }
133 catch (NoSuchMap& ) {
134 //initialize with default values
136 }
137
138 try {
139 string str=unitConfiguration.getParamsValueAtKey("words");
140 if (str=="no" || str=="false" || str=="") {
141 m_outputWords=false;
142 }
143 else {
144 m_outputWords=true;
145 m_wordTag=str;
146 LDEBUG << "GenericXmlDumper: outputSpecificEntities set to true (tag is " << str << ")";
147 }
148 }
149 catch (NoSuchParam& ) {// optional : do not output entities if param not specified
150 }
151
152 try {
153 string str=unitConfiguration.getParamsValueAtKey("specificEntities");
154 if (str!="no" && str!="false" && str!="") {
157 LDEBUG << "GenericXmlDumper: outputSpecificEntities set to true (tag is " << str << ")";
158 }
159 }
160 catch (NoSuchParam& ) {// optional : do not output entities if param not specified
161 }
162
163 try {
164 string str=unitConfiguration.getParamsValueAtKey("specificEntityParts");
165 if (str!="no" && str!="false" && str!="") {
167 }
168 }
169 catch (NoSuchParam& ) { }// optional : do not output entities if param not specified
170
171 try {
172 string str=unitConfiguration.getParamsValueAtKey("sentenceBoundaries");
173 if (str!="no" && str!="false" && str!="") {
176 LDEBUG << "GenericXmlDumper: outputSentenceBoundaries set to true (tag is " << str << ")";
177 }
178 }
179 catch (NoSuchParam& ) {} // optional : do not output entities if param not specified
180
181 try {
182 string str=unitConfiguration.getParamsValueAtKey("compounds");
183 if (str!="no" && str!="false" && str!="") {
185 m_compoundTag=str;
186 // initialize compound creator
188 m_bowGenerator->init(unitConfiguration, m_language);
189 LDEBUG << "GenericXmlDumper: outputCompounds set to true (tag is " << str << ")";
190 }
191 }
192 catch (NoSuchParam& ) {} // optional : do not output entities if param not specified
193
194 try {
195 string str=unitConfiguration.getParamsValueAtKey("compoundParts");
196 if (str!="no" && str!="false" && str!="") {
198 }
199 }
200 catch (NoSuchParam& ) { }// optional : do not output entities if param not specified
201
202 try {
203 string str=unitConfiguration.getParamsValueAtKey("allCompounds");
204 if (str!="no" && str!="false" && str!="") {
206 }
207 }
208 catch (NoSuchParam& ) { }// optional : do not output entities if param not specified
209
210
211/* const Common::PropertyCode::PropertyCodeManager& codeManager=static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager();
212 m_propertyAccessor=&codeManager.getPropertyAccessor(m_property);
213 m_propertyManager=&codeManager.getPropertyManager(m_property);
214
215 try {
216 std::string str=unitConfiguration.getParamsValueAtKey("outputTStatus");
217 if (str=="yes" || str=="1") {
218 m_outputTStatus=true;
219 }
220 }
221 catch (NoSuchParam& ) {} // keep default value
222
223 try {
224 std::string str=unitConfiguration.getParamsValueAtKey("outputVerbTense");
225 if (str=="yes" || str=="1") {
226 m_outputVerbTense=true;
227 m_tenseAccessor=&codeManager.getPropertyAccessor("TIME");
228 m_tenseManager=&codeManager.getPropertyManager("TIME");
229 }
230 }
231 catch (NoSuchParam& ) {} // keep default value
232*/
233}
234
235void GenericXmlDumper::initializeFeatures(const map<string,string>& featuresMap,
236 const std::deque<std::string>& featureOrder)
237{
239 bool useMapOrder(false);
240 if (! featureOrder.empty()) {
241 // use specified order
243 LDEBUG << "GenericXmlDumper: initialize features: use order";
244 for (deque<string>::const_iterator it=featureOrder.begin(),it_end=featureOrder.end();it!=it_end;it++) {
245 LDEBUG << "GenericXmlDumper: --"<< (*it);
246 const std::string& featureTag=(*it);
247 m_featureTags.push_back(featureTag);
248 map<string,string>::const_iterator f=featuresMap.find(featureTag);
249 if (f==featuresMap.end()) {
251 LWARN << "GenericXmlDumper: 'featureOrder' parameter mentions a feature '" << featureTag
252 << "' not in feature map parameter: order ignored";
253 useMapOrder=true;
254 m_featureNames.clear();
255 m_featureTags.clear();
256 break;
257 }
258 m_featureNames.push_back((*f).second);
259 }
260 }
261 else {
262 useMapOrder=true;
263 }
264
265 if (useMapOrder) {
266 for (map<string,string>::const_iterator it=featuresMap.begin(),it_end=featuresMap.end();it!=it_end; it++) {
267 const std::string& featureName=(*it).second;
268 const std::string& featureTag=(*it).first;
269 m_featureNames.push_back(featureName);
270 m_featureTags.push_back(featureTag);
271 }
272 }
275 if (m_features.size()!=m_featureTags.size() || m_bowFeatures.size()!=m_featureTags.size()) {
277 LERROR << "GenericXmlDumper: error: failed to initialize all features";
278 throw InvalidConfiguration();
279 }
280}
281
283process(AnalysisContent& analysis) const
284{
287 LDEBUG << "GenericXmlDumper::process";
288
289 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.getData("LinguisticMetaData"));
290 if (metadata == 0)
291 {
292 LERROR << "no LinguisticMetaData ! abort";
293 return MISSING_DATA;
294 }
295
296 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"));
297 if (anagraph==0)
298 {
299 LERROR << "no graph 'AnaGraph' available !";
300 return MISSING_DATA;
301 }
302 auto posgraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("PosGraph"));
303 if (posgraph==0)
304 {
305 LERROR << "no graph 'PosGraph' available !";
306 return MISSING_DATA;
307 }
308 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.getData("AnnotationData"));
309 if (annotationData==0)
310 {
311 LERROR << "no annotation graph available !";
312 return MISSING_DATA;
313 }
314
315 std::shared_ptr<SyntacticData> syntacticData;
317 syntacticData = std::dynamic_pointer_cast< SyntacticData >(analysis.getData("SyntacticData"));
318 if (annotationData==0)
319 {
320 LWARN << "compounds are supposed to be printed in output but no syntactic data available !";
321 }
322 }
323
324 auto dstream = initialize(analysis);
325 xmlOutput(dstream->out(), analysis, anagraph.get(), posgraph.get(), annotationData.get(), syntacticData.get());
326
327
328 TimeUtils::logElapsedTime("GenericXmlDumper");
329 return SUCCESS_ID;
330}
331
333xmlOutput(std::ostream& out,
334 AnalysisContent& analysis,
335 AnalysisGraph* anagraph,
336 AnalysisGraph* posgraph,
337 const Common::AnnotationGraphs::AnnotationData* annotationData,
338 const SyntacticAnalysis::SyntacticData* syntacticData) const
339{
341
342 out << "<text>" << endl;
343
344 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.getData("LinguisticMetaData"));
345
347
348 std::shared_ptr<SegmentationData> sb;
350 sb = std::dynamic_pointer_cast<SegmentationData>(analysis.getData("SentenceBoundaries"));
351 if (sb==0) {
352 LWARN << "GenericXmlDumper:: no SentenceBoundaries";
353 }
354 }
355
356 if (sb==0)
357 {
358 // no sentence bounds : dump all text at once
360 analysis,
361 anagraph,
362 posgraph,
363 annotationData,
364 syntacticData,
365 anagraph->firstVertex(),
366 anagraph->lastVertex(),
367 sp,
368 metadata->getStartOffset());
369 }
370 else
371 {
372 // ??OME2 uint64_t nbSentences(sb->size());
373 uint64_t nbSentences((sb->getSegments()).size());
374 LDEBUG << "GenericXmlDumper: "<< nbSentences << " sentences found";
375 for (uint64_t i=0; i<nbSentences; i++)
376 {
377 // ??OME2 LinguisticGraphVertex sentenceBegin=(*sb)[i].getFirstVertex();
378 // LinguisticGraphVertex sentenceEnd=(*sb)[i].getLastVertex();
379 LinguisticGraphVertex sentenceBegin=(sb->getSegments())[i].getFirstVertex();
380 LinguisticGraphVertex sentenceEnd=(sb->getSegments())[i].getLastVertex();
381
382 // if (sentenceEnd==posgraph->lastVertex()) {
383 // continue;
384 // }
385
386 LDEBUG << "dump sentence between " << sentenceBegin << " and " << sentenceEnd;
387 LDEBUG << "dump simple terms for this sentence";
388
389 ostringstream oss;
391 analysis,
392 anagraph,
393 posgraph,
394 annotationData,
395 syntacticData,
396 sentenceBegin,
397 sentenceEnd,
398 sp,
399 metadata->getStartOffset());
400 string str=oss.str();
401 if (str.empty()) {
402 LDEBUG << "nothing to dump in this sentence";
403 }
404 else {
405 out << "<" << m_sentenceBoundaryTag << " id=\"" << i << "\">" << endl
406 << str
407 << "</"<< m_sentenceBoundaryTag << ">" << endl;
408 }
409 }
410 }
411 out << "</text>" << endl;
412}
413
415xmlOutputVertices(std::ostream& out,
416 AnalysisContent& analysis,
417 AnalysisGraph* anagraph,
418 AnalysisGraph* posgraph,
419 const Common::AnnotationGraphs::AnnotationData* annotationData,
420 const SyntacticAnalysis::SyntacticData* syntacticData,
421 const LinguisticGraphVertex begin,
422 const LinguisticGraphVertex end,
423 const FsaStringsPool& sp,
424 const uint64_t offset) const
425{
426
428 LDEBUG << "GenericXmlDumper: ========================================";
429 LDEBUG << "GenericXmlDumper: outputXml from vertex " << begin << " to vertex " << end;
430
431 LinguisticGraph* graph=posgraph->getGraph();
432 LinguisticGraphVertex lastVertex=posgraph->lastVertex();
433
434 map<Token*, vector<LinguisticGraphVertex>, lTokenPosition> sortedTokens;
435
436 std::queue<LinguisticGraphVertex> toVisit;
437 std::set<LinguisticGraphVertex> visited;
438
439 LinguisticGraphOutEdgeIt outItr,outItrEnd;
440
441 // output vertices between begin and end,
442 // but do not include begin (beginning of text or previous end of sentence) and include end (end of sentence)
443 toVisit.push(begin);
444
445 bool first=true;
446 bool last=false;
447 while (!toVisit.empty()) {
448 LinguisticGraphVertex v=toVisit.front();
449 toVisit.pop();
450 if (last || v == lastVertex) {
451 continue;
452 }
453 if (v == end) {
454 last=true;
455 }
456
457 for (boost::tie(outItr,outItrEnd)=out_edges(v,*graph); outItr!=outItrEnd; outItr++)
458 {
459 LinguisticGraphVertex next=target(*outItr,*graph);
460 if (visited.find(next)==visited.end())
461 {
462 visited.insert(next);
463 toVisit.push(next);
464 }
465 }
466
467 if (first) {
468 first=false;
469 }
470 else {
471 Token* t=get(vertex_token,*graph,v);
472 if( t!=0) {
473 sortedTokens[t].push_back(v);
474 }
475 }
476 }
477
478 // for compounds
479 std::set<LinguisticGraphVertex> alreadyStoredVertices;
480
481 // store outputs, sorting them by their positions (useful for compounds)
482 map<uint64_t,vector<string> > xmlOutputs;
483
484 for (map< Token*,vector<LinguisticGraphVertex>,lTokenPosition >::const_iterator
485 it=sortedTokens.begin(),it_end=sortedTokens.end(); it!=it_end; it++)
486 {
487 const vector<LinguisticGraphVertex>& vertices=(*it).second;
488 if (vertices.size()==0) {
490 LERROR << "GenericXmlDumper: no vertices for token " << (*it).first->stringForm();
491 continue;
492 }
493
494 for (vector<LinguisticGraphVertex>::const_iterator d=vertices.begin(),
495 d_end=vertices.end(); d!=d_end; d++) {
496
497 /*if (alreadyStoredVertices.find(*d)!=alreadyStoredVertices.end()) {
498 // if already printed as a compound part
499 continue;
500 }*/
501 ostringstream oss;
502 xmlOutputVertex(oss,analysis,(*d),anagraph,posgraph,annotationData,syntacticData,
503 sp,offset,visited,alreadyStoredVertices);
504 uint64_t pos=(*it).first->position();
505 xmlOutputs[pos].push_back(oss.str());
506 }
507 }
508
509 for (map<uint64_t,vector<string> >::const_iterator it=xmlOutputs.begin(),it_end=xmlOutputs.end();it!=it_end;it++) {
510 for (vector<string>::const_iterator s=(*it).second.begin(),s_end=(*it).second.end();s!=s_end;s++) {
511 out << *s;
512 }
513 }
514
515}
516
518xmlOutputVertex(std::ostream& out,
519 AnalysisContent& analysis,
521 AnalysisGraph* anagraph,
522 AnalysisGraph* posgraph,
523 const Common::AnnotationGraphs::AnnotationData* annotationData,
524 const SyntacticAnalysis::SyntacticData* syntacticData,
525 const FsaStringsPool& sp,
526 uint64_t offset,
527 set<LinguisticGraphVertex>& visited,
528 std::set<LinguisticGraphVertex>& alreadyStoredVertices) const
529{
531 LDEBUG << "GenericXmlDumper: output vertex " << v;
532
533 // check if specific entity, event if m_outputSpecificEntities is false, to access parts
534 // and print them as simple words
535 std::pair<const SpecificEntityAnnotation*,AnalysisGraph*>
536 se=checkSpecificEntity(v,anagraph,posgraph,annotationData);
537 if (se.first!=0) {
538 LDEBUG << "GenericXmlDumper: -- is a specific entity ";
539 if (xmlOutputSpecificEntity(out,analysis,se.first,se.second,sp,offset)) {
540 return;
541 }
542 else {
544 LERROR << "failed to output specific entity for vertex " << v;
545 }
546 }
547
548 if (m_outputCompounds) {
549 // check if the word is head of a compound
550 std::vector< boost::shared_ptr< BoWToken > > compoundTokens=
551 checkCompound(v, anagraph, posgraph, annotationData, syntacticData, offset, visited);
552 if (compoundTokens.size()!=0) {
553 for (auto it=compoundTokens.begin(), it_end=compoundTokens.end();it!=it_end;it++) {
554
555 xmlOutputCompound(out,analysis,(*it),anagraph,posgraph,annotationData,sp,offset);
556 std::set<uint64_t> bowTokenVertices = (*it)->getVertices();
557 alreadyStoredVertices.insert(bowTokenVertices.begin(), bowTokenVertices.end());
558 }
559 }
560 }
561
562 LDEBUG << "GenericXmlDumper: -- is simple word ";
563 // if not a specific entity nor a compound, output simple word infos
564 if (m_outputWords) {
565 xmlOutputVertexInfos(out, analysis, v, posgraph, offset);
566 }
567}
568
569std::pair<const SpecificEntityAnnotation*,AnalysisGraph*>
572 AnalysisGraph* posgraph,
573 const Common::AnnotationGraphs::AnnotationData* annotationData) const
574{
575 // first, check if vertex corresponds to a specific entity found before pos tagging (i.e. in analysis graph)
576 std::set< AnnotationGraphVertex > anaVertices = annotationData->matches("PosGraph",v,"AnalysisGraph");
577 // note: anaVertices size should be 0 or 1
578 for (std::set< AnnotationGraphVertex >::const_iterator anaVerticesIt = anaVertices.begin();
579 anaVerticesIt != anaVertices.end(); anaVerticesIt++)
580 {
581 std::set< AnnotationGraphVertex > matches = annotationData->matches("AnalysisGraph",*anaVerticesIt,"annot");
582 for (std::set< AnnotationGraphVertex >::const_iterator it = matches.begin();
583 it != matches.end(); it++)
584 {
586 if (annotationData->hasAnnotation(vx, Common::Misc::utf8stdstring2limastring("SpecificEntity")))
587 {
588 const SpecificEntityAnnotation* se =
589 annotationData->annotation(vx, Common::Misc::utf8stdstring2limastring("SpecificEntity")).
590 pointerValue<SpecificEntityAnnotation>();
591 return make_pair(se,anagraph);
592 }
593 }
594 }
595
596 // then check if vertex corresponds to a specific entity found after POS tagging
597 std::set< AnnotationGraphVertex > matches = annotationData->matches("PosGraph",v,"annot");
598 for (std::set< AnnotationGraphVertex >::const_iterator it = matches.begin();
599 it != matches.end(); it++)
600 {
602 if (annotationData->hasAnnotation(vx, Common::Misc::utf8stdstring2limastring("SpecificEntity")))
603 {
604 //BoWToken* se = createSpecificEntity(v,*it, annotationData, anagraph, posgraph, offsetBegin);
605 const SpecificEntityAnnotation* se =
606 annotationData->annotation(vx, Common::Misc::utf8stdstring2limastring("SpecificEntity")).
607 pointerValue<SpecificEntityAnnotation>();
608 return make_pair(se,posgraph);
609 }
610 }
611 return std::pair<const SpecificEntityAnnotation*,AnalysisGraph*>((const SpecificEntityAnnotation*)0,(AnalysisGraph*)0);
612}
613
615xmlOutputSpecificEntity(std::ostream& out,
616 AnalysisContent& analysis,
619 const FsaStringsPool& sp,
620 uint64_t offset) const
621{
622 if (se == 0) {
624 LERROR << "missing specific entity annotation";
625 return false;
626 }
627
628 // output enclosing tag for entity with associated information
630 out << "<" << m_specificEntityTag;
631 for (unsigned int i=0,size=m_featureNames.size();i<size;i++) {
632 // try to get value directly from specific entity (defined for some features)
633 string value=xmlString(specificEntityFeature(se,m_featureNames[i],sp,offset));
634 if (value.empty()) {
635 // otherwise, get features from head
636 value=xmlString(m_features[i]->getValue(graph,se->getHead(),analysis));
637 }
638 out << " " << m_featureTags[i] << "=\"" << value << "\"";
639 }
640 //<< " inf=\"" << xmlString(Common::Misc::limastring2utf8stdstring(sp[se->getString()])) << "\""
641
643 // output tag as enclosing tag, with parts enclosed
644 out << ">" << endl;
645 for (std::vector< LinguisticGraphVertex>::const_iterator m(se->vertices().begin());
646 m != se->vertices().end(); m++)
647 {
648 xmlOutputVertexInfos(out,analysis,(*m),graph,offset);
649 }
650 out << "</" << m_specificEntityTag << ">" << endl;
651 }
652 else {
653 // output only the named entity tag
654 out << "/>" << endl;
655 }
656 }
657 else {
658 // output parts as simple words
659 for (std::vector< LinguisticGraphVertex>::const_iterator m(se->vertices().begin());
660 m != se->vertices().end(); m++)
661 {
662 xmlOutputVertexInfos(out,analysis,(*m),graph,offset);
663 }
664 }
665
666 // take as category for parts the category for the named entity
667 /*LinguisticCode category=m_propertyAccessor->readValue(data->begin()->properties);
668 DUMPERLOGINIT;
669 LDEBUG << "Using category " << m_propertyManager->getPropertySymbolicValue(category) << " for specific entity of type " << typeName;
670 */
671
672 return true;
673
674}
675
676std::vector< boost::shared_ptr< BoWToken > > GenericXmlDumper::
678 AnalysisGraph* anagraph,
679 AnalysisGraph* posgraph,
680 const Common::AnnotationGraphs::AnnotationData* annotationData,
681 const SyntacticAnalysis::SyntacticData* syntacticData,
682 uint64_t offset,
683 set<LinguisticGraphVertex>& visited) const
684{
686 LDEBUG << "GenericXmlDumper: check if compound for vertex " << v;
687
688 std::set< AnnotationGraphVertex > cpdsHeads = annotationData->matches("PosGraph", v, "cpdHead");
689 if (cpdsHeads.empty())
690 {
691 // not a compound
692 return std::vector< boost::shared_ptr< BoWToken > >();
693 }
694
695 LDEBUG << "GenericXmlDumper: -- is head of a compound ";
696 std::vector< boost::shared_ptr< BoWToken > > tokens;
697 std::set< std::string > alreadyStored;
698 for (std::set< AnnotationGraphVertex >::const_iterator it=cpdsHeads.begin(), it_end=cpdsHeads.end();
699 it!=it_end; it++)
700 {
701 const AnnotationGraphVertex& agv=*it;
702
703 // create compound using BoWGeneration : store in BoW
704 std::vector<std::pair<boost::shared_ptr< BoWRelation >, boost::shared_ptr<BoWToken> > > bowTokens =
705 m_bowGenerator->buildTermFor(agv, agv, *(anagraph->getGraph()), *(posgraph->getGraph()), offset,
706 syntacticData, annotationData, visited);
707 for (auto bowItr=bowTokens.begin();
708 bowItr!=bowTokens.end(); bowItr++)
709 {
710 std::string elem = (*bowItr).second->getIdUTF8String();
711 if (alreadyStored.find(elem) != alreadyStored.end())
712 {
713 // already stored
714 // LDEBUG << "BuildBoWTokenListVisitor: BoWToken already stored. Skipping it.";
715 }
716 else {
717 tokens.push_back((*bowItr).second);
718 alreadyStored.insert(elem);
719 }
720 }
721 }
722 return tokens;
723}
724
726xmlOutputCompound(std::ostream& out,
727 AnalysisContent& analysis,
728 boost::shared_ptr<Common::BagOfWords::AbstractBoWElement> token,
731 const AnnotationData* annotationData,
732 const FsaStringsPool& sp,
733 uint64_t offset) const
734{
736 LDEBUG << "GenericXmlDumper: output BoWToken [" << token->getOutputUTF8String() << "]";
737 switch (token->getType()) {
738 case BoWType::BOW_PREDICATE:{
739 // FIXME To implement
740 LERROR << "GenericXmlDumper: BoWType::BOW_PREDICATE support not implemented";
741 break;
742 }
743 case BoWType::BOW_TERM: {
744 LDEBUG << "GenericXmlDumper: output BoWTerm";
745 // compound informations
746 out << "<" << m_compoundTag;
747 xmlOutputBoWInfos(out,&*token,offset);
748
750 // close opening tag, then parts, then closing tag
751 out << ">" << endl;
752 }
753 else {
754 //single tag
755 out << "/>" << endl;
756 }
757
758 // go through parts in any case, at least to get enclosed compounds
760 // use iterator to create all partial compounds
761 BoWText t;
762 t.push_back(token);
763 BoWTokenIterator bit(t);
764 if (! bit.isAtEnd()) {
765 LDEBUG << "first token=" << bit.getElement()->getOutputUTF8String();
766 bit++; // first one is same BoWTerm
767 }
768 while (! bit.isAtEnd()) {
769 boost::shared_ptr< AbstractBoWElement > tok=bit.getElement();
770 LDEBUG << "next token=" << tok->getOutputUTF8String();
771 xmlOutputCompound(out,analysis,tok,anagraph,posgraph,annotationData,sp,offset);
772 bit++;
773 }
774 }
775 else {
776 // output only enclosed compounds
777 boost::shared_ptr< BoWTerm > term=boost::dynamic_pointer_cast<BoWTerm>(token);
778 const std::deque< BoWComplexToken::Part >& parts=term->getParts();
779 for (auto p=parts.begin(),p_end=parts.end();p!=p_end;p++) {
780 xmlOutputCompound(out,analysis,(*p).getBoWToken(),anagraph,posgraph,annotationData,sp,offset);
781 }
782 }
783
785 out << "</" << m_compoundTag << ">" << endl;
786 }
787 break;
788 }
789 case BoWType::BOW_NAMEDENTITY: {
791 LinguisticGraphVertex v=boost::dynamic_pointer_cast<BoWNamedEntity>(token)->getVertex();
792 LDEBUG << "GenericXmlDumper: output BoWNamedEntity of vertex " << v;
793 std::pair<const SpecificEntityAnnotation*,AnalysisGraph*>
794 se=checkSpecificEntity(v,anagraph,posgraph,annotationData);
795 if (se.first==0) {
797 LERROR << "GenericXmlDumper: for vertex " << v << ": specific entity not found";
798 }
799 else {
800 xmlOutputSpecificEntity(out,analysis,se.first,se.second,sp,offset);
801 }
802 }
803 break;
804 }
805 case BoWType::BOW_TOKEN: {
807 LinguisticGraphVertex v=boost::dynamic_pointer_cast<BoWToken>(token)->getVertex();
808 LDEBUG << "GenericXmlDumper: output BoWToken of vertex " << v;
809 xmlOutputVertexInfos(out,analysis,v,posgraph,offset);
810 }
811 break;
812 }
813 default: {
815 LERROR << "GenericXmlDumper: Error: BowToken has type BoWType::BOW_NOTYPE";
816
817 }
818 }
819}
820
822 AnalysisContent& analysis,
825 uint64_t offset) const
826{
827 out << "<" << m_wordTag;
828 for (unsigned int i=0,size=m_features.size();i<size;i++) {
829 std::string value;
830 // for position, correct with offset : hard coded name
831 if (m_features[i]->getName()=="position") {
832 unsigned int pos=atoi(m_features[i]->getValue(graph,v,analysis).c_str());
833 pos+=offset;
834 ostringstream oss;
835 oss << pos;
836 value=oss.str();
837 }
838 else {
839 value=xmlString(m_features[i]->getValue(graph,v,analysis));
840 }
841 out << " " << m_featureTags[i] << "=\"" << value << "\"";
842 }
843 out << "/>" << endl;
844}
845
846void GenericXmlDumper::xmlOutputBoWInfos(ostream& out, AbstractBoWElement* token, uint64_t offset) const
847{
848 for (unsigned int i=0,size=m_bowFeatures.size();i<size;i++) {
849 std::string value;
850 // for position, correct with offset : hard coded name
851 if (m_bowFeatures[i]->getName()=="position") {
852 unsigned int pos=atoi(m_bowFeatures[i]->getValue(token).c_str());
853 pos+=offset;
854 ostringstream oss;
855 oss << pos;
856 value=oss.str();
857 }
858 else {
859 value=xmlString(m_bowFeatures[i]->getValue(token));
860 }
861 out << " " << m_featureTags[i] << "=\"" << value << "\"";
862 }
863}
864
867 const std::string& featureName,
868 const FsaStringsPool& sp,
869 uint64_t offset) const
870{
871 // all hard-coded feature names : not really clean, but a clean definition of all features for
872 // a specialized class such as SpecificEntityAnnotation seems a bit too much...
873 if (featureName=="position") {
874 uint64_t pos=se->getPosition();
875 pos+=offset;
876 ostringstream oss;
877 oss << pos;
878 return oss.str();
879 }
880 if (featureName.find("property:MACRO")==0) { // put entity type in category
881 std::string typeName("");
882 try {
883 LimaString str= MediaticData::single().getEntityName(se->getType());
885 }
886 catch (std::exception& ) {
888 LERROR << "Undefined entity type " << se->getType();
889 return "";
890 }
891 return typeName;
892 }
893 else if (featureName=="lemma") {
895 }
896 else if (featureName=="word") {
898 }
899 return "";
900}
901
902//--------------------------------------------------------------------------------------
903// string manipulation functions to protect XML entities
904std::string GenericXmlDumper::xmlString(const std::string& inputStr) const
905{
906 // protect XML entities
907 std::string str(inputStr);
908 replace(str,"&", "&amp;");
909 replace(str,"<", "&lt;");
910 replace(str,">", "&gt;");
911 replace(str,"\"", "&quot;");
912 replace(str,"\n", "\\n");
913 return str;
914}
915
916void GenericXmlDumper::replace(std::string& str,
917 const std::string& toReplace,
918 const std::string& newValue) const
919{
920 string::size_type oldLen=toReplace.size();
921 string::size_type newLen=newValue.size();
922 string::size_type i=str.find(toReplace);
923 while (i!=string::npos) {
924 str.replace(i,oldLen,newValue);
925 i+=newLen;
926 i=str.find(toReplace,i);
927 }
928}
929
930
931} // AnalysisDumper
932} // LinguisticProcessing
933} // Lima
#define GENERICXMLDUMPER_CLASSID
#define LWARN
Definition LimaCommon.h:160
#define LDEBUG
Definition LimaCommon.h:157
#define LERROR
Definition LimaCommon.h:161
A graph structure for linguistic analysis.
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define DUMPERLOGINIT
Defines a Factory to create Object of type Base.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
Holds an annotation graph and gives an API to manipulate it.
std::set< AnnotationGraphVertex > matches(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second) const
Gets the set of vertices matched in the second graph by the given vertex of the first graph.
const GenericAnnotation & annotation(AnnotationGraphVertex v1, AnnotationGraphVertex v2, const LimaString &annot) const
This class is the abstract base class of all elements that can be stored in a BoWText.
This class represents a list of elements, that are pointers on polymmorphic tokens that can be simple...
Definition bowText.h:48
boost::shared_ptr< Lima::Common::BagOfWords::AbstractBoWElement > getElement()
const FsaStringsPool & stringsPool(MediaId med) const
std::deque< std::string > & getListsValueAtKey(const std::string &key)
std::map< std::string, std::string > & getMapAtKey(const std::string &key)
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
Use this exception to signal an error in one of the configuration files.
Definition LimaCommon.h:345
std::shared_ptr< DumperStream > initialize(AnalysisContent &analysis) const
virtual void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
void xmlOutputVertices(std::ostream &out, AnalysisContent &analysis, LinguisticAnalysisStructure::AnalysisGraph *anagraph, LinguisticAnalysisStructure::AnalysisGraph *posgraph, const Common::AnnotationGraphs::AnnotationData *annotationData, const SyntacticAnalysis::SyntacticData *syntacticData, const LinguisticGraphVertex begin, const LinguisticGraphVertex end, const FsaStringsPool &sp, const uint64_t offset) const
bool m_outputSentenceBoundaries
output sentence boundaries (enclosing sentence tags)
std::vector< std::string > m_featureTags
use additional vector (aligned) to store associated XML tags
BoWFeatures m_bowFeatures
use dedicated class for feature storage (easy initialization functions)
virtual LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
std::vector< boost::shared_ptr< Common::BagOfWords::BoWToken > > checkCompound(LinguisticGraphVertex v, LinguisticAnalysisStructure::AnalysisGraph *anagraph, LinguisticAnalysisStructure::AnalysisGraph *posgraph, const Common::AnnotationGraphs::AnnotationData *annotationData, const SyntacticAnalysis::SyntacticData *syntacticData, uint64_t offset, std::set< LinguisticGraphVertex > &visited) const
std::deque< std::string > m_featureNames
use additional vector (aligned) to store feature names
void initializeFeatures(const std::map< std::string, std::string > &features, const std::deque< std::string > &featureOrder=std::deque< std::string >())
void xmlOutputVertex(std::ostream &out, AnalysisContent &analysis, LinguisticGraphVertex v, LinguisticAnalysisStructure::AnalysisGraph *anagraph, LinguisticAnalysisStructure::AnalysisGraph *posgraph, const Common::AnnotationGraphs::AnnotationData *annotationData, const SyntacticAnalysis::SyntacticData *syntacticData, const FsaStringsPool &sp, uint64_t offset, std::set< LinguisticGraphVertex > &visited, std::set< LinguisticGraphVertex > &alreadyStoredVertices) const
void xmlOutputVertexInfos(std::ostream &out, Lima::AnalysisContent &analysis, LinguisticGraphVertex v, Lima::LinguisticProcessing::LinguisticAnalysisStructure::AnalysisGraph *graph, uint64_t offset) const
void xmlOutputCompound(std::ostream &out, AnalysisContent &analysis, boost::shared_ptr< Lima::Common::BagOfWords::AbstractBoWElement > token, Lima::LinguisticProcessing::LinguisticAnalysisStructure::AnalysisGraph *anagraph, Lima::LinguisticProcessing::LinguisticAnalysisStructure::AnalysisGraph *posgraph, const Lima::Common::AnnotationGraphs::AnnotationData *annotationData, const Lima::FsaStringsPool &sp, uint64_t offset) const
bool m_outputSpecificEntityParts
output parts of specific entities
std::string specificEntityFeature(const SpecificEntities::SpecificEntityAnnotation *se, const std::string &featureName, const FsaStringsPool &sp, uint64_t offset) const
void xmlOutput(std::ostream &out, AnalysisContent &analysis, LinguisticAnalysisStructure::AnalysisGraph *anagraph, LinguisticAnalysisStructure::AnalysisGraph *posgraph, const Common::AnnotationGraphs::AnnotationData *annotationData, const SyntacticAnalysis::SyntacticData *syntacticData) const
virtual void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
std::pair< const SpecificEntities::SpecificEntityAnnotation *, LinguisticAnalysisStructure::AnalysisGraph * > checkSpecificEntity(LinguisticGraphVertex v, LinguisticAnalysisStructure::AnalysisGraph *anagraph, LinguisticAnalysisStructure::AnalysisGraph *posgraph, const Common::AnnotationGraphs::AnnotationData *annotationData) const
check if a vertex is a specific entity: returns the specific entity annotation if it is the case,...
bool m_outputAllCompounds
output all partial compounds (created using BoWToken iterator)
WordFeatures m_features
use dedicated class for feature storage (easy initialization functions)
void xmlOutputBoWInfos(std::ostream &out, Common::BagOfWords::AbstractBoWElement *token, uint64_t offset) const
bool xmlOutputSpecificEntity(std::ostream &out, AnalysisContent &analysis, const SpecificEntities::SpecificEntityAnnotation *se, LinguisticAnalysisStructure::AnalysisGraph *anagraph, const FsaStringsPool &sp, uint64_t offset) const
void replace(std::string &str, const std::string &toReplace, const std::string &newValue) const
void initialize(const std::deque< std::string > &featureNames)
Parameters retrived in the configuration file:
std::vector< std::pair< boost::shared_ptr< Common::BagOfWords::BoWRelation >, boost::shared_ptr< Common::BagOfWords::BoWToken > > > buildTermFor(const AnnotationGraphVertex &vx, const AnnotationGraphVertex &tgt, const LinguisticGraph &anagraph, const LinguisticGraph &posgraph, const uint64_t offset, const SyntacticAnalysis::SyntacticData *syntacticData, const Common::AnnotationGraphs::AnnotationData *annotationData, std::set< LinguisticGraphVertex > &visited) const
Creates the terms reachable from the given annotation vertex.
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, MediaId language)
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
A representation of a specific entity to store in the annotation graph.
This class points to a graph, its dependency graph and the structure that holds the maping between th...
void initialize(const std::deque< std::string > &featureNames)
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
static void logElapsedTime(const std::string &mess, const std::string &taskCategory=std::string(""))
log the number of microseconds since last UpdateCurrentTime
static void updateCurrentTime(const std::string &taskCategory=std::string(""))
store current time for new elapsed time computation
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
bool hasAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
LimaString utf8stdstring2limastring(const std::string &src)
SimpleFactory< MediaProcessUnit, GenericXmlDumper > genericXmlDumperFactory(GENERICXMLDUMPER_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
@ MISSING_DATA
Definition LimaCommon.h:243
QString LimaString
Definition LimaString.h:33
STL namespace.
launch exception related to the configuration file parsing