LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
posGraphXmlDumper.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
25
26#include "posGraphXmlDumper.h"
28
29
45//TOTO include annotation.h
50
51#include <fstream>
52#include <queue>
53//ajout include
54#include <boost/config.hpp>
55
56using namespace std;
57//using namespace boost;
58using namespace boost::tuples;
59
60using namespace Lima::Common::AnnotationGraphs;
62using namespace Lima::Common::MediaticData;
63using namespace Lima::Common::BagOfWords;
64
68
69namespace Lima {
70namespace LinguisticProcessing {
71namespace AnalysisDumpers {
72
73
74//***********************************************************************
75// constructors
76//***********************************************************************
78
81 m_dumpFullTokens(true),
82 m_handler()
83
84{
85}
86
90
93 Manager* manager)
94
95{
97 LDEBUG << "posGraphXmlDumper init!";
98 m_language=manager->getInitializationParameters().media;
99 m_propertyCodeManager= &(static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager());
100 try
101 {
102 m_dumpFullTokens = (unitConfiguration.getParamsValueAtKey("dumpTokens") == "true");
103 }
104 catch (NoSuchParam& )
105 {
106 LWARN << "dumpTokens parameter not found, using default: "
107 << (m_dumpFullTokens?"true":"false");
108 }
109 try
110 {
111 m_graph=unitConfiguration.getParamsValueAtKey("graph");
112 }
113 catch (NoSuchParam& )
114 {
115 m_graph=string("PosGraph");
116 }
117 try
118 {
119 m_handler=unitConfiguration.getParamsValueAtKey("handler");
120 }
121 catch (NoSuchParam& )
122 {
124 LERROR << "posGraphXmlDumper::init: Missing parameter handler in posGraphXmlDumper configuration";
125 throw InvalidConfiguration();
126 }
127
129 m_bowGenerator->init(unitConfiguration, m_language);
130
131}
132
134{
135 Lima::TimeUtilsController timer("posGraphXmlDumper");
137
138 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.getData("LinguisticMetaData"));
139 if (metadata == 0) {
140 LERROR << "posGraphXmlDumper::process: no LinguisticMetaData ! abort";
141 return MISSING_DATA;
142 }
143 LDEBUG << "handler will be: " << m_handler;
144 auto h = std::dynamic_pointer_cast<AnalysisHandlerContainer>(analysis.getData("AnalysisHandlerContainer"));
145 AbstractTextualAnalysisHandler* handler = static_cast<AbstractTextualAnalysisHandler*>(h->getHandler(m_handler));
146 if (handler==0)
147 {
148 LERROR << "posGraphXmlDumper::process: handler " << m_handler << " has not been given to the core client";
149 return MISSING_DATA;
150 }
151
152 auto graph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData(m_graph));
153 if (graph==0) {
154 graph = std::make_shared<AnalysisGraph>(m_graph,m_language,true,true);
155 analysis.setData(m_graph,graph);
156 }
157
158 auto syntacticData = std::dynamic_pointer_cast<SyntacticData>(analysis.getData("SyntacticData"));
159 if (syntacticData==0)
160 {
161 syntacticData = std::make_shared<SyntacticAnalysis::SyntacticData>(std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData(m_graph)).get(), nullptr);
162 syntacticData->setupDependencyGraph();
163 analysis.setData("SyntacticData",syntacticData);
164 }
165
166 // Are sentences bounds right?
167 auto sb = std::dynamic_pointer_cast<SegmentationData>(analysis.getData("SentenceBoundaries"));
168 if (sb==0)
169 {
170 sb = std::make_shared<SegmentationData>(m_graph);
171 analysis.setData("SentenceBoundaries",sb);
172 }
173 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.getData("AnnotationData"));
174 if (annotationData==0)
175 {
176 annotationData = std::make_shared<AnnotationData>();
177 if (std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph")) != 0)
178 {
179 std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"))->populateAnnotationGraph(annotationData.get(), "AnalysisGraph");
180 }
181 analysis.setData("AnnotationData",annotationData);
182 }
183
184 handler->startAnalysis();
185 HandlerStreamBuf hsb(handler);
186 std::ostream outputStream(&hsb);
187 std::set< std::pair<size_t, size_t> > alreadyDumped;
188
189 outputStream << "<?xml version='1.0' encoding='UTF-8'?>" << std::endl;
190 outputStream << "<!DOCTYPE lima_analysis_dump SYSTEM \"lima-xml-output.dtd\">" << std::endl;
191 outputStream << "<lima_analysis_dump>" << std::endl;
192
193
194
195 // ??OME2 SegmentationData::iterator sbItr=sb->begin();
196 std::vector<Segment>::iterator sbItr=(sb->getSegments().begin());
197
198 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"));
199 auto posgraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("PosGraph"));
200 if (posgraph != 0)
201 {
202 std::vector< bool > alreadyDumpedTokens;
203 std::map< LinguisticAnalysisStructure::Token*, uint64_t > fullTokens;
205 uint64_t id = 0;
206 alreadyDumpedTokens.resize(num_vertices(*posgraph->getGraph()));
207 for (boost::tie(i, i_end) = vertices(*posgraph->getGraph()); i != i_end; ++i)
208 {
209 alreadyDumpedTokens[id] = false;
210 fullTokens[get(vertex_token, *posgraph->getGraph(), *i)] = id;
211 id++;
212 }
213 outputStream << "<PosGraph>" << std::endl;
214 int sentenceId = 0;
215 // ??OME2 while (sbItr!=sb->end())
216 while (sbItr!=(sb->getSegments().end()))
217 {
218 LinguisticGraphVertex sentenceBegin=sbItr->getFirstVertex();
219 LinguisticGraphVertex sentenceEnd=sbItr->getLastVertex();
220 dumpLimaData(outputStream,
221 sentenceBegin,
222 sentenceEnd,
223 anagraph.get(),
224 posgraph.get(),
225 syntacticData.get(),
226 annotationData.get(),
227 "PosGraph",
228 true, alreadyDumpedTokens, fullTokens, ++sentenceId);
229
230 sbItr++;
231 }
232 outputStream << "</PosGraph>" << std::endl;
233 }
234 outputStream << "</lima_analysis_dump>" << std::endl;
235 handler->endAnalysis();
236 return SUCCESS_ID;
237}
238
239
240//***********************************************************************
241// main function for outputing the graph
242//***********************************************************************
244 const LinguisticGraphVertex begin,
245 const LinguisticGraphVertex end,
246 const AnalysisGraph* anagraph,
247 const AnalysisGraph* posgraph,
248 const SyntacticData* syntacticData,
249 const AnnotationData* annotationData,
250 const std::string& graphId,
251 bool bySentence,
252 std::vector< bool >& alreadyDumpedTokens,
253 std::map< LinguisticAnalysisStructure::Token*, uint64_t >& fullTokens,
254 int sentenceId) const
255{
256
258
259 LDEBUG << "posGraphXmlDumper::dumpLimaData parameters: ";
260 LDEBUG << "begin = "<< begin;
261 LDEBUG << "end = " << end ;
262 LDEBUG << "posgraph fist vertex= " << posgraph->firstVertex() ;
263 LDEBUG << "posgraph last vertex= " << posgraph->lastVertex() ;
264 LDEBUG << "graphId= " << graphId ;
265 LDEBUG << "bySentence= " << bySentence ;
266// just in case we want to check alreadt dumped tokens' array
267// for (uint64_t i=0; i<alreadyDumpedTokens.size(); i++)
268// if (alreadyDumpedTokens[i]) LDEBUG << "already_dumped_tokens[" << i << "]=" << alreadyDumpedTokens[i];
269
270
271
272 LinguisticGraph* lanagraph = const_cast< LinguisticGraph* >(anagraph->getGraph());
273 LinguisticGraph* lposgraph = const_cast< LinguisticGraph* >(posgraph->getGraph());
274 if (bySentence)
275 {
276 os << " <sentence id=\""<<sentenceId<<"\">" << std::endl;
277 }
278 else
279 {
280 os << " <"<<graphId<<">" << std::endl;
281 }
282 std::queue<LinguisticGraphVertex> toVisit;
283 std::set<LinguisticGraphVertex> visited;
284 toVisit.push(begin);
285 LinguisticGraphOutEdgeIt outItr,outItrEnd;
286 while (!toVisit.empty()) {
287 LinguisticGraphVertex v=toVisit.front();
288 toVisit.pop();
289 outputVertex(v, *lanagraph, *lposgraph, syntacticData, annotationData, os, fullTokens, alreadyDumpedTokens, graphId);
290
291 if (v == end) {
292 continue;
293 }
294
295 for (boost::tie(outItr,outItrEnd)=out_edges(v,*lposgraph); outItr!=outItrEnd; outItr++)
296 {
297 LinguisticGraphVertex next=target(*outItr,*lposgraph);
298 if (visited.find(next)==visited.end())
299 {
300 visited.insert(next);
301 toVisit.push(next);
302 }
303 }
304 }
305 if (bySentence)
306 {
307 os << " </sentence>" << std::endl;
308 }
309 else
310 {
311 os << " </"<<graphId<<">" << std::endl;
312 }
313}
314
315//***********************************************************************
316// output functions
317//***********************************************************************
318
319
320LimaString posGraphXmlDumper::getPosition(const uint64_t position) const
321{
322 std::ostringstream pos;
323 pos << position;
325}
326
328 const LinguisticGraph& lanagraph,
329 const LinguisticGraph& lposgraph,
330 const SyntacticData* syntacticData,
331 const AnnotationData* annotationData,
332 std::ostream& xmlStream,
333 std::map< LinguisticAnalysisStructure::Token*, uint64_t >& fullTokens,
334 std::vector< bool >& alreadyDumpedTokens,
335 const std::string& graphId) const
336{
338 Token* token = get(vertex_token, lposgraph, v);
339 uint64_t tokenId = (*(fullTokens.find(token))).second;
340 bool alreadyDumped = alreadyDumpedTokens[tokenId];
341
342// without this condition, there's duplicate vertex in the XML output!!!
343 if (!alreadyDumped)
344 {
346 LDEBUG << "posGraphXmlDumper::outputVertex " << v;
347 if (v == syntacticData->iterator()->firstVertex() ||
348 v == syntacticData->iterator()->lastVertex())
349 {
350 xmlStream << " <vertex id=\"_" << v << "\" />" << std::endl;
351 return;
352 }
353 if (token == 0)
354 {
356 LWARN << "No token (vertex_token) for vertex " << v;
357 xmlStream << " <vertex id=\"_" << v << "\" />" << std::endl;
358 return;
359 }
360
361
362 xmlStream << " <vertex id=\"_" << v << "\"";
363 // debugging to take out JGF
364 // DUMPERLOGINIT;
365 const VertexChainIdProp& chains = get(vertex_chain_id, lposgraph,v);
366 if (chains.size() > 0)
367 {
368 xmlStream << " chains=\"";
369 VertexChainIdProp::const_iterator itChains, itChains_end;
370 itChains = chains.begin(); itChains_end = chains.end();
371 xmlStream << (*itChains); itChains++;
372 for (; itChains != itChains_end; itChains++)
373 {
374 xmlStream << "," << (*itChains);
375 }
376 xmlStream << "\"";
377 }
378 xmlStream << " >" << std::endl;
379
380 if (graphId != "AnalysisGraph")
381 {
382 const DependencyGraph* depGraph = syntacticData->dependencyGraph();
383 DependencyGraphVertex depV = syntacticData->depVertexForTokenVertex(v);
384 if (out_degree(depV, *depGraph) > 0)
385 {
386 xmlStream << " <deps>" << std::endl;
387 DependencyGraphOutEdgeIt depIt, depIt_end;
388 boost::tie(depIt, depIt_end) = out_edges(depV, *depGraph);
389 for (; depIt != depIt_end; depIt++)
390 {
391 DependencyGraphVertex depTargV = target(*depIt, *depGraph);
392 LinguisticGraphVertex targV = syntacticData-> tokenVertexForDepVertex(depTargV);
393 // CEdgeDepChainIdPropertyMap chainsMap = get(edge_depchain_id, *depGraph);
394 CEdgeDepRelTypePropertyMap relTypeMap = get(edge_deprel_type, *depGraph);
395 xmlStream << " <dep v=\"_" << targV;
396 // xmlStream << "\" c=\"" << chainsMap[*depIt];
397 xmlStream << "\" t=\"" <<
399 getSyntacticRelationName(relTypeMap[*depIt]);
400 xmlStream << "\" />" << std::endl;
401 }
402 xmlStream << " </deps>" << std::endl;
403 }
404 }
405
406 /* ADD here the output of coreference antecedents if any */
407 bool hasAntecedent = false;
408
409 std::set< AnnotationGraphVertex > matches = annotationData->matches("PosGraph",v,"annot");
410 if (!matches.empty())
411 {
412 AnnotationGraphVertex av = *matches.begin();
413
414 AnnotationGraphVertex referee = av;
415 // go through the referents chain to output the initial referee
416 while (annotationData->hasAnnotation(referee, Common::Misc::utf8stdstring2limastring("Coreferent")))
417 {
418 AnnotationGraphVertex newReferee = 0;
419 AnnotationGraphOutEdgeIt it, it_end;
420 boost::tie(it, it_end) = boost::out_edges(referee, annotationData->getGraph());
421 for (; it != it_end; it++)
422 {
423 if (annotationData->hasAnnotation(target(*it, annotationData->getGraph()), Common::Misc::utf8stdstring2limastring("Coreferent")))
424 {
425 newReferee = target(*it, annotationData->getGraph());
426 break;
427 }
428 }
429 if (newReferee != 0 && newReferee != referee)
430 {
431 hasAntecedent = true;
432 referee = newReferee;
433 }
434 else break;
435 } //while
436 if (hasAntecedent)
437 {
438 std::set< AnnotationGraphVertex > refereeMatches = annotationData->matches("annot",referee,"PosGraph");
439 if (refereeMatches.empty())
440 {
442 LERROR << "posGraphXmlDumper::outputVertex: No PoS graph vertex matches annotation graph vertex " << referee << ". This should not happen.";
443 }
444 AnnotationGraphVertex refereeAv = *refereeMatches.begin();
445 xmlStream << " <antecedent id=\"_" << refereeAv << "\" />" << endl;
446 }
447 }
448
449 MorphoSyntacticData* data = get(vertex_data, lposgraph, v);
450 if (data == 0)
451 {
453 LWARN << "No morphosyntactic (vertex_data) data for vertex " << v;
454 }
455 else
456 {
457 data->outputXml(xmlStream, *m_propertyCodeManager,sp);
458 }
459 if (m_dumpFullTokens && !alreadyDumped)
460 {
461 token->outputXml(xmlStream, *m_propertyCodeManager,sp);
462 }
463 else
464 {
465 xmlStream << " <ref>" << tokenId << "</ref>" << std::endl;
466 }
467 alreadyDumpedTokens[tokenId] = true;
468 xmlStream << " </vertex>" << std::endl;
469
470 // dump complex tokens this token is the head of
471
472 std::set<LinguisticGraphVertex> visited;
473 std::set< std::string > alreadyStored;
474
475 std::set< AnnotationGraphVertex > cpdsHeads = annotationData->matches("PosGraph", v, "cpdHead");
476 if (!cpdsHeads.empty())
477 {
478 std::set< AnnotationGraphVertex >::const_iterator cpdsHeadsIt, cpdsHeadsIt_end;
479 cpdsHeadsIt = cpdsHeads.begin(); cpdsHeadsIt_end = cpdsHeads.end();
480 for (; cpdsHeadsIt != cpdsHeadsIt_end; cpdsHeadsIt++)
481 {
482 AnnotationGraphVertex agv = *cpdsHeadsIt;
483 std::vector<std::pair< boost::shared_ptr< BoWRelation>, boost::shared_ptr< BoWToken > > > bowTokens =
484 m_bowGenerator->buildTermFor(agv, agv, lanagraph, lposgraph, 0, syntacticData, annotationData, visited);
485 for (auto bowItr=bowTokens.begin(); bowItr!=bowTokens.end(); bowItr++)
486 {
487 std::string elem = (*bowItr).second->getIdUTF8String();
488 if (alreadyStored.find(elem) != alreadyStored.end())
489 { // already stored
490 // LDEBUG << "BuildBoWTokenListVisitor: BoWToken already stored. Skipping it.";
491 }
492 else
493 {
494 boost::shared_ptr< BoWToken > compound = (*bowItr).second;
495 LDEBUG << "Outputing compound: " << *compound;
496// std::string cat = static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager().getPropertyManager("MACRO").getPropertySymbolicValue(compound->getCategory());
497
498 QVector<LimaString> compounds;
499 naturalCompoundTokenString(boost::dynamic_pointer_cast< Common::BagOfWords::BoWTerm >(compound).get(), compounds);
500 for(const auto& compoundString : compounds)
501 {
502// qDebug() << "naturalCompoundTokenString :" << compoundString;
503 xmlStream << " <vertex id=\"_compound\">" << std::endl;
504 xmlStream << " <string>"
505 << Common::Misc::transcodeToXmlEntities(compoundString).toStdString()
506 << "</string>" << std::endl;
507 xmlStream << " <position>" << compound->getPosition() << "</position>" << std::endl;
508 xmlStream << " <length>" << compound->getLength() << "</length>" << std::endl;
509 MorphoSyntacticData* data = get(vertex_data, lposgraph, v);
510 if (data == 0)
511 {
513 LWARN << "No morphosyntactic (vertex_data) data for vertex " << v;
514 }
515 else
516 {
517 xmlStream << " <data>" << std::endl;
518 xmlStream << " <compound>" << std::endl;
519 LimaString form=compoundString;
520 LimaString lemma=compoundString;
521 LimaString norm=compoundString;
522 xmlStream << " <form infl=\""
523 << Common::Misc::transcodeToXmlEntities(form).toStdString()
524 << "\" ";
525 xmlStream << "lemma=\""
526 << Common::Misc::transcodeToXmlEntities(lemma).toStdString()
527 << "\" ";
528 xmlStream << "norm=\""
529 << Common::Misc::transcodeToXmlEntities(norm).toStdString()
530 << "\">" << std::endl;
531 }
532 const auto& managers = m_propertyCodeManager->getPropertyManagers();
533 xmlStream << " <property>" << std::endl;
534 for (auto propItr = managers.cbegin(); propItr != managers.cend();
535 propItr++)
536 {
537 if (!propItr->second.getPropertyAccessor().empty(data->begin()->properties))
538 {
539 xmlStream << " <p prop=\"" << propItr->first
540 << "\" val=\""
541 << propItr->second.getPropertySymbolicValue(data->begin()->properties)
542 << "\"/>" << std::endl;
543 }
544 }
545 xmlStream << " </property>" << std::endl;
546 xmlStream << " </form>" << std::endl;
547 xmlStream << " </compound>" << std::endl;
548 xmlStream << " </data>" << std::endl;
549 xmlStream << " </vertex>" << std::endl;
550 }
551 }
552 // outputCompound();
553 alreadyStored.insert(elem);
554 }
555 }
556 }
557 }
558}
559
564void posGraphXmlDumper::naturalCompoundTokenString(const Common::BagOfWords::BoWTerm* compound, QVector< Lima::LimaString >& strings) const
565{
566// qDebug() << "posGraphXmlDumper::naturalCompoundTokenString IN" << compound->getOutputUTF8String();
567 if (compound == 0)
568 {
569// strings << Common::Misc::utf8stdstring2limastring(compound->getOutputUTF8String());
570 return;
571 }
572
573#ifndef WIN32
574#if __cplusplus >= 201103L || ( ( defined(__GXX_EXPERIMENTAL_CXX0X__) ) && ( not defined(BOOST_NO_LAMBDAS) ) )
575
576 std::deque< BoWComplexToken::Part > parts = compound->getParts();
577
578 QMap<int, QSet<LimaString> > subresults;
579
580
581 std::function<QSet<LimaString>(QMap <int, QSet <Lima::LimaString > >,int,int)> recurseResult;
582 recurseResult = [&recurseResult](QMap <int, QSet <Lima::LimaString > >subresults,int i, int head)
583 {
584 QSet<LimaString> recurseResultResult;
585 if (i < subresults.size())
586 {
587// qDebug() << "posGraphXmlDumper::naturalCompoundTokenString recurseResult i=" << i << " ; subresults size=" << subresults.size();
588 QSet<LimaString> E = subresults.values()[i]; // clazy:exclude=container-anti-pattern
589 QSet<LimaString> nextResult = recurseResult(subresults,i+1,head);
590 Q_FOREACH(const LimaString& e, E)
591 {
592 if (/*i == head || */nextResult.isEmpty())
593 {
594// qDebug() << "posGraphXmlDumper::naturalCompoundTokenString recurseResultResult" << i << e;
595 recurseResultResult << e;
596 }
597 Q_FOREACH(const LimaString& nextResultString, nextResult)
598 {
599// qDebug() << "posGraphXmlDumper::naturalCompoundTokenString recurseResultResult concat" << i << (e + " " + nextResultString);
600 recurseResultResult << (e + " " + nextResultString);
601 }
602 }
603 }
604// qDebug() << "posGraphXmlDumper::naturalCompoundTokenString recurseResult lambda "<<i<<" returns result of size" << recurseResultResult.size()<<recurseResultResult;
605 return recurseResultResult;
606 };
607
608
609 std::function< QMap<int, QSet<LimaString> >(std::deque< BoWComplexToken::Part >&,int)> recurse;
610 recurse = [&recurse,&recurseResult](std::deque< BoWComplexToken::Part >& parts,uint64_t head) -> QMap<int, QSet<LimaString> >
611 {
612// qDebug() << "posGraphXmlDumper::naturalCompoundTokenString entering recurse lambda "<<parts.size()<<head;
613 QMap<int, QSet<LimaString> > recurseresults;
614 for (std::deque< BoWComplexToken::Part >::size_type i = 0; i < parts.size(); i++)
615 {
616 boost::shared_ptr< BoWToken > partToken = parts[i].getBoWToken();
617 recurseresults.insert(partToken->getPosition(), QSet<LimaString>());
618 const BoWComplexToken::Part& part = parts[i];
619 LimaString relation;
620 if (part.getBoWRelation() != 0)
621 {
622 relation = part.getBoWRelation()->getRealization();
623 }
624 if (!relation.isEmpty())
625 {
626 QSet< LimaString > relationSet;
627 relationSet.insert(relation);
628 recurseresults.insert(partToken->getPosition()-1,relationSet);
629 }
630 if (boost::dynamic_pointer_cast<Common::BagOfWords::BoWTerm>(partToken) != 0)
631 {
632 std::deque< BoWComplexToken::Part > parts = boost::dynamic_pointer_cast< Common::BagOfWords::BoWTerm >(partToken)->getParts();
633 QMap<int, QSet<LimaString> > partTokenResults = recurse(parts,boost::dynamic_pointer_cast< Common::BagOfWords::BoWTerm >(partToken)->getHead());
634// naturalCompoundTokenString(dynamic_cast<const Common::BagOfWords::BoWTerm*>(partToken), partStrings);
635 QSet<LimaString> partStrings = recurseResult(partTokenResults,0,head);
636 // After building all terms for the parts, add the head
637 // @TODO Add all terms built from the head token if complex
638 partStrings.insert(parts[head].getBoWToken()->getLemma());
639
640// qDebug() << "posGraphXmlDumper::naturalCompoundTokenString recurse lambda after return from recurseResult with nb terms:" << partStrings.size();
641 recurseresults.insert(partToken->getPosition(), QSet<LimaString>());
642 Q_FOREACH(const QString& partString, partStrings)
643 {
644// qDebug() << "posGraphXmlDumper::naturalCompoundTokenString subresults part term" << partToken->getPosition() << partString;
645 recurseresults[partToken->getPosition()].insert(partString);
646 }
647 }
648 else
649 {
650// qDebug() << "posGraphXmlDumper::naturalCompoundTokenString subresults part simple" << partToken->getPosition() << partToken->getLemma();
651 recurseresults[partToken->getPosition()].insert(partToken->getLemma());
652 }
653 }
654// qDebug() << "posGraphXmlDumper::naturalCompoundTokenString going out of recurse lambda with nb positions"<<recurseresults.size()<<recurseresults;
655 return recurseresults;
656 };
657 subresults = recurse(parts,compound->getHead());
658
659 QSet<LimaString> result = recurseResult(subresults,0,compound->getHead());
660 Q_FOREACH(const LimaString& string, result)
661 {
662// qDebug() << "posGraphXmlDumper::naturalCompoundTokenString final result:" << string;
663 strings << string;
664 }
665
666#endif
667#endif
668 return;
669}
670
671
673 const LinguisticGraph& graph,
674 std::ostream& xmlStream) const
675{
676 xmlStream << " <edge src=\"" << source(e, graph)
677 << "\" targ=\"" << target(e, graph) << "\" />" << std::endl;
678}
679
680
681} // end namespace AnalysisDumpers
682} // end namespace LinguisticProcessings
683} // end namespace Lima
This file is the main header file for the data related to annotation graphs.
A graph that stores any data (annotations) referencing primarily nodes of a text anlaysis.
A graph that stores the relations of syntactic dependency between the elements of a DependencyGraph.
DependencyGraph::out_edge_iterator DependencyGraphOutEdgeIt
DependencyGraph::vertex_descriptor DependencyGraphVertex
boost::property_map< DependencyGraph, edge_deprel_type_t >::const_type CEdgeDepRelTypePropertyMap
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, DepVertexProperties, DepEdgeProperties > DependencyGraph
The dependency graph class.
@ edge_deprel_type
#define LWARN
Definition LimaCommon.h:160
#define LDEBUG
Definition LimaCommon.h:157
#define LERROR
Definition LimaCommon.h:161
A graph structure for linguistic analysis.
std::set< Lima::LinguisticProcessing::LinguisticAnalysisStructure::ChainIdStruct > VertexChainIdProp
Property to identify the chains in the graph.
boost::graph_traits< LinguisticGraph >::edge_descriptor LinguisticGraphEdge
typedefs to simplify the access to various graphs elements
LinguisticGraph::vertex_iterator LinguisticGraphVertexIt
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
@ vertex_chain_id
#define DUMPERLOGINIT
Defines a Factory to create Object of type Base.
virtual void startAnalysis()=0
function called by the LIMA analyzer on start of a new document
virtual void endAnalysis()=0
function called by the LIMA analyzer on the end of a text part that is to be analyzed
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Holds an annotation graph and gives an API to manipulate it.
std::set< AnnotationGraphVertex > matches(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second) const
Gets the set of vertices matched in the second graph by the given vertex of the first graph.
This class represents a part of a ComplexToken : it is composed of a pointer on the actual part (a Bo...
boost::shared_ptr< BoWRelation > getBoWRelation() const
uint64_t getHead() const
add a part in the list of parts of the complex token.
This is a complex token used to represent a multiword term.
Definition bowTerm.h:36
Holds linguistic data for one language.
const FsaStringsPool & stringsPool(MediaId med) const
const MediaData & mediaData(MediaId media) const
const std::map< std::string, PropertyManager > & getPropertyManagers() const
Get the map of all PropertyManagers.
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Use this exception to signal an error in one of the configuration files.
Definition LimaCommon.h:345
void dumpLimaData(std::ostream &os, const LinguisticGraphVertex begin, const LinguisticGraphVertex end, const LinguisticAnalysisStructure::AnalysisGraph *anagraph, const LinguisticAnalysisStructure::AnalysisGraph *posgraph, const SyntacticAnalysis::SyntacticData *syntacticData, const Common::AnnotationGraphs::AnnotationData *annotationData, const std::string &graphId, bool bySentence, std::vector< bool > &alreadyDumpedTokens, std::map< LinguisticAnalysisStructure::Token *, uint64_t > &fullTokens, int sentenceId) const
void naturalCompoundTokenString(const Common::BagOfWords::BoWTerm *compound, QVector< LimaString > &result) const
LinguisticProcessing::Compounds::BowGenerator * m_bowGenerator
void outputVertex(const LinguisticGraphVertex v, const LinguisticGraph &lanagraph, const LinguisticGraph &lposgraph, const SyntacticAnalysis::SyntacticData *syntacticData, const Common::AnnotationGraphs::AnnotationData *annotationData, std::ostream &xmlStream, std::map< LinguisticAnalysisStructure::Token *, uint64_t > &fullTokens, std::vector< bool > &alreadyDumpedFullTokens, const std::string &graphId) const
const Common::PropertyCode::PropertyCodeManager * m_propertyCodeManager
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
void outputEdge(const LinguisticGraphEdge e, const LinguisticGraph &graph, std::ostream &xmlStream) const
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
Parameters retrived in the configuration file:
std::vector< std::pair< boost::shared_ptr< Common::BagOfWords::BoWRelation >, boost::shared_ptr< Common::BagOfWords::BoWToken > > > buildTermFor(const AnnotationGraphVertex &vx, const AnnotationGraphVertex &tgt, const LinguisticGraph &anagraph, const LinguisticGraph &posgraph, const uint64_t offset, const SyntacticAnalysis::SyntacticData *syntacticData, const Common::AnnotationGraphs::AnnotationData *annotationData, std::set< LinguisticGraphVertex > &visited) const
Creates the terms reachable from the given annotation vertex.
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, MediaId language)
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
void outputXml(std::ostream &xmlStream, const Common::PropertyCode::PropertyCodeManager &pcm, const FsaStringsPool &sp) const
virtual void outputXml(std::ostream &xmlStream, const Common::PropertyCode::PropertyCodeManager &pcm, const FsaStringsPool &sp) const
Definition Token.cpp:60
This class points to a graph, its dependency graph and the structure that holds the maping between th...
DependencyGraphVertex depVertexForTokenVertex(const LinguisticGraphVertex &v) const
LinguisticAnalysisStructure::AnalysisGraph * iterator()
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
static MediaticData & changeable()
singleton accessor
Definition Singleton.h:71
This file contains a class to control log of informations about time, such as logging cumulated time ...
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
AnnotationGraph::out_edge_iterator AnnotationGraphOutEdgeIt
bool hasAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
LimaString transcodeToXmlEntities(const LimaString &str)
LimaString utf8stdstring2limastring(const std::string &src)
SimpleFactory< MediaProcessUnit, posGraphXmlDumper > posGraphXmlDumperFactory(POSGRAPHXMLDUMPER_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
@ MISSING_DATA
Definition LimaCommon.h:243
QString LimaString
Definition LimaString.h:33
STL namespace.
dump just the content of the PosGraph in XML format
#define POSGRAPHXMLDUMPER_CLASSID
launch exception related to the configuration file parsing