54#include <boost/config.hpp>
58using namespace boost::tuples;
70namespace LinguisticProcessing {
71namespace AnalysisDumpers {
81 m_dumpFullTokens(true),
97 LDEBUG <<
"posGraphXmlDumper init!";
106 LWARN <<
"dumpTokens parameter not found, using default: "
124 LERROR <<
"posGraphXmlDumper::init: Missing parameter handler in posGraphXmlDumper configuration";
138 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.
getData(
"LinguisticMetaData"));
140 LERROR <<
"posGraphXmlDumper::process: no LinguisticMetaData ! abort";
144 auto h = std::dynamic_pointer_cast<AnalysisHandlerContainer>(analysis.
getData(
"AnalysisHandlerContainer"));
148 LERROR <<
"posGraphXmlDumper::process: handler " <<
m_handler <<
" has not been given to the core client";
152 auto graph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
m_graph));
154 graph = std::make_shared<AnalysisGraph>(
m_graph,m_language,
true,
true);
158 auto syntacticData = std::dynamic_pointer_cast<SyntacticData>(analysis.
getData(
"SyntacticData"));
159 if (syntacticData==0)
161 syntacticData = std::make_shared<SyntacticAnalysis::SyntacticData>(std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
m_graph)).get(),
nullptr);
162 syntacticData->setupDependencyGraph();
163 analysis.
setData(
"SyntacticData",syntacticData);
167 auto sb = std::dynamic_pointer_cast<SegmentationData>(analysis.
getData(
"SentenceBoundaries"));
170 sb = std::make_shared<SegmentationData>(
m_graph);
171 analysis.
setData(
"SentenceBoundaries",sb);
173 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.
getData(
"AnnotationData"));
174 if (annotationData==0)
176 annotationData = std::make_shared<AnnotationData>();
177 if (std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"AnalysisGraph")) != 0)
179 std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"AnalysisGraph"))->populateAnnotationGraph(annotationData.get(),
"AnalysisGraph");
181 analysis.
setData(
"AnnotationData",annotationData);
186 std::ostream outputStream(&hsb);
187 std::set< std::pair<size_t, size_t> > alreadyDumped;
189 outputStream <<
"<?xml version='1.0' encoding='UTF-8'?>" << std::endl;
190 outputStream <<
"<!DOCTYPE lima_analysis_dump SYSTEM \"lima-xml-output.dtd\">" << std::endl;
191 outputStream <<
"<lima_analysis_dump>" << std::endl;
196 std::vector<Segment>::iterator sbItr=(sb->getSegments().begin());
198 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"AnalysisGraph"));
199 auto posgraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"PosGraph"));
202 std::vector< bool > alreadyDumpedTokens;
203 std::map< LinguisticAnalysisStructure::Token*, uint64_t > fullTokens;
206 alreadyDumpedTokens.resize(num_vertices(*posgraph->getGraph()));
207 for (boost::tie(i, i_end) = vertices(*posgraph->getGraph()); i != i_end; ++i)
209 alreadyDumpedTokens[id] =
false;
210 fullTokens[get(
vertex_token, *posgraph->getGraph(), *i)] = id;
213 outputStream <<
"<PosGraph>" << std::endl;
216 while (sbItr!=(sb->getSegments().end()))
226 annotationData.get(),
228 true, alreadyDumpedTokens, fullTokens, ++sentenceId);
232 outputStream <<
"</PosGraph>" << std::endl;
234 outputStream <<
"</lima_analysis_dump>" << std::endl;
250 const std::string& graphId,
252 std::vector< bool >& alreadyDumpedTokens,
253 std::map< LinguisticAnalysisStructure::Token*, uint64_t >& fullTokens,
254 int sentenceId)
const
259 LDEBUG <<
"posGraphXmlDumper::dumpLimaData parameters: ";
260 LDEBUG <<
"begin = "<< begin;
261 LDEBUG <<
"end = " << end ;
264 LDEBUG <<
"graphId= " << graphId ;
265 LDEBUG <<
"bySentence= " << bySentence ;
276 os <<
" <sentence id=\""<<sentenceId<<
"\">" << std::endl;
280 os <<
" <"<<graphId<<
">" << std::endl;
282 std::queue<LinguisticGraphVertex> toVisit;
283 std::set<LinguisticGraphVertex> visited;
286 while (!toVisit.empty()) {
289 outputVertex(v, *lanagraph, *lposgraph, syntacticData, annotationData, os, fullTokens, alreadyDumpedTokens, graphId);
295 for (boost::tie(outItr,outItrEnd)=out_edges(v,*lposgraph); outItr!=outItrEnd; outItr++)
298 if (visited.find(next)==visited.end())
300 visited.insert(next);
307 os <<
" </sentence>" << std::endl;
311 os <<
" </"<<graphId<<
">" << std::endl;
322 std::ostringstream pos;
332 std::ostream& xmlStream,
333 std::map< LinguisticAnalysisStructure::Token*, uint64_t >& fullTokens,
334 std::vector< bool >& alreadyDumpedTokens,
335 const std::string& graphId)
const
339 uint64_t tokenId = (*(fullTokens.find(token))).second;
340 bool alreadyDumped = alreadyDumpedTokens[tokenId];
346 LDEBUG <<
"posGraphXmlDumper::outputVertex " << v;
350 xmlStream <<
" <vertex id=\"_" << v <<
"\" />" << std::endl;
356 LWARN <<
"No token (vertex_token) for vertex " << v;
357 xmlStream <<
" <vertex id=\"_" << v <<
"\" />" << std::endl;
362 xmlStream <<
" <vertex id=\"_" << v <<
"\"";
366 if (chains.size() > 0)
368 xmlStream <<
" chains=\"";
369 VertexChainIdProp::const_iterator itChains, itChains_end;
370 itChains = chains.begin(); itChains_end = chains.end();
371 xmlStream << (*itChains); itChains++;
372 for (; itChains != itChains_end; itChains++)
374 xmlStream <<
"," << (*itChains);
378 xmlStream <<
" >" << std::endl;
380 if (graphId !=
"AnalysisGraph")
384 if (out_degree(depV, *depGraph) > 0)
386 xmlStream <<
" <deps>" << std::endl;
388 boost::tie(depIt, depIt_end) = out_edges(depV, *depGraph);
389 for (; depIt != depIt_end; depIt++)
395 xmlStream <<
" <dep v=\"_" << targV;
397 xmlStream <<
"\" t=\"" <<
399 getSyntacticRelationName(relTypeMap[*depIt]);
400 xmlStream <<
"\" />" << std::endl;
402 xmlStream <<
" </deps>" << std::endl;
407 bool hasAntecedent =
false;
409 std::set< AnnotationGraphVertex > matches = annotationData->
matches(
"PosGraph",v,
"annot");
410 if (!matches.empty())
420 boost::tie(it, it_end) = boost::out_edges(referee, annotationData->
getGraph());
421 for (; it != it_end; it++)
425 newReferee = target(*it, annotationData->
getGraph());
429 if (newReferee != 0 && newReferee != referee)
431 hasAntecedent =
true;
432 referee = newReferee;
438 std::set< AnnotationGraphVertex > refereeMatches = annotationData->
matches(
"annot",referee,
"PosGraph");
439 if (refereeMatches.empty())
442 LERROR <<
"posGraphXmlDumper::outputVertex: No PoS graph vertex matches annotation graph vertex " << referee <<
". This should not happen.";
445 xmlStream <<
" <antecedent id=\"_" << refereeAv <<
"\" />" << endl;
453 LWARN <<
"No morphosyntactic (vertex_data) data for vertex " << v;
465 xmlStream <<
" <ref>" << tokenId <<
"</ref>" << std::endl;
467 alreadyDumpedTokens[tokenId] =
true;
468 xmlStream <<
" </vertex>" << std::endl;
472 std::set<LinguisticGraphVertex> visited;
473 std::set< std::string > alreadyStored;
475 std::set< AnnotationGraphVertex > cpdsHeads = annotationData->
matches(
"PosGraph", v,
"cpdHead");
476 if (!cpdsHeads.empty())
478 std::set< AnnotationGraphVertex >::const_iterator cpdsHeadsIt, cpdsHeadsIt_end;
479 cpdsHeadsIt = cpdsHeads.begin(); cpdsHeadsIt_end = cpdsHeads.end();
480 for (; cpdsHeadsIt != cpdsHeadsIt_end; cpdsHeadsIt++)
483 std::vector<std::pair< boost::shared_ptr< BoWRelation>, boost::shared_ptr< BoWToken > > > bowTokens =
485 for (
auto bowItr=bowTokens.begin(); bowItr!=bowTokens.end(); bowItr++)
487 std::string elem = (*bowItr).second->getIdUTF8String();
488 if (alreadyStored.find(elem) != alreadyStored.end())
494 boost::shared_ptr< BoWToken > compound = (*bowItr).second;
495 LDEBUG <<
"Outputing compound: " << *compound;
498 QVector<LimaString> compounds;
500 for(
const auto& compoundString : compounds)
503 xmlStream <<
" <vertex id=\"_compound\">" << std::endl;
504 xmlStream <<
" <string>"
506 <<
"</string>" << std::endl;
507 xmlStream <<
" <position>" << compound->getPosition() <<
"</position>" << std::endl;
508 xmlStream <<
" <length>" << compound->getLength() <<
"</length>" << std::endl;
513 LWARN <<
"No morphosyntactic (vertex_data) data for vertex " << v;
517 xmlStream <<
" <data>" << std::endl;
518 xmlStream <<
" <compound>" << std::endl;
522 xmlStream <<
" <form infl=\""
525 xmlStream <<
"lemma=\""
528 xmlStream <<
"norm=\""
530 <<
"\">" << std::endl;
533 xmlStream <<
" <property>" << std::endl;
534 for (
auto propItr = managers.cbegin(); propItr != managers.cend();
537 if (!propItr->second.getPropertyAccessor().empty(data->begin()->properties))
539 xmlStream <<
" <p prop=\"" << propItr->first
541 << propItr->second.getPropertySymbolicValue(data->begin()->properties)
542 <<
"\"/>" << std::endl;
545 xmlStream <<
" </property>" << std::endl;
546 xmlStream <<
" </form>" << std::endl;
547 xmlStream <<
" </compound>" << std::endl;
548 xmlStream <<
" </data>" << std::endl;
549 xmlStream <<
" </vertex>" << std::endl;
553 alreadyStored.insert(elem);
574#if __cplusplus >= 201103L || ( ( defined(__GXX_EXPERIMENTAL_CXX0X__) ) && ( not defined(BOOST_NO_LAMBDAS) ) )
576 std::deque< BoWComplexToken::Part > parts = compound->
getParts();
578 QMap<int, QSet<LimaString> > subresults;
581 std::function<QSet<LimaString>(QMap <
int, QSet <Lima::LimaString > >,
int,
int)> recurseResult;
582 recurseResult = [&recurseResult](QMap <int, QSet <Lima::LimaString > >subresults,
int i,
int head)
584 QSet<LimaString> recurseResultResult;
585 if (i < subresults.size())
588 QSet<LimaString> E = subresults.values()[i];
589 QSet<LimaString> nextResult = recurseResult(subresults,i+1,head);
592 if (nextResult.isEmpty())
595 recurseResultResult << e;
597 Q_FOREACH(
const LimaString& nextResultString, nextResult)
600 recurseResultResult << (e +
" " + nextResultString);
605 return recurseResultResult;
609 std::function< QMap<int, QSet<LimaString> >(std::deque< BoWComplexToken::Part >&,int)> recurse;
610 recurse = [&recurse,&recurseResult](std::deque< BoWComplexToken::Part >& parts,uint64_t head) -> QMap<
int, QSet<LimaString> >
613 QMap<int, QSet<LimaString> > recurseresults;
614 for (std::deque< BoWComplexToken::Part >::size_type i = 0; i < parts.size(); i++)
616 boost::shared_ptr< BoWToken > partToken = parts[i].getBoWToken();
617 recurseresults.insert(partToken->getPosition(), QSet<LimaString>());
624 if (!relation.isEmpty())
626 QSet< LimaString > relationSet;
627 relationSet.insert(relation);
628 recurseresults.insert(partToken->getPosition()-1,relationSet);
630 if (boost::dynamic_pointer_cast<Common::BagOfWords::BoWTerm>(partToken) != 0)
632 std::deque< BoWComplexToken::Part > parts = boost::dynamic_pointer_cast< Common::BagOfWords::BoWTerm >(partToken)->getParts();
633 QMap<int, QSet<LimaString> > partTokenResults = recurse(parts,boost::dynamic_pointer_cast< Common::BagOfWords::BoWTerm >(partToken)->getHead());
635 QSet<LimaString> partStrings = recurseResult(partTokenResults,0,head);
638 partStrings.insert(parts[head].getBoWToken()->getLemma());
641 recurseresults.insert(partToken->getPosition(), QSet<LimaString>());
642 Q_FOREACH(
const QString& partString, partStrings)
645 recurseresults[partToken->getPosition()].insert(partString);
651 recurseresults[partToken->getPosition()].insert(partToken->getLemma());
655 return recurseresults;
657 subresults = recurse(parts,compound->
getHead());
659 QSet<LimaString> result = recurseResult(subresults,0,compound->
getHead());
674 std::ostream& xmlStream)
const
676 xmlStream <<
" <edge src=\"" << source(e, graph)
677 <<
"\" targ=\"" << target(e, graph) <<
"\" />" << std::endl;
This file is the main header file for the data related to annotation graphs.
A graph that stores any data (annotations) referencing primarily nodes of a text anlaysis.
A graph that stores the relations of syntactic dependency between the elements of a DependencyGraph.
DependencyGraph::out_edge_iterator DependencyGraphOutEdgeIt
DependencyGraph::vertex_descriptor DependencyGraphVertex
boost::property_map< DependencyGraph, edge_deprel_type_t >::const_type CEdgeDepRelTypePropertyMap
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, DepVertexProperties, DepEdgeProperties > DependencyGraph
The dependency graph class.
A graph structure for linguistic analysis.
std::set< Lima::LinguisticProcessing::LinguisticAnalysisStructure::ChainIdStruct > VertexChainIdProp
Property to identify the chains in the graph.
boost::graph_traits< LinguisticGraph >::edge_descriptor LinguisticGraphEdge
typedefs to simplify the access to various graphs elements
LinguisticGraph::vertex_iterator LinguisticGraphVertexIt
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
Defines a Factory to create Object of type Base.
virtual void startAnalysis()=0
function called by the LIMA analyzer on start of a new document
virtual void endAnalysis()=0
function called by the LIMA analyzer on the end of a text part that is to be analyzed
defines callback interface
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Holds an annotation graph and gives an API to manipulate it.
std::set< AnnotationGraphVertex > matches(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second) const
Gets the set of vertices matched in the second graph by the given vertex of the first graph.
This class represents a part of a ComplexToken : it is composed of a pointer on the actual part (a Bo...
boost::shared_ptr< BoWRelation > getBoWRelation() const
uint64_t getHead() const
add a part in the list of parts of the complex token.
std::deque< Part > & getParts(void)
This is a complex token used to represent a multiword term.
const std::map< std::string, PropertyManager > & getPropertyManagers() const
Get the map of all PropertyManagers.
std::string & getParamsValueAtKey(const std::string &key)
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Use this exception to signal an error in one of the configuration files.
void dumpLimaData(std::ostream &os, const LinguisticGraphVertex begin, const LinguisticGraphVertex end, const LinguisticAnalysisStructure::AnalysisGraph *anagraph, const LinguisticAnalysisStructure::AnalysisGraph *posgraph, const SyntacticAnalysis::SyntacticData *syntacticData, const Common::AnnotationGraphs::AnnotationData *annotationData, const std::string &graphId, bool bySentence, std::vector< bool > &alreadyDumpedTokens, std::map< LinguisticAnalysisStructure::Token *, uint64_t > &fullTokens, int sentenceId) const
void naturalCompoundTokenString(const Common::BagOfWords::BoWTerm *compound, QVector< LimaString > &result) const
LinguisticProcessing::Compounds::BowGenerator * m_bowGenerator
void outputVertex(const LinguisticGraphVertex v, const LinguisticGraph &lanagraph, const LinguisticGraph &lposgraph, const SyntacticAnalysis::SyntacticData *syntacticData, const Common::AnnotationGraphs::AnnotationData *annotationData, std::ostream &xmlStream, std::map< LinguisticAnalysisStructure::Token *, uint64_t > &fullTokens, std::vector< bool > &alreadyDumpedFullTokens, const std::string &graphId) const
const Common::PropertyCode::PropertyCodeManager * m_propertyCodeManager
virtual ~posGraphXmlDumper()
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
void outputEdge(const LinguisticGraphEdge e, const LinguisticGraph &graph, std::ostream &xmlStream) const
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
LimaString getPosition(const uint64_t position) const
Parameters retrived in the configuration file:
std::vector< std::pair< boost::shared_ptr< Common::BagOfWords::BoWRelation >, boost::shared_ptr< Common::BagOfWords::BoWToken > > > buildTermFor(const AnnotationGraphVertex &vx, const AnnotationGraphVertex &tgt, const LinguisticGraph &anagraph, const LinguisticGraph &posgraph, const uint64_t offset, const SyntacticAnalysis::SyntacticData *syntacticData, const Common::AnnotationGraphs::AnnotationData *annotationData, std::set< LinguisticGraphVertex > &visited) const
Creates the terms reachable from the given annotation vertex.
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, MediaId language)
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
Holds morphosyntactic informations.
void outputXml(std::ostream &xmlStream, const Common::PropertyCode::PropertyCodeManager &pcm, const FsaStringsPool &sp) const
holds surface data of a token
virtual void outputXml(std::ostream &xmlStream, const Common::PropertyCode::PropertyCodeManager &pcm, const FsaStringsPool &sp) const
This class points to a graph, its dependency graph and the structure that holds the maping between th...
DependencyGraphVertex depVertexForTokenVertex(const LinguisticGraphVertex &v) const
LinguisticAnalysisStructure::AnalysisGraph * iterator()
DependencyGraph * dependencyGraph()
static const MediaticData & single()
const singleton accessor
static MediaticData & changeable()
singleton accessor
This file contains a class to control log of informations about time, such as logging cumulated time ...
AnnotationGraph & getGraph()
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
AnnotationGraph::out_edge_iterator AnnotationGraphOutEdgeIt
bool hasAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
LimaString transcodeToXmlEntities(const LimaString &str)
LimaString utf8stdstring2limastring(const std::string &src)
SimpleFactory< MediaProcessUnit, posGraphXmlDumper > posGraphXmlDumperFactory(POSGRAPHXMLDUMPER_CLASSID)
dump just the content of the PosGraph in XML format
#define POSGRAPHXMLDUMPER_CLASSID
launch exception related to the configuration file parsing