31#include <boost/graph/properties.hpp>
40using namespace boost::tuples;
50namespace LinguisticProcessing {
51namespace AnalysisDumpers {
64m_outputSentenceBoundaries(false),
65m_outputSpecificEntities(false),
66m_outputSpecificEntityParts(false),
67m_outputCompounds(false),
68m_outputCompoundParts(false),
69m_outputAllCompounds(false),
71m_sentenceBoundaryTag(
"s"),
72m_specificEntityTag(
"e"),
95 for (WordFeatures::const_iterator it=
m_features.begin(),it_end=
m_features.end();it!=it_end;it++) {
123 map<string,string> featuresMap=unitConfiguration.
getMapAtKey(
"features");
140 if (str==
"no" || str==
"false" || str==
"") {
146 LDEBUG <<
"GenericXmlDumper: outputSpecificEntities set to true (tag is " << str <<
")";
154 if (str!=
"no" && str!=
"false" && str!=
"") {
157 LDEBUG <<
"GenericXmlDumper: outputSpecificEntities set to true (tag is " << str <<
")";
165 if (str!=
"no" && str!=
"false" && str!=
"") {
173 if (str!=
"no" && str!=
"false" && str!=
"") {
176 LDEBUG <<
"GenericXmlDumper: outputSentenceBoundaries set to true (tag is " << str <<
")";
183 if (str!=
"no" && str!=
"false" && str!=
"") {
189 LDEBUG <<
"GenericXmlDumper: outputCompounds set to true (tag is " << str <<
")";
196 if (str!=
"no" && str!=
"false" && str!=
"") {
204 if (str!=
"no" && str!=
"false" && str!=
"") {
236 const std::deque<std::string>& featureOrder)
239 bool useMapOrder(
false);
240 if (! featureOrder.empty()) {
243 LDEBUG <<
"GenericXmlDumper: initialize features: use order";
244 for (deque<string>::const_iterator it=featureOrder.begin(),it_end=featureOrder.end();it!=it_end;it++) {
245 LDEBUG <<
"GenericXmlDumper: --"<< (*it);
246 const std::string& featureTag=(*it);
248 map<string,string>::const_iterator f=featuresMap.find(featureTag);
249 if (f==featuresMap.end()) {
251 LWARN <<
"GenericXmlDumper: 'featureOrder' parameter mentions a feature '" << featureTag
252 <<
"' not in feature map parameter: order ignored";
266 for (map<string,string>::const_iterator it=featuresMap.begin(),it_end=featuresMap.end();it!=it_end; it++) {
267 const std::string& featureName=(*it).second;
268 const std::string& featureTag=(*it).first;
277 LERROR <<
"GenericXmlDumper: error: failed to initialize all features";
287 LDEBUG <<
"GenericXmlDumper::process";
289 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.
getData(
"LinguisticMetaData"));
292 LERROR <<
"no LinguisticMetaData ! abort";
296 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"AnalysisGraph"));
299 LERROR <<
"no graph 'AnaGraph' available !";
302 auto posgraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"PosGraph"));
305 LERROR <<
"no graph 'PosGraph' available !";
308 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.
getData(
"AnnotationData"));
309 if (annotationData==0)
311 LERROR <<
"no annotation graph available !";
315 std::shared_ptr<SyntacticData> syntacticData;
317 syntacticData = std::dynamic_pointer_cast< SyntacticData >(analysis.
getData(
"SyntacticData"));
318 if (annotationData==0)
320 LWARN <<
"compounds are supposed to be printed in output but no syntactic data available !";
325 xmlOutput(dstream->out(), analysis, anagraph.get(), posgraph.get(), annotationData.get(), syntacticData.get());
342 out <<
"<text>" << endl;
344 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.
getData(
"LinguisticMetaData"));
348 std::shared_ptr<SegmentationData> sb;
350 sb = std::dynamic_pointer_cast<SegmentationData>(analysis.
getData(
"SentenceBoundaries"));
352 LWARN <<
"GenericXmlDumper:: no SentenceBoundaries";
368 metadata->getStartOffset());
373 uint64_t nbSentences((sb->getSegments()).size());
374 LDEBUG <<
"GenericXmlDumper: "<< nbSentences <<
" sentences found";
375 for (uint64_t i=0; i<nbSentences; i++)
386 LDEBUG <<
"dump sentence between " << sentenceBegin <<
" and " << sentenceEnd;
387 LDEBUG <<
"dump simple terms for this sentence";
399 metadata->getStartOffset());
400 string str=oss.str();
402 LDEBUG <<
"nothing to dump in this sentence";
411 out <<
"</text>" << endl;
424 const uint64_t offset)
const
428 LDEBUG <<
"GenericXmlDumper: ========================================";
429 LDEBUG <<
"GenericXmlDumper: outputXml from vertex " << begin <<
" to vertex " << end;
434 map<Token*, vector<LinguisticGraphVertex>,
lTokenPosition> sortedTokens;
436 std::queue<LinguisticGraphVertex> toVisit;
437 std::set<LinguisticGraphVertex> visited;
447 while (!toVisit.empty()) {
450 if (last || v == lastVertex) {
457 for (boost::tie(outItr,outItrEnd)=out_edges(v,*graph); outItr!=outItrEnd; outItr++)
460 if (visited.find(next)==visited.end())
462 visited.insert(next);
473 sortedTokens[t].push_back(v);
479 std::set<LinguisticGraphVertex> alreadyStoredVertices;
482 map<uint64_t,vector<string> > xmlOutputs;
485 it=sortedTokens.begin(),it_end=sortedTokens.end(); it!=it_end; it++)
487 const vector<LinguisticGraphVertex>& vertices=(*it).second;
488 if (vertices.size()==0) {
490 LERROR <<
"GenericXmlDumper: no vertices for token " << (*it).first->stringForm();
494 for (vector<LinguisticGraphVertex>::const_iterator d=vertices.begin(),
495 d_end=vertices.end(); d!=d_end; d++) {
502 xmlOutputVertex(oss,analysis,(*d),anagraph,posgraph,annotationData,syntacticData,
503 sp,offset,visited,alreadyStoredVertices);
504 uint64_t pos=(*it).first->position();
505 xmlOutputs[pos].push_back(oss.str());
509 for (map<uint64_t,vector<string> >::const_iterator it=xmlOutputs.begin(),it_end=xmlOutputs.end();it!=it_end;it++) {
510 for (vector<string>::const_iterator s=(*it).second.begin(),s_end=(*it).second.end();s!=s_end;s++) {
527 set<LinguisticGraphVertex>& visited,
528 std::set<LinguisticGraphVertex>& alreadyStoredVertices)
const
531 LDEBUG <<
"GenericXmlDumper: output vertex " << v;
535 std::pair<const SpecificEntityAnnotation*,AnalysisGraph*>
538 LDEBUG <<
"GenericXmlDumper: -- is a specific entity ";
544 LERROR <<
"failed to output specific entity for vertex " << v;
550 std::vector< boost::shared_ptr< BoWToken > > compoundTokens=
551 checkCompound(v, anagraph, posgraph, annotationData, syntacticData, offset, visited);
552 if (compoundTokens.size()!=0) {
553 for (
auto it=compoundTokens.begin(), it_end=compoundTokens.end();it!=it_end;it++) {
555 xmlOutputCompound(out,analysis,(*it),anagraph,posgraph,annotationData,sp,offset);
556 std::set<uint64_t> bowTokenVertices = (*it)->getVertices();
557 alreadyStoredVertices.insert(bowTokenVertices.begin(), bowTokenVertices.end());
562 LDEBUG <<
"GenericXmlDumper: -- is simple word ";
569std::pair<const SpecificEntityAnnotation*,AnalysisGraph*>
576 std::set< AnnotationGraphVertex > anaVertices = annotationData->
matches(
"PosGraph",v,
"AnalysisGraph");
578 for (std::set< AnnotationGraphVertex >::const_iterator anaVerticesIt = anaVertices.begin();
579 anaVerticesIt != anaVertices.end(); anaVerticesIt++)
581 std::set< AnnotationGraphVertex > matches = annotationData->
matches(
"AnalysisGraph",*anaVerticesIt,
"annot");
582 for (std::set< AnnotationGraphVertex >::const_iterator it = matches.begin();
583 it != matches.end(); it++)
590 pointerValue<SpecificEntityAnnotation>();
591 return make_pair(se,anagraph);
597 std::set< AnnotationGraphVertex > matches = annotationData->
matches(
"PosGraph",v,
"annot");
598 for (std::set< AnnotationGraphVertex >::const_iterator it = matches.begin();
599 it != matches.end(); it++)
607 pointerValue<SpecificEntityAnnotation>();
608 return make_pair(se,posgraph);
620 uint64_t offset)
const
624 LERROR <<
"missing specific entity annotation";
645 for (std::vector< LinguisticGraphVertex>::const_iterator m(se->
vertices().begin());
659 for (std::vector< LinguisticGraphVertex>::const_iterator m(se->
vertices().begin());
683 set<LinguisticGraphVertex>& visited)
const
686 LDEBUG <<
"GenericXmlDumper: check if compound for vertex " << v;
688 std::set< AnnotationGraphVertex > cpdsHeads = annotationData->
matches(
"PosGraph", v,
"cpdHead");
689 if (cpdsHeads.empty())
692 return std::vector< boost::shared_ptr< BoWToken > >();
695 LDEBUG <<
"GenericXmlDumper: -- is head of a compound ";
696 std::vector< boost::shared_ptr< BoWToken > > tokens;
697 std::set< std::string > alreadyStored;
698 for (std::set< AnnotationGraphVertex >::const_iterator it=cpdsHeads.begin(), it_end=cpdsHeads.end();
704 std::vector<std::pair<boost::shared_ptr< BoWRelation >, boost::shared_ptr<BoWToken> > > bowTokens =
706 syntacticData, annotationData, visited);
707 for (
auto bowItr=bowTokens.begin();
708 bowItr!=bowTokens.end(); bowItr++)
710 std::string elem = (*bowItr).second->getIdUTF8String();
711 if (alreadyStored.find(elem) != alreadyStored.end())
717 tokens.push_back((*bowItr).second);
718 alreadyStored.insert(elem);
728 boost::shared_ptr<Common::BagOfWords::AbstractBoWElement> token,
733 uint64_t offset)
const
736 LDEBUG <<
"GenericXmlDumper: output BoWToken [" << token->getOutputUTF8String() <<
"]";
737 switch (token->getType()) {
738 case BoWType::BOW_PREDICATE:{
740 LERROR <<
"GenericXmlDumper: BoWType::BOW_PREDICATE support not implemented";
743 case BoWType::BOW_TERM: {
744 LDEBUG <<
"GenericXmlDumper: output BoWTerm";
769 boost::shared_ptr< AbstractBoWElement > tok=bit.
getElement();
770 LDEBUG <<
"next token=" << tok->getOutputUTF8String();
777 boost::shared_ptr< BoWTerm > term=boost::dynamic_pointer_cast<BoWTerm>(token);
778 const std::deque< BoWComplexToken::Part >& parts=term->getParts();
779 for (
auto p=parts.begin(),p_end=parts.end();p!=p_end;p++) {
780 xmlOutputCompound(out,analysis,(*p).getBoWToken(),anagraph,posgraph,annotationData,sp,offset);
789 case BoWType::BOW_NAMEDENTITY: {
792 LDEBUG <<
"GenericXmlDumper: output BoWNamedEntity of vertex " << v;
793 std::pair<const SpecificEntityAnnotation*,AnalysisGraph*>
797 LERROR <<
"GenericXmlDumper: for vertex " << v <<
": specific entity not found";
805 case BoWType::BOW_TOKEN: {
808 LDEBUG <<
"GenericXmlDumper: output BoWToken of vertex " << v;
815 LERROR <<
"GenericXmlDumper: Error: BowToken has type BoWType::BOW_NOTYPE";
825 uint64_t offset)
const
828 for (
unsigned int i=0,size=
m_features.size();i<size;i++) {
832 unsigned int pos=atoi(
m_features[i]->getValue(graph,v,analysis).c_str());
848 for (
unsigned int i=0,size=
m_bowFeatures.size();i<size;i++) {
852 unsigned int pos=atoi(
m_bowFeatures[i]->getValue(token).c_str());
867 const std::string& featureName,
869 uint64_t offset)
const
873 if (featureName==
"position") {
880 if (featureName.find(
"property:MACRO")==0) {
881 std::string typeName(
"");
886 catch (std::exception& ) {
893 else if (featureName==
"lemma") {
896 else if (featureName==
"word") {
907 std::string str(inputStr);
917 const std::string& toReplace,
918 const std::string& newValue)
const
920 string::size_type oldLen=toReplace.size();
921 string::size_type newLen=newValue.size();
922 string::size_type i=str.find(toReplace);
923 while (i!=string::npos) {
924 str.replace(i,oldLen,newValue);
926 i=str.find(toReplace,i);
#define GENERICXMLDUMPER_CLASSID
A graph structure for linguistic analysis.
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
Defines a Factory to create Object of type Base.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
Holds an annotation graph and gives an API to manipulate it.
std::set< AnnotationGraphVertex > matches(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second) const
Gets the set of vertices matched in the second graph by the given vertex of the first graph.
const GenericAnnotation & annotation(AnnotationGraphVertex v1, AnnotationGraphVertex v2, const LimaString &annot) const
This class is the abstract base class of all elements that can be stored in a BoWText.
This class represents a list of elements, that are pointers on polymmorphic tokens that can be simple...
boost::shared_ptr< Lima::Common::BagOfWords::AbstractBoWElement > getElement()
std::string & getParamsValueAtKey(const std::string &key)
std::deque< std::string > & getListsValueAtKey(const std::string &key)
std::map< std::string, std::string > & getMapAtKey(const std::string &key)
return a message when a 'list' was not found
return a message when a 'map' was not found
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
Use this exception to signal an error in one of the configuration files.
std::shared_ptr< DumperStream > initialize(AnalysisContent &analysis) const
virtual void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
std::string m_specificEntityTag
void xmlOutputVertices(std::ostream &out, AnalysisContent &analysis, LinguisticAnalysisStructure::AnalysisGraph *anagraph, LinguisticAnalysisStructure::AnalysisGraph *posgraph, const Common::AnnotationGraphs::AnnotationData *annotationData, const SyntacticAnalysis::SyntacticData *syntacticData, const LinguisticGraphVertex begin, const LinguisticGraphVertex end, const FsaStringsPool &sp, const uint64_t offset) const
bool m_outputSentenceBoundaries
output sentence boundaries (enclosing sentence tags)
std::vector< std::string > m_featureTags
use additional vector (aligned) to store associated XML tags
BoWFeatures m_bowFeatures
use dedicated class for feature storage (easy initialization functions)
bool m_outputWords
output simple words
std::string xmlString(const std::string &str) const
virtual LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
std::vector< boost::shared_ptr< Common::BagOfWords::BoWToken > > checkCompound(LinguisticGraphVertex v, LinguisticAnalysisStructure::AnalysisGraph *anagraph, LinguisticAnalysisStructure::AnalysisGraph *posgraph, const Common::AnnotationGraphs::AnnotationData *annotationData, const SyntacticAnalysis::SyntacticData *syntacticData, uint64_t offset, std::set< LinguisticGraphVertex > &visited) const
std::deque< std::string > m_featureNames
use additional vector (aligned) to store feature names
void initializeFeatures(const std::map< std::string, std::string > &features, const std::deque< std::string > &featureOrder=std::deque< std::string >())
virtual ~GenericXmlDumper()
void xmlOutputVertex(std::ostream &out, AnalysisContent &analysis, LinguisticGraphVertex v, LinguisticAnalysisStructure::AnalysisGraph *anagraph, LinguisticAnalysisStructure::AnalysisGraph *posgraph, const Common::AnnotationGraphs::AnnotationData *annotationData, const SyntacticAnalysis::SyntacticData *syntacticData, const FsaStringsPool &sp, uint64_t offset, std::set< LinguisticGraphVertex > &visited, std::set< LinguisticGraphVertex > &alreadyStoredVertices) const
void xmlOutputVertexInfos(std::ostream &out, Lima::AnalysisContent &analysis, LinguisticGraphVertex v, Lima::LinguisticProcessing::LinguisticAnalysisStructure::AnalysisGraph *graph, uint64_t offset) const
void xmlOutputCompound(std::ostream &out, AnalysisContent &analysis, boost::shared_ptr< Lima::Common::BagOfWords::AbstractBoWElement > token, Lima::LinguisticProcessing::LinguisticAnalysisStructure::AnalysisGraph *anagraph, Lima::LinguisticProcessing::LinguisticAnalysisStructure::AnalysisGraph *posgraph, const Lima::Common::AnnotationGraphs::AnnotationData *annotationData, const Lima::FsaStringsPool &sp, uint64_t offset) const
bool m_outputSpecificEntityParts
output parts of specific entities
bool m_outputCompoundParts
output also compound parts
std::string m_compoundTag
std::string specificEntityFeature(const SpecificEntities::SpecificEntityAnnotation *se, const std::string &featureName, const FsaStringsPool &sp, uint64_t offset) const
void xmlOutput(std::ostream &out, AnalysisContent &analysis, LinguisticAnalysisStructure::AnalysisGraph *anagraph, LinguisticAnalysisStructure::AnalysisGraph *posgraph, const Common::AnnotationGraphs::AnnotationData *annotationData, const SyntacticAnalysis::SyntacticData *syntacticData) const
virtual void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
std::pair< const SpecificEntities::SpecificEntityAnnotation *, LinguisticAnalysisStructure::AnalysisGraph * > checkSpecificEntity(LinguisticGraphVertex v, LinguisticAnalysisStructure::AnalysisGraph *anagraph, LinguisticAnalysisStructure::AnalysisGraph *posgraph, const Common::AnnotationGraphs::AnnotationData *annotationData) const
check if a vertex is a specific entity: returns the specific entity annotation if it is the case,...
bool m_outputCompounds
output compounds
bool m_outputAllCompounds
output all partial compounds (created using BoWToken iterator)
WordFeatures m_features
use dedicated class for feature storage (easy initialization functions)
void xmlOutputBoWInfos(std::ostream &out, Common::BagOfWords::AbstractBoWElement *token, uint64_t offset) const
Compounds::BowGenerator * m_bowGenerator
std::string m_sentenceBoundaryTag
std::map< std::string, std::string > m_defaultFeatures
bool xmlOutputSpecificEntity(std::ostream &out, AnalysisContent &analysis, const SpecificEntities::SpecificEntityAnnotation *se, LinguisticAnalysisStructure::AnalysisGraph *anagraph, const FsaStringsPool &sp, uint64_t offset) const
void replace(std::string &str, const std::string &toReplace, const std::string &newValue) const
bool m_outputSpecificEntities
output specific entities
void initialize(const std::deque< std::string > &featureNames)
void setLanguage(MediaId language)
Parameters retrived in the configuration file:
std::vector< std::pair< boost::shared_ptr< Common::BagOfWords::BoWRelation >, boost::shared_ptr< Common::BagOfWords::BoWToken > > > buildTermFor(const AnnotationGraphVertex &vx, const AnnotationGraphVertex &tgt, const LinguisticGraph &anagraph, const LinguisticGraph &posgraph, const uint64_t offset, const SyntacticAnalysis::SyntacticData *syntacticData, const Common::AnnotationGraphs::AnnotationData *annotationData, std::set< LinguisticGraphVertex > &visited) const
Creates the terms reachable from the given annotation vertex.
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, MediaId language)
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
holds surface data of a token
A representation of a specific entity to store in the annotation graph.
StringsPoolIndex getString() const
Common::MediaticData::EntityType getType() const
const std::vector< LinguisticGraphVertex > & vertices() const
uint64_t getPosition() const
LinguisticGraphVertex getHead() const
StringsPoolIndex getNormalizedForm() const
This class points to a graph, its dependency graph and the structure that holds the maping between th...
void initialize(const std::deque< std::string > &featureNames)
void setLanguage(MediaId language)
static const MediaticData & single()
const singleton accessor
static void logElapsedTime(const std::string &mess, const std::string &taskCategory=std::string(""))
log the number of microseconds since last UpdateCurrentTime
static void updateCurrentTime(const std::string &taskCategory=std::string(""))
store current time for new elapsed time computation
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
bool hasAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
LimaString utf8stdstring2limastring(const std::string &src)
SimpleFactory< MediaProcessUnit, GenericXmlDumper > genericXmlDumperFactory(GENERICXMLDUMPER_CLASSID)
launch exception related to the configuration file parsing