10#ifndef LIMA_LINGUISTICPROCESSING_SIMPLEXMLBOWDUMPER_H
11#define LIMA_LINGUISTICPROCESSING_SIMPLEXMLBOWDUMPER_H
31namespace LinguisticProcessing {
32namespace AnalysisDumpers {
34#define GENERICXMLDUMPER_CLASSID "GenericXmlDumper"
84 void initializeFeatures(
const std::map<std::string,std::string>& features,
85 const std::deque<std::string>& featureOrder=std::deque<std::string>());
87 void xmlOutput(std::ostream& out,
94 void xmlOutputVertices(std::ostream& out,
103 const uint64_t offset)
const;
105 void xmlOutputVertex(std::ostream& out,
114 std::set<LinguisticGraphVertex>& visited,
115 std::set<LinguisticGraphVertex>& alreadyStoredVertices)
const;
119 void xmlOutputBoWInfos(std::ostream& out,
121 uint64_t offset)
const;
129 std::pair<const SpecificEntities::SpecificEntityAnnotation*,LinguisticAnalysisStructure::AnalysisGraph*>
135 bool xmlOutputSpecificEntity(std::ostream& out,
140 uint64_t offset)
const;
145 const std::string& featureName,
147 uint64_t offset)
const;
149 std::vector< boost::shared_ptr< Common::BagOfWords::BoWToken > >
156 std::set<LinguisticGraphVertex>& visited)
const;
158 void xmlOutputCompound(std::ostream& out,
177 std::string xmlString(
const std::string& str)
const;
178 void replace(std::string& str,
const std::string& toReplace,
const std::string& newValue)
const;
#define LIMA_ANALYSISDUMPERS_EXPORT
A graph that stores any data (annotations) referencing primarily nodes of a text anlaysis.
A graph structure for linguistic analysis.
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
Data used for the syntactic analyzis of texts.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
Holds an annotation graph and gives an API to manipulate it.
This class is the abstract base class of all elements that can be stored in a BoWText.
Manage initialization of InitializableObjects using configuration module and parameters.
std::string m_specificEntityTag
bool m_outputSentenceBoundaries
output sentence boundaries (enclosing sentence tags)
std::vector< std::string > m_featureTags
use additional vector (aligned) to store associated XML tags
BoWFeatures m_bowFeatures
use dedicated class for feature storage (easy initialization functions)
bool m_outputWords
output simple words
std::deque< std::string > m_featureNames
use additional vector (aligned) to store feature names
bool m_outputSpecificEntityParts
output parts of specific entities
bool m_outputCompoundParts
output also compound parts
std::string m_compoundTag
bool m_outputCompounds
output compounds
bool m_outputAllCompounds
output all partial compounds (created using BoWToken iterator)
WordFeatures m_features
use dedicated class for feature storage (easy initialization functions)
Compounds::BowGenerator * m_bowGenerator
std::string m_sentenceBoundaryTag
std::map< std::string, std::string > m_defaultFeatures
bool m_outputSpecificEntities
output specific entities
Parameters retrived in the configuration file:
An AnalysisData containing a LinguisticGraph with a language and an id.
A representation of a specific entity to store in the annotation graph.
This class points to a graph, its dependency graph and the structure that holds the maping between th...