LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
GenericXmlDumper.h
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/***************************************************************************
7 * Copyright (C) 2010 by CEA LIST *
8 * *
9 ***************************************************************************/
10#ifndef LIMA_LINGUISTICPROCESSING_SIMPLEXMLBOWDUMPER_H
11#define LIMA_LINGUISTICPROCESSING_SIMPLEXMLBOWDUMPER_H
12
14
16#include "BoWFeatureExtractor.h" // for compounds (stored in BoWTerms)
17
19
28#include "BowGeneration.h"
29
30namespace Lima {
31namespace LinguisticProcessing {
32namespace AnalysisDumpers {
33
34#define GENERICXMLDUMPER_CLASSID "GenericXmlDumper"
35
40{
41public:
43
44 virtual ~GenericXmlDumper();
45
46 virtual void init(
48 Manager* manager) override;
49
50 virtual LimaStatusCode process(AnalysisContent& analysis) const override;
51
52protected:
53 std::string m_graph;
56 std::deque<std::string> m_featureNames;
57 std::vector<std::string> m_featureTags;
58 std::map<std::string,std::string> m_defaultFeatures;
66 std::string m_wordTag;
69 std::string m_compoundTag;
71
72 // std::string m_property;
73// const Common::PropertyCode::PropertyAccessor* m_propertyAccessor;
74// const Common::PropertyCode::PropertyManager* m_propertyManager;
75//
76// // output of some specific properties (temporary: should be inserted in WordFeatures with XML output)
77// bool m_outputVerbTense;
78// bool m_outputTStatus;
79// const Common::PropertyCode::PropertyAccessor* m_tenseAccessor;
80// const Common::PropertyCode::PropertyManager* m_tenseManager;
81
82 // private member functions
83 void clearFeatures();
84 void initializeFeatures(const std::map<std::string,std::string>& features,
85 const std::deque<std::string>& featureOrder=std::deque<std::string>());
86
87 void xmlOutput(std::ostream& out,
88 AnalysisContent& analysis,
91 const Common::AnnotationGraphs::AnnotationData* annotationData,
92 const SyntacticAnalysis::SyntacticData* syntacticData) const;
93
94 void xmlOutputVertices(std::ostream& out,
95 AnalysisContent& analysis,
98 const Common::AnnotationGraphs::AnnotationData* annotationData,
99 const SyntacticAnalysis::SyntacticData* syntacticData,
100 const LinguisticGraphVertex begin,
101 const LinguisticGraphVertex end,
102 const FsaStringsPool& sp,
103 const uint64_t offset) const;
104
105 void xmlOutputVertex(std::ostream& out,
106 AnalysisContent& analysis,
110 const Common::AnnotationGraphs::AnnotationData* annotationData,
111 const SyntacticAnalysis::SyntacticData* syntacticData,
112 const FsaStringsPool& sp,
113 uint64_t offset,
114 std::set<LinguisticGraphVertex>& visited,
115 std::set<LinguisticGraphVertex>& alreadyStoredVertices) const;
116
117 void xmlOutputVertexInfos(std::ostream& out, Lima::AnalysisContent& analysis, LinguisticGraphVertex v, Lima::LinguisticProcessing::LinguisticAnalysisStructure::AnalysisGraph* graph, uint64_t offset) const;
118
119 void xmlOutputBoWInfos(std::ostream& out,
121 uint64_t offset) const;
122
129 std::pair<const SpecificEntities::SpecificEntityAnnotation*,LinguisticAnalysisStructure::AnalysisGraph*>
130 checkSpecificEntity(LinguisticGraphVertex v,
133 const Common::AnnotationGraphs::AnnotationData* annotationData) const;
134
135 bool xmlOutputSpecificEntity(std::ostream& out,
136 AnalysisContent& analysis,
139 const FsaStringsPool& sp,
140 uint64_t offset) const;
141
142 // hack to get compatible features between specific entities and words
143 // without having to define abstract Feature Extractors for SpecificEntityAnnotation
144 std::string specificEntityFeature(const SpecificEntities::SpecificEntityAnnotation* se,
145 const std::string& featureName,
146 const FsaStringsPool& sp,
147 uint64_t offset) const;
148
149 std::vector< boost::shared_ptr< Common::BagOfWords::BoWToken > >
150 checkCompound(LinguisticGraphVertex v,
153 const Common::AnnotationGraphs::AnnotationData* annotationData,
154 const SyntacticAnalysis::SyntacticData* syntacticData,
155 uint64_t offset,
156 std::set<LinguisticGraphVertex>& visited) const;
157
158 void xmlOutputCompound(std::ostream& out,
159 AnalysisContent& analysis,
160 boost::shared_ptr<Lima::Common::BagOfWords::AbstractBoWElement> token, Lima::LinguisticProcessing::LinguisticAnalysisStructure::AnalysisGraph* anagraph, Lima::LinguisticProcessing::LinguisticAnalysisStructure::AnalysisGraph* posgraph, const Lima::Common::AnnotationGraphs::AnnotationData* annotationData, const Lima::FsaStringsPool& sp, uint64_t offset) const;
161
162 /*void xmlOutputVertexInfos(std::ostream& out,
163 const LinguisticAnalysisStructure::Token* ft,
164 const std::vector<LinguisticAnalysisStructure::MorphoSyntacticData*>& data,
165 const FsaStringsPool& sp,
166 uint64_t offset,
167 LinguisticCode category=LinguisticCode(0)) const;
168
169 bool outputSpecificEntity(std::ostream& out,
170 const SpecificEntities::SpecificEntityAnnotation* se,
171 LinguisticAnalysisStructure::MorphoSyntacticData* data,
172 const LinguisticGraph* graph,
173 const FsaStringsPool& sp,
174 const uint64_t offset) const;
175 */
176 // string manipulation functions to protect XML entities
177 std::string xmlString(const std::string& str) const;
178 void replace(std::string& str, const std::string& toReplace, const std::string& newValue) const;
179
180};
181
182} // AnalysisDumpers
183} // LinguisticProcessing
184} // Lima
185
186#endif
#define LIMA_ANALYSISDUMPERS_EXPORT
A graph that stores any data (annotations) referencing primarily nodes of a text anlaysis.
A graph structure for linguistic analysis.
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
Data used for the syntactic analyzis of texts.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
Holds an annotation graph and gives an API to manipulate it.
This class is the abstract base class of all elements that can be stored in a BoWText.
Manage initialization of InitializableObjects using configuration module and parameters.
bool m_outputSentenceBoundaries
output sentence boundaries (enclosing sentence tags)
std::vector< std::string > m_featureTags
use additional vector (aligned) to store associated XML tags
BoWFeatures m_bowFeatures
use dedicated class for feature storage (easy initialization functions)
std::deque< std::string > m_featureNames
use additional vector (aligned) to store feature names
bool m_outputSpecificEntityParts
output parts of specific entities
bool m_outputAllCompounds
output all partial compounds (created using BoWToken iterator)
WordFeatures m_features
use dedicated class for feature storage (easy initialization functions)
Parameters retrived in the configuration file:
An AnalysisData containing a LinguisticGraph with a language and an id.
A representation of a specific entity to store in the annotation graph.
This class points to a graph, its dependency graph and the structure that holds the maping between th...
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236