LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
easyXmlDumper.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
20
21#include "easyXmlDumper.h"
22#include "EasyDumper.h"
24// #include "linguisticProcessing/core/LinguisticProcessors/HandlerStreamBuf.h"
32// #include "linguisticProcessing/common/BagOfWords/bowFileHeader.h"
42
43using namespace std;
44//using namespace boost;
45using namespace boost::tuples;
46
48using namespace Lima::Common::BagOfWords;
49using namespace Lima::Common::MediaticData;
50using namespace Lima::Common::AnnotationGraphs;
51
53//using namespace Lima::LinguisticProcessing::Automaton;
56
57typedef boost::color_traits<boost::default_color_type> Color;
58
59namespace Lima {
60namespace LinguisticProcessing {
61namespace AnalysisDumpers {
62namespace EasyXmlDumper {
63
64//***********************************************************************
65// constructors
66//***********************************************************************
68
71m_handler()
72{
73}
74
78
80 Manager* manager)
81{
83 LDEBUG << "EasyXmlDumper:: easyXmlDumper init!";
84 m_language = manager->getInitializationParameters().media;
86 try
87 {
88 m_typeMapping = unitConfiguration.getMapAtKey("typeMapping");
89 m_srcTag = unitConfiguration.getMapAtKey("srcTag");
90 m_tgtTag = unitConfiguration.getMapAtKey("tgtTag");
91 }
92 catch (NoSuchParam& )
93 {
94 LERROR << "EasyXmlDumper::init: parameter not found (typeMapping, srcTag and tgtTag must be specified)";
95 return;
96 }
97 try
98 {
99 m_graph = unitConfiguration.getParamsValueAtKey("graph");
100 }
101 catch (NoSuchParam& )
102 {
103 LDEBUG << "EasyXmlDumper:: graph parameter not found, using PosGraph";
104 m_graph = string("PosGraph");
105 }
106 try
107 {
108 m_handler=unitConfiguration.getParamsValueAtKey("handler");
109 }
110 catch (NoSuchParam& )
111 {
113 LERROR << "EasyXmlDumper::init: Missing parameter handler in EasyXmlDumper configuration";
114 throw InvalidConfiguration();
115 }
116}
117
119{
122
123 LinguisticMetaData* metadata = static_cast<LinguisticMetaData*>(analysis.getData("LinguisticMetaData"));
124 if (metadata == nullptr) {
125 LERROR << "EasyXmlDumper::process no LinguisticMetaData ! abort";
126 return MISSING_DATA;
127 }
128 string filename = metadata->getMetaData("FileName");
129 LDEBUG << "EasyXmlDumper::process Filename: " << filename;
130
131 LDEBUG << "handler will be: " << m_handler;
132// MediaId langid = static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(metadata->getMetaData("Lang"))).getMedia();
133 auto h = std::dynamic_pointer_cast<AnalysisHandlerContainer>(analysis.getData("AnalysisHandlerContainer"));
134 AbstractTextualAnalysisHandler* handler = static_cast<AbstractTextualAnalysisHandler*>(h->getHandler(m_handler));
135 if (handler==0)
136 {
137 LERROR << "EasyXmlDumper::process: handler " << m_handler << " has not been given to the core client";
138 return MISSING_DATA;
139 }
140
141 auto graph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData(m_graph));
142 if (graph == nullptr)
143 {
144 graph = new AnalysisGraph(m_graph,m_language,true,true);
145 analysis.setData(m_graph,graph);
146 }
147
148 auto syntacticData = std::dynamic_pointer_cast<SyntacticData>(analysis.getData("SyntacticData"));
149 if (syntacticData == nullptr)
150 {
151 syntacticData = new SyntacticAnalysis::SyntacticData(std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData(m_graph)),0);
152 syntacticData->setupDependencyGraph();
153 analysis.setData("SyntacticData",syntacticData);
154 }
155
156 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.getData("AnnotationData"));
157 if (annotationData == nullptr)
158 {
159 annotationData = std::make_shared<AnnotationData>();
160 if (std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph")) != 0)
161 {
162 std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"))->populateAnnotationGraph(annotationData, "AnalysisGraph");
163 }
164 analysis.setData("AnnotationData",annotationData);
165 }
166
167 handler->startAnalysis();
168 HandlerStreamBuf hsb(handler);
169 std::ostream outputStream(&hsb);
170
171 LDEBUG << "EasyXmlDumper:: process before printing heading";
172 auto anaGraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"));
173 auto posGraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("PosGraph"));
174 if (anaGraph != 0 && posGraph != 0)
175 {
176 LDEBUG << "EasyXmlDumper:: begin of posgraph";
177 std::vector< bool > alreadyDumpedTokens;
178 std::map< LinguisticAnalysisStructure::Token*, uint64_t > fullTokens;
180 uint64_t id = 0;
181 alreadyDumpedTokens.resize(num_vertices(*posGraph->getGraph()));
182 for (boost::tie(i, i_end) = vertices(*posGraph->getGraph()); i != i_end; ++i)
183 {
184 LDEBUG << "EasyXmlDumper:: examine posgraph for " << id;
185 alreadyDumpedTokens[id] = false;
186 fullTokens[get(vertex_token, *posGraph->getGraph(), *i)] = id;
187 id++;
188 }
189 /* No need for sentence boundaries in Easy input
190 LinguisticGraphVertex sentenceBegin = sb->getStartVertex();
191 SegmentationData::iterator sbItr = sb->begin();
192 LinguisticGraphVertex sentenceBegin = sb->getStartVertex();
193 SegmentationData::iterator sbItr = sb->begin();
194 */
195 LinguisticGraphVertex sentenceBegin = posGraph->firstVertex();
196 LinguisticGraphVertex sentenceEnd = posGraph->lastVertex();
197 string sentIdPrefix;
198 try {
199 sentIdPrefix = metadata->getMetaData("docid");
200 LDEBUG << "EasyXmlDumper:: retrieve sentence id " << sentIdPrefix;
202 sentIdPrefix = "";
203 }
204 if(sentIdPrefix.length() <= 0)
205 sentIdPrefix = "E";
206 /* No need for sentence boundaries in Easy input
207 while (sbItr != sb->end())
208 {
209 LinguisticGraphVertex sentenceEnd = *sbItr;
210 */
211 LDEBUG << "EasyXmlDumper:: inside posgraph while ";
212 dumpLimaData(outputStream,
213 sentenceBegin,
214 sentenceEnd,
215 *anaGraph,
216 *posGraph,
217 *annotationData,
218 *syntacticData,
219 "PosGraph",
220 alreadyDumpedTokens,
221 fullTokens,
222 sentIdPrefix);
223 /* No need for sentence boundaries in Easy input
224 sentenceBegin = sentenceEnd;
225 sbItr++;
226 }
227 */
228 LDEBUG << "EasyXmlDumper:: end of posgraph";
229 }
230
231 return SUCCESS_ID;
232}
233
234
235//***********************************************************************
236// main function for outputing the graph
237//***********************************************************************
238void EasyXmlDumper::dumpLimaData(std::ostream& os,
239 const LinguisticGraphVertex& begin,
240 const LinguisticGraphVertex& end,
241 const AnalysisGraph& anaGraph,
242 const AnalysisGraph& posGraph,
243 const AnnotationData& annotationData,
244 const SyntacticData& syntacticData,
245 const std::string& graphId,
246 std::vector< bool >& alreadyDumpedTokens,
247 std::map< LinguisticAnalysisStructure::Token*, uint64_t >& fullTokens,
248 std::string sentIdPrefix) const
249{
250
252 LDEBUG << "EasyXmlDumper:: dumpLimaData parameters: ";
253 LDEBUG << "EasyXmlDumper:: begin = " << begin;
254 LDEBUG << "EasyXmlDumper:: end = " << end;
255 LDEBUG << "EasyXmlDumper:: posgraph first vertex = " << posGraph.firstVertex();
256 LDEBUG << "EasyXmlDumper:: posgraph last vertex = " << posGraph.lastVertex();
257 LDEBUG << "EasyXmlDumper:: graphId = " << graphId;
258 LDEBUG << "EasyXmlDumper:: sentIdPrefix = " << sentIdPrefix;
259
260 // just in case we want to check alreadt dumped tokens' array
261 for (uint64_t i = 0; i<alreadyDumpedTokens.size(); i++)
262 {
263 if (alreadyDumpedTokens[i])
264 {
265 LDEBUG << "EasyXmlDumper:: already_dumped_tokens[" << i << "] =" << alreadyDumpedTokens[i];
266 }
267 }
268
269 std::string sentIdStr = sentIdPrefix;
270 if(find(m_sentIds.begin(), m_sentIds.end(), sentIdStr) != m_sentIds.end() || sentIdStr == "E" )
271 {
272 uint64_t sentIdsuffix = 0;
273 do{
274 sentIdsuffix++;
275 std::stringstream sentIdStream;
276 sentIdStream << sentIdPrefix << sentIdsuffix;
277 sentIdStr = sentIdStream.str();
278 }while(find(m_sentIds.begin(), m_sentIds.end(), sentIdStr) != m_sentIds.end());
279 }
280
281 LDEBUG << "EasyXmlDumper:: searching and extracting vertices and relations";
282 LinguisticGraph* anaGraphL = const_cast<LinguisticGraph*>(anaGraph.getGraph());
283 LinguisticGraph* posGraphL = const_cast<LinguisticGraph*>(posGraph.getGraph());
285 care.visitBoostGraph(begin,
286 end,
287 *anaGraphL,
288 *posGraphL,
289 annotationData,
290 syntacticData,
291 fullTokens,
292 alreadyDumpedTokens,
293 m_language);
294
295 LDEBUG << "EasyXmlDumper:: all found vertices and relations extracted";
298 care.splitCompoundTenses();
301
302 EasyDumper ed(care, m_typeMapping, m_srcTag, m_tgtTag, sentIdStr);
303 std::stringstream sentEasyStream;
304 ed.dump(sentEasyStream);
305 if(sentEasyStream.str().length() > 0)
306 {
307 // Makes object mutable for adding sentence ID
308 EasyXmlDumper* self = const_cast<EasyXmlDumper*>(this);
309 self->m_sentIds.push_back(sentIdStr);
310 os << "<E id=\"" << sentIdStr << "\">" << std::endl;
311 os << sentEasyStream.str();
312 os << "</E>" << std::endl;
313 }
314
315}
316
317} // end namespace EasyXmlDumper
318} // end namespace AnalysisDumpers
319} // end namespace LinguisticProcessings
320} // end namespace Lima
This file is the main header file for the data related to annotation graphs.
extracts forms and relations from boost graph (origninally, from XML file)
A graph that stores the relations of syntactic dependency between the elements of a DependencyGraph.
dump the content of the analysis graph in Easy XML format
#define LDEBUG
Definition LimaCommon.h:157
#define LERROR
Definition LimaCommon.h:161
A graph structure for linguistic analysis.
LinguisticGraph::vertex_iterator LinguisticGraphVertexIt
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define DUMPERLOGINIT
Defines a Factory to create Object of type Base.
virtual void startAnalysis()=0
function called by the LIMA analyzer on start of a new document
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Holds an annotation graph and gives an API to manipulate it.
Holds linguistic data for one language.
const MediaData & mediaData(MediaId media) const
std::map< std::string, std::string > & getMapAtKey(const std::string &key)
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Use this exception to signal an error in one of the configuration files.
Definition LimaCommon.h:345
void visitBoostGraph(const LinguisticGraphVertex &v, const LinguisticGraphVertex &end, const LinguisticGraph &anaGraph, const LinguisticGraph &posGraph, const Common::AnnotationGraphs::AnnotationData &annotationData, const SyntacticAnalysis::SyntacticData &syntacticData, std::map< LinguisticAnalysisStructure::Token *, uint64_t > &fullTokens, std::vector< bool > &alreadyDumpedTokens, const MediaId &language)
Dumps all the content of the analysis on an XML stream.
const Common::PropertyCode::PropertyCodeManager * m_propertyCodeManager
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
void dumpLimaData(std::ostream &os, const LinguisticGraphVertex &begin, const LinguisticGraphVertex &end, const LinguisticAnalysisStructure::AnalysisGraph &anaGraph, const LinguisticAnalysisStructure::AnalysisGraph &posGraph, const Common::AnnotationGraphs::AnnotationData &annotationData, const SyntacticAnalysis::SyntacticData &syntacticData, const std::string &graphId, std::vector< bool > &alreadyDumpedTokens, std::map< LinguisticAnalysisStructure::Token *, uint64_t > &easyTokens, std::string sentIdPrefix) const
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
contains metadata to be kept during the analysis: global generic metadata: file name,...
const std::string & getMetaData(const std::string &id) const
This class points to a graph, its dependency graph and the structure that holds the maping between th...
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
static void updateCurrentTime(const std::string &taskCategory=std::string(""))
store current time for new elapsed time computation
boost::color_traits< boost::default_color_type > Color
dump the content of the analysis graph in Easy XML format
#define EASYXMLDUMPER_CLASSID
SimpleFactory< MediaProcessUnit, EasyXmlDumper > easyXmlDumperFactory(EASYXMLDUMPER_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
@ MISSING_DATA
Definition LimaCommon.h:243
STL namespace.
launch exception related to the configuration file parsing