LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
SimpleXmlDumper.cpp
Go to the documentation of this file.
1// Copyright 2002-2020 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/***************************************************************************
7 * Copyright (C) 2010 by CEA LIST *
8 * *
9 ***************************************************************************/
10#include "SimpleXmlDumper.h"
11#include "TextDumper.h" // for lTokenPosition comparison function to order tokens
12
13// #include "linguisticProcessing/core/LinguisticProcessors/HandlerStreamBuf.h"
26
27#include <boost/graph/properties.hpp>
28
29#include <fstream>
30#include <deque>
31#include <queue>
32#include <iostream>
33
34// using namespace std;
35//using namespace boost;
36using namespace boost::tuples;
37using namespace Lima::Common;
38using namespace Lima::Common::AnnotationGraphs;
39using namespace Lima::Common::MediaticData;
44
45namespace Lima {
46namespace LinguisticProcessing {
47namespace AnalysisDumpers {
48
50
53m_graph("PosGraph"),
54m_property("MICRO"),
55m_propertyAccessor(0),
56m_propertyManager(0),
57m_outputVerbTense(false),
58m_outputTStatus(false),
59m_encapsulatingTag(""),
60m_tenseAccessor(0),
61m_tenseManager(0)
62{
63}
64
68
71 Manager* manager)
72
73{
74 AbstractTextualAnalysisDumper::init(unitConfiguration,manager);
75
77 try
78 {
79 m_graph=unitConfiguration.getParamsValueAtKey("graph");
80 }
81 catch (NoSuchParam& ) {} // keep default value
82
83 try {
84 m_property=unitConfiguration.getParamsValueAtKey("property");
85 }
86 catch (NoSuchParam& ) {} // keep default value
87
88 const auto& codeManager=static_cast<const LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager();
89 m_propertyAccessor=&codeManager.getPropertyAccessor(m_property);
90 m_propertyManager=&codeManager.getPropertyManager(m_property);
91
92 try {
93 std::string str=unitConfiguration.getParamsValueAtKey("outputTStatus");
94 if (str=="yes" || str=="1") {
95 m_outputTStatus=true;
96 LINFO << "activate outputTStatus";
97 }
98 }
99 catch (NoSuchParam& ) {} // keep default value
100
101 try {
102 std::string str=unitConfiguration.getParamsValueAtKey("outputVerbTense");
103 if (str=="yes" || str=="1") {
105 QString timeCode = static_cast<const LanguageData&>(
106 Common::MediaticData::MediaticData::single().mediaData(m_language)).getLimaToLanguageCodeMappingValue("TIME");
107 m_tenseManager=&codeManager.getPropertyManager(timeCode.toUtf8().constData());
108 m_tenseAccessor=&codeManager.getPropertyAccessor(timeCode.toUtf8().constData());
109 LINFO << "activate outputVerbTense";
110 }
111 }
112 catch (NoSuchParam& ) {} // keep default value
113
114 try {
115 std::string str=unitConfiguration.getParamsValueAtKey("encapsulatingTag");
117 LINFO << "set encapsulatingTag: "<< m_encapsulatingTag;
118 }
119 catch (NoSuchParam& ) {
120 LDEBUG << "no encapsulatingTag option set. Keep default";
121 } // keep default value
122
123
124}
125
127process(AnalysisContent& analysis) const
128{
131 LDEBUG << "SimpleXmlDumper::process";
132
133 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.getData("LinguisticMetaData"));
134 if (metadata == 0)
135 {
136 LERROR << "no LinguisticMetaData ! abort";
137 return MISSING_DATA;
138 }
139
140 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"));
141 if (anagraph==0)
142 {
143 LERROR << "no graph 'AnaGraph' available !";
144 return MISSING_DATA;
145 }
146 auto posgraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("PosGraph"));
147 if (posgraph==0)
148 {
149 LERROR << "no graph 'PosGraph' available !";
150 return MISSING_DATA;
151 }
152 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.getData("AnnotationData"));
153 if (annotationData==0)
154 {
155 LERROR << "no annotation graph available !";
156 return MISSING_DATA;
157 }
158
159 auto dstream = initialize(analysis);
160 xmlOutput(dstream->out(), analysis, anagraph.get(), posgraph.get(), annotationData.get());
161
162 TimeUtils::logElapsedTime("SimpleXmlDumper");
163 return SUCCESS_ID;
164}
165
167 std::ostream& out,
168 AnalysisContent& analysis,
169 AnalysisGraph* anagraph,
170 AnalysisGraph* posgraph,
171 const Common::AnnotationGraphs::AnnotationData* annotationData) const
172{
174
175 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.getData("LinguisticMetaData"));
176
177 auto sb = std::dynamic_pointer_cast<SegmentationData>(analysis.getData("SentenceBoundaries"));
178
180
181 if (m_encapsulatingTag!="" && !m_append) {
182 out << "<?xml version=\"1.0\" encoding=\"UTF-8\" standalone=\"no\" ?>" << std::endl;
183 out << "<"<< m_encapsulatingTag <<">" << std::endl;
184 }
185
186 QString docId("");
187 try {
188 docId.fromStdString(metadata->getMetaData("FileName"));
189 }
191 }
192
194 out << "<DOC id=\"" << docId.toHtmlEscaped().toStdString()
195 << "\" lang=\"" << language << "\">" << std::endl;
196
197 if (sb == nullptr)
198 {
199 LWARN << "no SentenceBoundaries";
200 }
201
202 if (sb == nullptr)
203 {
204 // no sentence bounds : there can be specific entities,
205 // but no compounds (syntactic analysis depend on sentence bounds)
206 // dump whole text at once
208 anagraph,
209 posgraph,
210 annotationData,
211 anagraph->firstVertex(),
212 anagraph->lastVertex(),
213 sp,
214 metadata->getStartOffset());
215 }
216 else
217 {
218 // ??OME2 uint64_t nbSentences(sb->size());
219 auto nbSentences = (sb->getSegments()).size();
220 LDEBUG << "SimpleXmlDumper: "<< nbSentences << " sentences found";
221 for (uint64_t i = 0; i < nbSentences; i++)
222 {
223 // ??OME2 LinguisticGraphVertex sentenceBegin=(*sb)[i].getFirstVertex();
224 // LinguisticGraphVertex sentenceEnd=(*sb)[i].getLastVertex();
225 auto sentenceBegin = (sb->getSegments())[i].getFirstVertex();
226 auto sentenceEnd = (sb->getSegments())[i].getLastVertex();
227
228 // if (sentenceEnd==posgraph->lastVertex()) {
229 // continue;
230 // }
231
232 LDEBUG << "dump sentence between " << sentenceBegin << " and " << sentenceEnd;
233 LDEBUG << "dump simple terms for this sentence";
234
235 std::ostringstream oss;
237 anagraph,
238 posgraph,
239 annotationData,
240 sentenceBegin,
241 sentenceEnd,
242 sp,
243 metadata->getStartOffset());
244 std::string str = oss.str();
245 if (str.empty())
246 {
247 LDEBUG << "nothing to dump in this sentence";
248 }
249 else
250 {
251 out << "<s id=\"" << i << "\">" << std::endl
252 << str
253 << "</s>" << std::endl;
254 }
255 }
256 }
257 out << "</DOC>" << std::endl;
258 if(m_encapsulatingTag != "") {
259 out << "</"<< m_encapsulatingTag <<">" << std::endl;
260 }
261}
262
264 std::ostream& out,
265 AnalysisGraph* anagraph,
266 AnalysisGraph* posgraph,
267 const Common::AnnotationGraphs::AnnotationData* annotationData,
268 const LinguisticGraphVertex begin,
269 const LinguisticGraphVertex end,
270 const FsaStringsPool& sp,
271 const uint64_t offset) const
272{
274 LDEBUG << "SimpleXmlDumper: ========================================";
275 LDEBUG << "SimpleXmlDumper: outputXml from vertex " << begin << " to vertex " << end;
276
277 auto graph = posgraph->getGraph();
278 auto lastVertex = posgraph->lastVertex();
279
280 std::map<Token*,
281 std::vector< std::pair<LinguisticGraphVertex, MorphoSyntacticData*> >,
282 lTokenPosition> sortedTokens;
283
284 std::queue<LinguisticGraphVertex> toVisit;
285 std::set<LinguisticGraphVertex> visited;
286
287 LinguisticGraphOutEdgeIt outItr, outItrEnd;
288
289 // output vertices between begin and end,
290 // but do not include begin (beginning of text or previous end of sentence) and include end (end of sentence)
291 toVisit.push(begin);
292
293 auto first = true;
294 auto last = false;
295 while (!toVisit.empty())
296 {
297 auto v = toVisit.front();
298 toVisit.pop();
299 if (last || v == lastVertex)
300 {
301 continue;
302 }
303 if (v == end)
304 {
305 last = true;
306 }
307
308 for (boost::tie(outItr, outItrEnd) = out_edges(v, *graph);
309 outItr != outItrEnd; outItr++)
310 {
311 auto next = target(*outItr,*graph);
312 if (visited.find(next) == visited.end())
313 {
314 visited.insert(next);
315 toVisit.push(next);
316 }
317 }
318
319 if (first)
320 {
321 first=false;
322 }
323 else
324 {
325 auto t = get(vertex_token,*graph,v);
326 if( t != 0)
327 {
328 sortedTokens[t].push_back(make_pair(v, get(vertex_data, *graph, v)));
329 }
330 }
331 }
332
333 for (auto it = sortedTokens.begin(), it_end = sortedTokens.end();
334 it!=it_end; it++)
335 {
336 if ((*it).second.size() == 0)
337 {
338 continue;
339 }
340
341 // if several interpreation of the token (i.e several LinguisticGraphVertex associated),
342 // it is not a specific entity, => then just print all interpretations
343 else if ((*it).second.size() > 1)
344 {
345 std::vector<MorphoSyntacticData*> data;
346 for (auto d = (*it).second.begin(), d_end = (*it).second.end();
347 d != d_end; d++)
348 {
349 data.push_back((*d).second);
350 }
351 xmlOutputVertexInfos(out, (*it).first, data, sp, offset);
352 }
353 else
354 {
355 xmlOutputVertex(out, (*it).second[0].first, (*it).first, anagraph,
356 posgraph, annotationData, sp, offset);
357 }
358 }
359}
360
362 std::ostream& out,
364 const Token* ft,
365 AnalysisGraph* anagraph,
366 AnalysisGraph* posgraph,
367 const Common::AnnotationGraphs::AnnotationData* annotationData,
368 const FsaStringsPool& sp,
369 uint64_t offset) const
370{
371 auto data = get(vertex_data, *(posgraph->getGraph()), v);
372
373 // first, check if vertex corresponds to a specific entity found before pos
374 // tagging (i.e. in analysis graph)
375 auto anaVertices = annotationData->matches("PosGraph", v, "AnalysisGraph");
376 // note: anaVertices size should be 0 or 1
377 for (auto anaVerticesIt = anaVertices.begin();
378 anaVerticesIt != anaVertices.end(); anaVerticesIt++)
379 {
380 auto matches = annotationData->matches("AnalysisGraph",
381 *anaVerticesIt,
382 "annot");
383 for (auto vx : matches)
384 {
385 if (annotationData->hasAnnotation(vx, QString::fromUtf8("SpecificEntity")))
386 {
387 auto se = annotationData->annotation(
388 vx, QString::fromUtf8("SpecificEntity")).
389 pointerValue<SpecificEntityAnnotation>();
390 if (outputSpecificEntity(out, se, data, anagraph->getGraph(), sp, offset))
391 {
392 return;
393 }
394 else
395 {
397 LERROR << "failed to output specific entity for vertex " << v;
398 }
399 }
400 }
401 }
402
403 // then check if vertex corresponds to a specific entity found after POS tagging
404 auto matches = annotationData->matches("PosGraph", v, "annot");
405 for (auto vx : matches)
406 {
407 if (annotationData->hasAnnotation(vx, QString::fromUtf8("SpecificEntity")))
408 {
409 //BoWToken* se = createSpecificEntity(v,*it, annotationData, anagraph, posgraph, offsetBegin);
410 auto se = annotationData->annotation(vx, QString::fromUtf8("SpecificEntity")).
411 pointerValue<SpecificEntityAnnotation>();
412
413 if (outputSpecificEntity(out, se, data, posgraph->getGraph(), sp, offset))
414 {
415 return;
416 }
417 else
418 {
420 LERROR << "failed to output specific entity for vertex " << v;
421 }
422 }
423 }
424
425 // if not a specific entity at all, output simple word infos
426 xmlOutputVertexInfos(out, ft, std::vector<MorphoSyntacticData*>(1,data), sp, offset);
427}
428
430 std::ostream& out,
431 const Token* ft,
432 const std::vector<MorphoSyntacticData*>& data,
433 const FsaStringsPool& sp,
434 uint64_t offset,
435 LinguisticCode category) const
436{
438 auto position = ft->position() + offset;
439
440 auto output = false;
441
442 for (auto dataItr = data.cbegin(), dataItr_end = data.cend();
443 dataItr!=dataItr_end; dataItr++)
444 {
445 auto data = *dataItr;
446 sort(data->begin(), data->end(), sorter);
447 StringsPoolIndex norm(0),curNorm(0);
448 LinguisticCode micro,curMicro;
449 for (auto elemItr = data->cbegin(); elemItr != data->cend(); elemItr++)
450 {
451 curNorm = elemItr->normalizedForm;
452 curMicro = m_propertyAccessor->readValue(elemItr->properties);
453 if ((curNorm != norm) || (curMicro != micro))
454 {
455 norm = curNorm;
456 micro = curMicro;
457
458 // if category is specified, output first data (lemma) compatible with this category
459 if (category == L_NONE || category==curMicro)
460 {
461 out << "<w p=\"" << position << "\""
462 << " inf=\"" << xmlString(ft->stringForm().toStdString()) << "\""
463 << " pos=\"" << m_propertyManager->getPropertySymbolicValue(curMicro) << "\""
464 << " lem=\"" << xmlString(sp[norm].toStdString()) << "\"";
465 if (m_outputTStatus)
466 {
467 out << " tok=\"" << ft->status().defaultKey().toStdString() << "\"";
468 }
470 {
471 auto tense = m_tenseAccessor->readValue(elemItr->properties);
472 if (tense != NONE_1)
473 {
474 out << " tense=\""
475 << m_tenseManager->getPropertySymbolicValue(tense) << "\"";
476 }
477 }
478 out << "/>" << std::endl;
479 output=true;
480 }
481 }
482 }
483 }
484
485 // if category is specified and no matching data is found for this category: use first data
486 if (category != L_NONE && !output && data.front()->size()>0 )
487 {
488 auto norm = data.front()->begin()->normalizedForm;
489 out << "<w p=\"" << position << "\""
490 << " inf=\"" << xmlString(ft->stringForm().toStdString()) << "\""
491 << " pos=\"" << m_propertyManager->getPropertySymbolicValue(category) << "\""
492 << " lem=\"" << xmlString(sp[norm].toStdString()) << "\""
493 << "/>" << std::endl;
494 }
495}
496
498 std::ostream& out,
499 const SpecificEntityAnnotation* se,
501 const LinguisticGraph* graph,
502 const FsaStringsPool& sp,
503 const uint64_t offset) const
504{
505 if (se == nullptr)
506 {
508 LERROR << "missing specific entity annotation";
509 return false;
510 }
511
512 std::string typeName;
513 std::string norm;
514 try
515 {
517 typeName = str.toStdString();
518 }
519 catch (std::exception& ) {
521 LERROR << "Undefined entity type " << se->getType();
522 return false;
523 }
524 out
525 << "<e type=\"" << typeName << "\""
526 << " inf=\"" << xmlString(sp[se->getString()].toStdString()) << "\""
527 << " norm=\"" << xmlString(sp[se->getNormalizedForm()].toStdString()) << "\""
528 << ">" << std::endl;
529
530 // take as category for parts the category for the named entity
531 auto category = m_propertyAccessor->readValue(data->begin()->properties);
533 LDEBUG << "Using category "
535 << " for specific entity of type " << typeName;
536
537 // get the parts of the named entity match
538 // use the category of the named entity for all elements
539 for (auto m = se->vertices().cbegin(); m != se->vertices().cend(); m++)
540 {
541 auto token = get(vertex_token, *graph, *m);
542 if (token != nullptr)
543 {
544 auto vertexData = get(vertex_data, *graph, *m);
545 xmlOutputVertexInfos(out, token,
546 std::vector<MorphoSyntacticData*>(1, vertexData),
547 sp,offset,category);
548 }
549 }
550 out << "</e>" << std::endl;
551 return true;
552}
553
554// string manipulation functions to protect XML entities
555std::string SimpleXmlDumper::xmlString(const std::string& inputStr) const
556{
557 // protect XML entities
558 std::string str(inputStr);
559 replace(str,"&", "&amp;");
560 replace(str,"<", "&lt;");
561 replace(str,">", "&gt;");
562 replace(str,"\"", "&quot;");
563 replace(str,"\n", "\n");
564 return str;
565}
566
567void SimpleXmlDumper::replace(std::string& str,
568 const std::string& toReplace,
569 const std::string& newValue) const
570{
571 auto oldLen = toReplace.size();
572 auto newLen = newValue.size();
573 auto i = str.find(toReplace);
574 while (i != std::string::npos)
575 {
576 str.replace(i, oldLen, newValue);
577 i += newLen;
578 i = str.find(toReplace, i);
579 }
580}
581
582
583} // AnalysisDumper
584} // LinguisticProcessing
585} // Lima
#define LWARN
Definition LimaCommon.h:160
#define LDEBUG
Definition LimaCommon.h:157
#define LINFO
Definition LimaCommon.h:158
#define LERROR
Definition LimaCommon.h:161
A graph structure for linguistic analysis.
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define DUMPERLOGINIT
Defines a Factory to create Object of type Base.
#define SIMPLEXMLDUMPER_CLASSID
#define NONE_1
Definition StdBitset.h:339
#define L_NONE
Definition StdBitset.h:338
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
Holds an annotation graph and gives an API to manipulate it.
std::set< AnnotationGraphVertex > matches(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second) const
Gets the set of vertices matched in the second graph by the given vertex of the first graph.
const GenericAnnotation & annotation(AnnotationGraphVertex v1, AnnotationGraphVertex v2, const LimaString &annot) const
Holds linguistic data for one language.
const FsaStringsPool & stringsPool(MediaId med) const
const MediaData & mediaData(MediaId media) const
LimaString getEntityName(const EntityType &type) const
const std::string & media(MediaId media) const
LinguisticCode readValue(const LinguisticCode &code) const
read a property in a coded int.
const std::string & getPropertySymbolicValue(const LinguisticCode &value) const
The coded property value can hold several property data.
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
static std::size_t size() noexcept
Definition StdBitset.h:94
std::shared_ptr< DumperStream > initialize(AnalysisContent &analysis) const
virtual void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
void xmlOutputVertices(std::ostream &out, LinguisticAnalysisStructure::AnalysisGraph *anagraph, LinguisticAnalysisStructure::AnalysisGraph *posgraph, const Common::AnnotationGraphs::AnnotationData *annotationData, const LinguisticGraphVertex begin, const LinguisticGraphVertex end, const FsaStringsPool &sp, const uint64_t offset) const
void xmlOutputVertexInfos(std::ostream &out, const LinguisticAnalysisStructure::Token *ft, const std::vector< LinguisticAnalysisStructure::MorphoSyntacticData * > &data, const FsaStringsPool &sp, uint64_t offset, LinguisticCode category=L_NONE) const
bool outputSpecificEntity(std::ostream &out, const SpecificEntities::SpecificEntityAnnotation *se, LinguisticAnalysisStructure::MorphoSyntacticData *data, const LinguisticGraph *graph, const FsaStringsPool &sp, const uint64_t offset) const
const Common::PropertyCode::PropertyManager * m_tenseManager
virtual void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
const Common::PropertyCode::PropertyAccessor * m_tenseAccessor
void replace(std::string &str, const std::string &toReplace, const std::string &newValue) const
virtual LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
void xmlOutputVertex(std::ostream &out, LinguisticGraphVertex v, const LinguisticAnalysisStructure::Token *ft, LinguisticAnalysisStructure::AnalysisGraph *anagraph, LinguisticAnalysisStructure::AnalysisGraph *posgraph, const Common::AnnotationGraphs::AnnotationData *annotationData, const FsaStringsPool &sp, uint64_t offset) const
const Common::PropertyCode::PropertyManager * m_propertyManager
const Common::PropertyCode::PropertyAccessor * m_propertyAccessor
void xmlOutput(std::ostream &out, AnalysisContent &analysis, LinguisticAnalysisStructure::AnalysisGraph *anagraph, LinguisticAnalysisStructure::AnalysisGraph *posgraph, const Common::AnnotationGraphs::AnnotationData *annotationData) const
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
A representation of a specific entity to store in the annotation graph.
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
static void logElapsedTime(const std::string &mess, const std::string &taskCategory=std::string(""))
log the number of microseconds since last UpdateCurrentTime
static void updateCurrentTime(const std::string &taskCategory=std::string(""))
store current time for new elapsed time computation
bool hasAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
SimpleFactory< MediaProcessUnit, SimpleXmlDumper > simpleXmlDumperFactory(SIMPLEXMLDUMPER_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
@ MISSING_DATA
Definition LimaCommon.h:243
PUGI__FN void sort(I begin, I end, const Pred &pred)
Definition pugixml.cpp:7549
launch exception related to the configuration file parsing