LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
CompoundsXmlLogger.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/***************************************************************************
7 * Copyright (C) 2004-2012 by CEA LIST *
8 * *
9 ***************************************************************************/
10#include "CompoundsXmlLogger.h"
11#include "BowGeneration.h"
12
23#include <iostream>
24#include <fstream>
25#include <queue>
26
27//using namespace boost;
28using namespace boost::tuples;
29
30typedef boost::color_traits<boost::default_color_type> Color;
31
32using namespace Lima::Common;
33using namespace Lima::Common::BagOfWords;
34using namespace Lima::Common::MediaticData;
35using namespace Lima::Common::AnnotationGraphs;
40
41namespace Lima
42{
43namespace LinguisticProcessing
44{
45namespace Compounds
46{
47
49
51AbstractLinguisticLogger(".compounds.xml")
52{
53 m_bowGenerator = new BowGenerator();
54}
55
56
58{
59 delete m_bowGenerator;
60}
61
64 Manager* manager)
65
66{
67 AbstractLinguisticLogger::init(unitConfiguration,manager);
68 m_language=manager->getInitializationParameters().media;
69 m_bowGenerator->init(unitConfiguration, m_language);
70}
71
73 AnalysisContent& analysis) const
74{
75 Lima::TimeUtilsController timer("CompoundsXmlLogger");
76
77 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.getData("LinguisticMetaData"));
78 if (metadata == 0)
79 {
81 LERROR << "no LinguisticMetaData ! abort";
82 return MISSING_DATA;
83 }
84
85 std::ofstream outputStream;
86 if (!openLogFile(outputStream,metadata->getMetaData("FileName"))) {
88 LERROR << "Error: cannot open log file";
90 }
92
93 auto syntacticData = std::dynamic_pointer_cast<const SyntacticData>(analysis.getData("SyntacticData"));
94 if (syntacticData==0)
95 {
96 LERROR << "no SyntacticData ! abort";
97 return MISSING_DATA;
98 }
99
100 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"));
101 if (anagraph==0)
102 {
103 LERROR << "no AnalysisGraph ! abort";
104 return MISSING_DATA;
105 }
106 auto posgraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("PosGraph"));
107 if (posgraph==0)
108 {
109 LERROR << "no PosGraph ! abort";
110 return MISSING_DATA;
111 }
112 auto sb = std::dynamic_pointer_cast<SegmentationData>(analysis.getData("SentenceBoundaries"));
113 if (sb==0)
114 {
115 LERROR << "no SentenceBounds ! abort";
116 return MISSING_DATA;
117 }
118 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.getData("AnnotationData"));
119 if (annotationData==0)
120 {
121 LERROR << "no annotation graph available !";
122 return MISSING_DATA;
123 }
124
125 std::set< std::pair<size_t, size_t> > alreadyDumped;
126
127 outputStream << "<?xml version='1.0' encoding='UTF-8'?>" << std::endl;
128 outputStream << "<compounds_dump>" << std::endl;
129
130
131 //LinguisticGraphVertex sentenceBegin=sb->getStartVertex();
132 // ??OME2 SegmentationData::iterator sbItr=sb->begin();
133 std::vector<Segment>::iterator sbItr=(sb->getSegments()).begin();
134
135 uint64_t sentNum = 1;
136 // ??OME2 while (sbItr!=sb->end())
137 while (sbItr!=(sb->getSegments()).end())
138 {
139 LinguisticGraphVertex beginSentence=sbItr->getFirstVertex();
140 LinguisticGraphVertex endSentence=sbItr->getLastVertex();
141
142 dumpLimaData(outputStream,
143 sentNum,
144 beginSentence,
145 endSentence,
146 anagraph.get(),
147 posgraph.get(),
148 syntacticData.get(),
149 annotationData.get());
150
151 sbItr++;
152 sentNum++;
153 }
154
155 outputStream << "</compounds_dump>" << std::endl;
156 outputStream.close();
157
158 return SUCCESS_ID;
159}
160
161
162//***********************************************************************
163// main function for outputing the graph
164//***********************************************************************
165void CompoundsXmlLogger::dumpLimaData(
166 std::ostream& os,
167 uint64_t sentNum,
168 const LinguisticGraphVertex begin,
169 const LinguisticGraphVertex end,
170 const AnalysisGraph* anagraph,
171 const AnalysisGraph* posgraph,
172 const SyntacticData* syntacticData,
173 const Common::AnnotationGraphs::AnnotationData* annotationData,
174 const uint64_t offsetBegin) const
175{
176// COMPOUNDSLOGINIT;
177 // LinguisticGraph* graph = const_cast< LinguisticGraph* >(posgraph->getGraph());
178 // go through the graph, add BoWTokens that are not in complex terms
179
180 os << "<sentence num=\"" << sentNum << "\" >" << std::endl;
181 // dump compounds
182 os << "<compounds>" << std::endl;
183 LinguisticGraphVertex firstVx = posgraph->firstVertex();
184 LinguisticGraphVertex lastVx = posgraph->lastVertex();
185
186 const LinguisticGraph& lanagraph=*(anagraph->getGraph());
187 const LinguisticGraph& lposgraph=*(posgraph->getGraph());
188 std::set< std::string > alreadyStored;
189 std::set<LinguisticGraphVertex> visited;
190 std::queue<LinguisticGraphVertex> toVisit;
191 toVisit.push(begin);
192
193 LinguisticGraphOutEdgeIt outItr,outItrEnd;
194 while (!toVisit.empty())
195 {
196 LinguisticGraphVertex v=toVisit.front();
197 toVisit.pop();
198 if (v == end) {
199 continue;
200 }
201
202 for (boost::tie(outItr,outItrEnd)=out_edges(v,lposgraph);
203 outItr!=outItrEnd;
204 outItr++)
205 {
206 LinguisticGraphVertex next=target(*outItr,lposgraph);
207 if (visited.find(next)==visited.end())
208 {
209 visited.insert(next);
210 toVisit.push(next);
211 }
212 }
213
214 if (v != firstVx && v != lastVx)
215 {
217// LDEBUG << "hasAnnotation("<<v<<", CompoundTokenAnnotation): "
218// << annotationData->hasAnnotation(v, Common::Misc::utf8stdstring2limastring("CompoundTokenAnnotation"));
219 //std::set< uint64_t > cpdsHeads = annotationData->matches("PosGraph", v, "cpdHead"); portage 32 64
220 std::set< AnnotationGraphVertex > cpdsHeads = annotationData->matches("PosGraph", v, "cpdHead");
221 if (!cpdsHeads.empty())
222 {
223 std::set< AnnotationGraphVertex >::const_iterator cpdsHeadsIt, cpdsHeadsIt_end;
224 cpdsHeadsIt = cpdsHeads.begin(); cpdsHeadsIt_end = cpdsHeads.end();
225 for (; cpdsHeadsIt != cpdsHeadsIt_end; cpdsHeadsIt++)
226 {
227 AnnotationGraphVertex agv = *cpdsHeadsIt;
228 std::vector<std::pair< boost::shared_ptr< BoWRelation >, boost::shared_ptr< BoWToken > > > bowTokens =
229 m_bowGenerator->buildTermFor(agv, agv, lanagraph, lposgraph, offsetBegin,
230 syntacticData, annotationData, visited);
231 for (auto bowItr=bowTokens.begin();
232 bowItr!=bowTokens.end();
233 bowItr++)
234 {
235 std::string elem = (*bowItr).second->getIdUTF8String();
236 if (alreadyStored.find(elem) != alreadyStored.end())
237 { // already stored
238 // LDEBUG << "BuildBoWTokenListVisitor: BoWToken already stored. Skipping it.";
239 }
240 else
241 {
242 outputCompound(os,&*(*bowItr).second,offsetBegin);
243 alreadyStored.insert(elem);
244 }
245 }
246 }
247 }
248 }
249 }
250 os << "</compounds>" << std::endl;
251 os << "</sentence>" << std::endl;
252}
253
254//***********************************************************************
255// output functions
256//***********************************************************************
257
258
259uint64_t CompoundsXmlLogger::getPosition(const uint64_t position,
260 const uint64_t offsetBegin) const
261{
262 return (offsetBegin+position);
263}
264
265
266void CompoundsXmlLogger::outputCompound(
267 std::ostream& os,
268 const BoWToken* compound,
269 const uint64_t offsetBegin) const
270{
272 LDEBUG << "Outputing compound: " << compound;
273 auto pos = getPosition(compound->getPosition(),offsetBegin);
274 auto form = compound->getLemma();
275 auto cat = static_cast<const Common::MediaticData::LanguageData&>(
276 Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager().getPropertyManager("MACRO").getPropertySymbolicValue(compound->getCategory());
277 os << "<compound pos=\"" << pos << "\" form=\"" << form.toStdString()
278 << "\" cat=\"" << cat << "\" >" << std::endl;
279 auto term = dynamic_cast<const BoWComplexToken*>(compound);
280 if (term != nullptr)
281 {
282 auto headId = term->getHead();
283 auto partIt = term->getParts().cbegin();
284 auto partIt_end = term->getParts().cend();
285 for (auto partId = uint64_t(0); partIt != partIt_end; partIt++, partId++)
286 {
287 boost::shared_ptr< BoWToken > partTok = (*partIt).get<1>();
288 // bool head = (*partIt).second;
289 os << "<part head=\"" << std::boolalpha << (partId == headId) << "\" >" << std::endl;
290 if (boost::dynamic_pointer_cast<BoWComplexToken>(partTok))
291 {
292 outputCompound(os, &*partTok, offsetBegin);
293 }
294 else
295 {
296 LDEBUG << " part: " << *partTok;
297 uint64_t partPos = getPosition(partTok->getPosition(),offsetBegin);
298 LimaString partForm = partTok->getLemma();
299 std::string partCat = static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager().getPropertyManager("MACRO").getPropertySymbolicValue(partTok->getCategory()) ;
300 os << "<word pos=\"" << partPos << "\" form=\""
301 << partForm.toStdString()
302 << "\" cat=\"" << partCat << "\" />" << std::endl;
303 }
304 os << "</part>" << std::endl;
305 }
306 }
307 os << "</compound>" << std::endl;
308}
309
310} // SyntacticAnalysis
311} // LinguisticProcessing
312} // Lima
This file is the main header file for the data related to annotation graphs.
#define COMPOUNDSXMLLOGGER_CLASSID
#define LDEBUG
Definition LimaCommon.h:157
#define LERROR
Definition LimaCommon.h:161
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define COMPOUNDSLOGINIT
#define SALOGINIT
Defines a Factory to create Object of type Base.
Data used for the syntactic analyzis of texts.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
Holds an annotation graph and gives an API to manipulate it.
std::set< AnnotationGraphVertex > matches(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second) const
Gets the set of vertices matched in the second graph by the given vertex of the first graph.
This is a complex token for an index.
uint64_t getHead() const
add a part in the list of parts of the complex token.
This class contains the representation of an element of the bag of words.
Definition bowToken.h:46
LinguisticCode getCategory(void) const
Definition bowToken.cpp:285
uint64_t getPosition(void) const override
Definition bowToken.cpp:286
virtual Lima::LimaString getLemma(void) const
Definition bowToken.cpp:283
Holds linguistic data for one language.
const MediaData & mediaData(MediaId media) const
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
A generic process unit to log information in files: contains some common informations such as : outpu...
bool openLogFile(std::ofstream &output, const std::string &sourceFile) const
virtual void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
Parameters retrived in the configuration file:
std::vector< std::pair< boost::shared_ptr< Common::BagOfWords::BoWRelation >, boost::shared_ptr< Common::BagOfWords::BoWToken > > > buildTermFor(const AnnotationGraphVertex &vx, const AnnotationGraphVertex &tgt, const LinguisticGraph &anagraph, const LinguisticGraph &posgraph, const uint64_t offset, const SyntacticAnalysis::SyntacticData *syntacticData, const Common::AnnotationGraphs::AnnotationData *annotationData, std::set< LinguisticGraphVertex > &visited) const
Creates the terms reachable from the given annotation vertex.
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, MediaId language)
virtual LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
virtual void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
This class points to a graph, its dependency graph and the structure that holds the maping between th...
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
This file contains a class to control log of informations about time, such as logging cumulated time ...
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
boost::color_traits< boost::default_color_type > Color
Definition BowDumper.cpp:68
SimpleFactory< MediaProcessUnit, CompoundsXmlLogger > compoundsXmlLoggerFactory(COMPOUNDSXMLLOGGER_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ CANNOT_OPEN_FILE_ERROR
Definition LimaCommon.h:239
@ SUCCESS_ID
Definition LimaCommon.h:237
@ MISSING_DATA
Definition LimaCommon.h:243
QString LimaString
Definition LimaString.h:33