LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
AnalysisGraph.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/***************************************************************************
7 * Copyright (C) 2004-2012 by CEA LIST *
8 * *
9 ***************************************************************************/
10#include "AnalysisGraph.h"
11
17
18#include <queue>
19
20using namespace Lima::Common::MediaticData;
21using namespace Lima::Common::AnnotationGraphs;
22
23namespace Lima {
24
25namespace LinguisticProcessing {
26
27namespace LinguisticAnalysisStructure {
28
29//***********************************************************************
30// constructors
31//***********************************************************************
32AnalysisGraph::AnalysisGraph(const std::string& graphId,MediaId language,bool deleteTokenWhenDestroyed,bool deleteDataWhenDestroyed):
33 m_graph(0),
34 m_firstVertex(0),
35 m_lastVertex(0),
36 m_deleteTokenWhenDestroyed(deleteTokenWhenDestroyed),
37 m_deleteDataWhenDestroyed(deleteDataWhenDestroyed),
38 m_language(language),
39 m_graphId(graphId)
40{
41 createGraph();
42}
43
44AnalysisGraph::AnalysisGraph(const std::string& graphId,MediaId language,bool deleteTokenWhenDestroyed,bool deleteDataWhenDestroyed,const AnalysisGraph& anagraph) :
45 m_graph(new LinguisticGraph(*(anagraph.m_graph))),
46 m_firstVertex(anagraph.m_firstVertex),
47 m_lastVertex(anagraph.m_lastVertex),
48 m_deleteTokenWhenDestroyed(deleteTokenWhenDestroyed),
49 m_deleteDataWhenDestroyed(deleteDataWhenDestroyed),
50 m_language(language),
51 m_graphId(graphId)
52{}
53
54
57 m_graph(g.m_graph),
58 m_firstVertex(g.m_firstVertex),
59 m_lastVertex(g.m_lastVertex),
60 m_deleteTokenWhenDestroyed(g.m_deleteTokenWhenDestroyed),
61 m_deleteDataWhenDestroyed(g.m_deleteDataWhenDestroyed),
62 m_language(g.m_language)
63{}
64
65//***********************************************************************
66// destructor
67//***********************************************************************
69{
70 deleteGraph();
71}
72
73//***********************************************************************
74// assignment operator
75//***********************************************************************
76AnalysisGraph& AnalysisGraph::operator = (const AnalysisGraph& g)
77{
78 if (this != &g)
79 {
80 m_graph=g.m_graph;
81 m_firstVertex=g.m_firstVertex;
82 m_lastVertex=g.m_lastVertex;
83 m_language=g.m_language;
84 }
85 return *this;
86}
87
88//***********************************************************************
89// member functions
90//***********************************************************************
91void AnalysisGraph::createGraph()
92{
93
94 if (Common::MediaticData::MediaticData::single().releaseStringsPool())
96
97 if (m_graph == 0)
98 {
99 m_graph = new LinguisticGraph();
100 }
101
102 if (num_vertices(*m_graph) != 0)
103 {
104 throw LinguisticProcessingException("AnalysisGraph::createGraph() - graph is not empy");
105 }
106
107 VertexTokenPropertyMap tokenMap = get( vertex_token, *m_graph );
108 VertexDataPropertyMap dataMap = get( vertex_data, *m_graph );
109
110 // add first vertex
111 LinguisticGraphVertex vertex1 = add_vertex(*m_graph);
112 tokenMap[vertex1] = 0;
113 dataMap[vertex1] = 0;
114 m_firstVertex = vertex1;
115
116 // add last vertex
117 LinguisticGraphVertex vertex2 = add_vertex(*m_graph);
118 tokenMap[vertex2] = 0;
119 dataMap[vertex2] = 0;
120 m_lastVertex = vertex2;
121
122 // add edge between first and last vertex
123 bool b;
124 LinguisticGraphEdge beginEndEdge;
125 boost::tie(beginEndEdge, b) = add_edge(vertex1, vertex2, *m_graph);
126 if (!b)
127 {
128 throw LinguisticProcessingException("AnalysisGraph::createGraph - could not bind beginEndEdge");
129 }
130}
131
132void AnalysisGraph::deleteGraph()
133{
134#ifdef DEBUG_LP
136 LDEBUG << "deleteGraph";
137#endif
138 if (m_graph == 0) return;
139
140 if (m_deleteTokenWhenDestroyed)
141 {
142 LinguisticGraphVertexIt it, it_end;
143 VertexTokenPropertyMap tokenMap = get( vertex_token, *m_graph );
144
145 std::set<Token*> deletedToken;
146
147 boost::tie(it, it_end) = vertices(*m_graph);
148 for (; it != it_end; it++)
149 {
150 Token* ft = tokenMap[*it];
151 if ( ft != 0 && (deletedToken.find(ft) == deletedToken.end()) &&
152 ( (*it) != firstVertex() ) && ( (*it) != lastVertex() ) )
153 {
154 delete ft;
155 deletedToken.insert(ft);
156 }
157 tokenMap[*it] = 0;
158 }
159 }
160 if (m_deleteDataWhenDestroyed)
161 {
162 LinguisticGraphVertexIt it, it_end;
163 VertexDataPropertyMap dataMap = get( vertex_data, *m_graph );
164
165 std::set<MorphoSyntacticData*> deletedData;
166
167 boost::tie(it, it_end) = vertices(*m_graph);
168 for (; it != it_end; it++)
169 {
170 MorphoSyntacticData* data = dataMap[*it];
171 if ( data != 0 && (deletedData.find(data) == deletedData.end()) &&
172 ( (*it) != firstVertex() ) && ( (*it) != lastVertex() ) )
173 {
174 delete data;
175 deletedData.insert(data);
176 }
177 dataMap[*it] = 0;
178 }
179 }
180
181 delete m_graph;
182 m_graph = 0;
183
184 if (Common::MediaticData::MediaticData::single().releaseStringsPool())
186
187}
188
191 const Common::PropertyCode::PropertyAccessor& microAccessor,
192 const std::list<LinguisticCode> microFilters,
194{
195#ifdef DEBUG_LP
197#endif
198 /*
199 * Algorithm: we're using a Breadth First Search and keep track of the
200 * "thickness" of the lattice, and only stop if both condition apply:
201 * 1/ the thickness is 1, meaning that every path goes through this node
202 * 2/ the node is in microFilters (eg. a full stop in english)
203 * OR the node has t_sentence_break tokenization status
204 */
205 std::set<LinguisticGraphVertex> visited;
206 LinguisticGraphOutEdgeIt outItr,outItrEnd;
207 LinguisticGraphInEdgeIt inItr,inItrEnd;
208
209 std::queue<LinguisticGraphVertex,std::list<LinguisticGraphVertex> > toVisit;
210
211 // initialize
212 size_t accumulator=out_degree(start,*m_graph);
213 boost::tie (outItr,outItrEnd) = out_edges(start,*m_graph);
214 for (;outItr!=outItrEnd;outItr++)
215 {
216 toVisit.push(target(*outItr,*m_graph));
217 }
218
219 VertexTokenPropertyMap tokenMap = get( vertex_token, *m_graph );
220 // search
221 while (!toVisit.empty())
222 {
223 LinguisticGraphVertex current=toVisit.front();
224 toVisit.pop();
225 visited.insert(current);
226 if (current==end)
227 {
228 return end;
229 }
230 Token* ft = tokenMap[current];
231
232 accumulator-=in_degree(current,*m_graph);
233 if (accumulator==0)
234 {
235 // check unique category only if accumulator is 0
236 MorphoSyntacticData* msd=get(vertex_data,*m_graph,current);
237 if (msd!=0 && msd->hasUniqueMicro(microAccessor,microFilters))
238 {
239#ifdef DEBUG_LP
240 LDEBUG << "AnalysisGraph::nextMainPathVertex micro, return" << current;
241#endif
242 return current;
243 }
244 if (ft && ft->status().getStatus() == T_SENTENCE_BRK)
245 {
246#ifdef DEBUG_LP
247 LDEBUG << "AnalysisGraph::nextMainPathVertex sentence break, return" << current;
248#endif
249 return current;
250 }
251 }
252 accumulator+=out_degree(current,*m_graph);
253
254 boost::tie (outItr,outItrEnd) = out_edges(current,*m_graph);
255 if (outItr==outItrEnd)
256 {
258 LERROR << "no next vertex in graph whereas current vertex is not last vertex !!";
259 throw std::runtime_error("no next vertex in graph whereas current vertex is not last vertex !!");
260 }
261 for (;outItr!=outItrEnd;outItr++)
262 {
263 // Must Check if all predecessors have already been visited
264 LinguisticGraphVertex next=target(*outItr,*m_graph);
265 boost::tie(inItr,inItrEnd) = in_edges(next,*m_graph);
266 bool visitable=true;
267 for (;inItr!=inItrEnd;inItr++)
268 {
269 if (visited.find(source(*inItr,*m_graph))==visited.end())
270 {
271 visitable=false;
272 break;
273 }
274 }
275 if (visitable)
276 {
277 toVisit.push(next);
278 }
279 }
280 }
281
282 // reach this end if file doesn't end with a specified categ.
283 return end;
284}
285
286
302 const LinguisticGraphVertex& v,
303 const Common::PropertyCode::PropertyAccessor& macroAccessor,
304 const LinguisticCode& ponctu,
305 const Common::PropertyCode::PropertyAccessor& microAccessor,
306 LinguisticGraphVertex& nextSentenceBreak)
307{
308#ifdef DEBUG_LP
310#endif
311
312 LinguisticGraphVertex current = v;
313 size_t accumulator=out_degree(current,*m_graph);
314 while (true)
315 {
316 LinguisticGraphOutEdgeIt it, it_end;
317 boost::tie (it, it_end) = out_edges(current, *m_graph);
318
319 if (it == it_end)
320 return m_lastVertex;
321 if ( (source((*it), *m_graph) == m_firstVertex) && (target((*it), *m_graph) == m_lastVertex) )
322 {
323 it++;
324 if (it == it_end)
325 return m_lastVertex;
326 }
327
328 LinguisticGraphVertex next = target((*it), *m_graph);
329 if (next == m_lastVertex || next==nextSentenceBreak)
330 {
331 if (next != nextSentenceBreak)
332 {
334 LERROR << "In nextChainsBreakFrom: went beyond next sentence break " << nextSentenceBreak;
335 LERROR << " returning graph's last vertex " << next;
336 }
337#ifdef DEBUG_LP
338 LDEBUG << "Next chains break is: " << next;
339#endif
340 return next;
341 }
342 accumulator-=in_degree(next,*m_graph);
343 VertexDataPropertyMap dataMap = get( vertex_data, *m_graph );
344 MorphoSyntacticData* msd =dataMap[next];
345 if ( (accumulator == 0) && (msd->countValues(microAccessor) == 1) )
346 {
347 LinguisticCode macro = NONE_1;
348
349 /*if (!tok->morphoSyntacticData().isEmpty())
350 {
351 std::pair< std::list< WordForm >::const_iterator, std::list< WordForm >::const_iterator > basePair = tok->morphoSyntacticData().base();
352 std::pair< std::list< WordForm >::const_iterator, std::list< WordForm >::const_iterator > hyphensPair = tok->morphoSyntacticData().hyphens();
353 std::pair< WordFormProperties::const_iterator, WordFormProperties::const_iterator > defaultsPair = tok->morphoSyntacticData().properties();
354 if (basePair.first != basePair.second)
355 {
356 macro = macroAccessor.readValue(*((*(basePair.first)).properties().first));
357 }
358 else if (hyphensPair.first !=hyphensPair.second)
359 {
360 macro = macroAccessor.readValue(*((*(hyphensPair.first)).properties().first));
361 }
362 else if (defaultsPair.first != defaultsPair.second)
363 {
364 macro = macroAccessor.readValue(*(defaultsPair.first));
365 }
366 else
367 macro = NONE_1;
368 }*/
369 /* �valider : le code ci-dessus est remplac�par : */
370 if (msd->begin() != msd->end()) {
371 macro=macroAccessor.readValue(msd->begin()->properties);
372 }
373
374 /* else if (hyphensPair.first != hyphensPair.second)
375 {
376 const Token& alt = (*((tok-> getOrthographicAlternatives()).begin()));
377 macro = ((*(alt.morphoSyntacticData().base().first)).properties().first)->code(MACRO);
378 }
379 else
380 macro = NONE_1;*/
381
382 if ( macro == ponctu)
383 return next;
384 }
385 accumulator+=out_degree(next,*m_graph);
386 if (next == current)
387 {
389 LERROR << "In nextChainsBreakFrom: cannot go beyond " << current;
390 return m_lastVertex;
391 }
392 current = next;
393 }
394}
395
396
398 AnnotationData* annotData,
399 const std::string& src)
400{
401 LinguisticGraphVertexIt it, it_end;
402
403 boost::tie(it, it_end) = vertices(*m_graph);
404 for (; it != it_end; it++)
405 {
406 if (annotData->matches(src, *it, "annot").empty())
407 {
409 annotData->addMatching(src, *it, "annot", agv);
410 annotData->annotate(agv, Common::Misc::utf8stdstring2limastring(src), static_cast< uint64_t >(*it));
411 }
412 }
413
414}
415
416
417}
418
419}
420
421}
This file is the main header file for the data related to annotation graphs.
#define LDEBUG
Definition LimaCommon.h:157
#define LERROR
Definition LimaCommon.h:161
LinguisticGraph::in_edge_iterator LinguisticGraphInEdgeIt
boost::property_map< LinguisticGraph, vertex_data_t >::type VertexDataPropertyMap
boost::graph_traits< LinguisticGraph >::edge_descriptor LinguisticGraphEdge
typedefs to simplify the access to various graphs elements
LinguisticGraph::vertex_iterator LinguisticGraphVertexIt
boost::property_map< LinguisticGraph, vertex_token_t >::type VertexTokenPropertyMap
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define LASLOGINIT
#define NONE_1
Definition StdBitset.h:339
just for semantics: base class for analysisData
Holds an annotation graph and gives an API to manipulate it.
void addMatching(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second, AnnotationGraphVertex secondVx)
Adds a symetric matching between two vertices of two graphs identified by the two string parameters.
std::set< AnnotationGraphVertex > matches(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second) const
Gets the set of vertices matched in the second graph by the given vertex of the first graph.
AnnotationGraphVertex createAnnotationVertex()
Creates a new annotation vertex in the graph.
const FsaStringsPool & stringsPool(MediaId med) const
Provide function to read write and check a property.
LinguisticCode readValue(const LinguisticCode &code) const
read a property in a coded int.
void unregisterUser(void *p)
An AnalysisData containing a LinguisticGraph with a language and an id.
AnalysisGraph(const std::string &graphId, MediaId language, bool deleteTokenWhenDestroyed, bool deleteDataWhenDestroyed)
LinguisticGraphVertex nextChainsBreakFrom(const LinguisticGraphVertex &v, const Common::PropertyCode::PropertyAccessor &macroAccessor, const LinguisticCode &ponctu, const Common::PropertyCode::PropertyAccessor &microAccessor, LinguisticGraphVertex &nextSentenceBreak)
Finds the next vertex after the input vertex that:
LinguisticGraphVertex nextMainPathVertex(LinguisticGraphVertex start, const Common::PropertyCode::PropertyAccessor &microAccessor, const std::list< LinguisticCode > microFilters, LinguisticGraphVertex end)
Finds the next unambiguated vertex for which micro categories are all included in the microFilters li...
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
void populateAnnotationGraph(Common::AnnotationGraphs::AnnotationData *annotData, const std::string &src)
Creates the annotations in the agdata corresponding to this graphs vertices.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
uint64_t countValues(const Common::PropertyCode::PropertyAccessor &propertyAccessor)
bool hasUniqueMicro(const Lima::Common::PropertyCode::PropertyAccessor &microAccessor, const std::list< Lima::LinguisticCode > &microFilter)
return true if there is only one micro, and this micro is in microfilter
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
static MediaticData & changeable()
singleton accessor
Definition Singleton.h:71
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
void annotate(AnnotationGraphVertex v, const LimaString &annot, uint64_t value)
LimaString utf8stdstring2limastring(const std::string &src)
NAUTITIA.