LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
recognizerData.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/************************************************************************
7 *
8 * @file recognizerData.cpp
9 * @author besancon (besanconr@zoe.cea.fr)
10 * @date Tue Jan 25 2005
11 * @version $Id$
12 * copyright Copyright (C) 2005-2012 by CEA LIST
13 *
14 ***********************************************************************/
15
16#include "recognizerData.h"
17#include <queue>
18
20using namespace std;
21
22namespace Lima
23{
24namespace LinguisticProcessing
25{
26namespace ApplyRecognizer
27{
28
29//***********************************************************************
30// constructors
31//***********************************************************************
34 m_verticesToRemove(),
35 m_resultData(0),
36 m_currentSentence(0),
37 m_nextVertices(),
38 m_entityFeatures()
39{}
40
42 AnalysisData(d),
43 m_verticesToRemove(d.m_verticesToRemove),
44 m_resultData(d.m_resultData),
45 m_currentSentence(d.m_currentSentence),
46 m_nextVertices(d.m_nextVertices),
47 m_entityFeatures(d.m_entityFeatures)
48{}
49
50//***********************************************************************
51// destructor
52//***********************************************************************
56
57//***********************************************************************
58// assignment operator
59//***********************************************************************
60RecognizerData& RecognizerData::operator = (const RecognizerData& d)
61{
62 if (this != &d)
63 {
64 AnalysisData::operator=(d);
65 m_verticesToRemove=d.m_verticesToRemove;
66 m_resultData=d.m_resultData;
67 m_currentSentence=d.m_currentSentence;
68 m_nextVertices = d.m_nextVertices;
69 m_entityFeatures = d.m_entityFeatures;
70 }
71 return *this;
72}
73
74//***********************************************************************
75// member functions
76//***********************************************************************
79 m_currentSentence++;
80 if (m_resultData!=0) {
81 m_resultData->push_back(std::vector< Automaton::RecognizerMatch >());
82 }
83}
84
86matchOnRemovedVertices(const RecognizerMatch& result) const
87{
88 for (RecognizerMatch::const_iterator m(result.begin());
89 m!=result.end(); m++)
90 {
91 if ((*m).isKept())
92 {
93 if (m_verticesToRemove.find((*m).getVertex()) !=
94 m_verticesToRemove.end())
95 {
96 return true;
97 }
98 }
99 }
100 return false;
101}
102
104 LinguisticGraph* graph)
105{
106 // APPRLOGINIT;
107 // LDEBUG << "RecognizerData: storing vertices to remove";
108
109 for (RecognizerMatch::const_iterator match(result.begin());
110 match!=result.end(); match++)
111 {
112 if ((*match).isKept())
113 {
114 // LDEBUG << " storing "<< (*match).getVertex() << " to be removed";
115 // store this vertex to remove
116 m_verticesToRemove.insert((*match).getVertex());
117
118 // check previous vertices
119 std::queue<LinguisticGraphVertex> verticesToCheck;
120 verticesToCheck.push((*match).getVertex());
121 while (! verticesToCheck.empty())
122 {
123 // check previous vertices to see if they have only this vertex
124 // as target : if it is the case, remove them also
125 LinguisticGraphInEdgeIt it_begin,it_end;
126 boost::tie(it_begin,it_end)=in_edges(verticesToCheck.front(),*graph);
127 for (LinguisticGraphInEdgeIt it(it_begin); it!=it_end; it++)
128 {
129 LinguisticGraphVertex previousVertex=source(*it,*graph);
130 // LDEBUG << " checking if "<< previousVertex << " should be removed also";
131 bool vertexToRemove(false);
132 if (out_degree(previousVertex,*graph)==1)
133 {
134 vertexToRemove=true;
135 }
136 else
137 {
138 // test if all following vertices are already to be removed
139 LinguisticGraphOutEdgeIt it_out_begin,it_out_end;
140 boost::tie(it_out_begin,it_out_end)=out_edges(previousVertex,*graph);
141 vertexToRemove=true;
142 for (LinguisticGraphOutEdgeIt it_out(it_out_begin); it_out!=it_out_end; it_out++)
143 {
144 if (m_verticesToRemove.find(target(*it_out,*graph))==
145 m_verticesToRemove.end())
146 {
147 vertexToRemove=false;
148 break;
149 }
150 }
151 }
152 if (vertexToRemove &&
153 m_verticesToRemove.find(previousVertex)==
154 m_verticesToRemove.end())
155 {
156 // LDEBUG << " yes";
157 m_verticesToRemove.insert(previousVertex);
158 verticesToCheck.push(previousVertex);
159 }
160 // else
161 // {
162 // LDEBUG << " no";
163 // }
164 }
165 verticesToCheck.pop();
166 }
167
168 // check next vertices
169 verticesToCheck.push((*match).getVertex());
170 while (! verticesToCheck.empty())
171 {
172 // check next vertices to see if they have only this vertex
173 // as source : if it is the case, remove them also
174 LinguisticGraphOutEdgeIt it_begin,it_end;
175 boost::tie(it_begin,it_end)=out_edges(verticesToCheck.front(),*graph);
176 for (LinguisticGraphOutEdgeIt it(it_begin); it!=it_end; it++)
177 {
178 LinguisticGraphVertex nextVertex=target(*it,*graph);
179 bool vertexToRemove(false);
180 if (in_degree(nextVertex,*graph)==1)
181 {
182 vertexToRemove=true;
183 }
184 else
185 {// test if all preceding vertices are already to be removed
186 vertexToRemove=true;
187 LinguisticGraphInEdgeIt it_in_begin,it_in_end;
188 boost::tie(it_in_begin,it_in_end)=in_edges(nextVertex,*graph);
189 for (LinguisticGraphInEdgeIt it_in(it_in_begin); it_in!=it_in_end; it_in++)
190 {
191 if (m_verticesToRemove.find(source(*it_in,*graph))== m_verticesToRemove.end())
192 {
193 vertexToRemove=false;
194 break;
195 }
196 }
197 }
198 if (vertexToRemove &&
199 m_verticesToRemove.find(nextVertex)==
200 m_verticesToRemove.end())
201 {
202 m_verticesToRemove.insert(nextVertex);
203 verticesToCheck.push(nextVertex);
204 }
205 }
206 verticesToCheck.pop();
207 }
208 }
209 }
210}
211
213{
215 static_cast<LinguisticAnalysisStructure::AnalysisGraph*>(analysis.getData(m_resultData->getGraphId()).get());
216
217 // remove vertices and edges in reverse order, so that
218 // it does not affect the reordering of vertex numbers in
219 // the graph
220#ifdef DEBUG_LP
222 LDEBUG << "RecognizerData: removing vertices";
223#endif
224 LinguisticGraph& g=*(anagraph->getGraph());
225 for (set<LinguisticGraphVertex>::const_reverse_iterator
226 it=m_verticesToRemove.rbegin();
227 it!=m_verticesToRemove.rend(); it++)
228 {
229#ifdef DEBUG_LP
230 LDEBUG << " clearing vertex " << *it;
231#endif
232 clear_vertex(*it,g);
233 // remove FullToken;
234 //Data::FullToken* token=get(vertex_ling,g,*it);
235 // LDEBUG << "Idiomatic alternatives: removing vertex " << *it
236 // << "(" << *token << ")";
237 //delete token;
238 }
239
240}
241
243{
244 if (m_resultData==0) {
246 LERROR << "RecognizerData: cannot add result: missing data";
247 return;
248 }
249 m_resultData->insert(result,m_currentSentence);
250}
251
253{
254#ifdef DEBUG_LP
256 LDEBUG << "RecognizerData: removing edges to remove";
257#endif
259 static_cast<LinguisticAnalysisStructure::AnalysisGraph*>(analysis.getData(m_resultData->getGraphId()).get());
260 LinguisticGraph& g=*(anagraph->getGraph());
261 std::set< std::pair<LinguisticGraphVertex, LinguisticGraphVertex> >::const_iterator it, it_end;
262 it = m_edgesToRemove.begin(); it_end = m_edgesToRemove.end();
263 for (; it != it_end; it++)
264 {
265#ifdef DEBUG_LP
266 LDEBUG << "RecognizerData::removeEdges removing edge " << (*it).first << " - " << (*it).second;
267#endif
268 boost::remove_edge((*it).first,(*it).second, g);
269 clearUnreachableVertices(analysis, (*it).first);
270 clearUnreachableVertices(analysis, (*it).second);
271 }
272 m_edgesToRemove.clear();
273}
274
276{
277 // APPRLOGINIT;
278 // LDEBUG << "RecognizerData: setting edge "<<e<<" to be removed";
280 static_cast<LinguisticAnalysisStructure::AnalysisGraph*>(analysis.getData(m_resultData->getGraphId()).get());
281 LinguisticGraph& g=*(anagraph->getGraph());
282
283 std::pair<LinguisticGraphVertex, LinguisticGraphVertex> p = std::make_pair(source(e,g),target(e,g));
284 m_edgesToRemove.insert(p);
285}
286
288{
289 return (m_edgesToRemove.find(std::make_pair(s,t)) != m_edgesToRemove.end());
290}
291
292
294 AnalysisContent& analysis,
297 std::set< std::pair<LinguisticGraphVertex, LinguisticGraphVertex > >& storedEdges)
298{
299#ifdef DEBUG_LP
301 LDEBUG << "RecognizerData: clearing unreachable vertices from " << from << " and to " << to;
302#endif
303 std::deque< std::deque< LinguisticGraphVertex > > paths;
304 std::deque< LinguisticGraphVertex > current;
305 std::set< std::pair<LinguisticGraphVertex, LinguisticGraphVertex > > validated;
306
308 static_cast<LinguisticAnalysisStructure::AnalysisGraph*>(analysis.getData(m_resultData->getGraphId()).get());
309 LinguisticGraph& g=*(anagraph->getGraph());
310
311 current.push_back(from);
312 paths.push_back(current);
313 while (!paths.empty())
314 {
315 current = paths.front();
316 paths.pop_front();
317 if (current.empty()) continue;
318 LinguisticGraphOutEdgeIt it_out,it_out_end;
319 boost::tie(it_out,it_out_end)=out_edges(current.back(),g);
320 if (it_out == it_out_end)
321 { // current back vertex will not be reachable, remove edges from current
322 // path iff not in storedEdges and not last vertex
323 LinguisticGraphVertex tgt = current.back();
324 if (tgt == 1) continue;
325 current.pop_back();
326 while (!current.empty())
327 {
328 LinguisticGraphVertex src = current.back();
329 current.pop_back();
330 std::pair< LinguisticGraphVertex, LinguisticGraphVertex > p = std::make_pair(src,tgt);
331 if (storedEdges.find(p) == storedEdges.end())
332 {
333#ifdef DEBUG_LP
334 LDEBUG << "RecognizerData::clearUnreachableVertices removing edge " << src << " -> " << tgt;
335#endif
336 remove_edge(edge(src,tgt,g).first,g);
337 }
338 tgt = src;
339 }
340 }
341 else
342 {
343 for (; it_out != it_out_end; it_out++)
344 {
345 std::pair< LinguisticGraphVertex, LinguisticGraphVertex > p = std::make_pair(source(*it_out,g),source(*it_out,g));
346 if ( (target(*it_out,g) == to)
347 || (validated.find(p) != validated.end()) )
348 {
349 validated.insert(p);
350 LinguisticGraphVertex tgt = current.back();
351 current.pop_back();
352 while (!current.empty())
353 {
354 LinguisticGraphVertex src = current.back();
355 current.pop_back();
356 validated.insert(std::make_pair(src,tgt));
357 tgt = src;
358 }
359 }
360 else
361 {
362 std::deque< LinguisticGraphVertex > newpath = current;
363 newpath.push_back(target(*it_out,g));
364 paths.push_front(newpath);
365 }
366 }
367 }
368 }
369}
370
371
373 AnalysisContent& analysis,
375{
376#ifdef DEBUG_LP
378 LDEBUG << "RecognizerData: clearing unreachable vertices from " << from;
379#endif
380
382 static_cast<LinguisticAnalysisStructure::AnalysisGraph*>(analysis.getData(m_resultData->getGraphId()).get());
383 LinguisticGraph& g=*(anagraph->getGraph());
384
385 std::queue<LinguisticGraphVertex> verticesToCheck;
386 verticesToCheck.push( from );
387 while (! verticesToCheck.empty() )
388 {
389#ifdef DEBUG_LP
390 LDEBUG << " vertices to check size = " << verticesToCheck.size();
391#endif
392 LinguisticGraphVertex v = verticesToCheck.front();
393 verticesToCheck.pop();
394 bool toClear = false;
395#ifdef DEBUG_LP
396 LDEBUG << " out degree of " << v << " is " << out_degree(v, g);
397#endif
398 if (out_degree(v, g) == 0 && v != anagraph->lastVertex())
399 {
400 toClear = true;
401 LinguisticGraphInEdgeIt it,it_end;
402 boost::tie(it,it_end)=in_edges(v,g);
403 for (; it!=it_end; it++)
404 {
405 verticesToCheck.push(source(*it,g));
406 }
407 }
408#ifdef DEBUG_LP
409 LDEBUG << " in degree of " << v << " is " << in_degree(v, g);
410#endif
411 if (in_degree(v, g) == 0 && v != anagraph->firstVertex())
412 {
413 toClear = true;
414 LinguisticGraphOutEdgeIt it,it_end;
415 boost::tie(it,it_end)=out_edges(v,g);
416 for (; it!=it_end; it++)
417 {
418 verticesToCheck.push(target(*it,g));
419 }
420 }
421 if (toClear)
422 {
423#ifdef DEBUG_LP
424 LDEBUG << " clearing vertex " << v;
425#endif
426 clear_vertex(v,g);
427 }
428 }
429}
430
431
433{
434 m_resultData=data;
435}
436
438{
439 if (m_resultData != 0) {
440 delete m_resultData;
441 }
442}
443
444//**********************************************************************
445// use also this AnalysisData to store Entity Features
446
448{
449 m_entityFeatures.clear();
450}
451
452//**********************************************************************
453// Data to store the results
454
456RecognizerResultData(const std::string& sourceGraph):
457 AnalysisData(),
458 std::vector<std::vector< Automaton::RecognizerMatch > >(),
459 m_graphId(sourceGraph)
460{
461 // empty result contains one empty vector
462 // (for first sentence or whole text)
463 push_back(std::vector< Automaton::RecognizerMatch >());
464}
465
468 AnalysisData(d),
469 std::vector<std::vector< Automaton::RecognizerMatch > >(d),
470 m_graphId(d.m_graphId)
471{}
472
475
478{
479 if (this != &d)
480 {
481 AnalysisData::operator=(d);
482 std::vector<std::vector< Automaton::RecognizerMatch > >::operator=(d);
483 m_graphId=d.m_graphId;
484 }
485 return *this;
486}
487
489insert(const RecognizerMatch& m,
490 const uint64_t sentenceId)
491{
492 if (sentenceId>= size()) {
494 LERROR << "RecognizerResultData: try to access data oustide of vector (sentenceId=" << sentenceId << ",size=" << size() << ")";
495 return;
496 }
497 (*this)[sentenceId].push_back(m);
498}
499
500
501} // end namespace
502} // end namespace
503} // end namespace
#define LDEBUG
Definition LimaCommon.h:157
#define LERROR
Definition LimaCommon.h:161
LinguisticGraph::in_edge_iterator LinguisticGraphInEdgeIt
boost::graph_traits< LinguisticGraph >::edge_descriptor LinguisticGraphEdge
typedefs to simplify the access to various graphs elements
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
just for semantics: base class for analysisData
void addResult(const Automaton::RecognizerMatch &result)
void storeVerticesToRemove(const Automaton::RecognizerMatch &result, LinguisticGraph *graph)
void clearUnreachableVertices(AnalysisContent &analysis, LinguisticGraphVertex from, LinguisticGraphVertex to, std::set< std::pair< LinguisticGraphVertex, LinguisticGraphVertex > > &storedEdges)
remove edges linked to vertices that have no path between from and to excepted those in storedEdges
bool isEdgeToBeRemoved(LinguisticGraphVertex s, LinguisticGraphVertex t) const
void setEdgeToBeRemoved(AnalysisContent &analysis, LinguisticGraphEdge e)
bool matchOnRemovedVertices(const Automaton::RecognizerMatch &result) const
RecognizerResultData & operator=(const RecognizerResultData &)
void insert(const Automaton::RecognizerMatch &m, const uint64_t sentenceId=0)
␈rief A class for the description of automata
Definition automaton.h:87
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
SimpleFactory< MediaProcessUnit, ApplyRecognizer > ApplyRecognizer(APPLYRECOGNIZER_CLASSID)
NAUTITIA.
STL namespace.
#define APPRLOGINIT