LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
recognizerMatch.cpp
Go to the documentation of this file.
1// Copyright 2002-2020 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/************************************************************************
7 *
8 * @file recognizerMatch.cpp
9 * @author besancon (besanconr@zoe.cea.fr)
10 * @date Wed Oct 13 2004
11 * copyright Copyright (C) 2004 by CEA LIST
12 *
13 ***********************************************************************/
14
15#include "recognizerMatch.h"
16
17using namespace std;
19
20namespace Lima {
21namespace LinguisticProcessing {
22namespace Automaton {
23
24//***********************************************************************
25// constructors
26//***********************************************************************
33
35 const LinguisticGraphVertex& vertex,
36 const bool isKept):
37std::vector<MatchElement>(),
39m_graph(graph)
40{
41 addBackVertex(vertex,isKept);
42}
43
46
47// comparison operator
49 if (m_graph != m.m_graph) {
50 return false;
51 }
52 if (size() != m.size()) {
53 return false;
54 }
55 std::vector<MatchElement>::const_iterator
56 it1=begin(),
57 it1_end=end(),
58 it2=m.begin();
59 for (; it1!=it1_end; it1++,it2++) {
60 if ((*it1).m_elem!=(*it2).m_elem) {
61 return false;
62 }
63 }
64 if (! EntityProperties::operator==(m)) {
65 return false;
66 }
67 return true;
68}
69
70//***********************************************************************
71// member functions
72//***********************************************************************
74 // do not reinit graph
75 std::vector<MatchElement>::clear();
76 // clear entity properties
77 m_head=0;
80}
81
82// position of first element of the match
84 if (empty()) {
85 return 0;
86 }
87 return get(vertex_token,*(m_graph->getGraph()),
88 front().getVertex())->position();
89}
90
91// position after the last element of the match
93 if (empty()) {
94 return 0;
95 }
96 Token *t=get(vertex_token,*(m_graph->getGraph()),
97 back().getVertex());
98 return t->position()+t->length();
99}
100
101uint64_t RecognizerMatch::length() const {
102 return (positionEnd() - positionBegin());
103}
104
106 uint64_t n(0);
107// AULOGINIT;
108// LDEBUG << "RecognizerMatch:numberOfElements";
109// LDEBUG << "RecognizerMatch:this=" << *this;
110 for (RecognizerMatch::const_iterator it=begin(),it_end=end();
111 it!=it_end; it++) {
112 if ((*it).isKept()) {
113 n++;
114 }
115 }
116 return n;
117}
118
120 for (RecognizerMatch::const_iterator it=begin(),it_end=end();
121 it!=it_end; it++) {
122 if (! (*it).isKept()) {
123 return false;
124 }
125 }
126 return true;
127}
128
130 std::set<LinguisticGraphVertex> vertices;
131 for (RecognizerMatch::const_iterator it=begin(),it_end=end();
132 it!=it_end; it++) {
133 LinguisticGraphVertex v = (*it).m_elem.first;
134 if (vertices.find(v) != vertices.end())
135 return true;
136 vertices.insert(v);
137 }
138 return false;
139}
140
143 uint64_t currentPosition(0);
144 if (empty()) {
145 return str;
146 }
147 RecognizerMatch::const_iterator i(begin());
148 const LinguisticGraphVertex& v=(*i).getVertex();
149 bool firstHyphenPassed = false;
150 bool prevTokenIsSynthetic = false;
151 if (v != m_graph->firstVertex() &&
152 v != m_graph->lastVertex()) {
153 if ((*i).isKept()) {
154 Token *t = get(vertex_token,*(m_graph->getGraph()),v);
155 if (t->status().isAlphaHyphen()) {
156 firstHyphenPassed = true;
157 }
158 if (t!=0) {
159 str += t->stringForm();
160 }
161
162 if (t->length() != static_cast<uint64_t>(t->stringForm().size()))
163 prevTokenIsSynthetic = true;
164
165 currentPosition=t->position()+t->length();
166 }
167 }
168 i++;
169 while (i!=end()) {
170 const LinguisticGraphVertex& v=(*i).getVertex();
171 if (v != m_graph->firstVertex() &&
172 v != m_graph->lastVertex()) {
173 if ((*i).isKept()) {
174 Token *t = get(vertex_token,*(m_graph->getGraph()),v);
175 // hack to deal with missing information of what is bewteen
176 // the tokens : rely on positions
177 if (t->position() > currentPosition) {
178 if (t->status().isAlphaHyphen()) {
179 if (firstHyphenPassed) {
180 str += LimaChar('-');
181 }
182 else {
183 str += LimaChar(' ');
184 firstHyphenPassed = true;
185 }
186 }
187 else {
188 str += LimaChar(' ');
189 if (firstHyphenPassed) {
190 firstHyphenPassed = false;
191 }
192 }
193 } else if (prevTokenIsSynthetic)
194 str += LimaChar(' ');
195
196 str += t->stringForm();
197
198 if (t->length() != static_cast<uint64_t>(t->stringForm().size()))
199 prevTokenIsSynthetic = true;
200
201 currentPosition=t->position()+t->length();
202 }
203 }
204 i++;
205 }
206 return str;
207}
208
211 uint64_t currentPosition(0);
212 if (empty()) {
213 return str;
214 }
215 bool firstHyphenPassed = false;
216 RecognizerMatch::const_iterator i(begin());
217 const LinguisticGraphVertex& v=(*i).getVertex();
218 if (v != m_graph->firstVertex() &&
219 v != m_graph->lastVertex()) {
220 if ((*i).isKept()) {
221 Token* t = get(vertex_token,*(m_graph->getGraph()),v);
222
223 if (t->status().isAlphaHyphen()) {
224 firstHyphenPassed = true;
225 }
226 MorphoSyntacticData* data = get(vertex_data,*(m_graph->getGraph()),v);
227
228 if (data==0 || data->empty()) {
229 str += t->stringForm();
230 }
231 else {
232 // take first norm
233 str += sp[data->front().normalizedForm];
234 }
235 currentPosition=t->position()+t->length();
236 }
237 }
238 i++;
239 while (i!=end()) {
240 const LinguisticGraphVertex& v=(*i).getVertex();
241 if (v != m_graph->firstVertex() &&
242 v != m_graph->lastVertex()) {
243 if ((*i).isKept()) {
244 Token *t = get(vertex_token,*(m_graph->getGraph()),v);
245 MorphoSyntacticData* data = get(vertex_data,*(m_graph->getGraph()),v);
246
247 // hack to deal with missing information of what is bewteen
248 // the tokens : rely on positions
249 if (t->position() > currentPosition) {
250 if (t->status().isAlphaHyphen()) {
251 if (firstHyphenPassed) {
252 str += LimaChar('-');
253 }
254 else {
255 str += LimaChar(' ');
256 firstHyphenPassed = true;
257 }
258 }
259 else {
260 str += LimaChar(' ');
261 if (firstHyphenPassed) {
262 firstHyphenPassed = false;
263 }
264 }
265 }
266
267 if (data == 0 || data->empty()) {
268 str += t->stringForm();
269 }
270 else {
271 // take first norm
272 str += sp[data->front().normalizedForm];
273 }
274
275 currentPosition=t->position()+t->length();
276 }
277 }
278 i++;
279 }
280 return str;
281}
282
284isOverlapping(const RecognizerMatch& otherMatch) const {
285 if (positionBegin() <= otherMatch.positionBegin()) {
286 if (positionEnd() <= otherMatch.positionBegin()) {
287 return false;
288 }
289 }
290 else {
291 if (positionBegin() >= otherMatch.positionEnd()) {
292 return false;
293 }
294 }
295 return true;
296}
297//**********************************************************************
298// construction functions
299//**********************************************************************
301 bool isKept, const LimaString& ruleElementId ) {
302// #ifdef DEBUG_LP
303// AULOGINIT;
304// LDEBUG << "RecognizerMatch:addBackVertex(v:" << v << ", isKept:" << isKept << ", ruleElmtId:" << ruleElementId << ")";
305// LDEBUG << " in match " << *this;
306// #endif
307 push_back(MatchElement(v,isKept, ruleElementId));
308}
309
311 if (empty()) {
312 return;
313 }
314// #ifdef DEBUG_LP
315// AULOGINIT;
316// LDEBUG << "RecognizerMatch:popBackVertex() for match " << *this;
317// #endif
318 pop_back();
319}
320
322 bool isKept, const LimaString& ruleElementId) {
323#ifdef DEBUG_LP
324 AULOGINIT;
325 LDEBUG << "RecognizerMatch:addFrontVertex(v:" << v << ", isKept:" << isKept << ", ruleElmtId:" << ruleElementId << ")";
326#endif
327 insert(begin(),MatchElement(v,isKept,ruleElementId));
328}
329
331 if (empty()) {
332 return;
333 }
334 erase(begin());
335}
336
338 if( l.getHead() != 0 ){
339 setHead(l.getHead());
340 }
341 insert(end(),l.begin(),l.end());
342}
343
345 if( l.getHead() != 0 ){
346 setHead(l.getHead());
347 }
348 insert(begin(),l.begin(),l.end());
349}
350
352 // remove unkept at beginning
353 RecognizerMatch::iterator it=begin();
354 while (it != end() && ! (*it).isKept() ) {
355 it=erase(it);
356 }
357 if (it == end()) {
358 return;
359 }
360 // remove unkept at end
361 // cannot erase reverse_iterator => use forward iterators
362 // go to last element
363 RecognizerMatch::iterator next=it;
364 next++;
365 while (next != end()) {
366 next++; it++;
367 }
368 while (! (*it).isKept() ) {
369 next=it;
370 it--;
371 erase(next);
372 }
373}
374
375
376//***********************************************************************
377// output
378//***********************************************************************
379LIMA_AUTOMATON_EXPORT std::ostream& operator << (std::ostream& os, const RecognizerMatch& m) {
380 os << " /[-";
381 for (RecognizerMatch::const_iterator i(m.begin()); i != m.end(); i++) {
382 if ((*i).isKept()) {
383 os << (*i).getRuleElemtId().toUtf8().constData() << "." << (*i).getVertex() << "-";
384 }
385 else {
386 os << "(" << (*i).getRuleElemtId().toUtf8().constData() << "." << (*i).getVertex() << ")" << "-";
387 }
388 }
389 os << "]";
390 os.flush();
391 return os;
392}
393
395 os << "/[-";
396 for (RecognizerMatch::const_iterator i(m.begin()); i != m.end(); i++) {
397 if ((*i).isKept()) {
398 os << (*i).getRuleElemtId().toUtf8().constData() << "." << (*i).getVertex() << "-";
399 }
400 else {
401 os << "(" << (*i).getRuleElemtId().toUtf8().constData() << "." << (*i).getVertex() << ")" << "-";
402 }
403 }
404 os << "]";
405 return os;
406}
407
408} // end namespace
409} // end namespace
410} // end namespace
#define LIMA_AUTOMATON_EXPORT
#define LDEBUG
Definition LimaCommon.h:157
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
#define AULOGINIT
#define L_NONE
Definition StdBitset.h:338
LinguisticGraphVertex m_head
indicates which token (in the original text) is the head of the recognized entity
Common::MediaticData::EntityType m_type
the type of the recognized entity
LinguisticCode m_linguisticProperties
associated ling prop
bool isOverlapping(const RecognizerMatch &otherMatch) const
void addBackVertex(const LinguisticGraphVertex &, bool isKept=true, const LimaString &ruleElementId="")
RecognizerMatch(const LinguisticAnalysisStructure::AnalysisGraph *graph)
LimaString getNormalizedString(const FsaStringsPool &sp) const
void addFrontVertex(const LinguisticGraphVertex &, bool isKept=true, const LimaString &ruleElementId="")
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
std::ostream & operator<<(std::ostream &os, const DFFSPos &x)
NAUTITIA.
QChar LimaChar
Definition LimaString.h:30
QString LimaString
Definition LimaString.h:33
STL namespace.