LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
gazeteerTransition.cpp
Go to the documentation of this file.
1// Copyright 2002-2018 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/*************************************************************************
7*
8* File : gazeteerTransition.cpp
9* Author : Olivier Mesnard (olivier.mesnard@cea.fr)
10* @date Thu August 04 2015
11* copyright Copyright (C) 2002-2018 by CEA LIST
12* Version : $Id$
13*
14*************************************************************************/
15
16
17#include "gazeteerTransition.h"
19#include <iostream>
20#include <boost/tuple/tuple.hpp> // for tie
21#include "searchGraph.h"
22
23using namespace std;
25using namespace Lima::Common::MediaticData;
26
27
28namespace Lima {
29namespace LinguisticProcessing {
30namespace Automaton {
31
32/***********************************************************************/
33// constructors
34/***********************************************************************/
37m_wordSet(),
38m_alias()
39{
40}
41
42GazeteerTransition::GazeteerTransition(const std::vector<LimaString>& wordSet, const LimaString& alias, bool keep):
43TransitionUnit(keep),
44m_wordSet(wordSet.begin(),wordSet.end()),
45m_alias(alias)
46{
47}
48
51m_wordSet(t.m_wordSet),
52m_alias(t.m_alias)
53{
54// TODO ToBeDeleted ?
55 // copyProperties(t);
56}
57
59
61 if (this != &t) {
62 m_alias = t.alias();
64 }
65 return *this;
66}
67
68
69std::string GazeteerTransition::printValue() const {
70 ostringstream oss;
71 oss << "gazetteerT(" << Lima::Common::Misc::limastring2utf8stdstring(m_alias) << ":";
72 std::set<LimaString>::const_iterator it = m_wordSet.begin();
73 if( it != m_wordSet.end() ) {
74 const Lima::LimaString & word = *it;
76 }
77 int printMax=8;
78 for( it++ ; it != m_wordSet.end(); it++ ) {
79 const Lima::LimaString & word = *it;
80 if(--printMax==0) {
81 oss << "...";
82 break;
83 }
85 }
86 oss << ")";
87 return oss.str();
88}
89
90/***********************************************************************/
91// operators ==
92/***********************************************************************/
94 if ( (type() == tright.type())
95 && (m_alias == static_cast<const GazeteerTransition&>(tright).alias())
96 ) {
97 return compareProperties(tright);
98 }
99 else {
100 return false;
101 }
102}
103
106 const LinguisticGraphVertex& /*vertex*/,
107 AnalysisContent& /*analysis*/,
110{
111 //AULOGINIT;
112// LDEBUG << "GazeteerTransition compare " << Common::MediaticData::MediaticData::changeable().stringsPool()[token->form()] << " and " << Common::MediaticData::MediaticData::changeable().stringsPool()[m_word];
113 QString form(token->stringForm());
114 std::set<LimaString>::const_iterator it = m_wordSet.lower_bound(form);
115 if( it == m_wordSet.end() ) {
116 return false;
117 }
118 QString element = *it;
119 // If element is equal to form
120 if( element == form )
121 {
122 return true;
123 }
124 // Or element begin with form followed by a space character
125 if( element.startsWith(form) )
126 {
127 if( element.at(form.length()) == ' ')
128 {
129 return true;
130 }
131 }
132 /*
133 QString pattern(form);
134 pattern.append("\\b");
135 QRegularExpression rx(pattern);
136 int index = qStringList.indexOf(rx);
137 */
138// return true;
139 return false;
140}
141
144 const LinguisticGraphVertex& vertex,
145 const LinguisticGraphVertex& limit,
146 const SearchGraph* searchGraph,
147 AnalysisContent& analysis,
149 deque<LinguisticGraphVertex>& vertices,
151{
152 // TODO: use of limit???
153#ifdef DEBUG_LP
154 AULOGINIT;
155#endif
156 const LimaString firstSimpleTerm = token->stringForm();
157 /* build multi term list in gazeteer with firstSimpleTerm as first term */
158 std::vector<std::vector<LimaString> > additionalMultiTermList;
159 buildNextTermsList( firstSimpleTerm, additionalMultiTermList );
160 /* follow graph if tokens match other terms */
161 std::stack<std::deque<LinguisticGraphVertex>,std::vector<deque<LinguisticGraphVertex> > > triggerMatches;
162 checkMultiTerms(graph, vertex, limit, searchGraph, analysis, additionalMultiTermList, triggerMatches );
163 if( triggerMatches.empty() ) {
164#ifdef DEBUG_LP
165 LDEBUG << "GazeteerTransition::matchPath from" <<vertex<<": no match";
166#endif
167 return false;
168 }
169 else {
170 vertices = triggerMatches.top();
171#ifdef DEBUG_LP
172 ostringstream oss;
173 std::copy(vertices.begin(),vertices.end(),std::ostream_iterator<int>(oss,"-"));
174 LDEBUG << "GazeteerTransition::matchPath from" <<vertex<<": found match with vertices" << oss.str();
175#endif
176 return true;
177 }
178 return false;
179}
180
181 /* Gazeteer may contains multi-term elements like */
182/* "managing director","Managing Director","managing editor","managing comitee secretary"... */
183/* From wordSet, we build a list of multiple terms, each with parameter firstSimpleTerm as first simple term */
184/* [("managing,director");("managing,Director");("managing,editor");("managing,comitee,secretary")] */
185/* return false if there is no elements begining with "managing" */
186bool GazeteerTransition::
187buildNextTermsList( const LimaString& firstSimpleTerm, std::vector<std::vector<LimaString> >& multiTermList ) const
188{
189// #ifdef DEBUG_LP
190// AULOGINIT;
191// LDEBUG << "GazeteerTransition::buildNextTermsList(" << firstSimpleTerm << ")";
192// #endif
193
194 // Fill list of list of additional simple terms from list of elements
195 std::set<LimaString>::const_iterator it = m_wordSet.lower_bound(firstSimpleTerm);
196 if( it == m_wordSet.end() ) {
197// #ifdef DEBUG_LP
198// LDEBUG << "GazeteerTransition::buildNextTermsList: Error: first term not found";
199// #endif
200 return false;
201 }
202 for( ; it != m_wordSet.end() ; it++ )
203 {
204 LimaString element = *it;
205// #ifdef DEBUG_LP
206// LDEBUG << "GazeteerTransition::buildNextTermsList: Examining " << element.toStdString();
207// #endif
208 // if element does not start with firstSimpleTerm, there no more possible match
209 if( !element.startsWith(firstSimpleTerm) ) {
210// #ifdef DEBUG_LP
211// LDEBUG << "GazeteerTransition::buildNextTermsList: stop it!: first term not found";
212// #endif
213 break;
214 }
215 std::vector<LimaString> multiTerm;
216 // if element equals the token, we push a vector with a unique element, and go to the next element
217 if( element == firstSimpleTerm ) {
218// #ifdef DEBUG_LP
219// LDEBUG << "GazeteerTransition::buildNextTermsList: push back in multiTermList singleton " << firstSimpleTerm.toStdString();
220// #endif
221 multiTerm.push_back(firstSimpleTerm);
222 multiTermList.push_back(multiTerm);
223 continue;
224 }
225 // within element, if firstSimpleTerm is not followed by others simple terms separated with space
226 // first term is only a prefix and does not match exactly firstSimpleTerm, go to the next element
227 int pos(0);
228 int index = element.indexOf(' ', pos);
229// #ifdef DEBUG_LP
230// LDEBUG << "GazeteerTransition::buildNextTermsList: pos = " << pos << ", index=" << index;
231// #endif
232 if( index != firstSimpleTerm.length() ) {
233// #ifdef DEBUG_LP
234// LDEBUG << "GazeteerTransition::buildNextTermsList: no second term for " << element.toStdString();
235// #endif
236 continue;
237 }
238 else {
239// #ifdef DEBUG_LP
240// LDEBUG << "GazeteerTransition::buildNextTermsList: push back in multiterm " << firstSimpleTerm.toStdString();
241// #endif
242 multiTerm.push_back(firstSimpleTerm);
243 }
244 // build list of elements following firstSimpleTerm
245 for( ; ; ) {
246 pos = index+1;
247 index = element.indexOf(' ', pos);
248// #ifdef DEBUG_LP
249// LDEBUG << "GazeteerTransition::buildNextTermsList: pos = " << pos << ", index=" << index;
250// #endif
251 if( index == -1 ) {
252// #ifdef DEBUG_LP
253// LDEBUG << "GazeteerTransition::buildNextTermsList: push back last term " << element.mid(pos).toStdString();
254// #endif
255 multiTerm.push_back(element.mid(pos));
256 break;
257 }
258 else
259 {
260// #ifdef DEBUG_LP
261// LDEBUG << "GazeteerTransition::buildNextTermsList: add term " << element.mid(pos,index-pos).toStdString();
262// #endif
263 multiTerm.push_back(element.mid(pos,index-pos));
264 }
265 }
266// #ifdef DEBUG_LP
267// LDEBUG << "GazeteerTransition::buildNextTermsList: push back list of " << multiTerm.size() << " elements";
268// #endif
269 multiTermList.push_back(multiTerm);
270 }
271 return( multiTermList.size() > 0 );
272}
273
274bool GazeteerTransition::
275checkMultiTerms( const AnalysisGraph& graph,
276 const LinguisticGraphVertex& position,
277 const LinguisticGraphVertex& limit,
279 Lima::AnalysisContent& analysis, const vector< vector< Lima::LimaString > >& additionalMultiTermList,
280 stack< deque< LinguisticGraphVertex >, vector< deque< LinguisticGraphVertex > > >& matches
281 ) const
282{
283 LIMA_UNUSED(limit)
284 LIMA_UNUSED(analysis)
285
286// #ifdef DEBUG_LP
287// AULOGINIT;
288// LDEBUG << "GazeteerTransition::checkMultiTerms( from " << position << ")";
289// #endif
290 // Iteration on multi-terms from gazeteer whose first term matches current token
291 std::vector<std::vector<LimaString> >::const_iterator multiTermsIt = additionalMultiTermList.begin();
292 const LinguisticGraph* lGraph = graph.getGraph();
293 for( ; multiTermsIt != additionalMultiTermList.end() ; multiTermsIt++ ) {
294 // iterator for simpleterms
295 std::vector<LimaString>::const_iterator termsIt = (*multiTermsIt).begin();
296 std::vector<LimaString>::const_iterator termsIt_end = (*multiTermsIt).end();
297// #ifdef DEBUG_LP
298// LDEBUG << "GazeteerTransition::checkMultiTerms: check multi-term ("
299// << *termsIt << " and " << (*multiTermsIt).size()-1 << " more...)";
300// #endif
301 // For each list of simple Terms, we make a deep first search in the graph
302 // searchPos stores a stack of position in the graph to perform the deep first search
303 // the completed path is stored in a deque of vertices (initialized with position)
304 std::deque<LinguisticGraphVertex> triggerMatch;
305 triggerMatch.push_back(position);
306 termsIt++;
307 // init search from position
308 std::unique_ptr<SearchGraph> tempSearchGraph(searchGraph->createNew());
309 tempSearchGraph->findNextVertices(lGraph, position);
310 // init current position
311 LinguisticGraphVertex nextVertex = position;
312 // if list is not exhausted
313
314 // case of empty list of simple term
315 if(termsIt == termsIt_end ) {
316 // Error!
317// #ifdef DEBUG_LP
318// LDEBUG << "GazeteerTransition::checkMultiTerms: list of simple terms is a singleton!";
319// #endif
320 matches.push(triggerMatch);
321 //matches.push(triggerMatch);
322 }
323 else {
324 // go one step ahead from curentPosition if possible
325 while ( tempSearchGraph->getNextVertex(lGraph, nextVertex )) {
326 const LinguisticGraphVertex& firstVertex = graph.firstVertex(),
327 lastVertex = graph.lastVertex();
328 if (nextVertex == lastVertex || nextVertex == firstVertex)
329// return false;
330 break;
331// #ifdef DEBUG_LP
332// LDEBUG << "GazeteerTransition::checkMultiTerms: progress one step forward, nextVertex=" << nextVertex;
333// LDEBUG << "GazeteerTransition::checkMultiTerms: test " << *termsIt;
334// #endif
335 // test currentVertex
336 Token* token = get(vertex_token, *lGraph, nextVertex);
337 LimaString form(token->stringForm());
338 if( form == *termsIt ) {
339// #ifdef DEBUG_LP
340// LDEBUG << "GazeteerTransition::checkMultiTerms: match with " << *termsIt;
341// #endif
342 // If match, push vertex in triggerMatch and initialize next step
343 // Push out_edge is a better if we have to follow the path from the begining ???
344 triggerMatch.push_back(nextVertex);
345 // stack next step to continue the search
346 tempSearchGraph->findNextVertices(lGraph, nextVertex);
347 termsIt++;
348 if(termsIt == termsIt_end ) {
349// #ifdef DEBUG_LP
350// LDEBUG << "GazeteerTransition::checkMultiTerms: list of simple terms exhausted!";
351// #endif
352 // list of Simple term exhausted: success
353 // we push the path in the aGraph as a solution of triggerMatch
354 // Only if size of solution is greater than previous one !!
355 if( matches.empty() || (triggerMatch.size() > matches.top().size()) ) {
356// #ifdef DEBUG_LP
357// LDEBUG << "GazeteerTransition::checkMultiTerms: push (in matches) a deque of size " << triggerMatch.size();
358// #endif
359 matches.push(triggerMatch);
360 }
361 // no need to go forward
362 break;
363 }
364 // else we do not stack next steps, we obtain a cut
365 }
366 }
367 }
368 }
369
370 if( matches.empty() )
371 return false;
372 return true;
373}
374
375} // namespace end
376} // namespace end
377} // namespace end
#define LIMA_UNUSED(x)
Definition LimaCommon.h:224
#define LDEBUG
Definition LimaCommon.h:157
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define AULOGINIT
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
GazeteerTransition & operator=(const GazeteerTransition &)
bool matchPath(const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &vertex, const LinguisticGraphVertex &limit, const SearchGraph *searchGraph, AnalysisContent &analysis, const LinguisticAnalysisStructure::Token *token, std::deque< LinguisticGraphVertex > &vertices, const LinguisticAnalysisStructure::MorphoSyntacticData *) const
bool compare(const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &vertex, AnalysisContent &analysis, const LinguisticAnalysisStructure::Token *token, const LinguisticAnalysisStructure::MorphoSyntacticData *data) const override
comparison of the transition with a token : can use the graph and vertex info, associated with genera...
bool operator==(const TransitionUnit &) const override
virtual SearchGraph * createNew() const =0
bool compareProperties(const TransitionUnit &t) const
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
NAUTITIA.
QString LimaString
Definition LimaString.h:33
STL namespace.