LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
RegexMatcher.cpp
Go to the documentation of this file.
1// Copyright 2002-2019 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
17#include "RegexMatcher.h"
18
29
30#include <QtCore/QRegularExpression>
31
32#include <vector>
33
34#include <string>
35#include <queue>
36#include <set>
37
38// To use tuples
39#include <boost/tuple/tuple.hpp>
40//Comparison operators can be included with:
41#include <boost/tuple/tuple_comparison.hpp>
42// To use tuple input and output operators,
43#include <boost/tuple/tuple_io.hpp>
44
45typedef boost::tuple< size_t, size_t, QString, QString > RegexMatch;
46
47using namespace std;
50using namespace Lima::Common::Misc;
51using namespace Lima::Common::AnnotationGraphs;
52
53#define REGEXREPLACERLOGINIT LOGINIT("LP::RegexReplacer")
54
55
56namespace Lima
57{
58namespace LinguisticProcessing
59{
60namespace RegexMatching
61{
62
64{
65public:
68
70
71 MediaId m_language;
72 // map between a regex and the label it marks
73 QMap< QString, QString > m_regexes;
74};
75
77
81
83{
84 delete m_d;
85}
86
89 Manager* manager)
90
91{
93 m_d->m_language=manager->getInitializationParameters().media;
94
95 try
96 {
97 std::map <std::string, std::string >& regexes = unitConfiguration.getMapAtKey("regexes");
98 for (std::map <std::string, std::string >::const_iterator it = regexes.begin(); it != regexes.end(); it++)
99 {
100 m_d->m_regexes.insert( QString::fromUtf8 ((*it).first.c_str()), QString::fromUtf8((*it).second.c_str()) );
101 }
102 }
103 catch (NoSuchMap& )
104 {
105 LERROR << "no map 'regexes' in RegexReplacer group configuration (language="
106 << (int) m_d->m_language << ")";
107 throw InvalidConfiguration();
108 }
109 // when input XML file is syntactically wrong
110 catch (XmlSyntaxException &exc)
111 {
112 std::ostringstream mess;
113 mess << "XmlSyntaxException at line "<<exc._lineNumber<<" cause: ";
114 switch (exc._why)
115 {
116 case XmlSyntaxException::SYNTAX_EXC : mess << "SYNTAX_EXC"; break;
117 case XmlSyntaxException::NO_DATA_EXC : mess << "NO_DATA_EXC"; break;
118 case XmlSyntaxException::DOUBLE_EXC : mess << "DOUBLE_EXC"; break;
119 case XmlSyntaxException::FWD_CLASS_EXC : mess << "FWD_CLASS_EXC"; break;
120 case XmlSyntaxException::MULT_CLASS_EXC : mess << "MULT_CLASS_EXC"; break;
121 case XmlSyntaxException::EOF_EXC : mess << "EOF_EXC"; break;
122 case XmlSyntaxException::NO_CODE_EXC : mess << "NO_CODE_EXC"; break;
123 case XmlSyntaxException::BAD_CODE_EXC : mess << "BAD_CODE_EXC"; break;
124 case XmlSyntaxException::NO_CLASS_EXC : mess << "NO_CLASS_EXC"; break;
125 case XmlSyntaxException::UNK_CLASS_EXC : mess << "UNK_CLASS_EXC"; break;
126 case XmlSyntaxException::INT_ERROR_EXC : mess << "INT_ERROR_EXC"; break;
127 case XmlSyntaxException::INV_CLASS_EXC : mess << "INV_CLASS_EXC"; break;
128 default: mess << "??";
129 }
130 LERROR << mess.str();
131 throw InvalidConfiguration();
132 }
133 catch (std::exception &exc)
134 {
135 // @todo remove all causes of InfiniteLoopException
136 LERROR << exc.what();
137 throw InvalidConfiguration();
138 }
139
140}
141
143 AnalysisContent& analysis) const
144{
145 Lima::TimeUtilsController timer("RegexReplacer");
147 LINFO << "start RegexMatcher process";
148 AnalysisGraph* anagraph = static_cast< AnalysisGraph* >(analysis.getData("AnalysisGraph").get());
149 if (anagraph == 0) {
150 LERROR << "no AnalysisGraph available for RegexMatcher ! abort";
151 return MISSING_DATA;
152 }
153 if (!m_d->checkGraphIsString(anagraph)) {
154 LERROR << "AnalysisGraph is not a string ! abort";
155 return UNKNOWN_ERROR;
156 }
157 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.getData("AnnotationData"));
158 if (annotationData==0)
159 {
160 annotationData = std::make_shared<AnnotationData>();
161 anagraph->populateAnnotationGraph(annotationData.get(), "AnalysisGraph");
162 if (std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("PosGraph")) != 0)
163 {
164 std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("PosGraph"))->populateAnnotationGraph(
165 annotationData.get(), "PosGraph");
166 }
167 analysis.setData("AnnotationData",annotationData);
168 }
169
170 std::map< size_t, RegexMatch > matches;
171 LinguisticGraph* graph = anagraph->getGraph();
172
173 VertexTokenPropertyMap tokenMap = get( vertex_token, *graph );
174 VertexDataPropertyMap dataMap = get( vertex_data, *graph );
175
176 auto originalText = std::dynamic_pointer_cast<LimaStringText>(analysis.getData("Text"));
177 std::string text = Common::Misc::limastring2utf8stdstring(*originalText);
178 for (auto reit = m_d->m_regexes.begin(); reit != m_d->m_regexes.end(); reit++)
179 {
180 QRegularExpression re(reit.key());
181 QString& tstatus = reit.value();
182
183 qsizetype position = 0;
184 QRegularExpressionMatch qmatch;
185 while ((position = originalText->indexOf(re, position, &qmatch)) != -1) {
186 QString matchedString = originalText->mid(position, qmatch.capturedLength());
187#ifdef DEBUG_LP
188 LDEBUG << "Matched '" << matchedString << "' at " << position << " as " << tstatus;
189#endif
190 RegexMatch match = boost::make_tuple(position+1, matchedString.size(), matchedString, tstatus);
191 matches.insert(std::make_pair(position+1, match));
192
193 position += qmatch.capturedLength();
194 }
195 }
196#ifdef DEBUG_LP
197 LDEBUG << "Matching finished. Updating graph";
198#endif
199 // must ensure that we handle only one regex on a given token
200
201 if (!matches.empty())
202 {
203 std::map< size_t, RegexMatch >::const_iterator matchesIt = matches.begin();
204 size_t currentMatchPosition = (*matchesIt).second.get<0>();
205 size_t currentMatchLength = (*matchesIt).second.get<1>();
206 QString currentMatchString = (*matchesIt).second.get<2>();
207 QString currentMatchStatus = (*matchesIt).second.get<3>();
208
209 LinguisticGraphVertex v = anagraph->firstVertex();
210 LinguisticGraphOutEdgeIt outItr,outItrEnd;
211 for (boost::tie(outItr,outItrEnd)=boost::out_edges(v,*graph); outItr!=outItrEnd; outItr++)
212 {
213 v=target(*outItr,*graph);
214 break;
215 }
216
217 // hypothesis : the graph is a string of tokens, there is only one path
218 while (matchesIt != matches.end() && v != anagraph->lastVertex())
219 {
220 Token* token=get(vertex_token,*graph,v);
221 uint64_t tokenPosition = token->position();
222#ifdef DEBUG_LP
223 LDEBUG << "Working on current match '" << currentMatchString << "' (" << currentMatchPosition << ", " << currentMatchLength << ") and token" << v << "at position" << tokenPosition;
224#endif
225 while (v != anagraph->lastVertex() && tokenPosition < currentMatchPosition)
226 {
227 for (boost::tie(outItr,outItrEnd)=boost::out_edges(v,*graph); outItr != outItrEnd; outItr++)
228 {
229 v=target(*outItr, *graph);
230 break;
231 }
232 if (v!= anagraph->lastVertex())
233 {
234 token=get(vertex_token,*graph,v);
235 tokenPosition = token->position();
236 }
237 }
238
239#ifdef DEBUG_LP
240 LDEBUG << "current token ("<<v<<", "<<token->stringForm()<<") position is: " << tokenPosition
241 << " ; current match position is: " << currentMatchPosition;
242#endif
243 // current token is at current match position
244 if (tokenPosition == currentMatchPosition)
245 {
246#ifdef DEBUG_LP
247 LDEBUG << " current match found on token on token" << v << "at position" << tokenPosition << ". Now filling mach vertices" ;
248#endif
249 std::vector< LinguisticGraphVertex > matchVertices;
250 matchVertices.push_back(v);
251 // get all tokens inside current match
253 Token* nextToken = 0;
254 uint64_t nextTokenPosition = std::numeric_limits<uint64_t>::max();
255 for (boost::tie(outItr,outItrEnd)=boost::out_edges(next,*graph); outItr!=outItrEnd; outItr++)
256 {
257 next=target(*outItr,*graph);
258 nextToken=get(vertex_token,*graph,next);
259 nextTokenPosition = nextToken!=0?nextToken->position():0;
260 break;
261 }
262
263 while (next != anagraph->lastVertex() && nextTokenPosition < (currentMatchPosition + currentMatchLength))
264 {
265#ifdef DEBUG_LP
266 LDEBUG << " next token ("<<next<<") position is: " << nextTokenPosition;
267#endif
268 matchVertices.push_back(next);
269 for (boost::tie(outItr,outItrEnd)=boost::out_edges(next,*graph); outItr!=outItrEnd; outItr++)
270 {
271 next=target(*outItr,*graph);
272 nextToken=get(vertex_token,*graph,next);
273 nextTokenPosition = nextToken!=0?nextToken->position():0;
274 break;
275 }
276 }
277 auto matchVerticesIt = matchVertices.begin(), matchVerticesIt_end = matchVertices.end();
278
279#ifdef DEBUG_LP
280 LDEBUG << "Match Found. Vertices are: ";
281 for (; matchVerticesIt != matchVerticesIt_end; matchVerticesIt++)
282 {
283 LDEBUG << (*matchVerticesIt) << ", ";
284 }
285 LDEBUG;
286#endif
287 if (matchVertices.empty()) continue;
288
289 // before = get vertex before first match vertex
290 LinguisticGraphInEdgeIt inIt, inIt_end;
291 boost::tie (inIt, inIt_end) = boost::in_edges(*matchVertices.begin(), *graph);
292 LinguisticGraphVertex before = boost::source(*inIt, *graph);
293 // remove all before out_edges
294 for (; inIt != inIt_end; inIt++)
295 {
296 boost::remove_edge(*inIt, *graph);
297 }
298 // after = get vertex after last match vertex
299 LinguisticGraphOutEdgeIt outIt, outIt_end;
300 boost::tie (outIt, outIt_end) = boost::out_edges(*matchVertices.rbegin(), *graph);
301 LinguisticGraphVertex after = boost::target(*outIt, *graph);
302 if (outIt != outIt_end) after = boost::target(*outIt, *graph);
303 // remove all after in_edges
304 for (; outIt != outIt_end; outIt++)
305 {
306 boost::remove_edge(*outIt, *graph);
307 }
308 // create newvertex, with new fulltoken
309 StringsPoolIndex form=(Common::MediaticData::MediaticData::changeable().stringsPool(m_d->m_language))[currentMatchString];
310 Token* newToken=new Token( form, currentMatchString, currentMatchPosition, currentMatchLength );
311 newToken->setStatus(TStatus());
312 newToken->status().setDefaultKey(currentMatchStatus);
314 // add newvertex
315 LinguisticGraphVertex newVertex = boost::add_vertex(*graph);
316 tokenMap[newVertex]=newToken;
317 dataMap[newVertex]=newData;
318 newToken-> setPosition(currentMatchPosition);
319 // create edge before -> newvertex
320 boost::add_edge(before, newVertex, *graph);
321 // create edge newvertex -> before
322 boost::add_edge(newVertex, after, *graph);
323 AnnotationGraphVertex agv = annotationData->createAnnotationVertex();
324 annotationData->addMatching("AnalysisGraph", newVertex, "annot", agv);
325 annotationData->annotate(agv, Common::Misc::utf8stdstring2limastring("AnalysisGraph"), static_cast< uint64_t >(newVertex));
326
327 GenericAnnotation ga(matchVertices);
328 annotationData->annotate(agv, Common::Misc::utf8stdstring2limastring("regexmatches"), ga);
329 v = after;
330 matchesIt++;
331 if (matchesIt != matches.end())
332 {
333 currentMatchPosition = (*matchesIt).second.get<0>();
334 currentMatchLength = (*matchesIt).second.get<1>();
335 currentMatchString = (*matchesIt).second.get<2>();
336 currentMatchStatus = (*matchesIt).second.get<3>();
337#ifdef DEBUG_LP
338 LDEBUG << " current match is now: " << currentMatchPosition << currentMatchLength << currentMatchString << currentMatchStatus;
339#endif
340 }
341 else
342 {
343#ifdef DEBUG_LP
344 LDEBUG << " no more match";
345#endif
346 break;
347 }
348 }
349 else // tokenPosition > currentMatchPosition
350 {
351#ifdef DEBUG_LP
352 LDEBUG << "Skiping matches up to after next token position: " << tokenPosition << currentMatchPosition << currentMatchLength;
353#endif
354 // advances current match until we find one which starts after the end of the last token which is in the current match
355 while ((matchesIt != matches.end()) && ((currentMatchPosition ) <= tokenPosition))
356 {
357 matchesIt++;
358 if (matchesIt == matches.end()) break;
359 currentMatchPosition = (*matchesIt).second.get<0>();
360 currentMatchLength = (*matchesIt).second.get<1>();
361 currentMatchString = (*matchesIt).second.get<2>();
362 currentMatchStatus = (*matchesIt).second.get<3>();
363 }
364#ifdef DEBUG_LP
365 if (matchesIt != matches.end())
366 LDEBUG << " current match is now: " << currentMatchPosition << currentMatchLength << currentMatchString << currentMatchStatus;
367 else
368 LDEBUG << " no more match";
369#endif
370 }
371 }
372 }
373#ifdef DEBUG_LP
374 LDEBUG << "DONE";
375#endif
376 return SUCCESS_ID;
377}
378
380{
381 // there is a first vertex, a last vertex and at least one token vertex
382 if (boost::num_vertices(*anagraph->getGraph()) <= 2) return false;
383
384 LinguisticGraphVertex next = anagraph->firstVertex();
385 LinguisticGraphVertex last = anagraph->lastVertex();
386 while (next != last)
387 {
388 // each vertex must have exactly one next vertex
389 if (boost::out_degree(next, *anagraph->getGraph()) != 1) return false;
390 LinguisticGraphOutEdgeIt outItr,outItrEnd;
391 boost::tie(outItr,outItrEnd)=boost::out_edges(next,*anagraph->getGraph());
392 next=target(*outItr,*anagraph->getGraph());
393 }
394 return true;
395}
396
397} //namespace RegexReplace
398} // namespace LinguisticProcessing
399} // namespace Lima
This file is the main header file for the data related to annotation graphs.
#define LDEBUG
Definition LimaCommon.h:157
#define LINFO
Definition LimaCommon.h:158
#define LERROR
Definition LimaCommon.h:161
LinguisticGraph::in_edge_iterator LinguisticGraphInEdgeIt
boost::property_map< LinguisticGraph, vertex_data_t >::type VertexDataPropertyMap
boost::property_map< LinguisticGraph, vertex_token_t >::type VertexTokenPropertyMap
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define TOKENIZERLOGINIT
#define REGEXREPLACERLOGINIT
boost::tuple< size_t, size_t, QString, QString > RegexMatch
A process unit able to match regex against analysed text.
#define REGEXREPLACER_CLASSID
Defines a Factory to create Object of type Base.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
This class allows to convert any object into an annotation by inheritance.
const FsaStringsPool & stringsPool(MediaId med) const
std::map< std::string, std::string > & getMapAtKey(const std::string &key)
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Use this exception to signal an error in one of the configuration files.
Definition LimaCommon.h:345
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
void populateAnnotationGraph(Common::AnnotationGraphs::AnnotationData *annotData, const std::string &src)
Creates the annotations in the agdata corresponding to this graphs vertices.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
void setDefaultKey(const Lima::LimaString &defaultKey)
Definition TStatus.cpp:325
void setStatus(const TStatus &status)
Set the TStatus of a token.
Definition Token.h:100
bool checkGraphIsString(const LinguisticAnalysisStructure::AnalysisGraph *anagraph) const
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
static MediaticData & changeable()
singleton accessor
Definition Singleton.h:71
This file contains a class to control log of informations about time, such as logging cumulated time ...
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
LimaString utf8stdstring2limastring(const std::string &src)
SimpleFactory< MediaProcessUnit, RegexMatcher > tokenizerFactory(REGEXREPLACER_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ UNKNOWN_ERROR
Definition LimaCommon.h:238
@ SUCCESS_ID
Definition LimaCommon.h:237
@ MISSING_DATA
Definition LimaCommon.h:243
STL namespace.
launch exception related to the configuration file parsing