LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
ApproxStringMatcher.cpp
Go to the documentation of this file.
1// Copyright 2002-2020 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
7
22#include <tre/regex.h>
23#include <boost/regex.hpp>
24//#include "linguisticProcessing/common/annotationGraph/AnnotationData.h"
27
28using namespace std;
29using namespace Lima::Common::AnnotationGraphs;
30using namespace Lima::Common::MediaticData;
34
35namespace Lima
36{
37namespace LinguisticProcessing
38{
39namespace MorphologicAnalysis
40{
41typedef boost::basic_regex<wchar_t> wide_regex;
42
44
45QChar BLANK_SEPARATOR= ' ';
46
47std::ostream& operator<<(ostream& os, const Suggestion& suggestion)
48{
49 os << "Suggestion pos=(" << suggestion.startPosition << ","<< suggestion.endPosition<< ")"
50 << " nbErr=" << suggestion.nb_error << std::endl;
51 return os;
52}
53
54QDebug& operator<<(QDebug& os, const Suggestion& suggestion)
55{
56 os << "Suggestion pos=(" << suggestion.startPosition << ","<< suggestion.endPosition<< ")"
57 << " nbErr=" << suggestion.nb_error;
58 return os;
59}
60
61std::ostream& operator<<(ostream& os, const Solution& solution)
62{
63 os << "Solution: " << solution.suggestion
64 << " vertices=";
65 std::deque<LinguisticGraphVertex>::const_iterator it = solution.vertices.begin();
66 if( it != solution.vertices.end() ) {
67 os << *(it++);
68 }
69 for( ; it != solution.vertices.end() ; it++ ) {
70 os << "," << *it;
71 }
74 << "length=" << solution.length
75 << std::endl;
76 return os;
77}
78
79QDebug& operator<<(QDebug& os, const Solution& solution)
80{
81 os << "Solution: " << solution.suggestion
82 << " vertices=";
83 std::deque<LinguisticGraphVertex>::const_iterator it = solution.vertices.begin();
84 if( it != solution.vertices.end() ) {
85 os << *(it++);
86 }
87 for( ; it != solution.vertices.end() ; it++ ) {
88 os << "," << *it;
89 }
92 << "length=" << solution.length;
93 return os;
94}
95
96bool SolutionCompare::operator() (const Solution& s1, const Solution& s2) const {
98 return true;
99 }
100 if( s1.suggestion.nb_error > s2.suggestion.nb_error ) {
101 return false;
102 }
104 return true;
105 }
107 return false;
108 }
109 if( s1.length > s2.length) {
110 return true;
111 }
112 if( s1.length < s2.length) {
113 return false;
114 }
115 return true;
116}
117
119 m_language(0),
120 m_lexicon(0),
121 m_sp(0),
122 m_nbMaxNumError(0),
123 m_nbMaxDenError(1),
124 m_entityType(),
125 m_entityGroupId()
126{}
127
129{
130 // delete m_reader;
131}
132
135 Manager* manager)
136{
138
139 m_language = manager->getInitializationParameters().media;
141
142 // Get groupId and Entity type in group
143 Common::MediaticData::EntityGroupId foundGroup;
144 try
145 {
146 std::string entityGroupName=unitConfiguration.getParamsValueAtKey("entityGoup");
147 LimaString lsEntityGroupName = Lima::Common::Misc::utf8stdstring2limastring(entityGroupName);
148 m_entityGroupId = Common::MediaticData::MediaticData::single().getEntityGroupId(lsEntityGroupName);
149 }
150 catch (NoSuchParam& )
151 {
152 LERROR << "no param 'entityGoup' in ApproxStringMatcher group for language " << (int) m_language;
153 throw InvalidConfiguration();
154 }
155 try
156 {
157 std::string entityName=unitConfiguration.getParamsValueAtKey("entityName");
159 try {
160 m_entityType = Common::MediaticData::MediaticData::single().getEntityType(lsentityName);
161 } catch (const LimaException& e) {
162 QString errorString;
163 QTextStream qts(&errorString);
164 qts << __FILE__ << ", line" << __LINE__ << "Unknown entity type" << lsentityName;
165 LERROR << errorString;
166 throw InvalidConfiguration(errorString.toStdString());
167 }
168 }
169 catch (NoSuchParam& )
170 {
171 QString errorString;
172 QTextStream qts(&errorString);
173 qts << __FILE__ << ", line" << __LINE__
174 << "no param 'entityGoup' in ApproxStringMatcher group for language "
175 << (int) m_language;
176 LERROR << errorString;
177 throw InvalidConfiguration(errorString.toStdString());
178 }
179
180 try
181 {
182 std::string np = unitConfiguration.getParamsValueAtKey("NPCategory");
183 m_neCode=static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager().getPropertyManager("MACRO").getPropertyValue(np);
184 }
185 catch (NoSuchParam& )
186 {
187 LERROR << "no param 'NPCategory' in ApproxStringMatcher group for language " << (int) m_language;
188 throw InvalidConfiguration();
189 }
190
191 try
192 {
193 std::string microCategory=unitConfiguration.getParamsValueAtKey("NPMicroCategory");
194 m_neMicroCode=static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager().getPropertyManager("MICRO").getPropertyValue(microCategory);
195 }
196 catch (NoSuchParam& )
197 {
198 LERROR << "no param 'NPMicroCategory' in ApproxStringMatcher group for language " << (int) m_language;
199 throw InvalidConfiguration();
200 }
201
202 // get index of names
203 std::string nameindexId;
204 try
205 {
206 nameindexId = unitConfiguration.getParamsValueAtKey("nameindex");
207 }
208 catch (NoSuchParam& )
209 {
210 LERROR << "no param 'nameindex' in ApproxStringMatcher group for language " << (int) m_language;
211 throw InvalidConfiguration();
212 }
213 const auto res = LinguisticResources::single().getResource(m_language,nameindexId);
214 m_nameIndex = std::dynamic_pointer_cast<NameIndexResource>(res);
215
216 /*
217 // get dictionary of normalized forms
218 string dico;
219 try
220 {
221 dico=unitConfiguration.getParamsValueAtKey("dictionary");
222 }
223 catch (NoSuchParam& )
224 {
225 LERROR << "no param 'dictionary' in ApproxStringMatcher group for language " << (int) m_language;
226 throw InvalidConfiguration();
227 }
228 auto res = LinguisticResources::single().getResource(m_language,dico);
229 AbstractAccessResource* lexicon = lexicon=static_cast<AbstractAccessResource*>(res);
230 m_lexicon = lexicon->getAccessByString();
231 */
232
233 // get max edit distance
234 try
235 {
236 std::string nbMaxErrorStr=unitConfiguration.getParamsValueAtKey("nbMaxNumError");
237 std::istringstream iss(nbMaxErrorStr);
238 iss >> m_nbMaxNumError;
239 }
240 catch (NoSuchParam& )
241 {
242 LERROR << "no param 'nbMaxNumError' in ApproxStringMatcher group for language " << (int) m_language;
243 throw InvalidConfiguration();
244 }
245 try
246 {
247 std::string nbMaxErrorStr=unitConfiguration.getParamsValueAtKey("nbMaxDenError");
248 std::istringstream iss(nbMaxErrorStr);
249 iss >> m_nbMaxDenError;
250 }
251 catch (NoSuchParam& )
252 {
253 LERROR << "no param 'nbMaxDenError' in ApproxStringMatcher group for language " << (int) m_language;
254 throw InvalidConfiguration();
255 }
256
257 // get generalization pattern
258 try
259 {
260 std::map <std::string, std::string >& regexes = unitConfiguration.getMapAtKey("generalizationRules");
261 std::deque <std::string >& regexesOrder = unitConfiguration.getListsValueAtKey("generalizationRulesOrder");
262 for (std::deque <std::string >::const_iterator it = regexesOrder.begin(); it != regexesOrder.end(); it++)
263 {
264 const std::string& key = *it;
265 QString patternQ = QString::fromUtf8 (key.c_str());
266 std::basic_string<wchar_t> patternWS = NameIndexResource::LimaStr2wcharStr(patternQ);
267 const std::string& value=regexes.at(key);
268 QString substitutionQ = QString::fromUtf8 (value.c_str());
269 std::basic_string<wchar_t> substitutionWS = NameIndexResource::LimaStr2wcharStr(substitutionQ);
270 m_regexes.push_back( std::pair<std::basic_string<wchar_t>,std::basic_string<wchar_t> >(patternWS,
271 substitutionWS) );
272 }
273 }
274 catch (NoSuchParam& )
275 {
276 LERROR << "no map 'generalization' in RegexReplacer group configuration fot language "
277 << (int) m_language;
278 throw InvalidConfiguration();
279 }
280 catch (std::out_of_range& )
281 {
282 LERROR << "no value for some key in 'generalizationRules' in RegexReplacer group configuration fot language "
283 << (int) m_language;
284 throw InvalidConfiguration();
285 }
286}
287
288
290 AnalysisContent& analysis) const
291{
292 Lima::TimeUtilsController timer("ApproxStringMatcher");
294 LINFO << "starting process ApproxStringMatcher";
295
296 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.getData("LinguisticMetaData"));
297 std::basic_string<wchar_t> indexName;
298 try {
299 if (metadata != 0) {
300 std::string indexNametds = metadata->getMetaData("index");
301 QString indexNamels = QString::fromUtf8 (indexNametds.c_str());
302 indexName = NameIndexResource::LimaStr2wcharStr(indexNamels);
303 }
304 }
306 // do nothing: try full set of names
307 }
308#ifdef DEBUG_LP
309 QString name = wcharStr2LimaStr(indexName);
310 LDEBUG << "ApproxStringMatcher::process: index from metadata= "
312#endif
313
314 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"));
315 // initialize annotation data
316 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.getData("AnnotationData"));
317 if (annotationData==0)
318 {
319 LINFO << "ApproxStringMatcher::process no annotation data, creating and populating it";
320 annotationData = std::make_shared<AnnotationData>();
321 analysis.setData("AnnotationData",annotationData);
322 }
323 anagraph->populateAnnotationGraph(annotationData.get(), "AnalysisGraph");
324
325 // initialize annotation data for SpecificEntity???
326 if (annotationData==0)
327 {
328 LERROR << "ApproxStringMatcher::process: no AnnotationData, cannot store result";
330 }
331 // see annotationData...
332 if (annotationData->dumpFunction("SpecificEntity") == 0)
333 {
334 annotationData->dumpFunction("SpecificEntity", new Lima::LinguisticProcessing::SpecificEntities::DumpSpecificEntityAnnotation());
335 }
336
337 // Initalize list of suggestions, ordered by number of errors and position in text
338 OrderedSolution solutions;
339 // std::vector<Solution> result;
340
341 // Initalize set of names to search for
342 std::pair<NameIndex::const_iterator,NameIndex::const_iterator> nameRange;
343 if( (indexName.length() == 0) || !(m_nameIndex->withIndex()) ) {
344 nameRange.first = m_nameIndex->begin();
345 nameRange.second = m_nameIndex->end();
346 }
347 else {
348 nameRange = m_nameIndex->equal_range(indexName);
349 // nameRange.first = m_nameIndex->lower_bound(indexName);
350 // nameRange.second = m_nameIndex->upper_bound(indexName);
351 }
352#ifdef DEBUG_LP
353 const std::basic_string<wchar_t> firstNameW = (*(nameRange.first)).second;
354 std::basic_string<wchar_t> lastNameW;
355 for( NameIndex::const_iterator it= nameRange.first ; it != nameRange.second ; it++ )
356 lastNameW = (*it).second;
357 const QString firstNameQ = wcharStr2LimaStr(firstNameW);
358 const QString lastNameQ = wcharStr2LimaStr(lastNameW);
359 LDEBUG << "ApproxStringMatcher::process: nameRange= from " << firstNameQ << " to " << lastNameQ;
360#endif
361 LinguisticGraph & g = *(anagraph->getGraph());
362 matchApproxTokenAndFollowers(g, anagraph->firstVertex(), anagraph->lastVertex(), nameRange, solutions);
363#ifdef DEBUG_LP
364 LDEBUG << "ApproxStringMatcher::process: solution.size()=" << solutions.size();
365#endif
366 for( OrderedSolution::const_iterator sIt = solutions.begin() ; sIt != solutions.end() ; sIt++ ) {
367 // check that vertices of solution are still in graph
368#ifdef DEBUG_LP
369 LDEBUG << "ApproxStringMatcher::process: check " << *sIt;
370#endif
371 const Solution solution= *sIt;
372 bool outOfGraph=false;
373#ifdef DEBUG_LP
374 std::deque<LinguisticGraphVertex>::const_iterator vIt2 = solution.vertices.begin();
375 if( vIt2 != solution.vertices.end() ) {
376 LDEBUG << *(vIt2++);
377 }
378
379 for( ; vIt2 != solution.vertices.end() ; vIt2++ ) {
380 LDEBUG << "," << *vIt2;
381 }
382#endif
383 for( std::deque<LinguisticGraphVertex>::const_iterator vIt = solution.vertices.begin();
384 vIt != solution.vertices.end() ; vIt++ )
385 {
386#ifdef DEBUG_LP
387 LDEBUG << "ApproxStringMatcher::process: outOfGraph = " << outOfGraph << ", check vertex " << *vIt;
388#endif
389 // If one of vertices has no successor or no predecessor, it means is belongs no more to the graph
390 LinguisticGraphOutEdgeIt outEdge,outEdge_end;
391 boost::tie (outEdge,outEdge_end)=out_edges(*vIt,g);
392 if( outEdge == outEdge_end ) {
393 outOfGraph=true;
394 }
395 LinguisticGraphInEdgeIt inEdge,inEdge_end;
396 boost::tie (inEdge,inEdge_end)=in_edges(*vIt,g);
397 if( inEdge == inEdge_end ) {
398 outOfGraph=true;
399 }
400 if( outOfGraph )
401 break;
402 }
403 if( !outOfGraph ) {
404 // TODO: check if( solution.suggestion.nb_error <= (len*m_nbMaxNumError)/m_nbMaxDenError ) ??
405 createVertex(g, anagraph->firstVertex(), anagraph->lastVertex(), solution, annotationData.get() );
406 // result.push_back(solution);
407 }
408 }
409
410#ifdef DEBUG_LP
411 LDEBUG << "ending process ApproxStringMatcher";
412#endif
413 return SUCCESS_ID;
414}
415
416void ApproxStringMatcher::createVertex(
420 const Solution& solution,
421 AnnotationData* annotationData
422) const
423{
424#ifdef DEBUG_LP
426 LDEBUG << "ApproxStringMatcher::createVertex( solution=" << solution << ")";
427#endif
428 // VertexTokenPropertyMap tokenMap=get(vertex_token,g);
429 // VertexDataPropertyMap dataMap = get(vertex_data, g);
430
431 // create new vertex in analysis graph
432 LinguisticGraphVertex newVertex = add_vertex(g);
433 // Find previous vertex
434 LinguisticGraphOutEdgeIt outEdge,outEdge_end;
435 LinguisticGraphVertex previousVertex = vStart;
436 for( ; previousVertex != vEnd ; ) {
437 boost::tie (outEdge,outEdge_end)=out_edges(previousVertex,g);
438 LinguisticGraphVertex nextVertex = target(*outEdge,g);
439 if( nextVertex == solution.vertices.front() ) {
440 break;
441 }
442 else {
443 previousVertex = nextVertex;
444 }
445 }
446 // Find next vertex
447 boost::tie (outEdge,outEdge_end)=out_edges(solution.vertices.back(),g);
448 LinguisticGraphVertex nextVertex = target(*outEdge,g);
449 // remove edges
450 boost::remove_edge(previousVertex,solution.vertices.front(), g);
451 boost::remove_edge(solution.vertices.back(),nextVertex, g);
452 // replace edges
453 bool success;
455 boost::tie(e, success) = add_edge(previousVertex, newVertex, g);
456 boost::tie(e, success) = add_edge(newVertex, nextVertex, g);
457
458 // Create token for this vertex
459 StringsPoolIndex form = (*m_sp)[solution.form];
460 // Qu'est ce qu'on affecte comme chaîne et comme position à ce nouveau token ???
461 Token* newToken = new Token(
462 form,
463 (*m_sp)[form],
464 solution.suggestion.startPosition,
465 solution.suggestion.endPosition - solution.suggestion.startPosition);
466 // specify status for token
467 TStatus tStatus;
468 tStatus.reset();
469 tStatus.setStatus(T_PATTERN);
471 newToken->setStatus(tStatus);
472 // tokenMap[newVertex] = newToken;
473 put(vertex_token,g,newVertex,newToken);
474
475 // Create MorphoSyntacticData for this vertex
476 MorphoSyntacticData* newMorphoSyntacticData = new MorphoSyntacticData ();
478 elem.inflectedForm = newToken->form(); // (StringsPoolIndex)
479 elem.lemma = newToken->form(); // (StringsPoolIndex)
480 elem.normalizedForm = (*m_sp)[solution.normalizedForm];
481 elem.type = SPECIFIC_ENTITY; // MorphoSyntacticType
482 elem.properties = m_neCode; // LinguisticCode
483 elem.properties = m_neMicroCode; // LinguisticCode
484 newMorphoSyntacticData->push_back(elem);
485 // dataMap[newVertex] = newMorphoSyntacticData;
486 put(vertex_data,g,newVertex,newMorphoSyntacticData);
487
488 // Create vertex fot annotation graph
489 AnnotationGraphVertex agv = annotationData->createAnnotationVertex();
490 // make access to this annotation vertex from newVertex
491 annotationData->addMatching("AnalysisGraph", newVertex, "annot", agv);
492 // make access back to newVertex from this annotation
493 annotationData->annotate(agv, Common::Misc::utf8stdstring2limastring("AnalysisGraph"),
494 newVertex);
495 // Create annotation data of type 'SpecificEntity'
497 solution.vertices,
498 m_entityType,
499 solution.form, solution.normalizedForm, solution.suggestion.nb_error,
500 solution.suggestion.startPosition, solution.length, *m_sp);
501 GenericAnnotation spGa(spAnnot);
502 // make access to annotation data from annotation vertex
503 annotationData->annotate(agv, Common::Misc::utf8stdstring2limastring("SpecificEntity"), spGa);
504 // Créer une 'feature' 'value' sans information de localisation
505 // pour qu'elle soit produite comme 'normalization' dans le logger
506}
507
508QString ApproxStringMatcher::wcharStr2LimaStr(const std::basic_string<wchar_t>& wstring) const {
509 // convert std::basic_string<wchar_t> to QString
510 return QString::fromWCharArray(wstring.c_str(),wstring.length());
511}
512
513std::basic_string<wchar_t> ApproxStringMatcher::buildPattern(const std::basic_string<wchar_t>& normalizedForm) const {
514#ifdef DEBUG_LP
516#endif
517 // convert normalizedForm into std::basic_string<wchar_t>
518 // std::basic_string<wchar_t> wpattern = LimaStr2wcharStr(normalizedForm);
519 std::basic_string<wchar_t> wpattern = normalizedForm;
520 // escape 'regex special character', to avoid to interpret a character in a name as regex special character:
521 // '.', '^', '$', '|', '(', ')', '[', ']', '{', '}', '*', '+', '?', '\'
522 wide_regex esc(L"[.^$|()\\[\\]{}*+?\\\\]");
523 const std::basic_string<wchar_t> rep(L"\\\\&");
524 wpattern = boost::regex_replace(wpattern, esc, rep, boost::match_default | boost::format_sed);
525#ifdef DEBUG_LP
526 QString pattern = wcharStr2LimaStr(wpattern);
527 QString name = wcharStr2LimaStr(normalizedForm);
528 LDEBUG << "ApproxStringMatcher::buildPattern: (escaping) name "
529 << Lima::Common::Misc::limastring2utf8stdstring(name) << " changed to "
531#endif
532 // apply user supplied regex (generalization)
533 for(RegexMap::const_iterator regexIt = m_regexes.begin() ;
534 regexIt != m_regexes.end() ; regexIt++ )
535 {
536 // get regex as wstring
537 Regex a_regex = *regexIt;
538 wide_regex matching_rule(a_regex.first);
539 std::basic_string<wchar_t> substitution = a_regex.second;
540 wpattern = boost::regex_replace(wpattern, matching_rule,
541 substitution, boost::match_default | boost::format_sed);
542#ifdef DEBUG_LP
543 QString pattern = wcharStr2LimaStr(wpattern);
544 QString name = wcharStr2LimaStr(normalizedForm);
545 LDEBUG << "ApproxStringMatcher::buildPattern: name "
546 << Lima::Common::Misc::limastring2utf8stdstring(name) << " changed to "
548#endif
549 }
550 return wpattern;
551}
552
553void ApproxStringMatcher::matchApproxTokenAndFollowers(
557 std::pair<NameIndex::const_iterator,NameIndex::const_iterator> nameRange,
558 OrderedSolution& result) const
559{
560#ifdef DEBUG_LP
562#endif
563 VertexTokenPropertyMap tokenMap=get(vertex_token,g);
564 // VertexDataPropertyMap dataMap=get(vertex_data,g);
565
566 // Build string (text) where search will be done
567 LimaString text;
568 Token* currentToken=tokenMap[vStart];
569 // TODO: vérifier que vEndIt est le noeud 1 (sans token)
570 for( LinguisticGraphVertex currentVertex = vStart ; currentVertex != vEnd ; ) {
571 currentToken=tokenMap[currentVertex];
572#ifdef DEBUG_LP
573 LDEBUG << "ApproxStringMatcher::matchApproxTokenAndFollowers() from " << currentVertex;
574#endif
575 if (currentToken!=0)
576 {
577 // Add enough space characters to adjust text to beginning of token
578 if (currentToken->position() > (uint)text.length()) {
579 for( int i = currentToken->position() - text.length() ; i > 0 ; i-- )
580 text.append(BLANK_SEPARATOR);
581 }
582 assert( currentToken->length() == (uint64_t)(currentToken->stringForm().length()));
583 text.append(currentToken->stringForm());
584#ifdef DEBUG_LP
585 LDEBUG << "ApproxStringMatcher::matchApproxTokenAndFollowers() text= "
587#endif
588 }
589 // following nodes
590 LinguisticGraphOutEdgeIt outEdge,outEdge_end;
591 boost::tie (outEdge,outEdge_end)=out_edges(currentVertex,g);
592 currentVertex =target(*outEdge,g);
593 }
594
595 // search for names in text
596 for( NameIndex::const_iterator wordIt = nameRange.first ; wordIt != nameRange.second ; wordIt++ ) {
597 // get normalized form of name from lexicon
598 std::basic_string<wchar_t> normalizedForm = (*wordIt).second;
599 // build pattern from name
600 std::basic_string<wchar_t> wpattern = buildPattern(normalizedForm);
601 // compute nb max error for name
602 int nbMaxError = (normalizedForm.length()*m_nbMaxNumError)/m_nbMaxDenError;
603 // Search for pattern in text
604 std::vector<Suggestion> suggestions;
605 int ret = findApproxPattern( wpattern, text, suggestions, nbMaxError);
606 // keep all suggestions
607 if(ret == 0) {
608#ifdef DEBUG_LP
609 LDEBUG << "ApproxStringMatcher::matchApproxTokenAndFollowers(): findApproxPattern()="
610 << ret;
611 for( std::vector<Suggestion>::const_iterator sIt = suggestions.begin() ; sIt != suggestions.end() ; sIt++ ) {
612 LDEBUG << *sIt;
613 }
614#endif
615 // compute which vertices contribute to the match and compute (position,length) in original text
616 for( std::vector<Suggestion>::const_iterator sIt = suggestions.begin() ;
617 sIt != suggestions.end() ; sIt++ ) {
618 Solution tempResult;
619 computeVertexMatches( g, vStart, vEnd, *sIt, tempResult);
620 // TODO: à revoir, peut être qu'il faut recalculer la chaîne à partir du match au moment du calcul de suggestion.startPosition et suggestion.endPosition
621 //tempResult.form = text.mid(sIt->startPosition, sIt->endPosition-sIt->startPosition);
622 // tempResult.length = sIt->endPosition-sIt->startPosition;
623 tempResult.length = tempResult.suggestion.endPosition-tempResult.suggestion.startPosition;
624 tempResult.form = text.mid(tempResult.suggestion.startPosition,
625 tempResult.length);
626 if( (tempResult.suggestion.nb_error <= nbMaxError) ) {
627 tempResult.normalizedForm = wcharStr2LimaStr(normalizedForm);
628 result.insert(tempResult);
629 }
630#ifdef DEBUG_LP
631 LDEBUG << "ApproxStringMatcher::matchApproxTokenAndFollowers: tempResult= " << tempResult;
632#endif
633 }
634 }
635 }
636}
637
638
639void ApproxStringMatcher::computeVertexMatches(
640 const LinguisticGraph& g,
641 const LinguisticGraphVertex vStart,
642 const LinguisticGraphVertex vEnd,
643 const Suggestion& suggestion, Solution& tempResult) const
644{
645#ifdef DEBUG_LP
647#endif
648 Token* currentToken=get(vertex_token,g,vStart);
649
650 tempResult.suggestion.nb_error = suggestion.nb_error;
651 tempResult.vertices=std::deque<LinguisticGraphVertex>();
652 bool pushVertex=false;
653 for( LinguisticGraphVertex currentVertex = vStart ; currentVertex != vEnd ; ) {
654 currentToken=get(vertex_token,g,currentVertex);
655 if (currentToken!=0)
656 {
657 int startTok = currentToken->position();
658 int endTok = currentToken->position() + currentToken->length();
659#ifdef DEBUG_LP
660 LDEBUG << "ApproxStringMatcher::computeVertexMatches() compare with (start,end)="
661 << "(" << startTok
662 << "," << endTok << "}";
663#endif
664 // Search for token where matches begins
665 if( tempResult.vertices.size() == 0 ) {
666 if( (suggestion.startPosition <= startTok )
667 || ( (suggestion.startPosition >= startTok) && ( suggestion.startPosition < endTok) ) ) {
668 pushVertex=true;
669 if(suggestion.startPosition > startTok) {
670 tempResult.suggestion.nb_error += (suggestion.startPosition - startTok);
671#ifdef DEBUG_LP
672 LDEBUG << "ApproxStringMatcher::computeVertexMatches: error +="
673 << suggestion.startPosition - startTok;
674#endif
675 }
676 }
677 }
678 if( pushVertex ) {
679#ifdef DEBUG_LP
680 LDEBUG << "ApproxStringMatcher::computeVertexMatches: push "
681 << currentVertex;
682#endif
683 tempResult.vertices.push_back(currentVertex);
684 if( ( suggestion.endPosition >= startTok ) && ( suggestion.endPosition <= endTok) ) {
685 if(suggestion.endPosition < endTok) {
686 tempResult.suggestion.nb_error += (endTok-suggestion.endPosition);
687#ifdef DEBUG_LP
688 LDEBUG << "ApproxStringMatcher::computeVertexMatches: error +="
689 << endTok-suggestion.endPosition;
690#endif
691 }
692 break;
693 }
694 }
695 }
696 // following nodes
697 LinguisticGraphOutEdgeIt outEdge,outEdge_end;
698 boost::tie (outEdge,outEdge_end)=out_edges(currentVertex,g);
699 currentVertex =target(*outEdge,g);
700 }
701 Token* firstToken=get(vertex_token,g,tempResult.vertices.front());
702 tempResult.suggestion.startPosition = firstToken->position();
703 Token* lastToken=get(vertex_token,g,tempResult.vertices.back());
704 tempResult.suggestion.endPosition = lastToken->position()+lastToken->length();
705}
706
707int ApproxStringMatcher::findApproxPattern(
708 const std::basic_string<wchar_t>& pattern, LimaString text,
709 std::vector<Suggestion>& suggestions, int nbMaxError) const {
710 int returnStatus=1;
712 QString patternQ = wcharStr2LimaStr(pattern);
713#ifdef DEBUG_LP
714 LDEBUG << "ApproxStringMatcher::findApproxPattern("
716 << Lima::Common::Misc::limastring2utf8stdstring(text) << "), nbErr=" << nbMaxError;
717#endif
718
719 // pattern buffer structure (result of compilation)
720 regex_t preg;
721 int cflags = REG_EXTENDED|REG_ICASE|REG_NEWLINE;
722 // cflags |= REG_NOSUB;
723 // cflags |= REG_LITERAL;
724 // cflags |= REG_RIGHT_ASSOC;
725 // cflags |= REG_UNGREEDY;
726
727 // Compile pattern
728 /*
729#ifdef DEBUG_LP
730 LDEBUG << "ApproxStringMatcher::findApproxPattern: compilation...";
731#endif
732 */
733 int agrepStatus = regwncomp(&preg, pattern.c_str(), pattern.length(), cflags);
734 // TODO: agrepStatus???
735#ifdef DEBUG_LP
736 LDEBUG << "ApproxStringMatcher::findApproxPattern: agrepStatus=" << agrepStatus;
737#endif
738 if(agrepStatus!= 0) {
739 QString patternQ = wcharStr2LimaStr(pattern);
740 LWARN << "ApproxStringMatcher::findApproxPattern: error when compiling ="
742 return returnStatus;
743 }
744
745 regaparams_t params = {
746 1, // int cost_ins;
747 1, // int cost_del;
748 1, // int cost_subst;
749 nbMaxError, // int max_cost;
750 nbMaxError, // int max_ins;
751 nbMaxError, // int max_del;
752 nbMaxError, // int max_subst;
753 nbMaxError, // int max_err;
754 };
755 int eflags = REG_NOTBOL;
756 // eflags |= REGNOTEOL;
757 // pmatch has more than 1 emplacement to be filled with match of subexpression (parenthese expr. in pattern)
758 const size_t MAX_MATCH=10;
759 regmatch_t pmatch[MAX_MATCH];
760 regamatch_t amatch = {
761 MAX_MATCH, // size_t nmatch;
762 pmatch, // regmatch_t *pmatch
763 0, // int cost;
764 0, // int num_ins;
765 0, // int num_del;
766 0, //int num_subst;
767 };
768 wchar_t tarray[text.length()];
769 int tlength = text.toWCharArray(tarray);
770 /*
771#ifdef DEBUG_LP
772 LDEBUG << "ApproxStringMatcher::findApproxPattern: execution...";
773#endif
774 */
775 int offset=0;
776 int execStatus;
777 do {
778 execStatus = regawnexec(&preg, tarray+offset, tlength-offset, &amatch, params, eflags);
779#ifdef DEBUG_LP
780 LDEBUG << "ApproxStringMatcher::findApproxPattern: execStatus=" << execStatus;
781#endif
782 if( execStatus == 0 ) {
783 returnStatus=0;
784 regmatch_t* current_match=&(pmatch[0]);
785 Suggestion suggestion;
786 suggestion.startPosition = current_match->rm_so+offset;
787 suggestion.endPosition = current_match->rm_eo+offset;
788 suggestion.nb_error = amatch.num_del+amatch.num_ins+amatch.num_subst;
789 suggestions.push_back(suggestion);
790 offset += current_match->rm_eo;
791 }
792 } while ( (tlength-offset > 0) && (execStatus == 0) );
793 regfree(&preg);
794 return returnStatus;
795}
796
797
798} // MorphologicAnalysis
799} // LinguisticProcessing
800} // Lima
This file is the main header file for the data related to annotation graphs.
#define APPROX_STRING_MATCHER_CLASSID
#define LWARN
Definition LimaCommon.h:160
#define LDEBUG
Definition LimaCommon.h:157
#define LINFO
Definition LimaCommon.h:158
#define LERROR
Definition LimaCommon.h:161
LinguisticGraph::in_edge_iterator LinguisticGraphInEdgeIt
boost::graph_traits< LinguisticGraph >::edge_descriptor LinguisticGraphEdge
typedefs to simplify the access to various graphs elements
boost::property_map< LinguisticGraph, vertex_token_t >::type VertexTokenPropertyMap
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define MORPHOLOGINIT
Defines a Factory to create Object of type Base.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Holds an annotation graph and gives an API to manipulate it.
void addMatching(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second, AnnotationGraphVertex secondVx)
Adds a symetric matching between two vertices of two graphs identified by the two string parameters.
AnnotationGraphVertex createAnnotationVertex()
Creates a new annotation vertex in the graph.
This class allows to convert any object into an annotation by inheritance.
Holds linguistic data for one language.
const FsaStringsPool & stringsPool(MediaId med) const
const MediaData & mediaData(MediaId media) const
EntityType getEntityType(const LimaString &entityName) const
entity types manager
EntityGroupId getEntityGroupId(const LimaString &groupName) const
std::deque< std::string > & getListsValueAtKey(const std::string &key)
std::map< std::string, std::string > & getMapAtKey(const std::string &key)
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Use this exception to signal an error in one of the configuration files.
Definition LimaCommon.h:345
The main LIMA exception class.
Definition LimaCommon.h:262
void setDefaultKey(const Lima::LimaString &defaultKey)
Definition TStatus.cpp:325
void setStatus(const TStatus &status)
Set the TStatus of a token.
Definition Token.h:100
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
static std::basic_string< wchar_t > LimaStr2wcharStr(const QString &limastr)
Definition of a function suitable to be used as a dumper for specific entities annotations of an anno...
A representation of a specific entity to store in the annotation graph.
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
static MediaticData & changeable()
singleton accessor
Definition Singleton.h:71
This file contains a class to control log of informations about time, such as logging cumulated time ...
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
void annotate(AnnotationGraphVertex v, const LimaString &annot, uint64_t value)
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
LimaString utf8stdstring2limastring(const std::string &src)
struct Lima::LinguisticProcessing::MorphologicAnalysis::_Suggestion Suggestion
SimpleFactory< MediaProcessUnit, ApproxStringMatcher > ApproxStringMatcherFactory(APPROX_STRING_MATCHER_CLASSID)
struct Lima::LinguisticProcessing::MorphologicAnalysis::_Solution Solution
std::ostream & operator<<(ostream &os, const Suggestion &suggestion)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
@ INVALID_CONFIGURATION
Definition LimaCommon.h:242
QString LimaString
Definition LimaString.h:33
STL namespace.
bool operator()(const Solution &s1, const Solution &s2) const
launch exception related to the configuration file parsing