39namespace MorphologicAnalysis
50 <<
" nbErr=" << suggestion.
nb_error << std::endl;
65 std::deque<LinguisticGraphVertex>::const_iterator it = solution.
vertices.begin();
66 if( it != solution.
vertices.end() ) {
69 for( ; it != solution.
vertices.end() ; it++ ) {
74 <<
"length=" << solution.
length
83 std::deque<LinguisticGraphVertex>::const_iterator it = solution.
vertices.begin();
84 if( it != solution.
vertices.end() ) {
87 for( ; it != solution.
vertices.end() ; it++ ) {
92 <<
"length=" << solution.
length;
143 Common::MediaticData::EntityGroupId foundGroup;
152 LERROR <<
"no param 'entityGoup' in ApproxStringMatcher group for language " << (int) m_language;
163 QTextStream qts(&errorString);
164 qts << __FILE__ <<
", line" << __LINE__ <<
"Unknown entity type" << lsentityName;
172 QTextStream qts(&errorString);
173 qts << __FILE__ <<
", line" << __LINE__
174 <<
"no param 'entityGoup' in ApproxStringMatcher group for language "
187 LERROR <<
"no param 'NPCategory' in ApproxStringMatcher group for language " << (int) m_language;
198 LERROR <<
"no param 'NPMicroCategory' in ApproxStringMatcher group for language " << (int) m_language;
203 std::string nameindexId;
210 LERROR <<
"no param 'nameindex' in ApproxStringMatcher group for language " << (int) m_language;
214 m_nameIndex = std::dynamic_pointer_cast<NameIndexResource>(res);
237 std::istringstream iss(nbMaxErrorStr);
238 iss >> m_nbMaxNumError;
242 LERROR <<
"no param 'nbMaxNumError' in ApproxStringMatcher group for language " << (int) m_language;
248 std::istringstream iss(nbMaxErrorStr);
249 iss >> m_nbMaxDenError;
253 LERROR <<
"no param 'nbMaxDenError' in ApproxStringMatcher group for language " << (int) m_language;
260 std::map <std::string, std::string >& regexes = unitConfiguration.
getMapAtKey(
"generalizationRules");
261 std::deque <std::string >& regexesOrder = unitConfiguration.
getListsValueAtKey(
"generalizationRulesOrder");
262 for (std::deque <std::string >::const_iterator it = regexesOrder.begin(); it != regexesOrder.end(); it++)
264 const std::string& key = *it;
265 QString patternQ = QString::fromUtf8 (key.c_str());
267 const std::string& value=regexes.at(key);
268 QString substitutionQ = QString::fromUtf8 (value.c_str());
270 m_regexes.push_back( std::pair<std::basic_string<wchar_t>,std::basic_string<wchar_t> >(patternWS,
276 LERROR <<
"no map 'generalization' in RegexReplacer group configuration fot language "
280 catch (std::out_of_range& )
282 LERROR <<
"no value for some key in 'generalizationRules' in RegexReplacer group configuration fot language "
294 LINFO <<
"starting process ApproxStringMatcher";
296 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.
getData(
"LinguisticMetaData"));
297 std::basic_string<wchar_t> indexName;
300 std::string indexNametds = metadata->getMetaData(
"index");
301 QString indexNamels = QString::fromUtf8 (indexNametds.c_str());
309 QString name = wcharStr2LimaStr(indexName);
310 LDEBUG <<
"ApproxStringMatcher::process: index from metadata= "
314 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"AnalysisGraph"));
316 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.
getData(
"AnnotationData"));
317 if (annotationData==0)
319 LINFO <<
"ApproxStringMatcher::process no annotation data, creating and populating it";
320 annotationData = std::make_shared<AnnotationData>();
321 analysis.
setData(
"AnnotationData",annotationData);
323 anagraph->populateAnnotationGraph(annotationData.get(),
"AnalysisGraph");
326 if (annotationData==0)
328 LERROR <<
"ApproxStringMatcher::process: no AnnotationData, cannot store result";
332 if (annotationData->dumpFunction(
"SpecificEntity") == 0)
338 OrderedSolution solutions;
342 std::pair<NameIndex::const_iterator,NameIndex::const_iterator> nameRange;
343 if( (indexName.length() == 0) || !(m_nameIndex->withIndex()) ) {
344 nameRange.first = m_nameIndex->begin();
345 nameRange.second = m_nameIndex->end();
348 nameRange = m_nameIndex->equal_range(indexName);
353 const std::basic_string<wchar_t> firstNameW = (*(nameRange.first)).second;
354 std::basic_string<wchar_t> lastNameW;
355 for( NameIndex::const_iterator it= nameRange.first ; it != nameRange.second ; it++ )
356 lastNameW = (*it).second;
357 const QString firstNameQ = wcharStr2LimaStr(firstNameW);
358 const QString lastNameQ = wcharStr2LimaStr(lastNameW);
359 LDEBUG <<
"ApproxStringMatcher::process: nameRange= from " << firstNameQ <<
" to " << lastNameQ;
362 matchApproxTokenAndFollowers(g, anagraph->firstVertex(), anagraph->lastVertex(), nameRange, solutions);
364 LDEBUG <<
"ApproxStringMatcher::process: solution.size()=" << solutions.size();
366 for( OrderedSolution::const_iterator sIt = solutions.begin() ; sIt != solutions.end() ; sIt++ ) {
369 LDEBUG <<
"ApproxStringMatcher::process: check " << *sIt;
372 bool outOfGraph=
false;
374 std::deque<LinguisticGraphVertex>::const_iterator vIt2 = solution.
vertices.begin();
375 if( vIt2 != solution.
vertices.end() ) {
379 for( ; vIt2 != solution.
vertices.end() ; vIt2++ ) {
383 for( std::deque<LinguisticGraphVertex>::const_iterator vIt = solution.
vertices.begin();
384 vIt != solution.
vertices.end() ; vIt++ )
387 LDEBUG <<
"ApproxStringMatcher::process: outOfGraph = " << outOfGraph <<
", check vertex " << *vIt;
391 boost::tie (outEdge,outEdge_end)=out_edges(*vIt,g);
392 if( outEdge == outEdge_end ) {
396 boost::tie (inEdge,inEdge_end)=in_edges(*vIt,g);
397 if( inEdge == inEdge_end ) {
405 createVertex(g, anagraph->firstVertex(), anagraph->lastVertex(), solution, annotationData.get() );
411 LDEBUG <<
"ending process ApproxStringMatcher";
416void ApproxStringMatcher::createVertex(
426 LDEBUG <<
"ApproxStringMatcher::createVertex( solution=" << solution <<
")";
436 for( ; previousVertex != vEnd ; ) {
437 boost::tie (outEdge,outEdge_end)=out_edges(previousVertex,g);
439 if( nextVertex == solution.
vertices.front() ) {
443 previousVertex = nextVertex;
447 boost::tie (outEdge,outEdge_end)=out_edges(solution.
vertices.back(),g);
450 boost::remove_edge(previousVertex,solution.
vertices.front(), g);
451 boost::remove_edge(solution.
vertices.back(),nextVertex, g);
455 boost::tie(e, success) = add_edge(previousVertex, newVertex, g);
456 boost::tie(e, success) = add_edge(newVertex, nextVertex, g);
459 StringsPoolIndex form = (*m_sp)[solution.
form];
484 newMorphoSyntacticData->push_back(elem);
486 put(
vertex_data,g,newVertex,newMorphoSyntacticData);
491 annotationData->
addMatching(
"AnalysisGraph", newVertex,
"annot", agv);
508QString ApproxStringMatcher::wcharStr2LimaStr(
const std::basic_string<wchar_t>& wstring)
const {
510 return QString::fromWCharArray(wstring.c_str(),wstring.length());
513std::basic_string<wchar_t> ApproxStringMatcher::buildPattern(
const std::basic_string<wchar_t>& normalizedForm)
const {
519 std::basic_string<wchar_t> wpattern = normalizedForm;
523 const std::basic_string<wchar_t> rep(L
"\\\\&");
524 wpattern = boost::regex_replace(wpattern, esc, rep, boost::match_default | boost::format_sed);
526 QString pattern = wcharStr2LimaStr(wpattern);
527 QString name = wcharStr2LimaStr(normalizedForm);
528 LDEBUG <<
"ApproxStringMatcher::buildPattern: (escaping) name "
533 for(RegexMap::const_iterator regexIt = m_regexes.begin() ;
534 regexIt != m_regexes.end() ; regexIt++ )
537 Regex a_regex = *regexIt;
539 std::basic_string<wchar_t> substitution = a_regex.second;
540 wpattern = boost::regex_replace(wpattern, matching_rule,
541 substitution, boost::match_default | boost::format_sed);
543 QString pattern = wcharStr2LimaStr(wpattern);
544 QString name = wcharStr2LimaStr(normalizedForm);
545 LDEBUG <<
"ApproxStringMatcher::buildPattern: name "
553void ApproxStringMatcher::matchApproxTokenAndFollowers(
557 std::pair<NameIndex::const_iterator,NameIndex::const_iterator> nameRange,
558 OrderedSolution& result)
const
568 Token* currentToken=tokenMap[vStart];
571 currentToken=tokenMap[currentVertex];
573 LDEBUG <<
"ApproxStringMatcher::matchApproxTokenAndFollowers() from " << currentVertex;
578 if (currentToken->
position() > (uint)text.length()) {
579 for(
int i = currentToken->
position() - text.length() ; i > 0 ; i-- )
582 assert( currentToken->
length() == (uint64_t)(currentToken->
stringForm().length()));
585 LDEBUG <<
"ApproxStringMatcher::matchApproxTokenAndFollowers() text= "
591 boost::tie (outEdge,outEdge_end)=out_edges(currentVertex,g);
592 currentVertex =target(*outEdge,g);
596 for( NameIndex::const_iterator wordIt = nameRange.first ; wordIt != nameRange.second ; wordIt++ ) {
598 std::basic_string<wchar_t> normalizedForm = (*wordIt).second;
600 std::basic_string<wchar_t> wpattern = buildPattern(normalizedForm);
602 int nbMaxError = (normalizedForm.length()*m_nbMaxNumError)/m_nbMaxDenError;
604 std::vector<Suggestion> suggestions;
605 int ret = findApproxPattern( wpattern, text, suggestions, nbMaxError);
609 LDEBUG <<
"ApproxStringMatcher::matchApproxTokenAndFollowers(): findApproxPattern()="
611 for( std::vector<Suggestion>::const_iterator sIt = suggestions.begin() ; sIt != suggestions.end() ; sIt++ ) {
616 for( std::vector<Suggestion>::const_iterator sIt = suggestions.begin() ;
617 sIt != suggestions.end() ; sIt++ ) {
619 computeVertexMatches( g, vStart, vEnd, *sIt, tempResult);
623 tempResult.length = tempResult.suggestion.endPosition-tempResult.suggestion.startPosition;
624 tempResult.form = text.mid(tempResult.suggestion.startPosition,
626 if( (tempResult.suggestion.nb_error <= nbMaxError) ) {
627 tempResult.normalizedForm = wcharStr2LimaStr(normalizedForm);
628 result.insert(tempResult);
631 LDEBUG <<
"ApproxStringMatcher::matchApproxTokenAndFollowers: tempResult= " << tempResult;
639void ApproxStringMatcher::computeVertexMatches(
650 tempResult.suggestion.nb_error = suggestion.nb_error;
651 tempResult.vertices=std::deque<LinguisticGraphVertex>();
652 bool pushVertex=
false;
657 int startTok = currentToken->
position();
660 LDEBUG <<
"ApproxStringMatcher::computeVertexMatches() compare with (start,end)="
662 <<
"," << endTok <<
"}";
665 if( tempResult.vertices.size() == 0 ) {
666 if( (suggestion.startPosition <= startTok )
667 || ( (suggestion.startPosition >= startTok) && ( suggestion.startPosition < endTok) ) ) {
669 if(suggestion.startPosition > startTok) {
670 tempResult.suggestion.nb_error += (suggestion.startPosition - startTok);
672 LDEBUG <<
"ApproxStringMatcher::computeVertexMatches: error +="
673 << suggestion.startPosition - startTok;
680 LDEBUG <<
"ApproxStringMatcher::computeVertexMatches: push "
683 tempResult.vertices.push_back(currentVertex);
684 if( ( suggestion.endPosition >= startTok ) && ( suggestion.endPosition <= endTok) ) {
685 if(suggestion.endPosition < endTok) {
686 tempResult.suggestion.nb_error += (endTok-suggestion.endPosition);
688 LDEBUG <<
"ApproxStringMatcher::computeVertexMatches: error +="
689 << endTok-suggestion.endPosition;
698 boost::tie (outEdge,outEdge_end)=out_edges(currentVertex,g);
699 currentVertex =target(*outEdge,g);
702 tempResult.suggestion.startPosition = firstToken->
position();
704 tempResult.suggestion.endPosition = lastToken->
position()+lastToken->
length();
707int ApproxStringMatcher::findApproxPattern(
708 const std::basic_string<wchar_t>& pattern,
LimaString text,
709 std::vector<Suggestion>& suggestions,
int nbMaxError)
const {
712 QString patternQ = wcharStr2LimaStr(pattern);
714 LDEBUG <<
"ApproxStringMatcher::findApproxPattern("
721 int cflags = REG_EXTENDED|REG_ICASE|REG_NEWLINE;
733 int agrepStatus = regwncomp(&preg, pattern.c_str(), pattern.length(), cflags);
736 LDEBUG <<
"ApproxStringMatcher::findApproxPattern: agrepStatus=" << agrepStatus;
738 if(agrepStatus!= 0) {
739 QString patternQ = wcharStr2LimaStr(pattern);
740 LWARN <<
"ApproxStringMatcher::findApproxPattern: error when compiling ="
745 regaparams_t params = {
755 int eflags = REG_NOTBOL;
758 const size_t MAX_MATCH=10;
759 regmatch_t pmatch[MAX_MATCH];
760 regamatch_t amatch = {
768 wchar_t tarray[text.length()];
769 int tlength = text.toWCharArray(tarray);
778 execStatus = regawnexec(&preg, tarray+offset, tlength-offset, &amatch, params, eflags);
780 LDEBUG <<
"ApproxStringMatcher::findApproxPattern: execStatus=" << execStatus;
782 if( execStatus == 0 ) {
784 regmatch_t* current_match=&(pmatch[0]);
787 suggestion.endPosition = current_match->rm_eo+offset;
788 suggestion.nb_error = amatch.num_del+amatch.num_ins+amatch.num_subst;
789 suggestions.push_back(suggestion);
790 offset += current_match->rm_eo;
792 }
while ( (tlength-offset > 0) && (execStatus == 0) );