LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
recognizer.cpp
Go to the documentation of this file.
1// Copyright 2002-2020 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/************************************************************************
7*
8* File : recognizer.cpp
9* Author : Romaric Besancon (besanconr@zoe.cea.fr)
10* Created on : Tue Oct 15 2002
11* Copyright : (c) 2002 by CEA
12*
13************************************************************************/
14
15#include "recognizer.h"
16
18#include "automatonCommon.h"
19#include "transitionUnit.h"
20#include "recognizerData.h"
27#include <iostream>
28#include <fstream>
29#include <sstream>
30#include <string>
31#include <vector>
32#include <queue>
33#include <map>
34#include <stack>
35
36using namespace std;
39
40namespace Lima {
41namespace LinguisticProcessing {
42namespace Automaton {
43
44// a comparison operator on Rule pointer:
45// to sort SetOfRules on decreasing rule weights
47public:
48 bool operator()(Rule* r1,Rule* r2) const {
49 return (r1->getWeight() > r2->getWeight());
50 }
51};
52
53
54// a comparison operator on TriggerRule
56public:
58 const Recognizer::TriggerRule* r2) {
59 return (r1->setOfRules().front()->getWeight() >
60 r2->setOfRules().front()->getWeight());
61 }
62};
63
64
67
68//**********************************************************************
69// constructors
70//**********************************************************************
73 m_rules(0),
74 m_ruleStorage(0),
75 m_language(),
76 m_automatonControlParams(),
77 m_filename(),
78 m_searchStructure()
79{ }
80
81// copy is complex because of the pointers
84{
85#ifdef LDEBUG
87 LDEBUG << "Recognizer::Recognizer copy constructor";
88#endif
89 init();
90 copy(r);
91
92 // have to initialize the search structure of the new recognizer
94}
95
96//**********************************************************************
97// destructor
98//**********************************************************************
104
105//**********************************************************************
106// copy
107//**********************************************************************
109{
110#ifdef LDEBUG
111 AULOGINIT;
112 LDEBUG << "Recognizer::Recognizer operator=";
113#endif
114 if (this != &r)
115 {
116 freeMem();
117 init();
118 copy(r);
119 }
120
121 // do not copy the search structure : recompute it the new recognizer
122 // (not sure the copy is less complex than recomputing it)
124
125 return (*this);
126}
127
130 Manager* manager)
131
132{
133#ifdef LDEBUG
134 AULOGINIT;
135 LDEBUG << "Recognizer::init" << (void*)this;
136#endif
151 m_language=manager->getInitializationParameters().language;
153 try
154 {
155 QString rulesFile = unitConfiguration.getParamsValueAtKey("rules").c_str();
156 if (!rulesFile.isEmpty())
157 {
158 m_filename=rulesFile.toUtf8().constData();
159 rulesFile = Common::Misc::findFileInPaths(resourcesPath.c_str(), rulesFile);
160// LDEBUG << "read recognizer from file : " << rulesFile;
161 //readFromFile(rulesFile);
162 AutomatonReader reader;
163 reader.readRecognizer(rulesFile.toUtf8().constData(),*this);
164 }
165 }
167 AULOGINIT;
168 LERROR << "No param 'rules' in recognizer group for language " << (int)m_language;
169 throw InvalidConfiguration();
170 }
171
172 try
173 {
174 string str=unitConfiguration.getParamsValueAtKey("maxDepthStack");
175 uint64_t val=atol(str.c_str());
176 if (val==0) {
177 AULOGINIT;
178 LWARN << "maxDepthStack is 0: keep default value";
179 }
180 else {
182 }
183 }
185 // keep default value
186 }
187
188 try
189 {
190 string str=unitConfiguration.getParamsValueAtKey("maxTransitionsExplored");
191 uint64_t val=atol(str.c_str());
192 if (val==0) {
193 AULOGINIT;
194 LWARN << "maxTransitionsExplored is 0: keep default value";
195 }
196 else {
198 }
199 }
201 // keep default value
202 }
203
204 try
205 {
206 string str=unitConfiguration.getParamsValueAtKey("maxNbResults");
207 uint64_t val=atol(str.c_str());
208 if (val==0) {
209 AULOGINIT;
210 LWARN << "maxNbResults is 0: keep default value";
211 }
212 else {
214 }
215 }
217 // keep default value
218 }
219
220 try
221 {
222 string str=unitConfiguration.getParamsValueAtKey("maxResultSize");
223 uint64_t val=atol(str.c_str());
224 if (val==0) {
225 AULOGINIT;
226 LWARN << "maxResultSize is 0: keep default value";
227 }
228 else {
230 }
231 }
233 // keep default value
234 }
235
237}
238
239//**********************************************************************
240// helper functions for constructors and destructors
241//**********************************************************************
249
251{
252 map<Rule*,Rule*> pointersMap;
253
254 for (uint64_t i(0); i<r.m_ruleStorage.size(); i++)
255 {
256 Rule *rule = new Rule(*(r.m_ruleStorage[i]));
257 m_ruleStorage.push_back(rule);
258 pointersMap[r.m_ruleStorage[i]]=rule;
259 }
260
261 for (uint64_t i(0); i<r.m_rules.size(); i++)
262 {
263 TransitionUnit* t = r.m_rules[i].first->clone();
264 m_rules.push_back(TriggerRule(t,SetOfRules(0)));
265 for (uint64_t j(0); j<r.m_rules[i].second.size(); j++)
266 {
267 m_rules[i].second.push_back(pointersMap[r.m_rules[i].second[j]]);
268 }
269 }
270 pointersMap.clear();
271
274}
275
277{
278 for (uint64_t i(0); i<m_ruleStorage.size(); i++)
279 {
280 delete m_ruleStorage[i];
281 m_ruleStorage[i]=0;
282 }
283 m_ruleStorage.clear();
284 for (uint64_t i(0); i<m_rules.size(); i++)
285 {
286 delete m_rules[i].first;
287 m_rules[i].first=0;
288 m_rules[i].second.clear();
289 }
290 m_rules.clear();
291}
292
293//**********************************************************************
294// test the rules corresponding to one trigger on a text,
295// at a given position
296// stop at the first rule recognized
297//**********************************************************************
298/* TODO: tobe deleted?
299bool Recognizer::parse(const LinguisticAnalysisStructure::Token&,
300 const LinguisticAnalysisStructure::AnalysisGraph& graph,
301 const LinguisticGraphVertex& current,
302 uint64_t offset,
303 AnalysisContent& analysis,
304 RecognizerMatch& result) const {
305
306 vector<RecognizerMatch> results;
307 if (testSetOfRules(*(m_rules[offset].first),
308 m_rules[offset].second,
309 graph,
310 current,
311 graph.firstVertex(),
312 graph.lastVertex(),
313 analysis,
314 results))
315 {
316 result=results.front(); // only one result because stopAtFirstSuccess=true
317 return true;
318 }
319 return false;
320}
321*/
322
323//**********************************************************************
324// test a set of rules for a trigger
326 const SetOfRules& rules,
328 const LinguisticGraphVertex& position,
329 const LinguisticGraphVertex& begin,
330 const LinguisticGraphVertex& end,
331 AnalysisContent& analysis,
332 vector<RecognizerMatch>& matches,
333 std::set<Common::MediaticData::EntityType>* forbiddenTypes,
334 bool stopAtFirstSuccess,
335 bool onlyOneSuccessPerType,
336 bool applySameRuleWhileSuccess) const {
337 AULOGINIT;
338 // If the trigger is defined with a gazeteer, we must check the case of multi-term elements in the gazeteer
339 const GazeteerTransition* gazeteerTrigger = dynamic_cast<const GazeteerTransition*>(&trigger);
340 RecognizerMatch triggermatch(&graph);
341 LinguisticGraphVertex right=position;
342 if( gazeteerTrigger != 0 ) {
343 Token* token = get(vertex_token, *(graph.getGraph()), position);
344 MorphoSyntacticData* data = get(vertex_data, *(graph.getGraph()), position);
345 deque<LinguisticGraphVertex> vertices;
346 ForwardSearch searchGraph;
347 bool match = gazeteerTrigger->matchPath(graph, position, end, &searchGraph, analysis, token, vertices, data);
348 if( match ) {
349 for( std::deque<LinguisticGraphVertex>::const_iterator vIt = vertices.begin(); vIt != vertices.end() ; vIt++ ) {
350 triggermatch.addBackVertex(*vIt,trigger.keep(),"trigger");
351 right=*vIt;
352 }
353 if (trigger.isHead()) {
354 // if gazetteer trigger is head: assign head to the first word
355 triggermatch.setHead(position);
356 }
357 }
358 }
359 else {
360 triggermatch.addBackVertex(position,trigger.keep(),"trigger");
361 if (trigger.isHead()) {
362 triggermatch.setHead(position);
363 }
364 right=position;
365 }
366
367 RecognizerMatch leftmatch(&graph);
368 RecognizerMatch rightmatch(&graph);
369
370 if (onlyOneSuccessPerType && forbiddenTypes==0) {
371 LERROR << "Recognizer::testSetOfRules: cannot use onlyOneSuccessPerType "
372 << "when forbidden types are not allowed";
373 onlyOneSuccessPerType=false;
374 }
375
376
377 uint64_t nbSuccess(0);
378
379 // left context is same LinguisticAnalysisStructure::AnalysisGraph as current (current is in fact
380 // between the current token and the previous one)
381 LinguisticGraphVertex left=position;
382
383#ifdef DEBUG_LP
384 LDEBUG << "Recognizer::testSetOfRules: testing set of rules triggered by " << trigger << " on vertex " << position;
385 LDEBUG << "onlyOneSuccessPerType=" << onlyOneSuccessPerType;
386 if (logger.isDebugEnabled()) {
387 std::ostringstream oss;
388 for (const auto& rule: rules) {
389 oss << " - " << rule->getWeight();
390 }
391 LDEBUG << "Rule weights" << oss.str();
392 }
393#endif
394
395 bool reapplySameRule(false);
396
397 for (auto rule = rules.cbegin(); rule != rules.cend(); rule++) {
398 const auto& currentRule = *rule;
399#ifdef DEBUG_LP
400 if (logger.isDebugEnabled()) {
401 LDEBUG << "Recognizer::testSetOfRules: testing rule "<<*currentRule << ","
402 << currentRule->getRuleId() <<" of type "
403 << currentRule->getType() << ",reapply="
404 << reapplySameRule << " from " << position;
405 }
406#endif
407
408 if (forbiddenTypes && forbiddenTypes->find(currentRule->getType()) != forbiddenTypes->end()) {
409 // type previously forbidden by a negative rule
410/* LDEBUG << "type " << currentRule->getType()
411 << " is forbidden: continue";*/
412 continue;
413 }
414
415 // initializes the constraint checklist
416 ConstraintCheckList constraintCheckList(currentRule->numberOfConstraints(),
418
419 // treat the constraints for the trigger with the constraint
420 // checklist corresponding to this rule
421
422 // what to do with constraints on gazeteer triggers ?
423 // -> apply them on each vertex : useful for constraints which are actually actions
424 // such as AppendEntityFeature, maybe dangerous for constraints that are actual tests
425 // but suppose gazeteer triggers are not used in this case (mostly in syntactic analysis)
426
427 // if (!trigger.checkConstraints(graph,position,analysis,
428 // constraintCheckList)) {
429 bool constraintsVerified = true;
430 for (const auto& elt: triggermatch) {
431 if (!trigger.checkConstraints(graph, elt.getVertex(),analysis, constraintCheckList)) {
432 // one unary constraint was not verified
433 constraintsVerified=false;
434 break;
435 }
436 }
437 if (! constraintsVerified) {
438 // apply actions (for actions triggered by failure)
439 if (!currentRule->negative()) {
440 currentRule->executeActions(graph,
441 analysis,
442 constraintCheckList,
443 false,
444 0); // match is not used
445 }
446 continue;
447 }
448
449
450 leftmatch.reinit();
451 rightmatch.reinit();
452 ForwardSearch forward;
453 BackwardSearch backward;
454 bool success = currentRule->test(graph, left, right,
455 begin, end, analysis,
456 leftmatch, rightmatch,
457 constraintCheckList,forward,backward,
459 //LDEBUG << "success=" << success;
460
461 std::unique_ptr<RecognizerMatch> match=nullptr;
462
463 if (success) {
464 // build complete match
465
466 match=std::make_unique<RecognizerMatch>(leftmatch);
467 if (leftmatch.getHead() != 0) {
468 match->setHead(leftmatch.getHead());
469 }
470
471 // TODO: add node of gazeteerTrigger
472 //match->addBackVertex(position,trigger.keep(), "trigger");
473 /*
474 RecognizerMatch::const_iterator triggerMatchIt = triggermatch.begin();
475 for( ; triggerMatchIt != triggermatch.end(); triggerMatchIt++) {
476 match->addBackVertex(*triggerMatchIt,trigger.keep(), "trigger");
477 }
478 */
479 match->addBack(triggermatch);
480 // check if trigger is head
481 if (triggermatch.getHead() != 0) {
482 match->setHead(triggermatch.getHead());
483 }
484
485 match->addBack(rightmatch);
486 // remove elements not kept at begin and end of the expression
487 match->removeUnkeptAtExtremity();
488
489 match->setType(currentRule->getType());
490 match->setLinguisticProperties(currentRule->getLinguisticProperties());
491 match->setContextual(currentRule->contextual());
492 setNormalizedForm(currentRule->getNormalizedForm(),*match);
493 }
494
495 // execute possible actions associated to the rule iff current rule is
496 // positive
497 //LDEBUG << "Recognizer: executing actions: ";
498 bool actionSuccess = true;
499 if (!currentRule->negative()) {
500 actionSuccess = currentRule->executeActions(graph, analysis,
501 constraintCheckList,
502 success,
503 match.get());
504 //LDEBUG << "actionSuccess=" << actionSuccess;
505 }
506
507#ifdef DEBUG_LP
508 if (logger.isDebugEnabled()) {
509 LinguisticGraphVertex v=position;
510 LimaString str("");
511 Token* token=get(vertex_token,*(graph.getGraph()),position);
512 if (token!=0) {
513 str = token->stringForm();
514 }
515 if (success) {
516 LDEBUG << "Recognizer::testSetOfRules: trigger " << v << "[" << str << "]:rule "
517 << currentRule->getRuleId() << "-> success=" << success
518 << ",actionSuccess=" << actionSuccess;
519 LDEBUG << " matched:" << match->getNormalizedString(Common::MediaticData::MediaticData::single().stringsPool(m_language));
520 }
521 else {
522 LDEBUG << "Recognizer::testSetOfRules: vertex " << v << "[" << str << "]:rule "
523 << currentRule->getRuleId() << "-> success= false";
524 }
525 }
526#endif
527
528 if (success && actionSuccess) {
529 if (forbiddenTypes && currentRule->negative()) {
530 forbiddenTypes->insert(currentRule->getType());
531 success = false;
532 continue;
533 }
534 LINFO << "Recognizer::testSetOfRules: execute rule " << currentRule->getRuleId()
535 << " of type "<< currentRule->getType()
536 << "(" << Lima::Common::MediaticData::MediaticData::single().getEntityName(currentRule->getType())
537 << ") on vertex " << position;
538 RecognizerData* recoData = static_cast<RecognizerData*>(analysis.getData("RecognizerData").get());
539 if (stopAtFirstSuccess||(recoData != nullptr && !recoData->getNextVertices().empty())) {
540 matches.push_back(*match);
541#ifdef DEBUG_LP
542 if (logger.isDebugEnabled() && recoData != nullptr) {
543 LDEBUG << "Recognizer::testSetOfRules: Returning from testSetOfRules cause stopAtFirstSuccess ("
544 << stopAtFirstSuccess << ") or next vertices empty ("
545 << (recoData->getNextVertices().empty())
546 << ")";
547 }
548#endif
549 return 1;
550 }
551 else {
552 if (applySameRuleWhileSuccess) {
553 if (reapplySameRule) {
554 if (*match==matches.back()) {
555// AULOGINIT;
556// LDEBUG << "Reapplication of same rule gives same result: "
557// << "abort to avoid inifinite loop: "
558// << *match << ";" << matches.back();
559 reapplySameRule=false;
560 continue;
561 }
562/* else {
563 LDEBUG << "Reapplication of same rule gives new result";
564 }*/
565 }
566 // reapply same rule
567 rule--;
568 reapplySameRule=true;
569 }
570
571// LDEBUG << "add match to results " << *match;
572 matches.push_back(*match);
573
574 if (onlyOneSuccessPerType) {
575/* LDEBUG << "add " << currentRule->getType()
576 << " in forbiddenTypes";*/
577 forbiddenTypes->insert(currentRule->getType());
578 }
579 nbSuccess++;
580 }
581 }
582 else {
583// LDEBUG << "-> no success";
584 reapplySameRule=false;
585 }
586 }
587
588 return nbSuccess;
589}
590
591//**********************************************************************
592// normalization function
593//**********************************************************************
595setNormalizedForm(const LimaString& norm,
596 RecognizerMatch& match) const
597{
598 match.features().clear();
599
601 if (norm.isEmpty()) {
602 // use surface form of the expression as normalized form
604 }
605 else {
607 }
608}
609
610//**********************************************************************
611// main functions that applies the recognizer on a graph
612//**********************************************************************
613
614// Apply between two nodes and search between the same ones
617 const LinguisticGraphVertex& begin,
618 const LinguisticGraphVertex& end,
619 AnalysisContent& analysis,
620 std::vector<RecognizerMatch>& result,
621 bool testAllVertices,
622 bool stopAtFirstSuccess,
623 bool onlyOneSuccessPerType,
624 bool returnAtFirstSuccess,
625 bool applySameRuleWhileSuccess) const
626{
627 return apply(graph,
628 begin,
629 end,
630 begin,
631 end,
632 analysis,
633 result,
634 testAllVertices,
635 stopAtFirstSuccess,
636 onlyOneSuccessPerType,
637 returnAtFirstSuccess,
638 applySameRuleWhileSuccess);
639}
640
641// Apply between two nodes and search between two others.
642// precondition [begin, end] included in [upstreamBound,downstreamBound]
645 const LinguisticGraphVertex& begin,
646 const LinguisticGraphVertex& end,
647 const LinguisticGraphVertex& upstreamBound,
648 const LinguisticGraphVertex& downstreamBound,
649 AnalysisContent& analysis,
650 std::vector<RecognizerMatch>& result,
651 bool testAllVertices,
652 bool stopAtFirstSuccess,
653 bool onlyOneSuccessPerType,
654 bool returnAtFirstSuccess,
655 bool applySameRuleWhileSuccess) const
656{
657 if (begin == end)
658 {
659 return 0;
660 }
661
662 if (returnAtFirstSuccess) {
663 stopAtFirstSuccess=true; // implied by the other
664 }
665
666#ifdef DEBUG_LP
667 AULOGINIT;
668 LDEBUG << "apply recognizer " << m_filename << " from vertex "
669 << begin << " to vertex " << end;
670 LDEBUG << " up bound: " << upstreamBound << "; down bound: " << downstreamBound << "; testAllVertices: " << testAllVertices;
671 LDEBUG << " stopAtFirstSuccess: " << stopAtFirstSuccess << "; onlyOneSuccessPerType: " << onlyOneSuccessPerType;
672 LDEBUG << " returnAtFirstSuccess: " << returnAtFirstSuccess << "; applySameRuleWhileSuccess: " << applySameRuleWhileSuccess;
673#endif
674
675 uint64_t numberOfRecognized(0);
676 bool success(false);
677
678 // use deque instead of queue to be able to clear()
679 std::deque<LinguisticGraphVertex> toVisit;
680 std::set<LinguisticGraphVertex> visited;
681
682 toVisit.push_back(begin);
683 // patch for inifinite loop : avoid begin stopped at first step
684 //visited.insert(begin);
685
686 //match may include the last vertex: store following vertices to check on those
687 set<LinguisticGraphVertex> afterTheEnd;
688 if (downstreamBound!=graph.lastVertex()) {
689 auto [outEdge,outEdge_end]=out_edges(downstreamBound,*(graph.getGraph()));
690 for (; outEdge!=outEdge_end; outEdge++) {
691 LinguisticGraphVertex next=target(*outEdge,*(graph.getGraph()));
692 afterTheEnd.insert(next);
693 }
694 }
695
696 bool lastReached = false;
697 while (!toVisit.empty())
698 {
699 auto currentVertex = toVisit.front();
700 toVisit.pop_front();
701 // patch for inifinite loop : check if we already seen this node
702 if (visited.find(currentVertex) != visited.end())
703 {
704 continue;
705 }
706
707 visited.insert(currentVertex);
708#ifdef DEBUG_LP
709 LDEBUG << "to visit size=" << toVisit.size() << " ; currentVertex=" << currentVertex;
710#endif
711
712 if (lastReached || // limit given by argument
713 currentVertex == graph.lastVertex()) { // end of the graph
714 // LDEBUG << "vertex " << currentVertex << " is last vertex";
715 continue; // may be other nodes to test in queue
716 }
717 if (currentVertex == end ) { // limit given by argument
718 lastReached = true;
719 }
720
721 if (currentVertex != graph.firstVertex() && currentVertex != begin) {
722#ifdef DEBUG_LP
723 LDEBUG << "Recognizer: test on vertex " << currentVertex;
724#endif
725 success = testOnVertex(graph,currentVertex,
726 upstreamBound,downstreamBound,
727 analysis,result,
728 stopAtFirstSuccess,
729 onlyOneSuccessPerType,
730 applySameRuleWhileSuccess);
731 if (success) {
732 numberOfRecognized++;
733 if (returnAtFirstSuccess)
734 return numberOfRecognized;
735 if (! testAllVertices) { // restart from end of recognized expression
736 // with useSentenceBounds, may be the case that the last vertex in the sentence/segment
737 // is in the recognized expression: currentVertex may be outside the scope of the sentence
738 // how to test that ? => check if is a vertex following the downstreamBound
739 if (currentVertex==end || afterTheEnd.find(currentVertex)!=afterTheEnd.end()) {
740#ifdef DEBUG_LP
741 LDEBUG << "success: reached the end, stop";
742#endif
743 break;
744 }
745#ifdef DEBUG_LP
746 LDEBUG << "success: continue from vertex " << currentVertex;
747#endif
748 // GC on 20110803: the clearing below was problematic in case of rules like that:
749 // [<Location.LOCATION>]:(t_capital_1st|t_capital){1-3} [,]::LOCATION:N_LOCATION
750 // which matches text before (left) the trigger which is not included in the match.
751 // thus the next vertex explored was the newly created one ; the vertex following
752 // it is already visited (this is in this case the comma) and the content of
753 // toVisit (the vertex after the trigger) was removed. Thus the search stopped after
754 // the new vertex.
755 // Warning: what is the inpact on the use of the testAllVertices parameter ? And is there
756 // any other side effect ?
757// toVisit.clear();
758
759 }
760 }
761 }
762
763 // store following nodes to test
764 auto [outEdge,outEdge_end] = out_edges(currentVertex,*(graph.getGraph()));
765
766 for (; outEdge!=outEdge_end; outEdge++) {
767 auto next=target(*outEdge,*(graph.getGraph()));
768 if (visited.find(next)==visited.end()) {
769#ifdef DEBUG_LP
770 LDEBUG << "Recognizer: adding out edge target vertex to the 'to visit' list: " << next;
771#endif
772 toVisit.push_back(next);
773 // do not put in visited unless it is really visited
774 // (otherwise, may be suppressed when testAllVertices is false
775 // and never visited)
776 //visited.insert(next);
777 }
778 else {
779#ifdef DEBUG_LP
780 LDEBUG << "Recognizer: already visited:" << next;
781#endif
782 }
783 }
784 auto recoData = std::dynamic_pointer_cast<RecognizerData>(analysis.getData("RecognizerData"));
785 if (nullptr != recoData)
786 {
787 auto& nextVertices = recoData->getNextVertices();
788 if (!nextVertices.empty())
789 {
790#ifdef DEBUG_LP
791 LDEBUG << "Recognizer: adding next vertices to the 'to visit' list";
792#endif
793 for (auto nvit = nextVertices.begin(), nvit_end = nextVertices.end(); nvit != nvit_end; nvit++)
794 {
795#ifdef DEBUG_LP
796 LDEBUG << " - " << *nvit;
797#endif
798 toVisit.push_front(*nvit);
799 }
800 nextVertices.clear();
801 }
802 }
803#ifdef DEBUG_LP
804 LDEBUG << "Recognizer: 'to visit' list size is now: " << toVisit.size();
805#endif
806 }
807 return numberOfRecognized;
808}
809
810
811//**********************************************************************
812// test the recognizer on a vertex : test
813//**********************************************************************
816 LinguisticGraphVertex& current,
817 const LinguisticGraphVertex& begin,
818 const LinguisticGraphVertex& end,
819 AnalysisContent& analysis,
820 std::vector<RecognizerMatch>& result,
821 bool stopAtFirstSuccess,
822 bool onlyOneSuccessPerType,
823 bool applySameRuleWhileSuccess) const
824{
825 //AULOGINIT;
826 auto token = get(vertex_token, *(graph.getGraph()), current);
827 auto data = get(vertex_data, *(graph.getGraph()), current);
828
829 if (token==0) {
830 AULOGINIT;
831 LERROR << "no token for vertex " << current;
832 return 0;
833 }
834
835 if (data==0) {
836 AULOGINIT;
837 LERROR << "no data for vertex " << current;
838 return 0;
839 }
840
841 // TODO replace TriggerRule* raw pointer by a shared pointer
842 vector<TriggerRule*> matchingRules;
843 set<Common::MediaticData::EntityType> forbiddenTypes;
844 uint64_t nbSuccess=0;
845
846 findNextSetOfRules(graph, current, analysis, token, data, matchingRules);
847
848 if (! matchingRules.empty()) {
849 for (auto ruleSet=matchingRules.begin(), ruleSet_end=matchingRules.end(); ruleSet!=ruleSet_end; ruleSet++) {
850 auto nbSuccessForTheseRules=
851 testSetOfRules(*((*ruleSet)->transitionUnit()),
852 (*ruleSet)->setOfRules(),
853 graph, current, begin, end, analysis,
854 result, &forbiddenTypes,
855 stopAtFirstSuccess,
856 onlyOneSuccessPerType,
857 applySameRuleWhileSuccess);
858 if (nbSuccessForTheseRules>0) {
859 nbSuccess+=nbSuccessForTheseRules;
860 // skip recognized part (if the end of the recognized part is after
861 // current token)
862 auto& lastSuccess=result.back();
863 auto t=get(vertex_token,*(graph.getGraph()),current);
864 auto currentTokenEnd=t->position()+t->length();
865 auto recoData = std::dynamic_pointer_cast<RecognizerData>(analysis.getData("RecognizerData"));
866 if (stopAtFirstSuccess||(recoData != 0 && !recoData->getNextVertices().empty())) {
867 if (lastSuccess.positionEnd() >= currentTokenEnd) {
868 current=lastSuccess.getEnd();
869 }
870 break;
871 }
872 }
873 }
874 for(auto rule: matchingRules) {
875 if (rule!=nullptr) {
876 delete rule;
877 }
878 }
879 }
880 forbiddenTypes.clear();
881
882 // LDEBUG << "testOnVertex nb successes: " << nbSuccess;
883 return nbSuccess;
884}
885
886//**********************************************************************
887//resolve the problem of overlapping entities in the list of entities :
888// when two entities are overlaping, only one is kept
889//**********************************************************************
891resolveOverlappingEntities(std::vector<RecognizerMatch>& listEntities,
892 const OverlapResolutionStrategy& strategy) const
893{
894 typedef std::vector<RecognizerMatch>::iterator vectorRecognizerMatchIterator;
895
896 uint64_t numberOfOverlappingEntities(0);
897
898 if (listEntities.empty()) {
899 return numberOfOverlappingEntities;
900 }
901
902 switch (strategy) {
903 case IGNORE_FIRST: {
904 vectorRecognizerMatchIterator currentEntity(listEntities.begin());
905 vectorRecognizerMatchIterator nextEntity(currentEntity);
906 nextEntity++;
907 while (nextEntity != listEntities.end()) {
908 if (currentEntity->isOverlapping(*nextEntity)) {
909 numberOfOverlappingEntities++;
910 currentEntity=listEntities.erase(currentEntity);
911 nextEntity=currentEntity;
912 nextEntity++;
913 }
914 else {
915 currentEntity++;
916 nextEntity++;
917 }
918 }
919 break;
920 }
921 case IGNORE_SECOND: {
922 vectorRecognizerMatchIterator currentEntity(listEntities.begin());
923 vectorRecognizerMatchIterator previousEntity(currentEntity);
924 currentEntity++;
925 while (currentEntity != listEntities.end()) {
926 if (currentEntity->isOverlapping(*previousEntity)) {
927 numberOfOverlappingEntities++;
928 currentEntity=listEntities.erase(currentEntity);
929 }
930 else {
931 previousEntity++;
932 currentEntity++;
933 }
934 }
935 break;
936 }
937 case IGNORE_SMALLEST: {
938 vectorRecognizerMatchIterator currentEntity(listEntities.begin());
939 vectorRecognizerMatchIterator previousEntity(currentEntity);
940 currentEntity++;
941 while (currentEntity != listEntities.end()) {
942 if (currentEntity->isOverlapping(*previousEntity)) {
943 numberOfOverlappingEntities++;
944 if (currentEntity->numberOfElements()
945 < previousEntity->numberOfElements()) { // keep previous entity
946 currentEntity=listEntities.erase(currentEntity);
947 }
948 else { // keep current entity
949 previousEntity=listEntities.erase(previousEntity);
950 currentEntity=previousEntity;
951 currentEntity++;
952 }
953 }
954 else {
955 previousEntity++;
956 currentEntity++;
957 }
958 }
959 break;
960 }
961 default:
962 break;
963 }
964
965 return numberOfOverlappingEntities;
966}
967
968//**********************************************************************
969// find the set of rules in the recognizer that accept
970// a particular token as trigger
971//**********************************************************************
974 LinguisticGraphVertex& vertex,
975 AnalysisContent& analysis,
978 std::vector<TriggerRule*>& matchingSetOfRules) const
979{
980 matchingSetOfRules.clear();
981
982 // find matching rules
983 std::vector<const TriggerRule*> matchingRules;
984 m_searchStructure.findMatchingTransitions(graph,vertex,analysis,token,data,matchingRules);
985
986 // matching rules are gathered by common trigger (transition unit)
987 // we have to re-sort the rules by their weight at a global level, independently of the trigger
988 // create a vector of TriggerRule where each contains only one rule, then sort it
989 for (auto rule: matchingRules) {
990 for (auto r: rule->setOfRules()) {
991 matchingSetOfRules.push_back(new TriggerRule(rule->transitionUnit(), SetOfRules(1, r)));
992 }
993 }
994 sort(matchingSetOfRules.begin(),matchingSetOfRules.end(),CompareTriggerRule());
995
996 // then, gather rules with the same trigger that are consecutive in this new list
997 // (may save some constraint checking on trigger)
998 if (! matchingSetOfRules.empty()) {
999 auto it=matchingSetOfRules.begin();
1000 auto currentTrigger=(*it)->transitionUnit();
1001 auto next=it;
1002 next++;
1003 while (next!=matchingSetOfRules.end()) {
1004 if ((*next)->transitionUnit() == currentTrigger) {
1005 (*it)->second.push_back((*next)->setOfRules().front());
1006 delete *next;
1007 next=matchingSetOfRules.erase(next);
1008 }
1009 else {
1010 it++;
1011 currentTrigger=(*it)->transitionUnit();
1012 next++;
1013 }
1014 }
1015 }
1016}
1017
1019 auto macro = &(static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager().getPropertyAccessor("MACRO"));
1020 auto micro = &(static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager().getPropertyAccessor("MICRO"));
1021#ifdef LDEBUG
1022 AULOGINIT;
1023 LDEBUG << "Recognizer::initializeSearchStructure" << (void*)macro << (void*)micro;
1024#endif
1025 m_searchStructure.init(m_rules, macro, micro);
1026}
1027
1031
1032//**********************************************************************
1033// adding a rule
1034//**********************************************************************
1036{
1037 // add the rule in the storage
1038 m_ruleStorage.push_back(rule);
1039 // return the index of the rule in the storage
1040 return (m_ruleStorage.size() - 1);
1041}
1042
1043uint64_t Recognizer::addRule(TransitionUnit* trigger, Rule* rule)
1044{
1045 uint64_t indexRule=addRuleInStorage(rule);
1046
1047 // find if the trigger already exists in the set of triggers
1048 for (uint64_t i(0); i<m_rules.size(); i++)
1049 {
1050 if (*(m_rules[i].first) == *trigger)
1051 {
1052 m_rules[i].second.push_back(rule);
1053 return indexRule;
1054 }
1055 }
1056 m_rules.push_back(TriggerRule(trigger->clone(),
1057 SetOfRules(1,rule)));
1058
1059 return indexRule;
1060}
1061
1063 const uint64_t index)
1064{
1065 // find if the trigger already exists in the set of triggers
1066 for (uint64_t i(0); i<m_rules.size(); i++)
1067 {
1068 if (*(m_rules[i].first) == *trigger)
1069 {
1070 m_rules[i].second.push_back(m_ruleStorage[index]);
1071 return;
1072 }
1073 }
1074 m_rules.push_back(TriggerRule(trigger->clone(),
1075 SetOfRules(1,m_ruleStorage[index])));
1076}
1077
1078//**********************************************************************
1079// input/output in a binary format
1080//**********************************************************************
1081// void Recognizer::readFromTextFile(std::string filename) {
1082// RecognizerCompiler::buildRecognizer(*this,filename);
1083// }
1084
1085// simple linear search (called only with write function -> not optimized)
1087{
1088 for (uint64_t i(0); i<m_ruleStorage.size(); i++)
1089 {
1090 if (m_ruleStorage[i] == r)
1091 {
1092 return i;
1093 }
1094 }
1095 return m_ruleStorage.size()+1;
1096}
1097
1098//**********************************************************************
1099// output the list of triggers with their associated index
1100//**********************************************************************
1102{
1103 for (uint64_t i(0); i<m_rules.size(); i++)
1104 {
1105 cout << "<k>" << m_rules[i].first->printValue() << "</k>"
1106 << "<o>" << i << "</o>" << endl;
1107 }
1108}
1109
1110//***************************************************************************
1111// output
1112//***************************************************************************
1113ostream& operator << (ostream& os, const Recognizer& r)
1114{
1115 for (uint64_t i(0); i<r.m_rules.size(); i++)
1116 {
1117 os << "trigger "<< i << " = "
1118 << *(r.m_rules[i].first) << endl;
1119 for (uint64_t j(0); j<r.m_rules[i].second.size(); j++)
1120 {
1121 os << "rule " << j << ":"
1122 << *(r.m_rules[i].second[j]);
1123 }
1124 }
1125 return os;
1126}
1127
1128} // namespace end
1129} // namespace end
1130} // namespace end
#define DEFAULT_ATTRIBUTE
#define LWARN
Definition LimaCommon.h:160
#define UNDEFLANG
Definition LimaCommon.h:252
#define LDEBUG
Definition LimaCommon.h:157
#define LINFO
Definition LimaCommon.h:158
#define LERROR
Definition LimaCommon.h:161
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
#define AULOGINIT
Defines a Factory to create Object of type Base.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
Holds linguistic data for one language.
const FsaStringsPool & stringsPool(MediaId med) const
const std::string & getResourcesPath() const
const MediaData & mediaData(MediaId media) const
LimaString getEntityName(const EntityType &type) const
return a message when a 'param' was not found
Use this exception to signal an error in one of the configuration files.
Definition LimaCommon.h:345
const std::set< LinguisticGraphVertex > & getNextVertices() const
␈rief a class for control parameters for the search using the automata
Definition automaton.h:44
void readRecognizer(const std::string &filename, Recognizer &reco)
void setFeature(const std::string &name, const ValueType &value)
bool matchPath(const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &vertex, const LinguisticGraphVertex &limit, const SearchGraph *searchGraph, AnalysisContent &analysis, const LinguisticAnalysisStructure::Token *token, std::deque< LinguisticGraphVertex > &vertices, const LinguisticAnalysisStructure::MorphoSyntacticData *) const
void addBackVertex(const LinguisticGraphVertex &, bool isKept=true, const LimaString &ruleElementId="")
LimaString getNormalizedString(const FsaStringsPool &sp) const
bool operator()(const Recognizer::TriggerRule *r1, const Recognizer::TriggerRule *r2)
␈rief a class for the definition of a complete recognizer
Definition recognizer.h:78
Recognizer & operator=(const Recognizer &)
void findNextSetOfRules(const LinguisticAnalysisStructure::AnalysisGraph &graph, LinguisticGraphVertex &vertex, AnalysisContent &analysis, const LinguisticAnalysisStructure::Token *token, const LinguisticAnalysisStructure::MorphoSyntacticData *data, std::vector< TriggerRule * > &matchingSetOfRules) const
uint64_t testOnVertex(const LinguisticAnalysisStructure::AnalysisGraph &graph, LinguisticGraphVertex &current, const LinguisticGraphVertex &begin, const LinguisticGraphVertex &end, AnalysisContent &analysis, std::vector< RecognizerMatch > &result, bool stopAtFirstSuccess=true, bool onlyOneSuccessPerType=false, bool applySameRuleWhileSuccess=false) const
test the recognizer on a given vertex : check if this vertex is a trigger and if a rule applies,...
TransitionSearchStructure< TriggerRule > m_searchStructure
Definition recognizer.h:344
uint64_t testSetOfRules(const TransitionUnit &trigger, const SetOfRules &rules, const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &position, const LinguisticGraphVertex &begin, const LinguisticGraphVertex &end, AnalysisContent &analysis, std::vector< RecognizerMatch > &match, std::set< Common::MediaticData::EntityType > *forbiddenTypes=0, bool stopAtFirstSuccess=true, bool onlyOneSuccessPerType=false, bool applySameRuleWhileSuccess=false) const
TODO: toBeDeleted Parse tokens paths from the trigger point.
AutomatonControlParams m_automatonControlParams
parameters to control the search of the automaton
Definition recognizer.h:337
uint64_t addRuleInStorage(Rule *rule)
add a rule in the storage zone
uint64_t apply(const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &begin, const LinguisticGraphVertex &end, AnalysisContent &analysis, std::vector< RecognizerMatch > &result, bool testAllVertices=false, bool stopAtFirstSuccess=true, bool onlyOneSuccessPerType=false, bool returnAtFirstSuccess=false, bool applySameRuleWhileSuccess=false) const
apply the recognizer on a graph
void setNormalizedForm(const LimaString &norm, RecognizerMatch &match) const
uint64_t resolveOverlappingEntities(std::vector< RecognizerMatch > &listEntities, const OverlapResolutionStrategy &strategy=DEFAULT_OVERLAP_STRATEGY) const
resolve the problem of overlapping entities in the list of entities : when two entities are overlapin...
uint64_t addRule(TransitionUnit *trigger, Rule *rule)
add a rule in the recognizer : add the rule in the storage zone, and associate the trigger to the rul...
bool checkConstraints(const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &vertex, AnalysisContent &analysis, ConstraintCheckList &) const
virtual TransitionUnit * clone() const =0
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
static MediaticData & changeable()
singleton accessor
Definition Singleton.h:71
QString findFileInPaths(const QString &paths, const QString &fileName, const QChar &separator)
Find the given file in the given paths.
std::ostream & operator<<(std::ostream &os, const DFFSPos &x)
OverlapResolutionStrategy
an enumerated type to indicate which kind of strategy adopt to deal with two overlapping entities
Definition recognizer.h:43
@ IGNORE_SECOND
the second entity is ignored => assumes the leftmost trigger is more important
Definition recognizer.h:46
@ IGNORE_SMALLEST
the smallest entity is ignored (the one that covers the smallest number of words is assumed to be les...
Definition recognizer.h:48
@ IGNORE_FIRST
the first entity is ignored => assumes the rightmost trigger is more important
Definition recognizer.h:44
std::vector< ConstraintCheckListElement > ConstraintCheckList
SimpleFactory< AbstractResource, Recognizer > recognizerFactory(RECOGNIZER_CLASSID)
recognizer factory
std::vector< Rule * > SetOfRules
the SetOfRules type is defined as a vector of pointers on Rule
Definition recognizer.h:66
NAUTITIA.
QString LimaString
Definition LimaString.h:33
STL namespace.
PUGI__FN void sort(I begin, I end, const Pred &pred)
Definition pugixml.cpp:7549
#define RECOGNIZER_CLASSID
Definition recognizer.h:68
launch exception related to the configuration file parsing