LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
applyRecognizer.cpp
Go to the documentation of this file.
1// Copyright 2002-2020 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/************************************************************************
7 *
8 * @file applyRecognizer.cpp
9 * @author besancon (besanconr@zoe.cea.fr)
10 * @date Fri Jan 14 2005
11 * @version $Id$
12 * copyright Copyright (C) 2005-2020 by CEA LIST
13 *
14 ***********************************************************************/
15
16#include "applyRecognizer.h"
18
27
28using namespace Lima::Common::AnnotationGraphs;
31using namespace std;
32
33namespace Lima {
34namespace LinguisticProcessing {
35namespace ApplyRecognizer {
36
38
41m_recognizers(0),
42m_useSentenceBounds(false),
43m_sentenceBoundsData("SentenceBoundaries"),
44m_updateGraph(false),
45m_resolveOverlappingEntities(false),
46m_overlappingEntitiesStrategy(IGNORE_SMALLEST),
47m_testAllVertices(false),
48m_stopAtFirstSuccess(true),
49m_onlyOneSuccessPerType(false),
50m_graphId("PosGraph"),
51m_dataForStorage()
52{
53}
54
58
61 Manager* manager)
62
63{
65 MediaId language=manager->getInitializationParameters().media;
66 try {
67 // try to get a single automaton
68 string automaton=unitConfiguration.getParamsValueAtKey("automaton");
69 auto res = LinguisticResources::single().getResource(language,automaton);
70 m_recognizers.push_back(std::dynamic_pointer_cast<Recognizer>(res));
71 }
73 try {
74 // try to get a list of automatons
75 const deque<string>& automatonList=unitConfiguration.getListsValueAtKey("automatonList");
76 for (auto automaton: automatonList)
77 {
78 auto res = LinguisticResources::single().getResource(language, automaton);
79 m_recognizers.push_back(std::dynamic_pointer_cast<Recognizer>(res));
80 }
81 }
83 LERROR << "No 'automaton' or 'automatonList' in ApplyRecognizer group for language "
84 << (int)language << " !";
86 }
87 }
88
89 try {
90 m_useSentenceBounds=
91 getBooleanParameter(unitConfiguration,"useSentenceBounds");
92 }
94 // optional parameter: keep default value
95 }
96
97 try
98 {
99 string sentenceBoundsData=unitConfiguration.getParamsValueAtKey("sentenceBoundsData");
100 if (! sentenceBoundsData.empty()) {
101 m_sentenceBoundsData=sentenceBoundsData;
102 }
103 }
105 {
106 // optional parameter: keep default value
107 }
108
109 try {
110 m_updateGraph=
111 getBooleanParameter(unitConfiguration,"updateGraph");
112 }
114 // optional parameter: keep default value
115 }
116
117 try {
118 m_resolveOverlappingEntities=
119 getBooleanParameter(unitConfiguration,"resolveOverlappingEntities");
120 }
122 // optional parameter: keep default value
123 }
124
125 try {
126 string overlappingEntitiesStrategy=
127 unitConfiguration.getParamsValueAtKey("overlappingEntitiesStrategy");
128 if (overlappingEntitiesStrategy=="IgnoreSmallest") {
129 m_overlappingEntitiesStrategy=IGNORE_SMALLEST;
130 }
131 else if (overlappingEntitiesStrategy=="IgnoreFirst") {
132 m_overlappingEntitiesStrategy=IGNORE_FIRST;
133 }
134 else if (overlappingEntitiesStrategy=="IgnoreSecond") {
135 m_overlappingEntitiesStrategy=IGNORE_SECOND;
136 }
137 }
139 // optional parameter: keep default value
140 }
141
142 try {
143 m_testAllVertices=
144 getBooleanParameter(unitConfiguration,"testAllVertices");
145 }
147 // optional parameter: keep default value
148 }
149
150 try {
151 m_stopAtFirstSuccess=
152 getBooleanParameter(unitConfiguration,"stopAtFirstSuccess");
153 }
155 // optional parameter: keep default value
156 }
157
158
159 try {
160 m_onlyOneSuccessPerType=
161 getBooleanParameter(unitConfiguration,"onlyOneSuccessPerType");
162 }
164 // optional parameter: keep default value
165 }
166
167 try
168 {
169 m_graphId=unitConfiguration.getParamsValueAtKey("applyOnGraph");
170 }
172 {
173 // optional parameter: keep default value
174 }
175
176 try {
177 m_dataForStorage=unitConfiguration.getParamsValueAtKey("storeInData");
178 }
180 // optional parameter: keep default value
181 }
182
183}
184
185bool ApplyRecognizer::
186getBooleanParameter(Common::XMLConfigurationFiles::GroupConfigurationStructure& unitConfiguration,
187 const std::string& param) const {
188 string value=unitConfiguration.getParamsValueAtKey(param);
189 if (value == "yes" ||
190 value == "true" ||
191 value == "1") {
192 return true;
193 }
194 return false;
195}
196
198{
199 Lima::TimeUtilsController timer("ApplyRecognizer");
200 if (m_recognizers.empty()) {
202 LDEBUG << "ApplyRecognizer: No recognizer to apply";
203 return MISSING_DATA;
204 }
205#ifdef DEBUG_LP
207 LINFO << "start process";
208 LDEBUG << " parameters are:";
209 LDEBUG << " - useSentenceBounds :" << m_useSentenceBounds;
210 LDEBUG << " - updateGraph :" << m_updateGraph;
211 LDEBUG << " - resolveOverlappingEntities :" << m_resolveOverlappingEntities;
212 LDEBUG << " - overlappingEntitiesStrategy :" << m_overlappingEntitiesStrategy;
213 LDEBUG << " - testAllVertices :" << m_testAllVertices;
214 LDEBUG << " - stopAtFirstSuccess :" << m_stopAtFirstSuccess;
215 LDEBUG << " - onlyOneSuccessPerType :" << m_onlyOneSuccessPerType;
216 LDEBUG << " - graphId :" << m_graphId;
217 LDEBUG << " - dataForStorage :" << m_dataForStorage;
218#endif
219
220 LimaStatusCode returnCode(SUCCESS_ID);
221
222 auto recoData = std::dynamic_pointer_cast<RecognizerData>(analysis.getData("RecognizerData"));
223 if (recoData == 0)
224 {
225 recoData = std::make_shared<RecognizerData>();
226 analysis.setData("RecognizerData", recoData);
227 }
228
229 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.getData("AnnotationData"));
230 if (annotationData==0)
231 {
232 annotationData = std::make_shared<AnnotationData>();
233 if (std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph")) != 0)
234 {
235 std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"))->populateAnnotationGraph(
236 annotationData.get(), "AnalysisGraph");
237 }
238 analysis.setData("AnnotationData",annotationData);
239 }
240
241 // data to possibly store the result (according to the actions)
242 // assume all recognizers use the same entity type group (gloups)
243 RecognizerResultData* resultData=
244 new RecognizerResultData(m_graphId);
245 recoData->setResultData(resultData);
246
247 if (m_useSentenceBounds) {
248 for (auto reco: m_recognizers) {
249 returnCode = processOnEachSentence(analysis, reco.get(), recoData.get());
250 }
251 }
252 else {
253 for (auto reco: m_recognizers) {
254 returnCode = processOnWholeText(analysis, reco.get(), recoData.get());
255 }
256 }
257
258 if (m_updateGraph) {
259// LDEBUG << "";
260 recoData->removeVertices(analysis);
261 recoData->clearVerticesToRemove();
262 recoData->removeEdges(analysis);
263 recoData->clearEdgesToRemove();
264 }
265
266 if (! m_dataForStorage.empty()) {
267 analysis.setData(m_dataForStorage,resultData);
268 }
269 else {
270 // result data stored in recoData and resultData are same pointer
271 recoData->deleteResultData();
272 resultData=0;
273 }
274
275 // remove recognizer data (used only internally to this process unit)
276 analysis.removeData("RecognizerData");
277
278 return returnCode;
279}
280
281LimaStatusCode ApplyRecognizer::
282processOnEachSentence(AnalysisContent& analysis,
283 Recognizer* reco,
284 RecognizerData* recoData) const
285{
287
288 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData(recoData->getGraphId()));
289 if (nullptr==anagraph)
290 {
291 LERROR << "graph with id '"<< recoData->getGraphId() <<"' is not available";
292 return MISSING_DATA;
293 }
294
295 // get sentence bounds
296 auto sb=std::dynamic_pointer_cast<SegmentationData>(analysis.getData(m_sentenceBoundsData));
297 if (nullptr==sb)
298 {
299 LERROR << "no sentence bounds "<< m_sentenceBoundsData << " defined ! abort";
300 return MISSING_DATA;
301 }
302
303 std::vector<RecognizerMatch> seRecognizerResult;
304 // SegmentationData::const_iterator boundItr=sb->begin();
305 std::vector<Segment>::const_iterator boundItr=(sb->getSegments()).begin();
306 // ??OME2 while (boundItr!=sb->end())
307 while (boundItr!=(sb->getSegments()).end())
308 {
309 LinguisticGraphVertex beginSentence=boundItr->getFirstVertex();
310 LinguisticGraphVertex endSentence=boundItr->getLastVertex();
311 //LDEBUG << "ApplyRecognizer: analyze sentence from vertex " << beginSentence << " to vertex " << endSentence << QTENDL;
312
313 seRecognizerResult.clear();
314 reco->apply(*anagraph,beginSentence,
315 endSentence,analysis,seRecognizerResult);
316
317 //remove overlapping entities
318 if (m_resolveOverlappingEntities)
319 {
320 reco->resolveOverlappingEntities(seRecognizerResult,
321 m_overlappingEntitiesStrategy);
322 }
323
324 boundItr++;
325 recoData->nextSentence();
326 }
327
328 return SUCCESS_ID;
329}
330
331LimaStatusCode ApplyRecognizer::
332processOnWholeText(AnalysisContent& analysis,
333 Recognizer* reco,
334 RecognizerData* recoData ) const
335{
336 // APPRLOGINIT;
337 // LDEBUG << "apply recognizer on whole text";
338
339 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData(recoData->getGraphId()));
340 if (nullptr == anagraph)
341 {
343 LERROR << "graph with id '"<< recoData->getGraphId() <<"' is not available";
344 return MISSING_DATA;
345 }
346
347// auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"));
348
349 std::vector<RecognizerMatch> seRecognizerResult;
350
351 reco->apply(*anagraph,
352 anagraph->firstVertex(),
353 anagraph->lastVertex(),
354 analysis,seRecognizerResult,
355 m_testAllVertices,m_stopAtFirstSuccess,m_onlyOneSuccessPerType);
356
357 //remove overlapping entities
358 if (m_resolveOverlappingEntities)
359 {
360 reco->resolveOverlappingEntities(seRecognizerResult,
361 m_overlappingEntitiesStrategy);
362 }
363
364 return SUCCESS_ID;
365}
366
367} // end namespace
368} // end namespace
369} // end namespace
This file is the main header file for the data related to annotation graphs.
#define LDEBUG
Definition LimaCommon.h:157
#define LINFO
Definition LimaCommon.h:158
#define LERROR
Definition LimaCommon.h:161
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
Defines a Factory to create Object of type Base.
#define APPLYRECOGNIZER_CLASSID
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
void removeData(const QString &id)
remove the analysisData with the given id
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
std::deque< std::string > & getListsValueAtKey(const std::string &key)
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Use this exception to signal an error in one of the configuration files.
Definition LimaCommon.h:345
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
␈rief a class for the definition of a complete recognizer
Definition recognizer.h:78
uint64_t apply(const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &begin, const LinguisticGraphVertex &end, AnalysisContent &analysis, std::vector< RecognizerMatch > &result, bool testAllVertices=false, bool stopAtFirstSuccess=true, bool onlyOneSuccessPerType=false, bool returnAtFirstSuccess=false, bool applySameRuleWhileSuccess=false) const
apply the recognizer on a graph
uint64_t resolveOverlappingEntities(std::vector< RecognizerMatch > &listEntities, const OverlapResolutionStrategy &strategy=DEFAULT_OVERLAP_STRATEGY) const
resolve the problem of overlapping entities in the list of entities : when two entities are overlapin...
static const LinguisticResources & single()
const singleton accessor
Definition Singleton.h:51
This file contains a class to control log of informations about time, such as logging cumulated time ...
@ IGNORE_SECOND
the second entity is ignored => assumes the leftmost trigger is more important
Definition recognizer.h:46
@ IGNORE_SMALLEST
the smallest entity is ignored (the one that covers the smallest number of words is assumed to be les...
Definition recognizer.h:48
@ IGNORE_FIRST
the first entity is ignored => assumes the rightmost trigger is more important
Definition recognizer.h:44
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
@ MISSING_DATA
Definition LimaCommon.h:243
STL namespace.
#define APPRLOGINIT
launch exception related to the configuration file parsing