LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
SpecificEntitiesRecognizer.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/***************************************************************************
7 * Copyright (C) 2004-2012 by CEA LIST *
8 * *
9 ***************************************************************************/
10
14#include "common/misc/traceUtils.h"
21#include "linguisticProcessing/core/LinguisticAnalysisStructure/SentenceBounds.h"
24
25using namespace std;
26using namespace Lima::Common::AnnotationGraphs;
29
30namespace Lima
31{
32
33namespace LinguisticProcessing
34{
35
36namespace SpecificEntities
37{
38
40
42LinguisticProcessUnit(),
43m_recognizer(0),
44m_useSentenceBounds(true),
45m_sentenceBoundsData("SentenceBounds"),
46m_useDicoWords(false)
47{}
48
49
52
55 Manager* manager)
56
57{
59 MediaId language=manager->getInitializationParameters().language;
60 try
61 {
62 string automaton=unitConfiguration.getParamsValueAtKey("automaton");
63 auto res = LinguisticResources::single().getResource(language,automaton);
64 m_recognizer=static_cast<Automaton::Recognizer*>(res);
65 }
67 {
68 LERROR << "No param 'automaton' in SpecificEntitiesRecognizer group for language "
69 << (int)language << " !";
71 }
72
73 try {
74 string useDicoWords=unitConfiguration.getParamsValueAtKey("useDicoWords");
75 if (useDicoWords=="yes" ||
76 useDicoWords=="true" ||
77 useDicoWords=="1") {
78 m_useDicoWords=true;
79 }
80 else {
81 m_useDicoWords=false;
82 }
83 }
85 // optional parameter: keep default value
86 }
87
88 try
89 {
90 string useSentenceBounds=unitConfiguration.getParamsValueAtKey("useSentenceBounds");
91 if (useSentenceBounds=="yes" ||
92 useSentenceBounds=="true" ||
93 useSentenceBounds=="1") {
94 m_useSentenceBounds=true;
95 }
96 else {
97 m_useSentenceBounds=false;
98 }
99 }
101 {
102 // optional parameter: keep default value
103 }
104
105 try
106 {
107 string sentenceBoundsData=unitConfiguration.getParamsValueAtKey("sentenceBoundsData");
108 if (! sentenceBoundsData.empty()) {
109 m_sentenceBoundsData=sentenceBoundsData;
110 }
111 }
113 {
114 // optional parameter: keep default value
115 }
116
117
118 try
119 {
120 m_graph=unitConfiguration.getParamsValueAtKey("graph");
121 }
123 {
124 LWARN << "No 'graph' parameter in unit configuration '"
125 << unitConfiguration.getName() << "' ; using PosGraph";
126 m_graph=string("PosGraph");
127 }
128
129}
130
132 AnalysisContent& analysis) const
133{
135 SELOGINIT;
136 LINFO << "start process";
137
138 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.getData("AnnotationData"));
139 if (annotationData==0)
140 {
141 annotationData=new AnnotationData();
142 if (std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph")) != 0)
143 {
144 std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"))->populateAnnotationGraph(annotationData, "AnalysisGraph");
145 }
146 if (std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("PosGraph")) != 0)
147 {
148 std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("PosGraph"))->populateAnnotationGraph(annotationData, "PosGraph");
149 }
150
151 analysis.setData("AnnotationData",annotationData);
152 }
153 if (m_annotationData->dumpFunction("SpecificEntity") == 0)
154 {
155 m_annotationData->dumpFunction("SpecificEntity", new DumpSpecificEntityAnnotation());
156 }
157// if (analysis.getData("SyntacticData")==0)
158// {
159// SyntacticAnalysis::SyntacticData* syntacticData=new SyntacticAnalysis::SyntacticData(std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData(m_graph)),0);
160// syntacticData->setupDependencyGraph();
161// analysis.setData("SyntacticData",syntacticData);
162// }
163
164 AnalysisGraphId* gid(new AnalysisGraphId(m_graph));
165 analysis.setData("GraphId",gid);
166
167// SpecificEntityFound* seFound(new SpecificEntityFound(false));
168// analysis.setData("SEFound",seFound);
169
170 LimaStatusCode returnCode;
171
172 // process text
173// if (m_useSentenceBounds) {
174// returnCode=processOnEachSentence(analysis);
175// }
176// else {
177// returnCode=processOnWholeText(analysis);
178// }
179
180 // initialize the vertices to clear
181 std::set<LinguisticGraphVertex> verticesToRemove;
182
183 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData(m_graph));
184
185 LinguisticGraph* graph=anagraph->getGraph();
186 std::queue<LinguisticGraphVertex> toVisit;
187 VertexTokenPropertyMap tokenMap=get(vertex_token,*graph);
188 std::set<LinguisticGraphVertex> visited;
189
190 try
191 {
192 toVisit.push(anagraph->firstVertex());
193 visited.insert(anagraph->firstVertex());
194
195 while (!toVisit.empty())
196 {
197 std::set<LinguisticGraphVertex> addedVertices;
198 LinguisticGraphVertex currentVertex=toVisit.front();
199 toVisit.pop();
200
201 Token* currentToken=tokenMap[currentVertex];
202 if (currentToken)
203 {
204 // process current Token
205 bool specificEntityFound(false);
206 RecognizerMatch se(anagraph);
207 specificEntityFound= findSEFromRecognizer(currentVertex,
208 anagraph,
209 analysis,
210 se);
211
212// if (specificEntityFound) {
213// LDEBUG << "found specific entity for vertex "
214// << currentVertex << "("
215// << currentToken->stringForm() << ")";
216/* updateGraph(anagraph->getGraph(),se,verticesToRemove,
217 addedVertices,annotationData);*/
218// }
219// else {
220// LDEBUG << "no specific entity found for vertex "
221// << currentVertex << "("
222// << currentToken->stringForm() << ")";
223// }
224 }
225 std::set<LinguisticGraphVertex>::const_iterator addedIt, addedIt_end;
226 addedIt = addedVertices.begin(); addedIt_end = addedVertices.end();
227 for (; addedIt != addedIt_end; addedIt++)
228 {
229 if (visited.find(*addedIt) == visited.end())
230 {
231 toVisit.push(*addedIt);
232 visited.insert(*addedIt);
233 }
234 }
235
236 // go one step forward on the new path
237 LinguisticGraphAdjacencyIt adjItr,adjItrEnd;
238 boost::tie(adjItr,adjItrEnd) = adjacent_vertices(currentVertex,*graph);
239 for (;adjItr!=adjItrEnd;adjItr++)
240 {
241 if (addedVertices.find(*adjItr)==addedVertices.end() &&
242 visited.find(*adjItr)==visited.end())
243 {
244 toVisit.push(*adjItr);
245 visited.insert(*adjItr);
246 }
247 }
248 }
249
250 // at end, remove vertices
251// removeVertices(,*graph);
252 }
253 catch (std::exception &exc)
254 {
255 SELOGINIT;
256 LWARN << "Exception in Specific Entities : " << exc.what();
257// throw;
258 return UNKNOWN_ERROR;
259 }
260
261 return SUCCESS_ID;
262
263 TimeUtils::logElapsedTime("SpecificEntitiesRecognizer");
264 return returnCode;
265}
266
267LimaStatusCode SpecificEntitiesRecognizer::
268processOnEachSentence(AnalysisContent& analysis) const
269{
270 SELOGINIT;
271 LDEBUG << "SpecificEntitiesRecognizer::processOnEachSentence";
272
273 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData(m_graph));
274 if (anagraph==0) {
275 SELOGINIT;
276 LERROR << "SpecificEntitiesRecognizer::processOnEachSentence: no graph '" << m_graph << "' available !!";
277 return MISSING_DATA;
278 }
279
280 // get sentence bounds
281 SentenceBounds* sb=static_cast<SentenceBounds*>(analysis.getData("SentenceBounds"));
282 if (sb==0)
283 {
284 SELOGINIT;
285 LERROR << "SpecificEntitiesRecognizer::processOnEachSentence: no sentence bounds defined ! abort";
286 return MISSING_DATA;
287 }
288 if (sb->graphId() != m_graph) {
289 SELOGINIT;
290 LERROR << "SpecificEntitiesRecognizer::processOnEachSentence: SentenceBounds are computed on graph '" << sb->graphId() << "'";
291 LERROR << "can't compute specificEntities on graph '" << m_graph << "' !";
293 }
294
295 // resize data to the number of sentences
296 std::vector< Automaton::RecognizerMatch > seRecognizerResult;
297 LinguisticGraphVertex beginSentence=sb->getStartVertex();
298 SentenceBounds::const_iterator boundItr=sb->begin();
299 while (boundItr!=sb->end())
300 {
301 LinguisticGraphVertex endSentence=*boundItr;
302// LDEBUG << "analyze sentence from vertex " << beginSentence << " to vertex " << endSentence;
303
304 seRecognizerResult.clear();
305 m_recognizer->apply(
306 *anagraph,
307 beginSentence,
308 endSentence,
309 analysis,
310 seRecognizerResult
311/* bool testAllVertices=false,
312 bool stopAtFirstSuccess=true,
313 bool onlyOneSuccessPerType=false,
314 bool returnAtFirstSuccess=false,
315 bool applySameRuleWhileSuccess=false */
316 );
317
318 beginSentence=endSentence;
319 boundItr++;
320 }
321
322 return SUCCESS_ID;
323}
324
325LimaStatusCode SpecificEntitiesRecognizer::
326processOnWholeText(AnalysisContent& analysis) const
327{
328 SELOGINIT;
329 LDEBUG << "SpecificEntitiesRecognizer::processOnWholeText";
330
331 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData(m_graph));
332 if (anagraph == 0) {
333 SELOGINIT;
334 LERROR << "no graph '" << m_graph << "' available !!";
335 return MISSING_DATA;
336 }
337
338 // resize data to 1 : whole text taken as one
339 std::vector< Automaton::RecognizerMatch > seRecognizerResult;
340
341 m_recognizer->apply(*anagraph,
342 anagraph->firstVertex(),
343 anagraph->lastVertex(),
344 analysis,
345 seRecognizerResult
346
347// bool testAllVertices=false,
348// bool stopAtFirstSuccess=true,
349// bool onlyOneSuccessPerType=false,
350// bool returnAtFirstSuccess=false,
351// bool applySameRuleWhileSuccess=false) const;
352
353 );
354
355// SpecificEntityFound* seFound = static_cast<SpecificEntityFound*>(analysis.getData("SEFound"));
356// do
357// {
358// LDEBUG << "Applying automaton";
359// seFound->setFound(false);
360// m_recognizer->apply(*anagraph,
361// anagraph->firstVertex(),
362// anagraph->lastVertex(),
363// analysis,
364// seRecognizerResult,
365// false, // test all vertices=true
366// true, // stop rules search on a node at first success
367// true, // only one success per type
368// true // stop exploration at first success
369// );
370// } while (seFound->getFound());
371
372 return SUCCESS_ID;
373}
374
375bool SpecificEntitiesRecognizer::findSEFromRecognizer(
376 LinguisticGraphVertex& currentVertex,
377 AnalysisGraph* anagraph,
378 AnalysisContent& analysis,
379 RecognizerMatch& se) const
380{
381 SELOGINIT;
382 LDEBUG << "SpecificEntitiesRecognizer::findSEFromRecognizer " << currentVertex;
383
384 // search in recognizer if vertex matches one trigger
385 std::vector<RecognizerMatch> result;
386
387 if (m_recognizer->testOnVertex(
388 *anagraph,
389 currentVertex,
390 anagraph->firstVertex(),
391 anagraph->lastVertex(),
392 analysis,result
393/* bool stopAtFirstSuccess=true,
394 bool onlyOneSuccessPerType=false,
395 bool applySameRuleWhileSuccess=false */
396 ))
397 {
398 se=result.front();
399 return true;
400 }
401 return false;
402}
403
404
405} // end namespace
406} // end namespace
407} // end namespace
This file is the main header file for the data related to annotation graphs.
#define LWARN
Definition LimaCommon.h:160
#define LDEBUG
Definition LimaCommon.h:157
#define LINFO
Definition LimaCommon.h:158
#define LERROR
Definition LimaCommon.h:161
A graph structure for linguistic analysis.
boost::property_map< LinguisticGraph, vertex_token_t >::type VertexTokenPropertyMap
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
LinguisticGraph::adjacency_iterator LinguisticGraphAdjacencyIt
#define SELOGINIT
Defines a Factory to create Object of type Base.
#define SPECIFICENTITIESRECOGNIZER_CLASSID
Data used for the syntactic analyzis of texts.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Holds an annotation graph and gives an API to manipulate it.
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Use this exception to signal an error in one of the configuration files.
Definition LimaCommon.h:345
␈rief a class for the definition of a complete recognizer
Definition recognizer.h:78
uint64_t testOnVertex(const LinguisticAnalysisStructure::AnalysisGraph &graph, LinguisticGraphVertex &current, const LinguisticGraphVertex &begin, const LinguisticGraphVertex &end, AnalysisContent &analysis, std::vector< RecognizerMatch > &result, bool stopAtFirstSuccess=true, bool onlyOneSuccessPerType=false, bool applySameRuleWhileSuccess=false) const
test the recognizer on a given vertex : check if this vertex is a trigger and if a rule applies,...
uint64_t apply(const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &begin, const LinguisticGraphVertex &end, AnalysisContent &analysis, std::vector< RecognizerMatch > &result, bool testAllVertices=false, bool stopAtFirstSuccess=true, bool onlyOneSuccessPerType=false, bool returnAtFirstSuccess=false, bool applySameRuleWhileSuccess=false) const
apply the recognizer on a graph
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
Definition of a function suitable to be used as a dumper for specific entities annotations of an anno...
LimaStatusCode process(AnalysisContent &analysis) const
Process on data in analysisContent.
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager)
initialize with parameters from configuration file.
static const LinguisticResources & single()
const singleton accessor
Definition Singleton.h:51
static void logElapsedTime(const std::string &mess, const std::string &taskCategory=std::string(""))
log the number of microseconds since last UpdateCurrentTime
static void updateCurrentTime(const std::string &taskCategory=std::string(""))
store current time for new elapsed time computation
SimpleFactory< LinguisticProcessUnit, SpecificEntitiesRecognizer > specificEntitiesRecognizer(SPECIFICENTITIESRECOGNIZER_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ UNKNOWN_ERROR
Definition LimaCommon.h:238
@ SUCCESS_ID
Definition LimaCommon.h:237
@ INVALID_CONFIGURATION
Definition LimaCommon.h:242
@ MISSING_DATA
Definition LimaCommon.h:243
STL namespace.
launch exception related to the configuration file parsing