LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
applyRecognizerActions.cpp
Go to the documentation of this file.
1// Copyright 2002-2020 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/************************************************************************
7 *
8 * @file applyRecognizerActions.cpp
9 * @author besancon (besanconr@zoe.cea.fr)
10 * @date Tue Jan 25 2005
11 * copyright Copyright (C) 2005-2020 by CEA LIST
12 *
13 ***********************************************************************/
14
20
23using namespace Lima::Common;
24using namespace std;
25
26namespace Lima {
27namespace LinguisticProcessing {
28namespace ApplyRecognizer {
29
30//**********************************************************************
31// factories for constraint functions defined in this class
34
37
38//**********************************************************************
39// utility functions for CreateAlternative action
40std::pair<Token*,MorphoSyntacticData*> CreateAlternative::
41createAlternativeToken(const RecognizerMatch& recognizedExpression) const
42{
43 LimaString formStr=recognizedExpression.concatString();
44 StringsPoolIndex formId=(*m_stringsPool)[formStr];
45 Token* newToken=new Token(formId,
46 formStr,
47 recognizedExpression.positionBegin(),
48 recognizedExpression.length());
50
51 LimaString lemmaString;
52 EntityFeatures::const_iterator f=recognizedExpression.features().
54 if (f!=recognizedExpression.features().end()) {
55 lemmaString=boost::any_cast<const LimaString&>((*f).getValue());
56 }
57 else {
58 lemmaString=recognizedExpression.getString();
59 }
60
61 StringsPoolIndex lemma=(*m_stringsPool)[lemmaString];
62 LinguisticCode idiomProperty=recognizedExpression.getLinguisticProperties();
63// LOGINIT("LP::MorphologicAnalysis");
64// LDEBUG << "Idiomatic property=" << idiomProperty;
65
66 if (recognizedExpression.getHead() == 0) {
67// APPRLOGINIT;
68// LDEBUG << "Expression " << lemmaString << " has no head";
69
70 // set TStatus from first token
71 Token* firstToken=recognizedExpression.getToken(recognizedExpression.begin());
72 if (firstToken!=0) {
73 newToken->setStatus(firstToken->status());
74 }
75 else {
76 // set default t_status to null (changed from alpha small on 20110729 by GC)
77 // nothing to do: null by default
78/* TStatus* status=
79 new TStatus(T_SMALL, //AlphaCapitalType
80 T_NOT_ROMAN,//AlphaRomanType
81 false, //isHyphen,
82 false, //isPossessive,
83 false, //isConcatAbbrev,
84 T_NULL_NUM, //NumericType,
85 T_ALPHA // StatusType
86 );
87 status->setDefaultKey(Common::Misc::utf8stdstring2limastring("t_small"));
88 newToken->setStatus(status);*/
89 }
90
91 // set simple morphoData
92 LinguisticElement lingElt;
93 lingElt.inflectedForm=newToken->form();
94 lingElt.lemma=lemma;
95 lingElt.normalizedForm=lemma;
96 lingElt.properties=idiomProperty;
98 newData->push_back(lingElt);
99 return make_pair(newToken,newData);
100 }
101
102 Token* headToken=recognizedExpression.getHeadToken();
103 MorphoSyntacticData* headData=recognizedExpression.getHeadData();
104
105 // copy TStatus of the head
106 newToken->setStatus(headToken->status());
107
108 std::set<LinguisticCode> compatibleProperties;
109 getCompatibleProperties(headData,idiomProperty,
110 compatibleProperties);
111
112 // ensure that there is always a property,
113 // if none found, keep base one
114 if (compatibleProperties.empty()) {
116 LWARN << "head of expression " << (*m_stringsPool)[lemma] << " has no compatible properties, use"<< idiomProperty;
117 compatibleProperties.insert(idiomProperty);
118 }
119
120 LinguisticElement lingElt;
121 lingElt.inflectedForm=newToken->form();
122 lingElt.lemma=lemma;
123 lingElt.normalizedForm=lemma;
125
126 std::set<LinguisticCode>::const_iterator
127 prop=compatibleProperties.begin(),
128 prop_end=compatibleProperties.end();
129 for (; prop!=prop_end; prop++) {
130 lingElt.properties=*prop;
131 newData->push_back(lingElt);
132 }
133
134 return make_pair(newToken,newData);
135}
136
139 const LinguisticCode& baseProperty,
140 std::set<LinguisticCode>& newProperties) const {
141
142 LinguisticCode newProperty; // completed property
143 MorphoSyntacticData::const_iterator
144 it=headData->begin(),
145 it_end=headData->end();
146 for (; it!=it_end; it++) {
147 if (isCompatible(baseProperty,(*it).properties,newProperty)) {
148 newProperties.insert(newProperty);
149 }
150 }
151 return (!newProperties.empty());
152}
153
155isCompatible(const LinguisticCode& baseProperty,
156 const LinguisticCode& property,
157 LinguisticCode& newProperty) const {
158
159 //check compatibility only on macro
160 if (!m_macroAccessor->empty(baseProperty) &&
161 !m_macroAccessor->equal(property,baseProperty)) {
162 return false;
163 }
164 // if compatible, complete expression property (baseProperty)
165 // with head property (property)
166 newProperty=baseProperty;
167 const std::map<std::string,Common::PropertyCode::PropertyManager>&
168 managers=m_propertyCodeManager->getPropertyManagers();
169 for (map<string,Common::PropertyCode::PropertyManager>::const_iterator propIt=managers.begin();
170 propIt!=managers.end();
171 propIt++) {
172 const Common::PropertyCode::PropertyAccessor& acc=propIt->second.getPropertyAccessor();
173 if (acc.empty(newProperty)) {
174 acc.writeValue(property,newProperty);
175 }
176 }
177 return true;
178}
179
183 LinguisticGraph* graph) const
184{
185 LinguisticGraphVertex altVertex = add_vertex(*graph);
186 VertexTokenPropertyMap tokenMap = get(vertex_token, *graph);
187 VertexDataPropertyMap dataMap = get(vertex_data, *graph);
188 tokenMap[altVertex]= token;
189 dataMap[altVertex] = data;
190 return altVertex;
191}
192
203 LinguisticGraphVertex startVertex,
204 LinguisticGraphVertex alternativeFirstVertex,
205 LinguisticGraph& graph) const
206{
207// APPRLOGINIT;
208 // add edges from vertices preceding startVertex to alternativeFirstVertex
209 LinguisticGraphInEdgeIt it_begin,it_end;
210 boost::tie(it_begin,it_end)=in_edges(startVertex,graph);
211 for (LinguisticGraphInEdgeIt it(it_begin); it!=it_end; it++)
212 {
213 LinguisticGraphVertex previousVertex=source(*it,graph);
214 LinguisticGraphEdge newEdge;
215 bool ok;
216 boost::tie(newEdge, ok) = add_edge(previousVertex, alternativeFirstVertex, graph);
217 if (!ok) throw LinguisticProcessingException("CreateAlternative::createBeginAlternative() - Failed to add edge");
218// LDEBUG << " added initial edge " << newEdge;
219 }
220}
221
233 LinguisticGraphVertex alternativeLastVertex,
234 LinguisticGraphVertex endVertex,
235 LinguisticGraph& graph) const
236{
237// APPRLOGINIT;
238 // add edges from alternativeLastVertex to vertices following endVertex
239 LinguisticGraphOutEdgeIt it_begin,it_end;
240 boost::tie(it_begin,it_end)=out_edges(endVertex,graph);
241 for (LinguisticGraphOutEdgeIt it(it_begin); it!=it_end; it++)
242 {
243 LinguisticGraphVertex nextVertex=target(*it,graph);
244 LinguisticGraphEdge newEdge;
245 bool ok;
246 boost::tie(newEdge, ok) = add_edge(alternativeLastVertex, nextVertex, graph);
247 if (!ok) throw LinguisticProcessingException("CreateAlternative::attachEndOfAlternative() - Failed to add edge");
248// LDEBUG << " added final edge " << newEdge;
249 }
250}
251
252//**********************************************************************
254 const LimaString& complement):
255ConstraintFunction(language,complement),
256m_macroAccessor(0),
257m_propertyCodeManager(0)
258{
259
260 m_propertyCodeManager=&(static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(language)).getPropertyCodeManager());
261 m_macroAccessor=&(m_propertyCodeManager->getPropertyAccessor("MACRO"));
263
264 if (m_propertyCodeManager==0) {
266 LIMA_EXCEPTION( "cannot acces property code manager for language " << (int) language );
267 }
268 if (m_macroAccessor==0) {
270 LIMA_EXCEPTION( "cannot initialize MACRO property accessor for language " << (int) language );
271 }
272
273}
274
277 AnalysisContent& analysis) const
278{
279
280// APPRLOGINIT;
281// LDEBUG << "in CreateAlternative action";
282 // need dictionary
283 auto recoData = std::dynamic_pointer_cast<RecognizerData>(analysis.getData("RecognizerData"));
284 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"));
285 LinguisticGraph* graph=anagraph->getGraph();
286
287 if (result.isContiguous()) {
288 // only one part : terms in expression are adjacent -> easy part
289
290 // check if there is an overlap first
291 if (recoData->matchOnRemovedVertices(result)) {
292 // ignore current idiomatic expression, continue
294 LWARN << "recognized entity ignored: "
295 << result.concatString()
296 << ": overlapping with a previous one";
297 return false;
298 }
299
300 // create the new token
301 pair<Token*,MorphoSyntacticData*> newToken=
303 if (newToken.second->empty()) {
305 LERROR << "CreateAlternative::operator(): Got empty morphosyntactic data. Abort.";
306 delete newToken.first;
307 delete newToken.second;
308 return false;
309 }
310// LDEBUG << "create alternative token " << newToken.first->stringForm();
311
312 // add the vertex
313 LinguisticGraphVertex alternativeVertex =
314 addAlternativeVertex(newToken.first, newToken.second,graph);
315
316// LDEBUG << "add alternative vertex " << alternativeVertex;
317
318 //create the alternative with this only vertex
319 createBeginAlternative(result.front().getVertex(),
320 alternativeVertex,*graph);
321 attachEndOfAlternative(alternativeVertex,
322 result.back().getVertex(),*graph);
323
324 // if expression is not contextual, only keep alternative
325 if (! result.isContextual()) {
326 recoData->storeVerticesToRemove(result,graph);
327 }
328 }
329 else { // several parts
330// LDEBUG << "non adjacent recognized expression found: "
331// << result.concatString();
332 // check if there is an overlap first
333 if (recoData->matchOnRemovedVertices(result))
334 {
335 // ignore current expression, continue
337 LWARN << "alternative expression ignored: "
338 << result.concatString()
339 << ": overlapping with a previous one";
340 return false;
341 }
342
343 // create the new token
344 pair<Token*,MorphoSyntacticData*> newToken=
346 if (newToken.second->empty()) {
348 LERROR << "CreateAlternative::operator(): Got empty morphosyntactic data. Abort.";
349 delete newToken.first;
350 delete newToken.second;
351 return false;
352 }
353
354 // add the vertex
355 LinguisticGraphVertex altVertex =
356 addAlternativeVertex(newToken.first,newToken.second,graph);
357
358 //create the alternative with this vertex and duplicate of other vertives
359 deque<LinguisticGraphVertex> alternative;
360 LinguisticGraphVertex headVertex=result.getHead();
361// LDEBUG << "headVertex = " << headVertex;
362/* if (headVertex!=0) {
363 LDEBUG << "=> " << get(vertex_token,*graph,headVertex)->stringForm();
364 }*/
365 bool foundHead=false;
366 for (RecognizerMatch::const_iterator matchItr=result.begin();
367 matchItr!=result.end();
368 matchItr++)
369 {
370 if (!matchItr->isKept())
371 {
372 // duplicate this vertex
373// LDEBUG << "duplication vertex " << matchItr->getVertex();;
374 Token* token=get(vertex_token,*graph,matchItr->getVertex());
375 MorphoSyntacticData* data=new MorphoSyntacticData(*get(vertex_data,*graph,matchItr->getVertex()));
376 if (data->empty())
377 {
378 // ignore current idiomatic expression, continue
380 LERROR << "CreateAlternative::operator() Got empty morphosyntactic data. Abort";
381 delete data;
382 return false;
383 }
384 LinguisticGraphVertex dupVx=add_vertex(*graph);
385 put(vertex_token,*graph,dupVx,token);
386 put(vertex_data,*graph,dupVx,data);
387 alternative.push_back(dupVx);
388 }
389 else
390 {
391// LDEBUG << "kept vertex " << matchItr->getVertex();
392 if (matchItr->getVertex()==headVertex)
393 {
394 foundHead=true;
395// LDEBUG << "add head vertex " << altVertex;
396 alternative.push_back(altVertex);
397 }
398 }
399 }
400 if (!foundHead) {
402 LWARN << "head token has not been found in non contiguous expression. "
403 << "Alternative token is placed first";
404 alternative.push_front(altVertex);
405 }
406
407 // link alternatives
408// LDEBUG << "alternative has " << alternative.size() << " vertex";
409 createBeginAlternative(result.front().getVertex(),
410 alternative.front(),*graph);
411 {
412 deque<LinguisticGraphVertex>::const_iterator idItr=alternative.begin();
413 LinguisticGraphVertex lastAltVx=*idItr;
414 idItr++;
415 while (idItr!=alternative.end())
416 {
417 add_edge(lastAltVx,*idItr,*graph);
418 lastAltVx=*idItr;
419 idItr++;
420 }
421 }
422 attachEndOfAlternative(alternative.back(),
423 result.back().getVertex(),*graph);
424
425 // if expression is not contextual, only keep alternative
426 if (! result.isContextual())
427 {
428 recoData->storeVerticesToRemove(result,graph);
429 }
430 }
431 return true;
432}
433
434//**********************************************************************
436 const LimaString& complement):
437ConstraintFunction(language,complement)
438{
439}
440
443 AnalysisContent& analysis) const
444{
445// APPRLOGINIT;
446// LDEBUG << "add result in data:" << result;
447 auto recoData= std::dynamic_pointer_cast<RecognizerData>(analysis.getData("RecognizerData"));
448 if (recoData==0) {
450 LWARN << "StoreInData: cannot find RecognizerData in AnalysisContent";
451 return false;
452 }
453 recoData->addResult(result);
454 return true;
455}
456
457
458} // end namespace
459} // end namespace
460} // end namespace
#define DEFAULT_ATTRIBUTE
#define LWARN
Definition LimaCommon.h:160
#define LIMA_EXCEPTION(X)
This macro writes the message X to a previously configured error stream before throwing a LimaExcepti...
Definition LimaCommon.h:293
#define LERROR
Definition LimaCommon.h:161
LinguisticGraph::in_edge_iterator LinguisticGraphInEdgeIt
boost::property_map< LinguisticGraph, vertex_data_t >::type VertexDataPropertyMap
boost::graph_traits< LinguisticGraph >::edge_descriptor LinguisticGraphEdge
typedefs to simplify the access to various graphs elements
boost::property_map< LinguisticGraph, vertex_token_t >::type VertexTokenPropertyMap
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define CreateAlternativeId
#define StoreInDataId
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
Holds linguistic data for one language.
const FsaStringsPool & stringsPool(MediaId med) const
const MediaData & mediaData(MediaId media) const
Provide function to read write and check a property.
void writeValue(const LinguisticCode &value, LinguisticCode &code) const
write a property value in a coded int.
bool equal(const LinguisticCode &l1, const LinguisticCode &l2) const
check property equality for two linguisticCode
bool empty(const LinguisticCode &l) const
Test if the given code has property data.
const std::map< std::string, PropertyManager > & getPropertyManagers() const
Get the map of all PropertyManagers.
const PropertyAccessor & getPropertyAccessor(const std::string &propertyName) const
Get the PropertyAccessor associated to a property.
bool isCompatible(const LinguisticCode &baseProperty, const LinguisticCode &property, LinguisticCode &newProperty) const
LinguisticGraphVertex addAlternativeVertex(LinguisticAnalysisStructure::Token *, LinguisticAnalysisStructure::MorphoSyntacticData *, LinguisticGraph *graph) const
std::pair< LinguisticAnalysisStructure::Token *, LinguisticAnalysisStructure::MorphoSyntacticData * > createAlternativeToken(const Automaton::RecognizerMatch &recognizedExpression) const
Creates the new token corresponding to the idiomatic expression using information from the expression...
CreateAlternative(MediaId language, const LimaString &complement=LimaString())
void createBeginAlternative(LinguisticGraphVertex startVertex, LinguisticGraphVertex alternativeFirstVertex, LinguisticGraph &graph) const
create an alternative branch
void attachEndOfAlternative(LinguisticGraphVertex alternativeLastVertex, LinguisticGraphVertex endVertex, LinguisticGraph &graph) const
attach the end of an alternative to main path
bool getCompatibleProperties(const LinguisticAnalysisStructure::MorphoSyntacticData *headData, const LinguisticCode &baseProperty, std::set< LinguisticCode > &newProperties) const
virtual bool operator()(Automaton::RecognizerMatch &result, AnalysisContent &analysis) const override
zero-ary constraint function : applies the function without a vertex indication (used for actions,...
bool operator()(Automaton::RecognizerMatch &result, AnalysisContent &analysis) const override
zero-ary constraint function : applies the function without a vertex indication (used for actions,...
StoreInData(MediaId language, const LimaString &complement)
LinguisticAnalysisStructure::Token * getToken(RecognizerMatch::iterator) const
LinguisticAnalysisStructure::MorphoSyntacticData * getHeadData() const
LinguisticAnalysisStructure::Token * getHeadToken() const
void setStatus(const TStatus &status)
Set the TStatus of a token.
Definition Token.h:100
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
static MediaticData & changeable()
singleton accessor
Definition Singleton.h:71
ConstraintFunctionFactory< CreateAlternative > CreateAlternativeFactory(CreateAlternativeId)
ConstraintFunctionFactory< StoreInData > StoreInDataFactory(StoreInDataId)
SimpleFactory< MediaProcessUnit, ApplyRecognizer > ApplyRecognizer(APPLYRECOGNIZER_CLASSID)
NAUTITIA.
QString LimaString
Definition LimaString.h:33
STL namespace.
#define APPRLOGINIT