LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
SpecificEntitiesCompletion.cpp
Go to the documentation of this file.
1// Copyright 2002-2021 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
15
16#include <QRegularExpression>
17
18#include <queue>
19
20using namespace Lima::Common::AnnotationGraphs;
21using namespace Lima::Common::MediaticData;
22using namespace Lima::Common::Misc;
26using namespace std;
27
28namespace Lima {
29namespace LinguisticProcessing {
30namespace SpecificEntities {
31
33
36m_language(),
37m_graph("AnalysisGraph"),
38m_entityTypes()
39{}
40
41
44
47 Manager* manager)
48
49{
51 m_language=manager->getInitializationParameters().media;
52 try
53 {
54 deque<string> types=unitConfiguration.getListsValueAtKey("entityTypes");
55 for (const auto& s: types) {
56 m_entityTypes.insert(MediaticData::single().getEntityType(utf8stdstring2limastring(s)));
57 }
58 }
60 {
61 LDEBUG << "No param 'entityTypes' in SpecificEntitiesCompletion group: nothing to do";
62 }
63
64 try
65 {
66 m_graph=unitConfiguration.getParamsValueAtKey("graph");
67 }
68 catch (Common::XMLConfigurationFiles::NoSuchParam& ) {} // keep default value
69
70}
71
72// internal class holding information of found entities that is
73// useful for searching other occurrences
75public:
79 mutable std::set<std::pair<unsigned int,unsigned int> > occurrences;
80
81 EntityInfo(const EntityType& t, const LimaString& e, const LimaString& n):
82 entityType(t),
83 entityString(e),
84 entityNorm(n),
85 occurrences() {}
86
87 void addOccurrence(unsigned int posBegin, unsigned int posEnd)
88 {
89 occurrences.insert(make_pair(posBegin,posEnd));
90 }
91
92 LimaString regex() const {
93 // simplest for the moment: string equality
94 return entityString;
95 }
96
97 // comparison operator use equality on string and type (norm may be empty)
98 bool operator<(const EntityInfo& other) const {
99 if (this->entityType<other.entityType) {
100 return true;
101 }
102 else if (this->entityType==other.entityType) {
103 return this->entityString<other.entityString;
104 }
105 return false;
106 }
107
108 // debug
109 friend QDebug& operator<<(QDebug& os, const EntityInfo& ent) {
110 os << MediaticData::single().getEntityName(ent.entityType).toUtf8().constData()
111 << "/" << ent.entityString.toUtf8().constData()
112 << "/" << ent.entityNorm.toUtf8().constData()
113 << "/";
114 for (const auto& occ: ent.occurrences) {
115 os << "[" << occ.first << "-" << occ.second << "]";
116 }
117 return os;
118 }
119};
120
121class SpecificEntitiesCompletion::Entities: public set<EntityInfo>
122{
123public:
125
127 const SpecificEntityAnnotation* annot,
128 const VertexTokenPropertyMap& tokenMap)
129 {
130 unsigned int posBegin=annot->getPosition();
131 unsigned int posEnd=posBegin+annot->getLength();
132 EntityType entityType=annot->getType();
133 auto token=tokenMap[v];
134 if (token==0) {
135 SELOGINIT;
136 LERROR << "Empty token for vertex" << v;
137 return;
138 }
139 LimaString entityString=token->stringForm();
140 // get entity normalization in features
141 LimaString entityNorm;
142 for (const auto& f: annot->getFeatures())
143 {
144 if (f.getName()=="value") {
145 entityNorm=f.getValueLimaString();
146 break;
147 }
148 }
149 EntityInfo ent(entityType,entityString,entityNorm);
150 Entities::iterator it=find(ent);
151 if (it==end()) {
152 ent.addOccurrence(posBegin,posEnd);
153 insert(ent);
154 }
155 else {
156 // can't use addOccurrence (because iterator in a set is always const)
157 // byt can modify occurences member directly because I made it mutable
158 // (it is not used in the comparison function)
159 (*it).occurrences.insert(make_pair(posBegin,posEnd));
160 }
161 }
162
163 // debug
164 friend QDebug& operator<<(QDebug& os, const Entities& ent) {
165 for (const auto& e: ent) {
166 os << e << QTENDL;
167 }
168 return os;
169 }
170
171}; // end of class
172
173
174// internal class holding information of new entity occurrences
175// do not use the same EntityInfo class because, for new occurrence, having individual occurence is more efficient
177public:
181 unsigned int posBegin;
182 unsigned int posEnd;
183
185 const LimaString& s,
186 const LimaString& n,
187 const pair<unsigned int,unsigned int>& p):
188 entityType(t),entityString(s),entityNorm(n),posBegin(p.first),posEnd(p.second)
189 {}
190
191 // debug
192 friend QDebug& operator<<(QDebug& os, const EntityOccurrence& occ) {
193 os << MediaticData::single().getEntityName(occ.entityType).toUtf8().constData()
194 << "/" << occ.entityString.toUtf8().constData()
195 << "/" << occ.entityNorm.toUtf8().constData()
196 << "/[" << occ.posBegin << "-" << occ.posEnd << "]";
197 return os;
198 }
199
200 friend QDebug& operator<<(QDebug& os, const std::vector<EntityOccurrence>& occ) {
201 for (const auto& o: occ) { os << o << QTENDL; };
202 return os;
203 }
204
205};
206
207
209 AnalysisContent& analysis) const
210{
211 Lima::TimeUtilsController timer("SpecificEntitiesCompletion");
212 SELOGINIT;
213 LINFO << "start process";
214
215 // possible implementations:
216 // - create dynamic recognizers and apply them
217 // - use string matching on the text and report results on the graph
218 //
219 // use second approach
220
221 auto graphp = std::dynamic_pointer_cast<LinguisticAnalysisStructure::AnalysisGraph>(analysis.getData(m_graph));
222 if (graphp == nullptr)
223 {
224 SELOGINIT;
225 LERROR << "no graph "<< m_graph <<" ! abort";
226 return MISSING_DATA;
227 }
228 const auto& graph = *graphp;
229 auto lingGraph = const_cast<LinguisticGraph*>(graph.getGraph());
230 auto tokenMap = get(vertex_token, *lingGraph);
231
232 // gather found entities of the selected types
233 Entities foundEntities;
234 getEntities(analysis,foundEntities,tokenMap);
235 LDEBUG << "SpecificEntitiesCompletion: found entities" << QTENDL << foundEntities;
236
237 // find entities in text
238 std::vector<EntityOccurrence> newEntities;
239 findOccurrences(foundEntities,analysis,newEntities);
240 LDEBUG << "SpecificEntitiesCompletion: found new occurrences" << QTENDL << newEntities;
241
242 // report found entities in graph data
243 // use existing CreateSpecificEntity function, create RecognizerMatch
244 updateAnalysis(newEntities,analysis);
245
246 return SUCCESS_ID;
247}
248
249void SpecificEntitiesCompletion::getEntities(AnalysisContent& analysis, Entities& foundEntities,
250 const VertexTokenPropertyMap& tokenMap) const
251{
252 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.getData("AnnotationData"));
253 if (annotationData==0) {
254 SELOGINIT;
255 LDEBUG << "SpecificEntitiesCompletion: no annotation data";
256 return;
257 }
258
259 // take all annotations
260 AnnotationGraphVertexIt itv, itv_end;
261 boost::tie(itv, itv_end) = vertices(annotationData->getGraph());
262 for (; itv != itv_end; itv++)
263 {
264 if (annotationData->hasAnnotation(*itv,Common::Misc::utf8stdstring2limastring("SpecificEntity")))
265 {
266 const SpecificEntityAnnotation* annot = 0;
267 try
268 {
269 annot = annotationData->annotation(*itv,Common::Misc::utf8stdstring2limastring("SpecificEntity"))
270 .pointerValue<SpecificEntityAnnotation>();
271 }
272 catch (const boost::bad_any_cast& )
273 {
274 SELOGINIT;
275 LERROR << "This annotation is not a SpecificEntity; SE not logged";
276 continue;
277 }
278
279 // recuperer l'id du vertex morph cree
281 if (!annotationData->hasIntAnnotation(*itv,Common::Misc::utf8stdstring2limastring(m_graph)))
282 {
283 continue;
284 }
285 v = annotationData->intAnnotation(*itv,Common::Misc::utf8stdstring2limastring(m_graph));
286 if (m_entityTypes.find(annot->getType())!=m_entityTypes.end()) {
287 foundEntities.add(v,annot,tokenMap);
288 }
289 }
290 }
291}
292
293void SpecificEntitiesCompletion::
294findOccurrences(Entities& foundEntities,
295 AnalysisContent& analysis,
296 std::vector<EntityOccurrence>& newEntities) const
297{
298 auto text = std::dynamic_pointer_cast<LimaStringText>(analysis.getData("Text"));
299 for (const auto& e: foundEntities)
300 {
301 QRegularExpression re(e.regex());
302 qsizetype pos = 0;
303 QRegularExpressionMatch rx;
304 while ((pos = text->indexOf(re, pos, &rx)) != -1)
305 {
306 pair<unsigned int, unsigned int> matchpos(pos+1, pos+1+rx.capturedLength());
307 if (e.occurrences.find(matchpos)==e.occurrences.end()) {
308 // found a new occurrence
309 newEntities.push_back(EntityOccurrence(e.entityType, rx.captured(0), e.entityString, matchpos));
310 }
311 pos += rx.capturedLength();
312 }
313
314 }
315}
316
317void SpecificEntitiesCompletion::
318updateAnalysis(std::vector<EntityOccurrence>& occurrences,
319 AnalysisContent& analysis) const
320{
321 // use existing functions to create the entities:
322 // go through the graph to identify the vertices corresponding
323 // to the occurrences, build RecognizerMatch objects and use
324 // CreateSpecificEntity function
325 SELOGINIT;
326
327 // needs recognizerdata to make CreateSpecificEntity work
328 auto recoData = std::dynamic_pointer_cast<RecognizerData>(analysis.getData("RecognizerData"));
329 if (recoData == 0)
330 {
331 recoData = std::make_shared<RecognizerData>();
332 analysis.setData("RecognizerData", recoData);
333 }
334 // resultData is mandatory in recognizerData
335 RecognizerResultData* resultData=new RecognizerResultData(m_graph);
336 recoData->setResultData(resultData);
337
338 // create constraint functions
339 map<EntityType,std::unique_ptr<CreateSpecificEntity> > entityCreator;
340 for (auto t: m_entityTypes) {
341 LimaString entityName=MediaticData::single().getEntityName(t);
342 entityCreator[t]=std::make_unique<CreateSpecificEntity>(m_language,entityName);
343 }
344
345 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData(m_graph));
346
347 LinguisticGraph* graph=anagraph->getGraph();
348 VertexTokenPropertyMap tokenMap=get(vertex_token,*graph);
349 std::queue<LinguisticGraphVertex> toVisit;
350 std::set<LinguisticGraphVertex> visited;
351
352 const LinguisticGraphVertex firstVx = anagraph->firstVertex();
353 const LinguisticGraphVertex lastVx = anagraph->lastVertex();
354
355 RecognizerMatch currentMatch(anagraph.get());
356 int currentOccurrence=-1;
357
358 // todo: ensure the process is robust when the analysis graph is not linear...
359 try
360 {
361 toVisit.push(firstVx);
362
363 LinguisticGraphOutEdgeIt outItr,outItrEnd;
364 while (!toVisit.empty())
365 {
366 LinguisticGraphVertex v=toVisit.front();
367 toVisit.pop();
368 if (v == lastVx) {
369 continue;
370 }
371
372 for (boost::tie(outItr,outItrEnd)=out_edges(v,*graph); outItr!=outItrEnd; outItr++)
373 {
374 LinguisticGraphVertex next=target(*outItr,*graph);
375 if (visited.find(next)==visited.end())
376 {
377 visited.insert(next);
378 toVisit.push(next);
379 }
380 }
381
382 if (v != firstVx && v != lastVx)
383 {
384 processVertex(v,tokenMap,occurrences,currentOccurrence,currentMatch, entityCreator, anagraph.get(), analysis);
385 }
386 }
387 }
388 catch (std::exception &exc)
389 {
390 SELOGINIT;
391 LWARN << "Exception in SpecificEntitiesCompletion : " << exc.what();
392 }
393
394 // effective graph update
395 recoData->removeVertices(analysis);
396 recoData->clearVerticesToRemove();
397 recoData->removeEdges(analysis);
398 recoData->clearEdgesToRemove();
399 // clean recognizer data (used internally in the process unit)
400 recoData->deleteResultData();
401 analysis.removeData("RecognizerData");
402
403}
404
405void SpecificEntitiesCompletion::processVertex(LinguisticGraphVertex currentVertex,
406 const VertexTokenPropertyMap& tokenMap,
407 std::vector<EntityOccurrence>& occurrences,
408 int& currentOccurrence,
409 RecognizerMatch& currentMatch,
410 map<EntityType,std::unique_ptr<CreateSpecificEntity> >& entityCreator,
411 AnalysisGraph* anagraph,
412 AnalysisContent& analysis) const
413{
414 SELOGINIT;
415 Token* currentToken=tokenMap[currentVertex];
416 if (currentToken)
417 {
418 // process current Token
419 unsigned int pos=currentToken->position();
420 unsigned int posEnd=pos+currentToken->length();
421 //LDEBUG << "Exploring vertex" << currentVertex << "at position" << pos;
422 if (currentOccurrence!=-1) {
423 // in the process of matching the entity
424 if (pos>occurrences[currentOccurrence].posEnd) {
425 // we missed it
426 LDEBUG << "--at pos" << pos << ": missed occurrence" << occurrences[currentOccurrence];
427 currentMatch=RecognizerMatch(anagraph);
428 currentOccurrence=-1;
429 }
430 else {
431 // in the process: add this token to the match
432 LDEBUG << "--at pos" << pos << ": in occurrence" << occurrences[currentOccurrence];
433 currentMatch.addBackVertex(currentVertex);
434 }
435 }
436 else {
437 // does the position of the token matches the position of an occurrence
438 for (unsigned int i(0);i<occurrences.size();i++) {
439 if (pos==occurrences[i].posBegin) {
440 currentOccurrence=i;
441 currentMatch.addBackVertex(currentVertex);
442 LDEBUG << "--at pos" << pos << ": start occurrence" << occurrences[currentOccurrence];
443 break;
444 }
445 }
446 }
447 if (currentOccurrence!=-1 && posEnd==occurrences[currentOccurrence].posEnd) {
448 LDEBUG << "--at pos" << pos << ": -> end occurrence" << occurrences[currentOccurrence];
449 // found the end: success !
450 // add relevant info
451 EntityOccurrence& occ=occurrences[currentOccurrence];
452 currentMatch.setType(occ.entityType);
453 currentMatch.features().setFeature("value",occ.entityNorm);
454 // create the entity
455 LDEBUG << "found match for occcurrence" << occ << ":" << currentMatch;
456 (*entityCreator[occ.entityType])(currentMatch,analysis);
457 //reinit currents
458 currentMatch=RecognizerMatch(anagraph);
459 currentOccurrence=-1;
460 }
461 }
462}
463
464} // end namespace
465} // end namespace
466} // end namespace
This file is the main header file for the data related to annotation graphs.
#define LWARN
Definition LimaCommon.h:160
#define QTENDL
Definition LimaCommon.h:32
#define LDEBUG
Definition LimaCommon.h:157
#define LINFO
Definition LimaCommon.h:158
#define LERROR
Definition LimaCommon.h:161
boost::property_map< LinguisticGraph, vertex_token_t >::type VertexTokenPropertyMap
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define SELOGINIT
Defines a Factory to create Object of type Base.
#define SPECIFICENTITIESCOMPLETION_CLASSID
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
void removeData(const QString &id)
remove the analysisData with the given id
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
std::deque< std::string > & getListsValueAtKey(const std::string &key)
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
void setFeature(const std::string &name, const ValueType &value)
void setType(const Common::MediaticData::EntityType &t)
void addBackVertex(const LinguisticGraphVertex &, bool isKept=true, const LimaString &ruleElementId="")
An AnalysisData containing a LinguisticGraph with a language and an id.
void addOccurrence(unsigned int posBegin, unsigned int posEnd)
EntityInfo(const EntityType &t, const LimaString &e, const LimaString &n)
friend QDebug & operator<<(QDebug &os, const EntityInfo &ent)
std::set< std::pair< unsigned int, unsigned int > > occurrences
void add(LinguisticGraphVertex v, const SpecificEntityAnnotation *annot, const VertexTokenPropertyMap &tokenMap)
friend QDebug & operator<<(QDebug &os, const std::vector< EntityOccurrence > &occ)
EntityOccurrence(const EntityType &t, const LimaString &s, const LimaString &n, const pair< unsigned int, unsigned int > &p)
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager)
initialize with parameters from configuration file.
LimaStatusCode process(AnalysisContent &analysis) const
Process on data in analysisContent.
A representation of a specific entity to store in the annotation graph.
This file contains a class to control log of informations about time, such as logging cumulated time ...
AnnotationGraph::vertex_iterator AnnotationGraphVertexIt
LimaString utf8stdstring2limastring(const std::string &src)
SimpleFactory< MediaProcessUnit, SpecificEntitiesCompletion > specificEntitiesCompletion(SPECIFICENTITIESCOMPLETION_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
@ MISSING_DATA
Definition LimaCommon.h:243
QString LimaString
Definition LimaString.h:33
STL namespace.
launch exception related to the configuration file parsing