LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
EntityTracker.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6#include "EntityTracker.h"
7
17#include "CoreferenceEngine.h"
18#include "CoreferenceData.h"
19
20
21
22
23
24#include <fstream>
25#include <queue>
26
27using namespace std;
28using namespace boost;
29using namespace Lima::Common::AnnotationGraphs;
31using namespace Lima::Common::MediaticData;
32using namespace Lima::Common::AnnotationGraphs;
35using namespace Lima::Common::Misc;
36using namespace Lima::Common::MediaticData;
37using namespace Lima::Common::AnnotationGraphs;
39
40
41namespace Lima
42{
43namespace LinguisticProcessing
44{
45namespace EntityTracking
46{
47
49
53
54/*EntityTracker::EntityTracker(const Automaton::RecognizerMatch& entity,
55 FsaStringsPool& sp)
56{
57}*/
59
61 Manager* /*manager*/)
62
63{
64}
65
67{
70
71 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.getData("LinguisticMetaData"));
72 if (metadata == 0)
73 {
74 LERROR << "no LinguisticMetaData ! abort";
75 return MISSING_DATA;
76 }
77
78 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"));
79 if (anagraph==0)
80 {
81 LERROR << "no graph 'AnaGraph' available !";
82 return MISSING_DATA;
83 }
84
85 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.getData("AnnotationData"));
86 if (annotationData==0)
87 {
88 LERROR << "no annotation graph available !";
89 return MISSING_DATA;
90 }
91
92 // add new data to store co-references
93 CoreferenceData* corefData = new CoreferenceData;
94 analysis.setData("CoreferenceData",corefData);
95
97 LinguisticGraph* graph=anagraph->getGraph();
98 LinguisticGraphVertex lastVertex=anagraph->lastVertex();
99 LinguisticGraphVertex firstVertex=anagraph->firstVertex();
100
101 std::queue<LinguisticGraphVertex> toVisit;
102 std::set<LinguisticGraphVertex> visited;
103
104 LinguisticGraphOutEdgeIt outItr,outItrEnd;
105
106 // output vertices between begin and end,
107 // but do not include begin (beginning of text or previous end of sentence) and include end (end of sentence)
108 toVisit.push(firstVertex);
109
110 bool first=true;
111 bool last=false;
112 while (!toVisit.empty()) {
113 LinguisticGraphVertex v=toVisit.front();
114 toVisit.pop();
115 if (last || v == lastVertex) {
116 continue;
117 }
118 if (v == lastVertex) {
119 last=true;
120 }
121
122 for (boost::tie(outItr,outItrEnd)=out_edges(v,*graph); outItr!=outItrEnd; outItr++)
123 {
124 LinguisticGraphVertex next=target(*outItr,*graph);
125 if (visited.find(next)==visited.end())
126 {
127 visited.insert(next);
128 toVisit.push(next);
129 }
130 }
131
132 if (first) {
133 first=false;
134 }
135 else {
136 // first, check if vertex corresponds to a specific entity
137 std::set< AnnotationGraphVertex > matches = annotationData->matches("AnalysisGraph",v,"annot");
138 for (std::set< AnnotationGraphVertex >::const_iterator it = matches.begin();
139 it != matches.end(); it++)
140 {
142 Token* t=get(vertex_token,*graph,vx);
143 /* sauvegarde de tous les vertex */
144 if (t != 0)
145 {
146 //storeAllToken(t);
147 //allToken.push_back(t);
148 ref.storeAllToken(*t);
149 }
150 if (annotationData->hasAnnotation(vx, Common::Misc::utf8stdstring2limastring("SpecificEntity")))
151 {
152 /*const SpecificEntityAnnotation* se =
153 annotationData->annotation(vx, Common::Misc::utf8stdstring2limastring("SpecificEntity")).
154 pointerValue<SpecificEntityAnnotation>();*/
155 //storeSpecificEntity(se);
156 //Token* t=get(vertex_token,*graph,vx);
157 //storedAnnotations.push_back(*t);
158 ref.storeAnnot(*t);
159// std::cout<< "le vertex de nom "<< t->stringForm()<<std::endl;
160 }
161 }
162 }
163 }
164
165 /* recherche des coréferences entre les entitées nommées précédemment détectées */
166
167 vector<Token> vectTok;
168 vector<Token>::const_iterator it1=ref.getAnnotations().begin(), it1_end=ref.getAnnotations().end();
169 for (;
170 it1 != it1_end;
171 it1++)
172 {
173// checkCoreference (*it1,ref);
174 vectTok = ref.searchCoreference(*it1);
175 if (vectTok.size() > 0)
176 {
177 corefData->push_back(vectTok);
178 }
179 ref.searchCoreference(*it1);
180 }
181
182 /* get the text */
183// LimaStringText* text=static_cast<LimaStringText*>(analysis.getData("Text"));
184
185 return SUCCESS_ID;
186}
187
188
189// bool EntityTracker::checkCoreference(const Token& tok, CoreferenceEngine ref) const
190// {
191// //allToken = ref.getToken();
192// vector<Token> vectTok = ref.searchCoreference(tok);
193// if (vectTok.size() > 1)
194// {
195// ref.storeFindedToken(vectTok);
196// }
197// ref.searchCoreference(tok);
198// }
199
200
201/*
202void EntityTracker::storeAllToken(const Token* tok)
203{
204 allToken.push_back(*tok);
205}*/
206/*
207void EntityTracker::storeSpecificEntity (const Lima::LinguisticProcessing::
208 SpecificEntities::SpecificEntityAnnotation * se) const
209{
210 // look at the vertex
211 Token* t=get(vertex_token,*graph,v);
212 if (t!=0) {
213 la condition sera si l'annotation est une personne, un lieu ou bien une organisation
214}*/
215
216// //////////////////////////////////////////////////////////////////////////////////////////
217// //////////////////////////////////////////////////////////////////////////////////////////
218// bool EntityTracker::isInclude(const std::string original, const std::string currentWord) const
219// {
220// if (currentWord.size() > original.size())
221// return false;
222//
223// uint64_t comptCurrent(0);
224// for (uint64_t i(0); i< original.size(); i++)
225// {
226// if (original[i] == currentWord[comptCurrent])
227// {
228// comptCurrent++;
229// }
230// }
231// if (comptCurrent == currentWord.size())
232// return true;
233// else
234// return false;
235// }
236//
237// //////////////////////////////////////////////////////////////////////////////////////////
238// //////////////////////////////////////////////////////////////////////////////////////////
239// bool EntityTracker::isAcronym(const std::string original, const std::string currentWord) const
240// {
241// for (std::vector< std::vector<std::string> >::const_iterator it = Acronyms.begin(), it_end = Acronyms.end();
242// it != it_end;
243// it++)
244// {
245// if (strcmp((*(*it).begin()).c_str(),original.c_str()) == 0)
246// {
247// for (std::vector<string>::const_iterator it_intern = (*it).begin(), it_int_end = (*it).end();
248// it_intern != it_int_end;
249// it_intern++)
250// {
251// if ((strcmp((*it_intern).c_str(),currentWord.c_str()) == 0) ||
252// (isInclude(*it_intern,currentWord)))
253// return true;
254// }
255// }
256// }
257// return false;
258// }
259//
260// //////////////////////////////////////////////////////////////////////////////////////////
261// //////////////////////////////////////////////////////////////////////////////////////////
262// void EntityTracker::addNewForm(const std::string original, const std::string currentWord)
263// {
264// for (std::vector< std::vector<std::string> >::iterator it = Acronyms.begin(), it_end = Acronyms.end();
265// it != it_end;
266// it++)
267// {
268// if (strcmp((*(*it).begin()).c_str(),original.c_str()) == 0)
269// {
270// (*it).push_back(currentWord);
271// return;
272// }
273// }
274// }
275//
276// //////////////////////////////////////////////////////////////////////////////////////////
277// //////////////////////////////////////////////////////////////////////////////////////////
278// bool EntityTracker::exist(const std::string mot, const std::vector< std::vector<std::string> > Acronyms) const
279// {
280// std::vector<std::string>::const_iterator iterat;
281// for (std::vector< std::vector<std::string> >::const_iterator it = Acronyms.begin(), it_end = Acronyms.end();
282// it != it_end;
283// it++)
284// {
285// iterat = find((*it).begin(),(*it).end(),mot);
286// if (iterat != (*it).end())
287// {
288// return true;
289// }
290// }
291// return false;
292// }
293//
294// //////////////////////////////////////////////////////////////////////////////////////////
295// //////////////////////////////////////////////////////////////////////////////////////////
296// void EntityTracker::searchCoreference(const std::string text)
297// {
298//
299// string mot;
300//
301// /*
302// LimaStringText* text=static_cast<LimaStringText*>(analysis.getData("Text"));
303//
304// vector<string::size_type> paragraphPositions;
305// string::size_type currentPos=0;
306// string::size_type i=text->find(m_paragraphSeparator,currentPos);
307// while (i!=string::npos) {
308// paragraphPositions.push_back(i);
309// // goto next char that is not a carriage return
310// currentPos=text->find_first_not_of(m_paragraphSeparator,i+1);
311// i=text->find(m_paragraphSeparator,currentPos);
312// }
313// */
314//
315// /* parcourir tous les noeuds */
316//
317// /* Chercher si le type du noeud appartient à l'ensemble des personne, organisation ou bien lieu */
318//
319// /* Si le mot existe dans le vecteur des acronyms, ça ne sert à rien chercher ses acronyms parce qu'ils
320// sont recherchés */
321// if (!exist(mot, Acronyms))
322// {
323//
324// }
325// /* Recherche dans le texte de toutes les formes de mot précédent dans tout le text */
326//
327// }
328
329
330
331}
332}
333}
This file is the main header file for the data related to annotation graphs.
#define ENTITYTRACKER_CLASSID
#define LERROR
Definition LimaCommon.h:161
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define SELOGINIT
Defines a Factory to create Object of type Base.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Manage initialization of InitializableObjects using configuration module and parameters.
void storeAnnot(const LinguisticAnalysisStructure::Token &token)
std::vector< LinguisticAnalysisStructure::Token > & getAnnotations()
std::vector< LinguisticAnalysisStructure::Token > searchCoreference(const LinguisticAnalysisStructure::Token &tok)
void storeAllToken(const LinguisticAnalysisStructure::Token &token)
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
static void updateCurrentTime(const std::string &taskCategory=std::string(""))
store current time for new elapsed time computation
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
LimaString utf8stdstring2limastring(const std::string &src)
SimpleFactory< MediaProcessUnit, EntityTracker > EntityTrackerFactory(ENTITYTRACKER_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
@ MISSING_DATA
Definition LimaCommon.h:243
STL namespace.