LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
DynamicSvmToolPosTagger.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/***************************************************************************
7 * Copyright (C) 2004-2012 by CEA LIST *
8 * *
9 ***************************************************************************/
10
12
13
19
29#include "svmtool/tagger.h"
30#include "svmtool/nodo.h"
31
32#include <boost/algorithm/string/split.hpp>
33#include <boost/algorithm/string/classification.hpp>
34#include <boost/algorithm/string/replace.hpp>
35#include <boost/tuple/tuple.hpp>
36#include <boost/tuple/tuple_io.hpp>
37
38#include <cfloat> // LDBL_MIN/MAX
39#include <cmath> // log
40
41using namespace Lima::Common::MediaticData;
42
43namespace Lima
44{
45namespace LinguisticProcessing
46{
47namespace PosTagger
48{
49
50/* Factory boilerplate code only used to initialize the DynamicSvmToolPosTagger module */
51DynamicSvmToolPosTaggerFactory* DynamicSvmToolPosTaggerFactory::s_instance=new DynamicSvmToolPosTaggerFactory(DYNAMICSVMTOOLPOSTAGGER_CLASSID);
52
53DynamicSvmToolPosTaggerFactory::DynamicSvmToolPosTaggerFactory(const std::string& id) :
55{}
56
57std::shared_ptr<MediaProcessUnit> DynamicSvmToolPosTaggerFactory::create(
59 MediaProcessUnit::Manager* manager) const
60{
61 auto posTagger = std::make_shared<DynamicSvmToolPosTagger>();
62 posTagger->init(unitConfiguration,manager);
63
64 return posTagger;
65}
66
67/* Retrieves the needed values from the configuration and inits the SVMTool tagger */
68void DynamicSvmToolPosTagger::init(
70 Manager* manager)
71
72{
83 Lima::TimeUtilsController timer("DynamicSvmToolPosTagger init");
84
86 LDEBUG << "init!";
87
88 m_language=manager->getInitializationParameters().media;
89 const Common::MediaticData::LanguageData& ldata = static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language));
90 m_MicroManager=&(ldata.getPropertyCodeManager().getPropertyManager("MICRO"));
91 m_microAccessor=&(m_MicroManager->getPropertyAccessor());
92 string resourcesPath=MediaticData::single().getResourcesPath();
93
94
95 /* Retrieve the default category */
96 std::string defaultName;
97 try
98 {
99 defaultName = unitConfiguration.getParamsValueAtKey("defaultCategory");
100 }
102 {
103 LWARN << "No default microtageory for DynamicSvmToolPosTagger! using PONCTU_FORTE.";
104 defaultName = "PONCTU_FORTE";
105 }
106 m_defaultCateg = m_MicroManager->getPropertyValue(defaultName);
107
108
109 /* Retrieve the stop categories */
110 try
111 {
112 std::deque<std::string> cats=unitConfiguration.getListsValueAtKey("stopCategories");
113 for (std::deque<std::string>::iterator it=cats.begin();
114 it!=cats.end();
115 it++)
116 {
117 m_stopCategories.push_back(m_MicroManager->getPropertyValue(*it));
118 }
119 }
121 {
122 LWARN << "No stop categories defined! using the default category";
123 m_stopCategories.push_back(m_defaultCateg);
124 }
125
126 /* Retrieve the SVMTool model */
127 std::string model;
128 try {
129 model = unitConfiguration.getParamsValueAtKey("model");
131 LERROR << "No SVMTool model defined in the configuration file!";
132 throw InvalidConfiguration();
133 }
134
135
136 // Creates the tagger we use
137 erCompRegExp();
138 t = new tagger(Common::Misc::findFileInPaths(resourcesPath.c_str(), model.c_str()).toUtf8().constData());
139 t->taggerLoadModelsForTagging();
140 t->taggerShowComments();
141 t->taggerActiveShowScoresFlag();
142 t->taggerInit();
143
144}
145
146/* Actual pos-tagging code
147 *
148 * The goal of the DynamicSvmToolPosTagger is to apply the Viterbi algorithm
149 * on top of the SVMTool code. Each word has an associated score given a
150 * specific window. We try every possible window, and choose the best path
151 * in our treillis. See the wiki (lima:lima:svmtool) for more details.
152 *
153 * Implementation note: the code below could be faster but this won't improve
154 * the speed of the whole process: 99% of time and memory is spent in SVMTool++.
155 */
156LimaStatusCode DynamicSvmToolPosTagger::process(AnalysisContent& analysis) const
157{
158 Lima::TimeUtilsController timer("DynamicSvmToolPosTagger");
159 PTLOGINIT;
160
161 /* This modules creates a PosGraph from an AnalysisGraph */
164 /* We also store the actual graphs */
165 LinguisticGraph *srcGraph;
166 LinguisticGraph *resultGraph;
167
168 /* First retrieve the source graph (Analysis) */
169 analysisGraph = static_cast<LinguisticAnalysisStructure::AnalysisGraph*>
170 (analysis.getData("AnalysisGraph").get());
171 srcGraph = analysisGraph->getGraph();
172
173 /* Let's start with the "forward" algorithm
174 *
175 * The goal here is to go through each node microtag to be able to tell
176 * what is the best way to go throught this microtag. This will enable us
177 * to simply go back from the end of the graph and choose the most
178 * probable path.
179 */
180 std::queue<LinguisticGraphVertex> tokenQueue;
181 std::set<LinguisticGraphVertex> visited;
182 std::map<LinguisticGraphVertex, struct PathInfo > maxAncestor;
183
184 /* Push every vertex coming from vertex 0 onto the "tokens to be visited" list */
185 for(LinguisticGraphVertex vertex:
186 nextTokens(analysisGraph->firstVertex(), srcGraph))
187 {
188 tokenQueue.push(vertex);
189 }
190
191 posGraph = new LinguisticAnalysisStructure
192 ::AnalysisGraph("PosGraph", m_language, false, true);
193 analysis.setData("PosGraph", posGraph);
194
195 /* For every node in the graph (this is a BFS on a treillis, ie. a topological sort) */
196 while (!tokenQueue.empty()) {
197 LinguisticGraphVertex vertex = tokenQueue.front();
198 tokenQueue.pop();
199 LDEBUG << "\n" << vertex << " -> " << getWord(vertex, srcGraph);
200
201
202 double logMaxWeight = -LDBL_MAX;
203 int maxLength = 0;
204
205 /* For every ancestor of our node */
206 std::set<LinguisticGraphVertex> previousTokens = getPreviousTokens(vertex, srcGraph);
207 if(previousTokens.empty()) previousTokens.insert(posGraph->firstVertex());
208 for (auto it = previousTokens.begin(); it != previousTokens.end(); ++it) {
209 LinguisticGraphVertex prevVertex = *it;
210
211 std::string pos = "";
212 double logCurWeight = log(1.0), w;
213 if (vertex != 1) {
214 /* Call SVMTool */
215 boost::tie(pos, w) = SVMTool(srcGraph, vertex, prevVertex, maxAncestor);
216 logCurWeight = log(w);
217 LDEBUG << "weight = " << logCurWeight << " -> " << pos << "(" << w << ")";
218 }
219
220
221 /* find out the previous weight */
222 double logPrevPrice = log(1.0);
223 int prevLength = 0;
224 if(maxAncestor.find(prevVertex) != maxAncestor.end()) {
225 struct PathInfo prevPath = maxAncestor[prevVertex];
226 logPrevPrice = prevPath.score;
227 prevLength = prevPath.pathLength;
228 }
229
230 /* Did we find a better weight ? */
231 if((logPrevPrice + logCurWeight) / (prevLength+1)
232 > (logMaxWeight / (maxLength+1))) {
233 /* update the max ancestor */
234 struct PathInfo currentPath = { prevVertex, logCurWeight + logPrevPrice, pos, prevLength+1 };
235 maxAncestor[vertex] = currentPath;
236 logMaxWeight = logCurWeight + logPrevPrice;
237 maxLength = prevLength;
238 LDEBUG << " -> " << logMaxWeight << " (" << maxLength << ")";
239 }
240 }
241
242 LDEBUG << getWord(vertex, srcGraph) << " -> " << maxAncestor[vertex].pos;
243
244
245 /* we're only adding the vertices we never added before */
246 LinguisticGraphOutEdgeIt outItr,outItrEnd;
247 boost::tie(outItr,outItrEnd)=out_edges(vertex,*srcGraph);
248
249 for (;outItr!=outItrEnd;outItr++) {
250 LinguisticGraphVertex nextToken = target(*outItr,*srcGraph);
251
252 if (visited.find(nextToken) == visited.end()) {
253 tokenQueue.push(nextToken);
254 visited.insert(nextToken);
255 }
256 }
257 }
258
259 /* Construct the stack which is going to serve as a basis to build our resultGraph */
260 std::stack<boost::tuple<LinguisticGraphVertex, LinguisticCode> > chosenPath;
261 LinguisticGraphVertex backVertex = 1;
262 while ((backVertex = maxAncestor[backVertex].prev) != 0) {
263 LinguisticCode categ = m_MicroManager->getPropertyValue(maxAncestor[backVertex].pos);
264 chosenPath.push(boost::make_tuple(backVertex, categ));
265 }
266
267 /* Then start building the result graph (Pos) */
268 resultGraph = posGraph->getGraph();
269 // remove the edge between those two vertices will enable us to add nodes inbetween
270 remove_edge(posGraph->firstVertex(),posGraph->lastVertex(),*resultGraph);
271
272 /* Build everything needed to populate the PosGraph */
274 static_cast< Common::AnnotationGraphs::AnnotationData* >(analysis.getData("AnnotationData").get());
275
276 LinguisticGraphVertex previousPosVertex = 0;
277
278 while(!chosenPath.empty()) {
280 boost::tie(vertex, code) = chosenPath.top(); chosenPath.pop();
281
282 LinguisticGraphVertex newVertex = boost::add_vertex(*resultGraph);
283 LDEBUG << "create vertex " << newVertex;
284 annotationData->addMatching("PosGraph", newVertex, "annot", vertex);
285 annotationData->addMatching("AnalysisGraph", vertex, "PosGraph", newVertex);
286 AnnotationGraphVertex annotVertex = annotationData->createAnnotationVertex();
287 annotationData->annotate(annotVertex, Common::Misc::utf8stdstring2limastring("PosGraph"), newVertex);
288
289 // set linguistic infos
290 LinguisticAnalysisStructure::MorphoSyntacticData* morphoData=get(vertex_data,*srcGraph, vertex);
291 LinguisticAnalysisStructure::Token* srcToken=get(vertex_token,*srcGraph,vertex);
292 if (morphoData!=0)
293 {
295 LinguisticAnalysisStructure::CheckDifferentPropertyPredicate differentMicro(*m_microAccessor, code);
296 std::back_insert_iterator<LinguisticAnalysisStructure::MorphoSyntacticData> backInsertItr(*posData);
297 remove_copy_if(morphoData->begin(),morphoData->end(),backInsertItr,differentMicro);
298 if (posData->empty() || morphoData->empty()) {
299 LWARN << "No matching category found for tagger result " << getWord(vertex, srcGraph) << " " << m_MicroManager->getPropertySymbolicValue(code);
300 if (!morphoData->empty())
301 {
302 LWARN << "Taking any one";
303 posData->push_back(morphoData->front());
304 }
305 }
306 put(vertex_data,*resultGraph,newVertex,posData);
307 put(vertex_token,*resultGraph,newVertex,srcToken);
308 }
309
310 boost::add_edge(previousPosVertex, newVertex, *resultGraph);
311
312
313 LDEBUG << getWord(vertex, srcGraph) << " -> " << m_MicroManager->getPropertySymbolicValue(code);
314 previousPosVertex = newVertex;
315 }
316
317 boost::add_edge(previousPosVertex, posGraph->lastVertex(), *resultGraph);
318
319 return SUCCESS_ID;
320}
321
322boost::tuple<std::string, uint64_t> DynamicSvmToolPosTagger::SVMTool(
323 const LinguisticGraph* srcGraph,
325 LinguisticGraphVertex prevVertex,
326 std::map<LinguisticGraphVertex, struct PathInfo > &maxAncestor) const {
327 PTLOGINIT;
328 /* We now build the window to give to SVMTool. */
329 LinguisticGraphVertex prevPrevVertex = 0;
330 if(maxAncestor.find(prevVertex) != maxAncestor.end()) {
331 prevPrevVertex = maxAncestor[prevVertex].prev;
332 }
333 auto node_context = buildContext(srcGraph, prevPrevVertex, prevVertex, vertex);
334 std::vector<std::string> microsStr = getMicros(vertex, srcGraph);
335
336 if (microsStr.empty()) {
337 LERROR << getWord(vertex, srcGraph) << " has no attached microcategories";
338 return boost::make_tuple("", LDBL_MIN);
339 }
340
341
342 /* Call SVMTool */
343 t->sw->setWindow(node_context);
344 t->setPossibles(microsStr);
345 //showWindow(node_context);
346 t->taggerGenerateScore(node_context[2],1);
347
348 /* Normalize the weight */
349 node_context[2]->weight += 10.0;
350 node_context[2]->weight /= 20.0;
351
352 LDEBUG << "§" << node_context[2]->pos << "|" << (float)(node_context[2]->weight) << "§";
353
354 return boost::make_tuple(node_context[2]->pos, node_context[2]->weight);
355
356}
357
358std::vector<std::string> DynamicSvmToolPosTagger::getMicros(LinguisticGraphVertex token, const LinguisticGraph *srcGraph) const {
359 CVertexDataPropertyMap dataMap = get(vertex_data, *srcGraph);
360 LinguisticAnalysisStructure::MorphoSyntacticData *mdata = dataMap[token];
361 const Common::PropertyCode::PropertyManager& microManager = static_cast<const Common::MediaticData::LanguageData&>(MediaticData::single().mediaData(m_language)).getPropertyCodeManager().getPropertyManager("MICRO");
362
363 std::set<LinguisticCode> micros;
364 if (mdata == NULL) {
365 micros.insert(m_defaultCateg);
366 } else {
367 mdata->allValues(*m_microAccessor, micros);
368 }
369
370 std::vector<std::string> microsStr;
371
372 for(std::set<LinguisticCode>::iterator it = micros.begin(); it != micros.end(); ++it) {
373 std::string tag = microManager.getPropertySymbolicValue(*it);
374 microsStr.push_back(tag);
375 }
376
377 return microsStr;
378}
379
380std::set<LinguisticGraphVertex> DynamicSvmToolPosTagger::getPreviousTokens(LinguisticGraphVertex token, const LinguisticGraph *srcGraph) const {
381 std::set<LinguisticGraphVertex> previous;
382
383 LinguisticGraphInEdgeIt inItr,inItrEnd;
384 boost::tie(inItr,inItrEnd) = in_edges(token, *srcGraph);
385 for (;inItr!=inItrEnd;inItr++) {
386 previous.insert(source(*inItr, *srcGraph));
387 }
388
389 return previous;
390}
391
392std::vector<nodo*> DynamicSvmToolPosTagger::buildContext(
393 const LinguisticGraph *srcGraph,
394 LinguisticGraphVertex prevPrevVertex,
395 LinguisticGraphVertex prevVertex,
396 LinguisticGraphVertex vertex) const
397{
398 /* If we were to handle languages such as arabic, we would want to improve
399 * this part of the code. Indeed, the next token and the next next token
400 * don't mean much in languages such as arabic. Thus, the goal would be to
401 * "flatten" the next nodes: take every token with a distance of one, and
402 * create a single token with all the possibles microcategories. Same thing
403 * for tokens with a distance of two related to the current token.
404 */
405
406 std::string prevPrevWord = getWord(prevPrevVertex, srcGraph);
407 std::string prevWord = getWord(prevVertex, srcGraph);
408 std::string word = getWord(vertex, srcGraph);
409 std::string nextWord = getWord(nextToken(vertex, srcGraph), srcGraph);
410 std::string nextNextWord = getWord(nextToken(nextToken(vertex, srcGraph), srcGraph), srcGraph);
411
412 std::vector<nodo*> context;
413 for(int i = 0; i < 5; i++) {
414 context.push_back(NULL);
415 }
416
417 static int ord_id = 0;
418
419 if(prevPrevWord == "") {
420 context[0] = NULL;
421 } else {
422 context[0] = new nodo;
423 context[0]->ord = ord_id++;
424 context[0]->wrd = prevPrevWord;
425 context[0]->realWrd = prevPrevWord;
426 }
427
428 if(prevWord == "") {
429 context[1] = NULL;
430 } else {
431 context[1] = new nodo;
432 context[1]->ord = ord_id++;
433 context[1]->wrd = prevWord;
434 context[1]->realWrd = prevWord;
435 }
436
437 if(word == "") {
438 context[2] = NULL;
439 } else {
440 context[2] = new nodo;
441 context[2]->ord = ord_id++;
442 context[2]->wrd = word;
443 context[2]->realWrd = word;
444 }
445
446 if(nextWord == "") {
447 context[3] = NULL;
448 } else {
449 context[3] = new nodo;
450 context[3]->ord = ord_id++;
451 context[3]->wrd = nextWord;
452 context[3]->realWrd = nextWord;
453 }
454
455 if(nextNextWord == "") {
456 context[4] = NULL;
457 } else {
458 context[4] = new nodo;
459 context[4]->ord = ord_id++;
460 context[4]->wrd = nextNextWord;
461 context[4]->realWrd = nextNextWord;
462 }
463
464 return context;
465
466}
467
468/* Returns a word given its vertex in our treillis */
469std::string DynamicSvmToolPosTagger::getWord(LinguisticGraphVertex token, const LinguisticGraph* srcGraph) const {
470 if(token <= 1) {
471 return "";
472 } else {
473 std::string word = Common::Misc::limastring2utf8stdstring(get(vertex_token, *srcGraph, token)->stringForm());
474 boost::replace_all(word, " ", "_");
475 return word;
476 }
477}
478
479/* Returns the first following token of a given token */
480LinguisticGraphVertex DynamicSvmToolPosTagger::nextToken(LinguisticGraphVertex token, const LinguisticGraph* srcGraph) const {
481 std::set<LinguisticGraphVertex> tokens = nextTokens(token, srcGraph);
482 return tokens.empty() ? 1 : *(tokens.begin());
483}
484
485/* Return every token following a given token */
486std::set<LinguisticGraphVertex> DynamicSvmToolPosTagger::nextTokens(LinguisticGraphVertex token, const LinguisticGraph* srcGraph) const {
487 std::set<LinguisticGraphVertex> tokens;
488 LinguisticGraphOutEdgeIt outItr,outItrEnd;
489 boost::tie(outItr,outItrEnd)=out_edges(token,*srcGraph);
490 for(;outItr != outItrEnd; ++outItr) {
491 tokens.insert(target(*outItr, *srcGraph));
492 }
493
494 return tokens;
495
496}
497
498} // PosTagger
499
500} // LinguisticProcessing
501
502} // Lima
This file is the main header file for the data related to annotation graphs.
#define DYNAMICSVMTOOLPOSTAGGER_CLASSID
#define LWARN
Definition LimaCommon.h:160
#define LDEBUG
Definition LimaCommon.h:157
#define LERROR
Definition LimaCommon.h:161
A graph structure for linguistic analysis.
LinguisticGraph::in_edge_iterator LinguisticGraphInEdgeIt
boost::property_map< LinguisticGraph, vertex_data_t >::const_type CVertexDataPropertyMap
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define PTLOGINIT
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Holds an annotation graph and gives an API to manipulate it.
void addMatching(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second, AnnotationGraphVertex secondVx)
Adds a symetric matching between two vertices of two graphs identified by the two string parameters.
AnnotationGraphVertex createAnnotationVertex()
Creates a new annotation vertex in the graph.
Holds linguistic data for one language.
const PropertyCode::PropertyCodeManager & getPropertyCodeManager() const
const PropertyManager & getPropertyManager(const std::string &propertyName) const
Get the PropertyManager associated to a property.
Provide tools to manage a specific property.
const std::string & getPropertySymbolicValue(const LinguisticCode &value) const
The coded property value can hold several property data.
std::deque< std::string > & getListsValueAtKey(const std::string &key)
return a message when a 'param' was not found
Defines Factory for an Initializable Object.
Manage initialization of InitializableObjects using configuration module and parameters.
Use this exception to signal an error in one of the configuration files.
Definition LimaCommon.h:345
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
This file contains a class to control log of informations about time, such as logging cumulated time ...
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
void annotate(AnnotationGraphVertex v, const LimaString &annot, uint64_t value)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
launch exception related to the configuration file parsing