29#include "svmtool/tagger.h"
30#include "svmtool/nodo.h"
32#include <boost/algorithm/string/split.hpp>
33#include <boost/algorithm/string/classification.hpp>
34#include <boost/algorithm/string/replace.hpp>
35#include <boost/tuple/tuple.hpp>
36#include <boost/tuple/tuple_io.hpp>
45namespace LinguisticProcessing
53DynamicSvmToolPosTaggerFactory::DynamicSvmToolPosTaggerFactory(
const std::string&
id) :
57std::shared_ptr<MediaProcessUnit> DynamicSvmToolPosTaggerFactory::create(
61 auto posTagger = std::make_shared<DynamicSvmToolPosTagger>();
62 posTagger->init(unitConfiguration,manager);
68void DynamicSvmToolPosTagger::init(
88 m_language=manager->getInitializationParameters().media;
91 m_microAccessor=&(m_MicroManager->getPropertyAccessor());
92 string resourcesPath=MediaticData::single().getResourcesPath();
96 std::string defaultName;
103 LWARN <<
"No default microtageory for DynamicSvmToolPosTagger! using PONCTU_FORTE.";
104 defaultName =
"PONCTU_FORTE";
106 m_defaultCateg = m_MicroManager->getPropertyValue(defaultName);
113 for (std::deque<std::string>::iterator it=cats.begin();
117 m_stopCategories.push_back(m_MicroManager->getPropertyValue(*it));
122 LWARN <<
"No stop categories defined! using the default category";
123 m_stopCategories.push_back(m_defaultCateg);
131 LERROR <<
"No SVMTool model defined in the configuration file!";
138 t =
new tagger(Common::Misc::findFileInPaths(resourcesPath.c_str(), model.c_str()).toUtf8().constData());
139 t->taggerLoadModelsForTagging();
140 t->taggerShowComments();
141 t->taggerActiveShowScoresFlag();
170 (analysis.
getData(
"AnalysisGraph").get());
171 srcGraph = analysisGraph->
getGraph();
180 std::queue<LinguisticGraphVertex> tokenQueue;
181 std::set<LinguisticGraphVertex> visited;
182 std::map<LinguisticGraphVertex, struct PathInfo > maxAncestor;
186 nextTokens(analysisGraph->
firstVertex(), srcGraph))
188 tokenQueue.push(vertex);
191 posGraph =
new LinguisticAnalysisStructure
192 ::AnalysisGraph(
"PosGraph", m_language,
false,
true);
193 analysis.
setData(
"PosGraph", posGraph);
196 while (!tokenQueue.empty()) {
199 LDEBUG <<
"\n" << vertex <<
" -> " << getWord(vertex, srcGraph);
202 double logMaxWeight = -LDBL_MAX;
206 std::set<LinguisticGraphVertex> previousTokens = getPreviousTokens(vertex, srcGraph);
207 if(previousTokens.empty()) previousTokens.insert(posGraph->
firstVertex());
208 for (
auto it = previousTokens.begin(); it != previousTokens.end(); ++it) {
211 std::string pos =
"";
212 double logCurWeight = log(1.0), w;
215 boost::tie(pos, w) = SVMTool(srcGraph, vertex, prevVertex, maxAncestor);
216 logCurWeight = log(w);
217 LDEBUG <<
"weight = " << logCurWeight <<
" -> " << pos <<
"(" << w <<
")";
222 double logPrevPrice = log(1.0);
224 if(maxAncestor.find(prevVertex) != maxAncestor.end()) {
225 struct PathInfo prevPath = maxAncestor[prevVertex];
226 logPrevPrice = prevPath.score;
227 prevLength = prevPath.pathLength;
231 if((logPrevPrice + logCurWeight) / (prevLength+1)
232 > (logMaxWeight / (maxLength+1))) {
234 struct PathInfo currentPath = { prevVertex, logCurWeight + logPrevPrice, pos, prevLength+1 };
235 maxAncestor[vertex] = currentPath;
236 logMaxWeight = logCurWeight + logPrevPrice;
237 maxLength = prevLength;
238 LDEBUG <<
" -> " << logMaxWeight <<
" (" << maxLength <<
")";
242 LDEBUG << getWord(vertex, srcGraph) <<
" -> " << maxAncestor[vertex].pos;
247 boost::tie(outItr,outItrEnd)=out_edges(vertex,*srcGraph);
249 for (;outItr!=outItrEnd;outItr++) {
252 if (visited.find(nextToken) == visited.end()) {
253 tokenQueue.push(nextToken);
254 visited.insert(nextToken);
260 std::stack<boost::tuple<LinguisticGraphVertex, LinguisticCode> > chosenPath;
262 while ((backVertex = maxAncestor[backVertex].prev) != 0) {
263 LinguisticCode categ = m_MicroManager->getPropertyValue(maxAncestor[backVertex].pos);
264 chosenPath.push(boost::make_tuple(backVertex, categ));
278 while(!chosenPath.empty()) {
280 boost::tie(vertex, code) = chosenPath.top(); chosenPath.pop();
283 LDEBUG <<
"create vertex " << newVertex;
284 annotationData->
addMatching(
"PosGraph", newVertex,
"annot", vertex);
285 annotationData->
addMatching(
"AnalysisGraph", vertex,
"PosGraph", newVertex);
287 annotationData->
annotate(annotVertex, Common::Misc::utf8stdstring2limastring(
"PosGraph"), newVertex);
296 std::back_insert_iterator<LinguisticAnalysisStructure::MorphoSyntacticData> backInsertItr(*posData);
297 remove_copy_if(morphoData->begin(),morphoData->end(),backInsertItr,differentMicro);
298 if (posData->empty() || morphoData->empty()) {
299 LWARN <<
"No matching category found for tagger result " << getWord(vertex, srcGraph) <<
" " << m_MicroManager->getPropertySymbolicValue(code);
300 if (!morphoData->empty())
302 LWARN <<
"Taking any one";
303 posData->push_back(morphoData->front());
310 boost::add_edge(previousPosVertex, newVertex, *resultGraph);
313 LDEBUG << getWord(vertex, srcGraph) <<
" -> " << m_MicroManager->getPropertySymbolicValue(code);
314 previousPosVertex = newVertex;
317 boost::add_edge(previousPosVertex, posGraph->
lastVertex(), *resultGraph);
322boost::tuple<std::string, uint64_t> DynamicSvmToolPosTagger::SVMTool(
326 std::map<LinguisticGraphVertex, struct PathInfo > &maxAncestor)
const {
330 if(maxAncestor.find(prevVertex) != maxAncestor.end()) {
331 prevPrevVertex = maxAncestor[prevVertex].prev;
333 auto node_context = buildContext(srcGraph, prevPrevVertex, prevVertex, vertex);
334 std::vector<std::string> microsStr = getMicros(vertex, srcGraph);
336 if (microsStr.empty()) {
337 LERROR << getWord(vertex, srcGraph) <<
" has no attached microcategories";
338 return boost::make_tuple(
"", LDBL_MIN);
343 t->sw->setWindow(node_context);
344 t->setPossibles(microsStr);
346 t->taggerGenerateScore(node_context[2],1);
349 node_context[2]->weight += 10.0;
350 node_context[2]->weight /= 20.0;
352 LDEBUG <<
"§" << node_context[2]->pos <<
"|" << (float)(node_context[2]->weight) <<
"§";
354 return boost::make_tuple(node_context[2]->pos, node_context[2]->weight);
360 LinguisticAnalysisStructure::MorphoSyntacticData *mdata = dataMap[token];
363 std::set<LinguisticCode> micros;
365 micros.insert(m_defaultCateg);
367 mdata->allValues(*m_microAccessor, micros);
370 std::vector<std::string> microsStr;
372 for(std::set<LinguisticCode>::iterator it = micros.begin(); it != micros.end(); ++it) {
374 microsStr.push_back(tag);
381 std::set<LinguisticGraphVertex> previous;
384 boost::tie(inItr,inItrEnd) = in_edges(token, *srcGraph);
385 for (;inItr!=inItrEnd;inItr++) {
386 previous.insert(source(*inItr, *srcGraph));
392std::vector<nodo*> DynamicSvmToolPosTagger::buildContext(
406 std::string prevPrevWord = getWord(prevPrevVertex, srcGraph);
407 std::string prevWord = getWord(prevVertex, srcGraph);
408 std::string word = getWord(vertex, srcGraph);
409 std::string nextWord = getWord(nextToken(vertex, srcGraph), srcGraph);
410 std::string nextNextWord = getWord(nextToken(nextToken(vertex, srcGraph), srcGraph), srcGraph);
412 std::vector<nodo*> context;
413 for(
int i = 0; i < 5; i++) {
414 context.push_back(NULL);
417 static int ord_id = 0;
419 if(prevPrevWord ==
"") {
422 context[0] =
new nodo;
423 context[0]->ord = ord_id++;
424 context[0]->wrd = prevPrevWord;
425 context[0]->realWrd = prevPrevWord;
431 context[1] =
new nodo;
432 context[1]->ord = ord_id++;
433 context[1]->wrd = prevWord;
434 context[1]->realWrd = prevWord;
440 context[2] =
new nodo;
441 context[2]->ord = ord_id++;
442 context[2]->wrd = word;
443 context[2]->realWrd = word;
449 context[3] =
new nodo;
450 context[3]->ord = ord_id++;
451 context[3]->wrd = nextWord;
452 context[3]->realWrd = nextWord;
455 if(nextNextWord ==
"") {
458 context[4] =
new nodo;
459 context[4]->ord = ord_id++;
460 context[4]->wrd = nextNextWord;
461 context[4]->realWrd = nextNextWord;
473 std::string word = Common::Misc::limastring2utf8stdstring(get(
vertex_token, *srcGraph, token)->stringForm());
474 boost::replace_all(word,
" ",
"_");
481 std::set<LinguisticGraphVertex> tokens = nextTokens(token, srcGraph);
482 return tokens.empty() ? 1 : *(tokens.begin());
487 std::set<LinguisticGraphVertex> tokens;
489 boost::tie(outItr,outItrEnd)=out_edges(token,*srcGraph);
490 for(;outItr != outItrEnd; ++outItr) {
491 tokens.insert(target(*outItr, *srcGraph));
This file is the main header file for the data related to annotation graphs.
#define DYNAMICSVMTOOLPOSTAGGER_CLASSID
A graph structure for linguistic analysis.
LinguisticGraph::in_edge_iterator LinguisticGraphInEdgeIt
boost::property_map< LinguisticGraph, vertex_data_t >::const_type CVertexDataPropertyMap
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Holds an annotation graph and gives an API to manipulate it.
void addMatching(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second, AnnotationGraphVertex secondVx)
Adds a symetric matching between two vertices of two graphs identified by the two string parameters.
AnnotationGraphVertex createAnnotationVertex()
Creates a new annotation vertex in the graph.
const PropertyManager & getPropertyManager(const std::string &propertyName) const
Get the PropertyManager associated to a property.
Provide tools to manage a specific property.
const std::string & getPropertySymbolicValue(const LinguisticCode &value) const
The coded property value can hold several property data.
std::string & getParamsValueAtKey(const std::string &key)
std::deque< std::string > & getListsValueAtKey(const std::string &key)
return a message when a 'param' was not found
Defines Factory for an Initializable Object.
Manage initialization of InitializableObjects using configuration module and parameters.
Use this exception to signal an error in one of the configuration files.
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
Holds morphosyntactic informations.
holds surface data of a token
This file contains a class to control log of informations about time, such as logging cumulated time ...
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
void annotate(AnnotationGraphVertex v, const LimaString &annot, uint64_t value)
launch exception related to the configuration file parsing