LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
BowGeneration.cpp
Go to the documentation of this file.
1// Copyright 2002-2020 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/***************************************************************************
7 * Copyright (C) 2004-2020 by CEA LIST *
8 * *
9 ***************************************************************************/
10
11#include "BowGeneration.h"
38
39#include <boost/graph/properties.hpp>
40
41#include <fstream>
42#include <deque>
43#include <iostream>
44#include <QStringList>
45
46
48using namespace Lima::Common::MediaticData;
49using namespace Lima::Common::BagOfWords;
50using namespace Lima::Common::AnnotationGraphs;
53// using namespace Lima::LinguisticProcessing::Compounds;
58
59namespace Lima
60{
61
62namespace LinguisticProcessing
63{
64
65namespace Compounds
66{
67
68 struct A
69 {
70 bool operator()(const std::pair<boost::shared_ptr< Common::BagOfWords::BoWRelation >, boost::shared_ptr< Common::BagOfWords::BoWToken > > t1, const std::pair<boost::shared_ptr< Common::BagOfWords::BoWRelation >, boost::shared_ptr< Common::BagOfWords::BoWToken > > t2) const
71 {
72 return (t1.second->getPosition()<t2.second->getPosition()) ;
73 }
74 };
75
76
78{
79 friend class BowGenerator;
80
82
83 ~BowGeneratorPrivate() = default;
85 BowGeneratorPrivate& operator=(const BowGeneratorPrivate&) = delete;
86
87 std::vector< std::pair< boost::shared_ptr< Common::BagOfWords::BoWRelation >,
88 boost::shared_ptr< Common::BagOfWords::AbstractBoWElement > > >
89 createAbstractBoWElement(
91 const LinguisticGraph& anagraph,
92 const LinguisticGraph& posgraph,
93 const uint64_t offsetBegin,
94 const Common::AnnotationGraphs::AnnotationData* annotationData,
95 std::set<LinguisticGraphVertex>& visited,
96 bool keepAnyway = false) const;
97
98 boost::shared_ptr< Common::BagOfWords::BoWRelation > createBoWRelationFor(
99 const AnnotationGraphVertex& vx,
100 const AnnotationGraphVertex& tgt,
101 const Common::AnnotationGraphs::AnnotationData* annotationData,
102 const LinguisticGraph& posgraph,
103 const SyntacticAnalysis::SyntacticData* syntacticData) const;
104
105 class NamedEntityPart
106 {
107 public:
108 NamedEntityPart(): inflectedForm(), lemma(), position(0),
109 length(0) {}
110 NamedEntityPart(const LimaString& fl, const LimaString& l,
111 const LinguisticCode cat, const uint64_t pos,
112 const uint64_t len):
113 inflectedForm(fl), lemma(l), category(cat), position (pos),
114 length(len) {}
115
116 LimaString inflectedForm;
117 LimaString lemma;
118 LinguisticCode category;
119 uint64_t position;
120 uint64_t length;
121 };
122
123 typedef std::set< std::pair<uint64_t,uint64_t> > TokenPositions;
124
125 MediaId m_language;
126 std::shared_ptr<AnalysisDumpers::StopList> m_stopList;
127 bool m_useStopList;
128 bool m_useEmptyMacro;
129 bool m_useEmptyMicro;
130 LinguisticCode m_properNounCategory;
131 LinguisticCode m_commonNounCategory;
132 bool m_keepAllNamedEntityParts;
133 const Common::PropertyCode::PropertyAccessor* m_macroAccessor;
134 const Common::PropertyCode::PropertyAccessor* m_microAccessor;
135
136 // what to assign to the BoWNamedEntity lemma
137 typedef enum {
138 NORMALIZE_NE_INFLECTED, // assign inflected form
139 NORMALIZE_NE_LEMMA, // assign lemma
140 NORMALIZE_NE_NORMALIZEDFORM, // assign normalized form from NE
141 NORMALIZE_NE_NETYPE // assign type of NE
142 } NENormalization;
143 NENormalization m_NEnormalization;
144
145 boost::shared_ptr< Common::BagOfWords::BoWNamedEntity > createSpecificEntity(
146 const LinguisticGraphVertex& vertex,
147 const AnnotationGraphVertex& v,
148 const Common::AnnotationGraphs::AnnotationData* annotationData,
149 const LinguisticGraph& anagraph,
150 const LinguisticGraph& posgraph,
151 const uint64_t offset,
152 bool frompos = true) const;
153
154 boost::shared_ptr< Common::BagOfWords::BoWToken > createCompoundTense(
155 const AnnotationGraphVertex& v,
156 const Common::AnnotationGraphs::AnnotationData* annotationData,
157 const LinguisticGraph& anagraph,
158 const LinguisticGraph& posgraph,
159 const uint64_t offset,
160 std::set<LinguisticGraphVertex>& visited) const;
161
162// Common::BagOfWords::BoWPredicate* createPredicate(const Common::MediaticData::EntityType& t, QMultiMap<Common::MediaticData::EntityType, Common::BagOfWords::AbstractBoWElement*> roles) const;
163
167 QList< boost::shared_ptr< Common::BagOfWords::BoWPredicate > > createPredicate(
168 const LinguisticGraphVertex& lgv,
169 const AnnotationGraphVertex& agv,
170 const Common::AnnotationGraphs::AnnotationData* annotationData,
171 const LinguisticGraph& anagraph,
172 const LinguisticGraph& posgraph,
173 const uint64_t offset,
174 std::set<LinguisticGraphVertex>& visited,
175 bool keepAnyway)const;
176
177 bool checkStopWordInCompound(
178 boost::shared_ptr< Common::BagOfWords::BoWToken>&,
179 uint64_t offset,
180 std::set< std::string >& alreadyStored,
181 Common::BagOfWords::BoWText& bowText) const;
182
183 StringsPoolIndex getNamedEntityNormalization(
184 const AnnotationGraphVertex& v,
185 const Common::AnnotationGraphs::AnnotationData* annotationData) const;
186
187 bool shouldBeKept(const LinguisticAnalysisStructure::LinguisticElement& elem) const;
188
199 std::vector<NamedEntityPart> createNEParts(
200 const AnnotationGraphVertex& v,
201 const Common::AnnotationGraphs::AnnotationData* annotationData,
202 const LinguisticGraph& anagraph,
203 const LinguisticGraph& posgraph,
204 bool frompos = true) const;
205
206 void bowTokenPositions(TokenPositions& res,
207 const boost::shared_ptr< Common::BagOfWords::BoWToken > tok) const;
208
209 uint64_t computeCompoundLength(const TokenPositions& headTokPositions,
210 const TokenPositions& extensionPositions) const;
211};
212
213BowGeneratorPrivate::BowGeneratorPrivate():
214 m_language(0),
215 m_stopList(0),
216 m_useStopList(true),
217 m_useEmptyMacro(true),
218 m_useEmptyMicro(true),
219 m_keepAllNamedEntityParts(false),
220 m_macroAccessor(0),
221 m_microAccessor(0),
222 m_NEnormalization(NORMALIZE_NE_INFLECTED)
223{
224}
225
230
232{
233 delete m_d;
234}
235
238 MediaId language)
239{
241 m_d->m_language=language;
242 m_d->m_macroAccessor=&static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_d->m_language)).getPropertyCodeManager().getPropertyAccessor("MACRO");
243 m_d->m_microAccessor=&static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_d->m_language)).getPropertyCodeManager().getPropertyAccessor("MICRO");
244 try
245 {
246 auto use = unitConfiguration.getParamsValueAtKey("useStopList");
247 m_d->m_useStopList = (use=="true");
248 }
249 catch (NoSuchParam& )
250 {
251 LWARN << "No param 'useStopList' in"<<unitConfiguration.getName()
252 << "configuration group for language " << (int)m_d->m_language;
253 LWARN << "use default value : true";
254 }
255 if (m_d->m_useStopList)
256 {
257 try
258 {
259 auto stoplist = unitConfiguration.getParamsValueAtKey("stopList");
260 m_d->m_stopList = std::dynamic_pointer_cast<StopList>(LinguisticResources::single().getResource(m_d->m_language, stoplist));
261#ifdef DEBUG_LP
262 LDEBUG << "BowGenerator.init(): STOPLIST:";
263 for(const auto& word: *m_d->m_stopList)
264 {
265 LDEBUG << "BowGenerator.init(): " << word;
266 }
267#endif
268 }
269 catch (NoSuchParam& )
270 {
271 LWARN << "No param 'stopList' in" << unitConfiguration.getName()
272 << "configuration group for language "
273 << (int)m_d->m_language;
274// throw InvalidConfiguration();
275 }
276 }
277 try
278 {
279 auto use = unitConfiguration.getParamsValueAtKey("useEmptyMacro");
280 m_d->m_useEmptyMacro = (use=="true");
281 }
282 catch (NoSuchParam& )
283 {
284 LWARN << "No param 'useEmptyMacro' in" << unitConfiguration.getName()
285 << "configuration group for language " << (int)m_d->m_language;
286 LWARN << "use default value : true";
287 }
288 try
289 {
290 auto use = unitConfiguration.getParamsValueAtKey("useEmptyMicro");
291 m_d->m_useEmptyMicro = (use=="true");
292 }
293 catch (NoSuchParam& )
294 {
295 LWARN << "No param 'useEmptyMicro' in" << unitConfiguration.getName()
296 << "configuration group for language " << (int)m_d->m_language;
297 LWARN << "use default value : true";
298 }
299 try
300 {
301 auto np = unitConfiguration.getParamsValueAtKey("properNounCategory");
302 m_d->m_properNounCategory = static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_d->m_language)).getPropertyCodeManager().getPropertyManager("MACRO").getPropertyValue(np);
303 }
304 catch (NoSuchParam& )
305 {
306 LERROR << "No param 'properNounCategory' in" << unitConfiguration.getName()
307 << "configuration group for language " << (int)m_d->m_language;
308// throw InvalidConfiguration();
309 }
310 try
311 {
312 auto cn = unitConfiguration.getParamsValueAtKey("commonNounCategory");
313 m_d->m_commonNounCategory = static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_d->m_language)).getPropertyCodeManager().getPropertyManager("MACRO").getPropertyValue(cn);
314 }
315 catch (NoSuchParam& )
316 {
317 LERROR << "No param 'commonNounCategory' in" << unitConfiguration.getName()
318 << "configuration group for language " << (int)m_d->m_language;
319// throw InvalidConfiguration();
320 }
321
322 try
323 {
324 auto value = unitConfiguration.getParamsValueAtKey("keepAllNamedEntityParts");
325 m_d->m_keepAllNamedEntityParts = (value == "yes" || value == "true" || value == "1");
326 }
327 catch (NoSuchParam& ) { /* optional */ }
328
329 try
330 {
331 auto value = unitConfiguration.getParamsValueAtKey("NEnormalization");
332 if (value == "useInflectedForm")
333 {
334 m_d->m_NEnormalization = BowGeneratorPrivate::NORMALIZE_NE_INFLECTED;
335 }
336 else if (value == "useLemma")
337 {
338 m_d->m_NEnormalization = BowGeneratorPrivate::NORMALIZE_NE_LEMMA;
339 }
340 else if (value == "useNENormalizedForm")
341 {
342 m_d->m_NEnormalization = BowGeneratorPrivate::NORMALIZE_NE_NORMALIZEDFORM;
343 }
344 else if (value == "useNEType")
345 {
346 m_d->m_NEnormalization = BowGeneratorPrivate::NORMALIZE_NE_NETYPE;
347 }
348 }
349 catch (NoSuchParam& ) { /* optional */ }
350}
351
352
353std::vector< std::pair< boost::shared_ptr< BoWRelation >, boost::shared_ptr< BoWToken > > > BowGenerator::buildTermFor(
354 const AnnotationGraphVertex& vx,
355 const AnnotationGraphVertex& tgt,
356 const LinguisticGraph& anagraph,
357 const LinguisticGraph& posgraph,
358 const uint64_t offset,
359 const SyntacticData* syntacticData,
360 const AnnotationData* annotationData,
361 std::set< LinguisticGraphVertex >& visited) const
362{
363
364#ifdef DEBUG_LP
366 LDEBUG << "BowGenerator::buildTermFor annot:" << vx << "; pointing on annot:"<<tgt;
367#endif
368
369 LinguisticGraphVertex vxTokVertex =
370 *(annotationData->matches("cpd", vx, "PosGraph").begin());
371
372 // recuperation des noeuds tetes de relations pointant sur vx
373 std::vector<AnnotationGraphVertex> vxGovernors;
374 AnnotationGraphInEdgeIt inIt, inIt_end;
375 boost::tie(inIt, inIt_end) = boost::in_edges(vx, annotationData->getGraph());
376 for (; inIt != inIt_end; inIt++)
377 {
378 vxGovernors.push_back(source(*inIt, annotationData->getGraph()));
379 }
380
381 auto vxBoWTokens = createAbstractBoWElement(vxTokVertex, anagraph, posgraph,
382 offset, annotationData, visited);
383
384// #ifdef DEBUG_LP
385// LDEBUG << "BowGenerator::buildTermFor, line"<<__LINE__<<","<<vx<<vxTokVertex<<" There is " << vxBoWTokens.size() << " bow tokens";
386// #endif
387 if (vxGovernors.empty())
388 {
389
390 std::vector< std::pair<boost::shared_ptr< BoWRelation >,
391 boost::shared_ptr< BoWToken > > > vxBoWTk;
392// #ifdef DEBUG_LP
393// LDEBUG << "BowGenerator::buildTermFor empty governors ";
394// #endif
395 auto relation = m_d->createBoWRelationFor(vx, tgt, annotationData,
396 posgraph, syntacticData);
397
398 if (relation)
399 {
400 for (auto& vxBoWToken: vxBoWTokens)
401 {
402 vxBoWToken.first = relation;
403 }
404 }
405 for (auto& vxBoWToken: vxBoWTokens)
406 {
407 if (boost::dynamic_pointer_cast<BoWToken>(vxBoWToken.second) != 0)
408 {
409 vxBoWTk.push_back(std::make_pair(vxBoWToken.first,
410 boost::dynamic_pointer_cast<BoWToken>(vxBoWToken.second)));
411 }
412 else
413 {
415 LERROR << "BowGenerator::buildTermFor vxBoWTokensIt is not a BoWToken";
416 }
417 }
418// #ifdef DEBUG_LP
419// LDEBUG << "BowGenerator::buildTermFor == DONE buildTermFor " << vx << " (pointing on "<<tgt<<"):empty governors ";
420// LDEBUG << "BowGenerator::buildTermFor return result of size" << vxBoWTk.size();
421// #endif
422 return vxBoWTk;
423 }
424
425 std::vector< std::pair< boost::shared_ptr< BoWRelation >, boost::shared_ptr< BoWToken > > > result;
426 std::vector< std::vector< std::pair< boost::shared_ptr< BoWRelation > , boost::shared_ptr< BoWToken > > > > termsForVxGovernors;
427 for (const auto& gov: vxGovernors)
428 {
429 auto pairs = buildTermFor(gov, vx, anagraph, posgraph, offset,
430 syntacticData, annotationData, visited);
431 termsForVxGovernors.push_back(pairs);
432// #ifdef DEBUG_LP
433// LDEBUG << "BowGenerator::buildTermFor"<<vx<<vxTokVertex<<" For governor "
434// << i <<*govsIt<< ", there is " << pairs.size() << " terms.";
435// #endif
436 }
437
438
439 std::vector< std::vector< std::pair< boost::shared_ptr< BoWRelation >, boost::shared_ptr< BoWToken > > >::iterator > begins;
440 std::vector< std::vector< std::pair< boost::shared_ptr< BoWRelation >, boost::shared_ptr< BoWToken > > >::iterator > ends;
441 std::vector< std::vector< std::pair< boost::shared_ptr< BoWRelation >, boost::shared_ptr< BoWToken > > >::iterator > stack;
442 for (auto termsForVxGovernorsIt = termsForVxGovernors.begin();
443 termsForVxGovernorsIt != termsForVxGovernors.end();
444 termsForVxGovernorsIt++)
445 {
446 if (!(*termsForVxGovernorsIt).empty())
447 {
448 begins.push_back((*termsForVxGovernorsIt).begin());
449 ends.push_back((*termsForVxGovernorsIt).end());
450 stack.push_back((*termsForVxGovernorsIt).begin());
451 }
452 }
453
454 if (stack.empty())
455 {
456// #ifdef DEBUG_LP
457// LDEBUG << "BowGenerator::buildTermFor Stack is empty ! Returning bow tokens of " << vxTokVertex;
458// LDEBUG << "BowGenerator::buildTermFor == DONE buildTermFor " << vx << " (pointing on "<<tgt<<"):stack governors ";
459// #endif
460 auto relation = m_d->createBoWRelationFor(vx, tgt, annotationData,
461 posgraph, syntacticData);
462 if (relation)
463 {
464 for (auto& vxBoWToken: vxBoWTokens)
465 {
466 vxBoWToken.first = relation;
467 }
468 }
469 std::vector< std::pair< boost::shared_ptr< BoWRelation >, boost::shared_ptr< BoWToken > > > vxBoWTk;
470 auto vxBoWTokensIt = vxBoWTokens.begin(), vxBoWTokensIt_end = vxBoWTokens.end();
471 for (; vxBoWTokensIt != vxBoWTokensIt_end; vxBoWTokensIt++)
472 {
473 if (boost::dynamic_pointer_cast<BoWToken>((*vxBoWTokensIt).second) != 0)
474 {
475 vxBoWTk.push_back(std::make_pair((*vxBoWTokensIt).first,boost::dynamic_pointer_cast<BoWToken>((*vxBoWTokensIt).second)));
476 }
477 else
478 {
480 LERROR << "BowGenerator::buildTermFor vxBoWTokensIt is not a BoWToken";
481 }
482 }
483// #ifdef DEBUG_LP
484// LDEBUG << "BowGenerator::buildTermFor return result of size" << vxBoWTk.size();
485// #endif
486 return vxBoWTk;
487 }
488 std::vector< std::pair< boost::shared_ptr< BoWRelation >, boost::shared_ptr< BoWToken > > >::iterator t;
489 while (!stack.empty())
490 {
491// #ifdef DEBUG_LP
492// LDEBUG << "BowGenerator::buildTermFor There is " << vxBoWTokens.size() << " heads, " << vxGovernors.size() << " governors and stack size is " << stack.size();
493// #endif
494 for (auto vxBoWToken=vxBoWTokens.begin();
495 vxBoWToken!=vxBoWTokens.end();
496 vxBoWToken++)
497 {
498 boost::shared_ptr< BoWToken > head = boost::dynamic_pointer_cast<BoWToken>((*vxBoWToken).second);
499 if (head == 0)
500 {
502 LERROR << "BowGenerator::buildTermFor head" << &*(*vxBoWToken).second << "is not a BoWToken";
503 continue;
504 }
505// #ifdef DEBUG_LP
506// LDEBUG << "BowGenerator::buildTermFor Working on head " << *head << "(" << *head << ")";
507// #endif
508
509 std::set< std::pair< boost::shared_ptr< BoWRelation >, boost::shared_ptr< BoWToken > >, A > extensions;
510 auto govsIt = stack.begin(), govsIt_end = stack.end();
511 for (; govsIt != govsIt_end; govsIt++)
512 {
513// #ifdef DEBUG_LP
514// LDEBUG << "BowGenerator::buildTermFor Entering loop body";
515// #endif
516 boost::shared_ptr< BoWRelation > relation( (**govsIt).first);
517 boost::shared_ptr< BoWToken > bt = (**govsIt).second;
518// #ifdef DEBUG_LP
519// LDEBUG << "BowGenerator::buildTermFor ... done.";
520// #endif
521// LDEBUG << "BowGenerator: Inserting extension...";
522 extensions.insert(std::make_pair(relation,bt));
523// LDEBUG << "BowGenerator: ... done.";
524 }
525
526 LimaString lemma;
527 LimaString infl;
528 uint64_t position=0;
529 uint64_t length=0;
530 BowGeneratorPrivate::TokenPositions headPositions;
531 m_d->bowTokenPositions(headPositions, head);
532 BowGeneratorPrivate::TokenPositions extensionPositions;
533
534// #ifdef DEBUG_LP
535// LDEBUG << "BowGenerator::buildTermFor Working on extensions";
536// #endif
537 auto extensionsIt = extensions.begin(),
538 extensionsIt_end = extensions.end();
539 for (; extensionsIt != extensionsIt_end; extensionsIt++)
540 {
541 boost::shared_ptr< BoWToken > extension = (*extensionsIt).second;
542// #ifdef DEBUG_LP
543// LDEBUG << "BowGenerator::buildTermFor extension: " << *extension;
544// LDEBUG << "BowGenerator::buildTermFor extension: " << ((boost::dynamic_pointer_cast< BoWTerm >(extension)==0)?(*extension):(*(boost::dynamic_pointer_cast< BoWTerm >(extension))));
545// #endif
546 m_d->bowTokenPositions(extensionPositions, extension);
547 }
548
549// #ifdef DEBUG_LP
550// LDEBUG << "BowGenerator::buildTermFor Building term";
551// #endif
552 // position is the min of head min position and extension min position
553 position=headPositions.begin()->first;
554 if (position > extensionPositions.begin()->first)
555 {
556 position = extensionPositions.begin()->first;
557 }
558// #ifdef DEBUG_LP
559// LDEBUG << "BowGenerator::buildTermFor position: " << position;
560// #endif
561
562 // length is the length in original text: take end of term
563 length=m_d->computeCompoundLength(headPositions,extensionPositions);
564// #ifdef DEBUG_LP
565// LDEBUG << "BowGenerator::buildTermFor length : " << length;
566// #endif
567
568 boost::shared_ptr< BoWTerm > complex( new BoWTerm( lemma, head->getCategory(), position, length) );
569 complex->setVertex(head->getVertex());
570 complex->setInflectedForm(infl);
571 complex->setCategory(head->getCategory());
572 complex->addPart(head);
573// delete head; head = 0;
574
575 extensionsIt = extensions.begin();
576 extensionsIt_end = extensions.end();
577 for (; extensionsIt != extensionsIt_end; extensionsIt++)
578 {
579 boost::shared_ptr< BoWToken > extension = (*extensionsIt).second;
580// #ifdef DEBUG_LP
581// LDEBUG << "BowGenerator::buildTermFor extension: " << ((boost::dynamic_pointer_cast< BoWTerm >(extension)==0)?(*extension):(*(boost::dynamic_pointer_cast< BoWTerm >(extension))));
582// #endif
583 if ((*extensionsIt).first == 0)
584 complex->addPart(extension);
585 else
586 complex->addPart((*extensionsIt).first,extension);
587// LDEBUG << "Built complex: " << ((dynamic_cast< BoWTerm* >(complex)==0)?(*complex):(*(dynamic_cast< BoWTerm* >(complex))));
588// #ifdef DEBUG_LP
589// LDEBUG << "BowGenerator::buildTermFor Built complex: " << *complex;
590// #endif
591 }
592
593 auto relation = m_d->createBoWRelationFor(vx, tgt, annotationData,
594 posgraph, syntacticData);
595
596// #ifdef DEBUG_LP
597// LDEBUG << "BowGenerator::buildTermFor Filling result with: " << *complex;
598// #endif
599 result.push_back(std::make_pair(relation,complex));
600 }
601
602 // Mise a joueur de la pile d'iterateurs pour produire une nouvelle serie d'extensions
603// #ifdef DEBUG_LP
604// LDEBUG << "BowGenerator::buildTermFor Stack updating...";
605// #endif
606 {
607 t = *stack.rbegin();
608 t++;
609 while ( t == ends[stack.size()-1] )
610 {
611 stack.pop_back();
612 if (stack.empty())
613 {
614 break;
615 }
616 t = *stack.rbegin();
617 t++;
618 }
619 if (!stack.empty())
620 {
621 stack.pop_back();
622 stack.push_back(t);
623// #ifdef DEBUG_LP
624// LDEBUG << "BowGenerator::buildTermFor Stack filling...";
625// #endif
626 for (uint64_t i = stack.size(); i < begins.size();i++)
627 {
628 stack.push_back(begins[i]);
629 }
630 }
631 }
632 }
633
634// #ifdef DEBUG_LP
635// LDEBUG << "BowGenerator::buildTermFor == DONE" << vx << "(pointing on" << tgt << ")";
636// LDEBUG << "BowGenerator::buildTermFor, line"<<__LINE__<<", return result of size" << result.size();
637// #endif
638 return result;
639}
640
641boost::shared_ptr< BoWRelation > BowGeneratorPrivate::createBoWRelationFor(
642 const AnnotationGraphVertex& vx,
643 const AnnotationGraphVertex& tgt,
644 const AnnotationData* annotationData,
645 const LinguisticGraph& posgraph,
646 const SyntacticData* syntacticData) const
647{
648 LIMA_UNUSED(posgraph);
649 const DependencyGraph* depGraph = syntacticData->dependencyGraph();
650#ifdef DEBUG_LP
652 LDEBUG << "BowGenerator::createBoWRelationFor" << vx << tgt;
653#endif
654 boost::shared_ptr<BoWRelation> relation;
655 if (vx != tgt
656 && annotationData->hasAnnotation(vx, tgt,
657 Common::Misc::utf8stdstring2limastring("CompoundTokenAnnotation")) )
658 {
659#ifdef DEBUG_LP
660 LDEBUG << "BowGenerator: working on relation";
661#endif
662 const CompoundTokenAnnotation* annot = annotationData->annotation(vx,tgt, Common::Misc::utf8stdstring2limastring("CompoundTokenAnnotation")).pointerValue<CompoundTokenAnnotation>();
663 if (annot != 0 && !annot->empty())
664 {
665 const ConceptModifier& modifier = (*annot)[0];
666 StringsPoolIndex realizationIdx = modifier.getRealization();
667 LimaString realization = Common::MediaticData::MediaticData::changeable().stringsPool(m_language)[realizationIdx];
668 int type = modifier.getConceptType();
669 relation = boost::shared_ptr< BoWRelation >(new BoWRelation(realization, type));
670 }
671 else
672 {
673 relation = boost::shared_ptr< BoWRelation >(new BoWRelation());
674 }
675 LinguisticGraphVertex vxTokVertex = *(annotationData->matches("cpd", vx, "PosGraph").begin());
676 LinguisticGraphVertex tgtTokVertex = *(annotationData->matches("cpd", tgt, "PosGraph").begin());
677#ifdef DEBUG_LP
678 LDEBUG << "BowGenerator: working vx " << vxTokVertex;
679 LDEBUG << "BowGenerator: working tgt " << tgtTokVertex;
680#endif
681 DependencyGraphVertex depV = syntacticData->depVertexForTokenVertex(vxTokVertex);
682 if (out_degree(depV, *depGraph) > 0){
683 DependencyGraphOutEdgeIt depIt, depIt_end;
684 boost::tie(depIt, depIt_end) = out_edges(depV, *depGraph);
685 for (; depIt != depIt_end; depIt++)
686 {
687 DependencyGraphVertex depTargV = target(*depIt, *depGraph);
688 LinguisticGraphVertex targV = syntacticData-> tokenVertexForDepVertex(depTargV);
689 if (targV == tgtTokVertex){
690 CEdgeDepRelTypePropertyMap relTypeMap = get(edge_deprel_type, *depGraph);
691 relation->setSynType(relTypeMap[*depIt]);
692 }
693 }
694 }
695 }
696#ifdef DEBUG_LP
697 if (relation !=0)
698 {
699 LDEBUG << "BowGenerator: relation : " << *relation;
700 }
701#endif
702 return relation;
703}
704
705
706std::vector< std::pair< boost::shared_ptr< BoWRelation >,
707 boost::shared_ptr< AbstractBoWElement > > >
709 const LinguisticGraphVertex v,
710 const LinguisticGraph& anagraph,
711 const LinguisticGraph& posgraph,
712 const uint64_t offsetBegin,
713 const AnnotationData* annotationData,
714 std::set<LinguisticGraphVertex>& visited,
715 bool keepAnyway) const
716{
717 return m_d->createAbstractBoWElement(v, anagraph, posgraph, offsetBegin,
718 annotationData, visited, keepAnyway);
719}
720
721std::vector< std::pair< boost::shared_ptr< BoWRelation >,
722 boost::shared_ptr< AbstractBoWElement > > >
723 BowGeneratorPrivate::createAbstractBoWElement(
724 const LinguisticGraphVertex v,
725 const LinguisticGraph& anagraph,
726 const LinguisticGraph& posgraph,
727 const uint64_t offsetBegin,
728 const AnnotationData* annotationData,
729 std::set<LinguisticGraphVertex>& visited,
730 bool keepAnyway) const
731{
732#ifdef DEBUG_LP
734 LDEBUG << "BowGenerator::createAbstractBoWElement for " << v;
735#endif
736 std::vector<std::pair< boost::shared_ptr< BoWRelation >, boost::shared_ptr< AbstractBoWElement > > > abstractBowEl;
737 // Create bow tokens for specific entities created on the before PoS tagging
738 // analysis graph
739 //std::set< uint64_t > anaVertices = annotationData->matches("PosGraph",v,"AnalysisGraph"); portage 32 64
740 std::set< AnnotationGraphVertex > anaVertices = annotationData->matches("PosGraph",v,"AnalysisGraph");
741#ifdef DEBUG_LP
742 LDEBUG << "BowGenerator::createAbstractBoWElement " << v << " has " << anaVertices.size() << " matching vertices in analysis graph";
743#endif
744
745 bool createdSpecificEntity(false);
746
747 // note: anaVertices size should be 0 or 1
748 for (auto anaVertex = anaVertices.begin(); anaVertex != anaVertices.end(); ++anaVertex)
749 {
750#ifdef DEBUG_LP
751 LDEBUG << "BowGenerator::createAbstractBoWElement Looking at analysis graph vertex " << *anaVertex;
752#endif
753 std::set< AnnotationGraphVertex > matches = annotationData->matches("AnalysisGraph",*anaVertex,"annot");
754 for (auto matchVertex = matches.begin(); matchVertex != matches.end(); ++matchVertex)
755 {
756#ifdef DEBUG_LP
757 LDEBUG << "BowGenerator::createAbstractBoWElement Looking at annotation graph vertex " << *matchVertex;
758#endif
759 if (annotationData->hasAnnotation(*matchVertex, Common::Misc::utf8stdstring2limastring("SpecificEntity")))
760 {
761 // corresponding vertex from analysis graph
762 LinguisticGraphVertex matchv= annotationData->intAnnotation(*matchVertex,"AnalysisGraph");
763 boost::shared_ptr< BoWToken > se = createSpecificEntity(matchv,*matchVertex, annotationData, anagraph, posgraph, offsetBegin, false);
764 //boost::shared_ptr< BoWToken > se = createSpecificEntity(v,*matchVertex, annotationData, anagraph, posgraph, offsetBegin, false);
765 if (se != 0)
766 {
767#ifdef DEBUG_LP
768 LDEBUG << "BowGenerator::createAbstractBoWElement created specific entity: " << QString::fromUtf8(se->getOutputUTF8String().c_str());
769#endif
770 se->setVertex(v);
771 abstractBowEl.push_back(std::make_pair(boost::shared_ptr< BoWRelation >(),se));
772// visited.insert(v);
773 createdSpecificEntity=true;
774 break;
775 }
776 }
777 }
778 }
779#ifdef DEBUG_LP
780 LDEBUG << "BowGenerator::createAbstractBoWElement move on the PosGraph annot matching test";
781#endif
782
783 // check if there is specific entities or compound tenses associated to v.
784 // return them if any
785 std::set< AnnotationGraphVertex > matches = annotationData->matches("PosGraph",v,"annot");
786#ifdef DEBUG_LP
787 LDEBUG << "BowGenerator::createAbstractBoWElement there are " << matches.size() << " annotation graph vertices matching the current PsGraph vertex " << v;
788#endif
789 for (auto it = matches.begin(); it != matches.end(); ++it)
790 {
791 AnnotationGraphVertex vx = *it;
792#ifdef DEBUG_LP
793 LDEBUG << "BowGenerator::createAbstractBoWElement Looking at annotation graph vertex " << vx;
794#endif
795 if (annotationData->hasAnnotation(vx, Common::Misc::utf8stdstring2limastring("SpecificEntity")))
796 {
797 boost::shared_ptr< BoWToken > se = createSpecificEntity(v,vx, annotationData, anagraph, posgraph, offsetBegin);
798 if (se != 0)
799 {
800#ifdef DEBUG_LP
801 LDEBUG << "BowGenerator::createAbstractBoWElement created specific entity: " << QString::fromUtf8(se->getOutputUTF8String().c_str());
802#endif
803 se->setVertex(v);
804 abstractBowEl.push_back(std::make_pair(boost::shared_ptr< BoWRelation >(),se));
805// visited.insert(v);
806 return abstractBowEl;
807 }
808 }
809 else if (annotationData->hasIntAnnotation(vx, Common::Misc::utf8stdstring2limastring("CpdTense")))
810 {
811 boost::shared_ptr< BoWToken > ct = createCompoundTense(vx, annotationData, anagraph, posgraph, offsetBegin, visited);
812 if (ct != 0)
813 {
814 #ifdef DEBUG_LP
815 LDEBUG << "BowGenerator::createAbstractBoWElement created compound tense: " << *ct;
816#endif
817 ct->setVertex(v);
818 abstractBowEl.push_back(std::make_pair(boost::shared_ptr< BoWRelation >(),ct));
819// visited.insert(v);
820 return abstractBowEl;
821 }
822 }
823 else if (annotationData->hasStringAnnotation(vx, Common::Misc::utf8stdstring2limastring("Predicate")))
824 {
825
826#ifdef DEBUG_LP
827 LDEBUG << "BowGenerator::createAbstractBoWElement Found a predicate in the PosGraph annnotation graph matching";
828#endif
829
830 MorphoSyntacticData* data = get(vertex_data, posgraph, v);
831 bool toKeep = true;
832 if (data!=0)
833 {
834 for (auto elem = data->begin(); elem != data->end(); ++elem)
835 {
836 if (!keepAnyway && !shouldBeKept(*elem))
837 {
838 toKeep = false;
839 break;
840 }
841 }
842 }
843 if (toKeep)
844 {
845 auto pred = createPredicate(v, vx, annotationData, anagraph, posgraph,
846 offsetBegin, visited, keepAnyway);
847 for (auto bP = pred.begin(); bP != pred.end(); ++bP)
848 {
849 if (*bP!=0)
850 {
851 #ifdef DEBUG_LP
852 LDEBUG << "BowGenerator::createAbstractBoWElement created a predicate" ;
853 #endif
854 abstractBowEl.push_back(std::make_pair(boost::shared_ptr< BoWRelation >(),*bP));
855 // visited.insert(v);
856 // return abstractBowEl;
857 }
858 }
859 }
860 }
861 else
862 {
863#ifdef DEBUG_LP
864 LDEBUG << "BowGenerator::createAbstractBoWElement No SpecificEntity nor CpdTense nor Predicate found";
865#endif
866 }
867 }
868
869 // bow tokens have been created for specific entities on the before PoS
870 // tagging graph. return them
871 if (!abstractBowEl.empty())
872 {
873// return abstractBowEl;
874 }
875
877
878 MorphoSyntacticData* data = get(vertex_data, posgraph, v);
879 Token* token = get(vertex_token, posgraph, v);
880
881 std::set<std::pair<StringsPoolIndex,LinguisticCode> > alreadyCreated;
882 std::pair<StringsPoolIndex,LinguisticCode> predNormCode = std::make_pair(StringsPoolIndex(0),L_NONE);
883
884 if (createdSpecificEntity) {
885 // a specific entity has been created on the analysis graph: do not output a token
886 // (RB: do that here so that the vertex on the posgraph can also be analyzed: should test is this is
887 // needed or if we only need to place the return just after the creation of the named entity)
888 return abstractBowEl;
889 }
890
891 if (data!=0)
892 {
893 for (auto it=data->begin(); it!=data->end(); it++)
894 {
895 auto normCode=std::make_pair(it->normalizedForm, m_microAccessor->readValue(it->properties));
896 if (normCode != predNormCode)
897 {
898 if (alreadyCreated.find(normCode)==alreadyCreated.end())
899 {
900 if (keepAnyway || shouldBeKept(*it))
901 {
902 boost::shared_ptr< BoWToken > newbowtok( new BoWToken(sp[it->normalizedForm],
903 m_macroAccessor->readValue(it->properties),
904 offsetBegin+token->position(),
905 token->length()));
906 newbowtok->setVertex(v);
907 newbowtok->setInflectedForm(token->stringForm());
908#ifdef DEBUG_LP
909 LDEBUG << "BowGenerator::createAbstractBoWElement created bow token: " << *newbowtok;
910#endif
911 abstractBowEl.push_back(std::make_pair(boost::shared_ptr< BoWRelation >(),newbowtok));
912 }
913 alreadyCreated.insert(normCode);
914 }
915 predNormCode=normCode;
916 }
917 }
918 }
919/* if (!bowTokens.empty())
920 {
921 visited.insert(v);
922 }*/
923 return abstractBowEl;
924}
925
926
927
928bool BowGeneratorPrivate::shouldBeKept(const LinguisticAnalysisStructure::LinguisticElement& elem) const
929{
930/*
931 Critical function : comment logging messages
932*/
933#ifdef DEBUG_LP
935#endif
936 // bool result = false;
937
939 const LanguageData& ldata=static_cast<const LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language));
940 if (m_useEmptyMacro && ldata.isAnEmptyMacroCategory(m_macroAccessor->readValue(elem.properties)))
941 {
942#ifdef DEBUG_LP
943 LDEBUG << "BowGenerator::shouldBeKept token ("
944 << sp[elem.lemma] << "|"
945 << elem.properties << ") not kept : macro category is empty ";
946#endif
947 return false;
948 }
949
950
951 if (m_useEmptyMicro && ldata.isAnEmptyMicroCategory(m_microAccessor->readValue(elem.properties)))
952 {
953#ifdef DEBUG_LP
954 LDEBUG << "BowGenerator::shouldBeKept token ("
955 << sp[elem.lemma] << "|"
956 << elem.properties << ") not kept : micro category is empty ";
957#endif
958 return false;
959 }
960
961#ifdef DEBUG_LP
962 LDEBUG << "BowGenerator::shouldBeKept check token (" << sp[elem.normalizedForm] << ")";
963#endif
964 if (m_useStopList && m_stopList!=0 && (m_stopList->find(sp[elem.normalizedForm]) != m_stopList->end()))
965 {
966#ifdef DEBUG_LP
967 LDEBUG << "BowGenerator::shouldBeKept token ("
968 << sp[elem.lemma] << "|"
969 << elem.properties << ") not kept : normalization "
970 << sp[elem.normalizedForm]
971 << " is in stoplist";
972#endif
973 return false;
974 }
975
976#ifdef DEBUG_LP
977 LDEBUG << "BowGenerator::shouldBeKept token ("
978 << sp[elem.lemma] << "|"
979 << elem.properties << "), normalization "
980 << sp[elem.normalizedForm] << " kept";
981#endif
982
983 return true;
984}
985
986bool BowGeneratorPrivate::checkStopWordInCompound(
987 boost::shared_ptr< Common::BagOfWords::BoWToken >& tok,
988 uint64_t offset,
989 std::set< std::string >& alreadyStored,
990 Common::BagOfWords::BoWText& bowText) const
991{
992 LIMA_UNUSED(tok);
993 LIMA_UNUSED(offset);
994 LIMA_UNUSED(alreadyStored);
995 LIMA_UNUSED(bowText);
996 /* DUMPERLOGINIT;
997 LDEBUG << "BowGenerator: checkStopWord : " << tok->getIdUTF8String();
998 BoWTerm* bowTerm=dynamic_cast<BoWTerm*>(tok);
999 if (bowTerm != 0)
1000 {
1001 LDEBUG << "BowGenerator: is a bowTerm";
1002 // is a bowTerm : check parts and rearrange if necessary
1003 std::deque< BoWComplexToken::Part >& parts=bowTerm->getParts();
1004 uint64_t partIndex=0;
1005 std::vector<bool> partToRemove;
1006 bool removeHead=false;
1007
1008 // get part to remove
1009 for (std::deque< BoWComplexToken::Part >::iterator partItr=parts.begin();
1010 partItr!=parts.end();
1011 partItr++, partIndex++)
1012 {
1013
1014 if (checkStopWordInCompound(partItr->get<1>(),offset,alreadyStored,bowText))
1015 {
1016 // have to remove part and rearrange bowTerm
1017 if (partIndex == bowTerm->getHead())
1018 {
1019 removeHead=true;
1020 }
1021 // stop word is dependant add to remove list
1022 partToRemove.push_back(true);
1023 }
1024 else
1025 {
1026 partToRemove.push_back(false);
1027 }
1028 }
1029
1030 if (removeHead)
1031 {
1032 LDEBUG << "BowGenerator: remove head";
1033 // stop word is head : index other parts and return true
1034 std::vector<bool>::iterator toRemove=partToRemove.begin();
1035 for (std::deque< BoWComplexToken::Part >::iterator partItr=parts.begin();
1036 partItr!=parts.end();
1037 partItr++,toRemove++)
1038 {
1039 BoWToken* t=partItr->get<1>();
1040 if (*toRemove || (alreadyStored.find(t->getIdUTF8String())!=alreadyStored.end()) )
1041 {
1042 LDEBUG << "BowGenerator: remove part " << t->getIdUTF8String();
1043 if (!partItr->get<2>())
1044 {
1045 delete t;
1046 }
1047 }
1048 else
1049 {
1050 // add part to bowText
1051 LDEBUG << "BowGenerator: write part " << t->getIdUTF8String();
1052 t->addToPosition(offset);
1053 bowText.push_back(t);
1054 alreadyStored.insert(t->getIdUTF8String());
1055 }
1056 }
1057 parts.clear();
1058 return true;
1059 }
1060 else
1061 {
1062 LDEBUG << "BowGenerator: check part to remove";
1063 // check part to remove, and rearrange if only one part left
1064 std::vector<bool>::iterator toRemove=partToRemove.begin();
1065 uint64_t index=parts.size()-1;
1066 for (std::deque< BoWComplexToken::Part >::iterator partItr=parts.begin();
1067 partItr!=parts.end();
1068 toRemove++)
1069 {
1070 if (*toRemove)
1071 {
1072 LDEBUG << "BowGenerator: remove part " << partItr->get<1>()->getIdUTF8String();
1073 if (index < bowTerm->getHead())
1074 {
1075 bowTerm->setHead(bowTerm->getHead()-1);
1076 }
1077 if (!partItr->get<2>())
1078 {
1079 delete partItr->get<1>();
1080 }
1081 parts.erase(partItr);
1082 }
1083 else
1084 {
1085 partItr++;
1086 }
1087 }
1088 // size cannot be null, there should at least be the head
1089 if (parts.size()==1)
1090 {
1091 LDEBUG << "BowGenerator: replace bowTerm " << tok->getIdUTF8String() << " by his only part ";
1092 // replace bowToken by its only part
1093 tok=parts.begin()->get<1>()->clone();
1094 delete bowTerm;
1095 LDEBUG << "BowGenerator: bowTerm becomes " << tok->getIdUTF8String();
1096 }
1097 }
1098 return false;
1099 }
1100 else
1101 {
1102 // is a bowToken : check if stopword and return
1103 if (m_stopList->find(tok->getLemma()) != m_stopList->end())
1104 {
1105 LINFO << "found stopword " << tok->getIdUTF8String() << " in coumpound !";
1106 return true;
1107 }
1108 else
1109 {
1110 return false;
1111 }
1112 }*/
1113 return true;
1114}
1115
1116void BowGeneratorPrivate::bowTokenPositions(
1117 TokenPositions& res,
1118 const boost::shared_ptr< Common::BagOfWords::BoWToken > tok) const
1119{
1120 Common::Misc::PositionLengthList poslenlist=tok->getPositionLengthList();
1121 res.insert(poslenlist.begin(),poslenlist.end());
1122}
1123
1124uint64_t BowGeneratorPrivate::computeCompoundLength(
1125 const TokenPositions& headTokPositions,
1126 const TokenPositions& extensionPositions) const
1127{
1128 // extent (end-begin) is not a good measure: in "nice little cat",
1129 // nice_cat and nice_little_cat would have same length
1130
1131 // sum of length of part is not a good measure: in
1132 // in "products available in Minnesota",
1133 // products_available and products_Minnesota would
1134 // have same length
1135 //
1136
1137 // => keep extent and let further treatments
1138 // decide if they need to use more complex comparisons
1139 // (use complete PositionLengthList)
1140
1141 uint64_t length=0;
1142
1143 // extent
1144 uint64_t positionBegin=headTokPositions.begin()->first;
1145 if (positionBegin>extensionPositions.begin()->first)
1146 {
1147 positionBegin=extensionPositions.begin()->first;
1148 }
1149 uint64_t positionEnd=headTokPositions.rbegin()->first+
1150 headTokPositions.rbegin()->second;
1151 uint64_t positionEndExt=extensionPositions.rbegin()->first+
1152 extensionPositions.rbegin()->second;
1153 if (positionEndExt>positionEnd)
1154 {
1155 positionEnd=positionEndExt;
1156 }
1157 length=positionEnd-positionBegin;
1158
1159 // sum of lengths of components
1160 // TokenPositions::const_iterator
1161 // it=headTokPositions.begin(),
1162 // it_end=headTokPositions.end();
1163 // for (; it!=it_end; it++) {
1164 // length+=(*it).second;
1165 // }
1166 // it=extensionPositions.begin();
1167 // it_end=extensionPositions.end();
1168 // for (; it!=it_end; it++) {
1169 // length+=(*it).second;
1170 // }
1171
1172 return length;
1173}
1174
1175boost::shared_ptr< BoWNamedEntity > BowGenerator::createSpecificEntity(
1176 const LinguisticGraphVertex& vertex,
1177 const AnnotationGraphVertex& v,
1178 const AnnotationData* annotationData,
1179 const LinguisticGraph& anagraph,
1180 const LinguisticGraph& posgraph,
1181 const uint64_t offset,
1182 bool frompos) const
1183{
1184 return m_d->createSpecificEntity(vertex,v,annotationData,anagraph,posgraph,offset,frompos);
1185}
1186
1187boost::shared_ptr< BoWNamedEntity > BowGeneratorPrivate::createSpecificEntity(
1188 const LinguisticGraphVertex& vertex,
1189 const AnnotationGraphVertex& v,
1190 const AnnotationData* annotationData,
1191 const LinguisticGraph& anagraph,
1192 const LinguisticGraph& posgraph,
1193 const uint64_t offset,
1194 bool frompos) const
1195{
1196 if (!annotationData->hasAnnotation(v, Common::Misc::utf8stdstring2limastring("SpecificEntity")))
1197 {
1198 return boost::shared_ptr< BoWNamedEntity >();
1199 }
1200#ifdef DEBUG_LP
1202 LINFO << "BowGenerator: createSpecificEntity ling:" << vertex
1203 << "; annot:" << v << offset << frompos;
1204#endif
1205 const LinguisticGraph& graph = (frompos?posgraph:anagraph);
1207
1208 const SpecificEntityAnnotation* se =
1209 annotationData->annotation(v, Common::Misc::utf8stdstring2limastring("SpecificEntity")).
1210 pointerValue<SpecificEntityAnnotation>();
1211
1212#ifdef DEBUG_LP
1213 LINFO << "BowGenerator: specific entity type is " << se->getType();
1214#endif
1215
1216 std::set< std::string > alreadyStored;
1217
1218 // build BoWNamedEntity
1219 LimaString typeName("");
1220 try {
1221 typeName = MediaticData::single().getEntityName(se->getType());
1222 }
1223 catch (std::exception& e) {
1225 LERROR << "Undefined entity type " << se->getType();
1226 return boost::shared_ptr< BoWNamedEntity >();
1227 }
1228#ifdef DEBUG_LP
1229 LINFO << "BowGenerator: specific entity type name is " << typeName;
1230#endif
1231 // get the macro-category to use for this named entity
1232 MorphoSyntacticData* data = get(vertex_data, graph, vertex);
1233 if (data->empty())
1234 {
1236 LERROR << "Empty data for vertex " << vertex << " at " << __FILE__ << ", line " << __LINE__;
1237 LERROR << "This is a bug. Returning null entity for" << se->getString() << typeName;
1238 return boost::shared_ptr< BoWNamedEntity >();
1239 }
1240
1241 LinguisticCode category=m_macroAccessor->readValue(data->begin()->properties);
1242
1243 boost::shared_ptr< BoWNamedEntity > bowNE( new
1244 BoWNamedEntity(sp[getNamedEntityNormalization(v, annotationData)],
1245 category,
1246 se->getType(),
1247 offset+se->getPosition(),
1248 se->getLength()) );
1249 // add named entity parts
1250 auto neParts = createNEParts(v, annotationData, anagraph, posgraph, frompos);
1251
1252 if (neParts.empty())
1253 {
1255 LWARN << "No parts kept for named entity " << (*se).getString();
1256 // set named entity itself as part
1257 boost::shared_ptr< BoWToken > bowToken(new
1258 BoWToken(sp[getNamedEntityNormalization(v, annotationData)],
1259 category,
1260 offset+(*se).getPosition(),
1261 (*se).getLength()));
1262 Token* token = get(vertex_token, graph, vertex);
1263 bowToken->setInflectedForm(token->stringForm());
1264
1265 bowNE->addPart(bowToken);
1266 }
1267 else
1268 {
1269 for (auto p=neParts.begin(); p!=neParts.end(); p++)
1270 {
1271 //create a new BoWToken
1272 boost::shared_ptr< BoWToken > bowToken(new BoWToken((*p).lemma,(*p).category,
1273 offset+(*p).position,
1274 (*p).length));
1275 bowToken->setInflectedForm((*p).inflectedForm);
1276#ifdef DEBUG_LP
1277 LDEBUG << "BowGenerator: specific entity part infl " << (*p).inflectedForm;
1278#endif
1279 bowNE->addPart(bowToken);
1280 }
1281 }
1282
1283 // add the features
1284 const auto& features=(*se).getFeatures();
1285 for (auto f=features.begin(), f_end=features.end(); f!=f_end; f++)
1286 {
1287 bowNE->setFeature((*f).getName(),
1288 (*f).getValueLimaString());
1289 }
1290#ifdef DEBUG_LP
1291 LDEBUG << "CreateSpecificEntity: created features " << QString::fromUtf8(bowNE->getFeaturesUTF8String().c_str());
1292#endif
1293
1294 auto elem = bowNE->getIdUTF8String();
1295 if (alreadyStored.find(elem) != alreadyStored.end())
1296 { // already stored
1297#ifdef DEBUG_LP
1298 LDEBUG << "BowGenerator: BoWNE already stored. Skipping it.";
1299#endif
1300 return boost::shared_ptr< BoWNamedEntity >();
1301 }
1302 else
1303 {
1304// LDEBUG << "BowGenerator: created BoWNamedEntity " << bowNE->getOutputUTF8String();
1305 alreadyStored.insert(elem);
1306 return bowNE;
1307 }
1308 return bowNE;
1309}
1310
1311
1312QList< boost::shared_ptr< BoWPredicate > > BowGeneratorPrivate::createPredicate(
1313 const LinguisticGraphVertex& lgv,
1314 const AnnotationGraphVertex& agv,
1315 const AnnotationData* annotationData,
1316 const LinguisticGraph& anagraph,
1317 const LinguisticGraph& posgraph,
1318 const uint64_t offset,
1319 std::set< LinguisticGraphVertex >& visited,
1320 bool keepAnyway) const
1321{
1322#ifdef DEBUG_LP
1324 LINFO << "BowGenerator::createPredicate ling:" << lgv << "; annot:" << agv;
1325#endif
1326 QList< boost::shared_ptr< BoWPredicate > > result;
1327
1328 Token* token = get(vertex_token, posgraph, lgv);
1329
1330 // FIXME handle the ambiguous case when there is several class values for the predicate
1331 QStringList predicateIds=annotationData->stringAnnotation(agv,Common::Misc::utf8stdstring2limastring("Predicate")).split(",");
1332#ifdef DEBUG_LP
1333 if (predicateIds.size()>1)
1334 {
1335 LDEBUG << "BowGenerator::createPredicate Predicate has"
1336 << predicateIds.size() << "values:" << predicateIds;
1337 }
1338#endif
1339
1340
1341 // FIXED replace the hardcoded VerbNet by a value from configuration
1342 // LimaString predicate=predicateIds.first();
1343 // The fix should work only with FrameNet annotations. VerbNet does not assure to have the same
1344 // number of roles in each list as the number of predicates
1345 for (int i = 0 ; i < predicateIds.size(); i++)
1346 {
1347 auto predicate = predicateIds[i];
1348 try
1349 {
1350 auto predicateEntity = Common::MediaticData::MediaticData::single().getEntityType(predicate);
1351#ifdef DEBUG_LP
1352 LDEBUG << "BowGenerator::createPredicate The role(s) related to "
1353 << predicate << " is/are ";
1354#endif
1355 AnnotationGraph annotGraph = annotationData->getGraph();
1356 AnnotationGraphOutEdgeIt outIt, outIt_end;
1357 boost::tie(outIt, outIt_end) = boost::out_edges(agv, annotationData->getGraph());
1359 boost::shared_ptr< AbstractBoWElement > > roles;
1360 const LimaString typeAnnot = "SemanticRole";
1361 for (; outIt != outIt_end; outIt++)
1362 {
1363 // FIXME handle the ambiguous case when there is several values for each role
1364 auto semRoleVx = boost::target(*outIt, annotGraph);
1365 auto semRoleIds = annotationData->stringAnnotation(agv,
1366 semRoleVx,
1367 typeAnnot).split("|");
1368 if (predicateIds.size() != semRoleIds.size())
1369 {
1371 LERROR << "BowGenerator::createPredicate predicateIds and semRoleIds sizes are different:"
1372 << predicateIds.size() << "and" << semRoleIds.size();
1373 LERROR << "BowGenerator::createPredicate abort this predicate creation";
1374 return result;
1375 }
1376 Q_ASSERT(predicateIds.size() == semRoleIds.size());
1377 LimaString semRole = semRoleIds[i];
1378#ifdef DEBUG_LP
1379 LDEBUG << semRole;
1380#endif
1381 if (semRole.isEmpty()) continue;
1382 try
1383 {
1384 auto semRoleEntity = Common::MediaticData::MediaticData::single().getEntityType(semRole);
1385 auto posGraphSemRoleVertices = annotationData->matches("annot", semRoleVx,
1386 "PosGraph");
1387 if (!posGraphSemRoleVertices.empty())
1388 {
1389 auto posGraphSemRoleVertex = *(posGraphSemRoleVertices.begin());
1390 if (posGraphSemRoleVertex == lgv)
1391 {
1393 LERROR << "BowGenerator::createPredicate role vertex is the same as the trigger vertex. Abort this role.";
1394 continue;
1395 }
1396#ifdef DEBUG_LP
1397 LDEBUG << "BowGenerator::createPredicate Calling createAbstractBoWElement on PoS graph vertex"
1398 << posGraphSemRoleVertex;
1399#endif
1400 auto semRoleTokens = createAbstractBoWElement(posGraphSemRoleVertex,
1401 anagraph,
1402 posgraph,
1403 offset,
1404 annotationData,
1405 visited,
1406 keepAnyway);
1407#ifdef DEBUG_LP
1408 LDEBUG << "BowGenerator::createPredicate Created "
1409 << semRoleTokens.size()
1410 << "token for the role associated to" << predicate;
1411#endif
1412 // if (semRoleTokens[0].second!="")
1413 if (!semRoleTokens.empty())
1414 {
1415 roles.insert(semRoleEntity, semRoleTokens[0].second);
1416 }
1417 }
1418 else
1419 {
1420#ifdef DEBUG_LP
1421 LDEBUG << "BowGenerator::createPredicate Found no matching for the semRole in the annot graph";
1422#endif
1423 }
1424 }
1425 catch (const Lima::LimaException& e)
1426 {
1428 LERROR << "BowGenerator::createPredicate Unknown semantic role"
1429 << semRole << ";" << e.what();
1430 throw;
1431 }
1432 }
1433 boost::shared_ptr< BoWPredicate > bowP(new BoWPredicate());
1434 bowP->setPosition(offset+token->position());
1435 bowP->setLength(token->length());
1436 bowP->setPredicateType(predicateEntity);
1437 auto pEntityType = bowP->getPredicateType();
1438#ifdef DEBUG_LP
1439 LDEBUG << "BowGenerator::createPredicate Created a Predicate for the verbal class "
1441#endif
1442 if (!roles.empty())
1443 {
1444 bowP->setRoles(roles);
1445 auto pRoles = bowP->roles();
1446 for (auto it = pRoles.begin(); it != pRoles.end(); it++)
1447 {
1448 auto outputRoles = boost::dynamic_pointer_cast<BoWToken>(it.value());
1449 if (outputRoles != 0)
1450 {
1451 auto roleLabel = Common::MediaticData::MediaticData::single().getEntityName(it.key());
1452#ifdef DEBUG_LP
1453 LDEBUG << "BowGenerator::createPredicate Associated "
1454 << QString::fromStdString(outputRoles->getOutputUTF8String())
1455 << " to it" << "via the semantic role label "<< roleLabel ;
1456#endif
1457 }
1458 }
1459 }
1460 result.append(bowP);
1461 }
1462 catch (const Lima::LimaException& e)
1463 {
1465 LERROR << "BowGenerator::createPredicate Unknown predicate"
1466 << predicate << ";" << e.what();
1467 return QList< boost::shared_ptr< BoWPredicate > >();
1468 }
1469 }
1470 return result;
1471}
1472
1473QList< boost::shared_ptr< Common::BagOfWords::BoWPredicate > > BowGenerator::createSemanticRelationPredicate(
1474 const LinguisticGraphVertex& lgvs,
1475 const AnnotationGraphVertex& agvs,
1476 const AnnotationGraphVertex& agvt,
1477 const SemanticRelationAnnotation& annot ,
1478 const AnnotationData* annotationData,
1479 const LinguisticGraph& anagraph,
1480 const LinguisticGraph& posgraph,
1481 uint64_t offset,
1482 std::set< LinguisticGraphVertex >& visited,
1483 bool keepAnyway) const
1484{
1485#ifdef DEBUG_LP
1487 LINFO << "BowGenerator::createSemanticRelationPredicate " << lgvs << ", src:" << agvs << ", trgt:"<< agvt
1488 << ", rel:" << annot.type().c_str();
1489#endif
1490 QList< boost::shared_ptr< BoWPredicate > > result;
1491
1492 // FIXME handle the ambiguous case when there is several class values for the predicate
1493 auto predicateIds = QString::fromStdString(annot.type()).split("|");
1494#ifdef DEBUG_LP
1495 if (predicateIds.size()>1)
1496 {
1497 LDEBUG << "BowGenerator::createSemanticRelationPredicate Semantic relation has"
1498 << predicateIds.size() << "values:" << predicateIds;
1499 }
1500#endif
1501
1502 // FIXED replace the hardcoded VerbNet by a value from configuration
1503 // LimaString predicate=predicateIds.first();
1504 // The fix should work only with FrameNet annotations. VerbNet does not assure to have the same
1505 // number of roles in each list as the number of predicates
1506 for (int i = 0 ; i < predicateIds.size(); i++)
1507 {
1508 auto predicate = predicateIds[i];
1509 try
1510 {
1511 auto predicateEntity = Common::MediaticData::MediaticData::single().getEntityType(predicate);
1512
1513 boost::shared_ptr< BoWPredicate > bowP( new BoWPredicate() );
1514
1515 bowP->setPredicateType(predicateEntity);
1516
1517 bowP->setPosition(0);
1518 bowP->setLength(0);
1519
1520 std::vector<AnnotationGraphVertex> vertices;
1521 vertices.push_back(agvs);
1522 vertices.push_back(agvt);
1523 #ifdef DEBUG_LP
1524 LDEBUG << "BowGenerator::createSemanticRelationPredicate The role(s) related to "
1525 << annot.type() << " is/are ";
1526 #endif
1528 boost::shared_ptr< AbstractBoWElement > > roles;
1529 // const LimaString typeAnnot="SemanticRole";
1530 for (const auto& semRoleVx: vertices)
1531 {
1532 auto anaGraphSemRoleVertices = annotationData->matches("annot", semRoleVx,
1533 "AnalysisGraph");
1534 if (!anaGraphSemRoleVertices.empty())
1535 {
1536 auto anaGraphSemRoleVertex = *anaGraphSemRoleVertices.begin();
1537 auto posGraphSemRoleVertices = annotationData->matches("AnalysisGraph",
1538 anaGraphSemRoleVertex,
1539 "PosGraph");
1540 if (!posGraphSemRoleVertices.empty())
1541 {
1542 auto posGraphSemRoleVertex = *(posGraphSemRoleVertices.begin());
1543 if (posGraphSemRoleVertex == lgvs)
1544 {
1545 #ifdef DEBUG_LP
1546 LERROR << "BowGenerator::createSemanticRelationPredicate role vertex is the same as the trigger vertex. Abort this role.";
1547 #endif
1548 continue;
1549 }
1550 #ifdef DEBUG_LP
1551 LDEBUG << "BowGenerator::createSemanticRelationPredicate Calling createAbstractBoWElement on PoS graph vertex"
1552 << posGraphSemRoleVertex;
1553 #endif
1554 auto semRoleTokens = m_d->createAbstractBoWElement(posGraphSemRoleVertex,
1555 anagraph,
1556 posgraph,
1557 offset,
1558 annotationData,
1559 visited,
1560 keepAnyway);
1561 #ifdef DEBUG_LP
1562 LDEBUG << "BowGenerator::createSemanticRelationPredicate Created "
1563 << semRoleTokens.size() << "token for the role associated to "
1564 << annot.type().c_str();
1565 #endif
1566 if (!semRoleTokens.empty())
1567 {
1568 EntityType semRoleEntity;
1569 roles.insert(semRoleEntity, semRoleTokens[0].second);
1570 }
1571 }
1572#ifdef DEBUG_LP
1573 else
1574 {
1575 LDEBUG << "BowGenerator::createSemanticRelationPredicate Found no matching for the semRole in the annot graph";
1576 }
1577#endif
1578 }
1579 }
1580#ifdef DEBUG_LP
1581 LDEBUG << "BowGenerator::createSemanticRelationPredicate Created a Predicate for the semantic relation"
1582 << predicateEntity
1584#endif
1585 if (!roles.empty())
1586 {
1587 bowP->setRoles(roles);
1588#ifdef DEBUG_LP
1589 for (auto it = roles.begin(); it != roles.end(); it++)
1590 {
1591 auto outputRoles=boost::dynamic_pointer_cast<BoWToken>(it.value());
1592 if (outputRoles != 0)
1593 {
1594 auto roleLabel = it.key().isNull() ? QString()
1596 LDEBUG << "BowGenerator::createSemanticRelationPredicate Associated "
1597 << QString::fromStdString(outputRoles->getOutputUTF8String())
1598 << " to it" << "via the semantic role label "<< roleLabel ;
1599 }
1600 }
1601#endif
1602 }
1603 result.append(bowP);
1604 }
1605 catch (const Lima::LimaException& e)
1606 {
1608 LERROR << "BowGenerator::createSemanticRelationPredicate Unknown predicate"
1609 << predicate << ";" << e.what();
1610 return QList< boost::shared_ptr< BoWPredicate > >();
1611 }
1612 }
1613 return result;
1614}
1615
1616
1617StringsPoolIndex BowGeneratorPrivate::getNamedEntityNormalization(
1618 const AnnotationGraphVertex& v,
1619 const AnnotationData* annotationData) const
1620{
1621 if (!annotationData->hasAnnotation(v, Common::Misc::utf8stdstring2limastring("SpecificEntity")))
1622 {
1623#ifdef DEBUG_LP
1625 LDEBUG << "BowGenerator::getNamedEntityNormalization: no SpecificEntity annotation for " << v << " ; return 0";
1626#endif
1627 return static_cast<StringsPoolIndex>(0);
1628 }
1629#ifdef DEBUG_LP
1631 LINFO << "BowGenerator::getNamedEntityNormalization: m_NEnormalization is " << m_NEnormalization;
1632#endif
1633 StringsPoolIndex normalizedForm(0);
1634 switch (m_NEnormalization)
1635 {
1636 case NORMALIZE_NE_INFLECTED:
1637 normalizedForm=annotationData->annotation(v,Common::Misc::utf8stdstring2limastring("SpecificEntity"))
1638 .pointerValue< SpecificEntityAnnotation >()->getString();
1639 break;
1640 case NORMALIZE_NE_LEMMA:
1641 normalizedForm=annotationData->annotation(v,Common::Misc::utf8stdstring2limastring("SpecificEntity"))
1642 .pointerValue< SpecificEntityAnnotation >()->getNormalizedString();
1643 break;
1644 case NORMALIZE_NE_NORMALIZEDFORM:
1645 // list of "attribute=value" elements (not pretty but generic)
1646 // normalizedForm=se.getNormalizedForm().str();
1647 normalizedForm=annotationData->annotation(v,Common::Misc::utf8stdstring2limastring("SpecificEntity"))
1648 .pointerValue< SpecificEntityAnnotation >()->getNormalizedForm();
1649 break;
1650 case NORMALIZE_NE_NETYPE:
1652 const SpecificEntityAnnotation* annot = annotationData->annotation(v,Common::Misc::utf8stdstring2limastring("SpecificEntity"))
1654
1656
1657 normalizedForm=sp[typeStr];
1658 break;
1659 }
1660#ifdef DEBUG_LP
1661 LDEBUG << "BowGenerator::getNamedEntityNormalization return " << normalizedForm;
1662#endif
1663 return normalizedForm;
1664}
1665
1666std::vector<BowGeneratorPrivate::NamedEntityPart> BowGeneratorPrivate::createNEParts(
1667 const AnnotationGraphVertex& v,
1668 const AnnotationData* annotationData,
1669 const LinguisticGraph& anagraph,
1670 const LinguisticGraph& posgraph,
1671 bool frompos) const
1672{
1673#ifdef DEBUG_LP
1675#endif
1676 const LinguisticGraph& graph = (frompos?posgraph:anagraph);
1678
1679 const auto namedEntity =
1680 annotationData->annotation(v,Common::Misc::utf8stdstring2limastring("SpecificEntity"))
1682 std::vector<BowGeneratorPrivate::NamedEntityPart> parts;
1683
1684
1685 bool useOnePart(false); // use one part, tagged as proper noun
1686 bool useDefaultParts(true); // use parts of named entity as they are in the graph
1687 LinguisticCode useCategory = LinguisticCode::fromUInt((uint64_t)-1); // use this category for each of the parts (NP or NC)
1688 uint64_t position(0);
1689 uint64_t length(0);
1690
1692 /*
1693 LimaString typeName=MediaticData::single().getEntityName(namedEntity->getType());
1694 if (it==m_entityNames.end())
1695 {
1696 LERROR << "Undefined entity type " << namedEntity->getType();
1697 }
1698 else
1699 {
1700 typeName=(*it).second;
1701 }
1702 const Automaton::EntityFeatures& features=namedEntity->getFeatures();
1703 EntityFeatures::const_iterator featureIt;
1704
1705 if (typeName == "PERSON")
1706 {
1707 // could force two parts, firstname and lastname, both tagged as proper noun
1708 // but firstname in features can be added by normalization -> not found in text
1709 // => use real parts, but each with proper noun category
1710 useDefaultParts=true;
1711 useCategory=m_properNounCategory;
1712 }
1713 else if (typeName == "TIMEX")
1714 {
1715 // problems to get the positions and length for day/month/year features
1716 // => use real parts, each with common noun category
1717 useDefaultParts=true;
1718 useCategory=m_commonNounCategory;
1719 }
1720 else if (typeName == "NUMEX")
1721 {
1722 // problems to get the positions and length for value/unit features
1723 // => use real parts, each with common noun category
1724 useDefaultParts=true;
1725 useCategory=m_commonNounCategory;
1726 }
1727 else if (typeName == "EVENT")
1728 {
1729 // use default parts, with category assigned by analysis
1730 useDefaultParts=true;
1731 }
1732 else if (typeName == "LOCATION" ||
1733 typeName == "ORGANIZATION" ||
1734 typeName == "PRODUCT")
1735 {
1736 // only one part, tagged as proper noun
1737 useOnePart=true;
1738 }
1739 else
1740 { // other types
1741 LWARN << "unexpected type of NE to dump: use default treatment";
1742 useDefaultParts=true;
1743 }
1744 */
1745
1746 if (useOnePart)
1747 {
1748 position=namedEntity->getPosition();
1749 length=namedEntity->getLength();
1750 LimaString normalizedForm = sp[getNamedEntityNormalization(v, annotationData)];
1751 LimaString str = sp[namedEntity->getNormalizedString()];
1752 parts.
1753 push_back(NamedEntityPart(normalizedForm,
1754 str,
1755 m_properNounCategory,
1756 position,length));
1757 }
1758 else if (useDefaultParts)
1759 {
1760 // get the parts of the named entity match
1761 for (auto m = namedEntity->vertices().cbegin();
1762 m != namedEntity->vertices().cend(); m++)
1763 {
1764 const auto token = get(vertex_token, graph, *m);
1765 const auto data = get(vertex_data, graph, *m);
1766
1767 if (data != nullptr && !data->empty())
1768 {
1769 const auto& elem = *(data->begin());
1770
1771 if (! m_keepAllNamedEntityParts && ! shouldBeKept(elem))
1772 {
1773#ifdef DEBUG_LP
1774 LDEBUG << "BowGenerator: part of named entity not kept: " << token->stringForm();
1775#endif
1776 continue;
1777 }
1778
1779 LinguisticCode category;
1780 if (useCategory != LinguisticCode::fromUInt((uint64_t)-1))
1781 {
1782 category = useCategory;
1783 }
1784 else
1785 {
1786 category = m_macroAccessor->readValue(elem.properties);
1787 }
1788 parts.push_back(NamedEntityPart(token->stringForm(),
1789 sp[elem.normalizedForm],
1790 category,
1791 token->position(),
1792 token->length()));
1793 }
1794 }
1795 }
1796
1797 return parts;
1798}
1799
1800boost::shared_ptr< BoWToken > BowGeneratorPrivate::createCompoundTense(
1801 const AnnotationGraphVertex& v,
1802 const AnnotationData* annotationData,
1803 const LinguisticGraph& anagraph,
1804 const LinguisticGraph& posgraph,
1805 const uint64_t offset,
1806 std::set<LinguisticGraphVertex>& visited) const
1807{
1808#ifdef DEBUG_LP
1810 LINFO << "BowGenerator: createCompoundTense " << v;
1811#endif
1812 if (!annotationData->hasIntAnnotation(v, Common::Misc::utf8stdstring2limastring("CpdTense")))
1813 {
1814 return boost::shared_ptr< BoWToken >();
1815 }
1816
1817 // chercher l'aux et le pp ;
1818 AnnotationGraphVertex auxVertex, ppVertex;
1819 auxVertex = ppVertex = std::numeric_limits< AnnotationGraphVertex >::max();
1820 AnnotationGraphOutEdgeIt it, it_end;
1821 boost::tie(it, it_end) = boost::out_edges(v, annotationData->getGraph());
1822 for (; it != it_end; it++)
1823 {
1824 if (annotationData->hasIntAnnotation(*it,Common::Misc::utf8stdstring2limastring("Aux")))
1825 {
1826 auxVertex = boost::target(*it, annotationData->getGraph());
1827 }
1828 else if(annotationData->hasIntAnnotation(*it,Common::Misc::utf8stdstring2limastring("PastPart")))
1829 {
1830 ppVertex = boost::target(*it, annotationData->getGraph());
1831 }
1832 }
1833 if ( auxVertex == std::numeric_limits< AnnotationGraphVertex >::max()
1834 || ppVertex == std::numeric_limits< AnnotationGraphVertex >::max())
1835 {
1836 return boost::shared_ptr< BoWToken >();
1837 }
1838 // parcourir les liens entrants su v annotes par Aux et PastPart
1839 // si la source est annotee CpdTense, rappeler recursivement
1840 // createCompoundTense ; sinon, creer un BoWToken simple
1841 // creer eventuellement un compound tense pour chaque ;
1842 boost::shared_ptr< BoWToken> head, extension;
1843 if (annotationData->hasIntAnnotation(ppVertex, Common::Misc::utf8stdstring2limastring("CpdTense")))
1844 {
1845 head = createCompoundTense(ppVertex, annotationData, anagraph, posgraph, offset, visited);
1846 }
1847 else
1848 {
1849 LinguisticGraphVertex ppTokVertex =
1850 *(annotationData->matches("annot", ppVertex, "PosGraph").begin());
1851
1852 std::vector< std::pair<boost::shared_ptr< BoWRelation >, boost::shared_ptr< AbstractBoWElement > > > ppBoWTokens = createAbstractBoWElement(ppTokVertex, anagraph, posgraph, offset, annotationData, visited, true);
1853 if (ppBoWTokens.empty())
1854 {
1855 return boost::shared_ptr< BoWToken >();
1856 }
1857 else
1858 {
1859 head = boost::dynamic_pointer_cast<BoWToken>(ppBoWTokens.back().second);
1860 ppBoWTokens.pop_back();
1861 }
1862 }
1863 if (head == 0)
1864 {
1865 return boost::shared_ptr< BoWToken >();
1866 }
1867 if (annotationData->hasIntAnnotation(auxVertex, Common::Misc::utf8stdstring2limastring("CpdTense")))
1868 {
1869 extension = createCompoundTense(auxVertex, annotationData, anagraph, posgraph, offset, visited);
1870 }
1871 else
1872 {
1873 LinguisticGraphVertex auxTokVertex =
1874 *(annotationData->matches("annot", auxVertex, "PosGraph").begin());
1875
1876 std::vector<std::pair<boost::shared_ptr< BoWRelation >, boost::shared_ptr< AbstractBoWElement > > > auxBoWTokens = createAbstractBoWElement(auxTokVertex, anagraph, posgraph, offset, annotationData, visited, true);
1877 if (auxBoWTokens.empty())
1878 {
1879 return boost::shared_ptr< BoWToken >();
1880 }
1881 else
1882 {
1883 extension = boost::dynamic_pointer_cast<BoWToken>(auxBoWTokens.back().second);
1884 auxBoWTokens.pop_back();
1885 }
1886 }
1887 if (extension == 0 && head != 0)
1888 {
1889 return boost::shared_ptr< BoWToken >();
1890 }
1891 // Build a BoWTerm with the preposition group as head and the aux as extension
1892
1893 boost::shared_ptr< BoWToken > complex(
1894 new BoWToken(
1895 head->getLemma(),
1896 head->getCategory(),
1897 extension->getPosition(),
1898 (head->getPosition()+head->getLength()-extension->getPosition())));
1899 complex->setVertex(head->getVertex());
1900 complex->setInflectedForm(head->getInflectedForm());
1901#ifdef DEBUG_LP
1902 LDEBUG << "BowGenerator: Built complex: " << *complex;
1903#endif
1904
1905 return complex;
1906}
1907
1908} // AnalysisDumper
1909
1910} // LinguisticProcessing
1911
1912} // Lima
#define LIMA_ANALYSISDUMPERS_EXPORT
This file is the main header file for the data related to annotation graphs.
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, AGVertexProperties, AGEdgeProperties > AnnotationGraph
The graph class.
DependencyGraph::out_edge_iterator DependencyGraphOutEdgeIt
DependencyGraph::vertex_descriptor DependencyGraphVertex
boost::property_map< DependencyGraph, edge_deprel_type_t >::const_type CEdgeDepRelTypePropertyMap
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, DepVertexProperties, DepEdgeProperties > DependencyGraph
The dependency graph class.
@ edge_deprel_type
#define LWARN
Definition LimaCommon.h:160
#define LIMA_UNUSED(x)
Definition LimaCommon.h:224
#define LDEBUG
Definition LimaCommon.h:157
#define LINFO
Definition LimaCommon.h:158
#define LERROR
Definition LimaCommon.h:161
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define DUMPERLOGINIT
#define L_NONE
Definition StdBitset.h:338
Data used for the syntactic analyzis of texts.
Holds an annotation graph and gives an API to manipulate it.
uint64_t intAnnotation(AnnotationGraphVertex v1, AnnotationGraphVertex v2, const LimaString &annot) const
std::set< AnnotationGraphVertex > matches(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second) const
Gets the set of vertices matched in the second graph by the given vertex of the first graph.
const GenericAnnotation & annotation(AnnotationGraphVertex v1, AnnotationGraphVertex v2, const LimaString &annot) const
const LimaString & stringAnnotation(AnnotationGraphVertex v1, AnnotationGraphVertex v2, const LimaString &annot) const
This is a complex token used to represent a named entity token.
This is a BoW element used to represent a predicate (n-ary relation, template or semantic frame).
This class is used to type the relation between a part and its enclosing complex token.
Definition BoWRelation.h:50
This is a complex token used to represent a multiword term.
Definition bowTerm.h:36
This class represents a list of elements, that are pointers on polymmorphic tokens that can be simple...
Definition bowText.h:48
This class contains the representation of an element of the bag of words.
Definition bowToken.h:46
Holds linguistic data for one language.
bool isAnEmptyMicroCategory(LinguisticCode id) const
bool isAnEmptyMacroCategory(LinguisticCode id) const
const FsaStringsPool & stringsPool(MediaId med) const
const MediaData & mediaData(MediaId media) const
LimaString getEntityName(const EntityType &type) const
EntityType getEntityType(const LimaString &entityName) const
entity types manager
Provide function to read write and check a property.
LinguisticCode readValue(const LinguisticCode &code) const
read a property in a coded int.
return a message when a 'param' was not found
The main LIMA exception class.
Definition LimaCommon.h:262
virtual const char * what() const override
Definition LimaCommon.h:280
static LinguisticCode fromUInt(uint64_t v)
Definition StdBitset.h:65
Parameters retrived in the configuration file:
std::vector< std::pair< boost::shared_ptr< Common::BagOfWords::BoWRelation >, boost::shared_ptr< Common::BagOfWords::BoWToken > > > buildTermFor(const AnnotationGraphVertex &vx, const AnnotationGraphVertex &tgt, const LinguisticGraph &anagraph, const LinguisticGraph &posgraph, const uint64_t offset, const SyntacticAnalysis::SyntacticData *syntacticData, const Common::AnnotationGraphs::AnnotationData *annotationData, std::set< LinguisticGraphVertex > &visited) const
Creates the terms reachable from the given annotation vertex.
boost::shared_ptr< Common::BagOfWords::BoWNamedEntity > createSpecificEntity(const LinguisticGraphVertex &vertex, const AnnotationGraphVertex &v, const Common::AnnotationGraphs::AnnotationData *annotationData, const LinguisticGraph &anagraph, const LinguisticGraph &posgraph, const uint64_t offset, bool frompos) const
std::vector< std::pair< boost::shared_ptr< Common::BagOfWords::BoWRelation >, boost::shared_ptr< Common::BagOfWords::AbstractBoWElement > > > createAbstractBoWElement(const LinguisticGraphVertex v, const LinguisticGraph &anagraph, const LinguisticGraph &posgraph, const uint64_t offsetBegin, const Common::AnnotationGraphs::AnnotationData *annotationData, std::set< LinguisticGraphVertex > &visited, bool keepAnyway=false) const
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, MediaId language)
QList< boost::shared_ptr< Common::BagOfWords::BoWPredicate > > createSemanticRelationPredicate(const LinguisticGraphVertex &lgvs, const AnnotationGraphVertex &agvs, const AnnotationGraphVertex &agvt, const SemanticAnalysis::SemanticRelationAnnotation &annot, const Common::AnnotationGraphs::AnnotationData *annotationData, const LinguisticGraph &anagraph, const LinguisticGraph &posgraph, uint64_t offset, std::set< LinguisticGraphVertex > &visited, bool keepAnyway) const
Builds a BoWPredicate corresponding to a semantic relation (an edge in the annotation graph holding a...
This enumeration lists the types of annotations used to specify the role of an annotation in a compou...
Common::MediaticData::ConceptType getConceptType() const
An annotation between two annotation graph vertices denoting the semantic relation(s) holding between...
A representation of a specific entity to store in the annotation graph.
This class points to a graph, its dependency graph and the structure that holds the maping between th...
DependencyGraphVertex depVertexForTokenVertex(const LinguisticGraphVertex &v) const
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
static MediaticData & changeable()
singleton accessor
Definition Singleton.h:71
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
AnnotationGraph::in_edge_iterator AnnotationGraphInEdgeIt
AnnotationGraph::out_edge_iterator AnnotationGraphOutEdgeIt
bool hasStringAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
bool hasIntAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
bool hasAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
std::vector< std::pair< Position, Length > > PositionLengthList
LimaString utf8stdstring2limastring(const std::string &src)
NAUTITIA.
QString LimaString
Definition LimaString.h:33
bool operator()(const std::pair< boost::shared_ptr< Common::BagOfWords::BoWRelation >, boost::shared_ptr< Common::BagOfWords::BoWToken > > t1, const std::pair< boost::shared_ptr< Common::BagOfWords::BoWRelation >, boost::shared_ptr< Common::BagOfWords::BoWToken > > t2) const
launch exception related to the configuration file parsing