LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
SpecificEntitiesConstraints.cpp
Go to the documentation of this file.
1// Copyright 2002-2019 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/***************************************************************************
7 * Copyright (C) 2004-2019 by CEA LIST *
8 * *
9 ***************************************************************************/
21#include <QStringList>
22#include <queue>
23#include <iostream>
24#include <sstream>
25
26// #include <assert.h>
27
28using namespace Lima::Common::MediaticData;
29using namespace Lima::Common::AnnotationGraphs;
34
35namespace Lima
36{
37
38namespace LinguisticProcessing
39{
40
41namespace SpecificEntities
42{
43
44// factories for constraint functions defined in this file
47
50
53
56
59
62
65
68
71
72
74isAlphaPossessive(MediaId language,
75 const LimaString& complement):
76ConstraintFunction(language,complement)
77{
78}
79
81 const LinguisticGraphVertex& v,
82 AnalysisContent& /*analysis*/) const
83{
84 LinguisticGraph* lingGraph = const_cast<LinguisticGraph*>(graph.getGraph());
85// Token* token=get(vertex_token,*(graph.getGraph()),v);
86 VertexTokenPropertyMap tokenMap = get(vertex_token, *lingGraph);
87 const TStatus& status = tokenMap[v]->status();
88 return( status.isAlphaPossessive() );
89}
90
91
93isASpecificEntity(MediaId language,
94 const LimaString& complement):
95ConstraintFunction(language,complement),
96m_type()
97{
98 if (! complement.isEmpty()) {
100 }
101}
102
104 const LinguisticGraphVertex& v,
105 AnalysisContent& analysis) const
106{
107 auto recoData = std::dynamic_pointer_cast<RecognizerData>(analysis.getData("RecognizerData"));
108 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.getData("AnnotationData"));
109 if (annotationData == 0)
110 {
111 return false;
112 }
113 bool annotFound = false;
114 //std::set< uint64_t > matches = annotationData->matches(recoData->getGraphId(),v,"annot"); portage 32 64
115 //for (std::set< uint64_t >::const_iterator it = matches.begin(); portage 32 64
116 std::set< AnnotationGraphVertex > matches = annotationData->matches(recoData->getGraphId(),v,"annot");
117 for (std::set< AnnotationGraphVertex >::const_iterator it = matches.begin();
118 it != matches.end(); it++)
119 {
120 if (annotationData->hasAnnotation(*it, Common::Misc::utf8stdstring2limastring("SpecificEntity")))
121 {
122 annotFound = true;
123 break;
124 }
125 }
126
127 if (!annotFound)
128 {
129 return false;
130 }
131
132 if (m_type == Common::MediaticData::EntityType())
133 {
134 return true;
135 }
136 else
137 {
138 //for (std::set< uint64_t >::const_iterator it = matches.begin(); portage 32 64
139 for (std::set< AnnotationGraphVertex >::const_iterator it = matches.begin();
140 it != matches.end(); it++)
141 {
142 if (annotationData->annotation(*it,
144 .pointerValue<SpecificEntityAnnotation>()->getType() == m_type)
145 {
146 return true;
147 }
148 }
149 return false;
150 }
151 return false;
152}
153
154
156 const LimaString& complement):
157ConstraintFunction(language,complement),
158m_language(language)
159{
160#ifdef DEBUG_LP
161 SELOGINIT;
162#endif
163
164 LimaString str=complement; // copy for easier parse (modify)
166 if (!str.isEmpty()) {
167 LimaString typeName;
168 int j=str.indexOf(sep);
169 if (j!=-1) {
170 typeName=str.left(j);
171 str=str.mid(j+1);
172 }
173 else {
174 typeName=complement;
175 str.clear();
176 }
177#ifdef DEBUG_LP
178 LDEBUG << "CreateSpecificEntity: getting entity type "
180#endif
181 try {
183 } catch (const LimaException& e) {
184 SELOGINIT;
185 LIMA_EXCEPTION("Unknown entity type" << typeName << e.what());
186 }
187 }
188
189 m_microAccessor=&(static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(language)).getPropertyCodeManager().getPropertyAccessor("MICRO"));
190
191 const Common::PropertyCode::PropertyManager& microManager=
192 static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(language)).getPropertyCodeManager().
193getPropertyManager("MICRO");
194
195 if (!str.isEmpty())
196 {
197 //uint64_t currentPos = 0; portage 32 64
198 int currentPos = 0;
199 while (currentPos != str.size()+1)
200 {
201 int sepPos = str.indexOf(sep, currentPos);
202
203 if (sepPos == -1)
204 {
205 sepPos = str.size();
206 }
207 std::string smicro=Common::Misc::limastring2utf8stdstring(str.mid(currentPos, sepPos-currentPos));
208 LinguisticCode micro = microManager.getPropertyValue(smicro);
209 m_microsToKeep.insert(micro);
210// LDEBUG << "Added " << smicro << " / " << micro << " to micros to keep";
211 currentPos = sepPos+1;
212 }
213 }
214}
215
218 AnalysisContent& analysis) const
219{
220#ifdef DEBUG_LP
221 SELOGINIT;
222 LDEBUG << "CreateSpecificEntity: create entity of type " << match.getType() << " on vertices " << match;
223#endif
224 if (match.empty()) return false;
225 LinguisticGraphVertex v1 = (*(match.begin())).m_elem.first;
226 LinguisticGraphVertex v2 = (*(match.rbegin())).m_elem.first;
228
229// LDEBUG << "CreateSpecificEntity action between " << v1 << " and " << v2
230// << " with complement " << m_complement;
231 auto syntacticData = std::dynamic_pointer_cast<SyntacticData>(analysis.getData("SyntacticData"));
232
233 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.getData("AnnotationData"));
234 if (annotationData==0)
235 {
236 return false;
237 }
238 // do not create annotation if annotation of same type exists
239 if (match.size() == 1)
240 {
241 //&& (isASpecificEntity(0,LimaString())(graph,v1,analysis)))
242 //std::set< uint64_t > matches = annotationData->matches(graph.getGraphId(),v1,"annot"); portage 32 64
243 //for (std::set< uint64_t >::const_iterator it = matches.begin(); portage 32 64
244 std::set< AnnotationGraphVertex > matches = annotationData->matches(graph.getGraphId(),v1,"annot");
245 for (std::set< AnnotationGraphVertex >::const_iterator it = matches.begin();
246 it != matches.end(); it++) {
247 if (annotationData->hasAnnotation(*it, Common::Misc::utf8stdstring2limastring("SpecificEntity"))
248 && annotationData->annotation(*it,
250 .pointerValue<SpecificEntityAnnotation>()->getType() == match.getType() ) {
251 return false;
252 }
253 }
254 }
255
256/* {
257 std::set< uint32_t > annots = annotationData->matches(recoData.getGraphId(), v1, "annot");
258 for ( std::set< uint32_t >::iterator it = annots.begin(); it != annots.end(); it++)
259 {
260 if (annotationData->hasAnnotation( *it, Common::Misc::utf8stdstring2limastring("SpecificEntity")))
261 }
262 }*/
263 if (annotationData->dumpFunction("SpecificEntity") == 0)
264 {
265 annotationData->dumpFunction("SpecificEntity", new DumpSpecificEntityAnnotation());
266 }
267
268 auto rdata = analysis.getData("RecognizerData");
269 if (rdata == 0) {
270 SELOGINIT;
271 LERROR << "CreateSpecificEntity: missing data RecognizerData: entity will not be created";
272 return false;
273 }
274 auto recoData = std::dynamic_pointer_cast<RecognizerData>(rdata);
275 if (recoData==0) {
276 SELOGINIT;
277 LERROR << "CreateSpecificEntity: missing data RecognizerData: entity will not be created";
278 return false;
279 }
280 std::string graphId=recoData->getGraphId();
281
282// LDEBUG << " match is " << match;
283
284// LDEBUG << " Creating annotation ";
285 SpecificEntityAnnotation annot(match, Common::MediaticData::MediaticData::changeable().stringsPool(m_language));
286 std::ostringstream oss;
287 annot.dump(oss);
288#ifdef DEBUG_LP
289 LDEBUG << "CreateSpecificEntity: annot = " << oss.str();
290
291// LDEBUG << " Building new morphologic data for head "<< annot.getHead();
292#endif
293 // getting data
294 LinguisticGraph* lingGraph = const_cast<LinguisticGraph*>(graph.getGraph());
295// LDEBUG << "There is " << out_degree(v2, *lingGraph) << " edges out of " << v2;
296 VertexTokenPropertyMap tokenMap = get(vertex_token, *lingGraph);
297 VertexDataPropertyMap dataMap = get(vertex_data, *lingGraph);
298
299 LinguisticGraphVertex head = annot.getHead();
300 if( head == 0 ) {
301 // take status of last element in match for eng
302 head = v2;
303 // or take status of first element in match (in fre?)
304 // head = v1;
305 }
306 const MorphoSyntacticData* dataHead = dataMap[head];
307
308 // Preparer le Token et le MorphoSyntacticData pour le nouveau noeud. Construits
309 // a partir des infos de l'entitee nommee
310 StringsPoolIndex seFlex = annot.getString();
311 // !!! RB: use normalized form for lemma : better for cascading rules, since its easier to write rules that
312 // match the expected normalized form given in a previous rule than rules that match the lemma, which is computed by LIMA
313 // @todo : check the impact of this change
314 //StringsPoolIndex seLemma = annot.getNormalizedString();
315 StringsPoolIndex seLemma = annot.getNormalizedForm();
316 StringsPoolIndex seNorm = annot.getNormalizedForm();
317
318// LDEBUG << " Creating LinguisticElement";
319// LDEBUG << " Creating MorphoSyntacticData";
320 // creer un MorphoSyntacticData
321 MorphoSyntacticData* newMorphData = new MorphoSyntacticData();
322
323 // all linguisticElements of this morphosyntacticData share common SE information
325 elem.inflectedForm = seFlex; // StringsPoolIndex
326 elem.lemma = seLemma; // StringsPoolIndex
327 elem.normalizedForm = seNorm; // StringsPoolIndex
328 elem.type = SPECIFIC_ENTITY; // MorphoSyntacticType
329
330 if (! m_microsToKeep.empty()) {
331#ifdef DEBUG_LP
332 LDEBUG << "CreateSpecificEntity, use micros from the rule ";
333#endif
334 // micros are given in the rules
335 addMicrosToMorphoSyntacticData(newMorphData,dataHead,m_microsToKeep,elem);
336 }
337 else {
338#ifdef DEBUG_LP
339 LDEBUG << "CreateSpecificEntity, use micros from config file ";
340#endif
341 // use micros given in the config file : get the specific resource
342 // (specific to modex) AddEntityFeature
343 // WARN : some hard coded stuff here in resource names
344 EntityType seType=match.getType();
345 if (seType.getGroupId() == static_cast<EntityGroupId>(0))
346 {
347 SELOGINIT;
348 LERROR << "CreateSpecificEntity::operator() null group id:" << seType;
349 delete newMorphData;
350 return false;
351 }
352 const LimaString& resourceName =
354 auto res = LinguisticResources::single().getResource(m_language,resourceName.toUtf8().constData());
355#ifdef DEBUG_LP
356 LDEBUG << "Entities resource name is : " << resourceName;
357#endif
358 if (res!=0) {
359 auto entityMicros = std::dynamic_pointer_cast<SpecificEntitiesMicros>(res);
360 auto micros = entityMicros->getMicros(seType);
361#ifdef DEBUG_LP
362 if (logger.isDebugEnabled())
363 {
364 std::ostringstream oss;
365 for (std::set<LinguisticCode>::const_iterator it=micros->begin(),it_end=micros->end();it!=it_end;it++) {
366 oss << (*it) << ";";
367 }
368 LDEBUG << "CreateSpecificEntity, micros are " << oss.str();
369 }
370#endif
371 addMicrosToMorphoSyntacticData(newMorphData,dataHead,*micros,elem);
372 }
373 else {
374 // cannot find micros for this type: error
375 SELOGINIT;
376 LERROR << "CreateSpecificEntity: missing resource " << resourceName ;
377 delete newMorphData;
378 return false;
379 }
380 }
381
383 Token* newToken = new Token(
384 seFlex,
385 sp[seFlex],
386 match.positionBegin(),
387 match.length());
388
389 // take posessive tstatus from head
390 TStatus tStatus = tokenMap[head]->status();
391 const TStatus& headTStatus = tokenMap[v2]->status();
392 if(headTStatus.isAlphaPossessive()) {
393 tStatus.setAlphaPossessive(true);
394 }
395 newToken->setStatus(tStatus);
396
397 if (newMorphData->empty())
398 {
399 SELOGINIT;
400 LERROR << "CreateSpecificEntity::operator() Found no morphosyntactic data for new vertex. Abort.";
401 delete newToken;
402 delete newMorphData;
403 assert(false);
404 return false;
405 }
406// LDEBUG << " Updating morphologic graph "<< graphId;
407 // creer le noeud et ses 2 arcs
408 LinguisticGraphVertex newVertex;
409 DependencyGraphVertex newDepVertex = 0;
410 if (syntacticData != 0)
411 {
412 boost::tie (newVertex, newDepVertex) = syntacticData->addVertex();
413 }
414 else
415 {
416 newVertex = add_vertex(*lingGraph);
417 }
418 AnnotationGraphVertex agv = annotationData->createAnnotationVertex();
419 annotationData->addMatching(graphId, newVertex, "annot", agv);
420 annotationData->annotate(agv, Common::Misc::utf8stdstring2limastring(graphId), newVertex);
421 tokenMap[newVertex] = newToken;
422 dataMap[newVertex] = newMorphData;
423#ifdef DEBUG_LP
424 LDEBUG << " - new vertex " << newVertex << "("<<graphId<<"), " << newDepVertex
425 << "(dep), " << agv << "(annot) added";
426
427// LDEBUG << " Setting annotation ";
428#endif
429 GenericAnnotation ga(annot);
430
431 annotationData->annotate(agv, Common::Misc::utf8stdstring2limastring("SpecificEntity"), ga);
432// LDEBUG << " Creating SE annotation edges between SE match vertices "
433// "annotation and the new annotation vertex";
434 Automaton::RecognizerMatch::const_iterator matchIt, matchIt_end;
435 matchIt = match.begin(); matchIt_end = match.end();
436 for (; matchIt != matchIt_end; matchIt++)
437 {
438 std::set< AnnotationGraphVertex > matches = annotationData->matches(graphId,(*matchIt).m_elem.first,"annot");
439 if (matches.empty())
440 {
441 SELOGINIT;
442 LERROR << "CreateSpecificEntity::operator() No annotation 'annot' for" << (*matchIt).m_elem.first;
443 }
444 else
445 {
446 if( recoData->hasVertexAsEmbededEntity((*matchIt).m_elem.first) )
447 {
448#ifdef DEBUG_LP
449 LDEBUG << "CreateSpecificEntity::operator(): vertex " << *(matches.begin()) << " is embeded";
450#endif
451 AnnotationGraphVertex src = *(matches.begin());
452 annotationData->annotate( agv, src, Common::Misc::utf8stdstring2limastring("holds"), 1);
453 }
454 }
455 }
456
457 // creer les relations necessaires dans le graphe morphosyntaxique
458 // 1. entre les noeuds avant v1 et le nouveau noeud
459 std::vector< LinguisticGraphVertex > previous;
460 LinguisticGraphInEdgeIt firstInEdgesIt, firstInEdgesIt_end;
461 boost::tie(firstInEdgesIt, firstInEdgesIt_end) = in_edges(v1, *lingGraph);
462 std::set< std::pair<LinguisticGraphVertex,LinguisticGraphVertex > > newEdgesToRemove;
463 for (; firstInEdgesIt != firstInEdgesIt_end; firstInEdgesIt++)
464 {
465 LinguisticGraphVertex firstInVertex = source(*firstInEdgesIt, *lingGraph);
466 previous.push_back(firstInVertex);
467 if (shouldRemoveInitial(source(*firstInEdgesIt, *lingGraph),target(*firstInEdgesIt, *lingGraph), match))
468 {
469#ifdef DEBUG_LP
470 LDEBUG << " - storing edge " << source(*firstInEdgesIt, *lingGraph) <<" -> "<<target(*firstInEdgesIt, *lingGraph) << " to be removed";
471#endif
472/* recoData->setEdgeToBeRemoved(analysis, *firstInEdgesIt);*/
473 newEdgesToRemove.insert(std::make_pair(source(*firstInEdgesIt, *lingGraph),target(*firstInEdgesIt, *lingGraph)));
474 }
475 else
476 {
477#ifdef DEBUG_LP
478 LDEBUG << " - do not store initial edge " << source(*firstInEdgesIt, *lingGraph) <<" -> "<<target(*firstInEdgesIt, *lingGraph) << " to be removed";
479#endif
480 }
481 }
482 std::vector< LinguisticGraphVertex >::iterator pit, pit_end;
483 pit = previous.begin(); pit_end = previous.end();
484 for (; pit != pit_end; pit++)
485 {
486 /* Si X-Y doit etre supprime et que Z remplace Y , alors ne pas creer X-Z
487 autrement dit si *pit-v1 est dans recoData->m_edgesToRemove, ne pas creer l'arc */
488 if (!recoData->isEdgeToBeRemoved(*pit, v1))
489 {
490 bool success;
492 boost::tie(e, success) = add_edge(*pit, newVertex, *lingGraph);
493 if (success)
494 {
495#ifdef DEBUG_LP
496 LDEBUG << " - in edge " << e.m_source << " -> " << e.m_target << " added";
497#endif
498 }
499 else
500 {
502 LERROR << " - in edge " << *pit << " ->" << newVertex << " NOT added";
503 }
504 }
505 else
506 {
507#ifdef DEBUG_LP
508 LDEBUG << " - edge " << *pit << " - " << newVertex << " not added because " << *pit << " - " << v1 << " has to be removed";
509#endif
510 }
511 }
512 std::set< std::pair<LinguisticGraphVertex,LinguisticGraphVertex > >::iterator newEdgesToRemoveIt = newEdgesToRemove.begin();
513 for (; newEdgesToRemoveIt != newEdgesToRemove.end(); newEdgesToRemoveIt++)
514 {
515 recoData->setEdgeToBeRemoved(analysis, edge( newEdgesToRemoveIt->first, newEdgesToRemoveIt->second, *lingGraph).first);
516 }
517#ifdef DEBUG_LP
518 LDEBUG << " - in edges added";
519#endif
520
521
522 // 2. entre le nouveau noeud et les noeuds qui etaient apres v2
523#ifdef DEBUG_LP
524 LDEBUG << " there is " << out_degree(v2, *lingGraph) << " edges out of " << v2;
525#endif
526 LinguisticGraphOutEdgeIt secondOutEdgesIt, secondOutEdgesIt_end;
527 boost::tie(secondOutEdgesIt, secondOutEdgesIt_end) = out_edges(v2, *lingGraph);
528 std::vector< LinguisticGraphVertex > nexts;
529 for (; secondOutEdgesIt != secondOutEdgesIt_end; secondOutEdgesIt++)
530 {
531#ifdef DEBUG_LP
532 LDEBUG << " looking at edge " << source(*secondOutEdgesIt, *lingGraph) << " -> " << target(*secondOutEdgesIt, *lingGraph);
533#endif
534 LinguisticGraphVertex secondOutVertex = target(*secondOutEdgesIt, *lingGraph);
535 if (secondOutVertex == v2) continue;
536 nexts.push_back(secondOutVertex);
537 if (shouldRemoveFinal(source(*secondOutEdgesIt, *lingGraph),target(*secondOutEdgesIt, *lingGraph), match))
538 {
539#ifdef DEBUG_LP
540 LDEBUG << " - storing edge " << source(*secondOutEdgesIt, *lingGraph) << " -> " << target(*secondOutEdgesIt, *lingGraph) << " to be removed";
541#endif
542 recoData->setEdgeToBeRemoved(analysis, *secondOutEdgesIt);
543 }
544 else
545 {
546#ifdef DEBUG_LP
547 LDEBUG << " - do not store final edge " << source(*secondOutEdgesIt, *lingGraph) << " -> " << target(*secondOutEdgesIt, *lingGraph) << " to be removed";
548#endif
549 }
550 }
551 std::vector< LinguisticGraphVertex >::iterator nit, nit_end;
552 nit = nexts.begin(); nit_end = nexts.end();
553 for (; nit != nit_end; nit++)
554 {
555 bool success;
557 boost::tie(e, success) = add_edge(newVertex, *nit, *lingGraph);
558 if (success)
559 {
560#ifdef DEBUG_LP
561 LDEBUG << " - out edge " << e.m_source << " -> " << e.m_target << " added";
562#endif
563 }
564 else
565 {
566 SELOGINIT;
567 LERROR << " - out edge " << newVertex << " ->" << *nit << " NOT added";
568 }
569 }
570#ifdef DEBUG_LP
571 LDEBUG << " - out edges added";
572#endif
573
574 // 3. supprimer les arcs a remplacer
575 recoData->removeEdges( analysis );
576
577 // 4 specifier le noeud suivant a utiliser dans la recherche :
578 // - nouveau noeud si l'expression reconnue etait composee de plusieurs noeuds
579 // - les fils du nouveau noeud sinon (pour eviter les bouclages)
580 // correction 08/2016 : always restart rules application from following vertices
581 // to avoid unexpected behaviors (see issues #44 and #45)
582 /*if (annot.m_vertices.size() > 1)
583 {
584 recoData->setNextVertex(newVertex);
585 }
586 else*/
587 {
588 LinguisticGraphOutEdgeIt outItr,outItrEnd;
589 boost::tie(outItr,outItrEnd) = out_edges(newVertex,*lingGraph);
590 for (;outItr!=outItrEnd;outItr++)
591 {
592 recoData->setNextVertex(target(*outItr, *lingGraph));
593 }
594 }
595 RecognizerMatch::const_iterator matchItr=match.begin();
596 for (; matchItr!=match.end(); matchItr++)
597 {
598 recoData->clearUnreachableVertices( analysis, (*matchItr).getVertex());
599 }
600
601 return true;
602}
603
604void CreateSpecificEntity::addMicrosToMorphoSyntacticData(LinguisticAnalysisStructure::MorphoSyntacticData* newMorphData,
606 const std::set<LinguisticCode>& micros,
608{
609 // try to filter existing microcategories
610 for (MorphoSyntacticData::const_iterator it=oldMorphData->begin(),
611 it_end=oldMorphData->end(); it!=it_end; it++) {
612
613 if (micros.find(m_microAccessor->readValue((*it).properties)) !=
614 micros.end()) {
615 elem.properties=(*it).properties;
616 newMorphData->push_back(elem);
617 }
618 }
619 // if no categories kept : assign all micros to keep
620 if (newMorphData->empty()) {
621 for (std::set<LinguisticCode>::const_iterator it=micros.begin(),
622 it_end=micros.end(); it!=it_end; it++) {
623 elem.properties=*it;
624 newMorphData->push_back(elem);
625 }
626 }
627}
628
629bool CreateSpecificEntity::shouldRemoveInitial(
630 LinguisticGraphVertex /*src*/,
631 LinguisticGraphVertex /*tgt*/,
632 const RecognizerMatch& match) const
633{
634#ifdef DEBUG_LP
635 SELOGINIT;
636#endif
637 if (match.size() == 1)
638 {
639 return true;
640 }
642
643 std::set< LinguisticGraphVertex > matchVertices;
644 Automaton::RecognizerMatch::const_iterator matchIt, matchIt_end;
645
646 matchIt = match.begin();
647 matchIt_end = match.end();
648 for (; matchIt != matchIt_end; matchIt++)
649 {
650 matchVertices.insert((*matchIt).m_elem.first);
651 }
652
653 matchIt = match.begin();
654 matchIt_end = match.end()-1;
655 if (boost::out_degree((*matchIt).m_elem.first,*graph.getGraph()) > 1)
656 {
657#ifdef DEBUG_LP
658 LDEBUG << "removing edge (" << (*matchIt).m_elem.first << "," << (*(matchIt+1)).m_elem.first << ") because there is more than one path from the first vertex of the match.";
659#endif
660 boost::remove_edge((*matchIt).m_elem.first,(*(matchIt+1)).m_elem.first, *const_cast<LinguisticGraph*>(graph.getGraph()));
661 return false;
662 }
663 matchIt++;
664 for (; matchIt != matchIt_end; matchIt++)
665 {
666 if (boost::out_degree((*matchIt).m_elem.first,*graph.getGraph()) > 1)
667 {
668 LinguisticGraphOutEdgeIt outIt, outIt_end;
669 boost::tie (outIt, outIt_end) = boost::out_edges((*matchIt).m_elem.first, *graph.getGraph());
670 for (; outIt != outIt_end; outIt++)
671 {
672 if (matchVertices.find(source(*outIt, *graph.getGraph())) != matchVertices.end())
673 {
674// LDEBUG << "removing initial edge " << *outIt;
675// boost::remove_edge(*outIt, *const_cast<LinguisticGraph*>(graph.getGraph()));
676 break;
677 }
678 }
679 return false;
680 }
681 }
682
683 return true;
684}
685
686bool CreateSpecificEntity::shouldRemoveFinal(
687 LinguisticGraphVertex /*src*/,
688 LinguisticGraphVertex /*tgt*/,
689 const RecognizerMatch& match) const
690{
691#ifdef DEBUG_LP
692 SELOGINIT;
693#endif
694 if (match.size() == 1)
695 {
696 return true;
697 }
699
700 std::set< LinguisticGraphVertex > matchVertices;
701 Automaton::RecognizerMatch::const_iterator matchIt, matchIt_end;
702
703 matchIt = match.begin();
704 matchIt_end = match.end();
705 for (; matchIt != matchIt_end; matchIt++)
706 {
707 matchVertices.insert((*matchIt).m_elem.first);
708 }
709
710 matchIt = match.begin()+1;
711 matchIt_end = match.end();
712 for (; matchIt != matchIt_end; matchIt++)
713 {
714 if (boost::in_degree((*matchIt).m_elem.first,*graph.getGraph()) > 1)
715 {
716 LinguisticGraphInEdgeIt inIt, inIt_end;
717 boost::tie (inIt, inIt_end) = boost::in_edges((*matchIt).m_elem.first, *graph.getGraph());
718 for (; inIt != inIt_end; inIt++)
719 {
720 if (matchVertices.find(source(*inIt, *graph.getGraph())) != matchVertices.end())
721 {
722#ifdef DEBUG_LP
723 LDEBUG << "removing final edge " << source(*inIt, *graph.getGraph()) << " -> " << target(*inIt, *graph.getGraph());
724#endif
725 boost::remove_edge(*inIt, *const_cast<LinguisticGraph*>(graph.getGraph()));
726 break;
727 }
728 }
729 return false;
730 }
731 }
732
733 return true;
734}
735
736
737//----------------------------------------------------------------------------------------
738// SetEntityFeature : add a given feature to the recognized entity
739// we do not have direct access to the RecognizerMatch of the entity when calling this function
740// (called during the matching process) => hence, store features in an AnalysisData and use
741// this Data in normalization functions or CreateSpecificEntity function to get the features.
742// Use already existing RecognizerData (no need for another Data).
743// CAREFUL: the features must be cleaned after use: explicit call to clearFeatures in case of
744// matching failure must be added in the rule.
745
747 const LimaString& complement):
748ConstraintFunction(language,complement),
749m_featureName(""),
750m_featureType(QVariant::String)
751{
752 if (complement.size()) {
753 QStringList complementElements = complement.split(":");
754 m_featureName=complementElements.front().toUtf8().constData();
755 complementElements.pop_front();
756 if (!complementElements.empty()) {
757 const QString& complementType = complementElements.front();
758 m_featureType = QVariant::nameToType(complementType.toUtf8().constData());
759 if (m_featureType != QVariant::Invalid) {
760 if (complementType == "int") {
761 m_featureType = QVariant::Int;
762 }
763 else if (complementType == "double") {
764 m_featureType = QVariant::Double;
765 }
766 else {
767 m_featureType = QVariant::String;
768 }
769 }
770 }
771 }
772}
773
776 const LinguisticGraphVertex& vertex,
777 AnalysisContent& analysis) const
778{
779#ifdef DEBUG_LP
780 SELOGINIT;
781 LDEBUG << "SetEntityFeature:: (one argument) start... ";
782 LDEBUG << "SetEntityFeature::(feature:" << m_featureName << ", vertex:" << vertex << ")";
783#endif
784 // get RecognizerData: the data in which the features are stored
785 auto recoData = std::dynamic_pointer_cast<RecognizerData>(analysis.getData("RecognizerData"));
786 if (recoData==0) {
787 SELOGINIT;
788 LERROR << "SetEntityFeature:: Error: missing RecognizerData";
789 return false;
790 }
791
792 // get string from the vertex and associate it to the feature
793
794 // get string from the vertex :
795 // @todo: if named entity, take normalized string, otherwise take lemma
796 LimaString featureValue;
797 Token* token=get(vertex_token,*(graph.getGraph()),vertex);
798
799 if (token == nullptr)
800 throw LimaException("Token is equal to nullptr");
801
802 if (token!=0) {
803 featureValue=token->stringForm();
804 }
805 switch (m_featureType) {
806 case QVariant::String:
807#ifdef DEBUG_LP
808 LDEBUG << "SetEntityFeature:: recoData->setEntityFeature(feature:" << m_featureName << ", featureValue:" << featureValue<< ")";
809#endif
810 recoData->setEntityFeature(m_featureName,featureValue);
811 break;
812 case QVariant::Int:
813#ifdef DEBUG_LP
814 LDEBUG << "SetEntityFeature:: recoData->setEntityFeature(feature:" << m_featureName << ", featureValue:" << featureValue.toInt() << ")";
815#endif
816 recoData->setEntityFeature(m_featureName,featureValue.toInt());
817 break;
818 case QVariant::Double:
819#ifdef DEBUG_LP
820 LDEBUG << "SetEntityFeature:: recoData->setEntityFeature(feature:" << m_featureName << ", featureValue:" << featureValue.toDouble() << ")";
821#endif
822 recoData->setEntityFeature(m_featureName,featureValue.toDouble());
823 break;
824 default:
825 recoData->setEntityFeature(m_featureName,featureValue);
826 }
827 uint64_t pos = (int64_t)(token->position());
828 uint64_t len = (int64_t)(token->length());
829 Automaton::EntityFeatures& features = recoData->getEntityFeatures();
830
831 std::vector<EntityFeature>::iterator featureIt =
832 features.find(m_featureName);
833 if( featureIt != recoData->getEntityFeatures().end() )
834 {
835 featureIt->setPosition(pos);
836 featureIt->setLength(len);
837 }
838
839 return true;
840}
841
844 const LinguisticGraphVertex& v1,
845 const LinguisticGraphVertex& v2,
846 AnalysisContent& analysis) const
847{
848#ifdef DEBUG_LP
849 SELOGINIT;
850 LDEBUG << "SetEntityFeature:: (two arguments) start... ";
851 LDEBUG << "SetEntityFeature::(feature:" << m_featureName << ", v1:" << v1 << ", v2:" << v2 << ")";
852#endif
853
854 // get RecognizerData: the data in which the features are stored
855 auto recoData = std::dynamic_pointer_cast<RecognizerData>(analysis.getData("RecognizerData"));
856 if (recoData==0) {
857 SELOGINIT;
858 LERROR << "SetEntityFeature:: Error: missing RecognizerData";
859 return false;
860 }
861 // get string from the set of vertices between v1 and v2
862 // @todo: if named entity, take normalized string, otherwise take lemma
863 LimaString featureValue;
864 const LinguisticGraph& lGraph = *(graph.getGraph());
865
866 // (some code borrowed from SpecificEntitiesXmlLogger::process)
867 // assert v2 follows v1 within a path composed with a direct sequence of out_edges
868 // assert also there exist no ambiguities in the graph.
869 std::queue<LinguisticGraphVertex> toVisit;
870 std::set<LinguisticGraphVertex> visited;
871 toVisit.push(v1);
872 uint64_t pos = UNDEFPOSITION;
873 uint64_t len = UNDEFLENGTH;
874
875 LinguisticGraphOutEdgeIt outItr,outItrEnd;
876 unsigned int nbEdges(0);
877 while (!toVisit.empty()) {
878 LinguisticGraphVertex v=toVisit.front();
879 toVisit.pop();
880 if (v != v2) {
881 for (boost::tie(outItr,outItrEnd)=out_edges(v,lGraph); outItr!=outItrEnd; outItr++)
882 {
883 LinguisticGraphVertex next=target(*outItr,lGraph);
884 if (visited.find(next)==visited.end())
885 {
886 visited.insert(next);
887 toVisit.push(next);
888 nbEdges++;
889 }
890 }
891 }
892 if( nbEdges > 1 ) {
893 SELOGINIT;
894 LWARN << "SetEntityFeature:: Warning: ambiguities in graph";
895 }
896
897 Token* token=get(vertex_token,lGraph,v);
898 if (v == v1) {
899 pos = (int64_t)(token->position());
900 }
901 if (v == v2) {
902 if( pos != UNDEFPOSITION )
903 len = (int64_t)(token->position()) - pos + 1 + (int64_t)(token->length());
904 }
905 // @ todo: add separator, check non standard cases where separator is no whitespace.
906 // see RecognizeMatch::getString()
907 featureValue.append( token->stringForm());
908 }
909
910 switch (m_featureType) {
911 case QVariant::String:
912 recoData->setEntityFeature(m_featureName,featureValue);
913 break;
914 case QVariant::Int:
915 recoData->setEntityFeature(m_featureName,featureValue.toInt());
916 break;
917 case QVariant::Double:
918 recoData->setEntityFeature(m_featureName,featureValue.toDouble());
919 break;
920 default:
921 recoData->setEntityFeature(m_featureName,featureValue);
922 }
923 Automaton::EntityFeatures& features = recoData->getEntityFeatures();
924 std::vector<EntityFeature>::iterator featureIt =
925 features.find(m_featureName);
926 if( featureIt != recoData->getEntityFeatures().end() )
927 {
928 featureIt->setPosition(pos);
929 featureIt->setLength(len);
930 }
931 return true;
932}
933
934//----------------------------------------------------------------------------------------
935// AddEntityFeatureAsEntity : assert the vertex is a named entity.
936// Add it to the list of components as an embeded entity (the list is used to create the link
937// "holds" between the annotation of the embeded and the embedding entity.
938// Remember the embedding entity is no yet created.
939
941 const LimaString& complement):
942ConstraintFunction(language,complement),
943m_featureName(),
944m_type()
945{
946 if (complement.size()) {
947 QStringList complementElements = complement.split(":");
948 m_featureName=complementElements.front().toUtf8().constData();
949 complementElements.pop_front();
950 if (!complementElements.empty()) {
951#ifdef DEBUG_LP
952 SELOGINIT;
953 LERROR << "AddEntityFeatureAsEntity::AddEntityFeatureAsEntity(): no type specification authorized for the feature ("
954 << complementElements << ") the feature type is the type of the entity";
955#endif
956 }
957 }
958}
959
962 const LinguisticGraphVertex& vertex,
963 AnalysisContent& analysis) const
964{
965#ifdef DEBUG_LP
966 SELOGINIT;
967 LDEBUG << "AddEntityFeatureAsEntity:: (one argument) start... ";
968 LDEBUG << "AddEntityFeatureAsEntity::(feature:" << m_featureName << ", vertex:" << vertex << ")";
969#endif
970 // get RecognizerData: the data in which the features are stored
971 auto recoData = std::dynamic_pointer_cast<RecognizerData>(analysis.getData("RecognizerData"));
972 if (recoData==0) {
973 SELOGINIT;
974 LERROR << "AddEntityFeatureAsEntity:: Error: missing RecognizerData";
975 return false;
976 }
977 // add the vertex to the list of embeded named entities
978 recoData->addVertexAsEmbededEntity(vertex);
979 return true;
980}
981
982//----------------------------------------------------------------------------------------
983// AddEntityFeature : add a value for a given feature to the recognized entity
984// we do not have direct access to the RecognizerMatch of the entity when calling this function
985// (called during the matching process) => hence, store features in an AnalysisData and use
986// this Data in normalization functions or CreateSpecificEntity function to get the features.
987// Use already existing RecognizerData (no need for another Data).
988// CAREFUL: the features must be cleaned after use: explicit call to clearFeatures in case of
989// matching failure must be added in the rule.
990
992 const LimaString& complement):
993ConstraintFunction(language,complement),
994m_featureName(""),
995m_featureType(QVariant::String)
996{
997 if (complement.size()) {
998 QStringList complementElements = complement.split(":");
999 m_featureName=complementElements.front().toUtf8().constData();
1000 complementElements.pop_front();
1001 if (!complementElements.empty()) {
1002 const QString& complementType = complementElements.front();
1003 m_featureType = QVariant::nameToType(complementType.toUtf8().constData());
1004 if (m_featureType != QVariant::Invalid) {
1005 if (complementType == "int") {
1006 m_featureType = QVariant::Int;
1007 }
1008 else if (complementType == "double") {
1009 m_featureType = QVariant::Double;
1010 }
1011 else {
1012 m_featureType = QVariant::String;
1013 }
1014 }
1015 }
1016 }
1017}
1018
1021 const LinguisticGraphVertex& vertex,
1022 AnalysisContent& analysis) const
1023{
1024#ifdef DEBUG_LP
1025 SELOGINIT;
1026 LDEBUG << "AddEntityFeature:: (one argument) start... ";
1027 LDEBUG << "AddEntityFeature::(feature:" << m_featureName << ", vertex:" << vertex << ")";
1028#endif
1029 // get RecognizerData: the data in which the features are stored
1030 auto recoData = std::dynamic_pointer_cast<RecognizerData>(analysis.getData("RecognizerData"));
1031 if (recoData==0) {
1032 SELOGINIT;
1033 LERROR << "AddEntityFeature:: Error: missing RecognizerData";
1034 return false;
1035 }
1036
1037 // get string from the vertex and associate it to the feature
1038
1039 // get string from the vertex :
1040 // @todo: if named entity, take normalized string, otherwise take lemma
1041 LimaString featureValue;
1042 Token* token=get(vertex_token,*(graph.getGraph()),vertex);
1043
1044 if (token == nullptr)
1045 throw LimaException("Token is equal to nullptr");
1046
1047 if (token!=0) {
1048 featureValue=token->stringForm();
1049 }
1050 switch (m_featureType) {
1051 case QVariant::String:
1052#ifdef DEBUG_LP
1053 LDEBUG << "AddEntityFeature:: recoData->addEntityFeature(feature:" << m_featureName << ", featureValue:" << featureValue<< ")";
1054#endif
1055 recoData->addEntityFeature(m_featureName,featureValue);
1056 break;
1057 case QVariant::Int:
1058#ifdef DEBUG_LP
1059 LDEBUG << "AddEntityFeature:: recoData->addEntityFeature(feature:" << m_featureName << ", featureValue:" << featureValue.toInt()<< ")";
1060#endif
1061 recoData->addEntityFeature(m_featureName,featureValue.toInt());
1062 break;
1063 case QVariant::Double:
1064#ifdef DEBUG_LP
1065 LDEBUG << "AddEntityFeature:: recoData->addEntityFeature(feature:" << m_featureName << ", featureValue:" << featureValue.toDouble()<< ")";
1066#endif
1067 recoData->addEntityFeature(m_featureName,featureValue.toDouble());
1068 break;
1069 default:
1070 recoData->addEntityFeature(m_featureName,featureValue);
1071 }
1072 uint64_t pos = (int64_t)(token->position());
1073 uint64_t len = (int64_t)(token->length());
1074 Automaton::EntityFeatures& features = recoData->getEntityFeatures();
1075
1076 std::vector<EntityFeature>::iterator featureIt =
1077 features.findLast(m_featureName);
1078 if( featureIt != recoData->getEntityFeatures().end() )
1079 {
1080 featureIt->setPosition(pos);
1081 featureIt->setLength(len);
1082 }
1083
1084 return true;
1085}
1086
1089 const LinguisticGraphVertex& v1,
1090 const LinguisticGraphVertex& v2,
1091 AnalysisContent& analysis) const
1092{
1093#ifdef DEBUG_LP
1094 SELOGINIT;
1095 LDEBUG << "AddEntityFeature:: (two arguments) start... ";
1096 LDEBUG << "AddEntityFeature::(feature:" << m_featureName << ", v1:" << v1 << ", v2:" << v2 << ")";
1097#endif
1098
1099 // get RecognizerData: the data in which the features are stored
1100 auto recoData = std::dynamic_pointer_cast<RecognizerData>(analysis.getData("RecognizerData"));
1101 if (recoData==0) {
1102 SELOGINIT;
1103 LERROR << "AddEntityFeature:: Error: missing RecognizerData";
1104 return false;
1105 }
1106 // get string from the set of vertices between v1 and v2
1107 // @todo: if named entity, take normalized string, otherwise take lemma
1108 LimaString featureValue;
1109 const LinguisticGraph& lGraph = *(graph.getGraph());
1110
1111 // (some code borrowed from SpecificEntitiesXmlLogger::process)
1112 // assert v2 follows v1 within a path composed with a direct sequence of out_edges
1113 // assert also there exist no ambiguities in the graph.
1114 std::queue<LinguisticGraphVertex> toVisit;
1115 std::set<LinguisticGraphVertex> visited;
1116 toVisit.push(v1);
1117 uint64_t pos = UNDEFPOSITION;
1118 uint64_t len = UNDEFLENGTH;
1119
1120 LinguisticGraphOutEdgeIt outItr,outItrEnd;
1121 unsigned int nbEdges(0);
1122 while (!toVisit.empty()) {
1123 LinguisticGraphVertex v=toVisit.front();
1124 toVisit.pop();
1125 if (v != v2) {
1126 for (boost::tie(outItr,outItrEnd)=out_edges(v,lGraph); outItr!=outItrEnd; outItr++)
1127 {
1128 LinguisticGraphVertex next=target(*outItr,lGraph);
1129 if (visited.find(next)==visited.end())
1130 {
1131 visited.insert(next);
1132 toVisit.push(next);
1133 nbEdges++;
1134 }
1135 }
1136 }
1137 if( nbEdges > 1 ) {
1138 SELOGINIT;
1139 LWARN << "AddEntityFeature:: Warning: ambiguities in graph";
1140 }
1141
1142 Token* token=get(vertex_token,lGraph,v);
1143 if (v == v1) {
1144 pos = (int64_t)(token->position());
1145 }
1146 if (v == v2) {
1147 if( pos != UNDEFPOSITION )
1148 len = (int64_t)(token->position()) - pos + 1 + (int64_t)(token->length());
1149 }
1150 // @ todo: add separator, check non standard cases where separator is no whitespace.
1151 // see RecognizeMatch::getString()
1152 featureValue.append( token->stringForm());
1153 }
1154
1155 switch (m_featureType) {
1156 case QVariant::String:
1157 recoData->setEntityFeature(m_featureName,featureValue);
1158 break;
1159 case QVariant::Int:
1160 recoData->setEntityFeature(m_featureName,featureValue.toInt());
1161 break;
1162 case QVariant::Double:
1163 recoData->setEntityFeature(m_featureName,featureValue.toDouble());
1164 break;
1165 default:
1166 recoData->setEntityFeature(m_featureName,featureValue);
1167 }
1168 Automaton::EntityFeatures& features = recoData->getEntityFeatures();
1169 std::vector<EntityFeature>::iterator featureIt =
1170 features.findLast(m_featureName);
1171 if( featureIt != recoData->getEntityFeatures().end() )
1172 {
1173 featureIt->setPosition(pos);
1174 featureIt->setLength(len);
1175 }
1176 return true;
1177}
1178
1179//----------------------------------------------------------------------------------------
1180// AppendEntityFeature : add a given feature to the recognized entity or append the
1181// value of a given feature if it already exists.
1182// Same remark as for SetEntityFeature about the relation with RecognizerMatch
1183// and RecognizerData.
1184
1186 const LimaString& complement):
1187ConstraintFunction(language,complement),
1188m_featureName(""),
1189m_featureType(QVariant::String)
1190{
1191 if (complement.size()) {
1192 QStringList complementElements = complement.split(":");
1193 m_featureName=complementElements.front().toUtf8().constData();
1194 complementElements.pop_front();
1195 if (!complementElements.empty()) {
1196 const QString& complementType = complementElements.front();
1197 m_featureType = QVariant::nameToType(complementType.toUtf8().constData());
1198 if (m_featureType != QVariant::Invalid) {
1199 if (complementType == "int") {
1200 m_featureType = QVariant::Int;
1201 }
1202 else if (complementType == "double") {
1203 m_featureType = QVariant::Double;
1204 }
1205 else {
1206 m_featureType = QVariant::String;
1207 }
1208 }
1209 }
1210 }
1211}
1212
1213uint64_t AppendEntityFeature::minPos( const uint64_t pos1, const uint64_t pos2 ) const {
1214 if( pos1 == UNDEFPOSITION )
1215 return pos2;
1216 if( pos2 == UNDEFPOSITION )
1217 return pos1;
1218 if( pos1 < pos2 )
1219 return pos1;
1220 return pos2;
1221}
1222
1223uint64_t AppendEntityFeature::maxPos( const uint64_t pos1, const uint64_t pos2 ) const {
1224 if( pos1 == UNDEFPOSITION )
1225 return pos2;
1226 if( pos2 == UNDEFPOSITION )
1227 return pos1;
1228 if( pos1 > pos2 )
1229 return pos1;
1230 return pos2;
1231}
1232
1235 const LinguisticGraphVertex& vertex,
1236 AnalysisContent& analysis) const
1237{
1238#ifdef DEBUG_LP
1239 SELOGINIT;
1240 LDEBUG << "AppendEntityFeature::() (one argument) start... ";
1241 LDEBUG << "AppendEntityFeature::() feature:" << m_featureName << ", vertex:" << vertex << ")";
1242#endif
1243 // get RecognizerData: the data in which the features are stored
1244 auto recoData = std::dynamic_pointer_cast<RecognizerData>(analysis.getData("RecognizerData"));
1245 if (recoData==0) {
1246 SELOGINIT;
1247 LERROR << "AppendEntityFeature::() Error: missing RecognizerData";
1248 return false;
1249 }
1250
1251 // get position/length from feature
1252 // position is min of position.
1253 // length is span from min of position
1254 // to max of position + augmented with length of last element.
1255 uint64_t pos = UNDEFPOSITION;
1256 uint64_t len = UNDEFLENGTH;
1257 Automaton::EntityFeatures& features = recoData->getEntityFeatures();
1258 std::vector<EntityFeature>::iterator featureIt = features.find(m_featureName);
1259 Token* token=get(vertex_token,*(graph.getGraph()),vertex);
1260// Token* token=get(vertex_token,lGraph,vertex);
1261 // if feature already exists
1262 if( featureIt != recoData->getEntityFeatures().end() )
1263 {
1264 pos = minPos( featureIt->getPosition(), (int64_t)(token->position()) );
1265#ifdef DEBUG_LP
1266 LDEBUG << "AppendEntityFeature::() minPos=" << pos;
1267#endif
1268 uint64_t maxPos1 = UNDEFPOSITION;
1269 if( featureIt->getPosition() != UNDEFPOSITION ) {
1270 maxPos1 = featureIt->getPosition() + featureIt->getLength();
1271 }
1272#ifdef DEBUG_LP
1273 LDEBUG << "AppendEntityFeature::() maxPos1=" << maxPos1;
1274 #endif
1275 uint64_t maxPos2 = (int64_t)(token->position()) + (int64_t)(token->length());
1276#ifdef DEBUG_LP
1277 LDEBUG << "AppendEntityFeature::() maxPos2=" << maxPos2;
1278#endif
1279 uint64_t maxPos3 = maxPos( maxPos1, maxPos2 );
1280#ifdef DEBUG_LP
1281 LDEBUG << "AppendEntityFeature::() maxPos3=" << maxPos3;
1282#endif
1283 len = maxPos3 - pos;
1284#ifdef DEBUG_LP
1285 LDEBUG << "AppendEntityFeature::() len=" << len;
1286#endif
1287 }
1288 else {
1289 pos = (int64_t)(token->position());
1290 len = (int64_t)(token->length());
1291 }
1292
1293 // get string from the vertex and associate it to the feature
1294
1295 // get string from the vertex :
1296 // @todo: if named entity, take normalized string, otherwise take lemma
1297 LimaString featureValue;
1298 if (token!=0) {
1299 featureValue=token->stringForm();
1300 }
1301 switch (m_featureType) {
1302 case QVariant::String:
1303 recoData->appendEntityFeature(m_featureName,featureValue);
1304 break;
1305 case QVariant::Int:
1306 recoData->appendEntityFeature(m_featureName,featureValue.toInt());
1307 break;
1308 case QVariant::Double:
1309 recoData->appendEntityFeature(m_featureName,featureValue.toDouble());
1310 break;
1311 default:
1312 recoData->appendEntityFeature(m_featureName,featureValue);
1313 }
1314 featureIt = features.find(m_featureName);
1315#ifdef DEBUG_LP
1316 LDEBUG << "AppendEntityFeature::() pos before = ("
1317 << featureIt->getPosition() << ","
1318 << featureIt->getLength() << ")";
1319#endif
1320 featureIt->setPosition(pos);
1321 featureIt->setLength(len);
1322#ifdef DEBUG_LP
1323 LDEBUG << "AppendEntityFeature::() pos after = (" << pos << "," << len << ")";
1324#endif
1325
1326 return true;
1327}
1328
1329// clear stored entity features, added by the SetEntityFeature function
1331const LimaString& complement):
1332ConstraintFunction(language,complement)
1333{
1334}
1336operator()(AnalysisContent& analysis) const
1337{
1338 // get RecognizerData: the data in which the features are stored
1339 auto recoData = std::dynamic_pointer_cast<RecognizerData>(analysis.getData("RecognizerData"));
1340 if (recoData!=0) {
1341 recoData->clearEntityFeatures();
1342 }
1343 return true;
1344}
1345
1346
1347//----------------------------------------------------
1348// Normalize entity using stored features
1350const LimaString& complement):
1351ConstraintFunction(language,complement)
1352{
1353}
1356{
1357 // get stored features in recognizerData
1358 auto recoData = std::dynamic_pointer_cast<RecognizerData>(analysis.getData("RecognizerData"));
1359 if (recoData == nullptr) {
1360 SELOGINIT;
1361 LERROR << "NormalizeEntity:: Error: missing RecognizerData";
1362 return false;
1363 }
1364 // assign stored features to RecognizerMatch features (preserving DEFAULT_ATTIBUTE)
1365 //match.features()=recoData->getEntityFeatures();
1366 const auto& features = recoData->getEntityFeatures();
1367 for (auto it = features.cbegin(); it != features.cend(); ++it) {
1368 match.features().addFeature(it->getName(),it->getValue());
1369 if( it->getPosition() != UNDEFPOSITION ) {
1370 auto featureIt = match.features().findLast(it->getName());
1371 (*featureIt).setPosition(it->getPosition());
1372 (*featureIt).setLength(it->getLength());
1373 }
1374 }
1375 // must clear the stored features, once they are used (otherwise, will be kept for next entity)
1376 recoData->clearEntityFeatures();
1377 return true;
1378}
1379
1380} // SpecificEntities
1381} // LinguisticProcessing
1382} // Lima
This file is the main header file for the data related to annotation graphs.
DependencyGraph::vertex_descriptor DependencyGraphVertex
#define UNDEFLENGTH
#define UNDEFPOSITION
#define LWARN
Definition LimaCommon.h:160
#define LIMA_EXCEPTION(X)
This macro writes the message X to a previously configured error stream before throwing a LimaExcepti...
Definition LimaCommon.h:293
#define LDEBUG
Definition LimaCommon.h:157
#define LERROR
Definition LimaCommon.h:161
LinguisticGraph::in_edge_iterator LinguisticGraphInEdgeIt
boost::property_map< LinguisticGraph, vertex_data_t >::type VertexDataPropertyMap
boost::graph_traits< LinguisticGraph >::edge_descriptor LinguisticGraphEdge
typedefs to simplify the access to various graphs elements
boost::property_map< LinguisticGraph, vertex_token_t >::type VertexTokenPropertyMap
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define SELOGINIT
#define SetEntityFeatureId
#define CreateSpecificEntityId
#define isAlphaPossessiveId
#define NormalizeEntityId
#define ClearEntityFeaturesId
#define AddEntityFeatureId
#define isASpecificEntityId
#define AppendEntityFeatureId
#define AddEntityFeatureAsEntityId
Data used for the syntactic analyzis of texts.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
This class allows to convert any object into an annotation by inheritance.
Holds linguistic data for one language.
const LimaString & getEntityGroupName(EntityGroupId id) const
const FsaStringsPool & stringsPool(MediaId med) const
const MediaData & mediaData(MediaId media) const
EntityType getEntityType(const LimaString &entityName) const
entity types manager
LinguisticCode readValue(const LinguisticCode &code) const
read a property in a coded int.
Provide tools to manage a specific property.
LinguisticCode getPropertyValue(const std::string &symbolicValue) const
Get the coded property value from the symbolic value.
The main LIMA exception class.
Definition LimaCommon.h:262
virtual const char * what() const override
Definition LimaCommon.h:280
a list of generic features: each feature is unique (only one feature for a name)
EntityFeatures::const_iterator findLast(const std::string &featureName) const
void addFeature(const std::string &name, const ValueType &value)
EntityFeatures::const_iterator find(const std::string &featureName) const
const Common::MediaticData::EntityType & getType() const
const LinguisticAnalysisStructure::AnalysisGraph * getGraph() const
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
void setStatus(const TStatus &status)
Set the TStatus of a token.
Definition Token.h:100
bool operator()(const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &vertex, AnalysisContent &analysis) const override
unary constraint function : the constraint only applies on one vertices in the graph (the graph is al...
AddEntityFeatureAsEntity(MediaId language, const LimaString &complement=LimaString())
bool operator()(const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &vertex, AnalysisContent &analysis) const override
unary constraint function : the constraint only applies on one vertices in the graph (the graph is al...
AddEntityFeature(MediaId language, const LimaString &complement=LimaString())
uint64_t maxPos(const uint64_t pos1, const uint64_t pos2) const
bool operator()(const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &vertex, AnalysisContent &analysis) const override
unary constraint function : the constraint only applies on one vertices in the graph (the graph is al...
uint64_t minPos(const uint64_t pos1, const uint64_t pos2) const
AppendEntityFeature(MediaId language, const LimaString &complement=LimaString())
ClearEntityFeatures(MediaId language, const LimaString &complement=LimaString())
bool operator()(AnalysisContent &analysis) const override
zero-ary constraint function : applies the function without a vertex indication (used for actions,...
CreateSpecificEntity(MediaId language, const LimaString &complement=LimaString())
bool operator()(Automaton::RecognizerMatch &match, AnalysisContent &analysis) const override
Definition of a function suitable to be used as a dumper for specific entities annotations of an anno...
NormalizeEntity(MediaId language, const LimaString &complement=LimaString())
bool operator()(Automaton::RecognizerMatch &match, AnalysisContent &analysis) const override
zero-ary constraint function : applies the function without a vertex indication (used for actions,...
SetEntityFeature(MediaId language, const LimaString &complement=LimaString())
bool operator()(const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &vertex, AnalysisContent &analysis) const override
unary constraint function : the constraint only applies on one vertices in the graph (the graph is al...
A representation of a specific entity to store in the annotation graph.
void dump(std::ostream &os) const
The functions that dumps a SpecificEntityAnnotation on an output stream.
bool operator()(const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &v, AnalysisContent &analysis) const override
unary constraint function : the constraint only applies on one vertices in the graph (the graph is al...
isASpecificEntity(MediaId language, const LimaString &complement=LimaString())
isAlphaPossessive(MediaId language, const LimaString &complement=LimaString())
bool operator()(const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &v, AnalysisContent &analysis) const override
unary constraint function : the constraint only applies on one vertices in the graph (the graph is al...
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
static MediaticData & changeable()
singleton accessor
Definition Singleton.h:71
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
LimaString utf8stdstring2limastring(const std::string &src)
ConstraintFunctionFactory< AddEntityFeatureAsEntity > AddEntityFeatureAsEntityFactory(AddEntityFeatureAsEntityId)
ConstraintFunctionFactory< AppendEntityFeature > AppendEntityFeatureFactory(AppendEntityFeatureId)
ConstraintFunctionFactory< ClearEntityFeatures > ClearEntityFeaturesFactory(ClearEntityFeaturesId)
ConstraintFunctionFactory< isAlphaPossessive > isAlphaPossessiveFactory(isAlphaPossessiveId)
ConstraintFunctionFactory< NormalizeEntity > NormalizeEntityFactory(NormalizeEntityId)
ConstraintFunctionFactory< CreateSpecificEntity > CreateSpecificEntityFactory(CreateSpecificEntityId)
ConstraintFunctionFactory< isASpecificEntity > isASpecificEntityFactory(isASpecificEntityId)
ConstraintFunctionFactory< SetEntityFeature > SetEntityFeatureFactory(SetEntityFeatureId)
ConstraintFunctionFactory< AddEntityFeature > AddEntityFeatureFactory(AddEntityFeatureId)
NAUTITIA.
QString LimaString
Definition LimaString.h:33