39#include <boost/tokenizer.hpp>
45using namespace boost::tuples;
56typedef boost::color_traits<boost::default_color_type>
Color;
59namespace LinguisticProcessing {
60namespace AnalysisDumpers {
61namespace EasyXmlDumper {
65 m_macroA(propertyCodeManager->getPropertyAccessor(
"MACRO")),
66 m_macroPm(propertyCodeManager->getPropertyManager(
"MACRO")),
67 m_microA(propertyCodeManager->getPropertyAccessor(
"MICRO")),
68 m_microPm(propertyCodeManager->getPropertyManager(
"MICRO"))
83 std::map< LinguisticAnalysisStructure::Token*, uint64_t >& fullTokens,
84 std::vector< bool >& alreadyDumpedTokens,
85 const MediaId& language)
89 LDEBUG <<
"ConstituantAndRelationExtractor:: begin visiting vertex " << v;
92 boost::tie(itLing, itLing_end) = out_edges(v, posGraph);
95 for (; itLing != itLing_end; itLing++)
97 const auto& nextVertex = target(*itLing, posGraph);
98 LDEBUG <<
"ConstituantAndRelationExtractor:: visiting out edge " << nextVertex;
112 alreadyDumpedTokens, language);
117 LDEBUG <<
"ConstituantAndRelationExtractor:: insert form in index "
119 m_formesIndex[forme->id] = forme;
120 m_positionsFormsIds.insert(std::make_pair(forme->poslong.position,
122 m_vertexToFormeIds.insert(std::make_pair(v,forme->id));
123 m_formeIdsToVertex.insert(std::make_pair(forme->id,v));
126 std::set<AnnotationGraphVertex> vAnnot = annotationData.
matches(
"PosGraph",
130 std::set<AnnotationGraphVertex>::const_iterator vAnnotIt, vAnnotIt_end;
131 vAnnotIt = vAnnot.begin();
132 vAnnotIt_end = vAnnot.end();
133 for (; vAnnotIt != vAnnotIt_end; vAnnotIt++)
136 m_posAnnotMatching.insert(std::make_pair(forme->poslong.position,*vAnnotIt));
137 m_annotPosMatching.insert(std::make_pair(*vAnnotIt,forme->poslong.position));
138 if(annotationData.
hasIntAnnotation(*vAnnotIt, QString::fromUtf8(
"CpdTense")))
140 bool foundCpdAux =
false, foundCpdPp =
false;
142 auxVertex = ppVertex = std::numeric_limits< AnnotationGraphVertex >::max();
144 boost::tie(vAnnotOutIt, vAnnotOutIt_end) = boost::out_edges(*vAnnotIt, annotationData.
getGraph());
145 for (; vAnnotOutIt != vAnnotOutIt_end; vAnnotOutIt++)
147 if (annotationData.
hasIntAnnotation(*vAnnotOutIt, QString::fromUtf8(
"Aux")))
149 auxVertex = boost::target(*vAnnotOutIt, annotationData.
getGraph());
150 if(auxVertex != 0) foundCpdAux =
true;
152 else if(annotationData.
hasIntAnnotation(*vAnnotOutIt, QString::fromUtf8(
"PastPart")))
154 ppVertex = boost::target(*vAnnotOutIt, annotationData.
getGraph());
155 if(ppVertex != 0) foundCpdPp =
true; }
157 if(foundCpdAux && foundCpdPp)
159 auto auxV = *(annotationData.
matches(
"annot", auxVertex,
"PosGraph").begin());
169 auto ppV = *(annotationData.
matches(
"annot", ppVertex,
"PosGraph").begin());
179 if(m_formesIndex[auxV] != 0 && m_formesIndex[ppV] != 0)
181 LDEBUG <<
"ConstituantAndRelationExtractor:: register compound tense "
182 << *vAnnotIt <<
", " << forme->forme;
183 m_compoundTenses[*vAnnotIt] = std::make_pair(auxV, ppV);
191 vAnnot = annotationData.
matches(
"PosGraph", v,
"AnalysisGraph");
194 std::set<AnnotationGraphVertex>::const_iterator vAnnotIt, vAnnotIt_end;
195 vAnnotIt = vAnnot.begin();
196 vAnnotIt_end = vAnnot.end();
197 for (; vAnnotIt != vAnnotIt_end; vAnnotIt++)
200 LimaString idiomExprLimaString = QString::fromUtf8(
"IdiomExpr");
201 if(annotationData.
hasAnnotation(*vAnnotIt, idiomExprLimaString))
203 LDEBUG <<
"ConstituantAndRelationExtractor:: found idiomatic " << *vAnnotIt <<
", " << forme->forme;
204 splitCompoundAnalysisAnnotation<IdiomaticExpressionAnnotation>(
215 LimaString seLimaString = QString::fromUtf8(
"SpecificEntity");
218 LDEBUG <<
"ConstituantAndRelationExtractor:: found specific entity " << *vAnnotIt <<
", " << forme->forme;
219 splitCompoundAnalysisAnnotation<SpecificEntityAnnotation>(
236 boost::tie(itDep, itDep_end) = out_edges(depv, *depGraph);
237 for (; itDep != itDep_end; itDep++)
239 Relation* relation =
extractEdge(*itDep, posGraph, *depGraph, fullTokens, syntacticData, language);
242 forme->m_outRelations.push_back(relation);
246 LDEBUG <<
"ConstituantAndRelationExtractor:: end visiting vertex " << v;
256 bool checkFullTokens,
257 std::map< LinguisticAnalysisStructure::Token*, uint64_t >& fullTokens,
258 std::vector< bool >& alreadyDumpedTokens,
270 LDEBUG <<
"ConstituantAndRelationExtractor:: check token " << v <<
"("
272 auto tokenIter = fullTokens.find(token);
273 if(tokenIter != fullTokens.end())
275 uint64_t tokenId = tokenIter->second;
276 if(alreadyDumpedTokens[tokenId])
278 alreadyDumpedTokens[tokenId] =
true;
282 LDEBUG <<
"ConstituantAndRelationExtractor:: extract vertex " << v <<
"("
298 std::vector<LinguisticElement>::const_iterator itForms = (*data).begin();
299 if(itForms != (*data).end())
302 LDEBUG <<
"ConstituantAndRelationExtractor:: found morphosyntactic data ";
314 if(itForms != (*data).end())
316 LWARN <<
"ConstituantAndRelationExtractor:: ommitting multiple morphosyntactic data ";
328 std::map< LinguisticAnalysisStructure::Token*, uint64_t >& ,
337 LDEBUG <<
"ConstituantAndRelationExtractor:: extract dependency " << relationName;
348 for(std::vector<Relation*>::iterator relationsIt = m_outRelations.begin(), relationsIt_end = m_outRelations.end();
349 relationsIt != relationsIt_end;
353 if(rel->
type == relationName &&
360 LDEBUG <<
"ConstituantAndRelationExtractor:: found duplicate relation";
368 relation->
type = relationName;
371 if(relation->
type ==
"COORD1" || relation->
type ==
"COORD2")
373 std::vector<Relation*>::iterator it, it_toErase;
374 it_toErase = m_outRelations.end();
375 for(it = m_outRelations.begin(); it != m_outRelations.end(); it++)
377 if( ( (*it)->type ==
"COORD1" && relation->
type ==
"COORD2" )
378 || ( (*it)->type ==
"COORD2" && relation->
type ==
"COORD1") )
380 if((*it)->srcVertex == relation->
tgtVertex)
382 LDEBUG <<
"ConstituantAndRelationExtractor:: found coord second edge";
383 (*it)->secondaryVertex = relation->
srcVertex;
387 else if((*it)->tgtVertex == relation->
srcVertex)
389 LDEBUG <<
"ConstituantAndRelationExtractor:: found coord second edge";
395 if(it_toErase != m_outRelations.end())
397 Forme* srcForme = m_formesIndex[(*it_toErase)->srcVertex];
400 std::vector<Relation*>::iterator formIt, formIt_toErase;
404 if(*formIt == *it_toErase)
406 formIt_toErase = formIt;
414 m_outRelations.erase(it_toErase);
415 LDEBUG <<
"ConstituantAndRelationExtractor:: erased coord second edge";
419 m_outRelations.push_back(relation);
438 std::set< uint64_t > formesDone;
439 std::map<uint64_t, uint64_t>::const_iterator formesIt, formesIt_end;
440 formesIt = m_positionsFormsIds.begin();
441 formesIt_end = m_positionsFormsIds.end();
444 LDEBUG <<
"ConstituantAndRelationExtractor:: construction des groupes";
446 for (; formesIt != formesIt_end; formesIt++)
449 uint64_t formeId = (*formesIt).second;
450 if (formesDone.find(formeId) == formesDone.end())
452 formesDone.insert(formeId);
453 Forme* forme = m_formesIndex[formeId];
456 LWARN <<
"ConstituantAndRelationExtractor:: form not found in index: " << formeId;
459 LDEBUG <<
"ConstituantAndRelationExtractor:: construction des groupes " << forme->
forme <<
"(" << forme->
macro <<
"/" << forme->
micro <<
")";
460 if (addToGroupIfIsInsideAGroup(forme))
462 LDEBUG <<
"ConstituantAndRelationExtractor:: already inserted ";
466 std::set<std::string> relationsToFollow;
467 relationsToFollow.insert(
"SECOMPOUND");
468 std::set<std::string> relationsToFollow2;
469 relationsToFollow2.insert(
"SECOMPOUND");
473 LDEBUG <<
"ConstituantAndRelationExtractor:: verbe non pointe par une relation SujInv";
474 createGroupe(forme, relationsToFollow,
"NV",
true);
476 else if ( forme->
micro ==
"ADV" )
478 createGroupe(forme, relationsToFollow,
"GR",
true);
480 else if (forme->
micro ==
"CLS")
482 relationsToFollow.insert(
"SujInv");
483 relationsToFollow.insert(
"TIl");
485 newGrp = createGroupe(forme, relationsToFollow,
"NV");
488 newGrp = createGroupe(forme, relationsToFollow,
"GN",
true);
491 else if (forme->
macro ==
"DET" || forme->
macro ==
"ADJ" )
493 relationsToFollow.insert(
"CodAnaph");
494 LDEBUG <<
"ConstituantAndRelationExtractor:: construction de groupe déterminant";
495 newGrp = createGroupe(forme, relationsToFollow,
"NV");
498 relationsToFollow.insert(
"PREPSUB");
499 relationsToFollow.insert(
"PrepPron");
500 relationsToFollow.insert(
"PrepDetInt");
501 relationsToFollow.insert(
"DetIntSub");
502 relationsToFollow.insert(
"DETSUB");
503 relationsToFollow.insert(
"ADJPRENSUB");
504 relationsToFollow.insert(
"ADVADJ");
505 relationsToFollow.insert(
"ADVADV");
506 relationsToFollow.insert(
"PrepPronRelCa");
507 relationsToFollow.insert(
"SUBSUBJUX;NPP;NPP");
509 LDEBUG <<
"ConstituantAndRelationExtractor:: insert '" << forme->
forme <<
"' in GP group";
510 newGrp = createGroupe(forme, relationsToFollow,
"GP");
514 relationsToFollow2.insert(
"DETSUB");
515 relationsToFollow2.insert(
"DetAdj");
516 relationsToFollow2.insert(
"ADJPRENSUB");
517 relationsToFollow2.insert(
"ADVADJ");
518 relationsToFollow2.insert(
"ADVADV");
519 relationsToFollow2.insert(
"COORD1;CC;ADJ");
520 relationsToFollow2.insert(
"COORD2;CC;ADJ");
521 LDEBUG <<
"ConstituantAndRelationExtractor:: insert '" << forme->
forme <<
"' in GN group";
522 createGroupe(forme, relationsToFollow2,
"GN");
540 else if ( forme->
micro ==
"ADV" )
542 newGrp = createGroupe(forme, relationsToFollow,
"GR",
true);
544 else if ( forme->
macro ==
"NP" )
546 relationsToFollow.insert(
"SUBSUBJUX;NP;NP");
547 relationsToFollow.insert(
"SUBADJPOST");
548 newGrp = createGroupe(forme, relationsToFollow,
"GN",
true);
550 else if ( forme->
micro ==
"DETWH" )
552 relationsToFollow.insert(
"DETSUB");
553 relationsToFollow.insert(
"DetIntSub");
554 relationsToFollow.insert(
"ADJPRENSUB");
555 newGrp = createGroupe(forme, relationsToFollow,
"GN",
true);
559 relationsToFollow.insert(
"SUBSUBJUX");
561 newGrp = createGroupe(forme, relationsToFollow,
"GN",
true);
563 else if ( forme->
micro ==
"PROREL" )
565 relationsToFollow.insert(
"COMPADV");
566 newGrp = createGroupe(forme, relationsToFollow,
"GN",
true);
568 else if ( forme->
micro ==
"PROREL"
569 || forme->
micro ==
"NC" || forme->
micro ==
"CLS" )
571 newGrp = createGroupe(forme, relationsToFollow,
"GN",
true);
576 newGrp = createGroupe(forme, relationsToFollow,
"NV",
true);
578 else if (forme->
macro ==
"PREP" || forme->
macro ==
"DET" )
580 relationsToFollow.insert(
"PREPSUB");
581 relationsToFollow.insert(
"PrepPron");
582 relationsToFollow.insert(
"PrepDetInt");
583 relationsToFollow.insert(
"DetIntSub");
584 relationsToFollow.insert(
"DETSUB");
585 relationsToFollow.insert(
"ADJPRENSUB");
586 relationsToFollow.insert(
"ADVADJ");
587 relationsToFollow.insert(
"ADVADV");
588 relationsToFollow.insert(
"PrepPronRelCa");
589 relationsToFollow.insert(
"PrepPronRel");
590 relationsToFollow.insert(
"PrepAdv");
591 relationsToFollow.insert(
"COORD1");
592 relationsToFollow.insert(
"COORD2");
593 relationsToFollow.insert(
"SUBSUBJUX;NPP;NPP");
595 newGrp = createGroupe(forme, relationsToFollow,
"GP");
599 relationsToFollow2.insert(
"PrepInf");
600 relationsToFollow2.insert(
"Neg");
601 relationsToFollow2.insert(
"NePas");
602 relationsToFollow2.insert(
"ADVADV");
603 relationsToFollow2.insert(
"AdvVerbe");
604 relationsToFollow2.insert(
"PronReflVerbe");
605 relationsToFollow2.insert(
"CodPrev");
606 relationsToFollow2.insert(
"AuxCplPrev");
607 newGrp = createGroupe(forme, relationsToFollow2,
"PV");
611 else if (forme->
micro ==
"ADV" || forme->
micro ==
"CLS" || forme->
micro ==
"CLO" )
613 relationsToFollow.insert(
"Neg");
614 relationsToFollow.insert(
"NePas");
615 relationsToFollow.insert(
"PronSujVerbe");
616 relationsToFollow.insert(
"PronReflVerbe");
617 relationsToFollow.insert(
"CodPrev");
618 relationsToFollow.insert(
"CoiPrev");
619 relationsToFollow.insert(
"AuxCplPrev");
620 newGrp = createGroupe(forme, relationsToFollow,
"NV");
622 else if ( forme->
micro ==
"NC" )
624 relationsToFollow.insert(
"SUBSUBJUX");
625 newGrp = createGroupe(forme, relationsToFollow,
"GN",
true);
627 else if (forme->
micro ==
"VPP" )
629 newGrp = createGroupe(forme, relationsToFollow,
"NV",
true);
631 else if ( forme->
macro ==
"ADJ" )
633 newGrp = createGroupe(forme, relationsToFollow,
"GA",
true);
635 else if (forme->
micro ==
"CLR")
637 relationsToFollow.insert(
"PronReflVerbe");
638 relationsToFollow.insert(
"AuxCplPrev");
640 newGrp = createGroupe(forme, relationsToFollow,
"NV");
642 else if ( forme->
micro ==
"PROREL" )
644 newGrp = createGroupe(forme, relationsToFollow,
"GP",
true);
648 newGrp = createGroupe(forme, relationsToFollow,
"GP",
true);
650 else if ( forme->
micro ==
"PRO" || forme->
micro ==
"PROWH" )
652 newGrp = createGroupe(forme, relationsToFollow,
"GN",
true);
662 LDEBUG <<
"ConstituantAndRelationExtractor:: inserted " << forme->
forme <<
" in " << newGrp->
type() <<
" group";
674 LDEBUG <<
"ConstituantAndRelationExtractor:: constructionDesRelationsEntrantes";
675 std::map<uint64_t,Forme*>::iterator formesIt, formesIt_end;
676 formesIt = m_formesIndex.begin();
677 formesIt_end = m_formesIndex.end();
678 for (; formesIt != formesIt_end; formesIt++)
680 Forme* forme = formesIt->second;
683 std::vector<Relation*>::iterator relsIt, relsIt_end;
686 for (; relsIt != relsIt_end; relsIt++)
689 uint64_t tgtFormeId = m_vertexToFormeIds[rel->
tgtVertex];
692 Forme* tgtForme = m_formesIndex[tgtFormeId];
703Groupe* ConstituantAndRelationExtractor::createGroupe(
705 const std::set<std::string>& theRelationsToFollow,
706 const std::string& groupType)
708 return createGroupe(forme, theRelationsToFollow, groupType,
false);
711Groupe* ConstituantAndRelationExtractor::createGroupe(
713 const std::set<std::string>& theRelationsToFollow,
714 const std::string& groupType,
715 bool mayBeUnique =
false)
718 LDEBUG <<
"ConstituantAndRelationExtractor:: createGroupe " << forme->forme <<
", " << groupType <<
" : ";
719 std::set<std::string> relationsToFollow;
720 std::map< std::string, std::set< std::pair< std::string, std::string > > > followConds;
721 std::set<std::string>::const_iterator fit, fit_end;
722 fit = theRelationsToFollow.begin(); fit_end = theRelationsToFollow.end();
723 for (; fit != fit_end; fit++)
725 if ( (*fit).find(
';') != std::string::npos )
727 size_t first = (*fit).find(
";");
728 size_t second = (*fit).find(
";", first+1);
729 std::string rel = (*fit).substr(0, first);
730 std::string srcCond = (*fit).substr(first+1, second-first-1);
731 std::string targCond = (*fit).substr(second+1);
732 if ((srcCond !=
"*") || (targCond !=
"*"))
734 if (followConds.find(rel) == followConds.end())
736 std::set< std::pair< std::string, std::string > > conds;
737 conds.insert(std::make_pair(srcCond, targCond));
738 followConds.insert(std::make_pair(rel, conds));
741 followConds[rel].insert(std::make_pair(srcCond, targCond));
743 relationsToFollow.insert(rel);
747 relationsToFollow.insert(*fit);
750 LDEBUG <<
"ConstituantAndRelationExtractor:: collected relations to insert";
751 Groupe* newGrp =
new Groupe();
752 newGrp->type(groupType);
753 std::vector< uint64_t > formsToLookup;
754 std::set< uint64_t > formsAlreadyLookuped;
755 formsToLookup.push_back(forme->id);
756 while (formsToLookup.size() > 0)
758 Forme* currentForm = m_formesIndex[formsToLookup.back()];
759 LDEBUG <<
"ConstituantAndRelationExtractor:: current form is " << currentForm->forme;
760 formsToLookup.pop_back();
761 if (m_inGroupFormsPositions.find(currentForm->poslong.position) != m_inGroupFormsPositions.end())
763 formsAlreadyLookuped.insert(currentForm->id);
765 std::vector<Relation*>::iterator inIt, inIt_end;
766 inIt = currentForm->m_inRelations.begin();
767 inIt_end = currentForm->m_inRelations.end();
768 for (; inIt != inIt_end; inIt++)
770 Relation* rel = *inIt;
771 LDEBUG <<
"ConstituantAndRelationExtractor:: looking at in rel " << rel->type;
772 if(relationsToFollow.find(rel->type) != relationsToFollow.end() &&
773 formsAlreadyLookuped.find(m_vertexToFormeIds[rel->srcVertex]) == formsAlreadyLookuped.end() &&
774 m_inGroupFormsPositions.find(m_formesIndex[m_vertexToFormeIds[rel->srcVertex]]->poslong.position) == m_inGroupFormsPositions.end()
777 const Forme* srcForme = m_formesIndex[m_vertexToFormeIds[rel->srcVertex]];
778 const Forme* tgtForme = m_formesIndex[m_vertexToFormeIds[rel->tgtVertex]];
779 std::pair< std::string, std::string > pair1 = std::make_pair(srcForme->macro, tgtForme->macro);
780 std::pair< std::string, std::string > pair2 = std::make_pair(srcForme->macro, tgtForme->micro);
781 std::pair< std::string, std::string > pair3 = std::make_pair(srcForme->macro,
"*");
782 std::pair< std::string, std::string > pair4 = std::make_pair(srcForme->micro, tgtForme->macro);
783 std::pair< std::string, std::string > pair5 = std::make_pair(srcForme->micro, tgtForme->micro);
784 std::pair< std::string, std::string > pair6 = std::make_pair(srcForme->micro,
"*");
785 std::pair< std::string, std::string > pair7 = std::make_pair(
"*", tgtForme->macro);
786 std::pair< std::string, std::string > pair8 = std::make_pair(
"*", tgtForme->micro);
787 if ((followConds.find(rel->type) == followConds.end())
788 || (followConds[rel->type].find(pair1) != followConds[rel->type].end())
789 || (followConds[rel->type].find(pair2) != followConds[rel->type].end())
790 || (followConds[rel->type].find(pair3) != followConds[rel->type].end())
791 || (followConds[rel->type].find(pair4) != followConds[rel->type].end())
792 || (followConds[rel->type].find(pair5) != followConds[rel->type].end())
793 || (followConds[rel->type].find(pair6) != followConds[rel->type].end())
794 || (followConds[rel->type].find(pair7) != followConds[rel->type].end())
795 || (followConds[rel->type].find(pair8) != followConds[rel->type].end()) )
797 formsToLookup.push_back( m_vertexToFormeIds[rel->srcVertex]);
798 formsAlreadyLookuped.insert(m_vertexToFormeIds[rel->srcVertex]);
799 LDEBUG <<
"ConstituantAndRelationExtractor:: ins src form '" << srcForme->forme <<
"' in " << newGrp->type();
800 newGrp->insert( std::make_pair(srcForme->poslong.position, srcForme->id) );
804 std::vector<Relation*>::iterator outIt, outIt_end;
805 outIt = currentForm->m_outRelations.begin();
806 outIt_end = currentForm->m_outRelations.end();
807 for (; outIt != outIt_end; outIt++)
809 Relation& rel = **outIt;
810 LDEBUG <<
"ConstituantAndRelationExtractor:: looking at out rel " << rel.type;
813 (m_vertexToFormeIds[rel.tgtVertex] != 0) &&
814 (m_formesIndex[m_vertexToFormeIds[rel.tgtVertex]] != 0) &&
815 (relationsToFollow.find(rel.type) != relationsToFollow.end()) &&
816 (formsAlreadyLookuped.find(m_vertexToFormeIds[rel.tgtVertex]) == formsAlreadyLookuped.end() ) &&
817 (m_inGroupFormsPositions.find(m_formesIndex[m_vertexToFormeIds[rel.tgtVertex]]->poslong.position) == m_inGroupFormsPositions.end() ) )
819 LDEBUG <<
"ConstituantAndRelationExtractor:: first condition fullfilled";
820 const Forme* srcForme = m_formesIndex[m_vertexToFormeIds[rel.srcVertex]];
821 const Forme* tgtForme = m_formesIndex[m_vertexToFormeIds[rel.tgtVertex]];
822 std::pair< std::string, std::string > pair1 = std::make_pair(srcForme->macro, tgtForme->macro);
823 std::pair< std::string, std::string > pair2 = std::make_pair(srcForme->macro, tgtForme->micro);
824 std::pair< std::string, std::string > pair3 = std::make_pair(srcForme->macro,
"*");
825 std::pair< std::string, std::string > pair4 = std::make_pair(srcForme->micro, tgtForme->macro);
826 std::pair< std::string, std::string > pair5 = std::make_pair(srcForme->micro, tgtForme->micro);
827 std::pair< std::string, std::string > pair6 = std::make_pair(srcForme->micro,
"*");
828 std::pair< std::string, std::string > pair7 = std::make_pair(
"*", tgtForme->macro);
829 std::pair< std::string, std::string > pair8 = std::make_pair(
"*", tgtForme->micro);
830 if ((followConds.find(rel.type) == followConds.end())
831 || (followConds[rel.type].find(pair1) != followConds[rel.type].end())
832 || (followConds[rel.type].find(pair2) != followConds[rel.type].end())
833 || (followConds[rel.type].find(pair3) != followConds[rel.type].end())
834 || (followConds[rel.type].find(pair4) != followConds[rel.type].end())
835 || (followConds[rel.type].find(pair5) != followConds[rel.type].end())
836 || (followConds[rel.type].find(pair6) != followConds[rel.type].end())
837 || (followConds[rel.type].find(pair7) != followConds[rel.type].end())
838 || (followConds[rel.type].find(pair8) != followConds[rel.type].end()) )
840 LDEBUG <<
"ConstituantAndRelationExtractor:: second condition fullfilled";
841 formsToLookup.push_back( m_vertexToFormeIds[rel.tgtVertex]);
842 formsAlreadyLookuped.insert(m_vertexToFormeIds[rel.tgtVertex]);
843 LDEBUG <<
"ConstituantAndRelationExtractor:: insert tgt form '" << tgtForme->forme <<
"' in " << newGrp->type() <<
" group";
844 newGrp->insert( std::make_pair(tgtForme->poslong.position, tgtForme->id) );
849 if (mayBeUnique || !newGrp->empty())
851 LDEBUG <<
"ConstituantAndRelationExtractor:: insert forme '" << forme->forme <<
"' in " << newGrp->type() <<
" group";
852 newGrp->insert( std::make_pair(forme->poslong.position, forme->id) );
853 insertGroup(*newGrp);
859void ConstituantAndRelationExtractor::insertGroup(
const Groupe& groupe)
861 uint64_t grpPos = m_formesIndex[(*(groupe.begin())).second]->poslong.position;
863 LDEBUG <<
"ConstituantAndRelationExtractor:: inserting group " << grpPos <<
" : ";
864 m_groupes.insert(std::make_pair(grpPos,groupe));
865 Groupe::const_iterator grpsIt, grpsIt_end;
866 grpsIt = groupe.begin(); grpsIt_end = groupe.end();
867 for (; grpsIt != grpsIt_end; grpsIt++)
869 m_inGroupFormsPositions.insert((*grpsIt).first);
873bool ConstituantAndRelationExtractor::addToGroupIfIsInsideAGroup(
const Forme* forme)
875 uint64_t position = forme->poslong.position;
877 std::map<uint64_t, Groupe>::const_iterator itg, itg_end;
878 itg = m_groupes.begin(); itg_end = m_groupes.end();
880 for (; itg != itg_end ; itg++)
882 if ( (*itg).first > pos && position >= (*itg).first )
887 else if ( (*itg).first > pos && (*itg).first > position )
891 if (m_groupes.find(pos) != m_groupes.end() && !m_groupes[pos].empty())
893 if (m_groupes[pos].rbegin() == m_groupes[pos].rend())
897 else if ( ( (*(m_groupes[pos].rbegin())).first >= position ) &&
898 ( (*(m_groupes[pos].begin())).first <= position ) )
901 LDEBUG <<
"ConstituantAndRelationExtractor:: insert '" << forme->forme <<
"' in " << m_groupes[pos].type() <<
" group";
902 m_groupes[pos].insert( std::make_pair(position, forme->id) );
903 m_inGroupFormsPositions.insert(position);
917 LDEBUG <<
"ConstituantAndRelationExtractor:: add last forms in groups";
918 std::map<uint64_t, uint64_t>formsInGroups;
919 std::map<uint64_t, Groupe>::iterator itGr, itGr_end;
920 itGr = m_groupes.begin();
921 itGr_end = m_groupes.end();
922 for (;itGr!=itGr_end;itGr++)
924 std::map<uint64_t, uint64_t>::iterator itForms, itForms_end;
925 itForms = ((*itGr).second).begin();
926 itForms_end = ((*itGr).second).end();
928 for (;itForms!=itForms_end;itForms++)
930 formsInGroups.insert(std::make_pair((*itForms).first, (*itForms).second));
933 std::map<uint64_t, uint64_t>::iterator it, it_end;
934 it = m_positionsFormsIds.begin();
935 it_end = m_positionsFormsIds.end();
936 for (;it!=it_end;it++)
938 if (formsInGroups.find((*it).first) == formsInGroups.end() )
940 if(m_formesIndex[(*it).second] != 0)
942 addToGroupIfIsInsideAGroup(m_formesIndex[(*it).second]);
951 LDEBUG <<
"ConstituantAndRelationExtractor:: splitCompoundTenses";
952 std::map<uint64_t, uint64_t>::iterator it, it_end;
953 it = m_positionsFormsIds.begin(); it_end = m_positionsFormsIds.end();
954 uint64_t compoundSplitted = 0;
955 for (; it != it_end; it++)
957 LDEBUG <<
"ConstituantAndRelationExtractor:: pos/id/annot="<<(*it).first<<
"/"<<(*it).second<<
"/"<<m_posAnnotMatching[(*it).first];
959 if (m_compoundTenses.find(m_posAnnotMatching[(*it).first]) != m_compoundTenses.end() )
961 uint64_t position = (*it).first;
964 uint64_t cpdtenseid = m_positionsFormsIds[position];
965 uint64_t auxid = m_compoundTenses[m_posAnnotMatching[(*it).first]].first;
966 uint64_t pastpartid = m_compoundTenses[m_posAnnotMatching[(*it).first]].second;
968 LDEBUG <<
"ConstituantAndRelationExtractor:: cpd tense: "<<cpdtenseid<<
"->("<<auxid <<
"," << pastpartid <<
")";
969 Forme* cpdtenseForme = m_formesIndex[cpdtenseid];
970 Forme* auxForme = m_formesIndex[auxid];
971 Forme* pastpartForme = m_formesIndex[pastpartid];
973 if(cpdtenseForme != 0 && auxForme != 0 && pastpartForme != 0){
976 LDEBUG <<
"ConstituantAndRelationExtractor:: replacing at " << position <<
" by " << auxForme->
forme <<
" (" << auxid <<
")";
977 m_positionsFormsIds[position] = auxid;
982 std::vector<Relation*>::iterator cpdTenseInRelsIt, cpdTenseInRelsIt_end;
985 for (;cpdTenseInRelsIt != cpdTenseInRelsIt_end; cpdTenseInRelsIt++)
987 Relation* cpdTenseInRel = *cpdTenseInRelsIt;
988 LDEBUG <<
"ConstituantAndRelationExtractor:: compound tense input relation = " << cpdTenseInRel->
type;
989 if (cpdTenseInRel->
type ==
"SujInv" || cpdTenseInRel->
type ==
"SUJ_V" || cpdTenseInRel->
type ==
"Neg" || cpdTenseInRel->
type ==
"PronSujVerbe")
991 LDEBUG <<
"ConstituantAndRelationExtractor:: change it from (" << cpdTenseInRel->
srcVertex <<
"-> " << cpdTenseInRel->
tgtVertex <<
") to (" << cpdTenseInRel->
srcVertex <<
"->" << auxForme->
forme <<
")";
992 cpdTenseInRel->
tgtVertex = m_formeIdsToVertex[auxForme->
id];
997 LDEBUG <<
"ConstituantAndRelationExtractor:: change it from (" << cpdTenseInRel->
srcVertex <<
"-> " << cpdTenseInRel->
tgtVertex <<
") to (" << cpdTenseInRel->
srcVertex <<
"->" << pastpartForme->
forme <<
")";
998 cpdTenseInRel->
tgtVertex = m_formeIdsToVertex[pastpartForme->
id];
1001 if (cpdTenseInRel->
type ==
"CodPrev" || cpdTenseInRel->
type ==
"CoiPrev" || cpdTenseInRel->
type ==
"PronSujVerbe" || cpdTenseInRel->
type ==
"PronReflVerbe" || cpdTenseInRel->
type ==
"PrepInf")
1003 Forme* srcForme = m_formesIndex[m_vertexToFormeIds[cpdTenseInRel->
srcVertex]];
1005 LDEBUG <<
"ConstituantAndRelationExtractor:: specific compound tense case: " << srcForme->
forme;
1009 cplRel->
tgtVertex = m_formeIdsToVertex[auxForme->
id];
1010 cplRel->
type =
"AuxCplPrev";
1011 m_outRelations.push_back(cplRel);
1022 std::vector<Relation*>::iterator cpdTenseOutRelsIt, cpdTenseOutRelsIt_end;
1025 for (;cpdTenseOutRelsIt != cpdTenseOutRelsIt_end; cpdTenseOutRelsIt++)
1027 Relation* cpdTenseOutRel = *cpdTenseOutRelsIt;
1028 LDEBUG <<
"ConstituantAndRelationExtractor:: compound tense output relation = " << cpdTenseOutRel->
type;
1029 LDEBUG <<
"ConstituantAndRelationExtractor:: change it from (" << cpdTenseOutRel->
srcVertex <<
"-> " << cpdTenseOutRel->
tgtVertex <<
") to (" << pastpartForme->
forme <<
"->" << cpdTenseOutRel->
tgtVertex <<
")";
1030 cpdTenseOutRel->
srcVertex = m_formeIdsToVertex[pastpartForme->
id];
1039 LDEBUG <<
"ConstituantAndRelationExtractor:: compound tense not found part";
1043 LDEBUG <<
"ConstituantAndRelationExtractor:: splitCompoundTenses DONE";
1045 if(compoundSplitted > 0){
1046 LDEBUG <<
"ConstituantAndRelationExtractor:: trying recursive splitCompoundTenses";
1055 std::vector<uint64_t> formsToErase;
1057 if (m_formesIndex.empty())
1061 std::vector<Relation*>::iterator relIt, relIt_end;
1063 std::map<uint64_t,Forme*>::iterator It, It_end;
1064 It = m_formesIndex.begin();
1065 It_end = m_formesIndex.end();
1067 for (;It!=It_end;It++)
1069 uint64_t position = (*It).first;
1070 Forme* forme = (*It).second;
1072 if (m_namedEntitiesVertices.find(position) != m_namedEntitiesVertices.end())
1075 LDEBUG <<
"ConstituantAndRelationExtractor:: se at " << position <<
" for " << forme->
forme ;
1077 uint64_t matchingVertex = m_posAnaMatching[((*It).first)];
1079 std::vector<uint64_t> tmpVector = m_seCompounds[matchingVertex];
1080 std::vector<uint64_t>::iterator vectIt, vectIt_end;
1081 vectIt = tmpVector.begin(); vectIt_end = tmpVector.end();
1084 formsToErase.push_back(position);
1085 bool firstPassage =
true;
1086 Forme* precForm = 0;
1087 uint64_t maxVertex = (*(m_formesIndex.rbegin())).first;
1088 for (;vectIt!=vectIt_end;vectIt++)
1091 Forme* tmpForme = m_anaGraphVertices[*vectIt];
1092 LDEBUG <<
"ConstituantAndRelationExtractor:: se compound: " << tmpForme->
forme;
1094 tmpForme->
id = maxVertex+1;
1097 m_formesIndex[tmpForme->
id]=tmpForme;
1098 tmpForme->
macro = m_namedEntitiesVertices[(*It).first].macro;
1099 tmpForme->
micro = m_namedEntitiesVertices[(*It).first].micro;
1111 for (;relIt != relIt_end; relIt++)
1113 (*relIt)->srcVertex = tmpForme->
id;
1117 relIt = m_outRelations.begin();
1118 relIt_end = m_outRelations.end();
1119 for (;relIt != relIt_end;relIt++)
1121 if ((*relIt)->srcVertex == (*It).first)
1122 (*relIt)->srcVertex = tmpForme->
id;
1123 else if ((*relIt)->tgtVertex == (*It).first)
1124 (*relIt)->tgtVertex = tmpForme->
id;
1128 relIt = m_outRelations.begin();
1129 relIt_end = m_outRelations.end();
1130 for(; relIt!=relIt_end;relIt++)
1132 if ((*relIt)->tgtVertex == position)
1134 LDEBUG <<
"ConstituantAndRelationExtractor:: update relation target " << (*relIt)->tgtVertex;
1135 (*relIt)->tgtVertex = tmpForme->
id;
1138 firstPassage =
false;
1140 else if(precForm != 0)
1145 rel->
type =
"SECOMPOUND";
1147 m_outRelations.push_back(rel);
1151 LDEBUG <<
"ConstituantAndRelationExtractor:: adding compound " << tmpForme->
forme <<
", " << tmpForme->
id <<
"(" << tmpForme->
poslong.
position <<
")";
1152 std::map< LinguisticAnalysisStructure::Token*, uint64_t >::const_iterator tokenIter;
1154 m_positionsFormsIds.insert(std::make_pair(tmpForme->
poslong.
position, tmpForme->
id));
1155 m_vertexToFormeIds.insert(std::make_pair(tmpForme->
id, tmpForme->
id));
1156 m_formeIdsToVertex.insert(std::make_pair(tmpForme->
id, tmpForme->
id));
1158 precForm = tmpForme;
1166 std::vector<uint64_t>::const_iterator eraseIt, eraseIt_end;
1167 eraseIt = formsToErase.begin();
1168 eraseIt_end = formsToErase.end();
1169 for (;eraseIt!=eraseIt_end;eraseIt++)
1172 LDEBUG <<
"ConstituantAndRelationExtractor:: erase compound " << *eraseIt;
1173 m_formesIndex.erase(*eraseIt);
boost::color_traits< boost::default_color_type > Color
extracts forms and relations from boost graph (origninally, from XML file)
DependencyGraph::out_edge_iterator DependencyGraphOutEdgeIt
DependencyGraph::vertex_descriptor DependencyGraphVertex
boost::property_map< DependencyGraph, edge_deprel_type_t >::const_type CEdgeDepRelTypePropertyMap
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, DepVertexProperties, DepEdgeProperties > DependencyGraph
The dependency graph class.
boost::graph_traits< LinguisticGraph >::edge_descriptor LinguisticGraphEdge
typedefs to simplify the access to various graphs elements
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
Data used for the syntactic analyzis of texts.
Holds an annotation graph and gives an API to manipulate it.
std::set< AnnotationGraphVertex > matches(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second) const
Gets the set of vertices matched in the second graph by the given vertex of the first graph.
Provide function to read write and check a property.
LinguisticCode readValue(const LinguisticCode &code) const
read a property in a coded int.
Provide tools to parse a property file, and deal with the property coding system.
const PropertyManager & getPropertyManager(const std::string &propertyName) const
Get the PropertyManager associated to a property.
const PropertyAccessor & getPropertyAccessor(const std::string &propertyName) const
Get the PropertyAccessor associated to a property.
Provide tools to manage a specific property.
const std::string & getPropertySymbolicValue(const LinguisticCode &value) const
The coded property value can hold several property data.
void splitCompoundTenses()
void constructionDesGroupes()
construit les groupes syntaxiques apres la lecture du graphe
void addLastFormsInGroups()
void constructionDesRelationsEntrantes()
construit les relations entrantes des formes
void visitBoostGraph(const LinguisticGraphVertex &v, const LinguisticGraphVertex &end, const LinguisticGraph &anaGraph, const LinguisticGraph &posGraph, const Common::AnnotationGraphs::AnnotationData &annotationData, const SyntacticAnalysis::SyntacticData &syntacticData, std::map< LinguisticAnalysisStructure::Token *, uint64_t > &fullTokens, std::vector< bool > &alreadyDumpedTokens, const MediaId &language)
ConstituantAndRelationExtractor(const Common::PropertyCode::PropertyCodeManager *propertyCodeManager)
Relation * extractEdge(const LinguisticGraphEdge &e, const LinguisticGraph &posGraph, const DependencyGraph &depGraph, std::map< LinguisticAnalysisStructure::Token *, uint64_t > &fullTokens, const SyntacticAnalysis::SyntacticData &syntacticData, MediaId language)
void replaceSEWithCompounds()
virtual ~ConstituantAndRelationExtractor()
Forme * extractVertex(const LinguisticGraphVertex &v, const LinguisticGraph &graph, bool checkFullTokens, std::map< LinguisticAnalysisStructure::Token *, uint64_t > &fullTokens, std::vector< bool > &alreadyDumpedTokens, MediaId language)
Definition d'un groupe syntaxique.
const std::string & type() const
Holds morphosyntactic informations.
holds surface data of a token
uint64_t position() const
const LimaString & stringForm() const
This class points to a graph, its dependency graph and the structure that holds the maping between th...
LinguisticGraphVertex tokenVertexForDepVertex(const DependencyGraphVertex &v) const
DependencyGraphVertex depVertexForTokenVertex(const LinguisticGraphVertex &v) const
DependencyGraph * dependencyGraph()
static const MediaticData & single()
const singleton accessor
static MediaticData & changeable()
singleton accessor
dump the content of the analysis graph in Easy XML format
AnnotationGraph & getGraph()
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
AnnotationGraph::out_edge_iterator AnnotationGraphOutEdgeIt
bool hasIntAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
bool hasAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
uint64_t secondaryVertex
to store the third element of a 3-ary relation, currently only the source of the COORD2 relation.