LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
ConstituantAndRelationExtractor.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
22#include "easyXmlDumper.h"
23
24#include "common/LimaCommon.h"
33
34#include <algorithm>
35#include <iostream>
36#include <sstream>
37#include <set>
38
39#include <boost/tokenizer.hpp>
40
41using namespace Lima;
42
43using namespace std;
44//using namespace boost;
45using namespace boost::tuples;
46
47
48using namespace Lima::Common::PropertyCode;
49using namespace Lima::Common::MediaticData;
50using namespace Lima::Common::AnnotationGraphs;
55
56typedef boost::color_traits<boost::default_color_type> Color;
57
58namespace Lima {
59namespace LinguisticProcessing {
60namespace AnalysisDumpers {
61namespace EasyXmlDumper {
62
64 const Common::PropertyCode::PropertyCodeManager* propertyCodeManager) :
65 m_macroA(propertyCodeManager->getPropertyAccessor("MACRO")),
66 m_macroPm(propertyCodeManager->getPropertyManager("MACRO")),
67 m_microA(propertyCodeManager->getPropertyAccessor("MICRO")),
68 m_microPm(propertyCodeManager->getPropertyManager("MICRO"))
69{}
70
73
74#undef min
75#undef max
76
78 const LinguisticGraphVertex& end,
79 const LinguisticGraph& anaGraph,
80 const LinguisticGraph& posGraph,
81 const AnnotationData& annotationData,
82 const SyntacticData& syntacticData,
83 std::map< LinguisticAnalysisStructure::Token*, uint64_t >& fullTokens,
84 std::vector< bool >& alreadyDumpedTokens,
85 const MediaId& language)
86{
87
89 LDEBUG << "ConstituantAndRelationExtractor:: begin visiting vertex " << v;
90
91 LinguisticGraphOutEdgeIt itLing, itLing_end;
92 boost::tie(itLing, itLing_end) = out_edges(v, posGraph);
93 if(v != end)
94 {
95 for (; itLing != itLing_end; itLing++)
96 {
97 const auto& nextVertex = target(*itLing, posGraph);
98 LDEBUG << "ConstituantAndRelationExtractor:: visiting out edge " << nextVertex;
99 visitBoostGraph(nextVertex,
100 end,
101 anaGraph,
102 posGraph,
103 annotationData,
104 syntacticData,
105 fullTokens,
106 alreadyDumpedTokens,
107 language);
108 }
109 }
110
111 auto forme = extractVertex(v, posGraph, true, fullTokens,
112 alreadyDumpedTokens, language);
113
114 if (forme == 0)
115 return;
116
117 LDEBUG << "ConstituantAndRelationExtractor:: insert form in index "
118 << forme->id;
119 m_formesIndex[forme->id] = forme;
120 m_positionsFormsIds.insert(std::make_pair(forme->poslong.position,
121 forme->id));
122 m_vertexToFormeIds.insert(std::make_pair(v,forme->id));
123 m_formeIdsToVertex.insert(std::make_pair(forme->id,v));
124
125 // Looking for compound tense in annotation data
126 std::set<AnnotationGraphVertex> vAnnot = annotationData.matches("PosGraph",
127 v, "annot");
128 if (!vAnnot.empty())
129 {
130 std::set<AnnotationGraphVertex>::const_iterator vAnnotIt, vAnnotIt_end;
131 vAnnotIt = vAnnot.begin();
132 vAnnotIt_end = vAnnot.end();
133 for (; vAnnotIt != vAnnotIt_end; vAnnotIt++)
134 {
135 // if corresponding AnnotationGraph vertex, link it
136 m_posAnnotMatching.insert(std::make_pair(forme->poslong.position,*vAnnotIt));
137 m_annotPosMatching.insert(std::make_pair(*vAnnotIt,forme->poslong.position));
138 if(annotationData.hasIntAnnotation(*vAnnotIt, QString::fromUtf8("CpdTense")))
139 {
140 bool foundCpdAux = false, foundCpdPp = false;
141 AnnotationGraphVertex auxVertex, ppVertex;
142 auxVertex = ppVertex = std::numeric_limits< AnnotationGraphVertex >::max();
143 AnnotationGraphOutEdgeIt vAnnotOutIt, vAnnotOutIt_end;
144 boost::tie(vAnnotOutIt, vAnnotOutIt_end) = boost::out_edges(*vAnnotIt, annotationData.getGraph());
145 for (; vAnnotOutIt != vAnnotOutIt_end; vAnnotOutIt++)
146 {
147 if (annotationData.hasIntAnnotation(*vAnnotOutIt, QString::fromUtf8("Aux")))
148 {
149 auxVertex = boost::target(*vAnnotOutIt, annotationData.getGraph());
150 if(auxVertex != 0) foundCpdAux = true;
151 }
152 else if(annotationData.hasIntAnnotation(*vAnnotOutIt, QString::fromUtf8("PastPart")))
153 {
154 ppVertex = boost::target(*vAnnotOutIt, annotationData.getGraph());
155 if(ppVertex != 0) foundCpdPp = true; }
156 }
157 if(foundCpdAux && foundCpdPp)
158 {
159 auto auxV = *(annotationData.matches("annot", auxVertex, "PosGraph").begin());
160 visitBoostGraph(auxV,
161 end,
162 anaGraph,
163 posGraph,
164 annotationData,
165 syntacticData,
166 fullTokens,
167 alreadyDumpedTokens,
168 language);
169 auto ppV = *(annotationData.matches("annot", ppVertex, "PosGraph").begin());
170 visitBoostGraph(ppV,
171 end,
172 anaGraph,
173 posGraph,
174 annotationData,
175 syntacticData,
176 fullTokens,
177 alreadyDumpedTokens,
178 language);
179 if(m_formesIndex[auxV] != 0 && m_formesIndex[ppV] != 0)
180 {
181 LDEBUG << "ConstituantAndRelationExtractor:: register compound tense "
182 << *vAnnotIt << ", " << forme->forme;
183 m_compoundTenses[*vAnnotIt] = std::make_pair(auxV, ppV);
184 }
185 }
186 }
187 }
188 }
189
190 // Looking for idiomatic expressions in analysis data
191 vAnnot = annotationData.matches("PosGraph", v, "AnalysisGraph");
192 if (!vAnnot.empty())
193 {
194 std::set<AnnotationGraphVertex>::const_iterator vAnnotIt, vAnnotIt_end;
195 vAnnotIt = vAnnot.begin();
196 vAnnotIt_end = vAnnot.end();
197 for (; vAnnotIt != vAnnotIt_end; vAnnotIt++)
198 {
199
200 LimaString idiomExprLimaString = QString::fromUtf8("IdiomExpr");
201 if(annotationData.hasAnnotation(*vAnnotIt, idiomExprLimaString))
202 {
203 LDEBUG << "ConstituantAndRelationExtractor:: found idiomatic " << *vAnnotIt << ", " << forme->forme;
204 splitCompoundAnalysisAnnotation<IdiomaticExpressionAnnotation>(
205 *vAnnotIt,
206 *forme,
207 idiomExprLimaString,
208 annotationData,
209 anaGraph,
210 fullTokens,
211 alreadyDumpedTokens,
212 language);
213 }
214
215 LimaString seLimaString = QString::fromUtf8("SpecificEntity");
216 if(annotationData.hasAnnotation(*vAnnotIt, seLimaString))
217 {
218 LDEBUG << "ConstituantAndRelationExtractor:: found specific entity " << *vAnnotIt << ", " << forme->forme;
219 splitCompoundAnalysisAnnotation<SpecificEntityAnnotation>(
220 *vAnnotIt,
221 *forme,
222 seLimaString,
223 annotationData,
224 anaGraph,
225 fullTokens,
226 alreadyDumpedTokens,
227 language);
228 }
229
230 }
231 }
232
233 const DependencyGraph* depGraph = syntacticData.dependencyGraph();
234 DependencyGraphVertex depv = syntacticData.depVertexForTokenVertex(v);
235 DependencyGraphOutEdgeIt itDep, itDep_end;
236 boost::tie(itDep, itDep_end) = out_edges(depv, *depGraph);
237 for (; itDep != itDep_end; itDep++)
238 {
239 Relation* relation = extractEdge(*itDep, posGraph, *depGraph, fullTokens, syntacticData, language);
240 if(relation != 0)
241 {
242 forme->m_outRelations.push_back(relation);
243 }
244 }
245
246 LDEBUG << "ConstituantAndRelationExtractor:: end visiting vertex " << v;
247
248}
249
250//***********************************************************************
251// extract functions
252//***********************************************************************
253
255 const LinguisticGraph& graph,
256 bool checkFullTokens,
257 std::map< LinguisticAnalysisStructure::Token*, uint64_t >& fullTokens,
258 std::vector< bool >& alreadyDumpedTokens,
259 MediaId language)
260{
261
263
264 Token* token = get(vertex_token, graph, v);
266 if(token == 0)
267 return 0;
268 if(checkFullTokens)
269 {
270 LDEBUG << "ConstituantAndRelationExtractor:: check token " << v << "("
271 << token->stringForm() << ")";
272 auto tokenIter = fullTokens.find(token);
273 if(tokenIter != fullTokens.end())
274 {
275 uint64_t tokenId = tokenIter->second;
276 if(alreadyDumpedTokens[tokenId])
277 return 0;
278 alreadyDumpedTokens[tokenId] = true;
279 }
280 }
281
282 LDEBUG << "ConstituantAndRelationExtractor:: extract vertex " << v << "("
283 << token->stringForm() << ")";
284 Forme* forme = new Forme();
285 forme->id = v;
287
288 MorphoSyntacticData* data = get(vertex_data, graph, v);
289 if(data != 0)
290 {
291
292 const PropertyCodeManager pcm = static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(language)).getPropertyCodeManager();
293 const PropertyAccessor macroA = pcm.getPropertyAccessor("MACRO");
294 const PropertyManager macroPm = pcm.getPropertyManager("MACRO");
295 const PropertyAccessor microA = pcm.getPropertyAccessor("MICRO");
296 const PropertyManager microPm = pcm.getPropertyManager("MICRO");
297
298 std::vector<LinguisticElement>::const_iterator itForms = (*data).begin();
299 if(itForms != (*data).end())
300 {
301
302 LDEBUG << "ConstituantAndRelationExtractor:: found morphosyntactic data ";
303 forme->inflForme = Common::Misc::limastring2utf8stdstring(sp[itForms->inflectedForm]);
304
305 forme->poslong.position = token->position();
306 forme->poslong.longueur = token->length();
307
308 forme->macro = macroPm.getPropertySymbolicValue(macroA.readValue(itForms->properties));
309 forme->micro = microPm.getPropertySymbolicValue(microA.readValue(itForms->properties));
310
311 }
312
313 itForms++;
314 if(itForms != (*data).end())
315 {
316 LWARN << "ConstituantAndRelationExtractor:: ommitting multiple morphosyntactic data ";
317 }
318
319 }
320
321 return forme;
322
323}
324
326 const LinguisticGraph& /*posGraph*/,
327 const DependencyGraph& depGraph,
328 std::map< LinguisticAnalysisStructure::Token*, uint64_t >& /*fullTokens*/,
329 const SyntacticData& syntacticData,
330 MediaId language)
331{
332
333 CEdgeDepRelTypePropertyMap relTypeMap = get(edge_deprel_type, depGraph);
334 std::string relationName = static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(language)).getSyntacticRelationName(relTypeMap[e]);
335
337 LDEBUG << "ConstituantAndRelationExtractor:: extract dependency " << relationName;
338
339 uint64_t srcTokenId = syntacticData.tokenVertexForDepVertex(source(e, depGraph));
340 if(srcTokenId == 0)
341 return 0;
342
343 uint64_t tgtTokenId = syntacticData.tokenVertexForDepVertex(target(e, depGraph));
344 if(tgtTokenId == 0)
345 return 0;
346
347 // Check for duplicate relations
348 for(std::vector<Relation*>::iterator relationsIt = m_outRelations.begin(), relationsIt_end = m_outRelations.end();
349 relationsIt != relationsIt_end;
350 relationsIt++)
351 {
352 Relation* rel = *relationsIt;
353 if(rel->type == relationName &&
354 rel->srcVertex == srcTokenId &&
355 (rel->tgtVertex == tgtTokenId ||
356 (rel->secondaryVertex != 0 && rel->secondaryVertex == tgtTokenId)
357 )
358 )
359 {
360 LDEBUG << "ConstituantAndRelationExtractor:: found duplicate relation";
361 return 0;
362 }
363 }
364
365 Relation* relation = new Relation;
366 relation->srcVertex = srcTokenId;
367 relation->tgtVertex = tgtTokenId;
368 relation->type = relationName;
369
370 // Special case for coordination
371 if(relation->type == "COORD1" || relation->type == "COORD2")
372 {
373 std::vector<Relation*>::iterator it, it_toErase;
374 it_toErase = m_outRelations.end();
375 for(it = m_outRelations.begin(); it != m_outRelations.end(); it++)
376 {
377 if( ( (*it)->type == "COORD1" && relation->type == "COORD2" )
378 || ( (*it)->type == "COORD2" && relation->type == "COORD1") )
379 {
380 if((*it)->srcVertex == relation->tgtVertex)
381 {
382 LDEBUG << "ConstituantAndRelationExtractor:: found coord second edge";
383 (*it)->secondaryVertex = relation->srcVertex;
384 delete relation;
385 return 0;
386 }
387 else if((*it)->tgtVertex == relation->srcVertex)
388 {
389 LDEBUG << "ConstituantAndRelationExtractor:: found coord second edge";
390 relation->secondaryVertex = (*it)->srcVertex;
391 it_toErase = it;
392 }
393 }
394 }
395 if(it_toErase != m_outRelations.end())
396 {
397 Forme* srcForme = m_formesIndex[(*it_toErase)->srcVertex];
398 if(srcForme != 0)
399 {
400 std::vector<Relation*>::iterator formIt, formIt_toErase;
401 formIt_toErase = srcForme->m_outRelations.end();
402 for(formIt = srcForme->m_outRelations.begin(); formIt != srcForme->m_outRelations.end(); formIt++)
403 {
404 if(*formIt == *it_toErase)
405 {
406 formIt_toErase = formIt;
407 }
408 }
409 if(formIt_toErase != srcForme->m_outRelations.end())
410 {
411 srcForme->m_outRelations.erase(formIt_toErase);
412 }
413 }
414 m_outRelations.erase(it_toErase);
415 LDEBUG << "ConstituantAndRelationExtractor:: erased coord second edge";
416 }
417 }
418
419 m_outRelations.push_back(relation);
420 return relation;
421
422}
423
424
425/*
426char* annotId = Common::Misc::limastring2utf8stdstring(attrs.getValue(Common::Misc::limastring2utf8stdstring("k")));
427char* posId = Common::Misc::limastring2utf8stdstring(attrs.getValue(Common::Misc::limastring2utf8stdstring("v")));
428std::cout << "pos annot matching: " << atoi(posId) << " / " << atoi(annotId) << std::endl;
429m_posAnnotMatching.insert(std::make_pair(atoi(posId),atoi(annotId)));
430m_annotPosMatching.insert(std::make_pair(atoi(annotId),atoi(posId)));
431*/
432
437{
438 std::set< uint64_t > formesDone;
439 std::map<uint64_t, uint64_t>::const_iterator formesIt, formesIt_end;
440 formesIt = m_positionsFormsIds.begin();
441 formesIt_end = m_positionsFormsIds.end();
442
444 LDEBUG << "ConstituantAndRelationExtractor:: construction des groupes";
445
446 for (; formesIt != formesIt_end; formesIt++)
447 {
448 Groupe* newGrp = 0;
449 uint64_t formeId = (*formesIt).second;
450 if (formesDone.find(formeId) == formesDone.end())
451 {
452 formesDone.insert(formeId);
453 Forme* forme = m_formesIndex[formeId];
454 if(forme == 0)
455 {
456 LWARN << "ConstituantAndRelationExtractor:: form not found in index: " << formeId;
457 continue;
458 }
459 LDEBUG << "ConstituantAndRelationExtractor:: construction des groupes " << forme->forme << "(" << forme->macro << "/" << forme->micro << ")";
460 if (addToGroupIfIsInsideAGroup(forme))
461 {
462 LDEBUG << "ConstituantAndRelationExtractor:: already inserted ";
463 continue;
464 }
465 // always follow SECOMPOUND as it is a specific entity composition
466 std::set<std::string> relationsToFollow;
467 relationsToFollow.insert("SECOMPOUND");
468 std::set<std::string> relationsToFollow2;
469 relationsToFollow2.insert("SECOMPOUND");
470 // verbe non pointe par une relation SujInv
471 if ( (forme->macro == "V") && !(forme->hasInRelation("SujInv") ) && !(forme->hasOutRelation("SUBADJPOST") ) )
472 {
473 LDEBUG << "ConstituantAndRelationExtractor:: verbe non pointe par une relation SujInv";
474 createGroupe(forme, relationsToFollow, "NV", true);
475 }
476 else if ( forme->micro == "ADV" )
477 {
478 createGroupe(forme, relationsToFollow, "GR", true);
479 }
480 else if (forme->micro == "CLS")
481 {
482 relationsToFollow.insert("SujInv");
483 relationsToFollow.insert("TIl");
484 // relationsToFollow.insert("aux");
485 newGrp = createGroupe(forme, relationsToFollow, "NV");
486 if(newGrp == 0)
487 {
488 newGrp = createGroupe(forme, relationsToFollow, "GN", true);
489 }
490 }
491 else if (forme->macro == "DET" || forme->macro == "ADJ" )
492 {
493 relationsToFollow.insert("CodAnaph"); // "le veut" dans "a qui le veut."
494 LDEBUG << "ConstituantAndRelationExtractor:: construction de groupe déterminant";
495 newGrp = createGroupe(forme, relationsToFollow, "NV");
496 if (newGrp == 0)
497 {
498 relationsToFollow.insert("PREPSUB");
499 relationsToFollow.insert("PrepPron");
500 relationsToFollow.insert("PrepDetInt");
501 relationsToFollow.insert("DetIntSub");
502 relationsToFollow.insert("DETSUB");
503 relationsToFollow.insert("ADJPRENSUB");
504 relationsToFollow.insert("ADVADJ");
505 relationsToFollow.insert("ADVADV");
506 relationsToFollow.insert("PrepPronRelCa");
507 relationsToFollow.insert("SUBSUBJUX;NPP;NPP");
508// relationsToFollow.insert("SUBSUBJUX;U;NC");
509 LDEBUG << "ConstituantAndRelationExtractor:: insert '" << forme->forme << "' in GP group";
510 newGrp = createGroupe(forme, relationsToFollow, "GP");
511
512 if(newGrp == 0)
513 {
514 relationsToFollow2.insert("DETSUB");
515 relationsToFollow2.insert("DetAdj");
516 relationsToFollow2.insert("ADJPRENSUB");
517 relationsToFollow2.insert("ADVADJ");
518 relationsToFollow2.insert("ADVADV");
519 relationsToFollow2.insert("COORD1;CC;ADJ");
520 relationsToFollow2.insert("COORD2;CC;ADJ");
521 LDEBUG << "ConstituantAndRelationExtractor:: insert '" << forme->forme << "' in GN group";
522 createGroupe(forme, relationsToFollow2, "GN");
523 }
524
525 }
526 }
527/*
528 else if ( forme->micro == "ADV" && forme->hasOutRelation("AdvSub") )
529 {
530 relationsToFollow.insert("AdvSub");
531 // relationsToFollow.insert("aux");
532 newGrp = createGroupe(forme, relationsToFollow, "GP");
533 if (newGrp == 0)
534 {
535 LDEBUG << "ConstituantAndRelationExtractor:: insert '" << forme->forme << "' in GR group";
536 newGrp = createGroupe(forme, relationsToFollow, "GR");
537 }
538 }
539*/
540 else if ( forme->micro == "ADV" )
541 {
542 newGrp = createGroupe(forme, relationsToFollow, "GR", true);
543 }
544 else if ( forme->macro == "NP" )
545 {
546 relationsToFollow.insert("SUBSUBJUX;NP;NP");
547 relationsToFollow.insert("SUBADJPOST");
548 newGrp = createGroupe(forme, relationsToFollow, "GN", true);
549 }
550 else if ( forme->micro == "DETWH" )
551 {
552 relationsToFollow.insert("DETSUB");
553 relationsToFollow.insert("DetIntSub");
554 relationsToFollow.insert("ADJPRENSUB");
555 newGrp = createGroupe(forme, relationsToFollow, "GN", true);
556 }
557 else if ( forme->micro == "NC" && forme->hasInRelation("SUBSUBJUX") )
558 {
559 relationsToFollow.insert("SUBSUBJUX");
560 // relationsToFollow.insert("aux");
561 newGrp = createGroupe(forme, relationsToFollow, "GN", true);
562 }
563 else if ( forme->micro == "PROREL" )
564 {
565 relationsToFollow.insert("COMPADV");
566 newGrp = createGroupe(forme, relationsToFollow, "GN", true);
567 }
568 else if ( forme->micro == "PROREL"
569 || forme->micro == "NC" || forme->micro == "CLS" )
570 {
571 newGrp = createGroupe(forme, relationsToFollow, "GN", true);
572 }
573 else if ( forme->micro == "P" && forme->hasOutRelation("PrepPronCliv")
574 && !forme->hasOutRelation("PREPSUB") )
575 {
576 newGrp = createGroupe(forme, relationsToFollow, "NV", true);
577 }
578 else if (forme->macro == "PREP" || forme->macro == "DET" )
579 {
580 relationsToFollow.insert("PREPSUB");
581 relationsToFollow.insert("PrepPron");
582 relationsToFollow.insert("PrepDetInt");
583 relationsToFollow.insert("DetIntSub");
584 relationsToFollow.insert("DETSUB");
585 relationsToFollow.insert("ADJPRENSUB");
586 relationsToFollow.insert("ADVADJ");
587 relationsToFollow.insert("ADVADV");
588 relationsToFollow.insert("PrepPronRelCa");
589 relationsToFollow.insert("PrepPronRel");
590 relationsToFollow.insert("PrepAdv");
591 relationsToFollow.insert("COORD1");
592 relationsToFollow.insert("COORD2");
593 relationsToFollow.insert("SUBSUBJUX;NPP;NPP");
594// relationsToFollow.insert("SUBSUBJUX;U;U");
595 newGrp = createGroupe(forme, relationsToFollow, "GP");
596
597 if(newGrp == 0)
598 {
599 relationsToFollow2.insert("PrepInf");
600 relationsToFollow2.insert("Neg");
601 relationsToFollow2.insert("NePas");
602 relationsToFollow2.insert("ADVADV");
603 relationsToFollow2.insert("AdvVerbe");
604 relationsToFollow2.insert("PronReflVerbe");
605 relationsToFollow2.insert("CodPrev");
606 relationsToFollow2.insert("AuxCplPrev");
607 newGrp = createGroupe(forme, relationsToFollow2, "PV");
608 }
609
610 }
611 else if (forme->micro == "ADV" || forme->micro == "CLS" || forme->micro == "CLO" )
612 {
613 relationsToFollow.insert("Neg");
614 relationsToFollow.insert("NePas");
615 relationsToFollow.insert("PronSujVerbe");
616 relationsToFollow.insert("PronReflVerbe");
617 relationsToFollow.insert("CodPrev");
618 relationsToFollow.insert("CoiPrev");
619 relationsToFollow.insert("AuxCplPrev");
620 newGrp = createGroupe(forme, relationsToFollow, "NV");
621 }
622 else if ( forme->micro == "NC" )
623 {
624 relationsToFollow.insert("SUBSUBJUX");
625 newGrp = createGroupe(forme, relationsToFollow, "GN", true);
626 }
627 else if (forme->micro == "VPP" )
628 {
629 newGrp = createGroupe(forme, relationsToFollow, "NV", true);
630 }
631 else if ( forme->macro == "ADJ" )
632 {
633 newGrp = createGroupe(forme, relationsToFollow, "GA", true);
634 }
635 else if (forme->micro == "CLR")
636 {
637 relationsToFollow.insert("PronReflVerbe");
638 relationsToFollow.insert("AuxCplPrev");
639 // relationsToFollow.insert("aux");
640 newGrp = createGroupe(forme, relationsToFollow, "NV");
641 }
642 else if ( forme->micro == "PROREL" )
643 {
644 newGrp = createGroupe(forme, relationsToFollow, "GP", true);
645 }
646 else if ( forme->micro == "CS" && forme->hasInRelation("PrepPronCliv") )
647 {
648 newGrp = createGroupe(forme, relationsToFollow, "GP", true);
649 }
650 else if ( forme->micro == "PRO" || forme->micro == "PROWH" )
651 {
652 newGrp = createGroupe(forme, relationsToFollow, "GN", true);
653 }
654/*
655 else if ( forme->macro == "V" )
656 {
657 newGrp = createGroupe(forme, relationsToFollow, "NV", true);
658 }
659*/
660 if(newGrp != 0)
661 {
662 LDEBUG << "ConstituantAndRelationExtractor:: inserted " << forme->forme << " in " << newGrp->type() << " group";
663 }
664 }
665 }
666}
667
672{
674 LDEBUG << "ConstituantAndRelationExtractor:: constructionDesRelationsEntrantes";
675 std::map<uint64_t,Forme*>::iterator formesIt, formesIt_end;
676 formesIt = m_formesIndex.begin();
677 formesIt_end = m_formesIndex.end();
678 for (; formesIt != formesIt_end; formesIt++)
679 {
680 Forme* forme = formesIt->second;
681 if(forme != 0)
682 {
683 std::vector<Relation*>::iterator relsIt, relsIt_end;
684 relsIt = forme->m_outRelations.begin();
685 relsIt_end = forme->m_outRelations.end();
686 for (; relsIt != relsIt_end; relsIt++)
687 {
688 Relation* rel = (*relsIt);
689 uint64_t tgtFormeId = m_vertexToFormeIds[rel->tgtVertex];
690 if(tgtFormeId != 0)
691 {
692 Forme* tgtForme = m_formesIndex[tgtFormeId];
693 if(tgtForme != 0)
694 {
695 tgtForme->m_inRelations.push_back(rel);
696 }
697 }
698 }
699 }
700 }
701}
702
703Groupe* ConstituantAndRelationExtractor::createGroupe(
704 const Forme* forme,
705 const std::set<std::string>& theRelationsToFollow,
706 const std::string& groupType)
707{
708 return createGroupe(forme, theRelationsToFollow, groupType, false);
709}
710
711Groupe* ConstituantAndRelationExtractor::createGroupe(
712 const Forme* forme,
713 const std::set<std::string>& theRelationsToFollow,
714 const std::string& groupType,
715 bool mayBeUnique = false)
716{
718 LDEBUG << "ConstituantAndRelationExtractor:: createGroupe " << forme->forme << ", " << groupType << " : ";
719 std::set<std::string> relationsToFollow;
720 std::map< std::string, std::set< std::pair< std::string, std::string > > > followConds;
721 std::set<std::string>::const_iterator fit, fit_end;
722 fit = theRelationsToFollow.begin(); fit_end = theRelationsToFollow.end();
723 for (; fit != fit_end; fit++)
724 {
725 if ( (*fit).find(';') != std::string::npos )
726 {
727 size_t first = (*fit).find(";");
728 size_t second = (*fit).find(";", first+1);
729 std::string rel = (*fit).substr(0, first);
730 std::string srcCond = (*fit).substr(first+1, second-first-1);
731 std::string targCond = (*fit).substr(second+1);
732 if ((srcCond != "*") || (targCond != "*"))
733 {
734 if (followConds.find(rel) == followConds.end())
735 {
736 std::set< std::pair< std::string, std::string > > conds;
737 conds.insert(std::make_pair(srcCond, targCond));
738 followConds.insert(std::make_pair(rel, conds));
739 }
740 else
741 followConds[rel].insert(std::make_pair(srcCond, targCond));
742 }
743 relationsToFollow.insert(rel);
744 }
745 else
746 {
747 relationsToFollow.insert(*fit);
748 }
749 }
750 LDEBUG << "ConstituantAndRelationExtractor:: collected relations to insert";
751 Groupe* newGrp = new Groupe();
752 newGrp->type(groupType);
753 std::vector< uint64_t > formsToLookup;
754 std::set< uint64_t > formsAlreadyLookuped;
755 formsToLookup.push_back(forme->id);
756 while (formsToLookup.size() > 0)
757 {
758 Forme* currentForm = m_formesIndex[formsToLookup.back()];
759 LDEBUG << "ConstituantAndRelationExtractor:: current form is " << currentForm->forme;
760 formsToLookup.pop_back();
761 if (m_inGroupFormsPositions.find(currentForm->poslong.position) != m_inGroupFormsPositions.end())
762 continue;
763 formsAlreadyLookuped.insert(currentForm->id);
764
765 std::vector<Relation*>::iterator inIt, inIt_end;
766 inIt = currentForm->m_inRelations.begin();
767 inIt_end = currentForm->m_inRelations.end();
768 for (; inIt != inIt_end; inIt++)
769 {
770 Relation* rel = *inIt;
771 LDEBUG << "ConstituantAndRelationExtractor:: looking at in rel " << rel->type;
772 if(relationsToFollow.find(rel->type) != relationsToFollow.end() &&
773 formsAlreadyLookuped.find(m_vertexToFormeIds[rel->srcVertex]) == formsAlreadyLookuped.end() &&
774 m_inGroupFormsPositions.find(m_formesIndex[m_vertexToFormeIds[rel->srcVertex]]->poslong.position) == m_inGroupFormsPositions.end()
775 )
776 {
777 const Forme* srcForme = m_formesIndex[m_vertexToFormeIds[rel->srcVertex]];
778 const Forme* tgtForme = m_formesIndex[m_vertexToFormeIds[rel->tgtVertex]];
779 std::pair< std::string, std::string > pair1 = std::make_pair(srcForme->macro, tgtForme->macro);
780 std::pair< std::string, std::string > pair2 = std::make_pair(srcForme->macro, tgtForme->micro);
781 std::pair< std::string, std::string > pair3 = std::make_pair(srcForme->macro, "*");
782 std::pair< std::string, std::string > pair4 = std::make_pair(srcForme->micro, tgtForme->macro);
783 std::pair< std::string, std::string > pair5 = std::make_pair(srcForme->micro, tgtForme->micro);
784 std::pair< std::string, std::string > pair6 = std::make_pair(srcForme->micro, "*");
785 std::pair< std::string, std::string > pair7 = std::make_pair("*", tgtForme->macro);
786 std::pair< std::string, std::string > pair8 = std::make_pair("*", tgtForme->micro);
787 if ((followConds.find(rel->type) == followConds.end())
788 || (followConds[rel->type].find(pair1) != followConds[rel->type].end())
789 || (followConds[rel->type].find(pair2) != followConds[rel->type].end())
790 || (followConds[rel->type].find(pair3) != followConds[rel->type].end())
791 || (followConds[rel->type].find(pair4) != followConds[rel->type].end())
792 || (followConds[rel->type].find(pair5) != followConds[rel->type].end())
793 || (followConds[rel->type].find(pair6) != followConds[rel->type].end())
794 || (followConds[rel->type].find(pair7) != followConds[rel->type].end())
795 || (followConds[rel->type].find(pair8) != followConds[rel->type].end()) )
796 {
797 formsToLookup.push_back( m_vertexToFormeIds[rel->srcVertex]);
798 formsAlreadyLookuped.insert(m_vertexToFormeIds[rel->srcVertex]);
799 LDEBUG << "ConstituantAndRelationExtractor:: ins src form '" << srcForme->forme << "' in " << newGrp->type();
800 newGrp->insert( std::make_pair(srcForme->poslong.position, srcForme->id) );
801 }
802 }
803 }
804 std::vector<Relation*>::iterator outIt, outIt_end;
805 outIt = currentForm->m_outRelations.begin();
806 outIt_end = currentForm->m_outRelations.end();
807 for (; outIt != outIt_end; outIt++)
808 {
809 Relation& rel = **outIt;
810 LDEBUG << "ConstituantAndRelationExtractor:: looking at out rel " << rel.type;
811 if (
812 rel.doFollow &&
813 (m_vertexToFormeIds[rel.tgtVertex] != 0) &&
814 (m_formesIndex[m_vertexToFormeIds[rel.tgtVertex]] != 0) &&
815 (relationsToFollow.find(rel.type) != relationsToFollow.end()) &&
816 (formsAlreadyLookuped.find(m_vertexToFormeIds[rel.tgtVertex]) == formsAlreadyLookuped.end() ) &&
817 (m_inGroupFormsPositions.find(m_formesIndex[m_vertexToFormeIds[rel.tgtVertex]]->poslong.position) == m_inGroupFormsPositions.end() ) )
818 {
819 LDEBUG << "ConstituantAndRelationExtractor:: first condition fullfilled";
820 const Forme* srcForme = m_formesIndex[m_vertexToFormeIds[rel.srcVertex]];
821 const Forme* tgtForme = m_formesIndex[m_vertexToFormeIds[rel.tgtVertex]];
822 std::pair< std::string, std::string > pair1 = std::make_pair(srcForme->macro, tgtForme->macro);
823 std::pair< std::string, std::string > pair2 = std::make_pair(srcForme->macro, tgtForme->micro);
824 std::pair< std::string, std::string > pair3 = std::make_pair(srcForme->macro, "*");
825 std::pair< std::string, std::string > pair4 = std::make_pair(srcForme->micro, tgtForme->macro);
826 std::pair< std::string, std::string > pair5 = std::make_pair(srcForme->micro, tgtForme->micro);
827 std::pair< std::string, std::string > pair6 = std::make_pair(srcForme->micro, "*");
828 std::pair< std::string, std::string > pair7 = std::make_pair("*", tgtForme->macro);
829 std::pair< std::string, std::string > pair8 = std::make_pair("*", tgtForme->micro);
830 if ((followConds.find(rel.type) == followConds.end())
831 || (followConds[rel.type].find(pair1) != followConds[rel.type].end())
832 || (followConds[rel.type].find(pair2) != followConds[rel.type].end())
833 || (followConds[rel.type].find(pair3) != followConds[rel.type].end())
834 || (followConds[rel.type].find(pair4) != followConds[rel.type].end())
835 || (followConds[rel.type].find(pair5) != followConds[rel.type].end())
836 || (followConds[rel.type].find(pair6) != followConds[rel.type].end())
837 || (followConds[rel.type].find(pair7) != followConds[rel.type].end())
838 || (followConds[rel.type].find(pair8) != followConds[rel.type].end()) )
839 {
840 LDEBUG << "ConstituantAndRelationExtractor:: second condition fullfilled";
841 formsToLookup.push_back( m_vertexToFormeIds[rel.tgtVertex]);
842 formsAlreadyLookuped.insert(m_vertexToFormeIds[rel.tgtVertex]);
843 LDEBUG << "ConstituantAndRelationExtractor:: insert tgt form '" << tgtForme->forme << "' in " << newGrp->type() << " group";
844 newGrp->insert( std::make_pair(tgtForme->poslong.position, tgtForme->id) );
845 }
846 }
847 }
848 }
849 if (mayBeUnique || !newGrp->empty())
850 {
851 LDEBUG << "ConstituantAndRelationExtractor:: insert forme '" << forme->forme << "' in " << newGrp->type() << " group";
852 newGrp->insert( std::make_pair(forme->poslong.position, forme->id) );
853 insertGroup(*newGrp);
854 return newGrp;
855 }
856 return 0;
857}
858
859void ConstituantAndRelationExtractor::insertGroup(const Groupe& groupe)
860{
861 uint64_t grpPos = m_formesIndex[(*(groupe.begin())).second]->poslong.position;
863 LDEBUG << "ConstituantAndRelationExtractor:: inserting group " << grpPos << " : ";
864 m_groupes.insert(std::make_pair(grpPos,groupe));
865 Groupe::const_iterator grpsIt, grpsIt_end;
866 grpsIt = groupe.begin(); grpsIt_end = groupe.end();
867 for (; grpsIt != grpsIt_end; grpsIt++)
868 {
869 m_inGroupFormsPositions.insert((*grpsIt).first);
870 }
871}
872
873bool ConstituantAndRelationExtractor::addToGroupIfIsInsideAGroup(const Forme* forme)
874{
875 uint64_t position = forme->poslong.position;
876
877 std::map<uint64_t, Groupe>::const_iterator itg, itg_end;
878 itg = m_groupes.begin(); itg_end = m_groupes.end();
879 uint64_t pos = 0;
880 for (; itg != itg_end ; itg++)
881 {
882 if ( (*itg).first > pos && position >= (*itg).first )
883 {
884 pos = (*itg).first;
885
886 }
887 else if ( (*itg).first > pos && (*itg).first > position )
888 break;
889 }
890
891 if (m_groupes.find(pos) != m_groupes.end() && !m_groupes[pos].empty())
892 {
893 if (m_groupes[pos].rbegin() == m_groupes[pos].rend())
894 {
895 return false;
896 }
897 else if ( ( (*(m_groupes[pos].rbegin())).first >= position ) &&
898 ( (*(m_groupes[pos].begin())).first <= position ) )
899 {
901 LDEBUG << "ConstituantAndRelationExtractor:: insert '" << forme->forme << "' in " << m_groupes[pos].type() << " group";
902 m_groupes[pos].insert( std::make_pair(position, forme->id) );
903 m_inGroupFormsPositions.insert(position);
904 return true;
905 }
906 else
907 {
908 return false;
909 }
910 }
911 return false;
912}
913
915{
917 LDEBUG << "ConstituantAndRelationExtractor:: add last forms in groups";
918 std::map<uint64_t, uint64_t>formsInGroups;
919 std::map<uint64_t, Groupe>::iterator itGr, itGr_end;
920 itGr = m_groupes.begin();
921 itGr_end = m_groupes.end();
922 for (;itGr!=itGr_end;itGr++)
923 {
924 std::map<uint64_t, uint64_t>::iterator itForms, itForms_end;
925 itForms = ((*itGr).second).begin();
926 itForms_end = ((*itGr).second).end();
927
928 for (;itForms!=itForms_end;itForms++)
929 {
930 formsInGroups.insert(std::make_pair((*itForms).first, (*itForms).second));
931 }
932 }
933 std::map<uint64_t, uint64_t>::iterator it, it_end;
934 it = m_positionsFormsIds.begin();
935 it_end = m_positionsFormsIds.end();
936 for (;it!=it_end;it++)
937 {
938 if (formsInGroups.find((*it).first) == formsInGroups.end() )
939 {
940 if(m_formesIndex[(*it).second] != 0)
941 {
942 addToGroupIfIsInsideAGroup(m_formesIndex[(*it).second]);
943 }
944 }
945 }
946}
947
949{
951 LDEBUG << "ConstituantAndRelationExtractor:: splitCompoundTenses";
952 std::map<uint64_t, uint64_t>::iterator it, it_end;
953 it = m_positionsFormsIds.begin(); it_end = m_positionsFormsIds.end();
954 uint64_t compoundSplitted = 0;
955 for (; it != it_end; it++)
956 {
957 LDEBUG << "ConstituantAndRelationExtractor:: pos/id/annot="<<(*it).first<<"/"<<(*it).second<<"/"<<m_posAnnotMatching[(*it).first];
958 // le noeud a la position courante definit un temps compose
959 if (m_compoundTenses.find(m_posAnnotMatching[(*it).first]) != m_compoundTenses.end() )
960 {
961 uint64_t position = (*it).first;
962
963 // ids are in pos graph space
964 uint64_t cpdtenseid = m_positionsFormsIds[position];
965 uint64_t auxid = m_compoundTenses[m_posAnnotMatching[(*it).first]].first;
966 uint64_t pastpartid = m_compoundTenses[m_posAnnotMatching[(*it).first]].second;
967
968 LDEBUG << "ConstituantAndRelationExtractor:: cpd tense: "<<cpdtenseid<<"->("<<auxid << "," << pastpartid << ")";
969 Forme* cpdtenseForme = m_formesIndex[cpdtenseid];
970 Forme* auxForme = m_formesIndex[auxid];
971 Forme* pastpartForme = m_formesIndex[pastpartid];
972
973 if(cpdtenseForme != 0 && auxForme != 0 && pastpartForme != 0){
974
975 // remplacer la forme a la position courante par celle de l'auxiliaire
976 LDEBUG << "ConstituantAndRelationExtractor:: replacing at " << position <<" by " << auxForme->forme << " (" << auxid << ")";
977 m_positionsFormsIds[position] = auxid;
978
979 // pour chaque relation entrante sur temps compose
980 // - si sujinv ou suj ou advv, la repointer sur la forme de l'aux
981 // - si cod ou coi (cplv), la repointer sur la forme du past part
982 std::vector<Relation*>::iterator cpdTenseInRelsIt, cpdTenseInRelsIt_end;
983 cpdTenseInRelsIt = cpdtenseForme->m_inRelations.begin();
984 cpdTenseInRelsIt_end = cpdtenseForme->m_inRelations.end();
985 for (;cpdTenseInRelsIt != cpdTenseInRelsIt_end; cpdTenseInRelsIt++)
986 {
987 Relation* cpdTenseInRel = *cpdTenseInRelsIt;
988 LDEBUG << "ConstituantAndRelationExtractor:: compound tense input relation = " << cpdTenseInRel->type;
989 if (cpdTenseInRel->type == "SujInv" || cpdTenseInRel->type == "SUJ_V" || cpdTenseInRel->type == "Neg" || cpdTenseInRel->type == "PronSujVerbe")
990 {
991 LDEBUG << "ConstituantAndRelationExtractor:: change it from (" << cpdTenseInRel->srcVertex << "-> " << cpdTenseInRel->tgtVertex << ") to (" << cpdTenseInRel->srcVertex << "->" << auxForme->forme << ")";
992 cpdTenseInRel->tgtVertex = m_formeIdsToVertex[auxForme->id];
993 auxForme->m_inRelations.push_back(cpdTenseInRel);
994 }
995 else // at least COD_V and CPL_V
996 {
997 LDEBUG << "ConstituantAndRelationExtractor:: change it from (" << cpdTenseInRel->srcVertex << "-> " << cpdTenseInRel->tgtVertex << ") to (" << cpdTenseInRel->srcVertex << "->" << pastpartForme->forme << ")";
998 cpdTenseInRel->tgtVertex = m_formeIdsToVertex[pastpartForme->id];
999 pastpartForme->m_inRelations.push_back(cpdTenseInRel);
1000 // dans ce cas attention � ne pas suivre dans la construction des groupes, mais � suivre AuxCplPrev
1001 if (cpdTenseInRel->type == "CodPrev" || cpdTenseInRel->type == "CoiPrev" || cpdTenseInRel->type == "PronSujVerbe" || cpdTenseInRel->type == "PronReflVerbe" || cpdTenseInRel->type == "PrepInf")
1002 {
1003 Forme* srcForme = m_formesIndex[m_vertexToFormeIds[cpdTenseInRel->srcVertex]];
1004 if(srcForme != 0){
1005 LDEBUG << "ConstituantAndRelationExtractor:: specific compound tense case: " << srcForme->forme;
1006 cpdTenseInRel->doFollow = false;
1007 Relation* cplRel = new Relation();
1008 cplRel->srcVertex = cpdTenseInRel->srcVertex;
1009 cplRel->tgtVertex = m_formeIdsToVertex[auxForme->id];
1010 cplRel->type = "AuxCplPrev";
1011 m_outRelations.push_back(cplRel);
1012 srcForme->m_outRelations.push_back(cplRel);
1013 auxForme->m_inRelations.push_back(cplRel);
1014 }
1015 }
1016 }
1017 }
1018
1019 // pour chaque relation sortante d'un temps compose
1020 // - si sujinv ou suj ou advv, la repointer sur la forme de l'aux
1021 // - si cod, la repointer sur la forme du past part
1022 std::vector<Relation*>::iterator cpdTenseOutRelsIt, cpdTenseOutRelsIt_end;
1023 cpdTenseOutRelsIt = cpdtenseForme->m_outRelations.begin();
1024 cpdTenseOutRelsIt_end = cpdtenseForme->m_outRelations.end();
1025 for (;cpdTenseOutRelsIt != cpdTenseOutRelsIt_end; cpdTenseOutRelsIt++)
1026 {
1027 Relation* cpdTenseOutRel = *cpdTenseOutRelsIt;
1028 LDEBUG << "ConstituantAndRelationExtractor:: compound tense output relation = " << cpdTenseOutRel->type;
1029 LDEBUG << "ConstituantAndRelationExtractor:: change it from (" << cpdTenseOutRel->srcVertex << "-> " << cpdTenseOutRel->tgtVertex << ") to (" << pastpartForme->forme << "->" << cpdTenseOutRel->tgtVertex << ")";
1030 cpdTenseOutRel->srcVertex = m_formeIdsToVertex[pastpartForme->id];
1031 pastpartForme->m_outRelations.push_back(cpdTenseOutRel);
1032 }
1033
1034 compoundSplitted++;
1035
1036 }
1037 else
1038 {
1039 LDEBUG << "ConstituantAndRelationExtractor:: compound tense not found part";
1040 }
1041 }
1042 }
1043 LDEBUG << "ConstituantAndRelationExtractor:: splitCompoundTenses DONE";
1044
1045 if(compoundSplitted > 0){
1046 LDEBUG << "ConstituantAndRelationExtractor:: trying recursive splitCompoundTenses";
1047 //splitCompoundTenses();
1048 }
1049
1050}
1051
1052//Fonction qui permet de remplacer une entité nommée par les éléments qui la composent
1054{
1055 std::vector<uint64_t> formsToErase;
1056
1057 if (m_formesIndex.empty())
1058 {
1059 return;
1060 }
1061 std::vector<Relation*>::iterator relIt, relIt_end;
1062
1063 std::map<uint64_t,Forme*>::iterator It, It_end;
1064 It = m_formesIndex.begin();
1065 It_end = m_formesIndex.end();
1066
1067 for (;It!=It_end;It++)
1068 {
1069 uint64_t position = (*It).first;
1070 Forme* forme = (*It).second;
1071 // Pour toutes les formes de m_formesIndex, on regarde si leur id est présent dans les ids des entités nommées.
1072 if (m_namedEntitiesVertices.find(position) != m_namedEntitiesVertices.end())
1073 {
1075 LDEBUG << "ConstituantAndRelationExtractor:: se at " << position << " for " << forme->forme ;
1076 //on récupère l'id du vertex dans l'analysis graph qui correspond � l'id du posgraph
1077 uint64_t matchingVertex = m_posAnaMatching[((*It).first)];
1078 //on stocke dans un vecteur les composants de l'entité nommée
1079 std::vector<uint64_t> tmpVector = m_seCompounds[matchingVertex];
1080 std::vector<uint64_t>::iterator vectIt, vectIt_end;
1081 vectIt = tmpVector.begin(); vectIt_end = tmpVector.end();
1082
1083 //ici on ajoute l'id de l'entité nommée afin de pouvoir la supprimer de m_formesIndex par la suite
1084 formsToErase.push_back(position);
1085 bool firstPassage = true;
1086 Forme* precForm = 0;
1087 uint64_t maxVertex = (*(m_formesIndex.rbegin())).first;
1088 for (;vectIt!=vectIt_end;vectIt++)
1089 {
1090 //pour chacune des composantes de l'entité nommée, on en extrait la forme sur laquelle on fait quelques modification
1091 Forme* tmpForme = m_anaGraphVertices[*vectIt];
1092 LDEBUG << "ConstituantAndRelationExtractor:: se compound: " << tmpForme->forme;
1093 //afin qu'il n'y ait pas de conflit dans la numérotation des vertex, l'id de la première composante est égale � l'id la plus grande de m_formesIndex que l'on incrémente de 1
1094 tmpForme->id = maxVertex+1;
1095 maxVertex++;
1096 // Il faut alors ajouter cette forme dans le tableau des formes
1097 m_formesIndex[tmpForme->id]=tmpForme;
1098 tmpForme->macro = m_namedEntitiesVertices[(*It).first].macro;
1099 tmpForme->micro = m_namedEntitiesVertices[(*It).first].micro;
1100
1101 // si la composante traitée est la première composante de l'entité nommmée, on copie les relations de dépendances si elles existent de l'entité nommée sur celle ci (tout en modifiant srcVertex pour qu'il corresponde � son id) et on met � jour m_outRelations.
1102 if (firstPassage)
1103 {
1104
1105 //copie des relations de l'entité nommée
1106 tmpForme->m_outRelations = (*It).second->m_outRelations;
1107 relIt = tmpForme->m_outRelations.begin();
1108 relIt_end = tmpForme->m_outRelations.end();
1109
1110 // pour toute les relations copiées, modifie la source de la dépendance vers l'id de la composante
1111 for (;relIt != relIt_end; relIt++)
1112 {
1113 (*relIt)->srcVertex = tmpForme->id;
1114 }
1115
1116 // on met � jour des relations dans m_out_relations, qui ont pour sources ou cible l'entité nommée traitée
1117 relIt = m_outRelations.begin();
1118 relIt_end = m_outRelations.end();
1119 for (;relIt != relIt_end;relIt++)
1120 {
1121 if ((*relIt)->srcVertex == (*It).first)
1122 (*relIt)->srcVertex = tmpForme->id;
1123 else if ((*relIt)->tgtVertex == (*It).first)
1124 (*relIt)->tgtVertex = tmpForme->id;
1125 }
1126
1127 // ici, on met � jour m_formesIndex pour que toutes les relations qui pointaient vers l'entité nommée pointent désormais vers la composante.
1128 relIt = m_outRelations.begin();
1129 relIt_end = m_outRelations.end();
1130 for(; relIt!=relIt_end;relIt++)
1131 {
1132 if ((*relIt)->tgtVertex == position)
1133 {
1134 LDEBUG << "ConstituantAndRelationExtractor:: update relation target " << (*relIt)->tgtVertex;
1135 (*relIt)->tgtVertex = tmpForme->id;
1136 }
1137 }
1138 firstPassage = false;
1139 }
1140 else if(precForm != 0)
1141 {
1142 Relation* rel = new Relation;
1143 rel->srcVertex = precForm->id;
1144 rel->tgtVertex = tmpForme->id;
1145 rel->type = "SECOMPOUND";
1146 precForm->m_outRelations.push_back(rel);
1147 m_outRelations.push_back(rel);
1148 }
1149
1150 //on insère la forme créée dans les différents conteneurs
1151 LDEBUG << "ConstituantAndRelationExtractor:: adding compound " << tmpForme->forme << ", " << tmpForme->id << "(" << tmpForme->poslong.position << ")";
1152 std::map< LinguisticAnalysisStructure::Token*, uint64_t >::const_iterator tokenIter;
1153 m_positionsFormsIds.erase(tmpForme->poslong.position);
1154 m_positionsFormsIds.insert(std::make_pair(tmpForme->poslong.position, tmpForme->id));
1155 m_vertexToFormeIds.insert(std::make_pair(tmpForme->id, tmpForme->id));
1156 m_formeIdsToVertex.insert(std::make_pair(tmpForme->id, tmpForme->id));
1157
1158 precForm = tmpForme;
1159
1160 }
1161
1162 }
1163 }
1164
1165 //pour finir, on supprime les entités nommées traitées de m_formesIndex
1166 std::vector<uint64_t>::const_iterator eraseIt, eraseIt_end;
1167 eraseIt = formsToErase.begin();
1168 eraseIt_end = formsToErase.end();
1169 for (;eraseIt!=eraseIt_end;eraseIt++)
1170 {
1172 LDEBUG << "ConstituantAndRelationExtractor:: erase compound " << *eraseIt;
1173 m_formesIndex.erase(*eraseIt);
1174 }
1175}
1176
1177} // end namespace EasyXmlDumper
1178} // end namespace AnalysisDumpers
1179} // end namespace LinguisticProcessings
1180} // end namespace Lima
boost::color_traits< boost::default_color_type > Color
extracts forms and relations from boost graph (origninally, from XML file)
DependencyGraph::out_edge_iterator DependencyGraphOutEdgeIt
DependencyGraph::vertex_descriptor DependencyGraphVertex
boost::property_map< DependencyGraph, edge_deprel_type_t >::const_type CEdgeDepRelTypePropertyMap
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, DepVertexProperties, DepEdgeProperties > DependencyGraph
The dependency graph class.
@ edge_deprel_type
#define LWARN
Definition LimaCommon.h:160
#define LDEBUG
Definition LimaCommon.h:157
boost::graph_traits< LinguisticGraph >::edge_descriptor LinguisticGraphEdge
typedefs to simplify the access to various graphs elements
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define DUMPERLOGINIT
Data used for the syntactic analyzis of texts.
Holds an annotation graph and gives an API to manipulate it.
std::set< AnnotationGraphVertex > matches(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second) const
Gets the set of vertices matched in the second graph by the given vertex of the first graph.
Holds linguistic data for one language.
const FsaStringsPool & stringsPool(MediaId med) const
const MediaData & mediaData(MediaId media) const
Provide function to read write and check a property.
LinguisticCode readValue(const LinguisticCode &code) const
read a property in a coded int.
Provide tools to parse a property file, and deal with the property coding system.
const PropertyManager & getPropertyManager(const std::string &propertyName) const
Get the PropertyManager associated to a property.
const PropertyAccessor & getPropertyAccessor(const std::string &propertyName) const
Get the PropertyAccessor associated to a property.
Provide tools to manage a specific property.
const std::string & getPropertySymbolicValue(const LinguisticCode &value) const
The coded property value can hold several property data.
void visitBoostGraph(const LinguisticGraphVertex &v, const LinguisticGraphVertex &end, const LinguisticGraph &anaGraph, const LinguisticGraph &posGraph, const Common::AnnotationGraphs::AnnotationData &annotationData, const SyntacticAnalysis::SyntacticData &syntacticData, std::map< LinguisticAnalysisStructure::Token *, uint64_t > &fullTokens, std::vector< bool > &alreadyDumpedTokens, const MediaId &language)
ConstituantAndRelationExtractor(const Common::PropertyCode::PropertyCodeManager *propertyCodeManager)
Relation * extractEdge(const LinguisticGraphEdge &e, const LinguisticGraph &posGraph, const DependencyGraph &depGraph, std::map< LinguisticAnalysisStructure::Token *, uint64_t > &fullTokens, const SyntacticAnalysis::SyntacticData &syntacticData, MediaId language)
Forme * extractVertex(const LinguisticGraphVertex &v, const LinguisticGraph &graph, bool checkFullTokens, std::map< LinguisticAnalysisStructure::Token *, uint64_t > &fullTokens, std::vector< bool > &alreadyDumpedTokens, MediaId language)
This class points to a graph, its dependency graph and the structure that holds the maping between th...
LinguisticGraphVertex tokenVertexForDepVertex(const DependencyGraphVertex &v) const
DependencyGraphVertex depVertexForTokenVertex(const LinguisticGraphVertex &v) const
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
static MediaticData & changeable()
singleton accessor
Definition Singleton.h:71
dump the content of the analysis graph in Easy XML format
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
AnnotationGraph::out_edge_iterator AnnotationGraphOutEdgeIt
bool hasIntAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
bool hasAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
NAUTITIA.
QString LimaString
Definition LimaString.h:33
STL namespace.
uint64_t secondaryVertex
to store the third element of a 3-ary relation, currently only the source of the COORD2 relation.
Definition relation.h:52