LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
ConstituantAndRelationExtractor.h
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
21#ifndef LIMA_LINGUISTICPROCESSINGS_ANALYSISDUMPERS_EASYXMLDUMPER_CONSTITUANTANDRELATIONEXTRACTOR_H
22#define LIMA_LINGUISTICPROCESSINGS_ANALYSISDUMPERS_EASYXMLDUMPER_CONSTITUANTANDRELATIONEXTRACTOR_H
23
24#include "chaine.h"
25#include "poslong.h"
26#include "relation.h"
27#include "forme.h"
28#include "groupe.h"
29
35
42
43#include <limits>
44#include <string>
45#include <set>
46#include <vector>
47#include <map>
48#include <cstdint>
49
50#include <boost/tuple/tuple.hpp>
51#include <boost/graph/properties.hpp>
52
53namespace Lima {
54namespace LinguisticProcessing {
55namespace AnalysisDumpers {
56namespace EasyXmlDumper {
57
59{
60
61public:
62
65
67 const LinguisticGraphVertex& end,
68 const LinguisticGraph& anaGraph,
69 const LinguisticGraph& posGraph,
70 const Common::AnnotationGraphs::AnnotationData& annotationData,
71 const SyntacticAnalysis::SyntacticData& syntacticData,
72 std::map< LinguisticAnalysisStructure::Token*, uint64_t >& fullTokens,
73 std::vector< bool >& alreadyDumpedTokens,
74 const MediaId& language);
75
76 const std::map<Chaine,std::vector<uint64_t> >& getChaines() const { return m_chaines;}
77 const std::map<uint64_t,uint64_t>& getVertexToFormeIds() const { return m_vertexToFormeIds;}
78 const std::map<uint64_t,Forme*>& getFormesIndex() const { return m_formesIndex;}
79 const std::vector<Relation*>& getRelations() const { return m_outRelations;}
80 const std::vector<Relation*>& getInRelations() const { return m_inRelations;}
81 const std::map<uint64_t, Groupe>& getGroupes() const { return m_groupes;}
82 const std::map< uint64_t, uint64_t >& positionsFormsIds() const { return m_positionsFormsIds; };
83 const std::set< uint64_t >& inGroupFormsPositions() const { return m_inGroupFormsPositions; };
84
90
91protected:
92
94 const LinguisticGraph& graph,
95 bool checkFullTokens,
96 std::map< LinguisticAnalysisStructure::Token*, uint64_t >& fullTokens,
97 std::vector< bool >& alreadyDumpedTokens,
98 MediaId language);
99
101 const LinguisticGraph& posGraph,
102 const DependencyGraph& depGraph,
103 std::map< LinguisticAnalysisStructure::Token*, uint64_t >& fullTokens,
104 const SyntacticAnalysis::SyntacticData& syntacticData,
105 MediaId language);
106
111
112private:
113
114// std::string getName(const QString& localName,
115// const QString& qName);
116
117 Groupe* createGroupe(const Forme* forme,
118 const std::set<std::string>& relationsToFollow,
119 const std::string& groupType);
120 Groupe* createGroupe(const Forme* forme,
121 const std::set<std::string>& relationsToFollow,
122 const std::string& groupType,
123 bool mayBeUnique);
124
125 void insertGroup(const Groupe& groupe);
126 bool addToGroupIfIsInsideAGroup(const Forme* forme);
127
128 template <typename AnnotationType>
129 void splitCompoundAnalysisAnnotation(
131 Forme& forme,
132 LimaString& annotationTypeStr,
133 const Common::AnnotationGraphs::AnnotationData& annotationData,
134 const LinguisticGraph& anaGraph,
135 std::map< LinguisticAnalysisStructure::Token*, uint64_t >& fullTokens,
136 std::vector< bool >& alreadyDumpedTokens,
137 const MediaId& language)
138 {
139 Common::AnnotationGraphs::GenericAnnotation genAnnot = annotationData.annotation(v, annotationTypeStr);
140 AnnotationType genAnnotVect = genAnnot.value<AnnotationType>();
141 m_namedEntitiesVertices.insert(std::make_pair(forme.id, forme));
142 m_posAnaMatching.insert(std::make_pair(forme.id, v));
143
144 std::vector<uint64_t> genAnaVertices;
145 for(auto genAnnotVectIt = genAnnotVect.vertices().begin(),
146 genAnnotVectIt_end = genAnnotVect.vertices().end();
147 genAnnotVectIt != genAnnotVectIt_end; genAnnotVectIt++)
148 {
149 Forme* anaForme = extractVertex(*genAnnotVectIt, anaGraph, false, fullTokens, alreadyDumpedTokens, language);
150 if(anaForme != 0)
151 {
152 // if corresponding AnalysisGraph vertex, link it
154 LDEBUG << "ConstituantAndRelationExtractor:: got analysis " << *genAnnotVectIt << ", " << anaForme->forme;
155 m_anaGraphVertices[*genAnnotVectIt] = anaForme;
156 genAnaVertices.push_back(*genAnnotVectIt);
157 }
158 }
159 m_seCompounds[v] = genAnaVertices;
160 }
161
162 std::map<Chaine,std::vector<uint64_t> > m_chaines;
163 std::map<uint64_t,uint64_t> m_vertexToFormeIds;
164 std::map<uint64_t,uint64_t> m_formeIdsToVertex;
165
166 std::map<uint64_t,Forme*> m_formesIndex; // id in pos graph -> forme
167 std::map<uint64_t,Forme*> m_anaGraphVertices; // id in analysis graph -> forme
168
169 std::vector<Relation*> m_outRelations;
170 std::vector<Relation*> m_inRelations;
171
172 std::map<uint64_t,std::vector<uint64_t> > m_seCompounds; // List of specific entitites
173 std::map<uint64_t, Forme> m_namedEntitiesVertices; // List of named entitites
174 std::map< uint64_t, std::pair< uint64_t, uint64_t > > m_compoundTenses; // compound -> (aux id, participle id)
175
176 std::map<uint64_t,uint64_t> m_posAnaMatching; // id in pos graph -> id in ana graph
177 std::map<uint64_t,uint64_t> m_posAnnotMatching; // id in pos graph -> id in annot graph
178 std::map<uint64_t,uint64_t> m_annotPosMatching; // id in annot graph -> id in pos graph
179
180 std::map<uint64_t, Groupe> m_groupes; // position -> groupe
181 std::map< uint64_t, uint64_t > m_positionsFormsIds; // position -> id
182 std::set< uint64_t > m_inGroupFormsPositions; // positions
183
184};
185
186} // end namespace EasyXmlDumper
187} // end namespace AnalysisDumpers
188} // end namespace LinguisticProcessings
189} // end namespace Lima
190
191#endif // LIMA_LINGUISTICPROCESSINGS_ANALYSISDUMPERS_EASYXMLDUMPER_CONSTITUANTANDRELATIONEXTRACTOR_H
This file is the main header file for the data related to annotation graphs.
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, DepVertexProperties, DepEdgeProperties > DependencyGraph
The dependency graph class.
#define LDEBUG
Definition LimaCommon.h:157
A graph structure for linguistic analysis.
boost::graph_traits< LinguisticGraph >::edge_descriptor LinguisticGraphEdge
typedefs to simplify the access to various graphs elements
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define DUMPERLOGINIT
Data used for the syntactic analyzis of texts.
represents a chain
Holds an annotation graph and gives an API to manipulate it.
const GenericAnnotation & annotation(AnnotationGraphVertex v1, AnnotationGraphVertex v2, const LimaString &annot) const
This class allows to convert any object into an annotation by inheritance.
Provide function to read write and check a property.
Provide tools to parse a property file, and deal with the property coding system.
Provide tools to manage a specific property.
void visitBoostGraph(const LinguisticGraphVertex &v, const LinguisticGraphVertex &end, const LinguisticGraph &anaGraph, const LinguisticGraph &posGraph, const Common::AnnotationGraphs::AnnotationData &annotationData, const SyntacticAnalysis::SyntacticData &syntacticData, std::map< LinguisticAnalysisStructure::Token *, uint64_t > &fullTokens, std::vector< bool > &alreadyDumpedTokens, const MediaId &language)
Relation * extractEdge(const LinguisticGraphEdge &e, const LinguisticGraph &posGraph, const DependencyGraph &depGraph, std::map< LinguisticAnalysisStructure::Token *, uint64_t > &fullTokens, const SyntacticAnalysis::SyntacticData &syntacticData, MediaId language)
Forme * extractVertex(const LinguisticGraphVertex &v, const LinguisticGraph &graph, bool checkFullTokens, std::map< LinguisticAnalysisStructure::Token *, uint64_t > &fullTokens, std::vector< bool > &alreadyDumpedTokens, MediaId language)
This class points to a graph, its dependency graph and the structure that holds the maping between th...
represents an Easy form
represents an Easy group
NAUTITIA.
QString LimaString
Definition LimaString.h:33
represents an Easy position for a form
represents an Easy relation (dep))