LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
LTRTextBuilder.cpp
Go to the documentation of this file.
1// Copyright 2002-2020 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
21#include "LTRTextBuilder.h"
23
29
30
31using namespace Lima::Common::Misc;
32using namespace Lima::Common::BagOfWords;
33using namespace Lima::Common::MediaticData;
35using namespace std;
36//using namespace boost;
37
38#ifdef WIN32
39#undef min
40#undef max
41#endif
42
43namespace Lima {
44namespace LinguisticProcessing {
45namespace AnalysisDumpers {
46
47
49 const MediaId& language,
50 std::shared_ptr<StopList> stopList) : m_language(language), m_stopList(stopList) {
51
52 m_macroAccessor =
53 &static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(language)).getPropertyCodeManager().getPropertyAccessor("MACRO");
54 m_microAccessor =
55 &static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(language)).getPropertyCodeManager().getPropertyAccessor("MICRO");
56}
57
58
65 const LinguisticGraph& graph,
67 const LinguisticGraphVertex& graphFirstVertex,
68 const LinguisticGraphVertex& graphLastVertex,
69 boost::shared_ptr<LTR_Text> textRep,
70 uint64_t offset) {
71
72 if (sb==0) {
73 // no segmentation data: add tokens from all text
74 uint64_t tokenCounter = 0;
75 this->addTokensToLTRTextFrom(
76 graph,
77 graphFirstVertex, // from first vertex
78 graphLastVertex, // to last vertex
79 graphLastVertex,
80 textRep,
81 offset,
82 &tokenCounter);
83 // add a global sentence boundary (thay covers all the text)
85 LDEBUG << "LTR: add sentence bound at token" << tokenCounter;
86 textRep->addSentenceBound(tokenCounter);
87 }
88 else {
89 // ??OME2 SegmentationData::iterator sbIt = sb->begin();
90 std::vector<Segment>::iterator sbIt = (sb->getSegments()).begin();
91 uint64_t tokenCounter = 0;
92 // ??OME2 while (sbIt != sb->end()) {
93 while (sbIt != (sb->getSegments()).end()) {
94 LinguisticGraphVertex sentenceBegin = sbIt->getFirstVertex();
95 LinguisticGraphVertex sentenceEnd = sbIt->getLastVertex();
96 this->addTokensToLTRTextFrom(
97 graph,
98 sentenceBegin, // from sentence beginning
99 sentenceEnd, // to sentence end
100 graphLastVertex,
101 textRep,
102 offset,
103 &tokenCounter);
104 textRep->addSentenceBound(tokenCounter);
105 sbIt ++;
106 }
107 }
108}
109
110void LTRTextBuilder::addTokensToLTRTextFrom(
111 const LinguisticGraph& graph,
112 const LinguisticGraphVertex& firstVertex,
113 const LinguisticGraphVertex& lastVertex,
114 const LinguisticGraphVertex& graphLastVertex,
115 boost::shared_ptr<LTR_Text> textRep,
116 uint64_t offset,
117 uint64_t* tokenCounter) {
118
120 m_verticesToExplore.clear();
121 m_exploredVertices.clear();
122 this->exploreVerticesFrom(firstVertex, graphLastVertex, graph);
123 if (m_verticesToExplore.size() != 0) {
124 // find the beginning position of the first word
125 LinguisticGraphVertex firstTokenVertex = m_verticesToExplore.front();
126 Token* token = get(vertex_token, graph, firstTokenVertex);
127 m_currentOffset = token->position() + offset;
128 boost::shared_ptr<LTR_Token> currentLtrTok( new LTR_Token() );
129 textRep->addToken(currentLtrTok);
130 (*tokenCounter) ++;
131 // LTR_Token* sentenceBegin = currentLtrTok;
132 bool endVertexFlag = false;
133 while (! endVertexFlag && ! m_verticesToExplore.empty()) {
134 // find the closest vertex to the last visited vertex
135 LinguisticGraphVertex closestVertex;
136 uint64_t smallestPosDiff = std::numeric_limits<uint64_t>::max();
137 for (VERTICES_TO_EXPLORE_T::const_iterator itVert = m_verticesToExplore.begin();
138 itVert != m_verticesToExplore.end(); itVert ++) {
139 token = get(vertex_token, graph, *itVert);
140 uint64_t posDiff = token->position() + offset - m_currentOffset;
141 if (posDiff < smallestPosDiff) {
142 smallestPosDiff = posDiff;
143 closestVertex = *itVert;
144 }
145 }
146 // create a new LTR_Token for a new position
147 if (smallestPosDiff > 0) {
148 currentLtrTok = boost::shared_ptr<LTR_Token>(new LTR_Token());
149 m_currentOffset = token->position() + offset;
150 textRep->addToken(currentLtrTok);
151 (*tokenCounter) ++;
152 }
153 // update the current LTR_Token with the current vertex
154 this->updateLTR_TokenFromVertex(closestVertex, graph,
155 currentLtrTok, offset);
156 // remove the current vertex from the list of vertices to explore
157 // and add it to list of explored vertices
158 m_verticesToExplore.remove(closestVertex);
159 m_exploredVertices.insert(closestVertex);
160 // add the vertices that can be reached from the current vertex to the
161 // list of vertices to explore
162 this->exploreVerticesFrom(closestVertex, graphLastVertex, graph);
163 endVertexFlag = (closestVertex == lastVertex);
164 }
165 if (! m_verticesToExplore.empty()) {
166 LWARN << "all vertices not explored between " << firstVertex << " and "
167 << lastVertex;
168 }
169 }
170}
171
172void LTRTextBuilder::exploreVerticesFrom(
173 const LinguisticGraphVertex& vertex,
174 const LinguisticGraphVertex& graphLastVertex,
175 const LinguisticGraph& graph) {
176
177 LinguisticGraphOutEdgeIt it, itEnd;
178 boost::tie(it, itEnd) = out_edges(vertex, graph);
179 while (it != itEnd) {
180 LinguisticGraphVertex outVertex = target(*it, graph);
181 if ((outVertex != graphLastVertex) &&
182 (m_exploredVertices.find(outVertex) == m_exploredVertices.end())) {
183 m_verticesToExplore.push_back(outVertex);
184 }
185 it ++;
186 }
187}
188
189void LTRTextBuilder::updateLTR_TokenFromVertex(
190 const LinguisticGraphVertex& vertex,
191 const LinguisticGraph& graph,
192 boost::shared_ptr<LTR_Token> tokenRep,
193 uint64_t offset) const
194{
195
196 //LDEBUG << "LTRTextBuilder::updateLTR_TokenFromVertex(" << vertex << ")";
197 // get data from the result of the linguistic analysis
198 Token* fullToken = get(vertex_token, graph, vertex);
199 MorphoSyntacticData* data = get(vertex_data, graph, vertex);
201
202 if (data->size()==0) {
204 LERROR << "Empty MorphoSyntacticData for vertex" << vertex << ", token=" << fullToken->stringForm();
205 }
206
207 sort(data->begin(),data->end(),ltNormProperty(*m_macroAccessor));
208
209 StringsPoolIndex norm(0),lastNorm(0);
210 LinguisticCode macro,lastMacro;
211 for (MorphoSyntacticData::const_iterator elemItr=data->begin();
212 elemItr!=data->end(); elemItr++) {
213 norm = elemItr->normalizedForm;
214 macro = m_macroAccessor->readValue(elemItr->properties);
215 if (norm == lastNorm && macro == lastMacro) {
216 continue;
217 }
218 else {
219 lastNorm=norm;
220 lastMacro=macro;
221 LimaString normStr= sp[norm];
222 // test if the same word was not already met at this position
223 bool selectionFlag = true;
224 LTR_Token::const_iterator itTok = tokenRep->begin();
225 while (selectionFlag && (itTok != tokenRep->end())) {
226 selectionFlag = (itTok->first->getLemma() != normStr) ||
227 (itTok->first->getCategory() != macro);
228 itTok ++;
229 }
230 if (selectionFlag) {
231 // test if the current token is a plain word
232 bool plainWordFlag =
233 this->isWordToSelect(normStr, macro,m_microAccessor->readValue(elemItr->properties));
234 boost::shared_ptr< BoWToken> bowToken (new BoWToken(normStr,macro,
235 fullToken->position() + offset,
236 fullToken->length()));
237 bowToken->setInflectedForm(fullToken->stringForm());
238 //LDEBUG << "--add:" << fullToken->stringForm();
239 tokenRep->push_back(make_pair(bowToken, plainWordFlag));
240 }
241 //else { LDEBUG << "--ignored:" << fullToken->stringForm(); }
242 }
243 }
244}
245
246
247// -----------------------------------------------------------------------------
248// -- word selection
249// -----------------------------------------------------------------------------
250
252 const Lima::LimaString& lemma,
253 LinguisticCode macroCategory,
254 LinguisticCode microCategory) const
255{
256
257 if (this->isWordToSelect(lemma, macroCategory)) {
258 return ! static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).isAnEmptyMicroCategory(microCategory);
259 }
260 return false;
261}
262
264 const Lima::LimaString& lemma,
265 LinguisticCode macroCategory) const {
266
267 if (static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).isAnEmptyMacroCategory(macroCategory)) {
268 return false;
269 }
270 else {
271 if (m_stopList && (m_stopList->find(lemma) != m_stopList->end())) {
272 return false;
273 }
274 }
275 return true;
276}
277
278
279} // AnalysisDumpers
280} // LinguisticProcessing
281} // Lima
=========================================================================
#define LWARN
Definition LimaCommon.h:160
#define LDEBUG
Definition LimaCommon.h:157
#define LERROR
Definition LimaCommon.h:161
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define DUMPERLOGINIT
This class contains the representation of an element of the bag of words.
Definition bowToken.h:46
a token of a linear text representation Each LTR_Token represents the set of words that start at a po...
Definition ltrToken.h:48
Holds linguistic data for one language.
const FsaStringsPool & stringsPool(MediaId med) const
const MediaData & mediaData(MediaId media) const
LinguisticCode readValue(const LinguisticCode &code) const
read a property in a coded int.
LTRTextBuilder(const MediaId &language, std::shared_ptr< StopList > stopList)
void buildLTRTextFrom(const LinguisticGraph &graph, Lima::LinguisticProcessing::SegmentationData *sb, const LinguisticGraphVertex &graphFirstVertex, const LinguisticGraphVertex &graphLastVertex, boost::shared_ptr< Lima::Common::BagOfWords::LTR_Text > textRep, uint64_t offset)
build a LTRText representation of the analyzed text
bool isWordToSelect(const Lima::LimaString &lemma, LinguisticCode macroCategory) const
const std::vector< Segment > & getSegments() const
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
=========================================================================
NAUTITIA.
QString LimaString
Definition LimaString.h:33
STL namespace.
PUGI__FN void sort(I begin, I end, const Pred &pred)
Definition pugixml.cpp:7549