LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
indexElement.cpp
Go to the documentation of this file.
1// Copyright 2002-2020 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/************************************************************************
7 *
8 * @file indexElement.cpp
9 * @author Besancon Romaric (romaric.besancon@cea.fr)
10 * @date Tue Feb 7 2006
11 * copyright Copyright (C) 2006-2020 by CEA LIST
12 *
13 ***********************************************************************/
14
15#include "indexElement.h"
19
20using namespace Lima::Common::Misc;
21using namespace Lima::Common::BagOfWords;
22
23namespace Lima {
24namespace Common {
25namespace BagOfWords {
26
28{
29 friend class IndexElement;
30 friend std::ostream& operator<<(std::ostream& os, const IndexElement& elt);
31 friend QDebug& operator<<(QDebug& os, const IndexElement& elt);
32 friend QTextStream& operator<<(QTextStream& os, const IndexElement& elt);
33
35 IndexElementPrivate(const uint64_t id,
36 const BagOfWords::BoWType type,
37 const LimaString& word,
38 const LinguisticCode cat,
39 const uint64_t position=0,
40 const uint64_t length=0,
42 IndexElementPrivate(const uint64_t id,
43 const BagOfWords::BoWType type,
44 const std::vector<uint64_t>& structure,
45 const std::vector<uint64_t>& relations,
48 IndexElementPrivate& operator=(const IndexElementPrivate& iep);
50
51 uint64_t m_id;
53 LimaString m_word;
54 // for simple term, keep also some informations that
55 // may be useful, such as category and position
56 LinguisticCode m_category;
57 uint64_t m_position;
58 uint64_t m_length;
60 Misc::PositionLengthList m_poslenlist;
61 std::vector<uint64_t> m_structure;
62 std::vector<uint64_t> m_relations;
63};
64
65
66IndexElementPrivate::IndexElementPrivate():
67m_id(0),
68m_type(BoWType::BOW_NOTYPE),
69m_word(),
70m_position(0),
71m_length(0),
72m_neType(),
73m_poslenlist(0),
74m_structure(),
75m_relations()
76{
77}
78
79IndexElementPrivate::IndexElementPrivate(const IndexElementPrivate& iep):
80m_id(iep.m_id),
81m_type(iep.m_type),
82m_word(iep.m_word),
83m_category(iep.m_category),
84m_position(iep.m_position),
85m_length(iep.m_length),
86m_neType(iep.m_neType),
87m_poslenlist(iep.m_poslenlist),
88m_structure(iep.m_structure),
89m_relations(iep.m_relations)
90{
91}
92
93IndexElementPrivate::IndexElementPrivate(
94 const uint64_t id,
95 const BagOfWords::BoWType type,
96 const LimaString& word,
97 const LinguisticCode cat,
98 const uint64_t position,
99 const uint64_t length,
100 const MediaticData::EntityType neType):
101m_id(id),
102m_type(type),
103m_word(word),
104m_category(cat),
105m_position(position),
106m_length(length),
107m_neType(neType),
108m_poslenlist(1, std::make_pair(
109 Position(position),
110 Length(length))),
111m_structure(),
112m_relations()
113{
114}
115
116IndexElementPrivate::IndexElementPrivate(
117 const uint64_t id,
118 const BagOfWords::BoWType type,
119 const std::vector<uint64_t>& structure,
120 const std::vector<uint64_t>& relations,
121 const MediaticData::EntityType neType):
122m_id(id),
123m_type(type),
124m_word(),
125m_position(0),
126m_length(0),
127m_neType(neType),
128m_poslenlist(0),
129m_structure(structure),
130m_relations(relations)
131{
132}
133
134
135IndexElementPrivate& IndexElementPrivate::operator=(const IndexElementPrivate& iep)
136{
137 if (this != &iep)
138 {
139 m_id = iep.m_id;
140 m_type = iep.m_type;
141 m_word = iep.m_word;
142 m_category = iep.m_category;
143 m_position = iep.m_position;
144 m_length = iep.m_length;
145 m_neType = iep.m_neType;
146 m_poslenlist = iep.m_poslenlist;
147 m_structure = iep.m_structure;
148 m_relations = iep.m_relations;
149 }
150 return *this;
151}
152
153//***********************************************************************
154// constructors and destructors
158
160 const uint64_t id,
161 const BagOfWords::BoWType type,
162 const LimaString& word,
163 const LinguisticCode cat,
164 const uint64_t position,
165 const uint64_t length,
166 const MediaticData::EntityType neType):
167 m_d(new IndexElementPrivate(id, type, word, cat, position, length, neType))
168{
169}
170
172 const uint64_t id,
173 const BagOfWords::BoWType type,
174 const std::vector<uint64_t>& structure,
175 const std::vector<uint64_t>& relations,
176 const MediaticData::EntityType neType):
177 m_d(new IndexElementPrivate(id, type, structure, relations, neType))
178{
179}
180
182{
183}
184
186{
187 if (this != &ie)
188 {
189 *m_d = *ie.m_d;
190 }
191 return *this;
192}
193
195{
196 delete m_d;
197}
198
200{
201 if (isSimpleTerm()) {
202 return (other.isSimpleTerm() &&
203 m_d->m_type==other.m_d->m_type &&
204 m_d->m_word==other.getSimpleTerm() &&
205 m_d->m_category==other.getCategory());
206 }
207 else {
208 return (!other.isSimpleTerm() &&
209 m_d->m_type==other.m_d->m_type &&
210 getSimpleTerm() == other.getSimpleTerm() &&
211 m_d->m_structure==other.getStructure() &&
212 m_d->m_relations==other.getRelations());
213 }
214}
215
216bool IndexElement::empty() const { return m_d->m_id==0; }
217
218uint64_t IndexElement::getId() const { return m_d->m_id; }
219
220BagOfWords::BoWType IndexElement::getType() const { return m_d->m_type; }
221
222bool IndexElement::isSimpleTerm() const { return m_d->m_type == BoWType::BOW_TOKEN || (m_d->m_type == BoWType::BOW_NAMEDENTITY && m_d->m_structure.empty()); }
223
224bool IndexElement::isComposedTerm() const { return m_d->m_type == BoWType::BOW_TERM || (m_d->m_type == BoWType::BOW_NAMEDENTITY && ! m_d->m_structure.empty()); }
225
226bool IndexElement::isPredicate() const { return m_d->m_type == BoWType::BOW_PREDICATE; }
227
228const LimaString& IndexElement::getSimpleTerm() const { return m_d->m_word; }
229
230LinguisticCode IndexElement::getCategory() const { return m_d->m_category; }
231
232uint64_t IndexElement::getPosition() const { return m_d->m_position; }
233
234uint64_t IndexElement::getLength() const { return m_d->m_length; }
235
236bool IndexElement::isNamedEntity() const { return m_d->m_type == BoWType::BOW_NAMEDENTITY; }
237
238const MediaticData::EntityType& IndexElement::getNamedEntityType() const { return m_d->m_neType; }
239
241
242const PositionLengthList& IndexElement::getPositionLengthList() const { return m_d->m_poslenlist; }
243
244const std::vector<uint64_t>& IndexElement::getStructure() const { return m_d->m_structure; }
245
246std::vector<uint64_t>& IndexElement::getStructure() { return m_d->m_structure; }
247
248const std::vector<uint64_t>& IndexElement::getRelations() const { return m_d->m_relations; }
249
250std::vector<uint64_t>& IndexElement::getRelations() { return m_d->m_relations; }
251
252void IndexElement::setId(const uint64_t id) { m_d->m_id=id; }
253
254void IndexElement::setSimpleTerm(const LimaString& t) { m_d->m_word=t; }
255
256void IndexElement::setCategory(LinguisticCode category) { m_d->m_category = category; }
257
258void IndexElement::setStructure(const std::vector<uint64_t>& s,
259 const std::vector<uint64_t>& r)
260 { m_d->m_structure=s; m_d->m_relations=r; }
261
262void IndexElement::addInStructure(uint64_t id, uint64_t rel)
263 { m_d->m_structure.push_back(id); m_d->m_relations.push_back(rel); }
264
266{
267 if (isSimpleTerm())
268 {
269 return (other.isSimpleTerm() &&
270 m_d->m_position==other.getPosition() &&
271 m_d->m_length==other.getLength());
272 }
273 else
274 {
275 return (!other.isSimpleTerm() &&
276 m_d->m_poslenlist==other.getPositionLengthList());
277 }
278}
279
281{
282 // if both are simple terms or complex terms, compare them. Otherwise return false
283 if (isSimpleTerm() && other.isSimpleTerm())
284 {
285 // both are simple terms
286
287 // terms intersects: the beginning of one of them is inside the other one span
288 return ( ( (m_d->m_position<=other.getPosition()) && (other.getPosition() < m_d->m_position+m_d->m_length) )
289 ||( (other.getPosition()<=m_d->m_position) && (m_d->m_position < other.getPosition()+other.getLength()) )
290 );
291 }
292 else if (!isSimpleTerm() && !other.isSimpleTerm())
293 {
294 // both are complex terms
295
296 // compute the minimum and maximum positions in this index element as the
297 // minimal value of all positions and the maximum value of all positions
298 // plus the corresponding length
299 PositionLengthList::const_iterator pplIt = getPositionLengthList().begin();
300 Position posMin = pplIt->first;
301 Position posMax = static_cast<Position>(posMin + pplIt->second - 1);
302 for( ; pplIt != getPositionLengthList().end() ; pplIt++ )
303 {
304 if( pplIt->first < posMin )
305 {
306 posMin = pplIt->first;
307 }
308 if( (pplIt->first+pplIt->second - 1) > posMax )
309 {
310 posMax = pplIt->first+pplIt->second - 1;
311 }
312 }
313 // compute the minimum and maximum positions in the other index element in
314 // the same way
315 PositionLengthList::const_iterator otherPplIt = other.getPositionLengthList().begin();
316 Position otherPosMin = other.getPositionLengthList().back().first;
317 Position otherPosMax = static_cast<Position>(otherPosMin + getPositionLengthList().back().second - 1);
318 for( ; otherPplIt != other.getPositionLengthList().end() ; otherPplIt++ )
319 {
320 if( otherPplIt->first < otherPosMin )
321 {
322 otherPosMin = otherPplIt->first;
323 }
324 if( (otherPplIt->first+otherPplIt->second - 1) > otherPosMax )
325 {
326 otherPosMax = otherPplIt->first+otherPplIt->second - 1;
327 }
328 }
329
330 // The beginning of one of the complex terms is inside the other one span
331 return( ( (posMin <= otherPosMin) && (otherPosMin < posMax) )
332 || ( (otherPosMin <= posMin) && (posMin < otherPosMax) ) );
333 }
334 return false;
335}
336
337std::ostream& operator<<(std::ostream& os, const IndexElement& elt)
338{
339 os << "[IndexElement" << elt.m_d->m_id << "," << elt.m_d->m_type ;
340 if (elt.isSimpleTerm())
341 {
342 os << ":" << Common::Misc::limastring2utf8stdstring(elt.m_d->m_word);
343 if (elt.m_d->m_category != L_NONE)
344 {
345 os << "/" << elt.m_d->m_category.toString();
346 }
347 os << "/" << elt.m_d->m_position;
348 os << "," << elt.m_d->m_length;
349 }
350 else
351 {
352 if (elt.m_d->m_structure.empty())
353 {
354 return os << ":";
355 }
356 else
357 {
358 uint64_t i=0;
359 os << ":" << elt.m_d->m_structure[i] << " RE(" << elt.m_d->m_relations[i] << ")";
360 i++;
361 while (i<elt.m_d->m_structure.size())
362 {
363 os << "," << elt.m_d->m_structure[i] << " RE(" << elt.m_d->m_relations[i] << ")";
364 i++;
365 }
366 }
367 os << "/";
368 ::operator<<(os,elt.m_d->m_poslenlist);
369 }
370 if (! elt.m_d->m_neType.isNull())
371 {
372 os << "/NE(" << MediaticData::MediaticData::single().getEntityName(elt.m_d->m_neType).toUtf8().constData() << ")";
373 }
374 os << "]";
375 return os;
376}
377
378QDebug& operator<<(QDebug& os, const IndexElement& elt)
379{
380 os << "[IndexElement" << elt.m_d->m_id << "," << elt.m_d->m_type;
381 os << ":" << elt.m_d->m_word;
382 if (elt.m_d->m_category != L_NONE)
383 {
384 os << "/" << elt.m_d->m_category.toString();
385 }
386 os << "/" << elt.m_d->m_position;
387 os << "," << elt.m_d->m_length;
388 if (!elt.m_d->m_structure.empty())
389 {
390 uint64_t i=0;
391 os << ":" << elt.m_d->m_structure[i] << " RE(" << elt.m_d->m_relations[i] << ")";
392 i++;
393 while (i<elt.m_d->m_structure.size())
394 {
395 os << "," << elt.m_d->m_structure[i] << " RE(" << elt.m_d->m_relations[i] << ")";
396 i++;
397 }
398 }
399 os << "/" << elt.m_d->m_poslenlist;
400 if (elt.isNamedEntity())
401 {
402 os << "/NE(" << MediaticData::MediaticData::single().getEntityName(elt.m_d->m_neType) << ")";
403 }
404 else if (elt.isPredicate())
405 {
406 os << "/P(" << MediaticData::MediaticData::single().getEntityName(elt.m_d->m_neType) << ")";
407 }
408 os << "]";
409 return os;
410}
411
412QTextStream& operator<<(QTextStream& os, const IndexElement& elt)
413{
414 os << "[IndexElement" << elt.m_d->m_id << "," << elt.m_d->m_type;
415 if (elt.isSimpleTerm())
416 {
417 os << ":" << elt.m_d->m_word;
418 if (elt.m_d->m_category != L_NONE)
419 {
420 os << "/" << elt.m_d->m_category.toString().c_str();
421 }
422 os << "/" << elt.m_d->m_position;
423 os << "," << elt.m_d->m_length;
424 }
425 else
426 {
427 if (elt.m_d->m_structure.empty())
428 {
429 return os << ":";
430 }
431 if (!elt.m_d->m_structure.empty())
432 {
433 uint64_t i=0;
434 os << ":" << elt.m_d->m_structure[i] << " RE(" << elt.m_d->m_relations[i] << ")";
435 i++;
436 while (i<elt.m_d->m_structure.size())
437 {
438 os << "," << elt.m_d->m_structure[i] << " RE(" << elt.m_d->m_relations[i] << ")";
439 i++;
440 }
441 }
442 os << "/";
443 ::operator<<(os,elt.m_d->m_poslenlist);
444 }
445 if (! elt.m_d->m_neType.isNull())
446 {
447 os << "/NE(" << MediaticData::MediaticData::single().getEntityName(elt.m_d->m_neType) << ")";
448 }
449 os << "]";
450 return os;
451}
452
453
454} // end namespace
455} // end namespace
456} // end namespace
#define L_NONE
Definition StdBitset.h:338
friend std::ostream & operator<<(std::ostream &os, const IndexElement &elt)
Represent an element of an index If it is a predicate, its simple term is "PredicateElement" and its ...
void addInStructure(uint64_t id, uint64_t rel)
bool operator==(const IndexElement &other) const
equality operator just compare content, not positions
Lima::Common::BagOfWords::BoWType getType() const
const std::vector< uint64_t > & getStructure() const
void setCategory(LinguisticCode category)
void setSimpleTerm(const LimaString &t)
bool hasSamePosition(const IndexElement &other) const
compare position and length
const std::vector< uint64_t > & getRelations() const
const LimaString & getSimpleTerm() const
bool hasNearlySamePosition(const IndexElement &other) const
compare position and length: thereis some intersaction in position
Lima::Common::Misc::PositionLengthList & getPositionLengthList()
const Common::MediaticData::EntityType & getNamedEntityType() const
IndexElement & operator=(const IndexElement &)
void setStructure(const std::vector< uint64_t > &s, const std::vector< uint64_t > &r)
LimaString getEntityName(const EntityType &type) const
std::string toString() const
Definition StdBitset.h:152
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
BoWType
enum to characterize the type of the AbstractBoWElement
@ BOW_NOTYPE
the AbstractBoWElement is an abstract one that should not be instanciated
@ BOW_TERM
the AbstractBoWElement is a multi-term
@ BOW_NAMEDENTITY
the AbstractBoWElement is a named entity
@ BOW_TOKEN
the AbstractBoWElement is a simple token
@ BOW_PREDICATE
the AbstractBoWElement is a predicate (n-ary relation, template or semantic frame
T & operator<<(T &qd, const BoWType &bt)
std::vector< std::pair< Position, Length > > PositionLengthList
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
NAUTITIA.
QString LimaString
Definition LimaString.h:33
STL namespace.