LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
SegmentationData.cpp
Go to the documentation of this file.
1// Copyright 2011-2020 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/************************************************************************
7 *
8 * @file SegmentationData.cpp
9 * @author Romaric Besancon (romaric.besancon@cea.fr)
10 * @date Tue Jan 18 2011
11 * copyright Copyright (C) 2011 by CEA LIST
12 *
13 ***********************************************************************/
14
15#include "SegmentationData.h"
16#include <queue>
17#include <set>
18#include <algorithm>
19
20using namespace std;
21//using namespace boost;
23
24namespace Lima {
25namespace LinguisticProcessing {
26
27//***********************************************************************
28// Segment class
29Segment::Segment(const std::string& type) :
30 m_begin(0),
31 m_end(0),
32 m_posBegin(0),
33 m_length(0),
34 m_type(type)
35{
36}
37
38Segment::Segment(const std::string& type,
42 m_begin(begin),
43 m_end(end),
44 m_posBegin(0),
45 m_length(0),
46 m_type(type)
47{
48#ifdef DEBUG_LP
50 LDEBUG << "Segment::Segment :"<< begin << end;
51#endif
52 // find position and length in graph
53 LinguisticGraph* graph = anagraph->getGraph();
54
55 // begin vertex is the vertex before first element of the segment :
56 // use following vertices to find position of first element
57 uint64_t position(0);
58 bool foundPos(false);
59 LinguisticGraphOutEdgeIt it, it_end;
60 boost::tie(it, it_end) = boost::out_edges(begin, *graph);
61 for (; it != it_end; it++)
62 {
63 LinguisticGraphVertex v = target(*it, *graph);
64 if (v == 1)
65 { // vertex following begin vertex is end of graph vertex => empty segment
66#ifdef DEBUG_LP
67 LDEBUG << "Warning: empty segment";
68#endif
69 // keep default 0 values for (pos,len) to be informed that this is an empty segment
70// return;
71 }
72 Token* t = get(vertex_token,*graph,v);
73 if (t != nullptr)
74 {
75 if (foundPos && position!=t->position())
76 {
78 LWARN << "Warning: conflicting position for alternative vertices";
79 }
80 else
81 {
82 position=t->position();
83 foundPos=true;
84 }
85 }
86 else
87 {
88#ifdef DEBUG_LP
89 LDEBUG << "Warning: no token for vertex v after begin:" << v;
90#endif
91 }
92 }
93 if (foundPos)
94 {
95 m_posBegin=position;
96 }
97 else
98 {
100 LERROR << "Error: cannot find position of begin vertex "
101 << begin << " for segmentation data";
102 }
103
104 // last vertex is the last element of the segment except if it is vertex 1
105 // (last one in the graph). Then use previous vertices. If the (only) previous
106 // vertex is 0, then the graph is empty and length is null.
107
108 uint64_t positionEnd(0);
109 bool foundPosEnd(false);
110 LinguisticGraphInEdgeIt pit, pit_end;
111 if (end == 1)
112 {
113 boost::tie(pit, pit_end) = boost::in_edges(end, *graph);
114 for (; pit != pit_end; pit++)
115 {
116 LinguisticGraphVertex v=source(*pit,*graph);
117 if (v == 0)
118 {
119 m_length = 0;
120 return;
121 }
122 else
123 {
124 Token* t = get(vertex_token, *graph, v);
125 if (t != nullptr)
126 {
127 if (foundPosEnd && positionEnd != t->position()+t->length())
128 {
130 LWARN << "Warning: conflicting position for alternative vertices";
131 }
132 else
133 {
134 positionEnd=t->position()+t->length();
135 foundPosEnd=true;
136 }
137 }
138 else
139 {
141 LWARN << "Warning: no token for vertex before end" << v;
142 }
143 }
144 }
145 }
146 else
147 {
148 Token* t = get(vertex_token,*graph,end);
149 if (t != nullptr)
150 {
151 if (foundPosEnd && positionEnd!=t->position()+t->length())
152 {
154 LWARN << "Warning: conflicting position for alternative vertices";
155 }
156 else
157 {
158 positionEnd=t->position()+t->length();
159 foundPosEnd=true;
160 }
161 }
162 else
163 {
165 LWARN << "Warning: no token for vertex end" << end;
166 }
167 }
168 if (foundPosEnd)
169 {
170#ifdef DEBUG_LP
171 LDEBUG << "Segment::Segment m_length = "<< positionEnd << "-" << m_posBegin << "+1";
172#endif
173 m_length=positionEnd-m_posBegin+1;
174 }
175 else
176 {
178 LERROR << "Error: cannot determine length of segment for segmentation data ("
179 << begin << "," << end << ")";
180 }
181#ifdef DEBUG_LP
182 LDEBUG << "Segment::Segment :"<< m_begin << m_end << m_posBegin << m_length;
183#endif
184}
185
186// Segment::Segment(const std::string& type,
187// uint64_t posBegin,
188// uint64_t length,
189// LinguisticAnalysisStructure::AnalysisGraph* anagraph):
190// m_begin(0),
191// m_end(0),
192// m_posBegin(posBegin),
193// m_length(length),
194// m_type(type)
195
197 uint64_t length,
199{
200 m_posBegin=posBegin;
201 m_length=length;
202
203 // find first and last vertex in graph : have to go through the graph
204 LinguisticGraph* graph=anagraph->getGraph();
205
206 uint64_t posEnd=posBegin+length;
207
208 std::queue<std::pair<LinguisticGraphVertex,LinguisticGraphVertex> > toVisit;
209 std::set<LinguisticGraphVertex> visited;
210
211 LinguisticGraphOutEdgeIt outItr,outItrEnd;
212
213 // output vertices between begin and end,
214 // but do not include begin (beginning of text or previous end of sentence)
215 // and include end (end of sentence)
216 toVisit.push(make_pair(anagraph->firstVertex(),0));
217
218 bool first=true;
219 while (!toVisit.empty())
220 {
221 auto v = toVisit.front(); // clazy:exclude=rule-of-two-soft
222 toVisit.pop();
223 if (v.first == anagraph->lastVertex())
224 {
225 break;
226 }
227
228 bool endIsNextVertex(false);
229 if (first)
230 {
231 first = false;
232 }
233 else
234 {
235 Token* t = get(vertex_token,*graph,v.first);
236 if(t!=0) {
237 if (t->position() == posBegin)
238 {
239 m_begin = v.second;
240 }
241 if (t->position()+t->length() == posEnd)
242 {
243 // must take next vertex
244 endIsNextVertex = true;
245 }
246 }
247 }
248
249 // add next vertices
250 for (boost::tie(outItr,outItrEnd)=out_edges(v.first,*graph);
251 outItr != outItrEnd; outItr++)
252 {
253 LinguisticGraphVertex next = target(*outItr,*graph);
254 if (endIsNextVertex)
255 {
256 m_end=next;
257 break; // no need to go further in the graph
258 }
259 if (visited.find(next)==visited.end())
260 {
261 visited.insert(next);
262 toVisit.push(make_pair(next,v.first));
263 }
264 }
265 }
266}
267
268bool Segment::operator<(const Segment& s) const
269{
270 return (m_posBegin<s.getPosBegin());
271}
272
274{
275#ifdef DEBUG_LP
277 LDEBUG << "Segment::addSegment [" << s.getPosBegin() << "," << s.getLength()
278 << "] to [" << getPosBegin() << "," << getLength() << "]";
279#endif
280
281 // do not check types, keep type of current segment
282 // do not check adjacency, juste update end of segment
283 m_end = s.getLastVertex();
284 m_length = s.getPosEnd() - m_posBegin;
285}
286
287// debugging utility
288std::ostream& operator<<(std::ostream& os, const Segment& seg)
289{
290 os << "Segment(posBegin="<<seg.m_posBegin
291 <<", length="<<seg.m_length
292 <<", type="<<seg.m_type
293 <<", begin="<<seg.m_begin
294 <<", end="<<seg.m_end
295 <<")";
296 return os;
297}
298
299QDebug& operator<<(QDebug& os, const Segment& seg)
300{
301 os << "Segment(posBegin="<<seg.m_posBegin
302 <<", length="<<seg.m_length
303 <<", type="<<seg.m_type
304 <<", begin="<<seg.m_begin
305 <<", end="<<seg.m_end
306 <<")";
307 return os;
308}
309
310//***********************************************************************
311// constructors and destructors
312SegmentationData::SegmentationData(const std::string& graphId):
313m_graphId(graphId)
314{
315}
316
320
321//***********************************************************************
323{
324#ifdef DEBUG_LP
326 LDEBUG << "SegmentationData::add" << s.getType().c_str()
327 << s.getFirstVertex() << s.getLastVertex();
328#endif
329 // segments are sorted in the vector: use binary search to insert new segment
330 if (s.getLength() == 0)
331 {
332#ifdef DEBUG_LP
333 LDEBUG << "SegmentationData::add trying to add empty segment: ignored";
334#endif
335 }
336 else
337 {
338 // ??OME2 SegmentationData::iterator it=lower_bound( begin(),end(),s);
339 auto it = lower_bound( m_segments.begin(),m_segments.end(),s);
340 // ??OME2 insert(it,s);
341 m_segments.insert(it,s);
342 }
343}
344
345} // end namespace
346} // end namespace
#define LWARN
Definition LimaCommon.h:160
#define LDEBUG
Definition LimaCommon.h:157
#define LERROR
Definition LimaCommon.h:161
LinguisticGraph::in_edge_iterator LinguisticGraphInEdgeIt
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define SEGMENTATIONLOGINIT
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
void setVerticesFromPositions(uint64_t posBegin, uint64_t length, LinguisticAnalysisStructure::AnalysisGraph *anagraph)
LinguisticGraphVertex getFirstVertex() const
bool operator<(const Segment &s) const
const std::string & getType() const
LinguisticGraphVertex getLastVertex() const
SegmentationData(const std::string &sourceGraph="")
std::ostream & operator<<(std::ostream &os, const ChainIdStruct &ids)
NAUTITIA.
STL namespace.