LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
bowDocumentST.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/************************************************************************
7 *
8 * @file boWDocumentST.cpp
9 * @author Besancon Romaric (besanconr@zoe.cea.fr)
10 * @date Fri Oct 10 2003
11 * copyright Copyright (C) 2003 by CEA LIST
12 *
13 ***********************************************************************/
14
15#include "bowDocumentST.h"
16#include "bowToken.h"
17#include "common/LimaCommon.h"
19
20#include <vector>
21#include <map>
22
23namespace Lima {
24namespace Common {
25namespace BagOfWords {
26
27//***********************************************************************
28// constructors
29//***********************************************************************
32m_sentenceBreaks(),
33m_topicShifts()
34{
35}
36
39m_sentenceBreaks(),
40m_topicShifts()
41{
42 copySTData(d);
43}
44
45
46//***********************************************************************
47// destructor
48//***********************************************************************
50 m_sentenceBreaks.clear();
51 m_topicShifts.clear();
52}
53
54//***********************************************************************
55// assignment operator
56//***********************************************************************
58 if (&d != this) {
60 m_sentenceBreaks.clear();
61 m_topicShifts.clear();
62 copySTData(d);
63 }
64 return *this;
65}
66
67//***********************************************************************
68// member functions
69//***********************************************************************
70void BoWDocumentST::copySTData(const BoWDocumentST& d) {
71 // first build correspondance between iterators and indexes
72 std::map<BoWText::const_iterator, BoWText::const_iterator> iteratorMap;
73 BoWText::const_iterator thisIt=this->begin();
74 BoWText::const_iterator it = d.begin();
75 while (it != d.end() && thisIt != this->end()) {
76 iteratorMap[it] = thisIt;
77 it++;
78 thisIt++;
79 }
80 if (thisIt != this->end() || it != d.end()) {
81 throw std::runtime_error("cannot build iterator map in BoWDocumentST copy");
82 }
83
84 // copy sentenceBreaks and topicShifts;
85 for (std::vector<BoWText::const_iterator>::const_iterator
86 it=d.m_sentenceBreaks.begin(); it!=d.m_sentenceBreaks.end(); it++) {
87 m_sentenceBreaks.push_back(iteratorMap[*it]);
88 }
89 for (std::vector<BoWText::const_iterator>::const_iterator
90 it=d.m_topicShifts.begin(); it!=d.m_topicShifts.end(); it++) {
91 m_topicShifts.push_back(iteratorMap[*it]);
92 }
93}
94
97 m_sentenceBreaks.clear();
98 m_topicShifts.clear();
99}
100
101//***********************************************************************
102// text input/output
103//***********************************************************************
104/*
105std::wostream& operator << (std::wostream& os, const BoWDocumentST& d) {
106 // dump BoWDocument part
107 os << static_cast<BoWDocument>(d);
108 // dump sentence breaks
109 os << L"## sentence breaks" << std::endl;
110 std::vector<BoWText::const_iterator>::const_iterator itSentBrk;
111 for (itSentBrk = d.m_sentenceBreaks.begin();
112 itSentBrk != d.m_sentenceBreaks.end(); itSentBrk ++) {
113 os << ***itSentBrk << std::endl;
114 }
115
116 // dump topic shifts
117 os << L"## topic shifts" << std::endl;
118//if (d.m_topicShifts.empty()) {
119// os << L"no topic shifts" << std::endl;
120//}
121 std::vector<BoWText::const_iterator>::const_iterator itTopSht;
122 for (itTopSht = d.m_topicShifts.begin();
123 itTopSht != d.m_topicShifts.end(); itTopSht ++) {
124 os << ***itTopSht << std::endl;
125 }
126
127 return os;
128}
129*/
130
132
133std::ostream& operator << (std::ostream& os, const BoWDocumentST& d) {
134
135 // dump BoWDocument part
136 os << static_cast<BoWDocument>(d);
137
138 // dump sentence breaks
139 os << "## sentence breaks" << std::endl;
140 std::vector<BoWText::const_iterator>::const_iterator itSentBrk;
141 for (itSentBrk = d.m_sentenceBreaks.begin();
142 itSentBrk != d.m_sentenceBreaks.end(); itSentBrk ++) {
143 os << (**itSentBrk)->getOutputUTF8String() << std::endl;
144 }
145
146 // dump topic shifts
147 os << "## topic shifts" << std::endl;
148//if (d.m_topicShifts.empty()) {
149// os << L"no topic shifts" << std::endl;
150//}
151 std::vector<BoWText::const_iterator>::const_iterator itTopSht;
152 for (itTopSht = d.m_topicShifts.begin();
153 itTopSht != d.m_topicShifts.end(); itTopSht ++) {
154 os << (**itTopSht)->getOutputUTF8String() << std::endl;
155 }
156
157 return os;
158
159}
160
161
162//***********************************************************************
163// binary input/output
164//***********************************************************************
165
166void BoWDocumentST::read(std::istream& file) {
167
168 // a BoWDocument is a BoWDocument with more data; hence, first
169 // read the BoWDocument part
170 BoWDocument::read(file);
171
172 // build a dictionary for mapping numeric indexes in file
173 // and BoWTokens
174 // certainly not the most efficient way to do it
175 std::map<uint64_t, BoWText::const_iterator> indexToIterator;
176 uint64_t tokenCounter = 1;
177 for (BoWText::const_iterator itTok = this->begin();
178 itTok != this->end(); itTok ++) {
179 indexToIterator[tokenCounter] = itTok;
180 tokenCounter ++;
181 }
182
183 // sentence breaks
184 // read the number of sentence breaks
185 uint64_t sentBrkNb = Misc::readCodedInt(file);
186 // read sentence break indexes and map them to BoWTokens
187 for (uint64_t sentBrkInd = 1; sentBrkInd <= sentBrkNb; sentBrkInd ++) {
188 uint64_t sentBrkIndex = Misc::readCodedInt(file);
189 std::map<uint64_t, BoWText::const_iterator>::const_iterator itMap =
190 indexToIterator.find(sentBrkIndex);
191 if (itMap != indexToIterator.end()) {
192 m_sentenceBreaks.push_back(itMap->second);
193 }
194 else {
195 throw std::runtime_error("invalid sentence break reference\n");
196 }
197 }
198
199 // read topic shifts
200 // read the number of topic shifts
201 uint64_t topShtNb = Misc::readCodedInt(file);
202 // read topic shift indexes and map them to BoWTokens
203 for (uint64_t topShtInd = 1; topShtInd <= topShtNb; topShtInd ++) {
204 uint64_t topShtIndex = Misc::readCodedInt(file);
205 std::map<uint64_t, BoWText::const_iterator>::const_iterator itMap =
206 indexToIterator.find(topShtIndex);
207 if (itMap != indexToIterator.end()) {
208 m_topicShifts.push_back(itMap->second);
209 }
210 else {
211 throw std::runtime_error("invalid topic shift reference\n");
212 }
213 }
214
215}
216
217void BoWDocumentST::write(std::ostream& file) const {
218
219 // a BoWDocument is a BoWDocument with more data; hence, first
220 // write the BoWDocument part
221 BoWDocument::write(file);
222
223 // write BoWDocument specific data
224 this->writeSTData(file);
225
226}
227
228
229void BoWDocumentST::writeSTData(std::ostream& file) const
230{
231
232 // convert references to tokens into numeric indexes
233 std::vector<uint64_t> sentenceBreaks;
234 std::vector<uint64_t> topicShifts;
235 uint64_t tokenCounter = 1;
236 std::vector<BoWText::const_iterator>::const_iterator itSentBrk = m_sentenceBreaks.begin();
237 std::vector<BoWText::const_iterator>::const_iterator itTopSht = m_topicShifts.begin();
238 for (BoWText::const_iterator itTok = this->begin();
239 itTok != this->end(); itTok ++) {
240 if ((itSentBrk != m_sentenceBreaks.end()) && (*itTok == **itSentBrk)) {
241 sentenceBreaks.push_back(tokenCounter);
242 itSentBrk ++;
243 }
244 if ((itTopSht != m_topicShifts.end()) && (*itTok == **itTopSht)) {
245// std::cerr << "added topic shift index: " << tokenCounter
246// <<"(on token " << ***itTopSht << ")" << std::endl;
247 topicShifts.push_back(tokenCounter);
248 itTopSht ++;
249 }
250 tokenCounter ++;
251 }
252 if (itTopSht != m_topicShifts.end()) {
254 std::ostringstream oss;
255 do {
256 oss << (**itTopSht)->getOutputUTF8String() << " "; itTopSht++;
257 } while (itTopSht != m_topicShifts.end());
258 LERROR << "Write BoWDocumentST: missing topic shifts on tokens "
259 << oss.str();
260 }
261 if (itSentBrk != m_sentenceBreaks.end()) {
263 std::ostringstream oss;
264 do {
265 oss << (**itSentBrk)->getOutputUTF8String() << " "; itSentBrk++;
266 } while (itSentBrk != m_sentenceBreaks.end());
267 LERROR << "Write BoWDocumentST: missing sentence breaks on tokens "
268 << oss.str();
269 }
270
271 // write sentence breaks and topic shifts
272 // write the number of sentence breaks
273 Misc::writeCodedInt(file, sentenceBreaks.size());
274 // write sentence breaks
275 for (std::vector<uint64_t>::const_iterator itSent = sentenceBreaks.begin();
276 itSent != sentenceBreaks.end(); itSent ++) {
277 Misc::writeCodedInt(file, *itSent);
278 }
279
280 // write the number of topic shifts
281 Misc::writeCodedInt(file, topicShifts.size());
282 // write topic shifts
283 for (std::vector<uint64_t>::const_iterator itTop = topicShifts.begin();
284 itTop != topicShifts.end(); itTop ++) {
285 Misc::writeCodedInt(file, *itTop);
286 }
287
288}
289
290
291
292} // end namespace
293} // end namespace
294} // end namespace
#define LERROR
Definition LimaCommon.h:161
#define BOWLOGINIT
Definition LimaCommon.h:201
BoWDocumentST & operator=(const BoWDocumentST &)
void read(std::istream &file) override
specialization of read/write functions for taking into account sentence breaks and topic shifts
void writeSTData(std::ostream &file) const
void write(std::ostream &file) const
represent a document as a Bag Of Word, with associated document properties.
Definition bowDocument.h:57
BoWText & operator=(const BoWText &)
Definition bowText.cpp:52
T & operator<<(T &qd, const BoWType &bt)
uint64_t readCodedInt(std::istream &file)
read a integer coded in variable-byte format in a file
void writeCodedInt(std::ostream &file, const uint64_t number)
write a integer coded in variable-byte format in a file
NAUTITIA.