LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
treebank.cpp
Go to the documentation of this file.
1// Copyright 2021 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6#include <iostream>
7#include <fstream>
8#include <limits>
9#include <regex>
10
11#include <boost/filesystem.hpp>
12
13#include <unicode/uchar.h>
14
15#include "treebank.h"
16
17using namespace std;
18namespace fs = boost::filesystem;
19
20namespace deeplima
21{
22namespace CoNLLU
23{
24
25size_t Sentence::calc_num_of_words(const Annotation& annot) const
26{
27 size_t counter = 0;
28 size_t idx = m_first_token_line;
29 for (; idx < m_first_token_line + m_num_tokens; idx++)
30 {
31 const CoNLLULine& line = annot.get_line(idx);
32 if (line.is_real_word_line())
33 {
34 counter++;
35 }
36 }
37
38 assert(annot.get_line(idx).is_empty_line()); // double check the structure
39
40 return counter;
41}
42
43bool Treebank::load(const std::string& path)
44{
45 for (fs::directory_entry& entry : fs::directory_iterator(path))
46 {
47 if (entry.path().extension().string() == ".conllu")
48 {
49 map<string, string> fields;
50 if (!parse_ud_file_name(entry.path().filename().string(), fields)
51 || fields.end() == fields.find("part")
52 || fields["part"].size() == 0)
53 {
54 throw std::logic_error("Can't parse file name \"" + entry.path().filename().string() + "\"");
55 }
56 m_parts[fields["part"]] = AnnotatedDocument();
57
58 auto plain_text_fn = entry.path();
59 plain_text_fn.replace_extension("txt");
60 m_parts[fields["part"]].doc.load(plain_text_fn.string());
61 m_parts[fields["part"]].annot.m_pdoc = &m_parts[fields["part"]].doc;
62 m_parts[fields["part"]].annot.load(entry.path().string());
63 // std::cout << "Success" << std::endl;
64 }
65 }
66 return true;
67}
68
69void Document::load(const std::string& fn)
70{
71 ifstream input(fn);
72 if (!input.is_open())
73 {
74 throw std::runtime_error(
75 string("Document::load: companion plain-text file \"") + fn
76 + "\" is missing (it is needed for segmentation training; generate it from "
77 "the .conllu '# text =' lines, e.g. with make_trainable.py).");
78 }
79 m_original_text = std::string((istreambuf_iterator<char>(input)), istreambuf_iterator<char>());
80 if (m_original_text.empty())
81 {
82 throw std::runtime_error(
83 string("Document::load: companion plain-text file \"") + fn
84 + "\" is empty (it is needed for segmentation training; generate it from "
85 "the .conllu '# text =' lines, e.g. with make_trainable.py).");
86 }
87 std::wstring_convert<std::codecvt_utf8<wchar_t>> converter;
88 m_text = converter.from_bytes(m_original_text);
89}
90
91void Annotation::load(const std::string& fn)
92{
93 ifstream input(fn);
94 if (!input.is_open())
95 {
96 throw invalid_argument(string("Can't open file \"") + fn + "\"");
97 }
98
99 try
100 {
101 load(input);
102 }
103 catch (const std::runtime_error& e)
104 {
105 std::cerr << "Annotation::load runtime_error while loading "
106 << fn << " :" << std::endl << e.what() << std::endl;
107 throw;
108 }
109}
110
111void Annotation::load(std::istream& input)
112{
113 string line;
114 size_t num_sentences = 0;
115 size_t num_tokens = 0;
116 size_t num_words = 0;
117 while (getline(input, line))
118 {
119 CoNLLULine l(line);
120 m_lines.push_back(l);
121 if (l.is_empty_line())
122 {
123 ++num_sentences;
124 }
125
126 if (l.is_token_line())
127 {
128 ++num_tokens;
129 }
130
131 if (l.is_real_word_line())
132 {
133 ++num_words;
134 }
135 }
136
137 rebuild_structure(num_sentences, num_tokens, num_words);
138}
139
140wstring::size_type compare_ws_insensitive(const wchar_t *s1, const wchar_t *s2, size_t len)
141{
142 const wchar_t *p1 = s1, *p2 = s2;
143 size_t i = 0;
144
145 while (i < len)
146 {
147 if (*p1 == *p2 || (u_isUWhiteSpace(*p1) && u_isUWhiteSpace(*p2)))
148 {
149 i++; p1++; p2++;
150 }
151 else
152 {
153 if (u_isUWhiteSpace(*p1))
154 {
155 p1++;
156 }
157 else if (u_isUWhiteSpace(*p2))
158 {
159 p2++;
160 }
161 else
162 {
163 return wstring::npos;
164 }
165 }
166 }
167
168 return p1 - s1 - len;
169}
170
171void Annotation::rebuild_structure(size_t num_sentences, size_t num_tokens, size_t num_words)
172{
173 wstring_convert<codecvt_utf8<wchar_t>> converter;
174
175 m_sentences.reserve(num_sentences);
176 m_tokens.reserve(num_tokens);
177 m_words.reserve(num_words);
178
179 size_t sent_idx = 0;
180 size_t pos = 0;
181 size_t skip = 0;
182
183 for (size_t i = 0; i < m_lines.size(); i++)
184 {
185 const CoNLLULine& line = m_lines[i];
186 if (line.is_empty_line())
187 {
188 if (m_sentences.size() > 0)
189 {
190 sent_idx += 1;
191 assert(m_tokens.size() > 0);
192 m_tokens.back().m_flags = (token_t::token_flags_t)(m_tokens.back().m_flags | token_t::sentence_brk);
193 m_words[m_words.size()-1].m_flags = m_tokens[m_tokens.size()-1].m_flags;
194 }
195 continue;
196 }
197
198 if (sent_idx == m_sentences.size())
199 {
200 m_sentences.push_back(Sentence());
201 m_sentences[sent_idx].m_first_line = i;
202 m_sentences[sent_idx].m_first_word = m_words.size();
203 }
204 else
205 {
206 if ((m_sentences.size() > 1 && sent_idx < m_sentences.size() - 1)
207 || sent_idx > m_sentences.size() + 1)
208 {
209 throw std::runtime_error("rebuild_structure: wrong sent_idx.");
210 }
211 }
212
213 Sentence& sent = m_sentences[sent_idx];
214
215 if (line.is_comment_line())
216 {
217 continue;
218 }
219
220 if (sent.m_num_tokens == 0)
221 {
222 sent.m_first_token_line = i;
223 }
224
225 sent.m_num_tokens += 1; // TODO -> num_lines
226
227 // find token
228
229 if (line.idx().is_empty())
230 continue;
231
232 if (skip > 0)
233 {
234 m_words.emplace_back(word_t(i, m_tokens.size()));
235 skip -= 1;
236 if (0 == skip)
237 {
238 m_words[m_words.size()-1].m_flags = m_tokens[m_tokens.size()-1].m_flags;
239 }
240 continue;
241 }
242
243 wstring wstr = converter.from_bytes(line.form());
244 wstring::size_type size_ext = 0; // counts additional spaces in src text
245
246 if (nullptr != m_pdoc)
247 {
248 while (wcsncmp(m_pdoc->get_text().c_str() + pos, wstr.c_str(), wstr.size()) != 0)
249 {
250 if (u_isUWhiteSpace(m_pdoc->get_text()[pos]) && pos < m_pdoc->get_text().size())
251 {
252 pos += 1;
253 }
254 else
255 {
256 // there is a whitespace possible inside a "form" field
257 size_ext = compare_ws_insensitive(
258 m_pdoc->get_text().c_str() + pos,
259 wstr.c_str(), wstr.size());
260 if (size_ext != wstring::npos)
261 {
262 break;
263 }
264 else
265 {
266 throw std::runtime_error("rebuild_structure: size_ext not found.");
267 }
268 }
269 }
270 }
271
272 m_tokens.emplace_back(token_t(i, pos, wstr.size() + size_ext));
273 pos += m_tokens.back().m_len;
274
275 if (line.idx().is_multiword())
276 {
277 // Mark the surface token so segmentation training can learn to predict it.
278 m_tokens.back().m_flags =
280 skip = line.idx()._last - line.idx()._first + 1;
281 }
282 else
283 {
284 m_words.emplace_back(word_t(i, m_tokens.size()));
285 m_words[m_words.size()-1].m_flags = m_tokens[m_tokens.size()-1].m_flags;
286 }
287 }
288
289 if (!test_structure())
290 throw std::runtime_error("test_structure failed.");
291}
292
294{
295 size_t last_sent_end = 0;
296 for (const Sentence & sent : m_sentences)
297 {
298 if (sent.m_first_token_line < sent.m_first_line)
299 {
300 std::cerr << "test_structure failure: first token before first line" << sent.m_first_token_line
301 << " / " << sent.m_first_line << std::endl;
302 return false;
303 }
304 if (last_sent_end > 0)
305 {
306 if (sent.m_first_line <= last_sent_end)
307 {
308 std::cerr << "test_structure failure: first line before last sentence end"
309 << sent.m_first_line << " / " << last_sent_end << std::endl;
310 return false;
311 }
312 }
313
314 for (size_t i = sent.m_first_line; i < sent.m_first_token_line; i++)
315 {
316 if (i >= m_lines.size())
317 {
318 std::cerr << "test_structure failure at line " << i << ": " << std::endl;
319 return false;
320 }
321
322 if (!m_lines[i].is_comment_line())
323 {
324 std::cerr << "test_structure failure at line " << i << ": " << std::endl;
325 return false;
326 }
327
328 if (m_lines[i].is_empty_line() || m_lines[i].is_token_line())
329 {
330 std::cerr << "test_structure failure at line " << i << ": " << std::endl;
331 return false;
332 }
333 }
334
335 for (size_t i = sent.m_first_token_line; i < sent.m_first_token_line + sent.m_num_tokens; i++)
336 {
337 if (i >= m_lines.size())
338 {
339 std::cerr << "test_structure failure at line " << i << ": " << std::endl;
340 return false;
341 }
342
343 if (!m_lines[i].is_token_line())
344 {
345 std::cerr << "test_structure failure at line " << i << ": " << std::endl;
346 return false;
347 }
348
349 if (m_lines[i].is_empty_line() || m_lines[i].is_comment_line())
350 {
351 std::cerr << "test_structure failure at line " << i << ": " << std::endl;
352 return false;
353 }
354 }
355
356 last_sent_end = sent.m_first_line + sent.m_num_tokens;
357 }
358
359 return true;
360}
361
362bool parse_ud_file_name(const std::string& fn, map<string, string>& fields)
363{
364 smatch sm;
365 if (regex_match(fn, sm, regex("(\\w+)_(\\w+)-(\\w+)-(\\w+)\\.conllu")))
366 {
367 if (sm.size() == 5)
368 {
369 fields["lang"] = sm[1];
370 fields["corpus"] = sm[2];
371 fields["format"] = sm[3];
372 fields["part"] = sm[4];
373 return true;
374 }
375 }
376 return false;
377}
378
379
380const Sentence& Annotation::get_sentence(size_t idx) const
381{
382 return m_sentences[idx];
383}
384
385bool idx_t::parse(const std::string& s)
386{
387 _sub = 0;
388
389 size_t dash_pos = s.find('-');
390 if (string::npos != dash_pos)
391 {
392 _first = stoi(s);
393 _last = stoi(s.substr(dash_pos + 1));
394 return true;
395 }
396
397 size_t dot_pos = s.find('.');
398 if (string::npos != dot_pos)
399 {
400 _first = _last = stoi(s);
401 _sub = stoi(s.substr(dot_pos + 1));
402 return true;
403 }
404
405 size_t i = 0;
406 size_t value = stoi(s, &i);
407 if (value > std::numeric_limits<base_int_t>::max())
408 {
409 throw std::overflow_error("Id too big for word \"" + s + "\"");
410 }
411 _first = _last = value;
412 if (i < s.size())
413 {
414 throw logic_error("Wrong format of ID field: \"" + s + "\"");
415 }
416
417 return true;
418}
419
420string idx_t::serialize() const
421{
422 string rv = to_string(_first);
423
424 if (is_multiword())
425 rv += "-" + to_string(_last);
426
427 if (is_empty())
428 rv += "." + to_string(_sub);
429
430 return rv;
431}
432
433} // namespace CoNLLU
434} // namespace deeplima
435
std::vector< CoNLLULine > m_lines
Definition treebank.h:194
std::vector< token_t > m_tokens
Definition treebank.h:197
void load(const std::string &fn)
Definition treebank.cpp:91
void rebuild_structure(size_t num_sentences, size_t num_tokens, size_t num_words)
Definition treebank.cpp:171
const CoNLLULine & get_line(size_t line_idx) const
Definition treebank.h:184
std::vector< Sentence > m_sentences
Definition treebank.h:199
const Sentence & get_sentence(size_t idx) const
Definition treebank.cpp:380
std::vector< word_t > m_words
Definition treebank.h:198
bool is_empty_line() const
Definition line.h:138
const std::string & form() const
Definition line.h:153
bool is_comment_line() const
Definition line.h:133
const idx_t & idx() const
Definition line.h:148
bool is_real_word_line() const
Definition line.h:143
bool is_token_line() const
Definition line.h:128
std::string m_original_text
Definition treebank.h:79
void load(const std::string &fn)
Definition treebank.cpp:69
const std::wstring & get_text() const
Definition treebank.h:67
size_t calc_num_of_words(const Annotation &annot) const
Definition treebank.cpp:25
std::map< std::string, AnnotatedDocument > m_parts
Definition treebank.h:502
bool load(const std::string &path)
Definition treebank.cpp:43
bool parse_ud_file_name(const std::string &fn, map< string, string > &fields)
Definition treebank.cpp:362
wstring::size_type compare_ws_insensitive(const wchar_t *s1, const wchar_t *s2, size_t len)
Definition treebank.cpp:140
STL namespace.
base_int_t _first
Definition line.h:30
bool parse(const std::string &s)
Definition treebank.cpp:385
bool is_multiword() const
Definition line.h:48
bool is_empty() const
Definition line.h:53
base_int_t _sub
Definition line.h:33
std::string serialize() const
Definition treebank.cpp:420
base_int_t _last
Definition line.h:31