11#include <boost/filesystem.hpp>
13#include <unicode/uchar.h>
18namespace fs = boost::filesystem;
45 for (fs::directory_entry& entry : fs::directory_iterator(path))
47 if (entry.path().extension().string() ==
".conllu")
49 map<string, string> fields;
51 || fields.end() == fields.find(
"part")
52 || fields[
"part"].size() == 0)
54 throw std::logic_error(
"Can't parse file name \"" + entry.path().filename().string() +
"\"");
56 m_parts[fields[
"part"]] = AnnotatedDocument();
58 auto plain_text_fn = entry.path();
59 plain_text_fn.replace_extension(
"txt");
60 m_parts[fields[
"part"]].doc.load(plain_text_fn.string());
61 m_parts[fields[
"part"]].annot.m_pdoc = &
m_parts[fields[
"part"]].doc;
62 m_parts[fields[
"part"]].annot.load(entry.path().string());
74 throw std::runtime_error(
75 string(
"Document::load: companion plain-text file \"") + fn
76 +
"\" is missing (it is needed for segmentation training; generate it from "
77 "the .conllu '# text =' lines, e.g. with make_trainable.py).");
79 m_original_text = std::string((istreambuf_iterator<char>(input)), istreambuf_iterator<char>());
82 throw std::runtime_error(
83 string(
"Document::load: companion plain-text file \"") + fn
84 +
"\" is empty (it is needed for segmentation training; generate it from "
85 "the .conllu '# text =' lines, e.g. with make_trainable.py).");
87 std::wstring_convert<std::codecvt_utf8<wchar_t>> converter;
96 throw invalid_argument(
string(
"Can't open file \"") + fn +
"\"");
103 catch (
const std::runtime_error& e)
105 std::cerr <<
"Annotation::load runtime_error while loading "
106 << fn <<
" :" << std::endl << e.what() << std::endl;
114 size_t num_sentences = 0;
115 size_t num_tokens = 0;
116 size_t num_words = 0;
117 while (getline(input, line))
142 const wchar_t *p1 = s1, *p2 = s2;
147 if (*p1 == *p2 || (u_isUWhiteSpace(*p1) && u_isUWhiteSpace(*p2)))
153 if (u_isUWhiteSpace(*p1))
157 else if (u_isUWhiteSpace(*p2))
163 return wstring::npos;
168 return p1 - s1 - len;
173 wstring_convert<codecvt_utf8<wchar_t>> converter;
183 for (
size_t i = 0; i <
m_lines.size(); i++)
209 throw std::runtime_error(
"rebuild_structure: wrong sent_idx.");
243 wstring wstr = converter.from_bytes(line.
form());
244 wstring::size_type size_ext = 0;
248 while (wcsncmp(
m_pdoc->
get_text().c_str() + pos, wstr.c_str(), wstr.size()) != 0)
250 if (u_isUWhiteSpace(
m_pdoc->
get_text()[pos]) && pos < m_pdoc->get_text().size())
259 wstr.c_str(), wstr.size());
260 if (size_ext != wstring::npos)
266 throw std::runtime_error(
"rebuild_structure: size_ext not found.");
290 throw std::runtime_error(
"test_structure failed.");
295 size_t last_sent_end = 0;
298 if (sent.m_first_token_line < sent.m_first_line)
300 std::cerr <<
"test_structure failure: first token before first line" << sent.m_first_token_line
301 <<
" / " << sent.m_first_line << std::endl;
304 if (last_sent_end > 0)
306 if (sent.m_first_line <= last_sent_end)
308 std::cerr <<
"test_structure failure: first line before last sentence end"
309 << sent.m_first_line <<
" / " << last_sent_end << std::endl;
314 for (
size_t i = sent.m_first_line; i < sent.m_first_token_line; i++)
318 std::cerr <<
"test_structure failure at line " << i <<
": " << std::endl;
322 if (!
m_lines[i].is_comment_line())
324 std::cerr <<
"test_structure failure at line " << i <<
": " << std::endl;
330 std::cerr <<
"test_structure failure at line " << i <<
": " << std::endl;
335 for (
size_t i = sent.m_first_token_line; i < sent.m_first_token_line + sent.m_num_tokens; i++)
339 std::cerr <<
"test_structure failure at line " << i <<
": " << std::endl;
343 if (!
m_lines[i].is_token_line())
345 std::cerr <<
"test_structure failure at line " << i <<
": " << std::endl;
351 std::cerr <<
"test_structure failure at line " << i <<
": " << std::endl;
356 last_sent_end = sent.m_first_line + sent.m_num_tokens;
365 if (regex_match(fn, sm, regex(
"(\\w+)_(\\w+)-(\\w+)-(\\w+)\\.conllu")))
369 fields[
"lang"] = sm[1];
370 fields[
"corpus"] = sm[2];
371 fields[
"format"] = sm[3];
372 fields[
"part"] = sm[4];
389 size_t dash_pos = s.find(
'-');
390 if (string::npos != dash_pos)
393 _last = stoi(s.substr(dash_pos + 1));
397 size_t dot_pos = s.find(
'.');
398 if (string::npos != dot_pos)
401 _sub = stoi(s.substr(dot_pos + 1));
406 size_t value = stoi(s, &i);
407 if (value > std::numeric_limits<base_int_t>::max())
409 throw std::overflow_error(
"Id too big for word \"" + s +
"\"");
414 throw logic_error(
"Wrong format of ID field: \"" + s +
"\"");
422 string rv = to_string(
_first);
425 rv +=
"-" + to_string(
_last);
428 rv +=
"." + to_string(
_sub);
std::vector< CoNLLULine > m_lines
std::vector< token_t > m_tokens
void load(const std::string &fn)
bool test_structure() const
void rebuild_structure(size_t num_sentences, size_t num_tokens, size_t num_words)
const CoNLLULine & get_line(size_t line_idx) const
std::vector< Sentence > m_sentences
const Sentence & get_sentence(size_t idx) const
std::vector< word_t > m_words
bool is_empty_line() const
const std::string & form() const
bool is_comment_line() const
const idx_t & idx() const
bool is_real_word_line() const
bool is_token_line() const
std::string m_original_text
void load(const std::string &fn)
const std::wstring & get_text() const
size_t m_first_token_line
size_t calc_num_of_words(const Annotation &annot) const
std::map< std::string, AnnotatedDocument > m_parts
bool load(const std::string &path)
bool parse_ud_file_name(const std::string &fn, map< string, string > &fields)
wstring::size_type compare_ws_insensitive(const wchar_t *s1, const wchar_t *s2, size_t len)
bool parse(const std::string &s)
bool is_multiword() const
std::string serialize() const