LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
dumper_conllu.h
Go to the documentation of this file.
1// Copyright 2021 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6#ifndef DEEPLIMA_DUMPER_CONLLU_H
7#define DEEPLIMA_DUMPER_CONLLU_H
8
9#include <iostream>
10
11// #include "deeplima/segmentation/impl/segmentation_impl.h"
12
13#include "deeplima/token_type.h"
14
15namespace deeplima
16{
17namespace dumper
18{
19
21{
22 ConllToken() = default;
23 ConllToken(const ConllToken&) = default;
24 ConllToken& operator=(const ConllToken&) = default;
25 ~ConllToken() = default;
26
27 // ID: Word index, integer starting at 1 for each new sentence; may be a range for multiword tokens; may be a decimal number for empty nodes (decimal numbers can be lower than 1 but must be greater than 0).
28 uint32_t id = 0;
29 // FORM: Word form or punctuation symbol.
30 std::string form = "_";
31 // LEMMA: Lemma or stem of word form.
32 std::string lemma = "_";
33 // UPOS: Universal part-of-speech tag.
34 std::string upos = "_";
35 // XPOS: Language-specific part-of-speech tag; underscore if not available.
36 std::string xpos = "_";
37 // FEATS: List of morphological features from the universal feature inventory or from a defined language-specific extension; underscore if not available.
38 std::string feats = "_";
39 // HEAD: Head of the current word, which is either a value of ID or zero (0).
40 int head = 0;
41 // DEPREL: Universal dependency relation to the HEAD (root iff HEAD = 0) or a defined language-specific subtype of one.
42 std::string deprel = "dep";
43 // DEPS: Enhanced dependency graph in the form of a list of head-deprel pairs.
44 std::string deps = "_";
45 // MISC: Any other annotation.
46 std::string misc = "_";
47
48 // Multiword-token range. When mwt_last > 0, this token is the first sub-word
49 // of an expanded multiword token, and a UD "id-mwt_last mwt_surface" range
50 // line is emitted just before this token's own line.
51 uint32_t mwt_last = 0;
52 std::string mwt_surface;
53};
54
55std::ostream& operator<<(std::ostream& oss, const ConllToken& token)
56{
57 if (token.mwt_last > 0)
58 {
59 // UD multiword-token range line: "N-M\tsurface\t_\t_\t_\t_\t_\t_\t_\t_"
60 oss << token.id << "-" << token.mwt_last << "\t"
61 << token.mwt_surface
62 << "\t_\t_\t_\t_\t_\t_\t_\t_" << std::endl;
63 }
64 oss << token.id << "\t"
65 << token.form << "\t"
66 << token.lemma << "\t"
67 << token.upos << "\t"
68 << token.xpos << "\t"
69 << token.feats << "\t"
70 << token.head << "\t"
71 << token.deprel << "\t"
72 << token.deps << "\t"
73 << token.misc << std::endl << std::flush;
74 return oss;
75}
76
77
78template <typename T>
79std::ostream& operator<< (std::ostream& out, const std::vector<T>& v)
80{
81 out << '[';
82 if ( !v.empty() )
83 {
84 std::copy (v.begin(), v.end(), std::ostream_iterator<T>(out, ", "));
85 }
86 out << "]";
87 return out;
88}
89
90bool dfs(int v, std::vector<uint32_t>& heads, std::vector<int>& color,
91 uint32_t& cycle_start, uint32_t& cycle_end)
92{
93 // std::cerr << "dfs " << v << ", " << heads << ", " << color << ", " << cycle_start << ", " << cycle_end << std::endl;
94 color[v] = 1;
95 auto u = heads[v];
96 // A predicted head can be out of range (>= number of tokens) when the model
97 // produces a bad arc; treat it like "no head" so the cycle walk never
98 // indexes color[]/heads[] out of bounds.
99 if (u == 0 || u >= heads.size())
100 {
101 color[v] = 2;
102 return false;
103 }
104 if (color[u] == 0) {
105 // parent[u] = v;
106 if (dfs(u, heads, color, cycle_start, cycle_end))
107 return true;
108 } else if (color[u] == 1) {
109 cycle_end = v;
110 cycle_start = u;
111 return true;
112 }
113
114 color[v] = 2;
115 return false;
116}
117
118bool find_cycle(std::vector<uint32_t>& heads, uint32_t root)
119{
120 // std::cerr << "find_cycle " << heads << ", " << root << std::endl;
121 uint32_t n = heads.size();
122 std::vector<int> color;
123 uint32_t cycle_start = std::numeric_limits<uint32_t>::max();
124 uint32_t cycle_end = 0;
125 color.assign(n, 0);
126
127 for (uint32_t v = 1; v < n; v++)
128 {
129 if (v == root) continue;
130 if (color[v] == 0 && dfs(v, heads, color, cycle_start, cycle_end))
131 break;
132 }
133
134 if (cycle_start == std::numeric_limits<uint32_t>::max() || cycle_start == root)
135 {
136 // std::cerr << "Acyclic" << std::endl;
137 return false;
138 }
139 else
140 {
141 // if (cycle_start > 0 && cycle_start != root)
142 // heads[cycle_start] = root;
143 if (cycle_end > 0 && cycle_end != root && cycle_end < n)
144 {
145 // std::cerr << "Cycle found from " << cycle_start << ", " << cycle_end << " in " << heads << " with root: " << root << std::endl;
146 heads[cycle_end] = root;
147 return true;
148 }
149 else
150 {
151 // std::cerr << "Cycle not found from " << cycle_start << ", " << cycle_end << " in " << heads << " with root: " << root << std::endl;
152 return false;
153 }
154 }
155}
156
158{
159protected:
161
163 {
165 }
166
167public:
168 virtual void operator()(const std::vector<deeplima::segmentation::token_pos>& tokens, uint32_t len) = 0;
169
170 uint64_t get_token_counter() const
171 {
172 return m_token_counter;
173 }
174
177
178 virtual ~AbstractDumper() { }
179
180 void reset()
181 {
182 m_token_counter = 0;
183 }
184};
185
187{
188public:
190 {}
191
192 virtual void operator()(const std::vector<deeplima::segmentation::token_pos>& tokens, uint32_t len)
193 {
194 std::string temp;
195 for (size_t i = 0; i < len; i++)
196 {
197 const char* ptoken = tokens[i].m_pch;
198 std::ostringstream s;
199 if (temp.size() > 0)
200 {
201 s << temp;
202 temp.clear();
203 }
204
205 for (size_t j = 0; j < tokens[i].m_len; j++)
206 {
207 if (*ptoken == '\r' || *ptoken == '\n' || *ptoken == '\t')
208 {
209 s << " ";
210 }
211 else
212 {
213 s << *ptoken;
214 }
215 ptoken++;
216 }
217 std::string str = s.str();
218 if (std::string::npos == str.find_first_not_of(' '))
219 {
220 temp = str;
221 continue;
222 }
223 std::cout << str << " ";
224
225 if (tokens[i].m_flags & token_flags_t::sentence_brk ||
226 tokens[i].m_flags & token_flags_t::paragraph_brk)
227 {
228 // std::cerr << "Horizontal endl" << std::endl;
229 std::cout << std::endl;
230 }
231 }
232 }
233};
234
236{
237public:
241
243 {
244 if (m_next_token_idx > 1)
245 {
246 // std::cerr << "on destructor" << std::endl;
247 std::cout << std::endl;
248 }
249 }
250
251 virtual void operator()(const std::vector<deeplima::segmentation::token_pos>& tokens, uint32_t len)
252 {
253 std::string temp;
254 for (size_t i = 0; i < len; i++)
255 {
256 const char* ptoken = tokens[i].m_pch;
257 std::ostringstream s;
258 if (temp.size() > 0)
259 {
260 s << temp;
261 temp.clear();
262 }
263 for (size_t j = 0; j < tokens[i].m_len; j++)
264 {
265 if (*ptoken == '\r' || *ptoken == '\n' || *ptoken == '\t')
266 {
267 s << " ";
268 }
269 else
270 {
271 s << *ptoken;
272 }
273 ptoken++;
274 }
275 std::string str = s.str();
276 if (std::string::npos == str.find_first_not_of(' '))
277 {
278 temp = str;
279 continue;
280 }
281 // First sub-word of an expanded multiword token: emit the UD range line
282 // "N-M\tsurface\t_..." just before this sub-word's own line, so the
283 // tokenization-only output carries MWT (e.g. "28-29 au" then "à"/"le").
284 if (tokens[i].m_mwt_len > 0 && nullptr != tokens[i].m_mwt_surface_pch)
285 {
286 std::string surface;
287 const char* psurf = tokens[i].m_mwt_surface_pch;
288 for (size_t j = 0; j < tokens[i].m_mwt_surface_len; j++)
289 {
290 surface += (*psurf == '\r' || *psurf == '\n' || *psurf == '\t') ? ' ' : *psurf;
291 psurf++;
292 }
293 std::cout << m_next_token_idx << "-"
294 << (m_next_token_idx + tokens[i].m_mwt_len - 1) << "\t"
295 << surface << "\t_\t_\t_\t_\t_\t_\t_\t_" << std::endl;
296 }
297 std::cout << m_next_token_idx << "\t";
298 std::cout << str;
299 std::cout << "\t_\t_\t_\t_\t";
300 std::cout << m_next_token_idx - 1;
301 // std::cerr << "TokensToConllU::operator end of token" << std::endl;
302 std::cout << "\t_\t_\t_" << std::endl;
303
305
306 m_next_token_idx += 1;
307 if (tokens[i].m_flags & token_flags_t::sentence_brk ||
308 tokens[i].m_flags & token_flags_t::paragraph_brk)
309 {
310 // std::cerr << "TokensToConllU end of sentence" << std::endl;
311 std::cout << std::endl;
313 }
314 }
315 }
316
317protected:
319};
320
322{
323public:
324 virtual ~DumperBase() = default;
325 virtual uint64_t get_token_counter() const = 0;
326 virtual void flush() = 0;
327 virtual void reset() = 0;
328};
329
330template <class I>
332{
333protected:
336 std::vector<ConllToken> m_tokens;
337 uint32_t m_root;
338
339 void reset()
340 {
341 m_token_counter = 0;
342 }
343
345 {
347 }
348
349 std::vector<std::string> m_class_names;
350 std::vector<std::vector<std::string>> m_classes;
352
354
355public:
357 : m_token_counter(0),
359 m_root(0),
360 m_has_feats(false),
362 {
363 // std::cerr << "AnalysisToConllU()" << (void*)this << std::endl;
364 }
365
367 {
368 // std::cerr << "~AnalysisToConllU " << (void*)this << std::endl;
369 // if (m_next_token_idx > 1)
370 // {
371 // std::cout << std::endl;
372 // }
373 }
374
375 const std::vector<std::vector<std::string>> &getMClasses() const {
376 return m_classes;
377 }
378
379 virtual uint64_t get_token_counter() const
380 {
381 return m_token_counter;
382 }
383
384 void set_classes(size_t idx, const std::string& class_name, const std::vector<std::string>& data)
385 {
386 m_class_names.push_back(class_name);
387
388 if (idx + 1 > m_classes.size())
389 {
390 m_classes.resize(idx + 1);
391 }
392 assert(0 == m_classes[idx].size());
393 m_classes[idx] = data;
394
395 if (m_classes.size() > 1)
396 {
397 m_has_feats = true;
398 for (size_t i = 0; i < m_class_names.size(); ++i)
399 {
400 const std::string& feat_name = m_class_names[i];
401 if (feat_name == "upos" || feat_name == "xpos" || feat_name == "eos")
402 {
403 continue;
404 }
406 break;
407 }
408 }
409 }
410
411 std::string generate_feats(const I& iter)
412 {
413 std::string feat_str;
414
415 for (size_t i = m_first_feature_to_print; i < m_classes.size(); ++i)
416 {
417 if (0 != iter.token_class(i))
418 {
419 if (!feat_str.empty())
420 {
421 feat_str += "|";
422 }
423
424 feat_str += m_class_names[i];
425 feat_str += "=";
426 feat_str += m_classes[i][iter.token_class(i)];
427 }
428 }
429
430 return feat_str;
431 }
432
433 virtual void flush()
434 {
435 // Nothing pending: emit no separator. flush() is also called once at EOF
436 // (apps/deeplima.cpp) after the last sentence already flushed on its
437 // sentence break, so without this guard that final call would append a
438 // stray blank line, leaving a doubled blank at end-of-file that strict
439 // CoNLL-U scorers read as a trailing empty sentence. A genuinely pending
440 // last sentence with no sentence-break flag still has tokens here and is
441 // emitted normally.
442 if (m_tokens.empty())
443 {
444 return;
445 }
447 std::vector<uint32_t> heads(m_tokens.size()+1);
448 heads[0] = 0;
449 for (size_t i = 1; i < heads.size(); i++)
450 {
451 heads[i] = m_tokens[i-1].head;
452 }
453 // std::cerr << "AnalysisToConllU::flush() heads before find_cycle: " << heads << std::endl;
454 while (find_cycle(heads, m_root))
455 {
456 // std::cerr << "AnalysisToConllU::flush() heads after cycle found: " << heads << std::endl;
457 }
458 // std::cerr << "AnalysisToConllU::flush() heads after no more cycle: " << heads << std::endl;
459 for (size_t i = 1; i < heads.size(); i++)
460 {
461 m_tokens[i-1].head = heads[i];
462 }
463 for (const auto& token: m_tokens)
464 {
465 // std::cerr << "AnalysisToConllU::flush():567 " << token ;
466 std::cout << token ;
467 }
468 m_tokens.clear();
469 // std::cerr << "after clearing tokens. m_next_token_idx=" << m_next_token_idx << std::endl;
470 std::cout << std::endl;
471 m_root = 0;
472 }
473
474 void operator()(I& iter, uint32_t begin, uint32_t end, bool hasDeps = false)
475 {
476 // std::cerr << "AnalysisToConllU::operator() " << iter.form() << ", " << begin << ", " << end << ", "
477 // << m_next_token_idx << ", " << m_tokens << std::endl;
478 m_tokens.reserve(end);
479 // Inter-sentence blank lines are emitted by flush() only. When the previous
480 // buffer ended exactly on a sentence break, flush() already wrote the
481 // separator and reset m_next_token_idx to 1 and m_root to 0; emitting
482 // another std::endl here produced *doubled* blank lines between sentences at
483 // buffer boundaries, which strict CoNLL-U scorers (conll18) read as empty
484 // sentences. So only initialize on the very first call (m_next_token_idx==0).
485 if (m_next_token_idx == 0)
486 {
488 }
489 std::string temp;
490 while (!iter.end())
491 {
492 const char* ptoken = iter.form();
493 if (std::string(ptoken) == "<ROOT>")
494 {
495 iter.next();
496 continue;
497 }
498 ConllToken token;
499 std::ostringstream s;
500 if (temp.size() > 0)
501 {
502 s << temp;
503 temp.clear();
504 }
505 while (0 != *ptoken)
506 {
507 if (*ptoken == '\r' || *ptoken == '\n' || *ptoken == '\t')
508 {
509 s << " ";
510 }
511 else
512 {
513 s << *ptoken;
514 }
515 ptoken++;
516 }
517 std::string str = s.str();
518 if (std::string::npos == str.find_first_not_of(' '))
519 {
520 temp = str;
521 continue;
522 }
523 // std::cerr << m_next_token_idx << "\t" << str << "\t" << iter.lemma()
524 // << "\t" << m_classes[0][iter.token_class(0)] << std::endl;
525 token.id = m_next_token_idx;
526 token.form = str;
527 token.lemma = iter.lemma();
528 token.upos = m_classes[0][iter.token_class(0)];
529 // First sub-word of an expanded multiword token: record the "N-M surface"
530 // range to emit before this token's line (ids N..M are the sub-words).
531 if (iter.mwt_len() > 0)
532 {
533 token.mwt_last = m_next_token_idx + iter.mwt_len() - 1;
534 token.mwt_surface = iter.mwt_surface();
535 }
536 // std::cout << m_next_token_idx << "\t";
537 // std::cout << str << "\t";
538 // std::cout << iter.lemma();
539 // if (true)
540 // {
541 // std::cout << "\t" << m_classes[0][iter.token_class(0)];
542 // }
543 // else
544 // {
545 // std::cout << "\t_";
546 // }
547
548 // std::cerr << "\t_\t";
549 // std::cout << "\t_\t";
550
551 if (m_has_feats)
552 {
553 // std::cerr << generate_feats(iter);
554 // std::cout << generate_feats(iter);
555 token.feats = generate_feats(iter);
556 }
557 else
558 {
559 // std::cerr << "_";
560 // std::cout << "_";
561 }
562
563 // std::cerr << "\t";
564 // std::cout << "\t";
565 if (hasDeps)
566 {
567 if ((m_root == begin) && (iter.head() == 0))
568 {
570 token.head = 0;
571 token.deprel = "root";
572 }
573 else if (iter.head() >= end)
574 {
575 // std::cerr << "head is out of sentence. rehead to root or set it as root if no root is set " << m_root << "\tdep"<< std::endl;
576 if (m_root == 0)
577 {
578 token.head = 0;
579 token.deprel = "root";
581 }
582 else
583 {
584 // std::cout << root << "\tdep";
585 token.head = m_root;
586 token.deprel = "dep";
587 }
588 }
589 else if ((m_root != begin) && (iter.head() == 0))
590 {
591 // std::cerr << "multiple roots in the sentence. rehead to the first root" << std::endl;
592 // std::cerr << m_root << "\tdep";
593 // std::cout << root << "\tdep";
594 token.head = m_root;
595 token.deprel = "dep";
596 }
597 else if (iter.head() == m_next_token_idx)
598 {
599 token.head = m_root;
600 token.deprel = "dep";
601 }
602 else
603 {
604 // std::cerr << iter.head() << "\tdep";
605 // std::cout << iter.head() << "\tdep";
606 token.head = iter.head();
607 // Use the deprel predicted by the label decoder when available
608 // (falls back to "dep" if the model has no label decoder).
609 token.deprel = iter.deprel();
610 }
611 }
612 else
613 {
614 // std::cerr << m_next_token_idx - 1 << "\t_";
615 // std::cout << m_next_token_idx - 1 << "\t_";
616 token.head = m_next_token_idx - 1;
617 if (token.head == 0)
618 token.deprel = "root";
619 }
620 // std::cerr << "\t_\t_" << std::endl;
621 // std::cout << "\t_\t_" << std::endl;
622 m_tokens.push_back(token);
624
625 m_next_token_idx += 1;
626 if (iter.flags() & token_flags_t::sentence_brk ||
627 iter.flags() & token_flags_t::paragraph_brk)
628 {
629 // std::cerr << "AnalysisToConllU::operator() on sent/para break. m_next_token_idx="
630 // << m_next_token_idx << std::endl;
631 flush();
632 }
633 iter.next();
634 }
635 }
636};
637
638} // namespace dumper
639} // namespace deeplima
640
641#endif
virtual void operator()(const std::vector< deeplima::segmentation::token_pos > &tokens, uint32_t len)=0
void set_classes(size_t idx, const std::string &class_name, const std::vector< std::string > &data)
const std::vector< std::vector< std::string > > & getMClasses() const
std::string generate_feats(const I &iter)
std::vector< ConllToken > m_tokens
std::vector< std::string > m_class_names
std::vector< std::vector< std::string > > m_classes
void operator()(I &iter, uint32_t begin, uint32_t end, bool hasDeps=false)
virtual uint64_t get_token_counter() const
virtual uint64_t get_token_counter() const =0
virtual ~DumperBase()=default
virtual void operator()(const std::vector< deeplima::segmentation::token_pos > &tokens, uint32_t len)
virtual void operator()(const std::vector< deeplima::segmentation::token_pos > &tokens, uint32_t len)
bool dfs(int v, std::vector< uint32_t > &heads, std::vector< int > &color, uint32_t &cycle_start, uint32_t &cycle_end)
std::ostream & operator<<(std::ostream &oss, const ConllToken &token)
bool find_cycle(std::vector< uint32_t > &heads, uint32_t root)
@ paragraph_brk
Definition token_type.h:22
@ sentence_brk
Definition token_type.h:21
ConllToken(const ConllToken &)=default
ConllToken & operator=(const ConllToken &)=default