LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
mwt_dict.h
Go to the documentation of this file.
1// Copyright 2021 CEA LIST
2// SPDX-FileCopyrightText: 2026 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6#ifndef DEEPLIMA_CONLLU_MWT_DICT_H
7#define DEEPLIMA_CONLLU_MWT_DICT_H
8
9#include <ostream>
10#include <string>
11#include <vector>
12
13namespace deeplima
14{
15namespace CoNLLU
16{
17
18// Extract a multiword-token (MWT) expansion dictionary from CoNLL-U files.
19//
20// For every "N-M" range line, the surface form maps to the forms of the
21// following (M-N+1) real-word lines (e.g. French "du" -> ["de", "le"]).
22// Surfaces seen with several distinct expansions are resolved by frequency
23// (the most frequent expansion wins); discarded alternatives are logged to
24// std::cerr. The output has one line per surface:
25//
26// surface <TAB> count <TAB> word1 <TAB> word2 [<TAB> ...]
27//
28// where `count` is the frequency of the chosen expansion (usable downstream to
29// drop rare/noisy entries). Returns the number of distinct surfaces written.
30size_t extract_mwt_dict(const std::vector<std::string>& conllu_files, std::ostream& out);
31
32} // namespace CoNLLU
33} // namespace deeplima
34
35#endif
size_t extract_mwt_dict(const vector< string > &conllu_files, ostream &out)
Definition mwt_dict.cpp:22