LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
ConlluReader.cpp
Go to the documentation of this file.
1// Copyright 2002-2020 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6#include <QtCore/QTemporaryFile>
7#include <QtCore/QRegularExpression>
8#include <QDir>
9#include <QJsonDocument>
10
18
25
26#include "ConlluReader.h"
27
28
29#define DEBUG_THIS_FILE true
30
31using namespace std;
33using namespace Lima::Common::PropertyCode;
34using namespace Lima::Common::MediaticData;
35using namespace Lima::Common::Misc;
37
38
39namespace Lima
40{
41namespace LinguisticProcessing
42{
43namespace ConlluReader
44{
45
46namespace
47{
48inline string THIS_FILE_LOGGING_CATEGORY()
49{
51 return logger.zone().toStdString();
52}
53}
54
55static u32string SPACE = QString::fromUtf8(" ").toStdU32String();
56
58
59#define LOG_ERROR_AND_THROW(msg, exc) { \
60 TOKENIZERLOGINIT; \
61 LERROR << msg; \
62 throw exc; \
63 }
64
65#if defined(DEBUG_LP) && defined(DEBUG_THIS_FILE)
66 #define LOG_MESSAGE(stream, msg) stream << msg;
67 #define LOG_MESSAGE_WITH_PROLOG(stream, msg) TOKENIZERLOGINIT; LOG_MESSAGE(stream, msg);
68#else
69 #define LOG_MESSAGE(stream, msg) ;
70 #define LOG_MESSAGE_WITH_PROLOG(stream, msg) ;
71#endif
72
74{
75public:
77 virtual ~ConlluReaderPrivate();
78
80 {
82 TPrimitiveToken(const QString& w,
83 int pos,
84 const QString& orig=QString())
85 : wordText(w), originalText(orig), start(pos)
86 { }
87
88 QString wordText;
89 QString originalText;
90 int start;
91 };
92
93 void init(MediaId language, GroupConfigurationStructure& unitConfiguration);
94 vector< vector< TPrimitiveToken > > tokenize(const QString& text);
95 void computeDefaultStatus(Token& token);
96
97 MediaId m_language;
100 std::string m_data;
102
103protected:
104 void append_new_word(vector< TPrimitiveToken >& current_sentence,
105 const u32string& current_token,
106 int current_token_offset) const;
107
108 // Parameters
110};
111
113 ConfigurationHelper("ConlluReader", THIS_FILE_LOGGING_CATEGORY()),
114 m_stringsPool(nullptr),
115 m_currentVx(0),
116 m_ignoreEOL(false)
117{
118}
119
123
128
130{
131 delete m_d;
132}
133
135 Manager* manager)
136
137{
138 LOG_MESSAGE_WITH_PROLOG(LDEBUG, "ConlluReader::init");
139 m_d->init(manager->getInitializationParameters().media, unitConfiguration);
140}
141
143{
144 LOG_MESSAGE_WITH_PROLOG(LINFO, "start ConlluReader");
145 TimeUtilsController ConlluReaderProcessTime("ConlluReader");
146
147 auto anagraph = new AnalysisGraph("AnalysisGraph",m_d->m_language,true,true);
148 analysis.setData("AnalysisGraph",anagraph);
149 LinguisticGraph* graph=anagraph->getGraph();
150 m_d->m_currentVx = anagraph->firstVertex();
151 // Get text from analysis
152 auto originalText = std::dynamic_pointer_cast<LimaStringText>(analysis.getData("Text"));
153
154 // Evaluate TensorFlow model on the text
155 auto sentencesTokens = m_d->tokenize(*originalText);
156
157 // Insert the tokens in the graph and create sentence limits
158 SegmentationData* sb = nullptr;
159 if (m_d->m_data.size() > 0)
160 {
161 sb = new SegmentationData("AnalysisGraph");
162 analysis.setData(m_d->m_data, sb);
163 }
164
165 auto microAccessor = &(GET_PROPERTY_ACCESSOR(m_d->m_language, "MICRO"));
166
167 remove_edge(anagraph->firstVertex(),
168 anagraph->lastVertex(),
169 *graph);
170 LinguisticGraphVertex beginSentence = anagraph->firstVertex();
171
172 // Insert the tokens in the graph and create sentence limits
173 for (const auto& sentence: sentencesTokens)
174 {
175 if (sentence.size() < 1)
176 continue;
177
178 LinguisticGraphVertex endSentence = anagraph->lastVertex();
179 for (const auto& token: sentence)
180 {
181 const auto& str = token.wordText;
182
183 LOG_MESSAGE(LDEBUG, " Adding token '" << str << "'");
184
185 StringsPoolIndex form=(*m_d->m_stringsPool)[str];
186 Token *tToken = new Token(form, str, token.start, token.wordText.size());
187 if (tToken == 0)
188 LOG_ERROR_AND_THROW("ConlluReader::process: Can't allocate memory with \"new Token(...)\"",
190
191 /*if (token.originalText.size() > 0)
192 {
193 // tranduced token
194 // save original word as orph alternative
195 StringsPoolIndex orig = (*m_d->m_stringsPool)[token.originalText];
196 tToken->addOrthographicAlternatives(orig);
197 }*/
198
199 m_d->computeDefaultStatus(*tToken);
200
201 LOG_MESSAGE(LDEBUG, " status is " << tToken->status().toString());
202
203 // Adds on the path
204 LinguisticGraphVertex newVx = add_vertex(*graph);
205 endSentence = newVx;
206 put(vertex_token, *graph, newVx, tToken);
207 put(vertex_data, *graph, newVx, new MorphoSyntacticData());
208 add_edge(m_d->m_currentVx, newVx, *graph);
209 m_d->m_currentVx = newVx;
210 }
211
212 LOG_MESSAGE(LDEBUG, "adding sentence" << beginSentence << endSentence);
213
214 if (nullptr != sb)
215 {
216 sb->add(Segment("sentence", beginSentence, endSentence, anagraph));
217 }
218
219 Token *tToken = get(vertex_token, *graph, endSentence);
220 MorphoSyntacticData *morphoData = get(vertex_data, *graph, endSentence);
221 if (nullptr != morphoData)
222 {
224 elem.inflectedForm = tToken->form();
225 elem.lemma = tToken->form();
226 elem.normalizedForm = tToken->form();
227 elem.type = SIMPLE_WORD;
228 microAccessor->writeValue(m_d->m_boundaryMicro, elem.properties);
229 morphoData->push_back(elem);
230 }
231 else
232 {
233 LOG_MESSAGE(LERROR, "vertex" << endSentence << "has no MorphoSyntacticData");
234 }
235 beginSentence = endSentence;
236 }
237
238 add_edge(m_d->m_currentVx,anagraph->lastVertex(),*graph);
239
240 return SUCCESS_ID;
241}
242
243void ConlluReaderPrivate::init(MediaId language, GroupConfigurationStructure& unitConfiguration)
244{
245 m_language = language;
246 m_stringsPool = &MediaticData::changeable().stringsPool(m_language);
247
248 m_data = getStringParameter(unitConfiguration, "data", 0, "SentenceBoundaries");
249
250 try
251 {
252 string boundaryMicro = getStringParameter(unitConfiguration, "boundaryMicro", 0, "PONCTU_FORTE");
253
254 const auto& microManager = GET_PROPERTY_MANAGER(m_language, "MICRO");
255 m_boundaryMicro = microManager.getPropertyValue(boundaryMicro);
256 if (m_boundaryMicro == L_NONE)
257 {
258 LOG_ERROR_AND_THROW("ConlluReaderPrivate::init(): cannot find linguistic code for micro " << boundaryMicro,
260 }
261 }
263 {
264 throw InvalidConfiguration("ConlluReaderPrivate::init(): can't find boundary micro");
265 }
266}
267
268// set default key in status according to other elements in status
270{
271 static QRegularExpression reCapital("^[[:upper:]]+$", QRegularExpression::UseUnicodePropertiesOption);
272 static QRegularExpression reSmall("^[[:lower:]]+$", QRegularExpression::UseUnicodePropertiesOption);
273 static QRegularExpression reCapital1st("^[[:upper:]]\\w+$", QRegularExpression::UseUnicodePropertiesOption);
274 static QRegularExpression reAcronym("^([[:upper:]]\\.)+$", QRegularExpression::UseUnicodePropertiesOption);
275 static QRegularExpression reCapitalSmall("^([[:upper:][:lower:]])+$", QRegularExpression::UseUnicodePropertiesOption);
276 static QRegularExpression reAbbrev("^\\w+\\.$", QRegularExpression::UseUnicodePropertiesOption);
277 static QRegularExpression reTwitter("^[@#]\\w+$", QRegularExpression::UseUnicodePropertiesOption);
278 static QRegularExpression reAlphaHyphen("^\\w+[\\-]\\w+$", QRegularExpression::UseUnicodePropertiesOption);
279 static QRegularExpression reAlpha("^\\w+$", QRegularExpression::UseUnicodePropertiesOption);
280
281 // t_cardinal_roman
282 static QRegularExpression reCardinalRoman("^(?=[MDCLXVI])M*(C[MD]|D?C{0,3})(X[CL]|L?X{0,3})(I[XV]|V?I{0,3})$");
283 // t_ordinal_roman
284 static QRegularExpression reOrdinalRoman("^(?=[MDCLXVI])M*(C[MD]|D?C{0,3})(X[CL]|L?X{0,3})(I[XV]|V?I{0,3})(st|nd|d|th|er|ème)$");
285 // t_integer
286 static QRegularExpression reInteger("^\\d+$");
287 // t_comma_number
288 static QRegularExpression reCommaNumber("^\\d+,\\d$");
289 // t_dot_number
290 static QRegularExpression reDotNumber("^\\d\\.\\d$");
291 // t_fraction
292 static QRegularExpression reFraction("^\\d([.,]\\d+)?/\\d([.,]\\d+)?$");
293 // t_ordinal_integer
294 static QRegularExpression reOrdinalInteger("^\\d+(st|nd|d|th|er|ème)$");
295 // t_alphanumeric
296 static QRegularExpression reAlphanumeric("^[\\d[:lower:][:upper:]]+$", QRegularExpression::UseUnicodePropertiesOption);
297 static QRegularExpression reSentenceBreak("^[;.!?]$");
298
299 TStatus curSettings;
300 if (reCapital.match(token.stringForm()).hasMatch())
301 {
302// #ifdef DEBUG_LP
303// LDEBUG << "CppUppsalaTokenizerPrivate::computeDefaultStatus t_capital";
304// #endif
305 curSettings.setDefaultKey(QString::fromUtf8("t_capital"));
306 }
307 else if (reSmall.match(token.stringForm()).hasMatch())
308 {
309// #ifdef DEBUG_LP
310// LDEBUG << "CppUppsalaTokenizerPrivate::computeDefaultStatus t_small";
311// #endif
312 curSettings.setDefaultKey(QString::fromUtf8("t_small"));
313 }
314 else if (reCapital1st.match(token.stringForm()).hasMatch())
315 {
316// #ifdef DEBUG_LP
317// LDEBUG << "CppUppsalaTokenizerPrivate::computeDefaultStatus t_capital_1st";
318// #endif
319 curSettings.setDefaultKey(QString::fromUtf8("t_capital_1st"));
320 }
321 else if (reAcronym.match(token.stringForm()).hasMatch())
322 {
323// #ifdef DEBUG_LP
324// LDEBUG << "CppUppsalaTokenizerPrivate::computeDefaultStatus t_acronym";
325// #endif
326 curSettings.setDefaultKey(QString::fromUtf8("t_acronym"));
327 }
328 else if (reCapitalSmall.match(token.stringForm()).hasMatch())
329 {
330// #ifdef DEBUG_LP
331// LDEBUG << "CppUppsalaTokenizerPrivate::computeDefaultStatus t_capital_small";
332// #endif
333 curSettings.setDefaultKey(QString::fromUtf8("t_capital_small"));
334 }
335 else if (reAbbrev.match(token.stringForm()).hasMatch())
336 {
337// #ifdef DEBUG_LP
338// LDEBUG << "CppUppsalaTokenizerPrivate::computeDefaultStatus t_abbrev";
339// #endif
340 curSettings.setDefaultKey(QString::fromUtf8("t_abbrev"));
341 }
342 else if (reTwitter.match(token.stringForm()).hasMatch())
343 {
344// #ifdef DEBUG_LP
345// LDEBUG << "CppUppsalaTokenizerPrivate::computeDefaultStatus t_twitter";
346// #endif
347 curSettings.setDefaultKey(QString::fromUtf8("t_twitter"));
348 }
349 else if (reCardinalRoman.match(token.stringForm()).hasMatch())
350 {
351// #ifdef DEBUG_LP
352// LDEBUG << "CppUppsalaTokenizerPrivate::computeDefaultStatus t_cardinal_roman";
353// #endif
354 curSettings.setDefaultKey(QString::fromUtf8("t_cardinal_roman"));
355 }
356 else if (reOrdinalRoman.match(token.stringForm()).hasMatch())
357 {
358// #ifdef DEBUG_LP
359// LDEBUG << "CppUppsalaTokenizerPrivate::computeDefaultStatus t_ordinal_roman";
360// #endif
361 curSettings.setDefaultKey(QString::fromUtf8("t_ordinal_roman"));
362 }
363 else if (reInteger.match(token.stringForm()).hasMatch())
364 {
365// #ifdef DEBUG_LP
366// LDEBUG << "CppUppsalaTokenizerPrivate::computeDefaultStatus t_integer";
367// #endif
368 curSettings.setDefaultKey(QString::fromUtf8("t_integer"));
369 }
370 else if (reCommaNumber.match(token.stringForm()).hasMatch())
371 {
372// #ifdef DEBUG_LP
373// LDEBUG << "CppUppsalaTokenizerPrivate::computeDefaultStatus t_comma_number";
374// #endif
375 curSettings.setDefaultKey(QString::fromUtf8("t_comma_number"));
376 }
377 else if (reDotNumber.match(token.stringForm()).hasMatch())
378 {
379// #ifdef DEBUG_LP
380// LDEBUG << "CppUppsalaTokenizerPrivate::computeDefaultStatus t_dot_number";
381// #endif
382 curSettings.setDefaultKey(QString::fromUtf8("t_dot_number"));
383 }
384 else if (reFraction.match(token.stringForm()).hasMatch())
385 {
386// #ifdef DEBUG_LP
387// LDEBUG << "CppUppsalaTokenizerPrivate::computeDefaultStatus t_fraction";
388// #endif
389 curSettings.setDefaultKey(QString::fromUtf8("t_fraction"));
390 }
391 else if (reOrdinalInteger.match(token.stringForm()).hasMatch())
392 {
393// #ifdef DEBUG_LP
394// LDEBUG << "CppUppsalaTokenizerPrivate::computeDefaultStatus t_ordinal_integer";
395// #endif
396 curSettings.setDefaultKey(QString::fromUtf8("t_ordinal_integer"));
397 }
398 else if (reAlphanumeric.match(token.stringForm()).hasMatch())
399 {
400// #ifdef DEBUG_LP
401// LDEBUG << "CppUppsalaTokenizerPrivate::computeDefaultStatus t_alphanumeric";
402// #endif
403 curSettings.setDefaultKey(QString::fromUtf8("t_alphanumeric"));
404 }
405 else if (reAlphaHyphen.match(token.stringForm()).hasMatch())
406 {
407 curSettings.setDefaultKey(QString::fromUtf8("t_alpha_hyphen"));
408 curSettings.setAlphaHyphen(true);
409 }
410 else if (reSentenceBreak.match(token.stringForm()).hasMatch())
411 {
412// #ifdef DEBUG_LP
413// LDEBUG << "CppUppsalaTokenizerPrivate::computeDefaultStatus t_sentence_brk";
414// #endif
415 curSettings.setDefaultKey(QString::fromUtf8("t_sentence_brk"));
416 }
417 else // if (reSmall.match(token.stringForm()).hasMatch())
418 {
419// #ifdef DEBUG_LP
420// LDEBUG << "CppUppsalaTokenizerPrivate::computeDefaultStatus t_word_brk (default)";
421// #endif
422 curSettings.setDefaultKey(QString::fromUtf8("t_word_brk"));
423 }
424
425 /*
426 // t_not_roman
427 static QRegularExpression reNotRoman("^$");
428 // t_alpha_concat_abbrev
429 static QRegularExpression reAlphConcatAbbrev("^$");
430 // t_pattern
431 static QRegularExpression rePattern("^$");
432 // t_word_brk
433 static QRegularExpression reWordBreak("^$");
434 // t_sentence_brk
435 static QRegularExpression reSentenceBreak("^$");
436 // t_paragraph_brk
437 static QRegularExpression reParagraphBreak("^$");
438 */
439
440 token.setStatus(curSettings);
441}
442
443
444void ConlluReaderPrivate::append_new_word(vector< TPrimitiveToken >& current_sentence,
445 const u32string& current_token,
446 int current_token_offset) const
447{
448 current_sentence.push_back(TPrimitiveToken(QString::fromStdU32String(current_token), current_token_offset));
449}
450
451vector< vector< ConlluReaderPrivate::TPrimitiveToken > > ConlluReaderPrivate::tokenize(const QString& text)
452{
453 LOG_MESSAGE_WITH_PROLOG(LDEBUG, "ConlluReaderPrivate::tokenize" << text.left(100));
454
455 vector< vector< TPrimitiveToken > > sentences;
456 vector< TPrimitiveToken > current_sentence;
457 int current_token_offset = 0;
458
459 QStringList list = text.split(QString("\n"));
460 for (size_t i = 0; i < size_t(list.size()); i++)
461 {
462 const QString line = list[i];
463 QStringList fields = line.split(QString("\t"));
464
465 if (fields.size() < 2)
466 {
467 if (current_sentence.size() > 0)
468 {
469 sentences.push_back(current_sentence);
470 }
471 current_sentence.clear();
472 continue;
473 }
474
475 if (-1 != fields[0].indexOf("-") || -1 != fields[0].indexOf("."))
476 {
477 continue;
478 }
479
480 current_sentence.push_back(TPrimitiveToken(fields[1], current_token_offset));
481 current_token_offset += fields[1].size() + 1;
482 }
483
484 if (current_sentence.size() > 0)
485 {
486 sentences.push_back(current_sentence);
487 }
488
489 return sentences;
490}
491
492} // namespace ConlluReader
493} // namespace LinguisticProcessing
494} // namespace Lima
#define LOG_ERROR_AND_THROW(msg, exc)
#define LOG_MESSAGE_WITH_PROLOG(stream, msg)
#define LOG_MESSAGE(stream, msg)
#define CONLLUREADER_CLASSID
#define LDEBUG
Definition LimaCommon.h:157
#define LINFO
Definition LimaCommon.h:158
#define LERROR
Definition LimaCommon.h:161
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define TOKENIZERLOGINIT
#define GET_PROPERTY_MANAGER(language, property)
#define GET_PROPERTY_ACCESSOR(language, property)
Defines a Factory to create Object of type Base.
#define L_NONE
Definition StdBitset.h:338
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Use this exception to signal an error in one of the configuration files.
Definition LimaCommon.h:345
void getStringParameter(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, const std::string &name, std::string &value, int flags=Flags::REQUIRED, std::string default_value="")
void append_new_word(vector< TPrimitiveToken > &current_sentence, const u32string &current_token, int current_token_offset) const
void init(MediaId language, GroupConfigurationStructure &unitConfiguration)
vector< vector< TPrimitiveToken > > tokenize(const QString &text)
This is a MediaProcessUnit that is usually the first element of the pipeline.
void init(Lima::Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
An AnalysisData containing a LinguisticGraph with a language and an id.
void setDefaultKey(const Lima::LimaString &defaultKey)
Definition TStatus.cpp:325
void setStatus(const TStatus &status)
Set the TStatus of a token.
Definition Token.h:100
This file contains a class to control log of informations about time, such as logging cumulated time ...
static SimpleFactory< MediaProcessUnit, ConlluReader > conlluReaderFactory(CONLLUREADER_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
STL namespace.
TPrimitiveToken(const QString &w, int pos, const QString &orig=QString())
launch exception related to the configuration file parsing