6#include <QtCore/QTemporaryFile>
7#include <QtCore/QRegularExpression>
9#include <QJsonDocument>
29#define DEBUG_THIS_FILE true
41namespace LinguisticProcessing
48inline string THIS_FILE_LOGGING_CATEGORY()
51 return logger.zone().toStdString();
55static u32string
SPACE = QString::fromUtf8(
" ").toStdU32String();
59#define LOG_ERROR_AND_THROW(msg, exc) { \
65#if defined(DEBUG_LP) && defined(DEBUG_THIS_FILE)
66 #define LOG_MESSAGE(stream, msg) stream << msg;
67 #define LOG_MESSAGE_WITH_PROLOG(stream, msg) TOKENIZERLOGINIT; LOG_MESSAGE(stream, msg);
69 #define LOG_MESSAGE(stream, msg) ;
70 #define LOG_MESSAGE_WITH_PROLOG(stream, msg) ;
84 const QString& orig=QString())
94 vector< vector< TPrimitiveToken > >
tokenize(
const QString& text);
105 const u32string& current_token,
106 int current_token_offset)
const;
114 m_stringsPool(nullptr),
148 analysis.
setData(
"AnalysisGraph",anagraph);
152 auto originalText = std::dynamic_pointer_cast<LimaStringText>(analysis.
getData(
"Text"));
155 auto sentencesTokens = m_d->
tokenize(*originalText);
159 if (m_d->
m_data.size() > 0)
167 remove_edge(anagraph->firstVertex(),
168 anagraph->lastVertex(),
173 for (
const auto& sentence: sentencesTokens)
175 if (sentence.size() < 1)
179 for (
const auto& token: sentence)
181 const auto& str = token.wordText;
186 Token *tToken =
new Token(form, str, token.start, token.wordText.size());
188 LOG_ERROR_AND_THROW(
"ConlluReader::process: Can't allocate memory with \"new Token(...)\"",
216 sb->
add(
Segment(
"sentence", beginSentence, endSentence, anagraph));
221 if (
nullptr != morphoData)
229 morphoData->push_back(elem);
235 beginSentence = endSentence;
238 add_edge(m_d->
m_currentVx,anagraph->lastVertex(),*graph);
252 string boundaryMicro =
getStringParameter(unitConfiguration,
"boundaryMicro", 0,
"PONCTU_FORTE");
258 LOG_ERROR_AND_THROW(
"ConlluReaderPrivate::init(): cannot find linguistic code for micro " << boundaryMicro,
271 static QRegularExpression reCapital(
"^[[:upper:]]+$", QRegularExpression::UseUnicodePropertiesOption);
272 static QRegularExpression reSmall(
"^[[:lower:]]+$", QRegularExpression::UseUnicodePropertiesOption);
273 static QRegularExpression reCapital1st(
"^[[:upper:]]\\w+$", QRegularExpression::UseUnicodePropertiesOption);
274 static QRegularExpression reAcronym(
"^([[:upper:]]\\.)+$", QRegularExpression::UseUnicodePropertiesOption);
275 static QRegularExpression reCapitalSmall(
"^([[:upper:][:lower:]])+$", QRegularExpression::UseUnicodePropertiesOption);
276 static QRegularExpression reAbbrev(
"^\\w+\\.$", QRegularExpression::UseUnicodePropertiesOption);
277 static QRegularExpression reTwitter(
"^[@#]\\w+$", QRegularExpression::UseUnicodePropertiesOption);
278 static QRegularExpression reAlphaHyphen(
"^\\w+[\\-]\\w+$", QRegularExpression::UseUnicodePropertiesOption);
279 static QRegularExpression reAlpha(
"^\\w+$", QRegularExpression::UseUnicodePropertiesOption);
282 static QRegularExpression reCardinalRoman(
"^(?=[MDCLXVI])M*(C[MD]|D?C{0,3})(X[CL]|L?X{0,3})(I[XV]|V?I{0,3})$");
284 static QRegularExpression reOrdinalRoman(
"^(?=[MDCLXVI])M*(C[MD]|D?C{0,3})(X[CL]|L?X{0,3})(I[XV]|V?I{0,3})(st|nd|d|th|er|ème)$");
286 static QRegularExpression reInteger(
"^\\d+$");
288 static QRegularExpression reCommaNumber(
"^\\d+,\\d$");
290 static QRegularExpression reDotNumber(
"^\\d\\.\\d$");
292 static QRegularExpression reFraction(
"^\\d([.,]\\d+)?/\\d([.,]\\d+)?$");
294 static QRegularExpression reOrdinalInteger(
"^\\d+(st|nd|d|th|er|ème)$");
296 static QRegularExpression reAlphanumeric(
"^[\\d[:lower:][:upper:]]+$", QRegularExpression::UseUnicodePropertiesOption);
297 static QRegularExpression reSentenceBreak(
"^[;.!?]$");
300 if (reCapital.match(token.
stringForm()).hasMatch())
307 else if (reSmall.match(token.
stringForm()).hasMatch())
314 else if (reCapital1st.match(token.
stringForm()).hasMatch())
319 curSettings.
setDefaultKey(QString::fromUtf8(
"t_capital_1st"));
321 else if (reAcronym.match(token.
stringForm()).hasMatch())
328 else if (reCapitalSmall.match(token.
stringForm()).hasMatch())
333 curSettings.
setDefaultKey(QString::fromUtf8(
"t_capital_small"));
335 else if (reAbbrev.match(token.
stringForm()).hasMatch())
342 else if (reTwitter.match(token.
stringForm()).hasMatch())
349 else if (reCardinalRoman.match(token.
stringForm()).hasMatch())
354 curSettings.
setDefaultKey(QString::fromUtf8(
"t_cardinal_roman"));
356 else if (reOrdinalRoman.match(token.
stringForm()).hasMatch())
361 curSettings.
setDefaultKey(QString::fromUtf8(
"t_ordinal_roman"));
363 else if (reInteger.match(token.
stringForm()).hasMatch())
370 else if (reCommaNumber.match(token.
stringForm()).hasMatch())
375 curSettings.
setDefaultKey(QString::fromUtf8(
"t_comma_number"));
377 else if (reDotNumber.match(token.
stringForm()).hasMatch())
382 curSettings.
setDefaultKey(QString::fromUtf8(
"t_dot_number"));
384 else if (reFraction.match(token.
stringForm()).hasMatch())
391 else if (reOrdinalInteger.match(token.
stringForm()).hasMatch())
396 curSettings.
setDefaultKey(QString::fromUtf8(
"t_ordinal_integer"));
398 else if (reAlphanumeric.match(token.
stringForm()).hasMatch())
403 curSettings.
setDefaultKey(QString::fromUtf8(
"t_alphanumeric"));
405 else if (reAlphaHyphen.match(token.
stringForm()).hasMatch())
407 curSettings.
setDefaultKey(QString::fromUtf8(
"t_alpha_hyphen"));
410 else if (reSentenceBreak.match(token.
stringForm()).hasMatch())
415 curSettings.
setDefaultKey(QString::fromUtf8(
"t_sentence_brk"));
445 const u32string& current_token,
446 int current_token_offset)
const
448 current_sentence.push_back(
TPrimitiveToken(QString::fromStdU32String(current_token), current_token_offset));
455 vector< vector< TPrimitiveToken > > sentences;
456 vector< TPrimitiveToken > current_sentence;
457 int current_token_offset = 0;
459 QStringList list = text.split(QString(
"\n"));
460 for (
size_t i = 0; i < size_t(list.size()); i++)
462 const QString line = list[i];
463 QStringList fields = line.split(QString(
"\t"));
465 if (fields.size() < 2)
467 if (current_sentence.size() > 0)
469 sentences.push_back(current_sentence);
471 current_sentence.clear();
475 if (-1 != fields[0].indexOf(
"-") || -1 != fields[0].indexOf(
"."))
480 current_sentence.push_back(
TPrimitiveToken(fields[1], current_token_offset));
481 current_token_offset += fields[1].size() + 1;
484 if (current_sentence.size() > 0)
486 sentences.push_back(current_sentence);
#define LOG_ERROR_AND_THROW(msg, exc)
#define LOG_MESSAGE_WITH_PROLOG(stream, msg)
#define LOG_MESSAGE(stream, msg)
#define CONLLUREADER_CLASSID
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define GET_PROPERTY_MANAGER(language, property)
#define GET_PROPERTY_ACCESSOR(language, property)
Defines a Factory to create Object of type Base.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
return a message when a 'list' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Use this exception to signal an error in one of the configuration files.
void getStringParameter(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, const std::string &name, std::string &value, int flags=Flags::REQUIRED, std::string default_value="")
virtual ~ConlluReaderPrivate()
LinguisticGraphVertex m_currentVx
void append_new_word(vector< TPrimitiveToken > ¤t_sentence, const u32string ¤t_token, int current_token_offset) const
void init(MediaId language, GroupConfigurationStructure &unitConfiguration)
LinguisticCode m_boundaryMicro
FsaStringsPool * m_stringsPool
vector< vector< TPrimitiveToken > > tokenize(const QString &text)
void computeDefaultStatus(Token &token)
This is a MediaProcessUnit that is usually the first element of the pipeline.
void init(Lima::Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
An AnalysisData containing a LinguisticGraph with a language and an id.
Holds morphosyntactic informations.
void setDefaultKey(const Lima::LimaString &defaultKey)
void setAlphaHyphen(bool isAlphaHyphen)
std::string toString() const
holds surface data of a token
const TStatus & status() const
const LimaString & stringForm() const
StringsPoolIndex form() const
void setStatus(const TStatus &status)
Set the TStatus of a token.
void add(const Segment &s)
This file contains a class to control log of informations about time, such as logging cumulated time ...
static SimpleFactory< MediaProcessUnit, ConlluReader > conlluReaderFactory(CONLLUREADER_CLASSID)
TPrimitiveToken(const QString &w, int pos, const QString &orig=QString())
StringsPoolIndex inflectedForm
StringsPoolIndex normalizedForm
LinguisticCode properties
launch exception related to the configuration file parsing