5#include <QtCore/QTemporaryFile>
6#include <QtCore/QRegularExpression>
31#define DEBUG_THIS_FILE true
43namespace LinguisticProcessing
45namespace DeepLimaUnits
52#if defined(DEBUG_LP) && defined(DEBUG_THIS_FILE)
53 #define LOG_MESSAGE(stream, msg) stream << msg;
54 #define LOG_MESSAGE_WITH_PROLOG(stream, msg) TOKENIZERLOGINIT; LOG_MESSAGE(stream, msg);
56 #define LOG_MESSAGE(stream, msg) ;
57 #define LOG_MESSAGE_WITH_PROLOG(stream, msg) ;
74 const QString& orig=QString())
84 void tokenize(
const QString& text, std::vector<std::vector<TPrimitiveToken>>& sentences);
88 const QString& current_token,
89 int current_token_offset)
const;
103 std::shared_ptr<segmentation::impl::SegmentationImpl>
m_segm;
111 m_stringsPool(nullptr),
142 m_d->
init(unitConfiguration);
152 auto anagraph = std::make_shared<AnalysisGraph>(
"AnalysisGraph", m_d->
m_language,
true,
true);
153 analysis.
setData(
"AnalysisGraph", anagraph);
154 auto graph = anagraph->getGraph();
157 auto originalText = std::dynamic_pointer_cast<LimaStringText>(analysis.
getData(
"Text"));
158 if (originalText ==
nullptr)
161 LERROR <<
"Can't Process RnnTokenizer: missing data 'Text'";
166 std::vector< std::vector< RnnTokenizerPrivate::TPrimitiveToken > > sentencesTokens;
167 m_d->
tokenize(*originalText, sentencesTokens);
171 auto sb = std::make_shared<SegmentationData>(
"AnalysisGraph");
175 remove_edge(anagraph->firstVertex(),
176 anagraph->lastVertex(),
181 for (
const auto& sentence: sentencesTokens)
183 if (sentence.size() < 1)
188 auto endSentence = std::numeric_limits< LinguisticGraphVertex >::max();
189 for (
const auto& token: sentence)
191 const auto& str = token.wordText;
196 Token *tToken =
new Token(form, str, token.start+1, token.wordText.size());
197 if (tToken ==
nullptr)
200 LERROR <<
"RnnFlowTokenizer::process: Can't allocate memory with \"new Token(...)\"";
204 if (token.originalText.size() > 0)
217 auto newVx = add_vertex(*graph);
227 sb->add(
Segment(
"sentence", beginSentence, endSentence, anagraph.get()));
228 beginSentence = endSentence;
231 add_edge(m_d->
m_currentVx, anagraph->lastVertex(), *graph);
240 auto model_prefix = QString::fromStdString(
246 auto lang_str = QString::fromStdString(MediaticData::single().media(
m_language));
247 auto resources_path = QString::fromStdString(MediaticData::single().getResourcesPath());
248 auto model_name = model_prefix;
250 MediaticData::single().getOptionValue(
"udlang", udlang);
251 LOG_MESSAGE(
LDEBUG,
"RnnTokenizerPrivate::init lang_str=" << lang_str <<
", udlang=" << udlang);
256 "RnnTokenizerPrivate::init: Can't parse language id " << udlang.c_str(),
260 model_name.replace(QString(
"$udlang"), QString::fromStdString(udlang));
263 QString::fromUtf8(
"/RnnTokenizer/%1/%2.pt")
264 .arg(lang_str).arg(model_name));
265 if (model_file_name.isEmpty())
277 m_segm->load(model_file_name.toStdString());
290 const QString& current_token,
291 int current_token_offset)
const
293 auto ctoken_lower = current_token.toLower();
298 current_sentence.push_back(
TPrimitiveToken(current_token, current_token_offset));
303 for (
const auto& w : i->second)
307 current_sentence.push_back(
TPrimitiveToken(w, current_token_offset, current_token));
319 m_segm = std::make_shared<segmentation::impl::SegmentationImpl>();
326 sentences.reserve(text.size() / 15);
328 std::vector< TPrimitiveToken > current_sentence;
329 int current_token_offset = 0;
331 auto text_utf8 = text.toStdString();
333 m_segm->register_handler([
this, &sentences, ¤t_sentence, ¤t_token_offset]
334 (
const std::vector<segmentation::token_pos>& tokens,
337 for (
size_t i = 0; i < len; i++)
339 const auto& tok = tokens[i];
344 append_new_word(current_sentence, QString::fromUtf8(tok.m_pch, tok.m_len), current_token_offset);
345 current_token_offset += (tok.m_offset + tok.m_len);
346 if (tok.m_flags & token_flags_t::sentence_brk)
348 sentences.push_back(current_sentence);
349 current_sentence.clear();
354 size_t bytes_consumed = 0;
355 m_segm->parse_from_stream([&text_utf8, &bytes_consumed]
360 read = (text_utf8.size() - bytes_consumed) > max ? max : (text_utf8.size() - bytes_consumed);
361 memcpy(buffer, text_utf8.c_str() + bytes_consumed, read);
362 bytes_consumed += read;
363 return (text_utf8.size() - bytes_consumed) > max;
#define CONFIGURATIONHELPER_LOGGING_INIT(X)
#define LOG_MESSAGE_WITH_PROLOG(stream, msg)
#define LOG_MESSAGE(stream, msg)
#define LIMA_EXCEPTION_SELECT_LOGINIT(X, Y, Z)
This macro writes the message Y to the error stream configured by its first parameter X,...
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
#define RNNTOKENIZER_CLASSID
Defines a Factory to create Object of type Base.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Use this exception to signal an error in one of the configuration files.
void getStringParameter(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, const std::string &name, std::string &value, int flags=Flags::REQUIRED, std::string default_value="")
LinguisticGraphVertex m_currentVx
virtual ~RnnTokenizerPrivate()
void init(GroupConfigurationStructure &unitConfiguration)
void tokenize(const QString &text, std::vector< std::vector< TPrimitiveToken > > &sentences)
std::function< void()> m_load_fn
FsaStringsPool * m_stringsPool
std::shared_ptr< segmentation::impl::SegmentationImpl > m_segm
void append_new_word(std::vector< TPrimitiveToken > ¤t_sentence, const QString ¤t_token, int current_token_offset) const
std::map< QString, std::vector< QString > > m_trrules
This is a MediaProcessUnit that is usually the first element of the pipeline.
void init(Lima::Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
virtual void computeDefaultStatus(LinguisticAnalysisStructure::TStatus &curSettings)
Holds morphosyntactic informations.
std::string toString() const
holds surface data of a token
void addOrthographicAlternatives(StringsPoolIndex alt)
const TStatus & status() const
This file contains a class to control log of informations about time, such as logging cumulated time ...
static void logElapsedTime(const std::string &mess, const std::string &taskCategory=std::string(""))
log the number of microseconds since last UpdateCurrentTime
static void updateCurrentTime(const std::string &taskCategory=std::string(""))
store current time for new elapsed time computation
QString findFileInPaths(const QString &paths, const QString &fileName, const QChar &separator)
Find the given file in the given paths.
static SimpleFactory< MediaProcessUnit, RnnTokenizer > rnntokenizerFactory(RNNTOKENIZER_CLASSID)
bool fix_lang_codes(QString &lang_str, std::string &udlang)
TPrimitiveToken(const QString &w, int pos, const QString &orig=QString())