LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
Text.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6// NAUTITIA
7//
8// jys 24-JUL-2002
9//
10// Text is the class which reads original text and does its
11// 1st transformation into characters classes string.
12
13#include "Text.h"
14
15#include "CharChart.h"
16
22
23using namespace Lima;
25using namespace Lima::Common::Misc;
26
27namespace Lima
28{
29namespace LinguisticProcessing
30{
31namespace FlatTokenizer
32{
33
54
55
56TextPrivate::TextPrivate(MediaId lang, std::shared_ptr<CharChart> charChart) :
57 m_text(),
58 m_curPtr(0),
59 m_debPtr(0),
60 m_curSettings(),
61 m_tTokenGraph(0),
62 m_currentVx(0),
63 m_lastVx(0),
64 m_stringsPool(&Common::MediaticData::MediaticData::changeable().stringsPool(lang)),
65 m_charChart(charChart)
66{
67}
68
72
73Text::Text(MediaId lang, std::shared_ptr<CharChart> charChart) :
74 m_d(new TextPrivate(lang, charChart))
75{
76}
77
79{
80 delete m_d;
81}
82
84{
85 if (m_d->m_curPtr >= m_d->m_text.size())
86 {
87 return QChar();
88 }
89 return m_d->m_text[m_d->m_curPtr];
90}
91
93{
94 m_d->m_text = text;
95 m_d->m_curPtr = 0;
96 m_d->m_debPtr = 0;
97 m_d->m_tTokenGraph = 0;
98 m_d->m_currentVx = 0;
99 m_d->m_lastVx = 0;
100}
101
102int Text::position() const
103{
104 return m_d->m_curPtr;
105}
106
107int Text::size() const
108{
109 return m_d->m_text.size();
110}
111
113{
114 m_d->m_curSettings = status;
115}
116
117
118// Clear the entirely class and structure to accept new text
120{
121 // _tTokenList = 0; Not destroyed here
122}
123
125{
127 m_d->m_tTokenGraph = graph;
128 // go one step forward on the new path
129 LinguisticGraphAdjacencyIt adjItr,adjItrEnd;
130 boost::tie (adjItr,adjItrEnd) = adjacent_vertices(m_d->m_currentVx, *m_d->m_tTokenGraph);
131 if (adjItr==adjItrEnd)
132 {
134 LIMA_LP_EXCEPTION("Tokenizer Text : no token forward !");
135 }
136 m_d->m_lastVx=*adjItr;
137 if (++adjItr!=adjItrEnd) {
139 LIMA_LP_EXCEPTION("Tokenizer Text : more than one next token !");
140 }
141 //remove_edge(m_d->m_currentVx,m_d->m_lastVx,*m_d->m_tTokenGraph);
142}
143
145{
146 add_edge(m_d->m_currentVx,m_d->m_lastVx,*m_d->m_tTokenGraph);
147 m_d->m_tTokenGraph=0;
148}
149
150// increments text pointer
152{
153#ifdef DEBUG_LP
155#endif
156 if (m_d->m_curPtr+1 >= m_d->m_text.size())
157 {
158#ifdef DEBUG_LP
159 LDEBUG << "Trying to move after text end.";
160#endif
161 m_d->m_curPtr++;
162 return QChar();
163 }
164 if (m_d->m_text.at(m_d->m_curPtr).isHighSurrogate())
165 {
166 m_d->m_curPtr++;
167 }
168 m_d->m_curPtr++;
169#ifdef DEBUG_LP
170 LDEBUG << "Text::advance : new current=" << m_d->m_curPtr << " from='"
171 << m_d->m_text[m_d->m_curPtr-1] << "' to='" << m_d->m_text[m_d->m_curPtr] << "'";
172#endif
173 return m_d->m_text[m_d->m_curPtr];
174}
175
177{
178#ifdef DEBUG_LP
180 if (m_d->m_curPtr+1 >= m_d->m_text.size())
181 {
182 LDEBUG << "currentClass() at " << m_d->m_curPtr << ". No char after text end";
183 }
184 else
185 {
186 LDEBUG << "currentClass() at " << m_d->m_curPtr << ", for " << m_d->m_text[m_d->m_curPtr];
187 }
188#endif
189 if (m_d->m_curPtr >= m_d->m_text.size())
190 {
191 return m_d->m_charChart->charClass(QChar());
192 }
193 QChar c = m_d->m_text[m_d->m_curPtr];
194 if (c.isHighSurrogate())
195 {
196 if (m_d->m_curPtr+1 >= m_d->m_text.size())
197 {
198 return m_d->m_charChart->charClass(QChar());
199 }
200 return m_d->m_charChart->charClass( m_d->m_text[m_d->m_curPtr], m_d->m_text[m_d->m_curPtr+1] );
201 }
202 else
203 {
204 return m_d->m_charChart->charClass( m_d->m_text[m_d->m_curPtr] );
205 }
206
207}
208
209
210// flushes current token
212{
213 m_d->m_debPtr = m_d->m_curPtr;
214}
215
216// takes a token
218{
219#ifdef DEBUG_LP
221#endif
222 // Creates a new token
223 uint64_t delta = m_d->m_curPtr;
224 if (m_d->m_curPtr < m_d->m_text.size()
225 && (m_d->m_text.at(m_d->m_curPtr).isHighSurrogate() || m_d->m_curPtr == m_d->m_debPtr))
226 {
227 delta++;
228 }
229 if (m_d->m_debPtr >= m_d->m_text.size())
230 {
232 LERROR << "Empty token !";
233 m_d->m_debPtr = delta;
234 m_d->m_curSettings.reset();
235 return utf8stdstring2limastring("");
236 }
237 LimaString str=m_d->m_text.mid( m_d->m_debPtr, (delta-m_d->m_debPtr));
238#ifdef DEBUG_LP
239 LDEBUG << " Adding token '" << str << "'";
240#endif
241 StringsPoolIndex form=(*m_d->m_stringsPool)[str];
242 Token *tToken = new Token(form,str,m_d->m_debPtr+1,(delta-m_d->m_debPtr));
243 if (tToken == nullptr) throw MemoryErrorException();
244 // @todo: set default status here, according to structured status (alpha,numeric etc...)
245 // instead of setting it at each change of status (setAlphaCapital, setNumeric etc...)
246 tToken->setStatus(m_d->m_curSettings);
247// LDEBUG << " m_d->m_curSettings is " << m_d->m_curSettings.toString();
248#ifdef DEBUG_LP
249 LDEBUG << " status is " << tToken->status().toString();
250#endif
251 // Adds on the path
252 LinguisticGraphVertex newVx = add_vertex(*m_d->m_tTokenGraph);
253 put(vertex_token, *m_d->m_tTokenGraph, newVx, tToken);
254 put(vertex_data, *m_d->m_tTokenGraph, newVx, new MorphoSyntacticData());
255 add_edge(m_d->m_currentVx, newVx, *m_d->m_tTokenGraph);
256 m_d->m_currentVx=newVx;
257 m_d->m_debPtr = delta;
258 m_d->m_curSettings.reset();
259 return str;
260}
261
262// performs a trace
264{}
265
267 if (((static_cast<int>(m_d->m_curPtr)+i) < 0) || (static_cast<int>(m_d->m_curPtr)+i >= m_d->m_text.size()))
268 throw BoundsErrorException();
269 return m_d->m_text[m_d->m_curPtr+i];
270}
271
273{
274#ifdef DEBUG_LP
276 LDEBUG << "Text::setAlphaCapital " << alphaCapital;
277#endif
278 m_d->m_curSettings.setAlphaCapital(alphaCapital);
279 switch (alphaCapital)
280 {
281 case T_CAPITAL:
283 break;
284 case T_SMALL:
286 break;
287 case T_CAPITAL_1ST:
289 break;
290 case T_ACRONYM:
292 break;
293 case T_CAPITAL_SMALL:
295 break;
296 case T_ABBREV:
298 break;
299 default:
301 }
302}
303
305{
306#ifdef DEBUG_LP
308 LDEBUG << "Text::setAlphaRoman " << alphaRoman;
309#endif
310 m_d->m_curSettings.setAlphaRoman(alphaRoman);
311 switch (alphaRoman)
312 {
313 case T_CARDINAL_ROMAN:
315 break;
316 case T_ORDINAL_ROMAN:
318 break;
319 case T_NOT_ROMAN:
321 break;
322 default:;
323 }
324}
325
326void Text::setAlphaHyphen(const unsigned char isAlphaHyphen)
327{
328#ifdef DEBUG_LP
330 LDEBUG << "Text::setAlphaHyphen " << isAlphaHyphen;
331#endif
332 m_d->m_curSettings.setAlphaHyphen(isAlphaHyphen);
333}
334
335void Text::setAlphaPossessive(const unsigned char isAlphaPossessive)
336{
337#ifdef DEBUG_LP
339 LDEBUG << "Text::setAlphaPossessive " << isAlphaPossessive;
340#endif
341 m_d->m_curSettings.setAlphaPossessive(isAlphaPossessive);
342/* if (isAlphaPossessive > 0)
343 m_d->m_curSettings.setDefaultKey(Common::Misc::utf8stdstring2limastring("t_alpha_possessive"));*/
344}
345
346void Text::setAlphaConcatAbbrev(const unsigned char isConcatAbbreviation)
347{
348#ifdef DEBUG_LP
350 LDEBUG << "Text::setAlphaConcatAbbrev " << isConcatAbbreviation;
351#endif
352 m_d->m_curSettings.setAlphaConcatAbbrev(isConcatAbbreviation);
353 if (isConcatAbbreviation> 0)
355}
356
357void Text::setTwitter(const unsigned char isTwitter)
358{
359#ifdef DEBUG_LP
361 LDEBUG << "Text::setTwitter " << isTwitter;
362#endif
363 m_d->m_curSettings.setTwitter(isTwitter);
364 if (isTwitter> 0)
366}
367
369{
370#ifdef DEBUG_LP
372 LDEBUG << "Text::setNumeric " << numeric;
373#endif
374 m_d->m_curSettings.setNumeric(numeric);
375 switch (numeric)
376 {
377 case T_INTEGER:
379 break;
380 case T_COMMA_NUMBER:
382 break;
383 case T_DOT_NUMBER:
385 break;
386 case T_FRACTION:
388 break;
391 break;
392 default:;
393 }
394}
395
397{
398#ifdef DEBUG_LP
400 LDEBUG << "Text::setStatus " << status;
401#endif
403 m_d->m_curSettings.setStatus(status);
404 switch (status)
405 {
406 case T_ALPHA:
407 if (previousStatus != T_ALPHA)
409 break;
410 case T_NUMERIC:
412 break;
413 case T_ALPHANUMERIC:
415 break;
416 case T_PATTERN:
418 break;
419 case T_WORD_BRK:
421 break;
422 case T_SENTENCE_BRK:
424 break;
425 case T_PARAGRAPH_BRK:
427 break;
428 default:;
429 }
430
431}
432
434{
435#ifdef DEBUG_LP
437 LDEBUG << "Text::setDefaultKey " << Common::Misc::limastring2utf8stdstring(defaultKey);
438#endif
439 m_d->m_curSettings.setDefaultKey(defaultKey);
440}
441
442// set default key in status according to other elements in status
444{
445 std::string defaultKey;
446 switch (m_d->m_curSettings.getStatus()) {
447 case T_ALPHA : {
448 switch (m_d->m_curSettings.getAlphaCapital()) {
449 case T_CAPITAL : defaultKey = "t_capital" ; break;
450 case T_SMALL : defaultKey = "t_small" ; break;
451 case T_CAPITAL_1ST : defaultKey = "t_capital_1st" ; break;
452 case T_ACRONYM : defaultKey = "t_acronym" ; break;
453 case T_CAPITAL_SMALL : defaultKey = "t_capital_small"; break;
454 case T_ABBREV : defaultKey = "t_abbrev" ; break;
455 default : break;
456 }
457 switch (m_d->m_curSettings.getAlphaRoman()) { // Roman supersedes Cardinal
458 case T_CARDINAL_ROMAN : defaultKey = "t_cardinal_roman"; break;
459 case T_ORDINAL_ROMAN : defaultKey = "t_ordinal_roman" ; break;
460 case T_NOT_ROMAN : defaultKey = "t_not_roman" ; break;
461 default : break;
462 }
463 if (m_d->m_curSettings.isAlphaHyphen()) {
464 //no change
465 //defaultKey = "t_alpha_hyphen";
466 }
467 if (m_d->m_curSettings.isAlphaPossessive()) {
468 defaultKey = "t_alpha_possessive";
469 }
470 break;
471 } // end T_ALPHA
472 case T_NUMERIC : {
473 switch (m_d->m_curSettings.getNumeric()) {
474 case T_INTEGER : defaultKey = "t_integer" ; break;
475 case T_COMMA_NUMBER : defaultKey = "t_comma_number" ; break;
476 case T_DOT_NUMBER : defaultKey = "t_dot_number" ; break;
477 case T_FRACTION : defaultKey = "t_fraction" ; break;
478 case T_ORDINAL_INTEGER : defaultKey = "t_ordinal_integer"; break;
479 default: break;
480 }
481 break;
482 }
483 case T_ALPHANUMERIC : defaultKey = "t_alphanumeric" ; break;
484 case T_PATTERN : defaultKey = "t_pattern" ; break;
485 case T_WORD_BRK : defaultKey = "t_word_brk" ; break;
486 case T_SENTENCE_BRK : defaultKey = "t_sentence_brk" ; break;
487 case T_PARAGRAPH_BRK: defaultKey = "t_paragraph_brk" ; break;
488 default: defaultKey = "t_fallback";
489 }
490#ifdef DEBUG_LP
492 LDEBUG << "Text::computeDefaultKey " << defaultKey;
493#endif
495}
496
497
498} //namespace FlatTokenizer
499} // namespace LinguisticProcessing
500} // namespace Lima
#define LDEBUG
Definition LimaCommon.h:157
#define LERROR
Definition LimaCommon.h:161
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
LinguisticGraph::adjacency_iterator LinguisticGraphAdjacencyIt
#define TOKENIZERLOGINIT
#define LIMA_LP_EXCEPTION(X)
This macro writes the message X to a previously configured error stream before throwing a LinguisticP...
holds data about codes and names for grammatical categories, etc.
TextPrivate & operator=(const Text &)=delete
TextPrivate(MediaId lang, std::shared_ptr< CharChart > charChart)
Definition Text.cpp:56
LinguisticAnalysisStructure::TStatus m_curSettings
Definition Text.cpp:46
void setGraph(LinguisticGraphVertex position, LinguisticGraph *graph)
Definition Text.cpp:124
void setAlphaRoman(const LinguisticAnalysisStructure::AlphaRomanType alphaRoman)
Definition Text.cpp:304
void setAlphaConcatAbbrev(const unsigned char isConcatAbbreviation)
Definition Text.cpp:346
void setAlphaHyphen(const unsigned char isAlphaHyphen)
Definition Text.cpp:326
void setDefaultKey(const Lima::LimaString &defaultKey)
Definition Text.cpp:433
const CharClass * currentClass() const
Definition Text.cpp:176
void setNumeric(const LinguisticAnalysisStructure::NumericType numeric)
Definition Text.cpp:368
void setTwitter(const unsigned char isTwitter)
Definition Text.cpp:357
void setStatus(const LinguisticAnalysisStructure::StatusType status)
Definition Text.cpp:396
void setAlphaCapital(const LinguisticAnalysisStructure::AlphaCapitalType alphaCapital)
Definition Text.cpp:272
void setAlphaPossessive(const unsigned char isAlphaPossessive)
Definition Text.cpp:335
void setText(const Lima::LimaString &text)
Definition Text.cpp:92
Lima::LimaChar operator[](int i) const
Definition Text.cpp:266
Text(MediaId lang, std::shared_ptr< CharChart > charChart)
Definition Text.cpp:73
void setDefaultKey(const Lima::LimaString &defaultKey)
Definition TStatus.cpp:325
void setAlphaCapital(const AlphaCapitalType alphaCapital)
Definition TStatus.cpp:254
void setStatus(const TStatus &status)
Set the TStatus of a token.
Definition Token.h:100
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
LimaString utf8stdstring2limastring(const std::string &src)
NAUTITIA.
QChar LimaChar
Definition LimaString.h:30
QString LimaString
Definition LimaString.h:33