41namespace LinguisticProcessing
43namespace FlatTokenizer
54 "SET_T_ALPHA_POSSESSIVE",
59 "SET_T_CAPITAL_SMALL",
60 "SET_T_CARDINAL_ROMAN",
61 "SET_T_ORDINAL_ROMAN",
67 "SET_T_ORDINAL_INTEGER",
68 "SET_T_ALPHA_CONCAT_ABBREV",
69 "SET_T_PARAGRAPH_BRK",
73 "SET_T_ACRONYM_ARABIC",
74 "SET_T_ACRONYM_LATIN_ARABIC",
82 _events(state->automaton().charChart()),
83 _condition(
Condition(state->automaton().charChart())),
101 if (currentClass ==
nullptr)
109 LDEBUG <<
"| | looking at transition "<<
this<<
" with char (" << chcl <<
" ; " << currentClass->
id() <<
" ; " << currentClass->
name() <<
")";
114 LDEBUG <<
"| | event " << chcl <<
" not recognized.";
115 LDEBUG <<
"| | transition failed";
122 LDEBUG <<
"| | event " << chcl <<
" recognized but conditions not fullfilled.";
123 LDEBUG <<
"| | transition failed";
128 LDEBUG <<
"| | event " << chcl <<
" recognized: taking actions length="<<m_defaultKey.length()<<
", tokenize: "<<m_tokenize<<
", flush: "<<m_flush<<
".";
138 if (m_defaultKey.length() != 0)
149 LDEBUG <<
"| | | adding token";
156 LDEBUG <<
"| | | flushing";
167 LDEBUG <<
"| | "<<
this<<
" transition succeeded (next state is "
169 << currentClass->
name();
175void Transition::applySettings(
Text& text)
const
180 for (
auto it = m_settings.cbegin(), it_end = m_settings.end();
228 LDEBUG <<
" Setting transition status setting to " << str;
230 if (str ==
"T_CAPITAL")
234 else if (str ==
"T_SMALL")
238 else if (str ==
"T_CAPITAL_1ST")
242 else if (str ==
"T_ACRONYM")
246 else if (str ==
"T_CAPITAL_SMALL")
250 else if (str ==
"T_CARDINAL_ROMAN")
254 else if (str ==
"T_ORDINAL_ROMAN")
258 else if (str ==
"T_NOT_ROMAN")
262 else if (str ==
"T_INTEGER")
266 else if (str ==
"T_COMMA_NUMBER")
270 else if (str ==
"T_DOT_NUMBER")
274 else if (str ==
"T_FRACTION")
278 else if (str ==
"T_ORDINAL_INTEGER")
282 else if (str ==
"T_ALPHA")
286 else if (str ==
"T_NUMERIC")
290 else if (str ==
"T_ALPHANUMERIC")
294 else if (str ==
"T_PATTERN")
298 else if (str ==
"T_WORD_BRK")
302 else if (str ==
"T_SENTENCE_BRK")
306 else if (str ==
"T_PARAGRAPH_BRK")
311 else if (str ==
"T_HYPHEN_WORD")
315 else if (str ==
"T_POSSESSIVE")
319 else if (str ==
"T_ALPHA_CONCAT_ABBREV")
323 else if (str ==
"T_ARABIC")
327 else if (str ==
"T_LATIN_ARABIC")
331 else if (str ==
"T_ART_DEF")
335 else if (str ==
"T_ACRONYM_ARABIC")
339 else if (str ==
"T_ACRONYM_LATIN_ARABIC")
343 else if (str ==
"T_TWITTER")
347 else if (str ==
"T_ABBREV")
354 LERROR <<
"Transition::setSetting at " << __FILE__ <<
", line " << __LINE__
355 <<
": Unkown satus setting '"<<str<<
"'";
#define TOKENIZERLOADERLOGINIT
const Lima::LimaString & id() const
const Lima::LimaString & name() const
bool isFulfilled(const Text &text) const
bool isRecognized(const Lima::LimaChar &event) const
void setAlphaRoman(const LinguisticAnalysisStructure::AlphaRomanType alphaRoman)
void setAlphaConcatAbbrev(const unsigned char isConcatAbbreviation)
void setAlphaHyphen(const unsigned char isAlphaHyphen)
void setDefaultKey(const Lima::LimaString &defaultKey)
const CharClass * currentClass() const
void setNumeric(const LinguisticAnalysisStructure::NumericType numeric)
void setTwitter(const unsigned char isTwitter)
void setStatus(const LinguisticAnalysisStructure::StatusType status)
Lima::LimaChar currentChar() const
void setAlphaCapital(const LinguisticAnalysisStructure::AlphaCapitalType alphaCapital)
void setAlphaPossessive(const unsigned char isAlphaPossessive)
Transition(const State *state)
const State * run(Text &text) const
bool setSetting(const LimaString &s)
const Lima::LimaString nextStateName() const
static const char * SettingNames[]
@ SET_T_ALPHA_CONCAT_ABBREV
@ SET_T_ACRONYM_LATIN_ARABIC
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
LimaString utf8stdstring2limastring(const std::string &src)