LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
Transition.cpp
Go to the documentation of this file.
1// Copyright 2002-2019 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6// NAUTITIA
7//
8// jys 15-JUL-2002
9//
10// Transition is the main composant of automatons. Actions
11// are performed only during transitions. Automatons evoluate
12// only with transitions.
13// Components of transitions are :
14// o "check" components to determine if transition is "open"
15// o allowed events : to compare the current character
16// class with. (mandatory)
17// o static conditions : to check characters classes
18// before and after current character class and
19// inner automaton return status. (optionnal)
20// o "action" components to determine what to do if transition
21// has been found open.
22// o next state : on the next character class, automaton
23// will be on that state (optionnal)
24// o action : move the pointer on character classes buffer,
25// take a token, flush the token memory, ... (optionnal)
26// o setting : set up flags into tokens data structure
27// o return_status if automaton is an inner one. (optionnal)
28
29#include "Transition.h"
30
31#include "Condition.h"
32#include "State.h"
35
37using namespace Lima::Common::Misc;
38
39namespace Lima
40{
41namespace LinguisticProcessing
42{
43namespace FlatTokenizer
44{
45
46const char* Transition::SettingNames[] = {
47 "SET_T_ALPHA", // 0
48 "SET_T_NUMERIC", // 1
49 "SET_T_PATTERN", // 2
50 "SET_T_WORD_BRK", // 3
51 "SET_T_SENTENCE_BRK", // 4
52 "SET_T_ALPHANUMERIC", // 5
53 "SET_T_ALPHA_HYPHEN", // 6
54 "SET_T_ALPHA_POSSESSIVE", // 7
55 "SET_T_CAPITAL", // 8
56 "SET_T_SMALL", // 9
57 "SET_T_CAPITAL_1ST", // 10
58 "SET_T_ACRONYM", // 11
59 "SET_T_CAPITAL_SMALL", // 12
60 "SET_T_CARDINAL_ROMAN", // 13
61 "SET_T_ORDINAL_ROMAN", // 14
62 "SET_T_NOT_ROMAN", // 15
63 "SET_T_INTEGER", // 16
64 "SET_T_COMMA_NUMBER", // 17
65 "SET_T_DOT_NUMBER", // 18
66 "SET_T_FRACTION", // 19
67 "SET_T_ORDINAL_INTEGER", // 20
68 "SET_T_ALPHA_CONCAT_ABBREV", // 21
69 "SET_T_PARAGRAPH_BRK", // 22
70 "SET_T_ARABIC", // 23
71 "SET_T_LATIN_ARABIC", // 24
72 "SET_T_ART_DEF", // 25
73 "SET_T_ACRONYM_ARABIC", // 26
74 "SET_T_ACRONYM_LATIN_ARABIC",// 27
75 "SET_T_TWITTER", // 28
76 "SET_T_ABBREV" // 29
77};
78
80 m_state(state),
81 _toState(0),
82 _events(state->automaton().charChart()),
83 _condition(Condition(state->automaton().charChart())),
84 m_tokenize(false),
85 m_flush(false),
86 m_defaultKey()
87{
88}
89
93
94// for run-time use. Transition does its works
95const State* Transition::run(Text& text) const
96{
97#ifdef DEBUG_LP
99#endif
100 const CharClass* currentClass = text.currentClass();
101 if (currentClass == nullptr)
102 {
104 LERROR << "Transition::run Null Class for char '"<< limastring2utf8stdstring(LimaString()+text.currentChar()) << "'";
105 return 0;
106 }
107 LimaChar chcl = text.currentChar();
108#ifdef DEBUG_LP
109 LDEBUG << "| | looking at transition "<<this<<" with char (" << chcl << " ; " << currentClass->id() << " ; " << currentClass->name() << ")";
110#endif
111 if (!_events.isRecognized(chcl))
112 {
113#ifdef DEBUG_LP
114 LDEBUG << "| | event " << chcl << " not recognized.";
115 LDEBUG << "| | transition failed";
116#endif
117 return 0;
118 }
119 else if (!_condition.isFulfilled(text))
120 {
121#ifdef DEBUG_LP
122 LDEBUG << "| | event " << chcl << " recognized but conditions not fullfilled.";
123 LDEBUG << "| | transition failed";
124#endif
125 return 0;
126 }
127#ifdef DEBUG_LP
128 LDEBUG << "| | event " << chcl << " recognized: taking actions length="<<m_defaultKey.length()<<", tokenize: "<<m_tokenize<<", flush: "<<m_flush<<".";
129#endif
130 if (text.position() == 0)
131 {
132 applySettings(text);
133 }
134 // Transition is opened
135// LDEBUG << "Setting token StatusType to " << m_status.getStatus();
136// text.setStatus(m_status);
137
138 if (m_defaultKey.length() != 0)
139 {
140#ifdef DEBUG_LP
141 LDEBUG << "Setting token default key to " << limastring2utf8stdstring(m_defaultKey);
142#endif
143 text.setDefaultKey(m_defaultKey);
144 }
145
146 if (m_tokenize)
147 {
148#ifdef DEBUG_LP
149 LDEBUG << "| | | adding token";
150#endif
151 text.token();
152 }
153 if (m_flush)
154 {
155#ifdef DEBUG_LP
156 LDEBUG << "| | | flushing";
157#endif
158 text.flush();
159 }
160
161 if (text.position() != 0)
162 {
163 applySettings(text);
164 }
165
166#ifdef DEBUG_LP
167 LDEBUG << "| | "<<this<<" transition succeeded (next state is "
168 << nextStateName() << ") on "
169 << currentClass->name();
170#endif
171 text.advance();
172 return _toState;
173}
174
175void Transition::applySettings(Text& text) const
176{
177#ifdef DEBUG_LP
179#endif
180 for (auto it = m_settings.cbegin(), it_end = m_settings.end();
181 it != it_end; it++)
182 {
183#ifdef DEBUG_LP
184 LDEBUG << "Putting status setting to text: " << Transition::SettingNames[*it];
185#endif
186 switch (*it)
187 {
188 case SET_T_ALPHA : text.setStatus(T_ALPHA); break;
189 case SET_T_NUMERIC : text.setStatus(T_NUMERIC); break;
190 case SET_T_PATTERN : text.setStatus(T_PATTERN); break;
191 case SET_T_WORD_BRK : text.setStatus(T_WORD_BRK); break;
192 case SET_T_SENTENCE_BRK : text.setStatus(T_SENTENCE_BRK); break;
193 case SET_T_ALPHANUMERIC : text.setStatus(T_ALPHANUMERIC); break;
194 case SET_T_ALPHA_HYPHEN : text.setAlphaHyphen(true); break;
195 case SET_T_ALPHA_POSSESSIVE : text.setAlphaPossessive(true); break;
196 case SET_T_CAPITAL : text.setAlphaCapital(T_CAPITAL); break;
197 case SET_T_SMALL : text.setAlphaCapital(T_SMALL); break;
199 case SET_T_ACRONYM : text.setAlphaCapital(T_ACRONYM); break;
201 case SET_T_ABBREV : text.setAlphaCapital(T_ABBREV); break;
204 case SET_T_NOT_ROMAN : text.setAlphaRoman(T_NOT_ROMAN); break;
205 case SET_T_INTEGER : text.setNumeric(T_INTEGER); break;
206 case SET_T_COMMA_NUMBER : text.setNumeric(T_COMMA_NUMBER); break;
207 case SET_T_DOT_NUMBER : text.setNumeric(T_DOT_NUMBER); break;
208 case SET_T_FRACTION : text.setNumeric(T_FRACTION); break;
210 case SET_T_ALPHA_CONCAT_ABBREV : text.setAlphaConcatAbbrev(true); break;
211 case SET_T_TWITTER : text.setTwitter(true); break;
214 case SET_T_LATIN_ARABIC : text.setDefaultKey(Common::Misc::utf8stdstring2limastring("t_latin_arabic")); break;
216 case SET_T_ACRONYM_ARABIC : text.setDefaultKey(Common::Misc::utf8stdstring2limastring("t_acronym_arabic")); break;
217 case SET_T_ACRONYM_LATIN_ARABIC : text.setDefaultKey(Common::Misc::utf8stdstring2limastring("t_acronym_latin_arabic")); break;
218 default: ;
219 }
220 }
221}
222
224{
225 std::string str = limastring2utf8stdstring(s);
226#ifdef DEBUG_LP
228 LDEBUG << " Setting transition status setting to " << str;
229#endif
230 if (str == "T_CAPITAL")
231 {
232 m_settings.push_back(SET_T_CAPITAL);
233 }
234 else if (str == "T_SMALL")
235 {
236 m_settings.push_back(SET_T_SMALL);
237 }
238 else if (str == "T_CAPITAL_1ST")
239 {
240 m_settings.push_back(SET_T_CAPITAL_1ST);
241 }
242 else if (str == "T_ACRONYM")
243 {
244 m_settings.push_back(SET_T_ACRONYM);
245 }
246 else if (str == "T_CAPITAL_SMALL")
247 {
248 m_settings.push_back(SET_T_CAPITAL_SMALL);
249 }
250 else if (str == "T_CARDINAL_ROMAN")
251 {
252 m_settings.push_back(SET_T_CARDINAL_ROMAN);
253 }
254 else if (str == "T_ORDINAL_ROMAN")
255 {
256 m_settings.push_back(SET_T_ORDINAL_ROMAN);
257 }
258 else if (str == "T_NOT_ROMAN")
259 {
260 m_settings.push_back(SET_T_NOT_ROMAN);
261 }
262 else if (str == "T_INTEGER")
263 {
264 m_settings.push_back(SET_T_INTEGER);
265 }
266 else if (str == "T_COMMA_NUMBER")
267 {
268 m_settings.push_back(SET_T_COMMA_NUMBER);
269 }
270 else if (str == "T_DOT_NUMBER")
271 {
272 m_settings.push_back(SET_T_DOT_NUMBER);
273 }
274 else if (str == "T_FRACTION")
275 {
276 m_settings.push_back(SET_T_FRACTION);
277 }
278 else if (str == "T_ORDINAL_INTEGER")
279 {
280 m_settings.push_back(SET_T_ORDINAL_INTEGER);
281 }
282 else if (str == "T_ALPHA")
283 {
284 m_settings.push_back(SET_T_ALPHA);
285 }
286 else if (str == "T_NUMERIC")
287 {
288 m_settings.push_back(SET_T_NUMERIC);
289 }
290 else if (str == "T_ALPHANUMERIC")
291 {
292 m_settings.push_back(SET_T_ALPHANUMERIC);
293 }
294 else if (str == "T_PATTERN")
295 {
296 m_settings.push_back(SET_T_PATTERN);
297 }
298 else if (str == "T_WORD_BRK")
299 {
300 m_settings.push_back(SET_T_WORD_BRK);
301 }
302 else if (str == "T_SENTENCE_BRK")
303 {
304 m_settings.push_back(SET_T_SENTENCE_BRK);
305 }
306 else if (str == "T_PARAGRAPH_BRK")
307 {
308 m_settings.push_back(SET_T_PARAGRAPH_BRK);
309 m_settings.push_back(SET_T_SENTENCE_BRK);
310 }
311 else if (str == "T_HYPHEN_WORD")
312 {
313 m_settings.push_back(SET_T_ALPHA_HYPHEN);
314 }
315 else if (str == "T_POSSESSIVE")
316 {
317 m_settings.push_back(SET_T_ALPHA_POSSESSIVE);
318 }
319 else if (str == "T_ALPHA_CONCAT_ABBREV")
320 {
321 m_settings.push_back(SET_T_ALPHA_CONCAT_ABBREV);
322 }
323 else if (str == "T_ARABIC")
324 {
325 m_settings.push_back(SET_T_ARABIC);
326 }
327 else if (str == "T_LATIN_ARABIC")
328 {
329 m_settings.push_back(SET_T_LATIN_ARABIC);
330 }
331 else if (str == "T_ART_DEF")
332 {
333 m_settings.push_back(SET_T_ART_DEF);
334 }
335 else if (str == "T_ACRONYM_ARABIC")
336 {
337 m_settings.push_back(SET_T_ACRONYM_ARABIC);
338 }
339 else if (str == "T_ACRONYM_LATIN_ARABIC")
340 {
341 m_settings.push_back(SET_T_ACRONYM_LATIN_ARABIC);
342 }
343 else if (str == "T_TWITTER")
344 {
345 m_settings.push_back(SET_T_TWITTER);
346 }
347 else if (str == "T_ABBREV")
348 {
349 m_settings.push_back(SET_T_ABBREV);
350 }
351 else
352 {
354 LERROR << "Transition::setSetting at " << __FILE__ << ", line " << __LINE__
355 << ": Unkown satus setting '"<<str<<"'";
356 return false;
357 }
358 return true;
359}
360
361} //namespace FlatTokenizer
362} // namespace LinguisticProcessing
363} // namespace Lima
#define LDEBUG
Definition LimaCommon.h:157
#define LERROR
Definition LimaCommon.h:161
#define TOKENIZERLOGINIT
#define TOKENIZERLOADERLOGINIT
const Lima::LimaString & name() const
Definition CharClass.h:27
bool isRecognized(const Lima::LimaChar &event) const
Definition Events.cpp:42
void setAlphaRoman(const LinguisticAnalysisStructure::AlphaRomanType alphaRoman)
Definition Text.cpp:304
void setAlphaConcatAbbrev(const unsigned char isConcatAbbreviation)
Definition Text.cpp:346
void setAlphaHyphen(const unsigned char isAlphaHyphen)
Definition Text.cpp:326
void setDefaultKey(const Lima::LimaString &defaultKey)
Definition Text.cpp:433
const CharClass * currentClass() const
Definition Text.cpp:176
void setNumeric(const LinguisticAnalysisStructure::NumericType numeric)
Definition Text.cpp:368
void setTwitter(const unsigned char isTwitter)
Definition Text.cpp:357
void setStatus(const LinguisticAnalysisStructure::StatusType status)
Definition Text.cpp:396
void setAlphaCapital(const LinguisticAnalysisStructure::AlphaCapitalType alphaCapital)
Definition Text.cpp:272
void setAlphaPossessive(const unsigned char isAlphaPossessive)
Definition Text.cpp:335
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
LimaString utf8stdstring2limastring(const std::string &src)
NAUTITIA.
QChar LimaChar
Definition LimaString.h:30
QString LimaString
Definition LimaString.h:33