LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
CharChart.cpp
Go to the documentation of this file.
1// Copyright 2002-2019 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6// NAUTITIA
7//
8// jys 21-JUL-2002
9//
10// Char is the array of valid characters. It is used to find
11// corresponding characters class, maj, min, unmark.
12// Performances at run-time are very important.
13
14#include "CharChart.h"
15#include "Char.h"
16#include "CharClass.h"
21
24
25#include <QtGlobal>
26
27#include <string>
28#include <algorithm>
29#include <fstream>
30
31#include <boost/spirit/include/qi_match.hpp>
32
33using namespace Lima;
34using namespace Common;
35using namespace Common::Misc;
36using namespace Common::MediaticData;
37using namespace Common::XMLConfigurationFiles;
38
39
40namespace Lima {
41namespace LinguisticProcessing {
42namespace FlatTokenizer {
43
45
46CharChart::CharChart() : AbstractResource(), m_classes(), m_chars(),
47 m_unicodeCategories(),
48 m_unicodeCategories2LimaClasses()
49{
50#if QT_VERSION < 0x050000
51 m_unicodeCategories
52 << "NoCategory";
53#endif
54 m_unicodeCategories
55 << "Mark_NonSpacing"
56 << "Mark_SpacingCombining"
57 << "Mark_Enclosing"
58 << "Number_DecimalDigit"
59 << "Number_Letter"
60 << "Number_Other"
61 << "Separator_Space"
62 << "Separator_Line"
63 << "Separator_Paragraph"
64 << "Other_Control"
65 << "Other_Format"
66 << "Other_Surrogate"
67 << "Other_PrivateUse"
68 << "Other_NotAssigned"
69 << "Letter_Uppercase"
70 << "Letter_Lowercase"
71 << "Letter_Titlecase"
72 << "Letter_Modifier"
73 << "Letter_Other"
74 << "Punctuation_Connector"
75 << "Punctuation_Dash"
76 << "Punctuation_Open"
77 << "Punctuation_Close"
78 << "Punctuation_InitialQuote"
79 << "Punctuation_FinalQuote"
80 << "Punctuation_Other"
81 << "Symbol_Math"
82 << "Symbol_Currency"
83 << "Symbol_Modifier"
84 << "Symbol_Other";
85
86// c_all, c_del, c_b, c_par, c_dot, c_comma, c_slash, c_hyphen, c_lowline, c_quote, c_fraction, c_percent, c_del1, c_plus, c_del2, c_Mm, c_degree, c_M, c_A, c_O, c_S, c_N, c_V, c_m, c_o, c_l_o, c_a, c_s, c_n, c_a_t, c_5, c_other, m_pattern, m_end_pattern, m_line, m_parag, unknwn
87// m_unicodeCategories2LimaClasses.insert("Mark_NonSpacing","");
88// m_unicodeCategories2LimaClasses.insert("Mark_SpacingCombining","");
89// m_unicodeCategories2LimaClasses.insert("Mark_Enclosing","");
90 m_unicodeCategories2LimaClasses.insert("Number_DecimalDigit","c_5");
91 m_unicodeCategories2LimaClasses.insert("Number_Letter","c_5");
92 m_unicodeCategories2LimaClasses.insert("Number_Other","c_5");
93 m_unicodeCategories2LimaClasses.insert("Separator_Space","c_b");
94 m_unicodeCategories2LimaClasses.insert("Separator_Line","c_par");
95 m_unicodeCategories2LimaClasses.insert("Separator_Paragraph","c_par");
96 m_unicodeCategories2LimaClasses.insert("Other_Control","c_b");
97// m_unicodeCategories2LimaClasses.insert("Other_Format","");
98// m_unicodeCategories2LimaClasses.insert("Other_Surrogate","");
99 m_unicodeCategories2LimaClasses.insert("Other_PrivateUse","c_hyphen");
100 m_unicodeCategories2LimaClasses.insert("Other_NotAssigned","unknwn");
101 m_unicodeCategories2LimaClasses.insert("Letter_Uppercase","c_M");
102 m_unicodeCategories2LimaClasses.insert("Letter_Lowercase","c_m");
103 m_unicodeCategories2LimaClasses.insert("Letter_Titlecase","c_M");
104 m_unicodeCategories2LimaClasses.insert("Letter_Modifier","c_Mm");
105 m_unicodeCategories2LimaClasses.insert("Letter_Other","c_Mm");
106 m_unicodeCategories2LimaClasses.insert("Punctuation_Connector","c_hyphen");
107 m_unicodeCategories2LimaClasses.insert("Punctuation_Dash","c_hyphen");
108 m_unicodeCategories2LimaClasses.insert("Punctuation_Open","c_par");
109 m_unicodeCategories2LimaClasses.insert("Punctuation_Close","c_par");
110 m_unicodeCategories2LimaClasses.insert("Punctuation_InitialQuote","c_quote");
111 m_unicodeCategories2LimaClasses.insert("Punctuation_FinalQuote","c_quote");
112 m_unicodeCategories2LimaClasses.insert("Punctuation_Other","c_dot");
113 m_unicodeCategories2LimaClasses.insert("Symbol_Math","c_plus");
114 m_unicodeCategories2LimaClasses.insert("Symbol_Currency","c_del1");
115 m_unicodeCategories2LimaClasses.insert("Symbol_Modifier","c_del1");
116 m_unicodeCategories2LimaClasses.insert("Symbol_Other","c_del1");
117}
118
120{
121 std::vector<CharClass*>::iterator itcc, itcc_end;
122 itcc = m_classes.begin(); itcc_end = m_classes.end();
123 for (; itcc != itcc_end; itcc++)
124 {
125 delete *itcc;
126 }
127 std::vector<Char*>::iterator itc, itc_end;
128 itc = m_chars.begin(); itc_end = m_chars.end();
129 for (; itc != itc_end; itc++)
130 {
131 delete *itc;
132 }
133}
134
135
138 Manager* manager)
139
140{
142 LDEBUG << "Creating a CharChart (loads file)";
143 MediaId language=manager->getInitializationParameters().language;
144
145 try {
146 QString charChartFileName=Common::Misc::findFileInPaths(Common::MediaticData::MediaticData::single().getResourcesPath().c_str(),unitConfiguration.getParamsValueAtKey("charFile").c_str());
147 loadFromFile(charChartFileName.toUtf8().constData());
148
150 {
151 LERROR << "no parameter 'charFile' in charchart group for language " << (int) language << " !";
152 throw InvalidConfiguration();
153 }
154}
155
156
157
158// Gets the Class of the specified character.
159// If specified character was not defined,
160// try to use the mapping with Qt Unicode categories and if this fails too,
161// class named unknwn is returned
163{
164 if (c.unicode() >= m_chars.size() || m_chars[c.unicode()] == 0 || m_chars[c.unicode()]->charClass() == 0)
165 {
166 if (c.category() < m_unicodeCategories.size())
167 {
168 QString unicodeCategory = m_unicodeCategories[c.category()];
169 if (m_unicodeCategories2LimaClasses.contains(unicodeCategory))
170 {
171#ifdef DEBUG_LP
173 LDEBUG << "CharChart::charClass using unicode category" << unicodeCategory << "and LIMA class" << m_unicodeCategories2LimaClasses[unicodeCategory] ;
174#endif
175 return classNamed(m_unicodeCategories2LimaClasses[unicodeCategory]);
176 }
177 }
179 LNOTICE << "CharChart::charClass undefined char: " << c;
180 return classNamed(utf8stdstring2limastring("unknwn"));
181 }
182#ifdef DEBUG_LP
184 LTRACE << "CharChart::charClass" << c << c.unicode() << m_chars.size() << m_chars[c.unicode()]<< m_chars[c.unicode()]->charClass();
185#endif
186 return m_chars[c.unicode()]->charClass();
187}
188
189const CharClass* CharChart::charClass (const LimaChar& c1, const LimaChar& c2) const
190{
191 if (m_surrogates.find(c1) == m_surrogates.end())
192 {
194 LNOTICE << "CharChart::charClass undefined 32 bytes char: " << c1 << c2 ;
195 return classNamed(utf8stdstring2limastring("unknwn"));
196 }
197 const std::map<LimaChar,Char*>& c1classes = (*(m_surrogates.find(c1))).second;
198 if (c1classes.find(c2) == c1classes.end())
199 {
201 LNOTICE << "CharChart::charClass undefined 32 bytes char: " << c1 << c2 ;
202 return classNamed(utf8stdstring2limastring("unknwn"));
203 }
204 else
205 {
206 return (*(c1classes.find(c2))).second->charClass();
207 }
208}
209
210// Gets the upper case corresponding of the specified
211// character.
213{
214 return c.toUpper();
215}
216
217// Gets the lower case corresponding of the specified
218// character.
220{
221 return c.toLower();
222}
223
224// Gets the unmark character corresponding to the specified
225// character. If specified character was not defined,
226// InvalidCharException is raised.
228#ifdef DEBUG_LP
230#endif
231 if (c.unicode() >= m_chars.size())
232 throw InvalidCharException();
233 if (m_chars[c.unicode()] == 0)
234 return LimaChar();
235 if (m_chars[c.unicode()]->charClass() == 0)
236 throw InvalidCharException();
237 if (m_chars[c.unicode()]->longUnmark() != 0)
238 return LimaChar();
239 if (m_chars[c.unicode()]->min() != 0
240 && m_chars[c.unicode()]->min()->unmark() != 0)
241 return (*(m_chars[c.unicode()]->min()->unmark()))();
242 if (m_chars[c.unicode()]->unmark() != 0)
243 return (*(m_chars[c.unicode()]->unmark()))();
244 return LimaChar();
245}
246
247// Gets the unmark string corresponding of the specified
248// character. If specified character was not defined,
249// InvalidCharException is raised.
250// Null string, one, two and more characters string can be returned.
252{
253#ifdef DEBUG_LP
255 LDEBUG << "CharChart::unmarkByString" << c;
256#endif
257 if (c.unicode() >= m_chars.size())
258 throw InvalidCharException();
259 if (m_chars[c.unicode()] == 0)
260 {
261 LimaString result;
262 result.push_back(c);
263#ifdef DEBUG_LP
264 LDEBUG << "CharChart::unmarkByString" << result;
265#endif
266 return result;
267 }
268 if (!m_chars[c.unicode()]->charClass())
269 throw InvalidCharException();
270
271 LimaString result;
272 if (m_chars[c.unicode()]->unmark() != 0 && m_chars[c.unicode()]->unmark() != m_chars[c.unicode()])
273 result.push_back(m_chars[c.unicode()]->unmark()->code());
274 if (m_chars[c.unicode()]->longUnmark() != 0 && m_chars[c.unicode()]->longUnmark() != m_chars[c.unicode()])
275 result.push_back(m_chars[c.unicode()]->longUnmark()->code());
276#ifdef DEBUG_LP
277 LDEBUG << "CharChart::unmarkByString" << result;
278#endif
279 return result;
280}
281
283{
284 LimaString desaccented;
285 desaccented.reserve(str.size());
286 for (int i = 0; i < str.size(); i++)
287 {
288 try
289 {
290 LimaChar chr = unmark(str.at(i));
291 if (!chr.isNull())
292 {
293 desaccented.push_back(chr);
294 if (m_chars[str.at(i).unicode()]->hasLongUnmark())
295 desaccented.push_back(m_chars[str.at(i).unicode()]->longUnmark()->code());
296
297 }
298 else
299 {
300 LimaString s = unmarkByString(str.at(i));
301 if (!s.isEmpty())
302 desaccented.append(s);
303// else
304// desaccented.push_back(str.at(i));
305 }
306 }
307 // silently discard invalid character
308 catch (InvalidCharException&) {}
309 }
310 return desaccented;
311}
312
313LimaString CharChart::unmarkWithMapping(const LimaString& str,std::vector<unsigned char>& mapping) const
314{
315 LimaString desaccented;
316 desaccented.reserve(str.size());
317 mapping.resize(str.size());
318 LimaChar chr;
319 LimaString s;
320 unsigned char desaccIndex=0;
321 for (int i = 0; i < str.size(); i++)
322 {
323 try
324 {
325 chr = unmark(str.at(i));
326 if (!chr.isNull()) {
327 desaccented.push_back(chr);
328 mapping[i]=desaccIndex;
329 desaccIndex++;
330 }
331 else
332 {
333 s = unmarkByString(str.at(i));
334 desaccented.append(s);
335 mapping[i]=desaccIndex;
336 desaccIndex+=s.size();
337 }
338 }
339 // discard invalid character
340 catch (InvalidCharException&) {}
341 }
342 return desaccented;
343}
344
345
348{
349#ifdef DEBUG_LP
351 LDEBUG << "toLower("<<src<<") = " << src.toLower();
352#endif
353 return src.toLower();
354}
355
357{
358// #ifdef DEBUG_LP
359// TOKENIZERLOGINIT;
360// LDEBUG << "Searching class " << limastring2utf8stdstring(name);
361// #endif
362 std::vector<CharClass*>::const_iterator itcc, itcc_end;
363 itcc = m_classes.begin(); itcc_end = m_classes.end();
364 for (; itcc != itcc_end; itcc++)
365 {
366// LDEBUG << " looking at " << limastring2utf8stdstring((*itcc)->id());
367 if ( (*itcc)->id() == name)
368 {
369// LDEBUG << "Found " << limastring2utf8stdstring((*itcc)->id());
370 return (*itcc);
371 }
372 }
373 {
375 LERROR << "CharChart::classNamed "<<Common::Misc::limastring2utf8stdstring(name)<<" NOT Found ";
376 }
377 return 0;
378}
379
380bool CharChart::loadFromFile(const std::string& fileName)
381{
382#ifdef DEBUG_LP
384 LDEBUG << "Loading CharChart from " << fileName;
385#endif
386 std::ifstream file(fileName.c_str(), std::ifstream::binary);
387 if (!file.good())
388 {
390 LERROR << " CharChart::loadFromFile Unable to open" << fileName;
391 return false;
392 }
393 std::string str;
394 Common::Misc::readStream(file, str);
395
396 std::string::const_iterator iter = str.begin();
397 std::string::const_iterator end = str.end();
398
399// typedef std::string::const_iterator iterator_type;
400// typedef LinguisticProcessing::FlatTokenizer::charchart_parser<iterator_type> charchart_parser;
401
405
406 bool r = phrase_parse(iter, end, parser, skipper, charchart);
407 if (r)
408 {
409 std::vector<charchart_class>::const_iterator classIt, classItend;
410 classIt = charchart.classes.begin(); classItend = charchart.classes.end();
411 for (; classIt != classItend; classIt++)
412 {
413 const charchart_class& ch = *classIt;
414 CharClass* newClass = new CharClass();
417 if (ch.parent.is_initialized() && !ch.parent.get().empty()
419 {
421 }
422#ifdef DEBUG_LP
423 if (newClass->superClass() != 0)
424 {
425 LDEBUG << " Loaded class " << ch.name << " < " << Common::Misc::limastring2utf8stdstring(newClass->superClass()->id());
426 }
427 else
428 {
429 LDEBUG << " Loaded class " << ch.name << " < NONE";
430 }
431#endif
432 m_classes.push_back(newClass);
433 }
434 std::vector<charchart_char>::const_iterator charIt, charItend;
435 charIt = charchart.chars.begin(); charItend = charchart.chars.end();
436 for (; charIt != charItend; charIt++)
437 {
438 const charchart_char& ch = *charIt;
439 Char* newChar = lazyGetChar(QChar(ch.code));
440 if (newChar == 0)
441 {
443 LERROR << "Error loading char '" << ch.code << "'";
444 continue;
445 }
447 if (newCharClass == 0)
448 {
450 LERROR << "Error loading char '" << ch.code << "' : unknown class '" << ch.charclass << "'";
451 continue;
452 }
453 newChar->setCharClass(newCharClass);
455#ifdef DEBUG_LP
456 LDEBUG << "Adding char" << newChar->name() << newCharClass->name();
457#endif
458 }
459 charIt = charchart.chars.begin(); charItend = charchart.chars.end();
460 for (; charIt != charItend; charIt++)
461 {
462 const charchart_char& ch = *charIt;
463 Char* newChar = lazyGetChar(QChar(ch.code));
464#ifdef DEBUG_LP
465 LDEBUG << "Modifiers for" << newChar->name();
466#endif
467// const CharClass* newCharClass = classNamed(Common::Misc::utf8stdstring2limastring(ch.charclass));
468 std::vector<modifierdef>::const_iterator modIt, modItEnd;
469 modIt = ch.modifiers.begin(); modItEnd = ch.modifiers.end();
470 for (; modIt != modItEnd; modIt++)
471 {
472#ifdef DEBUG_LP
473 LDEBUG << " modifier "<< (*modIt).first <<":" << lazyGetChar(QChar{(*modIt).second})->name();
474#endif
475 switch ( (*modIt).first )
476 {
477 case MIN:
478 newChar->setMin(lazyGetChar(QChar{(*modIt).second}));
479 break;
480 case MAJ:
481 newChar->setMaj(lazyGetChar(QChar{(*modIt).second}));
482 break;
483 case UNMARK:
484 newChar->setUnmark(lazyGetChar(QChar{(*modIt).second}));
485 break;
486 default: ;
487 }
488 }
489 if (newChar->code().isHighSurrogate())
490 {
491 if (surrogates().find(newChar->code()) == surrogates().end())
492 {
493 surrogates().insert(std::make_pair(newChar->code(),std::map< LimaChar, Char* >()));
494 }
495 surrogates()[newChar->code()].insert(std::make_pair(newChar->surrogate(),newChar));
496 }
497 }
498 }
499 else
500 {
502 LERROR << "Error while parsing: " << fileName;
503 }
504 return true;
505}
506
507Char* CharChart::lazyGetChar(const LimaChar& code)
508{
509 Char* newChar = 0;
510// if (code <= 0xD800)
511// {
512 if (chars().size() <= code.unicode())
513 {
514 chars().resize(code.unicode()+1);
515 }
516 if (chars()[code.unicode()] == 0)
517 {
518 newChar = new Char(code);
519 chars()[code.unicode()] = newChar;
520 }
521 else
522 {
523 newChar = chars()[code.unicode()];
524 }
525// }
526 return newChar;
527}
528
529} // Tokenizer
530} // LinguisticProcessing
531} // Lima
#define FLATTOKENIZERCHARCHART_CLASSID
Definition CharChart.h:29
#define LTRACE
Definition LimaCommon.h:156
#define LDEBUG
Definition LimaCommon.h:157
#define LERROR
Definition LimaCommon.h:161
#define LNOTICE
Definition LimaCommon.h:159
#define TOKENIZERLOGINIT
#define TOKENIZERLOADERLOGINIT
Defines a Factory to create Object of type Base.
#define skipper
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Use this exception to signal an error in one of the configuration files.
Definition LimaCommon.h:345
Lima::LimaChar unmark(const Lima::LimaChar &c) const
Gets the unmark character corresponding to the specified character.
LimaString unmarkWithMapping(const LimaString &str, std::vector< unsigned char > &mapping) const
Lima::LimaString unmarkByString(const Lima::LimaChar &c) const
Gets the unmark string corresponding to the specified character.
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &, Manager *) override
const CharClass * classNamed(const Lima::LimaString &name) const
const CharClass * charClass(const Lima::LimaChar &c) const
Gets the Class of the specified character.
Lima::LimaString toLower(const Lima::LimaString &src) const
Converts src to its lowercase equivalent.
Lima::LimaChar maj(const Lima::LimaChar &c) const
Gets the upper case corresponding to the specified character.
const std::map< LimaChar, std::map< LimaChar, Char * > > & surrogates() const
Definition CharChart.h:88
bool loadFromFile(const std::string &fileName)
Lima::LimaChar min(const Lima::LimaChar &c) const
Gets the lower case corresponding to the specified character.
const std::vector< Char * > & chars() const
Definition CharChart.h:43
void setName(const Lima::LimaString &name)
Definition CharClass.h:31
void setId(const Lima::LimaString &id)
Definition CharClass.h:30
const Lima::LimaString & name() const
Definition CharClass.h:27
const Lima::LimaString & name() const
Definition Char.cpp:61
void setName(const Lima::LimaString &name)
Definition Char.cpp:66
void setCharClass(const CharClass *cl)
Definition Char.cpp:67
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
void readStream(std::istream &is, std::string &dest)
Read an input stream into a string, until EOF.
QString findFileInPaths(const QString &paths, const QString &fileName, const QChar &separator)
Find the given file in the given paths.
LimaString utf8stdstring2limastring(const std::string &src)
SimpleFactory< AbstractResource, CharChart > flatTokenizerCharChartFactory(FLATTOKENIZERCHARCHART_CLASSID)
NAUTITIA.
QChar LimaChar
Definition LimaString.h:30
QString LimaString
Definition LimaString.h:33
std::vector< modifierdef > modifiers
boost::optional< std::string > parent
std::vector< charchart_char > chars
std::vector< charchart_class > classes