LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
OrthographicAlternatives.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6// NAUTITIA
7//
8// jys 25-NOV-2002
9//
10// OrthographicAlternatives is the module which creates alternatives
11// for given tokens. It "unmarks" the string then searchs into dictionnary
12// with this new entry for new reaccented words.
13// Each token from the supplied tokens path is processed.
14// There are 2 modes :
15// o confident mode : only tokens unknown into dictionnary as simple word
16// are processed
17// o unconfident mode : all tokens are processed.
18
20
21// #include "common/linguisticData/linguisticData.h"
22#include "common/misc/traceUtils.h"
30
34using namespace Lima::Common::LinguisticData;
36using namespace std;
37
38namespace Lima
39{
40namespace LinguisticProcessing
41{
42namespace MorphologicAnalysis
43{
44
46
49
52
55 Manager* manager)
56{
58 m_language = manager->getInitializationParameters().language;
59 try
60 {
61 string dico=unitConfiguration.getParamsValueAtKey("dictionary");
62 auto res = LinguisticResources::single().getResource(m_language,dico);
63 m_dictionary=static_cast<AbstractAnalysisDictionary*>(res);
64 }
65 catch (NoSuchParam& )
66 {
67 LERROR << "no param 'dictionary' in OrthographicAlternatives group for language " << (int) m_language;
69 }
70
71 try
72 {
73 string dico=unitConfiguration.getParamsValueAtKey("charChart");
74 auto res = LinguisticResources::single().getResource(m_language,dico);
75 m_charChart=static_cast<CharChart*>(res);
76 }
77 catch (NoSuchParam& )
78 {
79 LERROR << "no param 'charChart' in OrthographicAlternatives group for language " << (int) m_language;
81 }
82
83 try
84 {
85 string confident=unitConfiguration.getParamsValueAtKey("confidentMode");
86 m_confidentMode=(confident=="true");
87 }
88 catch (NoSuchParam& )
89 {
90 LWARN << "no param 'confidentMode' in OrthographicAlternatives group for language " << (int) m_language;
91 LWARN << "use default value : 'true'";
92 m_confidentMode=true;
93 }
94
95}
96
97
99 AnalysisContent& analysis) const
100{
101
104 LINFO << "MorphologicalAnalysis: starting process OrthographicAlternatives";
105
106 StringsPool& sp=Common::LinguisticData::LinguisticData::changeable().stringsPool(m_language);
107 auto tokenList=std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"));
108 LinguisticGraph* g=tokenList->getGraph();
110 VertexDataPropertyMap dataMap=get(vertex_data,*g);
111 VertexTokenPropertyMap tokenMap=get(vertex_token,*g);
112 boost::tie(it,itEnd)=vertices(*g);
113 for (;it!=itEnd;it++)
114 {
115 LDEBUG << "processing vertex " << *it;
116 MorphoSyntacticData* currentTokenData=dataMap[*it];
117 Token* tok=tokenMap[*it];
118 if (currentTokenData!=0)
119 {
120
121 // if in confidentMode and token has already ling infos, skip
122 if ( m_confidentMode && (currentTokenData->size()>0) ) continue;
123
124 // set orthographic alternatives given by dictionary
125 // using the alternatives directly given by the morphosyntactic data
126 {
127 LDEBUG << "processing alternatives from dico";
128 DictionaryEntry* entry=tok->dictionaryEntry();
129 entry->reset();
130 if (entry->hasAccented()) {
131 LimaString oa = entry->nextAccented();
132 while ( oa.size() > 0 )
133 {
134 createAlternative(tok,currentTokenData,oa,m_dictionary,sp);
135 oa = entry->nextAccented();
136 }
137 }
138 }
139
140 // if in confidentMode and token has already ling infos, skip
141 if (m_confidentMode && (currentTokenData->size() > 0) ) continue;
142
143 // if no ling infos, then lower and unmark string
144 LDEBUG << "set unmark alternatives";
146 tok,
147 currentTokenData,
148 m_dictionary,
149 m_charChart,
150 sp);
151 }
152 }
153 LINFO << "MorphologicalAnalysis: ending process OrthographicAlternatives";
154 TimeUtils::logElapsedTime("OrthographicAlternatives");
155 return SUCCESS_ID;
156}
157
159 Token* token,
160 MorphoSyntacticData* tokenData,
162 CharChart* charChart,
163 StringsPool& sp)
164{
165 // try to find simple Uncapitalization
167 const LimaString& tokenStr=token->stringForm();
168 LimaString lowerWord = charChart->toLower(tokenStr);
169 if (!(lowerWord == "") && !(lowerWord == tokenStr) )
170 {
171 LDEBUG << "createAlternative for lowerWord " << lowerWord;
172 createAlternative(token,tokenData,lowerWord,dictionary,sp);
173 }
174 if (tokenData->size()>0)
175 {
176 return;
177 }
178
179 // desaccent token string
180 LimaString unmarked=charChart->unmark(tokenStr);
181 if (!(unmarked=="") && !(unmarked==tokenStr))
182 {
183 LDEBUG << "createAlternative for unmarked " << unmarked;
184 createAlternative(token,tokenData,unmarked,dictionary,sp);
185 }
186
187}
188
189
191 Token* srcToken,
192 MorphoSyntacticData* tokenData,
193 LimaString& str,
195 StringsPool& sp)
196{
198 LDEBUG << "OrthographicAlternatives::createAlternative" << str;
199 DictionaryEntry* dicoEntry = new DictionaryEntry(dictionary->getEntry(str));
200 if (!dicoEntry->isEmpty())
201 {
202 // add orthographic alternative to Token;
203 StringsPoolIndex infl=sp[str];
204 Token* altToken=new Token(infl,str,srcToken->position(),srcToken->length(),new TStatus(*(srcToken->status())));
205 altToken->setDictionaryEntry(dicoEntry);
206 srcToken->addOrthographicAlternative(altToken);
207
208 tokenData->appendLingInfo(infl,dicoEntry,ORTHOGRAPHIC_ALTERNATIVE,sp);
209
210 // if entry has other accented forms,
211 // keep them ("PARIS" -> "paris" -> "Paris")
212 if (dicoEntry->hasAccented())
213 {
214 dicoEntry->reset();
215 Lima::LimaString alternativeStr = dicoEntry->nextAccented();
216 while (alternativeStr.size() != 0)
217 {
218 // give it its simple word entry into dictionary
219 DictionaryEntry* altDicoEntry = new DictionaryEntry(dictionary->getEntry(alternativeStr));
220 StringsPoolIndex infl2=sp[alternativeStr];
221 tokenData->appendLingInfo(infl2,altDicoEntry,ORTHOGRAPHIC_ALTERNATIVE,sp);
222
223 // add orthographic alternative to Token
224 Token* altToken2=new Token(infl2,alternativeStr,srcToken->position(),srcToken->length(),new TStatus(*(srcToken->status())));
225 altToken2->setDictionaryEntry(altDicoEntry);
226 srcToken->addOrthographicAlternative(altToken2);
227
228 alternativeStr = dicoEntry->nextAccented();
229 }
230 }
231 } else {
232 delete dicoEntry;
233 }
234}
235
236
237} // MorphologicAnalysis
238} // LinguisticProcessing
239} // Lima
#define LWARN
Definition LimaCommon.h:160
#define LDEBUG
Definition LimaCommon.h:157
#define LINFO
Definition LimaCommon.h:158
#define LERROR
Definition LimaCommon.h:161
boost::property_map< LinguisticGraph, vertex_data_t >::type VertexDataPropertyMap
LinguisticGraph::vertex_iterator LinguisticGraphVertexIt
boost::property_map< LinguisticGraph, vertex_token_t >::type VertexTokenPropertyMap
@ vertex_token
@ vertex_data
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define MORPHOLOGINIT
#define ORTHOGRAPHALTERNATIVES_CLASSID
Defines a Factory to create Object of type Base.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Use this exception to signal an error in one of the configuration files.
Definition LimaCommon.h:345
virtual DictionaryEntry getEntry(const Lima::LimaString &word) const =0
get dictionary entry for a word
Lima::LimaChar unmark(const Lima::LimaChar &c) const
Gets the unmark character corresponding to the specified character.
Lima::LimaString toLower(const Lima::LimaString &src) const
Converts src to its lowercase equivalent.
LimaStatusCode process(AnalysisContent &analysis) const
Process on data in analysisContent.
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager)
initialize with parameters from configuration file.
static void setOrthographicAlternatives(LinguisticAnalysisStructure::Token *token, LinguisticAnalysisStructure::MorphoSyntacticData *tokenData, AnalysisDict::AbstractAnalysisDictionary *dictionary, FlatTokenizer::CharChart *charChart, FsaStringsPool &sp)
static void createAlternative(LinguisticAnalysisStructure::Token *srcToken, LinguisticAnalysisStructure::MorphoSyntacticData *tokenData, LimaString &str, AnalysisDict::AbstractAnalysisDictionary *dictionary, FsaStringsPool &sp)
static const LinguisticResources & single()
const singleton accessor
Definition Singleton.h:51
static void logElapsedTime(const std::string &mess, const std::string &taskCategory=std::string(""))
log the number of microseconds since last UpdateCurrentTime
static void updateCurrentTime(const std::string &taskCategory=std::string(""))
store current time for new elapsed time computation
SimpleFactory< LinguisticProcessUnit, OrthographicAlternatives > orthographicAlternativeFactory(ORTHOGRAPHALTERNATIVES_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
QString LimaString
Definition LimaString.h:33
STL namespace.
launch exception related to the configuration file parsing