LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
NormalizeNumber.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/************************************************************************
7 *
8 * @file NormalizeNumber.cpp
9 * @author Besancon Romaric (romaric.besancon@cea.fr)
10 * @date Tue Jun 13 2006
11 * copyright Copyright (C) 2006-2012 by CEA LIST
12 *
13 ***********************************************************************/
14
15#include "NormalizeNumber.h"
16#include "NormalizationUtils.h"
24#include <cmath>
25
26using namespace Lima::Common::MediaticData;
27using namespace Lima::Common::AnnotationGraphs;
30using namespace std;
31
32namespace Lima {
33namespace LinguisticProcessing {
34namespace SpecificEntities {
35
36#define NUMVALUE_FEATURE_NAME "numvalue"
37#define UNIT_FEATURE_NAME "unit"
38
39//**********************************************************************
40// factories for actions defined in this file
43
44//**********************************************************************
46NormalizeNumber(MediaId language,
47 const LimaString& complement):
48Automaton::ConstraintFunction(language,complement),
49m_language(language),
50m_microsForNumber(nullptr),
51m_microsForUnit(nullptr),
52m_microsForConjunction(nullptr),
53m_microAccessor(nullptr)
54{
55 m_microAccessor=&(static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(language)).getPropertyCodeManager().getPropertyAccessor("MICRO"));
56
57 if (language != UNDEFLANG) {
58 try {
59 auto res = LinguisticResources::single().getResource(language,"microsForNumberNormalization");
60 auto micros = std::dynamic_pointer_cast<MicrosForNormalization>(res);
61 m_microsForNumber=micros->getMicros("NumberMicros");
62 m_microsForUnit=micros->getMicros("UnitMicros");
63 m_microsForConjunction=micros->getMicros("ConjCoordMicros");
64 }
65 catch (exception& e) {
67 LWARN << "Exception caught: " << e.what();
68 LWARN << "-> micros for number normalization are not initialized";
69 }
70 }
71}
72
73/***********************************************************************/
74// specific helper functions for the normalization of numbers
75// not part of the class: are language independant
76
77// some local typedefs
78 // use a single double: is 0, indicates it is a conjunction
79typedef double NumberPart;
80// current state: must add or multiply with following value
82
83bool isFraction(Token* t);
86bool isIntegerWithDot(Token* t);
87double getValueHeuristic(const LimaString& str, LimaChar sep);
88bool isMultiplierNumber(double n);
89double computeNumberValue(vector<NumberPart>& m,
90 vector<NumberPart>::iterator itBegin,
91 vector<NumberPart>::iterator itEnd,
93
94double NormalizeNumber::
95getNumberValue(Token* t,MorphoSyntacticData* data) const
96{
97 if (isInteger(t))
98 { // t is in numeric format
99 return LimaStringToInt(t->stringForm());
100 }
101 else if (isIntegerWithComma(t))
102 {
103 return getValueHeuristic(t->stringForm(),LimaChar(','));
104 }
105 else if (isIntegerWithDot(t))
106 {
107 return getValueHeuristic(t->stringForm(),LimaChar('.'));
108 }
109
110 // get the normalized form : contains the numeric form
111 // (normalized form corresponding to a macro_micro identifying a number)
112 MorphoSyntacticData::const_iterator
113 it=data->begin(),
114 it_end=data->end();
115 for (; it!=it_end; it++) {
116 if (testMicroCategory(m_microsForNumber,m_microAccessor,(*it).properties)) {
117 const LimaString& str=
118 Common::MediaticData::MediaticData::single().stringsPool(m_language)[(*it).normalizedForm];
119 bool ok;
120 double d = str.toDouble(&ok);
121 if(ok) {
122#ifdef DEBUG_LP
123 SELOGINIT;
124 LDEBUG << "getNumberValue of " << t->stringForm() << ". Found number "<< d;
125#endif
126 return d;
127 }
128 }
129 }
130#ifdef DEBUG_LP
131 SELOGINIT;
132 LDEBUG << "getNumberValue of " << t->stringForm() << ". Could not read number.";
133#endif
134 return 0;
135}
136
137/***********************************************************************/
138// normalization of a number
139/***********************************************************************/
142 AnalysisContent& analysis) const
143{
144#ifdef DEBUG_LP
145 SELOGINIT;
146 LDEBUG << "NormalizeNumber " << m;
147#endif
148
149 // annotation data is used to get numeric value of already recognized number entities
150 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.getData("AnnotationData"));
151
152 vector<NumberPart> values;
153
154 // a first pass on the match to eliminate what isn't a number
155 for (RecognizerMatch::iterator it=m.begin(),it_end=m.end();
156 it!=it_end; it++) {
157
158 if (! (*it).isKept()) {
159 continue;
160 }
161
162 // find if vertex is a specific entity of type number
163 // in this case, get the numeric value from the features
164 bool hasNumericValue(false);
165 std::set< AnnotationGraphVertex > matches = annotationData->matches(m.getGraph()->getGraphId(),(*it).m_elem.first,"annot");
166 for (std::set< AnnotationGraphVertex >::const_iterator annot = matches.begin(),
167 annot_end=matches.end(); annot != annot_end; annot++) {
168 if (annotationData->hasAnnotation(*annot, Common::Misc::utf8stdstring2limastring("SpecificEntity"))) {
169#ifdef DEBUG_LP
170 LDEBUG << "NormalizeNumber: vertex " << (*it).m_elem.first << " has specific entity annotation";
171#endif
172 const SpecificEntityAnnotation* se =
173 annotationData->annotation(*annot, Common::Misc::utf8stdstring2limastring("SpecificEntity")).
174 pointerValue<SpecificEntityAnnotation>();
175 const EntityFeatures& features=se->getFeatures();
176 for (EntityFeatures::const_iterator f=features.begin(),f_end=features.end();
177 f!=f_end; f++) {
178#ifdef DEBUG_LP
179 LDEBUG << "NormalizeNumber: looking at feature " << (*f).getName();
180#endif
181 if ((*f).getName()==NUMVALUE_FEATURE_NAME) {
182 try {
183 double value=boost::any_cast<double>((*f).getValue());
184#ifdef DEBUG_LP
185 LDEBUG << "NormalizeNumber: add value " << value;
186#endif
187 values.push_back(value);
188 hasNumericValue=true;
189 }
190 catch (const boost::bad_any_cast& e) {
191 // check if it is a string containing a number (may be the case if a setEntityFeature
192 // has been used explicitely in a rule
193 std::string strval=(*f).getValueString();
194 char *end;
195 double val = std::strtod(strval.c_str(), &end);
196 if (val!=0 || strval=="0") {
197 values.push_back(val);
198 hasNumericValue=true;
199 }
200 else {
201 SELOGINIT;
202 LERROR << "Error: failed to get numeric value for feature" << (*f).getName() << ":" << (*f).getValueString();
203 }
204 }
205 }
206 }
207 }
208 }
209 if (hasNumericValue) {
210 continue;
211 }
212 // find the numeric value in the normalization form of tokens
213 Token* t = m.getToken(it);
214 MorphoSyntacticData* data = m.getData(it);
215 double value=getNumberValue(t,data);
216 if (value != 0.0) { // it is a number
217 values.push_back(value);
218 }
219 else if (testMicroCategory(m_microsForConjunction,m_microAccessor,data)) {
220 // conjunction needed to compute the value
221 values.push_back(0.0);
222 }
223 else if (testMicroCategory(m_microsForUnit,m_microAccessor,data)) {
224#ifdef DEBUG_LP
225 LDEBUG << "NormalizeNumber: add feature UNIT " << t->stringForm();
226#endif
228 }
229 // ignore other non numbers (can be % or "de"...)
230 }
231
232 if (values.empty()) {
233 SELOGINIT;
234 LWARN << "Warning: cannot normalize number \"" << m.getString()
235 << "\": no numeric information available for components";
237 // return true even if value could not be computed: a value has been set in
238 // features
239 return true;
240 }
241
242 // then compute the number
243 double number=computeNumberValue(values,values.begin(),values.end());
244#ifdef DEBUG_LP
245 LDEBUG << "NormalizeNumber: add feature NUMVALUE " << number;
246#endif
249 return true;
250}
251
252//**********************************************************************
253// definitions of helper functions
255 const TStatus& status=t->status();
256 switch(status.getNumeric()) {
257 case T_COMMA_NUMBER:
258 case T_DOT_NUMBER: return true;
259 default: return false;
260 }
261}
262
264 const TStatus& status=t->status();
265 switch(status.getNumeric()) {
266 case T_COMMA_NUMBER: return true;
267 default: return false;
268 }
269}
270
272 const TStatus& status=t->status();
273 switch(status.getNumeric()) {
274 case T_DOT_NUMBER: return true;
275 default: return false;
276 }
277}
278
280 const TStatus& status=t->status();
281 return (status.getNumeric() == T_FRACTION);
282}
283
284double getValueHeuristic(const LimaString& str, LimaChar sep) {
285 // heuristic to find the value of a number with separator '.' or ','
286 // -> should be language dependent !... (in TODO list)
287
288 // if there is at least two separators -> 20.000.000 20,000,000
289 //uint64_t firstSep(str.find(sep,0)); portage 32 64
290 int firstSep(str.indexOf(sep));
291 if (firstSep == -1) { // error
292 return LimaStringToInt(str);
293 }
294
295 //uint64_t nextSep(str.find(sep,firstSep+1)); portage 32 64
296 int nextSep(str.indexOf(sep,firstSep+1));
297
298 if (nextSep == -1) {
299 // no second separator
300 LimaString beforeSep = str.left(firstSep);
301 LimaString afterSep = str.mid(firstSep+1);
302
303 if (afterSep.size() == 3) {
304 // three digits after separator -> assume 1000
305 return ( (LimaStringToInt(beforeSep)*1000)
306 +
307 LimaStringToInt(afterSep) );
308 }
309 else { // assume decimal part
310 return ( LimaStringToInt(beforeSep) +
311 ( (double)LimaStringToInt(afterSep)
312 /pow((double)10,(int)afterSep.size())) ) ;
313 }
314 }
315 else {
316 // there is at least a second separator -> should be multiplier
317 LimaString beforeSep = str.left(firstSep);
318 double result(beforeSep.toDouble());
319 LimaString afterSep;
320 while (nextSep != -1) {
321 afterSep=str.mid(firstSep+1,nextSep-firstSep-1);
322 if (afterSep.size() != 3 ) {
323 // error : cant be multiplier -> don't know what it is
324 return 0;
325 }
326 result = result*1000 + afterSep.toDouble();
327 firstSep = nextSep;
328 nextSep = str.indexOf(sep,firstSep+1);
329 }
330 // last part
331 afterSep=str.mid(firstSep+1);
332 if (afterSep.size() != 3 ) {
333 // error : cant be multiplier -> don't know what it is
334 return 0;
335 }
336 result = result*1000 + LimaStringToInt(afterSep);
337 return result;
338 }
339}
340
341// !!! I doubt that this is language independant !!!...
342bool isMultiplierNumber(double n) {
343 if (n==100 ||
344 n==1000 ||
345 n==1000000 ||
346 n==1000000000) {
347 return true;
348 }
349 return false;
350}
351
352// recursive function to compute the value of a composed number
353// (for instance "trente trois mille vingt et un" or "22 millions")
354double computeNumberValue(vector<NumberPart>& m,
355 vector<NumberPart>::iterator itBegin,
356 vector<NumberPart>::iterator itEnd,
358{
359
360 if (itBegin == itEnd) { // neutral element
361 switch(mode) {
362 case ADDITIVE: return 0;
363 case MULTIPLICATIVE: return 1;
364 }
365 }
366
367 vector<NumberPart>::iterator tmp(itBegin);
368 if (++tmp == itEnd) { // single element
369 return *itBegin;
370 }
371
372 // search for multipliers -> find the biggest
373 // (the biggest is not always the first : trois cent douze mille)
374 vector<NumberPart>::iterator ConjunctionPosition(itEnd);
375 vector<NumberPart>::iterator biggestMultiplier(itEnd);
376 double biggestMultiplierValue(0);
377
378 for (vector<NumberPart>::iterator i(itBegin); i!=itEnd; i++) {
379 double value=*i;
380 if (*i==0) {
381 // is conjunction
382 ConjunctionPosition=i;
383 }
384 if (isMultiplierNumber(value)) {
385 if (value > biggestMultiplierValue) {
386 biggestMultiplier = i;
387 biggestMultiplierValue = value;
388 }
389 }
390 }
391
392 // if there is a multiplier
393 if (biggestMultiplier != itEnd) {
394 const double d= (computeNumberValue(m,itBegin,biggestMultiplier,MULTIPLICATIVE)*
395 biggestMultiplierValue
396 +computeNumberValue(m,biggestMultiplier+1,itEnd));
397 return d;
398 }
399
400 // if there is a conjunction
401 if (ConjunctionPosition != itEnd) {
402 const double d= ( computeNumberValue(m,itBegin,ConjunctionPosition) +
403 computeNumberValue(m,ConjunctionPosition+1,itEnd) );
404 return d;
405 }
406 return 0;
407}
408
409
410} // end namespace
411} // end namespace
412} // end namespace
This file is the main header file for the data related to annotation graphs.
#define DEFAULT_ATTRIBUTE
#define LWARN
Definition LimaCommon.h:160
#define UNDEFLANG
Definition LimaCommon.h:252
#define LDEBUG
Definition LimaCommon.h:157
#define LERROR
Definition LimaCommon.h:161
#define SELOGINIT
#define UNIT_FEATURE_NAME
#define NUMVALUE_FEATURE_NAME
#define NormalizeNumberId
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
Holds linguistic data for one language.
const FsaStringsPool & stringsPool(MediaId med) const
const MediaData & mediaData(MediaId media) const
␈rief A class for the description of automata
Definition automaton.h:87
a list of generic features: each feature is unique (only one feature for a name)
void setFeature(const std::string &name, const ValueType &value)
void addFeature(const std::string &name, const ValueType &value)
LinguisticAnalysisStructure::Token * getToken(RecognizerMatch::iterator) const
const LinguisticAnalysisStructure::AnalysisGraph * getGraph() const
LinguisticAnalysisStructure::MorphoSyntacticData * getData(RecognizerMatch::iterator) const
NormalizeNumber(MediaId language, const LimaString &complement=LimaString())
bool operator()(Automaton::RecognizerMatch &m, AnalysisContent &analysis) const override
zero-ary constraint function : applies the function without a vertex indication (used for actions,...
A representation of a specific entity to store in the annotation graph.
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
LimaString utf8stdstring2limastring(const std::string &src)
double computeNumberValue(vector< NumberPart > &m, vector< NumberPart >::iterator itBegin, vector< NumberPart >::iterator itEnd, NumberNormalizationMode mode=ADDITIVE)
Automaton::ConstraintFunctionFactory< NormalizeNumber > NormalizeNumberFactory(NormalizeNumberId)
double getValueHeuristic(const LimaString &str, LimaChar sep)
bool testMicroCategory(const std::set< LinguisticCode > *micros, const Common::PropertyCode::PropertyAccessor *microAccessor, const LinguisticCode properties)
NAUTITIA.
QChar LimaChar
Definition LimaString.h:30
QString LimaString
Definition LimaString.h:33
STL namespace.