LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
basicConstraintFunctions.cpp
Go to the documentation of this file.
1// Copyright 2002-2020 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/************************************************************************
7 *
8 * @file basicConstraintFunctions.cpp
9 * @author besancon (besanconr@zoe.cea.fr)
10 * @date Wed Mar 16 2005
11 * copyright Copyright (C) 2005-2020 by CEA LIST
12 *
13 ***********************************************************************/
14
20
21#include <algorithm>
22
23using namespace std;
24using namespace Lima::Common::MediaticData;
26
27namespace Lima
28{
29namespace LinguisticProcessing
30{
31namespace Automaton
32{
33
34//**********************************************************************
35// factories for constraints defined in this file
36ConstraintFunctionFactory<AgreementConstraint>
38
41
44
47
50
53
56
57//**********************************************************************
59{
60public:
62 bool operator()(const LinguisticCode& code) { return m_acc.empty(code); }
63private:
65};
66
67
69AgreementConstraint(MediaId language,
70 const LimaString& complement):
71 ConstraintFunction(language,complement),
72 m_categoryForAgreementAccessor(0)
73{
74 if (complement == Common::Misc::utf8stdstring2limastring("PERSON"))
75 {
76 m_categoryForAgreementAccessor=
77 &(static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(language)).getPropertyCodeManager().getPropertyAccessor("PERSON"));
78 }
79 else if (complement == Common::Misc::utf8stdstring2limastring("GENDER"))
80 {
81 m_categoryForAgreementAccessor=
82 &(static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(language)).getPropertyCodeManager().getPropertyAccessor("GENDER"));
83 }
84 else if (complement == Common::Misc::utf8stdstring2limastring("NUMBER"))
85 {
86 m_categoryForAgreementAccessor=
87 &(static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(language)).getPropertyCodeManager().getPropertyAccessor("NUMBER"));
88 }
89}
90
92operator()(const AnalysisGraph& anagraph,
93 const LinguisticGraphVertex& vertex1,
94 const LinguisticGraphVertex& vertex2,
95 AnalysisContent& /*ac*/ ) const
96{
97 const LinguisticGraph& graph = *(anagraph.getGraph());
98
99 if (vertex1 == anagraph.firstVertex() ||
100 vertex1 == anagraph.firstVertex() ||
101 vertex2 == anagraph.lastVertex() ||
102 vertex2 == anagraph.lastVertex() )
103 {
104 return false;
105 }
106
107 // compare the categories of the two vertices
108 MorphoSyntacticData* data1=get(vertex_data,graph,vertex1);
109 MorphoSyntacticData* data2=get(vertex_data,graph,vertex2);
110
111 MorphoSyntacticData::const_iterator
112 it1=data1->begin(),it1_end=data1->end();
113 for (; it1!=it1_end; it1++)
114 {
115 MorphoSyntacticData::const_iterator
116 it2=data2->begin(),it2_end=data2->end();
117 for (; it2!=it2_end; it2++)
118 {
119 if (m_categoryForAgreementAccessor->empty((*it1).properties)
120 || m_categoryForAgreementAccessor->empty((*it2).properties)
121 || m_categoryForAgreementAccessor->equal((*it1).properties,
122 (*it2).properties))
123 {
124 return true;
125 }
126 }
127 }
128 return false;
129}
130
131//**********************************************************************
133GenderAgreement(MediaId language,
134 const LimaString& /* unused complement*/ ):
135AgreementConstraint(language,Common::Misc::utf8stdstring2limastring("GENDER")) {}
136
138NumberAgreement(MediaId language,
139 const LimaString& /* unused complement */):
140AgreementConstraint(language,Common::Misc::utf8stdstring2limastring("NUMBER")) {}
141
142//**********************************************************************
144LinguisticPropertyIs(MediaId language,
145 const LimaString& complement):
146 ConstraintFunction(language,complement),
147 m_propertyAccessor(0),
148 m_values(0)
149{
150 //AULOGINIT;
151 //complement contains the name of the property, the value
152 //to test and the language, separated by a comma
153 std::string str=Common::Misc::limastring2utf8stdstring(complement);
154 //LDEBUG << "init constraint LinguisticPropertyIs with complement " << str;
155
156 auto j(string::npos),k(string::npos);
157 auto i=str.find(",");
158 if (i!=string::npos)
159 {
160 j=str.find(",",i+1);
161 }
162
163 if (i==string::npos || j==string::npos)
164 {
165 AULOGINIT;
166 LIMA_EXCEPTION( "Constraint LinguisticPropertyIs : invalid complement \""
167 << str.c_str() << "\": three arguments needed");
168 }
169
170 string propertyString(str,0,i);
171 string valueString(str,i+1,j-i-1);
172 string lang(str,j+1);
173
174 const auto& manager = static_cast<const Common::MediaticData::LanguageData&>(
176 .getPropertyCodeManager().getPropertyManager(propertyString);
177
178 m_propertyAccessor=&(manager.getPropertyAccessor());
179 i=0;
180 j=valueString.find("|");
181 if (j==std::string::npos) j=valueString.size();
182 for (;i<valueString.size();)
183 {
184 //LDEBUG << "read part " << valueString.substr(i,j-i);
185 pair<LinguisticCode,LinguisticProcessing::LinguisticAnalysisStructure::MorphoSyntacticType> value;
186 k=valueString.find("#",i);
187 if (k!=string::npos && k<j)
188 {
189 //LDEBUG << "found # : " << valueString.substr(i,k-i) << " # " << valueString.substr(k+1,j-k-1);
190 value.first=manager.getPropertyValue(valueString.substr(i,k-i));
191 std::string mtype=valueString.substr(k+1,j-k-1);
192 if (mtype == "SIMPLE_WORD") { value.second=SIMPLE_WORD; }
193 else if (mtype == "ABBREV_ALTERNATIVE") { value.second=ABBREV_ALTERNATIVE; }
194 else if (mtype == "HYPHEN_ALTERNATIVE") { value.second=HYPHEN_ALTERNATIVE; }
195 else if (mtype == "IDIOMATIC_EXPRESSION") { value.second=IDIOMATIC_EXPRESSION; }
196 else if (mtype == "CONCATENATED_ALTERNATIVE") { value.second=CONCATENATED_ALTERNATIVE; }
197 else if (mtype == "HYPERWORD_ALTERNATIVE") { value.second=HYPERWORD_ALTERNATIVE; }
198 else if (mtype == "UNKNOWN_WORD") { value.second=UNKNOWN_WORD; }
199 else if (mtype == "CAPITALFIRST_WORD") { value.second=CAPITALFIRST_WORD; }
200 else if (mtype == "AGGLUTINATED_WORD") { value.second=AGGLUTINATED_WORD; }
201 else if (mtype == "DESAGGLUTINATED_WORD") { value.second=DESAGGLUTINATED_WORD; }
202 else if (mtype == "CHINESE_SEGMENTER") { value.second=CHINESE_SEGMENTER; }
203 else if (mtype == "SPECIFIC_ENTITY") { value.second=SPECIFIC_ENTITY; }
204 else if (mtype == "SPELLING_ALTERNATIVE") { value.second=SPELLING_ALTERNATIVE; }
205 else
206 {
207 AULOGINIT;
208 LIMA_EXCEPTION( "Constraint LinguisticPropertyIs : invalid morphosyntactic type \""
209 << mtype.c_str() << "\" !");
210 }
211 } else {
212 value.first=manager.getPropertyValue(valueString.substr(i,j-i));
213 value.second=NO_MORPHOSYNTACTICTYPE;
214 }
215 //LDEBUG << "add value pair (linguisticCode=" << value.first << ",type=" << value.second << ")";
216 m_values.push_back(value);
217 i=j+1;
218 j=valueString.find("|",i);
219 if (j==string::npos) j=valueString.size();
220 }
221}
222
225 const LinguisticGraphVertex& v,
226 AnalysisContent& /*ac*/ ) const
227{
228 MorphoSyntacticData* data=get(vertex_data,*(graph.getGraph()),v);
229
230 MorphoSyntacticData::const_iterator
231 it=data->begin(),
232 it_end=data->end();
233
234 for (; it!=it_end; it++)
235 {
236 for (std::vector<pair<LinguisticCode,LinguisticProcessing::LinguisticAnalysisStructure::MorphoSyntacticType> >::const_iterator pItr=m_values.begin();
237 pItr!=m_values.end();
238 pItr++)
239 {
240 if (m_propertyAccessor->equal((*it).properties,pItr->first) &&
241 (pItr->second==NO_MORPHOSYNTACTICTYPE || it->type==pItr->second))
242 {
243 return true;
244 }
245 }
246 }
247 return false;
248}
249
250//***********************************************************************
252LengthInInterval(MediaId language,
253 const LimaString& complement):
254 ConstraintFunction(language,complement),
255 m_min(0),
256 m_max(0)
257{
258 //complement contains min and max values for length of the token
259 //(max value is optional)
260 std::string str=Common::Misc::limastring2utf8stdstring(complement);
261 auto i=str.find(",");
262 if (i==string::npos)
263 {
264 m_min=atoi(str.c_str());
265 m_max=m_min;
266 }
267 else
268 {
269 m_min=atoi(string(str,0,i).c_str());
270 m_max=atoi(string(str,i+1).c_str());
271 }
272}
273
276 const LinguisticGraphVertex& v,
277 AnalysisContent& /*ac*/ ) const
278{
279 Token* token = get(vertex_token,*(graph.getGraph()),v);
280 if (token == nullptr)
281 {
282 AULOGINIT;
283 LERROR << "Null token on vertex " << v;
284 return false;
285 }
286
287 uint64_t length=token->length();
288
289 // AULOGINIT;
290 // LDEBUG << "testing length of token " << *token
291 // << "(" << length << ") with interval ["
292 // << min << "-" << max << "]";
293
294 //std::cerr << "testing length(" << token->stringForm().toUtf8().constData()
295 // << "(" << length << ") with interval [" << m_min << "-" << m_max << "]" << endl;
296 return (length >= m_min && length <= m_max);
297}
298
299//***********************************************************************
301NumericValueInInterval(MediaId language,
302 const LimaString& complement):
303 ConstraintFunction(language,complement),
304 m_language(language),
305 m_min(0),
306 m_max(0)
307{
308 //complement contains min and max values for length of the token
309 //(max value is optional)
310 std::string str=Common::Misc::limastring2utf8stdstring(complement);
311 std::string::size_type i=str.find(",");
312 if (i==std::string::npos)
313 {
314 m_min=atoi(str.c_str());
315 m_max=m_min;
316 }
317 else
318 {
319 string minString(str,0,i), maxString(str,i+1);
320 if (minString.empty()) {
321 m_min=0;
322 }
323 else {
324 m_min=atoi(minString.c_str());
325 }
326 if (maxString.empty()) {
327 m_max=0;
328 }
329 else {
330 m_max=atoi(maxString.c_str());
331 }
332 }
333}
334
337 const LinguisticGraphVertex& v,
338 AnalysisContent& /*ac*/ ) const
339{
340 Token* token = get(vertex_token,*(graph.getGraph()),v);
341 if (token == nullptr)
342 {
343 AULOGINIT;
344 LERROR << "Null token on vertex " << v;
345 return false;
346 }
347
348 uint64_t numValue(0);
349
350 const TStatus& status=token->status();
351 if (status.getNumeric() == T_INTEGER) {
352 numValue=token->stringForm().toULong();
353 }
354 else {
355 // get the normalized form : contains the numeric form
356 // (normalized form corresponding to a macro_micro identifying a number)
357 MorphoSyntacticData* data = get(vertex_data,*(graph.getGraph()),v);
358 for (MorphoSyntacticData::const_iterator it=data->begin(),
359 it_end=data->end(); it!=it_end; it++) {
360 numValue=Common::MediaticData::MediaticData::single().stringsPool(m_language)[(*it).normalizedForm].toULong();
361 if (numValue!=0) {
362 break;
363 }
364 }
365 }
366 if (numValue==0) { // cannot determine the numeric value
367 return false;
368 }
369
370 if (m_max == 0) {
371 return ( numValue >= m_min);
372 }
373 else {
374 return ( numValue >= m_min && numValue <= m_max);
375 }
376}
377
378//***********************************************************************
380 const LimaString& complement):
381ConstraintFunction(language,complement)
382{
383}
384
386 const LinguisticGraphVertex& v1,
387 const LinguisticGraphVertex& v2,
388 AnalysisContent& /*analysis*/) const
389{
390 Token* token1 = get(vertex_token,*(graph.getGraph()),v1);
391 Token* token2 = get(vertex_token,*(graph.getGraph()),v2);
392 if (token1 == nullptr)
393 {
394 AULOGINIT;
395 LERROR << "Null token on vertex " << v1;
396 return false;
397 }
398 if (token2 == nullptr)
399 {
400 AULOGINIT;
401 LERROR << "Null token on vertex " << v2;
402 return false;
403 }
404
405 if (token2->position() > token1->position()) {
406 return token2->position()==token1->position()+token1->length();
407 }
408 else {
409 return token1->position()==token2->position()+token2->length();
410 }
411}
412
413} // end namespace
414} // end namespace
415} // end namespace
#define LIMA_EXCEPTION(X)
This macro writes the message X to a previously configured error stream before throwing a LimaExcepti...
Definition LimaCommon.h:293
#define LERROR
Definition LimaCommon.h:161
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define AULOGINIT
#define AgreementConstraintId
#define LinguisticPropertyIsId
#define GenderAgreementId
#define LengthInIntervalId
#define NumberAgreementId
#define NoSpaceWithId
#define NumericValueInIntervalId
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
Holds linguistic data for one language.
const FsaStringsPool & stringsPool(MediaId med) const
const MediaData & mediaData(MediaId media) const
Provide function to read write and check a property.
bool equal(const LinguisticCode &l1, const LinguisticCode &l2) const
check property equality for two linguisticCode
bool empty(const LinguisticCode &l) const
Test if the given code has property data.
generic agreement constraint function: complement must contain the element on which the agreement mus...
bool operator()(const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &v1, const LinguisticGraphVertex &v2, AnalysisContent &analysis) const override
binary constraint function : the constraint only applies on two vertices in the graph (the graph is a...
AgreementConstraint(MediaId language, const LimaString &complement=LimaString())
CheckIfEmptyPredicate(const Common::PropertyCode::PropertyAccessor &propAcc)
GenderAgreement(MediaId language, const LimaString &complement=LimaString())
LengthInInterval(MediaId language, const LimaString &complement=LimaString())
bool operator()(const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &v, AnalysisContent &analysis) const override
unary constraint function : the constraint only applies on one vertices in the graph (the graph is al...
LinguisticPropertyIs(MediaId language, const LimaString &complement=LimaString())
bool operator()(const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &v, AnalysisContent &analysis) const override
unary constraint function : the constraint only applies on one vertices in the graph (the graph is al...
NoSpaceWith(MediaId language, const LimaString &complement=LimaString())
bool operator()(const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &v1, const LinguisticGraphVertex &v2, AnalysisContent &analysis) const override
binary constraint function : the constraint only applies on two vertices in the graph (the graph is a...
NumberAgreement(MediaId language, const LimaString &complement=LimaString())
NumericValueInInterval(MediaId language, const LimaString &complement=LimaString())
bool operator()(const LinguisticAnalysisStructure::AnalysisGraph &graph, const LinguisticGraphVertex &v, AnalysisContent &analysis) const override
unary constraint function : the constraint only applies on one vertices in the graph (the graph is al...
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
LimaString utf8stdstring2limastring(const std::string &src)
ConstraintFunctionFactory< GenderAgreement > GenderAgreementFactory(GenderAgreementId)
ConstraintFunctionFactory< AgreementConstraint > AgreementConstraintFactory(AgreementConstraintId)
ConstraintFunctionFactory< NumericValueInInterval > NumericValueInIntervalFactory(NumericValueInIntervalId)
ConstraintFunctionFactory< NoSpaceWith > NoSpaceWithFactory(NoSpaceWithId)
ConstraintFunctionFactory< LengthInInterval > LengthInIntervalFactory(LengthInIntervalId)
ConstraintFunctionFactory< LinguisticPropertyIs > LinguisticPropertyIsFactory(LinguisticPropertyIsId)
ConstraintFunctionFactory< NumberAgreement > NumberAgreementFactory(NumberAgreementId)
NAUTITIA.
QString LimaString
Definition LimaString.h:33
STL namespace.