LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
NormalizePersonName.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/************************************************************************
7 *
8 * @file NormalizePersonName.cpp
9 * @author Besancon Romaric (romaric.besancon@cea.fr)
10 * @date Tue Jun 13 2006
11 * copyright Copyright (C) 2006-2012 by CEA LIST
12 *
13 ***********************************************************************/
14
15#include "NormalizePersonName.h"
16#include "NormalizationUtils.h"
21
22#include <QRegularExpression>
23
24using namespace Lima::Common::MediaticData;
27using namespace std;
28
29namespace Lima {
30namespace LinguisticProcessing {
31namespace SpecificEntities {
32
33#define FIRSTNAME_FEATURE_NAME "firstname"
34#define LASTNAME_FEATURE_NAME "lastname"
35
36//**********************************************************************
37// factories for actions defined in this file
40
41// utility function to normalize the case
43 // capitalize single word
44 return str.left(1).toUpper()+str.mid(1).toLower();
45}
46
48 //capitalize each word, separated by a space or '-'
49 QRegularExpression sep("[ -]");
50 QString capitalized;
51 int current=0;
52 int i=str.indexOf(sep,current);
53 while (i!=-1) {
54 capitalized+=capitalizeWord(str.mid(current,i-current))+str[i];
55 current=i+1;
56 i=str.indexOf(sep,current);
57 }
58 // last one
59 capitalized+=capitalizeWord(str.mid(current));
60 return capitalized;
61}
62
63
64//**********************************************************************
66NormalizePersonName(MediaId language,
67 const LimaString& complement):
68Automaton::ConstraintFunction(language,complement),
69m_language(language),
70m_firstname(),
71m_lastname(),
72m_microsForFirstname(0),
73m_microAccessor(0)
74{
75 m_microAccessor=&(static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(language)).getPropertyCodeManager().getPropertyAccessor("MICRO"));
76
77 if (language != UNDEFLANG) {
78 try {
79 auto res = LinguisticResources::single().getResource(language,"microsForPersonNameNormalization");
80 auto micros = std::dynamic_pointer_cast<MicrosForNormalization>(res);
81 m_microsForFirstname = micros->getMicros("FirstnameMicros");
82 }
83 catch (exception& e) {
85 LWARN << "Exception caught: " << e.what();
86 LWARN << "-> micros for person name normalization are not initialized";
87 }
88 }
89
90 if (!complement.isEmpty()) {
91 //uint64_t i=complement.find(LimaChar(',')); portage 32 64
92 int i=complement.indexOf(LimaChar(','));
93 if (i==-1) {
94 m_lastname=complement;
95 }
96 else {
97 m_firstname=complement.left(i);
98 m_lastname=complement.mid(i+1);
99 }
100 }
101}
102
103// function to normalize person names using a simple heuristic on result
104// to separate firstname from lastname
123 AnalysisContent& /*unused analysis*/) const
124{
125 // if firstname or lastname were given as arguments to the action
126 if (!m_firstname.isEmpty() || !m_lastname.isEmpty()) {
128 std::vector<EntityFeature>::iterator firstnameFeatureIt = m.features().find(FIRSTNAME_FEATURE_NAME);
129 (*firstnameFeatureIt).setPosition(0);
130 (*firstnameFeatureIt).setLength(0);
132 std::vector<EntityFeature>::iterator lastnameFeatureIt = m.features().find(LASTNAME_FEATURE_NAME);
133 (*lastnameFeatureIt).setPosition(0);
134 (*lastnameFeatureIt).setLength(0);
135 // modified stored normalized string to given normalization:
136 m.features().setFeature(DEFAULT_ATTRIBUTE,m_firstname+LimaChar(' ')+m_lastname);
137 return true;
138 }
139
140 // modified stored normalized string to inflected form :
141 // ensure no normalization is applied to person names
143
144 LimaString firstname;
145 LimaString lastname;
146
147 for (RecognizerMatch::const_iterator i(m.begin()); i!=m.end(); i++) {
148 if (! (*i).isKept()) {
149 continue;
150 }
151 Token* t = m.getToken(i);
152 MorphoSyntacticData* data = m.getData(i);
153 // test if is firstname
154 if (testMicroCategory(m_microsForFirstname,m_microAccessor,data)) {
155 if (!firstname.isEmpty()) { firstname += LimaChar(' '); }
156 firstname += t->stringForm();
157 }
158 else {
159 if (!lastname.isEmpty()) { lastname += LimaChar(' '); }
160 lastname += t->stringForm();
161 }
162 }
163
164 // if firstname and lastname are attributed (for at least two words)
165 if (((!firstname.isEmpty()) && (!lastname.isEmpty()))
166 || m.size() == 1) {
168 RecognizerMatch::const_iterator i(m.begin());
169 Token* t = m.getToken(i);
170 uint64_t pos = (int64_t)(t->position());
171 std::vector<EntityFeature>::iterator firstnameFeatureIt = m.features().find(FIRSTNAME_FEATURE_NAME);
172 uint64_t len = (int64_t)(t->length());
173 (*firstnameFeatureIt).setPosition(pos);
174 (*firstnameFeatureIt).setLength(len);
176 return true;
177 }
178
179 // if there isn't a firstname or a lastname, use a heuristic
180
181 //if only two elements, first is firstname, second is lastname
182 if (m.size() == 2) {
183 RecognizerMatch::const_iterator i(m.begin());
184 Token* t = m.getToken(i);
185 firstname = t->stringForm();
187 std::vector<EntityFeature>::iterator firstnameFeatureIt = m.features().find(FIRSTNAME_FEATURE_NAME);
188 uint64_t pos = (int64_t)(t->position());
189 (*firstnameFeatureIt).setPosition(pos);
190 uint64_t len = (int64_t)(t->length());
191 (*firstnameFeatureIt).setLength(len);
192 i++;
193 t = m.getToken(i);
194 lastname = t->stringForm();
196 std::vector<EntityFeature>::iterator lastnameFeatureIt = m.features().find(LASTNAME_FEATURE_NAME);
197 pos = (int64_t)(t->position());
198 (*lastnameFeatureIt).setPosition(pos);
199 len = (int64_t)(t->length());
200 (*lastnameFeatureIt).setLength(len);
201 return true;
202 }
203
204 // else loop again on the elements : last element is the lastname
205 // all others are firstname, unless something indicates we are passing
206 // in the lastname part (initial, "de")
207
208 lastname=LimaString();
209 firstname=LimaString();
210 uint64_t lastnamePos = 2001;
211 uint64_t lastnameLen = 2002;
212 uint64_t firstnamePos = 2003;
213 uint64_t firstnameLen = 2004;
214 bool inLastname(false);
215 bool initial(false);
216 RecognizerMatch::const_iterator next;
217 for (RecognizerMatch::const_iterator i(m.begin()); i!=m.end(); i++) {
218 Token* t = m.getToken(i);
219 if (initial) {
220 inLastname=1;
221 initial=false;
223 {
224 firstname += t->stringForm();
225 continue;
226 }
227 }
228 if ((t->length() == 1 ||
229 (t->length() == 2 && t->stringForm()[1]==LimaChar('.')))
230 && ! inLastname) {
231 initial=true;
232 if (!firstname.isEmpty()) { firstname += LimaChar(' '); }
233 firstname += t->stringForm();
234 continue;
235 }
236 else if (t->stringForm() == Common::Misc::utf8stdstring2limastring("de") ||
238 inLastname=1;
239 }
240 // if last element -> last name
241 next=i;
242 next++;
243 if (next == m.end()) { inLastname=1; }
244 else { // peek to find if next elt is something like "Jr"
245 Token* nextToken = m.getToken(next);
246 const LimaString& str=nextToken->stringForm();
251 inLastname=1;
252 }
253 }
254
255 if (inLastname) {
256 if (!lastname.isEmpty() &&
258 lastname += LimaChar(' ');
259 }
260 lastname += t->stringForm();
261 lastnamePos = t->position();
262 lastnameLen = t->length();
263 }
264 else {
265 if (!firstname.isEmpty()) { firstname += LimaChar(' '); }
266 firstname += t->stringForm();
267 firstnamePos = t->position();
268 firstnameLen = t->length();
269 }
270 }
271
272 if (firstname.isEmpty() && lastname.isEmpty()) {
273
276 }
277 else {
279 std::vector<EntityFeature>::iterator featureIt = m.features().find(FIRSTNAME_FEATURE_NAME);
280 (*featureIt).setPosition(firstnamePos);
281 (*featureIt).setLength(firstnameLen);
283 featureIt = m.features().find(LASTNAME_FEATURE_NAME);
284 (*featureIt).setPosition(lastnamePos);
285 (*featureIt).setLength(lastnameLen);
286 }
287 return true;
288}
289
290} // end namespace
291} // end namespace
292} // end namespace
#define DEFAULT_ATTRIBUTE
#define LWARN
Definition LimaCommon.h:160
#define UNDEFLANG
Definition LimaCommon.h:252
#define SELOGINIT
#define NormalizePersonNameId
#define LASTNAME_FEATURE_NAME
#define FIRSTNAME_FEATURE_NAME
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
Holds linguistic data for one language.
const FsaStringsPool & stringsPool(MediaId med) const
const MediaData & mediaData(MediaId media) const
␈rief A class for the description of automata
Definition automaton.h:87
void setFeature(const std::string &name, const ValueType &value)
EntityFeatures::const_iterator find(const std::string &featureName) const
LinguisticAnalysisStructure::Token * getToken(RecognizerMatch::iterator) const
LinguisticAnalysisStructure::MorphoSyntacticData * getData(RecognizerMatch::iterator) const
LimaString getNormalizedString(const FsaStringsPool &sp) const
bool operator()(Automaton::RecognizerMatch &m, AnalysisContent &analysis) const override
compute the normalized form of a person name heuristic :
NormalizePersonName(MediaId language, const LimaString &complement=LimaString())
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
static MediaticData & changeable()
singleton accessor
Definition Singleton.h:71
LimaString utf8stdstring2limastring(const std::string &src)
LimaString capitalize(const LimaString &str)
Automaton::ConstraintFunctionFactory< NormalizePersonName > NormalizePersonNameFactory(NormalizePersonNameId)
LimaString capitalizeWord(const LimaString &str)
bool testMicroCategory(const std::set< LinguisticCode > *micros, const Common::PropertyCode::PropertyAccessor *microAccessor, const LinguisticCode properties)
NAUTITIA.
QChar LimaChar
Definition LimaString.h:30
QString LimaString
Definition LimaString.h:33
STL namespace.