LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
NormalizePerson.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/************************************************************************
7 *
8 * @file NormalizePerson.cpp
9 * @author Olivier Mesnarrd (olivier.mesnard@cea.fr)
10 * @date Wed Jan 13 2015
11 * copyright Copyright (C) 2006-2015 by CEA LIST
12 * Project lima_linguisticprocessing
13 *
14 ***********************************************************************/
15
16#include "NormalizePerson.h"
17#include "NormalizationUtils.h"
22
23using namespace Lima::Common::MediaticData;
26using namespace std;
27
28namespace Lima {
29namespace LinguisticProcessing {
30namespace SpecificEntities {
31
32#define FIRSTNAME_FEATURE_NAME "firstname"
33#define LASTNAME_FEATURE_NAME "lastname"
34
35//**********************************************************************
36// factories for actions defined in this file
39
40//**********************************************************************
42NormalizePerson(MediaId language,
43 const LimaString& complement):
44Automaton::ConstraintFunction(language,complement),
45m_language(language),
46m_firstname(),
47m_lastname(),
48m_microsForFirstname(0),
49m_microAccessor(0)
50{
51 m_microAccessor=&(static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(language)).getPropertyCodeManager().getPropertyAccessor("MICRO"));
52
53 if (language != UNDEFLANG) {
54 try {
55 auto res = LinguisticResources::single().getResource(language,"microsForPersonNameNormalization");
56 auto micros = std::dynamic_pointer_cast<MicrosForNormalization>(res);
57 m_microsForFirstname=micros->getMicros("FirstnameMicros");
58 }
59 catch (exception& e) {
61 LWARN << "Exception caught: " << e.what();
62 LWARN << "-> micros for person name normalization are not initialized";
63 }
64 }
65
66 if (!complement.isEmpty()) {
67 //uint64_t i=complement.find(LimaChar(',')); portage 32 64
68 int i=complement.indexOf(LimaChar(','));
69 if (i==-1) {
70 m_lastname=complement;
71 }
72 else {
73 m_firstname=complement.left(i);
74 m_lastname=complement.mid(i+1);
75 }
76 }
77}
78
79// function to normalize person names using a simple heuristic on result
80// to separate firstname from lastname
99 AnalysisContent& /*unused analysis*/) const
100{
101 // if firstname or lastname were given as arguments to the action
102 if (!m_firstname.isEmpty() || !m_lastname.isEmpty()) {
104 std::vector<EntityFeature>::iterator firstnameFeatureIt = m.features().find(FIRSTNAME_FEATURE_NAME);
105 (*firstnameFeatureIt).setPosition(0);
106 (*firstnameFeatureIt).setLength(0);
108 std::vector<EntityFeature>::iterator lastnameFeatureIt = m.features().find(LASTNAME_FEATURE_NAME);
109 (*lastnameFeatureIt).setPosition(0);
110 (*lastnameFeatureIt).setLength(0);
111 // modified stored normalized string to given normalization:
112 m.features().setFeature(DEFAULT_ATTRIBUTE,m_firstname+LimaChar(' ')+m_lastname);
113 return true;
114 }
115
116 // modified stored normalized string to inflected form :
117 // ensure no normalization is applied to person names
119
120 LimaString firstname;
121 LimaString lastname;
122
123 for (RecognizerMatch::const_iterator i(m.begin()); i!=m.end(); i++) {
124 if (! (*i).isKept()) {
125 continue;
126 }
127 Token* t = m.getToken(i);
128 MorphoSyntacticData* data = m.getData(i);
129 // test if is firstname
130 if (testMicroCategory(m_microsForFirstname,m_microAccessor,data)) {
131 if (!firstname.isEmpty()) { firstname += LimaChar(' '); }
132 firstname += t->stringForm();
133 }
134 else {
135 if (!lastname.isEmpty()) { lastname += LimaChar(' '); }
136 lastname += t->stringForm();
137 }
138 }
139
140 // if firstname and lastname are attributed (for at least two words)
141 if (((!firstname.isEmpty()) && (!lastname.isEmpty()))
142 || m.size() == 1) {
144 RecognizerMatch::const_iterator i(m.begin());
145 Token* t = m.getToken(i);
146 uint64_t pos = (int64_t)(t->position());
147 std::vector<EntityFeature>::iterator firstnameFeatureIt = m.features().find(FIRSTNAME_FEATURE_NAME);
148 uint64_t len = (int64_t)(t->length());
149 (*firstnameFeatureIt).setPosition(pos);
150 (*firstnameFeatureIt).setLength(len);
152 return true;
153 }
154
155 // if there isn't a firstname or a lastname, use a heuristic
156
157 //if only two elements, first is firstname, second is lastname
158 if (m.size() == 2) {
159 RecognizerMatch::const_iterator i(m.begin());
160 Token* t = m.getToken(i);
161 firstname = t->stringForm();
163 std::vector<EntityFeature>::iterator firstnameFeatureIt = m.features().find(FIRSTNAME_FEATURE_NAME);
164 uint64_t pos = (int64_t)(t->position());
165 (*firstnameFeatureIt).setPosition(pos);
166 uint64_t len = (int64_t)(t->length());
167 (*firstnameFeatureIt).setLength(len);
168 i++;
169 t = m.getToken(i);
170 lastname = t->stringForm();
172 std::vector<EntityFeature>::iterator lastnameFeatureIt = m.features().find(LASTNAME_FEATURE_NAME);
173 pos = (int64_t)(t->position());
174 (*lastnameFeatureIt).setPosition(pos);
175 len = (int64_t)(t->length());
176 (*lastnameFeatureIt).setLength(len);
177 return true;
178 }
179
180 // else loop again on the elements : last element is the lastname
181 // all others are firstname, unless something indicates we are passing
182 // in the lastname part (initial, "de")
183
184 lastname=LimaString();
185 firstname=LimaString();
186 uint64_t lastnamePos = 2001;
187 uint64_t lastnameLen = 2002;
188 uint64_t firstnamePos = 2003;
189 uint64_t firstnameLen = 2004;
190 bool inLastname(false);
191 bool initial(false);
192 RecognizerMatch::const_iterator next;
193 for (RecognizerMatch::const_iterator i(m.begin()); i!=m.end(); i++) {
194 Token* t = m.getToken(i);
195 if (initial) {
196 inLastname=1;
197 initial=false;
199 {
200 firstname += t->stringForm();
201 continue;
202 }
203 }
204 if ((t->length() == 1 ||
205 (t->length() == 2 && t->stringForm()[1]==LimaChar('.')))
206 && ! inLastname) {
207 initial=true;
208 if (!firstname.isEmpty()) { firstname += LimaChar(' '); }
209 firstname += t->stringForm();
210 continue;
211 }
212 else if (t->stringForm() == Common::Misc::utf8stdstring2limastring("de") ||
214 inLastname=1;
215 }
216 // if last element -> last name
217 next=i;
218 next++;
219 if (next == m.end()) { inLastname=1; }
220 else { // peek to find if next elt is something like "Jr"
221 Token* nextToken = m.getToken(next);
222 const LimaString& str=nextToken->stringForm();
227 inLastname=1;
228 }
229 }
230
231 if (inLastname) {
232 if (!lastname.isEmpty() &&
234 lastname += LimaChar(' ');
235 }
236 lastname += t->stringForm();
237 lastnamePos = t->position();
238 lastnameLen = t->length();
239 }
240 else {
241 if (!firstname.isEmpty()) { firstname += LimaChar(' '); }
242 firstname += t->stringForm();
243 firstnamePos = t->position();
244 firstnameLen = t->length();
245 }
246 }
247
248 if (firstname.isEmpty() && lastname.isEmpty()) {
249
252 }
253 else {
255 std::vector<EntityFeature>::iterator featureIt = m.features().find(FIRSTNAME_FEATURE_NAME);
256 (*featureIt).setPosition(firstnamePos);
257 (*featureIt).setLength(firstnameLen);
259 featureIt = m.features().find(LASTNAME_FEATURE_NAME);
260 (*featureIt).setPosition(lastnamePos);
261 (*featureIt).setLength(lastnameLen);
262 }
263 return true;
264}
265
266} // end namespace
267} // end namespace
268} // end namespace
#define DEFAULT_ATTRIBUTE
#define LWARN
Definition LimaCommon.h:160
#define UNDEFLANG
Definition LimaCommon.h:252
#define SELOGINIT
#define LASTNAME_FEATURE_NAME
#define FIRSTNAME_FEATURE_NAME
#define NormalizePersonId
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
Holds linguistic data for one language.
const FsaStringsPool & stringsPool(MediaId med) const
const MediaData & mediaData(MediaId media) const
␈rief A class for the description of automata
Definition automaton.h:87
void setFeature(const std::string &name, const ValueType &value)
EntityFeatures::const_iterator find(const std::string &featureName) const
LinguisticAnalysisStructure::Token * getToken(RecognizerMatch::iterator) const
LinguisticAnalysisStructure::MorphoSyntacticData * getData(RecognizerMatch::iterator) const
LimaString getNormalizedString(const FsaStringsPool &sp) const
NormalizePerson(MediaId language, const LimaString &complement=LimaString())
bool operator()(Automaton::RecognizerMatch &m, AnalysisContent &analysis) const override
compute the normalized form of a person name heuristic :
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
static MediaticData & changeable()
singleton accessor
Definition Singleton.h:71
LimaString utf8stdstring2limastring(const std::string &src)
Automaton::ConstraintFunctionFactory< NormalizePerson > NormalizePersonFactory(NormalizePersonId)
bool testMicroCategory(const std::set< LinguisticCode > *micros, const Common::PropertyCode::PropertyAccessor *microAccessor, const LinguisticCode properties)
NAUTITIA.
QChar LimaChar
Definition LimaString.h:30
QString LimaString
Definition LimaString.h:33
STL namespace.