LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
AbbreviationSplitAlternatives.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
32
49
50
51using namespace std;
52using namespace Lima::Common::AnnotationGraphs;
55
56namespace Lima
57{
58namespace LinguisticProcessing
59{
60namespace MorphologicAnalysis
61{
62
64
66m_tokenizer(0),
67m_dictionary(0),
68m_abbreviations(),
69m_language(),
70m_confidentMode(true),
71m_reader(0),
72m_charSplitRegexp()
73{
74 // default split regexp: split on simple quote or UTF-8 right quotation mark
76 m_charSplitRegexp=QRegularExpression(quotes);
77
78}
79
81{
82 // if (m_reader) {
83 // delete m_reader;
84 // }
85}
86
87
90 Manager* manager)
91
92{
94 m_language=manager->getInitializationParameters().media;
95 try
96 {
97 string dico=unitConfiguration.getParamsValueAtKey("dictionary");
98 auto res = LinguisticResources::single().getResource(m_language,dico);
99 m_dictionary = std::dynamic_pointer_cast<AnalysisDict::AbstractAnalysisDictionary>(res);
100 }
102 {
103 LERROR << "no param 'dictionary' in AbbreviationSplitAlternatives group for language " << (int) m_language;
104 throw InvalidConfiguration();
105 }
106 try
107 {
108 auto tok = unitConfiguration.getParamsValueAtKey("tokenizer");
109 auto res = manager->getObject(tok);
110 m_tokenizer = std::dynamic_pointer_cast<FlatTokenizer::Tokenizer>(res);
111 }
113 {
114 LERROR << "no param 'tokenizer' in AbbreviationSplitAlternatives group for language " << (int) m_language;
115 throw InvalidConfiguration();
116 }
117 try
118 {
119 deque<string> abbs=unitConfiguration.getListsValueAtKey("abbreviations");
120 for (deque<string>::iterator it=abbs.begin();
121 it!=abbs.end();
122 it++)
123 {
124 m_abbreviations.push_back(Common::Misc::utf8stdstring2limastring(*it));
125 }
126 }
128 {
129 LERROR << "no list 'abbreviations' in AbbreviationSplitAlternatives group for language " << (int) m_language;
130 throw InvalidConfiguration();
131 }
132
133 std::shared_ptr<FlatTokenizer::CharChart> charChart;
134 try
135 {
136 auto charchart = unitConfiguration.getParamsValueAtKey("charChart");
137 auto res = LinguisticResources::single().getResource(m_language,charchart);
138 charChart = std::dynamic_pointer_cast<FlatTokenizer::CharChart>(res);
139 }
141 {
142 LERROR << "no param 'charChart' in AbbreviationSplitAlternatives group for language " << (int) m_language;
143 throw InvalidConfiguration();
144 }
145 try
146 {
147 auto confident = unitConfiguration.getParamsValueAtKey("confidentMode");
148 m_confidentMode = (confident=="true");
149 }
151 {
152 LWARN << "no param 'confidentMode' in AbbreviationSplitAlternatives group for language " << (int) m_language;
153 LWARN << "use default value : 'true'";
154 m_confidentMode = true;
155 }
156
157 try
158 {
159 string charSplit=unitConfiguration.getParamsValueAtKey("charSplitRegexp");
160 m_charSplitRegexp=QRegularExpression(Common::Misc::utf8stdstring2limastring(charSplit));
161 }
163 {
164 LWARN << "no param 'confidentMode' in AbbreviationSplitAlternatives group for language " << (int) m_language;
165 LWARN << "use default value : 'true'";
166 m_confidentMode=true;
167 }
168
170 m_reader = std::make_shared<AlternativesReader>(m_confidentMode, true, true, true, charChart, sp);
171
172}
173
175 AnalysisContent& analysis) const
176{
177 Lima::TimeUtilsController timer("AbbreviationSplitAlternatives");
179 LINFO << "MorphologicalAnalysis: starting process AbbreviationSplitAlternatives";
180
181 auto tokenList = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"));
182 auto graph = tokenList->getGraph();
183
184 VertexDataPropertyMap dataMap = get( vertex_data, *graph );
185 VertexTokenPropertyMap tokenMap = get( vertex_token, *graph );
186
187 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.getData("AnnotationData"));
188 if (annotationData==0)
189 {
190 LDEBUG << "AbbreviationSplitAlternatives::process: Misssing AnnotationData. Create it";
191 annotationData = std::make_shared<AnnotationData>();
192 if (std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph")) != 0)
193 {
194 std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"))->populateAnnotationGraph(
195 annotationData.get(), "AnalysisGraph");
196 }
197 analysis.setData("AnnotationData",annotationData);
198 }
199
200 try
201 {
202
203 LinguisticGraphVertexIt it, it_end;
204 boost::tie(it, it_end) = vertices(*graph);
205 for (; it != it_end; it++)
206 {
207 MorphoSyntacticData* currentData = dataMap[*it];
208 if (currentData == 0) continue;
209 Token* currentToken= tokenMap[*it];
210
218 bool isSplitted=false;
219 if (m_confidentMode)
220 {
221 // si la forme a deja ete trouvee dans le dictionnaire, il s'agit d'une
222 // contraction connue, donc on ne fait rien
223 if (currentData->size()>0) continue;
224 // si il s'agit d'un possessif alors on traite le cas possessif.
225 if (currentToken->status().isAlphaPossessive())
226 {
227 isSplitted=makePossessiveAlternativeFor(*it, graph, annotationData.get());
228 }
229 // On ne traite la forme comme une abbreviation, uniquement si elle n'a pas ete
230 // traitee comme un possessif
231 if ((!isSplitted) && currentToken->status().isAlphaConcatAbbrev())
232 {
233 isSplitted=makeConcatenatedAbbreviationSplitAlternativeFor(*it,graph, annotationData.get());
234 }
235 }
236 else
237 {
238 // En mode non confiance, on effectue tous les traitements
239 if (currentToken->status().isAlphaPossessive())
240 {
241 isSplitted=makePossessiveAlternativeFor(*it, graph, annotationData.get());
242 }
243 // On traite la forme comme une abbreviation, meme si elle a ete traitee comme un possessif
244 if (currentToken->status().isAlphaConcatAbbrev())
245 {
246 isSplitted=makeConcatenatedAbbreviationSplitAlternativeFor(*it,graph, annotationData.get()) || isSplitted;
247 }
248 }
249
250 // si la forme a ete decoupee, alors on supprime le vertex d'origine
251 if (isSplitted)
252 {
253 // unlink the previous vertex
254 clear_vertex(*it,*graph);
255 }
256
257 }
258 }
259 catch (std::exception &exc)
260 {
262 LWARN << "Exception in AbbreviationSplitAlternatives : " << exc.what();
263 return UNKNOWN_ERROR;
264 }
265
266 LINFO << "MorphologicalAnalysis: ending process AbbreviationSplitAlternatives";
267 return SUCCESS_ID;
268}
269
270
271bool AbbreviationSplitAlternatives::makeConcatenatedAbbreviationSplitAlternativeFor(
272 LinguisticGraphVertex splitted,
273 LinguisticGraph* graph,
274 AnnotationData* annotationData) const
275{
277 VertexTokenPropertyMap tokenMap = get( vertex_token,*graph );
278 Token* ftok = tokenMap[splitted];
279 const LimaString& ft = ftok->stringForm();
280 LDEBUG << "AbbreviationSplitAlternatives::makeConcatenatedAbbreviationSplitAlternativeFor " << Common::Misc::limastring2utf8stdstring(ft);
281
282 //int aposPos = ft.indexOf(Common::Misc::utf8stdstring2limastring("'"), 0);
283 int aposPos = ft.indexOf(m_charSplitRegexp, 0);
284 //LDEBUG << "AbbreviationSplitAlternatives: split chars found at " << aposPos;
285 if (aposPos==-1 || aposPos==0) {
286 return false;
287 }
288 LimaString beforeAbbrev(ft.left(aposPos-1));
289
290 std::vector< LimaString >::const_iterator itAbb = m_abbreviations.begin();
291 std::vector< LimaString >::const_iterator itAbb_end = m_abbreviations.end();
292 LimaString abbrev;
293 bool found = false;
294 for (; itAbb != itAbb_end ; itAbb++)
295 {
296 abbrev = *itAbb;
297 found = ft.endsWith(abbrev);
298 if (found)
299 {
300 beforeAbbrev = ft.left(ft.size() - abbrev.size());
301 break;
302 }
303 }
304 if (!found) return false;
305
306 // submit first string to Tokenizer
307 LimaStringText* beforeAbbrevText=new LimaStringText(beforeAbbrev);
308 AnalysisContent toTokenize;
309 toTokenize.setData("Text",beforeAbbrevText);
310 LimaStatusCode status=m_tokenizer->process(toTokenize);
311 if (status != SUCCESS_ID) return false;
312 auto tokenizerList = std::dynamic_pointer_cast<AnalysisGraph>(toTokenize.getData("AnalysisGraph"));
313 LinguisticGraph* tokGraph=tokenizerList->getGraph();
314
315 // insert the first abreviated word
316 uint64_t beginPos = ftok->position()-1;
317 LinguisticGraphVertex firstToken=tokenizerList->firstVertex();
318 {
319 LinguisticGraphAdjacencyIt adjItr,adjItrEnd;
320 boost::tie(adjItr,adjItrEnd) = adjacent_vertices(firstToken,*tokGraph);
321 if (adjItr==adjItrEnd)
322 {
324 LERROR << "AbbreviationSplitAlternatives::makeConcatenatedAbbreviationSplitAlternativeFor : no token forward !";
325 throw LinguisticProcessingException("AbbreviationSplitAlternatives::makeConcatenatedAbbreviationSplitAlternativeFor : no token forward !");
326 }
327 firstToken=*adjItr;
328 }
329
331 Token* tokenizerToken = new Token(*(get(vertex_token,*tokGraph,firstToken)));
332 MorphoSyntacticData* tokenizerData = new MorphoSyntacticData();
333 tokenizerToken->setPosition(tokenizerToken->position() + beginPos);
334 MorphoSyntacticDataHandler tokDataHandler(*tokenizerData,ABBREV_ALTERNATIVE);
335 // read alternatives, but don't try concatenated
336 m_reader->readAlternatives(
337 *tokenizerToken,
338 *m_dictionary,
339 &tokDataHandler,
340 0,
341 &tokDataHandler);
342
343 LinguisticGraphVertex beforeVertex = add_vertex(*graph);
344 put(vertex_token,*graph,beforeVertex,tokenizerToken);
345 put(vertex_data,*graph,beforeVertex,tokenizerData);
346
347 AnnotationGraphVertex agv = annotationData->createAnnotationVertex();
348 annotationData->addMatching("AnalysisGraph", beforeVertex, "annot", agv);
349 annotationData->annotate(agv, "AnalysisGraph", beforeVertex);
350
351
352 // insert the second abreviated word
353 StringsPoolIndex abbrevId=sp[abbrev];
354 Token* newFT = new Token(abbrevId,abbrev,tokenizerToken->position()+tokenizerToken->length(),abbrev.size(),ftok->status());
356
357 // retrieve ling infos for abbreviated word
358 DictionaryEntry entry(m_dictionary->getEntry(newFT->form(),newFT->stringForm()));
359 if (!entry.isEmpty())
360 {
361 MorphoSyntacticDataHandler newDataHandler(*newData,ABBREV_ALTERNATIVE);
362 if (entry.hasLingInfos())
363 {
364 entry.parseLingInfos(&newDataHandler);
365 }
366 else
367 {
368 LERROR << "AbbreviationSplitAlternatives::makeConcatenatedAbbreviationSplitAlternativeFor: dictionary entry for abbreviated word " << Lima::Common::Misc::limastring2utf8stdstring(abbrev) << " has no linguistic info";
369 }
370 }
371 else
372 {
373 LERROR << "AbbreviationSplitAlternatives::makeConcatenatedAbbreviationSplitAlternativeFor: Cannot find a dictionary entry for abbreviated word " << Lima::Common::Misc::limastring2utf8stdstring(abbrev);
374 }
375 if (newData->empty())
376 {
378 LERROR << "AbbreviationSplitAlternatives::makeConcatenatedAbbreviationSplitAlternativeFor Got empty morphosyntactic data. Abort.";
379 delete newFT;
380 delete newData;
381 return false;
382 }
383// LinguisticGraphVertex afterVertex = listIterator.createVertexFor(newFT);
384 LinguisticGraphVertex afterVertex = add_vertex(*graph);
385 put(vertex_token,*graph,afterVertex,newFT);
386 VertexDataPropertyMap dataMap = get(vertex_data, *graph);
387 dataMap[afterVertex] = newData;
388
389 AnnotationGraphVertex agvafter = annotationData->createAnnotationVertex();
390 annotationData->addMatching("AnalysisGraph", afterVertex, "annot", agvafter);
391 annotationData->annotate(agvafter, "AnalysisGraph", afterVertex);
392
393 // links the first newly created vertex to the predecessors of the old vertex
394 LinguisticGraphInEdgeIt itie, itie_end;
395 boost::tie(itie, itie_end) = in_edges(splitted, *graph);
396 for (; itie != itie_end; itie++)
397 {
398 add_edge(source(*itie,*graph), beforeVertex, *graph);
399 }
400
401 // links both newly created vertices
402 add_edge(beforeVertex, afterVertex, *graph);
403
404 // links the second newly created vertex to the succesors of the old vertex
405 LinguisticGraphOutEdgeIt itoe, itoe_end;
406 boost::tie(itoe, itoe_end) = out_edges(splitted, *graph);
407 for (; itoe != itoe_end; itoe++)
408 {
409 add_edge(afterVertex, target(*itoe,*graph), *graph);
410 }
411 return true;
412}
413
414bool AbbreviationSplitAlternatives::makePossessiveAlternativeFor(
415 LinguisticGraphVertex splitted,
416 LinguisticGraph* graph,
417 AnnotationData* annotationData) const
418{
420 VertexTokenPropertyMap tokenMap = get( vertex_token, *graph );
421 Token* ftok = tokenMap[splitted];
422 const LimaString& ft = ftok->stringForm();
423 LDEBUG << "AbbreviationSplitAlternatives::makePossessiveAlternativeFor " << Common::Misc::limastring2utf8stdstring(ft);
424
425 //int aposPos = ft.indexOf(LimaChar('\''), 0);
426 int aposPos = ft.indexOf(m_charSplitRegexp, 0);
427 if (aposPos==-1 || aposPos==0) return false;
428 LimaString possessivedWord(ft.left(aposPos));
429 LDEBUG << "AbbreviationSplitAlternatives::makePossessiveAlternativeFor possesive word: " << Common::Misc::limastring2utf8stdstring(possessivedWord);
430
431 QRegularExpression pronounre("^(he|she|it|let)$",
432 QRegularExpression::CaseInsensitiveOption);
433 auto match = pronounre.match(possessivedWord);
434 if (match.hasMatch())
435 {
436 return false;
437 }
438
439 // submit first string to Tokenizer
440 LimaStringText* possessivedWordText=new LimaStringText(possessivedWord);
441 AnalysisContent toTokenize;
442 toTokenize.setData("Text",possessivedWordText);
443 LimaStatusCode status=m_tokenizer->process(toTokenize);
444 if (status != SUCCESS_ID)
445 {
446 LERROR << "AbbreviationSplitAlternatives::makePossessiveAlternativeFor: Failed to tokenize possesive word";
447 return false;
448 }
449 auto tokenizerList = std::dynamic_pointer_cast<AnalysisGraph>(toTokenize.getData("AnalysisGraph"));
450 LinguisticGraph* tokGraph=tokenizerList->getGraph();
451
452 // insert the first abreviated word
453 uint64_t beginPos = ftok->position()-1;
454 LinguisticGraphVertex firstToken=tokenizerList->firstVertex();
455 {
456 LinguisticGraphAdjacencyIt adjItr,adjItrEnd;
457 boost::tie(adjItr,adjItrEnd) = adjacent_vertices(firstToken,*tokGraph);
458 if (adjItr==adjItrEnd)
459 {
461 LERROR << "AbbreviationSplitAlternatives::makePossessiveAlternativeFor : no token forward !";
462 throw LinguisticProcessingException("AbbreviationSplitAlternatives::makePossessiveAlternativeFor : no token forward !");
463 }
464 firstToken=*adjItr;
465 }
466 Token* tokenizerToken = new Token(*(get(vertex_token,*tokGraph,firstToken)));
467 MorphoSyntacticData* tokenizerData = new MorphoSyntacticData();
468 tokenizerToken->setPosition(tokenizerToken->position() + beginPos);
469 tokenizerToken->status().setAlphaPossessive(true);
470 MorphoSyntacticDataHandler tokDataHandler(*tokenizerData,ABBREV_ALTERNATIVE);
471 m_reader->readAlternatives(
472 *tokenizerToken,
473 *m_dictionary,
474 &tokDataHandler, // linginfos
475 0, // Concat
476 &tokDataHandler); // Accented
477
478// LinguisticGraphVertex possessivedVertex = listIterator.createVertexFor(tokenizerToken);
479 LinguisticGraphVertex possessivedVertex = add_vertex(*graph);
480 put(vertex_token,*graph,possessivedVertex,tokenizerToken);
481 put(vertex_data,*graph,possessivedVertex,tokenizerData);
482
483 AnnotationGraphVertex agvposs = annotationData->createAnnotationVertex();
484 annotationData->addMatching("AnalysisGraph", possessivedVertex, "annot", agvposs);
485 annotationData->annotate(agvposs, "AnalysisGraph", possessivedVertex);
486
487 // links the newly created vertex to the predecessors of the old vertex
488 LinguisticGraphInEdgeIt itie, itie_end;
489 boost::tie(itie, itie_end) = in_edges(splitted, *graph);
490 for (; itie != itie_end; itie++)
491 {
492 add_edge(source(*itie,*graph), possessivedVertex, *graph);
493 }
494
495 // links the newly created vertex to the succesors of the old vertex
496 LinguisticGraphOutEdgeIt itoe, itoe_end;
497 boost::tie(itoe, itoe_end) = out_edges(splitted, *graph);
498 for (; itoe != itoe_end; itoe++)
499 {
500 add_edge(possessivedVertex, target(*itoe,*graph), *graph);
501 }
502
503 return true;
504}
505
506
507
508} // closing namespace MorphologicAnalysis
509} // closing namespace LinguisticProcessing
510} // closing namespace Lima
AbbreviationSplitAlternatives is the module which creates split alternatives for hyphen word tokens.
#define ABBREVIATIONSPLITALTERNATIVESFACTORY_CLASSID
This file is the main header file for the data related to annotation graphs.
#define LWARN
Definition LimaCommon.h:160
#define LDEBUG
Definition LimaCommon.h:157
#define LINFO
Definition LimaCommon.h:158
#define LERROR
Definition LimaCommon.h:161
LinguisticGraph::in_edge_iterator LinguisticGraphInEdgeIt
boost::property_map< LinguisticGraph, vertex_data_t >::type VertexDataPropertyMap
LinguisticGraph::vertex_iterator LinguisticGraphVertexIt
boost::property_map< LinguisticGraph, vertex_token_t >::type VertexTokenPropertyMap
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
LinguisticGraph::adjacency_iterator LinguisticGraphAdjacencyIt
#define MORPHOLOGINIT
Defines a Factory to create Object of type Base.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Holds an annotation graph and gives an API to manipulate it.
void addMatching(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second, AnnotationGraphVertex secondVx)
Adds a symetric matching between two vertices of two graphs identified by the two string parameters.
AnnotationGraphVertex createAnnotationVertex()
Creates a new annotation vertex in the graph.
const FsaStringsPool & stringsPool(MediaId med) const
std::deque< std::string > & getListsValueAtKey(const std::string &key)
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Use this exception to signal an error in one of the configuration files.
Definition LimaCommon.h:345
define the limaStringText resource which is the text in limaString formal
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
virtual std::shared_ptr< Object > getObject(const std::string &id)
If object doesn't exists, call the create method.
static const LinguisticResources & single()
const singleton accessor
Definition Singleton.h:51
static MediaticData & changeable()
singleton accessor
Definition Singleton.h:71
This file contains a class to control log of informations about time, such as logging cumulated time ...
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
void annotate(AnnotationGraphVertex v, const LimaString &annot, uint64_t value)
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
LimaString utf8stdstring2limastring(const std::string &src)
SimpleFactory< MediaProcessUnit, AbbreviationSplitAlternatives > abbreviationSplitAlternativesFactory(ABBREVIATIONSPLITALTERNATIVESFACTORY_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ UNKNOWN_ERROR
Definition LimaCommon.h:238
@ SUCCESS_ID
Definition LimaCommon.h:237
QString LimaString
Definition LimaString.h:33
STL namespace.
launch exception related to the configuration file parsing