LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
HyphenWordAlternatives.cpp
Go to the documentation of this file.
1// Copyright 2002-2020 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
32
46
47using namespace std;
48using namespace Lima::Common::MediaticData;
49using namespace Lima::Common::AnnotationGraphs;
52
53namespace Lima
54{
55namespace LinguisticProcessing
56{
57namespace MorphologicAnalysis
58{
59
61
63 : m_sentBoundariesName("SentenceBoundaries")
64{}
65
67{
68 // delete m_reader;
69}
70
73 Manager* manager)
74
75{
77 m_language = manager->getInitializationParameters().media;
78 try
79 {
80 string dico=unitConfiguration.getParamsValueAtKey("dictionary");
81 auto res = LinguisticResources::single().getResource(m_language,dico);
82 m_dictionary = std::dynamic_pointer_cast<AnalysisDict::AbstractAnalysisDictionary>(res);
83 }
85 {
86 LERROR << "no param 'dictionary' in HyphenWordAlternatives group for language " << (int) m_language;
88 }
89 try
90 {
91 string charchart=unitConfiguration.getParamsValueAtKey("charChart");
92 auto res = LinguisticResources::single().getResource(m_language,charchart);
93 m_charChart = std::dynamic_pointer_cast<FlatTokenizer::CharChart>(res);
94 }
96 {
97 LERROR << "no param 'charChart' in HyphenWordAlternatives group for language " << (int) m_language;
99 }
100 try
101 {
102 string tok=unitConfiguration.getParamsValueAtKey("tokenizer");
103 auto res = manager->getObject(tok);
104 m_tokenizer = std::dynamic_pointer_cast<FlatTokenizer::Tokenizer>(res);
105 }
107 {
108 LERROR << "no param 'dictionary' in HyphenWordAlternatives group for language " << (int) m_language;
109 throw InvalidConfiguration();
110 }
111 try
112 {
113 m_deleteHyphenWord =( unitConfiguration.getParamsValueAtKey("deleteHyphenWord") == "true");
114 }
116 {
117 LWARN << "no param 'deleteHyphenWord' in HyphenAlternatives group for language " << (int) m_language;
118 LWARN << "use default value : true";
119 m_deleteHyphenWord=true;
120 }
121 try
122 {
123 string confident=unitConfiguration.getParamsValueAtKey("confidentMode");
124 m_confidentMode=(confident=="true");
125 }
127 {
128 LWARN << "no param 'confidentMode' in HyphenWordAlternatives group for language " << (int) m_language;
129 LWARN << "use default value : 'true'";
130 m_confidentMode=true;
131 }
133 m_reader = std::make_shared<AlternativesReader>(m_confidentMode, true, true, true, m_charChart, sp);
134
135 const auto &theMediaticData = static_cast<const Common::MediaticData::MediaticData&>(Common::MediaticData::MediaticData::single());
136 m_engLanguageId = theMediaticData.getMediaId("eng");
137
138 try
139 {
140 m_sentBoundariesName=unitConfiguration.getParamsValueAtKey("sentBoundaries");
141 }
143 {
144 LINFO << "no param 'sentBoundaries' in HyphenWordAlternatives group for language " << (int) m_language;
145 }
146}
147
149 AnalysisContent& analysis) const
150{
151 Lima::TimeUtilsController timer("HyphenWordAlternatives");
153 LINFO << "MorphologicalAnalysis: starting process HyphenWordAlternatives";
154
155 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.getData("AnnotationData"));
156 if (nullptr == annotationData)
157 {
158 LDEBUG << "HyphenWordAlternatives::process: Misssing AnnotationData. Create it";
159 annotationData = std::make_shared<AnnotationData>();
160 if (std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph")) != 0)
161 {
162 std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"))->populateAnnotationGraph(
163 annotationData.get(), "AnalysisGraph");
164 }
165 analysis.setData("AnnotationData",annotationData);
166 }
167
168 auto tokenList=std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"));
169 LinguisticGraph* graph=tokenList->getGraph();
170 auto sb = std::dynamic_pointer_cast<SegmentationData>(analysis.getData(m_sentBoundariesName));
171
172 VertexDataPropertyMap dataMap = get( vertex_data, *graph );
173 VertexTokenPropertyMap tokenMap = get( vertex_token, *graph );
174
175 try
176 {
177 LinguisticGraphVertexIt it, it_end;
178 boost::tie(it, it_end) = vertices(*graph);
179 for (; it != it_end; it++)
180 {
181 MorphoSyntacticData* currentToken = dataMap[*it];
182 Token* tok= tokenMap[*it];
183 if (currentToken==0) continue;
184 //<if a token has a linguistic data
185 // it is not decomposed>
186 if (currentToken->size() == 0)
187 {
188 if (tok->status().isAlphaHyphen() && isWorthSplitting(*it, graph))
189 {
190 makeHyphenSplitAlternativeFor(*it, graph, annotationData.get(), sb.get());
191 }
192 }
193 }
194 }
195 catch (std::exception &exc)
196 {
198 LWARN << "Exception in HyphenWordAlternatives : " << exc.what();
199 return UNKNOWN_ERROR;
200 }
201
202 LINFO << "MorphologicalAnalysis: ending process HyphenWordAlternatives";
203 return SUCCESS_ID;
204}
205
206/*
207 * Hyphenated words are often absent in the dictionary. This leads to inability to treat them properly
208 * on the level of rules. It's worth to split hyphenated words if their parts would bring more information
209 * in the later stages of the analysis.
210 *
211 * Verification is implemented for English only. Always true for other languages.
212 *
213 * It's considered that it makes sense to split English hypthenated words in following cases:
214 * - its right part is capitalized
215 * - its right part exists in the dictionary.
216 *
217 * Examples:
218 *
219 * April-June (right part capitalized)
220 * Ben-Elissar (right part capitalized)
221 * Singapore-listed ("listed" exists in the dictionary)
222 * record-breaking ("breaking" exists in the dictionary)
223 * EU-wide ("wide" exists in the dictionary)
224 */
225bool HyphenWordAlternatives::isWorthSplitting(
226 LinguisticGraphVertex splitted,
227 LinguisticGraph* graph) const
228{
229 if (m_engLanguageId != m_language)
230 return true;
231
232 VertexTokenPropertyMap tokenMap = get( vertex_token, *graph );
233 Token* currentToken = tokenMap[splitted];
234
235 LimaString hyphenWord(currentToken->stringForm());
236
237#ifdef DEBUG_LP
239 LDEBUG << "isWorthSplitting for " << hyphenWord;
240#endif
241
242 // find last hyphen
243 int pos = hyphenWord.lastIndexOf(LimaChar(L'-'));
244 if (pos < 0 || (hyphenWord.length() - pos < 3)) {
245#ifdef DEBUG_LP
246 LDEBUG << "isWorthSplitting: pos = " << pos;
247#endif
248 return false;
249 }
250
251 LimaString rightPart = hyphenWord.right(hyphenWord.length() - pos - 1);
252
253#ifdef DEBUG_LP
254 LDEBUG << "isWorthSplitting: rightPart = " << rightPart;
255#endif
256
257 QChar firstChar = rightPart[0];
258 if (firstChar.isLetter() && firstChar.isUpper()) {
259#ifdef DEBUG_LP
260 LDEBUG << "isWorthSplitting: first char is upper letter";
261#endif
262 return true;
263 }
264
265 DictionaryEntry dicoEntry(m_dictionary->getEntry(rightPart));
266 if (dicoEntry.isEmpty()) {
267#ifdef DEBUG_LP
268 LDEBUG << "isWorthSplitting: dicoEntry is empty. Don't split.";
269#endif
270 return false;
271 }
272
273 return true;
274}
275
276void HyphenWordAlternatives::makeHyphenSplitAlternativeFor(
277 LinguisticGraphVertex splitted,
278 LinguisticGraph* graph,
279 AnnotationData* annotationData,
280 SegmentationData* sb) const
281{
282 VertexTokenPropertyMap tokenMap = get( vertex_token, *graph );
283 VertexDataPropertyMap dataMap = get( vertex_data, *graph );
284 Token* currentToken = tokenMap[splitted];
285
286 // first, get a copy of token string
287 LimaString hyphenWord(currentToken->stringForm());
288
289 // first replace hyphens by spaces
290 int pos = hyphenWord.indexOf(LimaChar(L'-'), 0);
291 while (pos != -1)
292 {
293 hyphenWord[(int)pos] = LimaChar(L' ');
294 pos = hyphenWord.indexOf(LimaChar(L'-'), pos+1);
295 }
296 // then submit string to Tokenizer
297 AnalysisContent toTokenize;
298 toTokenize.setData("Text",new LimaStringText(hyphenWord));
299 LimaStatusCode status=m_tokenizer->process(toTokenize);
300 if (status != SUCCESS_ID) return;
301 auto agTokenizer = std::dynamic_pointer_cast<AnalysisGraph>(toTokenize.getData("AnalysisGraph"));
302 LinguisticGraph* tokgraph=agTokenizer->getGraph();
303
304 // setup position field
305 // insert each new FullToken into alternative path
306 uint64_t beginPos = currentToken->position()-1;
307 LinguisticGraphVertex previous = splitted;
308 LinguisticGraphVertex currentVx=agTokenizer->firstVertex();
309 // go one step forward on the new path
310 {
311 LinguisticGraphAdjacencyIt adjItr,adjItrEnd;
312 boost::tie(adjItr,adjItrEnd) = adjacent_vertices(currentVx,*tokgraph);
313 if (adjItr==adjItrEnd)
314 {
316 LERROR << "HypenWordAlternatives : no token forward !";
317 throw LinguisticProcessingException("HypenWordAlternatives : no token forward !");
318 }
319 currentVx=*adjItr;
320 }
321 // LinguisticGraphVertex lastVx=agTokenizer->lastVertex();
322 VertexTokenPropertyMap tokTokenMap=get(vertex_token,*tokgraph);
323 Token* tokenizerToken=tokTokenMap[currentVx];
324
325 bool isFirst=true;
326
327 LinguisticGraphVertex firstVertex;
328 LinguisticGraphVertex lastVertex;
329 size_t numVertices = 0;
330
331 while (tokenizerToken)
332 {
333 // prepare the new vertex
334 Token* newFT=new Token(*tokenizerToken);
335 newFT->status().setAlphaHyphen( true );
337 LinguisticGraphVertex newVertex = add_vertex(*graph);
338
339 if (0 == numVertices)
340 {
341 firstVertex = newVertex;
342 }
343 else
344 {
345 lastVertex = newVertex;
346 }
347
348 numVertices++;
349
350 AnnotationGraphVertex agv = annotationData->createAnnotationVertex();
351 annotationData->addMatching("AnalysisGraph", newVertex, "annot", agv);
352 annotationData->annotate(agv, Common::Misc::utf8stdstring2limastring("AnalysisGraph"), newVertex);
353
354 tokenMap[newVertex]=newFT;
355 dataMap[newVertex]=newData;
356 newFT-> setPosition(newFT->position() + beginPos);
357 const LimaString& newTokenStr=newFT->stringForm();
358 MorphoSyntacticDataHandler handler(*newData,HYPHEN_ALTERNATIVE);
359
360 if (isFirst)
361 {
362 LimaString newTokHyphen(newTokenStr);
363 newTokHyphen.append(LimaChar('-'));
364 DictionaryEntry dicoEntry(m_dictionary->getEntry(newTokHyphen));
365 if (!dicoEntry.isEmpty() && dicoEntry.hasLingInfos())
366 {
368 Token* newFT2=new Token((*sp)[newTokHyphen],newTokHyphen,newFT->position(),newFT->length()+1);
369 tokenMap[newVertex]=newFT2;
370 delete newFT;
371 newFT = newFT2;
372 dicoEntry.parseLingInfos(&handler);
373 }
374 else
375 {
376 m_reader->readAlternatives(
377 *newFT,
378 *m_dictionary,
379 &handler,
380 0,
381 &handler);
382 }
383 }
384 else
385 {
386 m_reader->readAlternatives(
387 *newFT,
388 *m_dictionary,
389 &handler,
390 0,
391 &handler);
392 }
393
394 // links the new vertex to its predecessor in the graph
395 if (previous == splitted)
396 {
397 LinguisticGraphInEdgeIt ite, ite_end;
398 boost::tie(ite, ite_end) = in_edges(splitted, *graph);
399 for (; ite != ite_end; ite++)
400 {
401 add_edge(source(*ite,*graph), newVertex, *graph);
402 }
403 }
404 else
405 {
406 add_edge(previous, newVertex, *graph);
407 }
408 previous = newVertex;
409 // go one step forward on the new path
410 LinguisticGraphAdjacencyIt adjItr,adjItrEnd;
411 boost::tie(adjItr,adjItrEnd) = adjacent_vertices(currentVx,*tokgraph);
412 if (adjItr==adjItrEnd)
413 {
415 LERROR << "HypenWordAlternatives : no token forward !";
416 throw LinguisticProcessingException("HypenWordAlternatives : no token forward !");
417 }
418 currentVx=*adjItr;
419 tokenizerToken=tokTokenMap[currentVx];
420 }
421
422 // links the last new vertex created to the successors of the splitted vertex
423 LinguisticGraphOutEdgeIt ite, ite_end;
424 boost::tie(ite, ite_end) = out_edges(splitted, *graph);
425 for (; ite != ite_end; ite++)
426 {
427 add_edge(previous, target(*ite,*graph), *graph);
428 }
429
430 // if have to delete hyphen word, then clear it in the graph
431 if (m_deleteHyphenWord)
432 {
433 clear_vertex(splitted,*graph);
434 }
435
436 if (nullptr != sb)
437 {
438 std::vector<Segment>& segments = sb->getSegments();
439 for (size_t i = 0; i < segments.size(); i++)
440 {
441 if (splitted == segments[i].getFirstVertex())
442 {
443 segments[i].setFirstVertex(firstVertex);
444 if (i > 0)
445 {
446 segments[i-1].setLastVertex(firstVertex);
447 }
448 }
449
450 if (splitted == segments[i].getLastVertex())
451 {
452 segments[i].setLastVertex(lastVertex);
453 if (i + 1 < segments.size())
454 {
455 segments[i+1].setFirstVertex(lastVertex);
456 }
457 }
458 }
459 }
460}
461
462} // closing namespace MorphologicAnalysis
463} // closing namespace LinguisticProcessing
464} // closing namespace Lima
This file is the main header file for the data related to annotation graphs.
HyphenWordAlternatives is the module which creates split alternatives for hyphen word tokens.
#define HYPHENWORDALTERNATIVESFACTORY_CLASSID
#define LWARN
Definition LimaCommon.h:160
#define LDEBUG
Definition LimaCommon.h:157
#define LINFO
Definition LimaCommon.h:158
#define LERROR
Definition LimaCommon.h:161
LinguisticGraph::in_edge_iterator LinguisticGraphInEdgeIt
boost::property_map< LinguisticGraph, vertex_data_t >::type VertexDataPropertyMap
LinguisticGraph::vertex_iterator LinguisticGraphVertexIt
boost::property_map< LinguisticGraph, vertex_token_t >::type VertexTokenPropertyMap
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
LinguisticGraph::adjacency_iterator LinguisticGraphAdjacencyIt
#define MORPHOLOGINIT
Defines a Factory to create Object of type Base.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Holds an annotation graph and gives an API to manipulate it.
void addMatching(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second, AnnotationGraphVertex secondVx)
Adds a symetric matching between two vertices of two graphs identified by the two string parameters.
AnnotationGraphVertex createAnnotationVertex()
Creates a new annotation vertex in the graph.
holds data about codes and names for grammatical categories, etc.
const FsaStringsPool & stringsPool(MediaId med) const
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Use this exception to signal an error in one of the configuration files.
Definition LimaCommon.h:345
define the limaStringText resource which is the text in limaString formal
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
const std::vector< Segment > & getSegments() const
virtual std::shared_ptr< Object > getObject(const std::string &id)
If object doesn't exists, call the create method.
static const LinguisticResources & single()
const singleton accessor
Definition Singleton.h:51
static MediaticData & changeable()
singleton accessor
Definition Singleton.h:71
This file contains a class to control log of informations about time, such as logging cumulated time ...
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
void annotate(AnnotationGraphVertex v, const LimaString &annot, uint64_t value)
LimaString utf8stdstring2limastring(const std::string &src)
SimpleFactory< MediaProcessUnit, HyphenWordAlternatives > hyphenwordAlternativesFactory(HYPHENWORDALTERNATIVESFACTORY_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ UNKNOWN_ERROR
Definition LimaCommon.h:238
@ SUCCESS_ID
Definition LimaCommon.h:237
QChar LimaChar
Definition LimaString.h:30
QString LimaString
Definition LimaString.h:33
STL namespace.
launch exception related to the configuration file parsing