58namespace LinguisticProcessing
60namespace MorphologicAnalysis
76 m_charSplitRegexp=QRegularExpression(quotes);
99 m_dictionary = std::dynamic_pointer_cast<AnalysisDict::AbstractAnalysisDictionary>(res);
103 LERROR <<
"no param 'dictionary' in AbbreviationSplitAlternatives group for language " << (int) m_language;
110 m_tokenizer = std::dynamic_pointer_cast<FlatTokenizer::Tokenizer>(res);
114 LERROR <<
"no param 'tokenizer' in AbbreviationSplitAlternatives group for language " << (int) m_language;
120 for (deque<string>::iterator it=abbs.begin();
129 LERROR <<
"no list 'abbreviations' in AbbreviationSplitAlternatives group for language " << (int) m_language;
133 std::shared_ptr<FlatTokenizer::CharChart> charChart;
138 charChart = std::dynamic_pointer_cast<FlatTokenizer::CharChart>(res);
142 LERROR <<
"no param 'charChart' in AbbreviationSplitAlternatives group for language " << (int) m_language;
148 m_confidentMode = (confident==
"true");
152 LWARN <<
"no param 'confidentMode' in AbbreviationSplitAlternatives group for language " << (int) m_language;
153 LWARN <<
"use default value : 'true'";
154 m_confidentMode =
true;
164 LWARN <<
"no param 'confidentMode' in AbbreviationSplitAlternatives group for language " << (int) m_language;
165 LWARN <<
"use default value : 'true'";
166 m_confidentMode=
true;
170 m_reader = std::make_shared<AlternativesReader>(m_confidentMode,
true,
true,
true, charChart, sp);
179 LINFO <<
"MorphologicalAnalysis: starting process AbbreviationSplitAlternatives";
181 auto tokenList = std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"AnalysisGraph"));
182 auto graph = tokenList->getGraph();
187 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.
getData(
"AnnotationData"));
188 if (annotationData==0)
190 LDEBUG <<
"AbbreviationSplitAlternatives::process: Misssing AnnotationData. Create it";
191 annotationData = std::make_shared<AnnotationData>();
192 if (std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"AnalysisGraph")) != 0)
194 std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"AnalysisGraph"))->populateAnnotationGraph(
195 annotationData.get(),
"AnalysisGraph");
197 analysis.
setData(
"AnnotationData",annotationData);
204 boost::tie(it, it_end) = vertices(*graph);
205 for (; it != it_end; it++)
208 if (currentData == 0)
continue;
209 Token* currentToken= tokenMap[*it];
218 bool isSplitted=
false;
223 if (currentData->size()>0)
continue;
227 isSplitted=makePossessiveAlternativeFor(*it, graph, annotationData.get());
233 isSplitted=makeConcatenatedAbbreviationSplitAlternativeFor(*it,graph, annotationData.get());
241 isSplitted=makePossessiveAlternativeFor(*it, graph, annotationData.get());
246 isSplitted=makeConcatenatedAbbreviationSplitAlternativeFor(*it,graph, annotationData.get()) || isSplitted;
254 clear_vertex(*it,*graph);
259 catch (std::exception &exc)
262 LWARN <<
"Exception in AbbreviationSplitAlternatives : " << exc.what();
266 LINFO <<
"MorphologicalAnalysis: ending process AbbreviationSplitAlternatives";
271bool AbbreviationSplitAlternatives::makeConcatenatedAbbreviationSplitAlternativeFor(
278 Token* ftok = tokenMap[splitted];
283 int aposPos = ft.indexOf(m_charSplitRegexp, 0);
285 if (aposPos==-1 || aposPos==0) {
290 std::vector< LimaString >::const_iterator itAbb = m_abbreviations.begin();
291 std::vector< LimaString >::const_iterator itAbb_end = m_abbreviations.end();
294 for (; itAbb != itAbb_end ; itAbb++)
297 found = ft.endsWith(abbrev);
300 beforeAbbrev = ft.left(ft.size() - abbrev.size());
304 if (!found)
return false;
309 toTokenize.
setData(
"Text",beforeAbbrevText);
312 auto tokenizerList = std::dynamic_pointer_cast<AnalysisGraph>(toTokenize.
getData(
"AnalysisGraph"));
316 uint64_t beginPos = ftok->
position()-1;
320 boost::tie(adjItr,adjItrEnd) = adjacent_vertices(firstToken,*tokGraph);
321 if (adjItr==adjItrEnd)
324 LERROR <<
"AbbreviationSplitAlternatives::makeConcatenatedAbbreviationSplitAlternativeFor : no token forward !";
336 m_reader->readAlternatives(
345 put(
vertex_data,*graph,beforeVertex,tokenizerData);
348 annotationData->
addMatching(
"AnalysisGraph", beforeVertex,
"annot", agv);
349 annotationData->
annotate(agv,
"AnalysisGraph", beforeVertex);
353 StringsPoolIndex abbrevId=sp[abbrev];
359 if (!entry.isEmpty())
362 if (entry.hasLingInfos())
364 entry.parseLingInfos(&newDataHandler);
375 if (newData->empty())
378 LERROR <<
"AbbreviationSplitAlternatives::makeConcatenatedAbbreviationSplitAlternativeFor Got empty morphosyntactic data. Abort.";
387 dataMap[afterVertex] = newData;
390 annotationData->
addMatching(
"AnalysisGraph", afterVertex,
"annot", agvafter);
391 annotationData->
annotate(agvafter,
"AnalysisGraph", afterVertex);
395 boost::tie(itie, itie_end) = in_edges(splitted, *graph);
396 for (; itie != itie_end; itie++)
398 add_edge(source(*itie,*graph), beforeVertex, *graph);
402 add_edge(beforeVertex, afterVertex, *graph);
406 boost::tie(itoe, itoe_end) = out_edges(splitted, *graph);
407 for (; itoe != itoe_end; itoe++)
409 add_edge(afterVertex, target(*itoe,*graph), *graph);
414bool AbbreviationSplitAlternatives::makePossessiveAlternativeFor(
421 Token* ftok = tokenMap[splitted];
426 int aposPos = ft.indexOf(m_charSplitRegexp, 0);
427 if (aposPos==-1 || aposPos==0)
return false;
431 QRegularExpression pronounre(
"^(he|she|it|let)$",
432 QRegularExpression::CaseInsensitiveOption);
433 auto match = pronounre.match(possessivedWord);
434 if (match.hasMatch())
442 toTokenize.
setData(
"Text",possessivedWordText);
446 LERROR <<
"AbbreviationSplitAlternatives::makePossessiveAlternativeFor: Failed to tokenize possesive word";
449 auto tokenizerList = std::dynamic_pointer_cast<AnalysisGraph>(toTokenize.
getData(
"AnalysisGraph"));
453 uint64_t beginPos = ftok->
position()-1;
457 boost::tie(adjItr,adjItrEnd) = adjacent_vertices(firstToken,*tokGraph);
458 if (adjItr==adjItrEnd)
461 LERROR <<
"AbbreviationSplitAlternatives::makePossessiveAlternativeFor : no token forward !";
471 m_reader->readAlternatives(
480 put(
vertex_token,*graph,possessivedVertex,tokenizerToken);
481 put(
vertex_data,*graph,possessivedVertex,tokenizerData);
484 annotationData->
addMatching(
"AnalysisGraph", possessivedVertex,
"annot", agvposs);
485 annotationData->
annotate(agvposs,
"AnalysisGraph", possessivedVertex);
489 boost::tie(itie, itie_end) = in_edges(splitted, *graph);
490 for (; itie != itie_end; itie++)
492 add_edge(source(*itie,*graph), possessivedVertex, *graph);
497 boost::tie(itoe, itoe_end) = out_edges(splitted, *graph);
498 for (; itoe != itoe_end; itoe++)
500 add_edge(possessivedVertex, target(*itoe,*graph), *graph);
AbbreviationSplitAlternatives is the module which creates split alternatives for hyphen word tokens.
#define ABBREVIATIONSPLITALTERNATIVESFACTORY_CLASSID
This file is the main header file for the data related to annotation graphs.
LinguisticGraph::in_edge_iterator LinguisticGraphInEdgeIt
boost::property_map< LinguisticGraph, vertex_data_t >::type VertexDataPropertyMap
LinguisticGraph::vertex_iterator LinguisticGraphVertexIt
boost::property_map< LinguisticGraph, vertex_token_t >::type VertexTokenPropertyMap
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
LinguisticGraph::adjacency_iterator LinguisticGraphAdjacencyIt
Defines a Factory to create Object of type Base.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Holds an annotation graph and gives an API to manipulate it.
void addMatching(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second, AnnotationGraphVertex secondVx)
Adds a symetric matching between two vertices of two graphs identified by the two string parameters.
AnnotationGraphVertex createAnnotationVertex()
Creates a new annotation vertex in the graph.
std::string & getParamsValueAtKey(const std::string &key)
std::deque< std::string > & getListsValueAtKey(const std::string &key)
return a message when a 'list' was not found
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Use this exception to signal an error in one of the configuration files.
define the limaStringText resource which is the text in limaString formal
Holds morphosyntactic informations.
bool isAlphaPossessive() const
void setAlphaPossessive(bool isAlphaPossessive)
bool isAlphaConcatAbbrev() const
holds surface data of a token
uint64_t position() const
const TStatus & status() const
const LimaString & stringForm() const
StringsPoolIndex form() const
void setPosition(uint64_t pos)
AbbreviationSplitAlternatives()
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
virtual ~AbbreviationSplitAlternatives()
virtual std::shared_ptr< Object > getObject(const std::string &id)
If object doesn't exists, call the create method.
static const LinguisticResources & single()
const singleton accessor
static MediaticData & changeable()
singleton accessor
This file contains a class to control log of informations about time, such as logging cumulated time ...
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
void annotate(AnnotationGraphVertex v, const LimaString &annot, uint64_t value)
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
LimaString utf8stdstring2limastring(const std::string &src)
SimpleFactory< MediaProcessUnit, AbbreviationSplitAlternatives > abbreviationSplitAlternativesFactory(ABBREVIATIONSPLITALTERNATIVESFACTORY_CLASSID)
launch exception related to the configuration file parsing