55namespace LinguisticProcessing
57namespace MorphologicAnalysis
63 : m_sentBoundariesName(
"SentenceBoundaries")
82 m_dictionary = std::dynamic_pointer_cast<AnalysisDict::AbstractAnalysisDictionary>(res);
86 LERROR <<
"no param 'dictionary' in HyphenWordAlternatives group for language " << (int) m_language;
93 m_charChart = std::dynamic_pointer_cast<FlatTokenizer::CharChart>(res);
97 LERROR <<
"no param 'charChart' in HyphenWordAlternatives group for language " << (int) m_language;
104 m_tokenizer = std::dynamic_pointer_cast<FlatTokenizer::Tokenizer>(res);
108 LERROR <<
"no param 'dictionary' in HyphenWordAlternatives group for language " << (int) m_language;
113 m_deleteHyphenWord =( unitConfiguration.
getParamsValueAtKey(
"deleteHyphenWord") ==
"true");
117 LWARN <<
"no param 'deleteHyphenWord' in HyphenAlternatives group for language " << (int) m_language;
118 LWARN <<
"use default value : true";
119 m_deleteHyphenWord=
true;
124 m_confidentMode=(confident==
"true");
128 LWARN <<
"no param 'confidentMode' in HyphenWordAlternatives group for language " << (int) m_language;
129 LWARN <<
"use default value : 'true'";
130 m_confidentMode=
true;
133 m_reader = std::make_shared<AlternativesReader>(m_confidentMode,
true,
true,
true, m_charChart, sp);
136 m_engLanguageId = theMediaticData.getMediaId(
"eng");
144 LINFO <<
"no param 'sentBoundaries' in HyphenWordAlternatives group for language " << (int) m_language;
153 LINFO <<
"MorphologicalAnalysis: starting process HyphenWordAlternatives";
155 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.
getData(
"AnnotationData"));
156 if (
nullptr == annotationData)
158 LDEBUG <<
"HyphenWordAlternatives::process: Misssing AnnotationData. Create it";
159 annotationData = std::make_shared<AnnotationData>();
160 if (std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"AnalysisGraph")) != 0)
162 std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"AnalysisGraph"))->populateAnnotationGraph(
163 annotationData.get(),
"AnalysisGraph");
165 analysis.
setData(
"AnnotationData",annotationData);
168 auto tokenList=std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"AnalysisGraph"));
170 auto sb = std::dynamic_pointer_cast<SegmentationData>(analysis.
getData(m_sentBoundariesName));
178 boost::tie(it, it_end) = vertices(*graph);
179 for (; it != it_end; it++)
182 Token* tok= tokenMap[*it];
183 if (currentToken==0)
continue;
186 if (currentToken->size() == 0)
190 makeHyphenSplitAlternativeFor(*it, graph, annotationData.get(), sb.get());
195 catch (std::exception &exc)
198 LWARN <<
"Exception in HyphenWordAlternatives : " << exc.what();
202 LINFO <<
"MorphologicalAnalysis: ending process HyphenWordAlternatives";
225bool HyphenWordAlternatives::isWorthSplitting(
229 if (m_engLanguageId != m_language)
233 Token* currentToken = tokenMap[splitted];
239 LDEBUG <<
"isWorthSplitting for " << hyphenWord;
243 int pos = hyphenWord.lastIndexOf(
LimaChar(L
'-'));
244 if (pos < 0 || (hyphenWord.length() - pos < 3)) {
246 LDEBUG <<
"isWorthSplitting: pos = " << pos;
251 LimaString rightPart = hyphenWord.right(hyphenWord.length() - pos - 1);
254 LDEBUG <<
"isWorthSplitting: rightPart = " << rightPart;
257 QChar firstChar = rightPart[0];
258 if (firstChar.isLetter() && firstChar.isUpper()) {
260 LDEBUG <<
"isWorthSplitting: first char is upper letter";
266 if (dicoEntry.isEmpty()) {
268 LDEBUG <<
"isWorthSplitting: dicoEntry is empty. Don't split.";
276void HyphenWordAlternatives::makeHyphenSplitAlternativeFor(
284 Token* currentToken = tokenMap[splitted];
290 int pos = hyphenWord.indexOf(
LimaChar(L
'-'), 0);
293 hyphenWord[(int)pos] =
LimaChar(L
' ');
294 pos = hyphenWord.indexOf(
LimaChar(L
'-'), pos+1);
301 auto agTokenizer = std::dynamic_pointer_cast<AnalysisGraph>(toTokenize.
getData(
"AnalysisGraph"));
306 uint64_t beginPos = currentToken->
position()-1;
312 boost::tie(adjItr,adjItrEnd) = adjacent_vertices(currentVx,*tokgraph);
313 if (adjItr==adjItrEnd)
316 LERROR <<
"HypenWordAlternatives : no token forward !";
323 Token* tokenizerToken=tokTokenMap[currentVx];
329 size_t numVertices = 0;
331 while (tokenizerToken)
339 if (0 == numVertices)
341 firstVertex = newVertex;
345 lastVertex = newVertex;
351 annotationData->
addMatching(
"AnalysisGraph", newVertex,
"annot", agv);
354 tokenMap[newVertex]=newFT;
355 dataMap[newVertex]=newData;
356 newFT-> setPosition(newFT->
position() + beginPos);
365 if (!dicoEntry.isEmpty() && dicoEntry.hasLingInfos())
369 tokenMap[newVertex]=newFT2;
372 dicoEntry.parseLingInfos(&handler);
376 m_reader->readAlternatives(
386 m_reader->readAlternatives(
395 if (previous == splitted)
398 boost::tie(ite, ite_end) = in_edges(splitted, *graph);
399 for (; ite != ite_end; ite++)
401 add_edge(source(*ite,*graph), newVertex, *graph);
406 add_edge(previous, newVertex, *graph);
408 previous = newVertex;
411 boost::tie(adjItr,adjItrEnd) = adjacent_vertices(currentVx,*tokgraph);
412 if (adjItr==adjItrEnd)
415 LERROR <<
"HypenWordAlternatives : no token forward !";
419 tokenizerToken=tokTokenMap[currentVx];
424 boost::tie(ite, ite_end) = out_edges(splitted, *graph);
425 for (; ite != ite_end; ite++)
427 add_edge(previous, target(*ite,*graph), *graph);
431 if (m_deleteHyphenWord)
433 clear_vertex(splitted,*graph);
438 std::vector<Segment>& segments = sb->
getSegments();
439 for (
size_t i = 0; i < segments.size(); i++)
441 if (splitted == segments[i].getFirstVertex())
443 segments[i].setFirstVertex(firstVertex);
446 segments[i-1].setLastVertex(firstVertex);
450 if (splitted == segments[i].getLastVertex())
452 segments[i].setLastVertex(lastVertex);
453 if (i + 1 < segments.size())
455 segments[i+1].setFirstVertex(lastVertex);
This file is the main header file for the data related to annotation graphs.
HyphenWordAlternatives is the module which creates split alternatives for hyphen word tokens.
#define HYPHENWORDALTERNATIVESFACTORY_CLASSID
LinguisticGraph::in_edge_iterator LinguisticGraphInEdgeIt
boost::property_map< LinguisticGraph, vertex_data_t >::type VertexDataPropertyMap
LinguisticGraph::vertex_iterator LinguisticGraphVertexIt
boost::property_map< LinguisticGraph, vertex_token_t >::type VertexTokenPropertyMap
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
LinguisticGraph::adjacency_iterator LinguisticGraphAdjacencyIt
Defines a Factory to create Object of type Base.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Holds an annotation graph and gives an API to manipulate it.
void addMatching(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second, AnnotationGraphVertex secondVx)
Adds a symetric matching between two vertices of two graphs identified by the two string parameters.
AnnotationGraphVertex createAnnotationVertex()
Creates a new annotation vertex in the graph.
std::string & getParamsValueAtKey(const std::string &key)
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Use this exception to signal an error in one of the configuration files.
define the limaStringText resource which is the text in limaString formal
Holds morphosyntactic informations.
void setAlphaHyphen(bool isAlphaHyphen)
bool isAlphaHyphen() const
holds surface data of a token
uint64_t position() const
const TStatus & status() const
const LimaString & stringForm() const
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
virtual ~HyphenWordAlternatives()
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
const std::vector< Segment > & getSegments() const
virtual std::shared_ptr< Object > getObject(const std::string &id)
If object doesn't exists, call the create method.
static const LinguisticResources & single()
const singleton accessor
static MediaticData & changeable()
singleton accessor
This file contains a class to control log of informations about time, such as logging cumulated time ...
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
void annotate(AnnotationGraphVertex v, const LimaString &annot, uint64_t value)
LimaString utf8stdstring2limastring(const std::string &src)
SimpleFactory< MediaProcessUnit, HyphenWordAlternatives > hyphenwordAlternativesFactory(HYPHENWORDALTERNATIVESFACTORY_CLASSID)
launch exception related to the configuration file parsing