LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
DicoConcatenatedAlternatives.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6// NAUTITIA
7//
8// jys 17-JAN-2003
9//
10// DicoConcatenatedAlternatives is the module which creates split alternatives
11// for concatenated expression tokens found into dictionary.
12// Rules :
13// <each FullToken of the main path is processed. Alternative paths
14// are not processed>
15// <there are as many created alternative paths as there are concatenated
16// entries associated with main Token and orthographic alternative Tokens>
17// <each concatenated entry gives FullToken path. Each token has the original
18// FullToken localization>
19// <if concatenated entry supplies dictionary entry, main Token of the just
20// created FullToken takes this entry. Otherwise, a dictionary access is
21// performed to find dictionary entry>
22
24
25#include "common/misc/LimaString.h"
26// #include "common/linguisticData/linguisticData.h"
27#include "common/misc/traceUtils.h"
34
35#include <list>
36
37using namespace std;
38using namespace boost;
40
41namespace Lima
42{
43namespace LinguisticProcessing
44{
45namespace MorphologicAnalysis
46{
47
49
52
55
56
63
65 AnalysisContent& analysis) const
66{
69 LINFO << "starting process DicoConcatenated";
70
71 auto tokenList=std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"));
72 LinguisticGraph* g=tokenList->getGraph();
74 VertexDataPropertyMap dataMap=get(vertex_data,*g);
75 VertexTokenPropertyMap tokenMap=get(vertex_token,*g);
76 boost::tie(it,itEnd)=vertices(*g);
77 for (;it!=itEnd;it++)
78 {
79 MorphoSyntacticData* currentData=dataMap[*it];
80 Token* currentToken=tokenMap[*it];
81 if (currentToken==0) continue;
82
83 // if no concatenated entries, skip
84 DictionaryEntry* entry=currentToken->dictionaryEntry();
85 if (entry->hasConcatenated())
86 {
87
88 // if some, then insert it in the graph and remove source token
89 // retrieve preds and succs vertices
90 expandConcatenatedEntries(*it,g,currentToken,entry);
91
92 // delete source token
93 clear_vertex(*it,*g);
94 }
95 else if (!currentToken->orthographicAlternatives().empty())
96 {
97 const std::vector< Token* >& orthos=currentToken->orthographicAlternatives();
98 bool expanded=false;
99 for (vector<Token*>::const_iterator tokItr=orthos.begin();
100 tokItr!=orthos.end();
101 tokItr++)
102 {
103 DictionaryEntry* entry=(*tokItr)->dictionaryEntry();
104 if (entry->hasConcatenated())
105 {
106 expandConcatenatedEntries(*it,g,currentToken,entry);
107 expanded=true;
108 }
109 }
110 // if expanded and no more linguistic info available remove source token
111 if (expanded && currentData->empty())
112 {
113 clear_vertex(*it,*g);
114 }
115 }
116 }
117 LINFO << "ending process DicoConcatenated";
118 TimeUtils::logElapsedTime("DicoConcatenatedAlternatives");
119 return SUCCESS_ID;
120
121}
122
123void DicoConcatenatedAlternatives::expandConcatenatedEntries(
126 Token* token,
127 DictionaryEntry* entry) const
128{
130 LDEBUG << "DicoConcatenatedAlternatives::expandConcatenatedEntries" << v;
131 list<LinguisticGraphVertex> preds,succs;
132 LinguisticGraphInEdgeIt inItr,inItrEnd;
133 boost::tie(inItr,inItrEnd)=in_edges(v,*g);
134 for (;inItr!=inItrEnd;inItr++)
135 {
136 preds.push_back(source(*inItr,*g));
137 }
138 LinguisticGraphOutEdgeIt outItr,outItrEnd;
139 boost::tie(outItr,outItrEnd)=out_edges(v,*g);
140 for (;outItr!=outItrEnd;outItr++)
141 {
142 succs.push_back(target(*outItr,*g));
143 }
144
145 entry->reset();
146 if (entry->hasConcatenated())
147 {
148 StringsPool& sp=Common::LinguisticData::LinguisticData::changeable().stringsPool(m_language);
149 ConcatenatedEntry concatEntry=entry->nextConcatenated();
150 while (!concatEntry.isEmpty())
151 {
152 list<LinguisticGraphVertex> localpreds(preds);
153 concatEntry.reset();
154 SingleConcatenatedEntry singleEntry=concatEntry.nextSingleConcatenated();
155 while (!singleEntry.isEmpty())
156 {
157 Lima::LimaString component = singleEntry.component();
158 unsigned char* adr = singleEntry.dictionaryEntryAddress();
159 Dictionary::DictionaryEntry* ent=new Dictionary::DictionaryEntry(component, entry->stringStartAddr(), entry->lingPropertiesStartAddr(), adr);
160
161 // create Token
162 Token* nft=new Token(*token);
164 ndata->appendLingInfo(nft->form(),ent,CONCATENATED_ALTERNATIVE,sp);
165 nft->setDictionaryEntry(ent);
166 // add vertex in graph
167 LinguisticGraphVertex nv=add_vertex(*g);
168 put(vertex_token,*g,nv,nft);
169 put(vertex_data,*g,nv,ndata);
170 // link it to predecessors
171 for (list<LinguisticGraphVertex>::const_iterator predItr=localpreds.begin();
172 predItr!=localpreds.end();
173 predItr++)
174 {
175 add_edge(*predItr,nv,*g);
176 }
177 // nv is the predecessor for the next vertex
178 localpreds.clear();
179 localpreds.push_back(nv);
180 singleEntry=concatEntry.nextSingleConcatenated();
181 }
182 // link to successors
183 for (list<LinguisticGraphVertex>::const_iterator predItr=localpreds.begin();
184 predItr!=localpreds.end();
185 predItr++)
186 {
187 for (list<LinguisticGraphVertex>::const_iterator succItr=succs.begin();
188 succItr!=succs.end();
189 succItr++)
190 {
191 add_edge(*predItr,*succItr,*g);
192 }
193 }
194 concatEntry=entry->nextConcatenated();
195 }
196 }
197}
198
199
200} // MorphologicAnalysis
201} // LinguisticProcessing
202} // Lima
#define DICOCONCATENATEDALTERNATIVES_CLASSID
#define LDEBUG
Definition LimaCommon.h:157
#define LINFO
Definition LimaCommon.h:158
LinguisticGraph::in_edge_iterator LinguisticGraphInEdgeIt
boost::property_map< LinguisticGraph, vertex_data_t >::type VertexDataPropertyMap
LinguisticGraph::vertex_iterator LinguisticGraphVertexIt
boost::property_map< LinguisticGraph, vertex_token_t >::type VertexTokenPropertyMap
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define MORPHOLOGINIT
Defines a Factory to create Object of type Base.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
const std::vector< StringsPoolIndex > & orthographicAlternatives() const
Definition Token.h:112
LimaStatusCode process(AnalysisContent &analysis) const
Process on data in analysisContent.
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager)
initialize with parameters from configuration file.
static void logElapsedTime(const std::string &mess, const std::string &taskCategory=std::string(""))
log the number of microseconds since last UpdateCurrentTime
static void updateCurrentTime(const std::string &taskCategory=std::string(""))
store current time for new elapsed time computation
SimpleFactory< LinguisticProcessUnit, DicoConcatenatedAlternatives > dicoConcatenatedAlternativesFactory(DICOCONCATENATEDALTERNATIVES_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
QString LimaString
Definition LimaString.h:33
STL namespace.
launch exception related to the configuration file parsing