LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
SyntacticAnalyzer-disamb.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
18
19// #include "ChainsDisambiguator.h"
23
24using namespace Lima::Common::MediaticData;
26
27namespace Lima
28{
29namespace LinguisticProcessing
30{
31namespace SyntacticAnalysis
32{
33
34static const uint64_t DEFAULT_DEPGRAPHMAXBRANCHINGFACTOR = 30;
35
37
41
44 Manager* manager)
45
46{
49 try
50 {
51 std::string depGraphMaxBranchingFactorS=unitConfiguration.getParamsValueAtKey("depGraphMaxBranchingFactor");
52 std::istringstream iss(depGraphMaxBranchingFactorS);
54 }
56 {
57 LWARN << "no parameter 'depGraphMaxBranchingFactor' in syntacticAnalyzerDisamb group for language " << (int) m_language << " ! Using default value << " << DEFAULT_DEPGRAPHMAXBRANCHINGFACTOR << ".";
58 }
59}
60
62 AnalysisContent& analysis) const
63{
64 Lima::TimeUtilsController timer("SyntacticAnalysis");
66 LINFO << "start syntactic analysis - disambiguation";
67 // create syntacticData
68 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("PosGraph"));
69 if (anagraph==0)
70 {
71 LERROR << "no AnalysisGraph ! abort";
72 return MISSING_DATA;
73 }
74 auto sb = std::dynamic_pointer_cast<SegmentationData>(analysis.getData("SentenceBoundaries"));
75 if (sb==0)
76 {
77 LERROR << "no sentence bounds ! abort";
78 return MISSING_DATA;
79 }
80 if (sb->getGraphId() != "PosGraph") {
81 LERROR << "SentenceBounds have been computed on " << sb->getGraphId() << " !";
82 LERROR << "SyntacticAnalyzer-deps needs SentenceBounds on PosGraph";
84 }
85
86 // if (analysis.getData("SyntacticData")==0)
87 // {
88 // auto syntacticData = std::make_shared< SyntacticData >(anagraph.get(), nullptr);
89 // syntacticData->setupDependencyGraph();
90 // analysis.setData("SyntacticData", syntacticData);
91 // }
92
93
94// bool l2r = true;
95 // ??OME2 for (SegmentationData::const_iterator boundItr=sb->begin();
96 // boundItr!=sb->end();
97// for (auto boundItr=(sb->getSegments()).begin(); boundItr!=(sb->getSegments()).end(); boundItr++)
98// {
99// LinguisticGraphVertex beginSentence=boundItr->getFirstVertex();
100// LinguisticGraphVertex endSentence=boundItr->getLastVertex();
101// #ifdef DEBUG_LP
102// LDEBUG << "analyze sentence from vertex " << beginSentence << " to vertex " << endSentence;
103// #endif
104 // ChainsDisambiguator cd(std::dynamic_pointer_cast<SyntacticData>(analysis.getData("SyntacticData")).get(),
105 // beginSentence, endSentence, m_language, m_depGraphMaxBranchingFactor);
106 // cd.initPaths();
107 // cd.computePaths();
108 // cd.applyDisambiguisation();
109/* LinguisticGraphVertex current, next;
110 current = beginSentence; next = current;
111 while (next != endSentence)
112 {
113 next = nextChainsDisambBreakFrom(current, *anagraph, endSentence);
114 LDEBUG << "Disambiguate chains between " << current << " and " << next;
115 ChainsDisambiguator cd(dynamic_cast<SyntacticData*>(analysis.getData("SyntacticData")),
116 current, next);
117 cd.initPaths();
118 cd.computePaths();
119 cd.applyDisambiguisation();
120 current = next;
121 }*/
122 // }
123
124 LINFO << "end syntactic analysis - disambiguation";
125 return SUCCESS_ID;
126}
127
146// LinguisticGraphVertex SyntacticAnalyzerDisamb::nextChainsDisambBreakFrom(
147// const LinguisticGraphVertex& v,
148// const AnalysisGraph& anagraph,
149// const LinguisticGraphVertex& nextSentenceBreak) const
150// {
151//
152// const LinguisticGraph& graph = *(anagraph.getGraph());
153//
154// LinguisticGraphVertex current = v;
155// while (boost::out_degree(current, graph) == 1)
156// {
157// // LDEBUG << "On " << current;
158// if (current == anagraph.lastVertex() || current == nextSentenceBreak) return current;
159// current = boost::target(*(boost::out_edges(current, graph).first), graph);
160// }
161// // LDEBUG << "Entering loop on " << current;
162// while(true)
163// {
164// std::list< LinguisticGraphVertex > fifo;
165// std::set< LinguisticGraphVertex > infifo;
166// std::set< LinguisticGraphVertex > finished;
167// fifo.push_back(current);
168// LinguisticGraphInEdgeIt init, init_end;
169// boost::tie(init, init_end) = boost::in_edges(current, graph);
170// for (; init != init_end; init++)
171// {
172// finished.insert(source(*init,graph));
173// }
174// bool first = true;
175// while (!fifo.empty())
176// {
177// current = fifo.front();
178// fifo.pop_front();
179// // LDEBUG << "On " << current;
180// bool curfinished = true;
181// if (finished.find(current) == finished.end())
182// {
183// LinguisticGraphInEdgeIt init, init_end;
184// boost::tie(init, init_end) = boost::in_edges(current, graph);
185// while (curfinished && init != init_end)
186// {
187// if ( finished.find(source(*init,graph)) == finished.end() )
188// curfinished = false;
189// init++;
190// }
191// if (curfinished) finished.insert(current);
192// }
193// if (fifo.empty() && curfinished && !first) break;
194// else
195// {
196// first = false;
197// LinguisticGraphOutEdgeIt it, it_end;
198// boost::tie(it, it_end) = boost::out_edges(current, graph);
199// for (; it != it_end; it++)
200// {
201// if (infifo.find(target(*it, graph)) == infifo.end())
202// {
203// fifo.push_back(target(*it, graph));
204// infifo.insert(target(*it, graph));
205// }
206// }
207// // fifo.push_back(current);
208// }
209// if (current == anagraph.lastVertex() || current==nextSentenceBreak)
210// {
211// if (current != nextSentenceBreak)
212// {
213// SADLOGINIT;
214// LERROR << "In nextChainsBreakFrom: went beyond next sentence break " << nextSentenceBreak;
215// LERROR << " returning graph's last vertex " << current;
216// }
217// // LDEBUG << "Next chains break is: " << current;
218// return current;
219// }
220// }
221// // LDEBUG << "Testing end only on " << current;
222// CVertexChainIdPropertyMap chainsMap = boost::get( vertex_chain_id, graph );
223// const std::set< Lima::LinguisticProcessing::LinguisticAnalysisStructure::ChainIdStruct >& chains = chainsMap[current];
224// if (chains.empty())
225// return current;
226// std::set< Lima::LinguisticProcessing::LinguisticAnalysisStructure::ChainIdStruct >::const_iterator itc, itc_end;
227// itc = chains.begin(); itc_end = chains.end();
228// bool endonly = true;
229// while (endonly && itc != itc_end)
230// {
231// if ( ( (*itc).elemType() == BEGIN ) || ( ( (*itc).elemType() == PART ) ) )
232// endonly = false;
233// itc++;
234// }
235// if (endonly) return current;
236// }
237// }
238
239
240} // closing namespace SyntacticAnalysis
241} // closing namespace LinguisticProcessing
242} // closing namespace Lima
#define LWARN
Definition LimaCommon.h:160
#define LINFO
Definition LimaCommon.h:158
#define LERROR
Definition LimaCommon.h:161
#define SALOGINIT
Defines a Factory to create Object of type Base.
#define SYNTACTICANALYZERDISAMB_CLASSID
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
This file contains a class to control log of informations about time, such as logging cumulated time ...
SimpleFactory< MediaProcessUnit, SyntacticAnalyzerDisamb > syntacticAnalyzerDisambFactory(SYNTACTICANALYZERDISAMB_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
@ INVALID_CONFIGURATION
Definition LimaCommon.h:242
@ MISSING_DATA
Definition LimaCommon.h:243