LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
CorefSolvingXmlLogger.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
27
28#include <iostream>
29#include <fstream>
30#include <QtCore/QDateTime>
31
32using namespace std;
33using namespace boost;
36using namespace Lima::Common::MediaticData;
38using namespace Lima::Common::AnnotationGraphs;
39using namespace Lima::Common::Misc;
40
41
42namespace Lima
43{
44namespace LinguisticProcessing
45{
46namespace Coreferences
47{
48
50
54
57
60 Manager* manager)
61
62{
63 AbstractLinguisticLogger::init(unitConfiguration,manager);
64 try
65 {
66 // DTD
67 m_outputSuffix=unitConfiguration.getParamsValueAtKey("outputSuffix")+".xml";
68 }
69 catch (NoSuchParam& )
70 {
71 m_outputSuffix=string(".coref")+".xml";
72 }
73 m_language=manager->getInitializationParameters().media;
74
75
76 //m_propertyCodeManager= &(static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(m_language)).getPropertyCodeManager());
77}
78
79
80// Datas are extracted from coref annotations and written on the xml file according to the given dtd format
82 AnalysisContent& analysis) const
83{
85 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.getData("LinguisticMetaData"));
86 if (metadata == 0)
87 {
89 LERROR << "no LinguisticMetaData ! abort";
90 return MISSING_DATA;
91 }
92
93 string textFileName = metadata->getMetaData("FileName");
94 string outputFile = textFileName + m_outputSuffix;
95 ofstream out(outputFile.c_str(), std::ofstream::binary);
96 if (!out.good()) {
97 throw runtime_error("can't open file " + outputFile);
98 }
99
100 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("PosGraph"));
101 if (anagraph==0)
102 {
104 LERROR << "no PosGraph ! abort";
105 return MISSING_DATA;
106 }
107
108
109 dump(out, anagraph.get(),/* static_cast<SyntacticData*>(analysis.getData("SyntacticData")),*/
110 std::dynamic_pointer_cast<AnnotationData>(analysis.getData("AnnotationData")).get());
111 out.flush();
112 out.close();
113 TimeUtils::logElapsedTime("CorefSolverXmlLogger");
114 return SUCCESS_ID;
115}
116
117//**********************************************************************
118// define a visitor to go through the graph and output the annotations
119
120
122 const LinguisticGraph& g)
123{
125 LDEBUG << "DumpXMLAnnotationVisitor::examine_edge";
126 LinguisticGraphVertex v = target(e, g);
127 // let process sentences like (...) have automatically tuned (...) where the graph has one token "have_tuned" with one branch "automatically" "tuned" and another one with the following of the sentence
128 LinguisticGraphOutEdgeIt it, it_end;
129 boost::tie(it, it_end) = boost::out_edges(v,g);
130 if (it == it_end)
131 return;
132 // let process sentences where one tag has not been fully determined and there is still two (or more) tag options
133 LinguisticGraphVertex v2 = target(m_lastEdge, g);
134 if (v2==v)
135 return;
136 if (m_lastEdge!=LinguisticGraphEdge() && are_equivalent(e, v2, v, g))
137 return;
138 // begin
139 // store this edge for the future tests
140 if (get(vertex_token, g,v)!=0)
141 m_lastEdge = e;
142// const FsaStringsPool& stringsPool= Common::MediaticData::MediaticData::single().stringsPool(m_language);
143 Token* token = get(vertex_token, g, v);
144 // processing of cases like "s'y introduire", tokenized as "y s'introduire"
145 if (token != 0 && (token->stringForm() == "en" || token->stringForm() =="y"))
146 {
147 LinguisticGraphOutEdgeIt it, it_end;
148 boost::tie(it, it_end) = boost::out_edges(v,g);
149 if (it != it_end)
150 {
151 Token* t = get(vertex_token, g,target(*it, g));
152 if (t!=0 && Common::Misc::limastring2utf8stdstring(t->stringForm()).substr(0,2)=="s'")
153 {
154 m_ostream << "s'";
155 }
156 }
157 }
158 // process
159 std::set< AnnotationGraphVertex > matches = m_ad->matches("PosGraph",v,"annot");
160 if (matches.empty())
161 {
163 LERROR << "DumpXMLAnnotationVisitor::examine_edge No annotation graph vertex matches PoS graph vertex " << v << ". This should not happen.";
164 return;
165 }
166 AnnotationGraphVertex av = *matches.begin();
167
168
169
170 if (m_ad->hasAnnotation(av, Common::Misc::utf8stdstring2limastring("Coreferent")))
171 {
172 GenericAnnotation ga = (m_ad->annotation(av,utf8stdstring2limastring("Coreferent")));
174 try
175 {
177 ca.outputXml(m_ostream,g,m_ad);
178 }
179 catch (const boost::bad_any_cast& )
180 {
182 LERROR << "non coreferent annotation";
183 }
184 }
185 else
186 {
187 Token* token = get(vertex_token, g, v);
188 if (token != 0)
189 {
190 std::string s = Common::Misc::limastring2utf8stdstring(token->stringForm());
191 // processing of cases like "s'y introduire", tokenized as "y s'introduire"
192 if (s.substr(0,2) == "s'")
193 {
194 Token* t = get(vertex_token,g,source(e, g));
195 if (t!=0 && (Common::Misc::limastring2utf8stdstring(t->stringForm()).substr(0,2)=="en" || Common::Misc::limastring2utf8stdstring(t->stringForm()).substr(0,2)=="y"))
196 {
197 s = s.substr(2,s.size());
198 }
199 }
200 // // processing of cases like "le Canada a-t-il envisagé...", où le mot entre "a" et "envisagé" se retrouverait rejeté après "a_envisagé". Nécessaire de traiter car problématique pour l'évaluation quand il s'agit d'un pronom clitique comme dans ce cas-ci.
201 // std::string formerMemo = m_memo;
202 // match_results<std::string::const_iterator> what;
203 // string::const_iterator start = s.begin();
204 // string::const_iterator end = s.end();
205 // if (regex_search(s, what, regex("_")))
206 // {
207 // m_memo = std::string(what[0].second,end) + " ";
208 // s = std::string(start,what[0].first);
209 // }
210 // else m_memo = "";
211 // m_ostream << formerMemo << s;
212
213 m_ostream << s;
214 if (token->status().isAlphaPossessive())
215 {
216 m_ostream << "'s ";
217 }
218 }
219 }
220 m_ostream << " ";
221}
222
223
224
226 LinguisticGraphEdge currentEdge,
227 LinguisticGraphVertex vProcessed,
228 LinguisticGraphVertex vNotProcessed,
229 const LinguisticGraph& g)
230{
232 LinguisticGraphInEdgeIt it, it_end;
233 boost::tie(it, it_end) = boost::in_edges(vProcessed,g);
234 if (it != it_end)
235 {
236 refEdge = *it;
237 }
238 if (refEdge!=LinguisticGraphEdge()
239 && (vProcessed != vNotProcessed)
240 && get(vertex_token, g,vProcessed)!=0
241 && get(vertex_token, g,vNotProcessed)!=0
242 && Common::Misc::limastring2utf8stdstring(get(vertex_token, g,vProcessed)->stringForm())==Common::Misc::limastring2utf8stdstring(get(vertex_token, g,vNotProcessed)->stringForm())
243 && source(refEdge,g)==source(currentEdge,g))
244 {
245 return true;
246 }
247 if (currentEdge!=DependencyGraphEdge() && source(currentEdge,g)!=vProcessed)
248 {
249 if (refEdge!=LinguisticGraphEdge())
250 {
251 vProcessed = source(*it,g);
252 }
253 vNotProcessed = source(currentEdge,g);
254 boost::tie(it, it_end) = boost::in_edges(vNotProcessed,g);
255 if (it != it_end)
256 {
257 currentEdge = *it;
258 }
259 return are_equivalent(currentEdge, vProcessed, vNotProcessed,g);
260 }
261 return false;
262}
263//**********************************************************************
264// dumps memory structure on XML file
265
266
267
268void CorefSolvingXmlLogger::dump(std::ostream& xmlStream, AnalysisGraph* anagraph, /*SyntacticAnalysis::SyntacticData* sd,*/ AnnotationData* ad) const
269{
270 if (m_outputSuffix == ".wh.xml")
271 {
272 xmlStream << "<?xml version='1.0' standalone='no'?>" << std::endl;
273 xmlStream << "<!--generated by Amose on ";
274 xmlStream << QDateTime::currentDateTime().toString().toUtf8().data();
275 xmlStream << "-->" << std::endl;
276 xmlStream << "<!DOCTYPE COREF SYSTEM \"coref.dtd\">" << std::endl;
277 xmlStream << std::endl;
278 xmlStream << "<TEXT>" << std::endl;
279 // dump the graph
280 DumpXMLAnnotationVisitor vis(xmlStream, /*sd,*/ ad, m_language);
281 breadth_first_search(*(anagraph->getGraph()), anagraph->firstVertex(),visitor(vis));
282
283
284 xmlStream << std::endl << std::endl << "</TEXT>" << std::endl;
285 }
286}
287
288} // Coreferences
289} // LinguisticProcessing
290} // Lima
AgglutinatedToken is a fulltoken that is a agglutinated compound word.
This file is the main header file for the data related to annotation graphs.
xml logger for coreferences
#define COREFSOLVINGXMLLOGGER_CLASSID
DependencyGraph::edge_descriptor DependencyGraphEdge
typedefs to simplify the acces to various graphs elements
#define LDEBUG
Definition LimaCommon.h:157
#define LERROR
Definition LimaCommon.h:161
LinguisticGraph::in_edge_iterator LinguisticGraphInEdgeIt
boost::graph_traits< LinguisticGraph >::edge_descriptor LinguisticGraphEdge
typedefs to simplify the access to various graphs elements
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define COREFSOLVERLOGINIT
Defines a Factory to create Object of type Base.
Data used for the syntactic analyzis of texts.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
Holds an annotation graph and gives an API to manipulate it.
std::set< AnnotationGraphVertex > matches(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second) const
Gets the set of vertices matched in the second graph by the given vertex of the first graph.
const GenericAnnotation & annotation(AnnotationGraphVertex v1, AnnotationGraphVertex v2, const LimaString &annot) const
This class allows to convert any object into an annotation by inheritance.
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
A generic process unit to log information in files: contains some common informations such as : outpu...
virtual void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
virtual void outputXml(std::ostream &xmlStream, const LinguisticGraph &g, const Common::AnnotationGraphs::AnnotationData *ad) const
void examine_edge(LinguisticGraphEdge e, const LinguisticGraph &g)
bool are_equivalent(LinguisticGraphEdge currentEdge, LinguisticGraphVertex vProcessed, LinguisticGraphVertex vNotProcessed, const LinguisticGraph &g)
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
static void logElapsedTime(const std::string &mess, const std::string &taskCategory=std::string(""))
log the number of microseconds since last UpdateCurrentTime
static void updateCurrentTime(const std::string &taskCategory=std::string(""))
store current time for new elapsed time computation
AnnotationGraph::vertex_descriptor AnnotationGraphVertex
bool hasAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
LimaString utf8stdstring2limastring(const std::string &src)
SimpleFactory< MediaProcessUnit, CorefSolvingXmlLogger > fullTokenXmlLoggerFactory(COREFSOLVINGXMLLOGGER_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
@ MISSING_DATA
Definition LimaCommon.h:243
STL namespace.
launch exception related to the configuration file parsing