LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
WordSenseXmlLogger.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
12#include "WordSenseXmlLogger.h"
13#include "WordSenseAnnotation.h"
26
27#include <iostream>
28#include <fstream>
29
30using namespace std;
31using namespace boost;
34using namespace Lima::Common::MediaticData;
36using namespace Lima::Common::AnnotationGraphs;
37using namespace Lima::Common::Misc;
38
39
40namespace Lima
41{
42namespace LinguisticProcessing
43{
44namespace WordSenseDisambiguation
45{
46
48
52
55
58 Manager* manager)
59
60{
61 AbstractLinguisticLogger::init(unitConfiguration,manager);
62 try
63 {
64 // DTD
65 m_outputSuffix=unitConfiguration.getParamsValueAtKey("outputSuffix");
66 }
67 catch (NoSuchParam& )
68 {
69 m_outputSuffix=string(".senses")+".xml";
70 }
71 m_language=manager->getInitializationParameters().media;
72
73
74}
75
76
77// Datas are extracted from word sense annotations and written on the xml file according to the given dtd format
79 AnalysisContent& analysis) const
80{
82 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.getData("LinguisticMetaData"));
83 if (metadata == 0)
84 {
85 LOGINIT("WordSenseDisambiguator");
86 LERROR << "no LinguisticMetaData ! abort";
87 return MISSING_DATA;
88 }
89
90 string textFileName = metadata->getMetaData("FileName");
91 string outputFile = textFileName + m_outputSuffix;
92 ofstream out(outputFile.c_str(), std::ofstream::binary);
93 if (!out.good()) {
94 throw runtime_error("can't open file " + outputFile);
95 }
96
97 auto anagraph=std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("PosGraph"));
98 if (anagraph==0)
99 {
100 LOGINIT("WordSenseDisambiguator");
101 LERROR << "no AnalysisGraph ! abort";
102 return MISSING_DATA;
103 }
104
105
106 dump(out, anagraph.get(),/* static_cast<SyntacticData*>(analysis.getData("SyntacticData")),*/
107 std::dynamic_pointer_cast<AnnotationData>(analysis.getData("AnnotationData")).get());
108 out.flush();
109 out.close();
110 TimeUtils::logElapsedTime("WordSenseDisambiguatorXmlLogger");
111 return SUCCESS_ID;
112}
113
114//**********************************************************************
115// define a visitor to go through the graph and output the annotations
116
117
119 const LinguisticGraph& g)
120{
121 LinguisticGraphVertex v = target(e, g);
122 // process
123 if (m_ad->hasAnnotation(v, Common::Misc::utf8stdstring2limastring("WordSense")))
124 {
125 GenericAnnotation ga = (m_ad->annotation(v,utf8stdstring2limastring("WordSense")));
127 try
128 {
130 wsa.outputXml(m_ostream,g);
131 }
132 catch (const boost::bad_any_cast& e)
133 {
134 LOGINIT("WordSenseDisambiguator");
135 LERROR << "non word sense annotation";
136 }
137 }
138 else
139 {
140 Token* token = get(vertex_token, g, v);
141 if (token != 0)
142 {
143 std::string s = Common::Misc::limastring2utf8stdstring(token->stringForm());
144 m_ostream << s;
145 }
146 }
147 m_ostream << " ";
148}
149
150
151
152//**********************************************************************
153// dumps memory structure on XML file
154
155
156
157void WordSenseXmlLogger::dump(std::ostream& xmlStream, AnalysisGraph* anagraph, /*SyntacticAnalysis::SyntacticData* sd,*/ AnnotationData* ad) const
158{
159 if (m_outputSuffix == ".senses.xml")
160 {
161 xmlStream << "<?xml version='1.0' standalone='no'?>" << std::endl;
162 xmlStream << "<!--generated by MM project on ";
163 time_t aclock;
164 time(&aclock); /* Get time in seconds */
165 std::string str(ctime(&aclock));
166 str = str.substr(0,str.size()-1);
167 xmlStream << str;
168 xmlStream << "-->" << std::endl;
169 xmlStream << "<!DOCTYPE WORDSENSE SYSTEM \"wordsense.dtd\">" << std::endl;
170 xmlStream << std::endl;
171 xmlStream << "<TEXT>" << std::endl;
172 // dump the graph
173 DumpXMLAnnotationVisitor vis(xmlStream, /*sd,*/ ad, m_language);
174 breadth_first_search(*(anagraph->getGraph()), anagraph->firstVertex(),visitor(vis));
175
176
177 xmlStream << std::endl << std::endl << "</TEXT>" << std::endl;
178 }
179}
180
181} // WordSenseDisambiguation
182} // LinguisticProcessing
183} // Lima
AgglutinatedToken is a fulltoken that is a agglutinated compound word.
This file is the main header file for the data related to annotation graphs.
#define LOGINIT(X)
Definition LimaCommon.h:187
#define LERROR
Definition LimaCommon.h:161
boost::graph_traits< LinguisticGraph >::edge_descriptor LinguisticGraphEdge
typedefs to simplify the access to various graphs elements
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
Defines a Factory to create Object of type Base.
Data used for the syntactic analyzis of texts.
xml logger for Word Senses
#define WORDSENSEXMLLOGGER_CLASSID
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
Holds an annotation graph and gives an API to manipulate it.
const GenericAnnotation & annotation(AnnotationGraphVertex v1, AnnotationGraphVertex v2, const LimaString &annot) const
This class allows to convert any object into an annotation by inheritance.
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
A generic process unit to log information in files: contains some common informations such as : outpu...
virtual void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
virtual void outputXml(std::ostream &xmlStream, const LinguisticGraph &g) const
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
static void logElapsedTime(const std::string &mess, const std::string &taskCategory=std::string(""))
log the number of microseconds since last UpdateCurrentTime
static void updateCurrentTime(const std::string &taskCategory=std::string(""))
store current time for new elapsed time computation
bool hasAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
LimaString utf8stdstring2limastring(const std::string &src)
SimpleFactory< MediaProcessUnit, WordSenseXmlLogger > fullTokenXmlLoggerFactory(WORDSENSEXMLLOGGER_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
@ MISSING_DATA
Definition LimaCommon.h:243
STL namespace.
launch exception related to the configuration file parsing