LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
TextFeaturesDumper.cpp
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/***************************************************************************
7 * Copyright (C) 2004-2012 by CEA LIST *
8 * *
9 ***************************************************************************/
10#include "TextFeaturesDumper.h"
11
25
26#include <boost/algorithm/string/replace.hpp>
27
28#include <fstream>
29#include <queue>
30
31using namespace std;
32//using namespace boost;
33using namespace boost::tuples;
35using namespace Lima::Common::MediaticData;
37
38namespace Lima {
39namespace LinguisticProcessing {
40namespace AnalysisDumpers {
41
43
46m_graph("PosGraph"),
47m_sep(" "),
48m_sepReplace("_"),
49m_sepPOS("#"),
50m_features()
51{}
52
53
56
58 Manager* manager)
59
60{
61 AbstractTextualAnalysisDumper::init(unitConfiguration,manager);
62
64 try
65 {
66 m_graph=unitConfiguration.getParamsValueAtKey("graph");
67 }
68 catch (NoSuchParam& ) {} // keep default value
69
70 try {
71 m_sep=unitConfiguration.getParamsValueAtKey("sep");
72 }
73 catch (NoSuchParam& ) {} // keep default value
74
75 try {
76 m_sepReplace=unitConfiguration.getParamsValueAtKey("sep_replace");
77 }
78 catch (NoSuchParam& ) {} // keep default value
79
80 try {
81 m_sepPOS=unitConfiguration.getParamsValueAtKey("sepPOS");
82 }
83 catch (NoSuchParam& ) {} // keep default value
84
85 try {
86 std::deque<string> featureList=unitConfiguration.getListsValueAtKey("features");
87 // initialize feature access
88 m_features.setLanguage(m_language);
89 m_features.initialize(featureList);
90 }
91 catch (NoSuchList& ) { // keep default value
92 LOGINIT("LP::Dumper");
93 LERROR << "Warning: no features selected in TextFeaturesDumper: output will be empty";
94 }
95
96}
97
99 AnalysisContent& analysis) const
100{
102 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.getData("LinguisticMetaData"));
103 if (metadata == 0) {
104 LERROR << "no LinguisticMetaData ! abort";
105 return MISSING_DATA;
106 }
107
108 auto dstream = initialize(analysis);
109
110 map<Token*,LinguisticGraphVertex,lTokenPosition > categoriesMapping;
111
112 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData(m_graph));
113 if (anagraph==0) {
114 LERROR << "graph " << m_graph << " has not been produced: check pipeline";
115 return MISSING_DATA;
116 }
117 LinguisticGraph* graph=anagraph->getGraph();
118 // const FsaStringsPool& sp=Common::MediaticData::MediaticData::single().stringsPool(m_language);
119
120 // instead of looking to all vertices, follow the graph (in
121 // morphological graph, some vertices are not related to main graph:
122 // idiomatic expressions parts and named entity parts)
123
124 std::queue<LinguisticGraphVertex> toVisit;
125 std::set<LinguisticGraphVertex> visited;
126 toVisit.push(anagraph->firstVertex());
127
128 LinguisticGraphOutEdgeIt outItr,outItrEnd;
129 while (!toVisit.empty()) {
130 LinguisticGraphVertex v=toVisit.front();
131 toVisit.pop();
132 if (v == anagraph->lastVertex()) {
133 continue;
134 }
135
136 for (boost::tie(outItr,outItrEnd)=out_edges(v,*graph); outItr!=outItrEnd; outItr++)
137 {
138 LinguisticGraphVertex next=target(*outItr,*graph);
139 if (visited.find(next)==visited.end())
140 {
141 visited.insert(next);
142 toVisit.push(next);
143 }
144 }
145
146 Token* ft=get(vertex_token,*graph,v);
147 if( ft!=0) {
148 categoriesMapping[ft]=v;
149 }
150 }
151
152 for (map<Token*,LinguisticGraphVertex,lTokenPosition >::const_iterator ftItr=categoriesMapping.begin();
153 ftItr!=categoriesMapping.end();
154 ftItr++)
155 {
156 outputVertex(dstream->out(),anagraph.get(),ftItr->second,analysis,metadata->getStartOffset());
157 }
158
159
160 return SUCCESS_ID;
161}
162
163
164void TextFeaturesDumper::outputVertex(ostream& out,
165 const AnalysisGraph* graph,
167 AnalysisContent& analysis,
168 uint64_t offset) const
169{
170 //TODO : use offset
171 LIMA_UNUSED(offset)
172
173 bool first=true;
174 for (WordFeatures::const_iterator it=m_features.begin(),it_end=m_features.end();
175 it!=it_end; it++)
176 {
177 if (first) { first=false; }
178 else {
179 out << m_sep;
180 }
181 // take only first morphosyntactic data
182 string str=(*it)->getValue(graph,v,analysis);
183 boost::replace_all(str,m_sep,m_sepReplace);
184 out << str;
185 }
186 out << endl;
187}
188
189} // end namespace
190} // end namespace
191} // end namespace
#define LIMA_UNUSED(x)
Definition LimaCommon.h:224
#define LOGINIT(X)
Definition LimaCommon.h:187
#define LERROR
Definition LimaCommon.h:161
A graph structure for linguistic analysis.
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define DUMPERLOGINIT
Defines a Factory to create Object of type Base.
#define TEXTFEATURESDUMPER_CLASSID
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
std::deque< std::string > & getListsValueAtKey(const std::string &key)
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
std::shared_ptr< DumperStream > initialize(AnalysisContent &analysis) const
virtual void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
An AnalysisData containing a LinguisticGraph with a language and an id.
void initialize(const std::deque< std::string > &featureNames)
SimpleFactory< MediaProcessUnit, TextFeaturesDumper > textFeaturesDumperFactory(TEXTFEATURESDUMPER_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
@ MISSING_DATA
Definition LimaCommon.h:243
STL namespace.
launch exception related to the configuration file parsing