27#include <boost/graph/properties.hpp>
36using namespace boost::tuples;
46namespace LinguisticProcessing {
47namespace AnalysisDumpers {
57m_outputVerbTense(false),
58m_outputTStatus(false),
59m_encapsulatingTag(
""),
94 if (str==
"yes" || str==
"1") {
96 LINFO <<
"activate outputTStatus";
103 if (str==
"yes" || str==
"1") {
107 m_tenseManager=&codeManager.getPropertyManager(timeCode.toUtf8().constData());
108 m_tenseAccessor=&codeManager.getPropertyAccessor(timeCode.toUtf8().constData());
109 LINFO <<
"activate outputVerbTense";
120 LDEBUG <<
"no encapsulatingTag option set. Keep default";
131 LDEBUG <<
"SimpleXmlDumper::process";
133 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.
getData(
"LinguisticMetaData"));
136 LERROR <<
"no LinguisticMetaData ! abort";
140 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"AnalysisGraph"));
143 LERROR <<
"no graph 'AnaGraph' available !";
146 auto posgraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.
getData(
"PosGraph"));
149 LERROR <<
"no graph 'PosGraph' available !";
152 auto annotationData = std::dynamic_pointer_cast< AnnotationData >(analysis.
getData(
"AnnotationData"));
153 if (annotationData==0)
155 LERROR <<
"no annotation graph available !";
160 xmlOutput(dstream->out(), analysis, anagraph.get(), posgraph.get(), annotationData.get());
175 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.
getData(
"LinguisticMetaData"));
177 auto sb = std::dynamic_pointer_cast<SegmentationData>(analysis.
getData(
"SentenceBoundaries"));
182 out <<
"<?xml version=\"1.0\" encoding=\"UTF-8\" standalone=\"no\" ?>" << std::endl;
188 docId.fromStdString(metadata->getMetaData(
"FileName"));
194 out <<
"<DOC id=\"" << docId.toHtmlEscaped().toStdString()
195 <<
"\" lang=\"" << language <<
"\">" << std::endl;
199 LWARN <<
"no SentenceBoundaries";
214 metadata->getStartOffset());
219 auto nbSentences = (sb->getSegments()).size();
220 LDEBUG <<
"SimpleXmlDumper: "<< nbSentences <<
" sentences found";
221 for (uint64_t i = 0; i < nbSentences; i++)
225 auto sentenceBegin = (sb->getSegments())[i].getFirstVertex();
226 auto sentenceEnd = (sb->getSegments())[i].getLastVertex();
232 LDEBUG <<
"dump sentence between " << sentenceBegin <<
" and " << sentenceEnd;
233 LDEBUG <<
"dump simple terms for this sentence";
235 std::ostringstream oss;
243 metadata->getStartOffset());
244 std::string str = oss.str();
247 LDEBUG <<
"nothing to dump in this sentence";
251 out <<
"<s id=\"" << i <<
"\">" << std::endl
253 <<
"</s>" << std::endl;
257 out <<
"</DOC>" << std::endl;
271 const uint64_t offset)
const
274 LDEBUG <<
"SimpleXmlDumper: ========================================";
275 LDEBUG <<
"SimpleXmlDumper: outputXml from vertex " << begin <<
" to vertex " << end;
281 std::vector< std::pair<LinguisticGraphVertex, MorphoSyntacticData*> >,
284 std::queue<LinguisticGraphVertex> toVisit;
285 std::set<LinguisticGraphVertex> visited;
295 while (!toVisit.empty())
297 auto v = toVisit.front();
299 if (last || v == lastVertex)
308 for (boost::tie(outItr, outItrEnd) = out_edges(v, *graph);
309 outItr != outItrEnd; outItr++)
311 auto next = target(*outItr,*graph);
312 if (visited.find(next) == visited.end())
314 visited.insert(next);
328 sortedTokens[t].push_back(make_pair(v, get(
vertex_data, *graph, v)));
333 for (
auto it = sortedTokens.begin(), it_end = sortedTokens.end();
336 if ((*it).second.size() == 0)
343 else if ((*it).second.size() > 1)
345 std::vector<MorphoSyntacticData*> data;
346 for (
auto d = (*it).second.begin(), d_end = (*it).second.end();
349 data.push_back((*d).second);
356 posgraph, annotationData, sp, offset);
369 uint64_t offset)
const
375 auto anaVertices = annotationData->
matches(
"PosGraph", v,
"AnalysisGraph");
377 for (
auto anaVerticesIt = anaVertices.begin();
378 anaVerticesIt != anaVertices.end(); anaVerticesIt++)
380 auto matches = annotationData->
matches(
"AnalysisGraph",
383 for (
auto vx : matches)
385 if (annotationData->
hasAnnotation(vx, QString::fromUtf8(
"SpecificEntity")))
388 vx, QString::fromUtf8(
"SpecificEntity")).
389 pointerValue<SpecificEntityAnnotation>();
397 LERROR <<
"failed to output specific entity for vertex " << v;
404 auto matches = annotationData->
matches(
"PosGraph", v,
"annot");
405 for (
auto vx : matches)
407 if (annotationData->
hasAnnotation(vx, QString::fromUtf8(
"SpecificEntity")))
410 auto se = annotationData->
annotation(vx, QString::fromUtf8(
"SpecificEntity")).
411 pointerValue<SpecificEntityAnnotation>();
420 LERROR <<
"failed to output specific entity for vertex " << v;
432 const std::vector<MorphoSyntacticData*>& data,
438 auto position = ft->
position() + offset;
442 for (
auto dataItr = data.cbegin(), dataItr_end = data.cend();
443 dataItr!=dataItr_end; dataItr++)
445 auto data = *dataItr;
446 sort(data->begin(), data->end(), sorter);
447 StringsPoolIndex norm(0),curNorm(0);
449 for (
auto elemItr = data->cbegin(); elemItr != data->cend(); elemItr++)
451 curNorm = elemItr->normalizedForm;
453 if ((curNorm != norm) || (curMicro != micro))
459 if (category ==
L_NONE || category==curMicro)
461 out <<
"<w p=\"" << position <<
"\""
464 <<
" lem=\"" <<
xmlString(sp[norm].toStdString()) <<
"\"";
478 out <<
"/>" << std::endl;
486 if (category !=
L_NONE && !output && data.front()->
size()>0 )
488 auto norm = data.front()->begin()->normalizedForm;
489 out <<
"<w p=\"" << position <<
"\""
492 <<
" lem=\"" <<
xmlString(sp[norm].toStdString()) <<
"\""
493 <<
"/>" << std::endl;
503 const uint64_t offset)
const
508 LERROR <<
"missing specific entity annotation";
512 std::string typeName;
517 typeName = str.toStdString();
519 catch (std::exception& ) {
525 <<
"<e type=\"" << typeName <<
"\""
533 LDEBUG <<
"Using category "
535 <<
" for specific entity of type " << typeName;
542 if (token !=
nullptr)
546 std::vector<MorphoSyntacticData*>(1, vertexData),
550 out <<
"</e>" << std::endl;
558 std::string str(inputStr);
568 const std::string& toReplace,
569 const std::string& newValue)
const
571 auto oldLen = toReplace.size();
572 auto newLen = newValue.size();
573 auto i = str.find(toReplace);
574 while (i != std::string::npos)
576 str.replace(i, oldLen, newValue);
578 i = str.find(toReplace, i);
A graph structure for linguistic analysis.
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
Defines a Factory to create Object of type Base.
#define SIMPLEXMLDUMPER_CLASSID
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
Holds an annotation graph and gives an API to manipulate it.
std::set< AnnotationGraphVertex > matches(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second) const
Gets the set of vertices matched in the second graph by the given vertex of the first graph.
const GenericAnnotation & annotation(AnnotationGraphVertex v1, AnnotationGraphVertex v2, const LimaString &annot) const
LinguisticCode readValue(const LinguisticCode &code) const
read a property in a coded int.
const std::string & getPropertySymbolicValue(const LinguisticCode &value) const
The coded property value can hold several property data.
std::string & getParamsValueAtKey(const std::string &key)
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
static std::size_t size() noexcept
std::shared_ptr< DumperStream > initialize(AnalysisContent &analysis) const
virtual void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
void xmlOutputVertices(std::ostream &out, LinguisticAnalysisStructure::AnalysisGraph *anagraph, LinguisticAnalysisStructure::AnalysisGraph *posgraph, const Common::AnnotationGraphs::AnnotationData *annotationData, const LinguisticGraphVertex begin, const LinguisticGraphVertex end, const FsaStringsPool &sp, const uint64_t offset) const
std::string m_encapsulatingTag
void xmlOutputVertexInfos(std::ostream &out, const LinguisticAnalysisStructure::Token *ft, const std::vector< LinguisticAnalysisStructure::MorphoSyntacticData * > &data, const FsaStringsPool &sp, uint64_t offset, LinguisticCode category=L_NONE) const
bool outputSpecificEntity(std::ostream &out, const SpecificEntities::SpecificEntityAnnotation *se, LinguisticAnalysisStructure::MorphoSyntacticData *data, const LinguisticGraph *graph, const FsaStringsPool &sp, const uint64_t offset) const
const Common::PropertyCode::PropertyManager * m_tenseManager
virtual ~SimpleXmlDumper()
virtual void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
const Common::PropertyCode::PropertyAccessor * m_tenseAccessor
void replace(std::string &str, const std::string &toReplace, const std::string &newValue) const
std::string xmlString(const std::string &str) const
virtual LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
void xmlOutputVertex(std::ostream &out, LinguisticGraphVertex v, const LinguisticAnalysisStructure::Token *ft, LinguisticAnalysisStructure::AnalysisGraph *anagraph, LinguisticAnalysisStructure::AnalysisGraph *posgraph, const Common::AnnotationGraphs::AnnotationData *annotationData, const FsaStringsPool &sp, uint64_t offset) const
const Common::PropertyCode::PropertyManager * m_propertyManager
const Common::PropertyCode::PropertyAccessor * m_propertyAccessor
void xmlOutput(std::ostream &out, AnalysisContent &analysis, LinguisticAnalysisStructure::AnalysisGraph *anagraph, LinguisticAnalysisStructure::AnalysisGraph *posgraph, const Common::AnnotationGraphs::AnnotationData *annotationData) const
An AnalysisData containing a LinguisticGraph with a language and an id.
const LinguisticGraphVertex & lastVertex(void) const
Returns the last vertex of the graph.
const LinguisticGraph * getGraph(void) const
Returns the underlying graph structure.
const LinguisticGraphVertex & firstVertex(void) const
Returns the first vertex of the graph.
Holds morphosyntactic informations.
const Lima::LimaString & defaultKey() const
holds surface data of a token
uint64_t position() const
const TStatus & status() const
const LimaString & stringForm() const
A representation of a specific entity to store in the annotation graph.
StringsPoolIndex getString() const
Common::MediaticData::EntityType getType() const
const std::vector< LinguisticGraphVertex > & vertices() const
StringsPoolIndex getNormalizedForm() const
static const MediaticData & single()
const singleton accessor
static void logElapsedTime(const std::string &mess, const std::string &taskCategory=std::string(""))
log the number of microseconds since last UpdateCurrentTime
static void updateCurrentTime(const std::string &taskCategory=std::string(""))
store current time for new elapsed time computation
bool hasAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
SimpleFactory< MediaProcessUnit, SimpleXmlDumper > simpleXmlDumperFactory(SIMPLEXMLDUMPER_CLASSID)
PUGI__FN void sort(I begin, I end, const Pred &pred)
launch exception related to the configuration file parsing