LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
ConllDumper.cpp
Go to the documentation of this file.
1// Copyright 2002-2020 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6#include "ConllDumper.h"
13#include "common/misc/escaper.h"
30
31#include <QQueue>
32#include <QSet>
33#include <QStringList>
34
35#include <fstream>
37
38using namespace Lima::Common;
39using namespace Lima::Common::MediaticData;
40using namespace Lima::Common::Misc;
41using namespace Lima::Common::PropertyCode;
43using namespace Lima::Common::AnnotationGraphs;
48
51
52namespace Lima
53{
54
55namespace LinguisticProcessing
56{
57
58namespace AnalysisDumpers
59{
60
61QStringList FORMATS = {"CoNLL-U", "CoNLL-03"};
62
64
66{
67 friend class ConllDumper;
69
70 ~ConllDumperPrivate() = default;
71
72 LimaStatusCode dumpPosGraphVertex(
73 std::shared_ptr<DumperStream>& dstream,
75 int& tokenId,
76 LinguisticGraphVertex vEndDone,
77 std::map<LinguisticGraphVertex,int>& segmentationMapping,
78 const QString& neType,
79 bool first);
80
84 LimaStatusCode dumpAnalysisGraphVertex(
85 std::shared_ptr<DumperStream>& dstream,
87 LinguisticGraphVertex posGraphVertex,
88 int& tokenId,
89 LinguisticGraphVertex vEndDone,
90 const QString& parentNeType,
91 bool first,
92 const Automaton::EntityFeatures& features);
93
97 QString getNeType(LinguisticGraphVertex posGraphVertex);
98
99 std::pair<QString, QString> getConllRelName(
101 std::map<LinguisticGraphVertex,int>& segmentationMapping);
102
103 QStringList getPredicate(LinguisticGraphVertex v);
104
105 bool hasSpaceAfter(LinguisticGraphVertex v, LinguisticGraph* graph);
106
107 QString getMicro(MorphoSyntacticData* morphoData);
108
109 QString getFeats(MorphoSyntacticData* morphoData);
110
111 void getTransducedTokens(const LinguisticGraph& graph,
114 std::map<int, size_t>& out);
115
119 void dumpNamedEntity(std::shared_ptr<DumperStream>& dstream,
121 int& tokenId,
122 LinguisticGraphVertex vEndDone,
123 std::map<LinguisticGraphVertex,int>& segmentationMapping,
124 const QString& neType);
125
130 void collectPredicateTokens(
131 Lima::AnalysisContent& analysis,
132 LinguisticGraphVertex sentenceBegin,
133 LinguisticGraphVertex sentenceEnd);
134
135 void dumpToken(
136 std::shared_ptr<DumperStream>& dstream,
137 int tokenId, // ID
138 const QString& inflectedToken, // FORM
139 const QString& lemmatizedToken, // LEMMA
140 const QString& micro, // UPOS
141 const QString& xpos, // XPOS
142 const QString& features,// FEATS
143 const QString& targetConllIdString, // HEAD
144 const QString& conllRelName, // DEPREL
145 const QString& deps, // DEPS @TODO
146 const QStringList& miscField,
147 const QString& neType,
148 const QString& previousNeType);
149
150 const SpecificEntityAnnotation* getSpecificEntityAnnotation(
151 LinguisticGraphVertex v) const;
152
153 QString m_format = "CoNLL-U";
154
155 MediaId m_language;
156 bool m_withColsHeader;
157 QString m_graph;
158 QMap<QString, QString> m_conllLimaDepMapping;
159 const Common::PropertyCode::PropertyAccessor* m_propertyAccessor;
160 LinguisticGraph* posGraph;
161 LinguisticGraph* anaGraph;
162 DependencyGraph* depGraph;
163 AnnotationData* annotationData;
164 const LanguageData* languageData;
165 const PropertyCodeManager* propertyCodeManager;
166 const PropertyManager* microManager;
167 const std::map< std::string, PropertyManager >* managers;
168 const FsaStringsPool* sp;
169 QMultiMap<LinguisticGraphVertex, AnnotationGraphVertex> predicates;
170 QString previousNeType;
171 std::map< LinguisticGraphVertex,
172 std::pair<LinguisticGraphVertex,
173 std::string> > vertexDependencyInformations;
174};
175
176ConllDumperPrivate::ConllDumperPrivate():
177 m_language(0),
178 m_withColsHeader(false),
179 m_graph("PosGraph"),
180 m_conllLimaDepMapping(),
181 posGraph(nullptr),
182 anaGraph(nullptr),
183 depGraph(nullptr),
184 annotationData(nullptr),
185 previousNeType("_")
186{
187}
188
194
196{
197 delete m_d;
198}
199
201 Manager* manager)
202{
204 AbstractTextualAnalysisDumper::init(unitConfiguration, manager);
205
206 try
207 {
208 m_d->m_graph = QString::fromUtf8(unitConfiguration.getParamsValueAtKey("graph").c_str());
209 }
210 catch (NoSuchParam& ) {} // keep default value
211
212 m_d->m_language = manager->getInitializationParameters().media;
213 m_d->languageData = &static_cast<const LanguageData&>(MedData::single().mediaData(m_d->m_language));
214 m_d->propertyCodeManager = &m_d->languageData->getPropertyCodeManager();
215 m_d->microManager = &m_d->propertyCodeManager->getPropertyManager("MICRO");
216 m_d->managers = &m_d->propertyCodeManager->getPropertyManagers();
217 m_d->m_propertyAccessor = &m_d->propertyCodeManager->getPropertyAccessor("MICRO");
218 m_d->sp = &MedData::single().stringsPool(m_d->m_language);
219
220 try
221 {
222 m_d->m_withColsHeader= QString(
223 unitConfiguration.getParamsValueAtKey(
224 "withColsHeader").c_str() ).toLower() == "true";
225 }
226 catch (NoSuchParam& ) {} // keep default value
227
228 try
229 {
230 auto resourcePath = MedData::single().getResourcesPath();
231 auto mappingFile = findFileInPaths(resourcePath.c_str(),
232 unitConfiguration.getParamsValueAtKey(
233 "mappingFile").c_str());
234 std::ifstream ifs(mappingFile.toStdString(), std::ifstream::binary);
235 if (!ifs.good())
236 {
237 LERROR << "ERROR: cannot open" << mappingFile;
238 throw InvalidConfiguration();
239 }
240 while (ifs.good() && !ifs.eof())
241 {
242 auto line = readLine(ifs);
243 QStringList strs = QString::fromUtf8(line.c_str()).split('\t');
244 if (strs.size() == 2)
245 {
246 m_d->m_conllLimaDepMapping.insert(strs[0],strs[1]);
247 }
248 }
249
250 }
252 {
253 LINFO << "no parameter 'mappingFile' in ConllDumper group" << " !";
254// throw InvalidConfiguration();
255 }
256
257 try
258 {
259 m_d->m_format = QString::fromStdString(
260 unitConfiguration.getParamsValueAtKey("format") );
261 if (!FORMATS.contains(m_d->m_format))
262 {
263 QString errorMessage;
264 QTextStream qts(&errorMessage);
265 qts << "Invalid CoNLL dumper configuration. Known formats are"
266 << FORMATS.join(",") << ". Got" << m_d->m_format;
268 LERROR << errorMessage;
269 throw InvalidConfiguration(errorMessage.toStdString());
270 }
271 }
272 catch (NoSuchParam& ) {} // keep default value
273}
274
276{
277#ifdef DEBUG_LP
279 LDEBUG << "ConllDumper::process";
280#endif
281
282 auto metadata = std::dynamic_pointer_cast<LinguisticMetaData>(analysis.getData("LinguisticMetaData"));
283 if (metadata == 0)
284 {
286 LERROR << "ConllDumper::process no LinguisticMetaData ! abort";
287 return MISSING_DATA;
288 }
289
290 auto annotationData = std::dynamic_pointer_cast<AnnotationData>(analysis.getData("AnnotationData"));
291 if (annotationData == nullptr)
292 {
294 LINFO << "ConllDumper::process no AnnotationData ! Will not contain NE nor predicates";
295 }
296 m_d->annotationData = annotationData.get();
297 auto posGraphData=std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData(m_d->m_graph.toStdString()));
298 // posGraphData est de type PosGraph et non pas AnalysisGraph
299 if (posGraphData==0)
300 {
302 LERROR << "ConllDumper::process graph" << m_d->m_graph << "has not been produced: check pipeline";
303 return MISSING_DATA;
304 }
305 auto posGraph = posGraphData->getGraph();
306 m_d->posGraph = posGraph;
307
308 auto anaGraphData=std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData("AnalysisGraph"));
309 if (anaGraphData==0)
310 {
312 LERROR << "ConllDumper::process graph AnalysisGraph has not been produced: check pipeline";
313 return MISSING_DATA;
314 }
315 auto anaGraph = anaGraphData->getGraph();
316 m_d->anaGraph = anaGraph;
317
318 auto sd = std::dynamic_pointer_cast<SegmentationData>(analysis.getData("SentenceBoundaries"));
319 if (sd == nullptr)
320 {
322 LERROR << "ConllDumper::process no SentenceBoundaries! abort";
323 return MISSING_DATA;
324 }
325
326 auto syntacticData = std::dynamic_pointer_cast<SyntacticData>(analysis.getData("SyntacticData"));
327 if (syntacticData == nullptr)
328 {
329 syntacticData = std::make_shared<SyntacticData>(posGraphData.get(), nullptr);
330 syntacticData->setupDependencyGraph();
331 analysis.setData("SyntacticData", syntacticData);
332 }
333 auto depGraph = syntacticData-> dependencyGraph();
334 m_d->depGraph = depGraph;
335
336 auto dstream = initialize(analysis);
337
338 uint64_t nbSentences((sd->getSegments()).size());
339 if (nbSentences == 0)
340 {
342 LERROR << "ConllDumper::process 0 sentence to process";
343 return SUCCESS_ID;
344 }
345
346 auto sbItr = sd->getSegments().begin();
347#ifdef DEBUG_LP
348 LDEBUG << "ConllDumper::process There are "<< nbSentences << " sentences";
349#endif
350 auto sentenceBegin = sbItr->getFirstVertex();
351 auto sentenceEnd = sbItr->getLastVertex();
352
353 auto limaConllTokenIdMapping = std::dynamic_pointer_cast<LimaConllTokenIdMapping>(
354 analysis.getData("LimaConllTokenIdMapping"));
355 if (limaConllTokenIdMapping == nullptr)
356 {
357 limaConllTokenIdMapping = std::make_shared<LimaConllTokenIdMapping>();
358 analysis.setData("LimaConllTokenIdMapping", limaConllTokenIdMapping);
359 }
360 int sentenceNb = 0;
361 LinguisticGraphVertex vEndDone = 0;
362
363 const auto originalText = std::dynamic_pointer_cast<LimaStringText>(analysis.getData("Text"));
364
365 if (m_d->m_format == "CoNLL-U")
366 {
367 }
368 else if (m_d->m_format == "CoNLL-03")
369 {
370 dstream->out() << "-DOCSTART- -X- O O" << std::endl << std::endl;
371 }
372 else
373 {
375 QString errorMessage;
376 QTextStream qts(&errorMessage);
377 qts << "ConllDumper::process unknown format" << m_d->m_format;
378 LERROR << errorMessage;
379 return UNKNOWN_FORMAT;
380 }
381 while (sbItr != sd->getSegments().end()) //for each sentence
382 {
383 sentenceNb++;
384 // The cols list below is optionnal
385 if (m_d->m_format == "CoNLL-U")
386 {
387 if( sentenceNb==1 && m_d->m_withColsHeader )
388 {
389 dstream->out()
390 << "# global.columns = ID\tFORM\tLEMMA\tUPOS\tXPOS\tFEATS\tHEAD\tDEPREL\tDEPS\tMISC"
391 << std::endl;
392 }
393 dstream->out() << "# sent_id = " << sentenceNb << std::endl;
394 }
395 else if (m_d->m_format == "CoNLL-03")
396 {
397 }
398 else
399 {
401 QString errorMessage;
402 QTextStream qts(&errorMessage);
403 qts << "ConllDumper::process unknown format" << m_d->m_format;
404 LERROR << errorMessage;
405 return UNKNOWN_FORMAT;
406 }
407 sentenceBegin=sbItr->getFirstVertex();
408 sentenceEnd=sbItr->getLastVertex();
409 std::map<LinguisticGraphVertex,int> segmentationMapping;//mapping the two types of segmentations (Lima and conll)
410 std::map<int,LinguisticGraphVertex> segmentationMappingReverse;
411 std::map<int,size_t> transducedTokens;
412 m_d->getTransducedTokens(*anaGraph, sentenceBegin, sentenceEnd, transducedTokens);
413
414#ifdef DEBUG_LP
415 LDEBUG << "ConllDumper::process begin - end: " << sentenceBegin
416 << " - " << sentenceEnd;
417#endif
418
419 LinguisticGraphOutEdgeIt outItr,outItrEnd;
420 QQueue<LinguisticGraphVertex> toVisit;
421 QSet<LinguisticGraphVertex> visited;
422 toVisit.enqueue(sentenceBegin);
423 int vertexId = 0;
425 while (v != sentenceEnd && !toVisit.empty())
426 {
427 v = toVisit.dequeue();
428#ifdef DEBUG_LP
429 LDEBUG << "ConllDumper::process Vertex index : " << v;
430#endif
431 visited.insert(v);
432 segmentationMapping.insert(std::make_pair(v,vertexId));
433 segmentationMappingReverse.insert(std::make_pair(vertexId,v));
434#ifdef DEBUG_LP
435 LDEBUG << "ConllDumper::process conll id : " << vertexId
436 << " Lima id : " << v;
437#endif
438
439 auto dcurrent = syntacticData->depVertexForTokenVertex(v);
440 DependencyGraphOutEdgeIt dit, dit_end;
441 boost::tie(dit,dit_end) = boost::out_edges(dcurrent,*depGraph);
442 for (; dit != dit_end; dit++)
443 {
444#ifdef DEBUG_LP
445 LDEBUG << "ConllDumper::process Dumping dependency edge "
446 << (*dit).m_source << " -> " << (*dit).m_target;
447#endif
448 try
449 {
450 auto typeMap = get(edge_deprel_type, *depGraph);
451 auto type = typeMap[*dit];
452 auto syntRelName = m_d->languageData->getSyntacticRelationName(type);
453#ifdef DEBUG_LP
454 LDEBUG << "ConllDumper::process relation = " << syntRelName;
455 LDEBUG << "ConllDumper::process Src : Dep vertex= "
456 << boost::source(*dit, *depGraph);
457 auto src = syntacticData->tokenVertexForDepVertex(
458 boost::source(*dit, *depGraph));
459 LDEBUG << "ConllDumper::process Src : Morph vertex= " << src;
460 LDEBUG << "ConllDumper::process Targ : Dep vertex= "
461 << boost::target(*dit, *depGraph);
462#endif
463 auto dest = syntacticData->tokenVertexForDepVertex(
464 boost::target(*dit, *depGraph));
465#ifdef DEBUG_LP
466 LDEBUG << "ConllDumper::process Targ : Morph vertex= " << dest;
467#endif
468 if (syntRelName!="")
469 {
470#ifdef DEBUG_LP
471 LDEBUG << "ConllDumper::process saving target for"
472 << v << ":" << dest << syntRelName;
473#endif
474 m_d->vertexDependencyInformations.insert(
475 std::make_pair(v, std::make_pair(dest, syntRelName)));
476 }
477 }
478 catch (const std::range_error& )
479 {
480 }
481 catch (...)
482 {
483#ifdef DEBUG_LP
484 LDEBUG << "ConllDumper::process: catch others.....";
485#endif
486 throw;
487 }
488 }
489 if (v == sentenceEnd)
490 {
491 continue;
492 }
493 LinguisticGraphOutEdgeIt outItr,outItrEnd;
494 for (boost::tie(outItr,outItrEnd)=boost::out_edges(v, *posGraph);
495 outItr!=outItrEnd; outItr++)
496 {
497 LinguisticGraphVertex next=boost::target(*outItr,*posGraph);
498 if (!visited.contains(next) && next != posGraphData->lastVertex())
499 {
500 toVisit.enqueue(next);
501 }
502 }
503 ++vertexId;
504 }
505
506 // instead of looking to all vertices, follow the graph (in
507 // morphological graph, some vertices are not related to main graph:
508 // idiomatic expressions parts and named entity parts)
509
510 toVisit.clear();
511 visited.clear();
512
513 sentenceBegin=sbItr->getFirstVertex();
514 sentenceEnd=sbItr->getLastVertex();
515
516 // get the list of predicates for the current sentence
517 m_d->collectPredicateTokens( analysis,
518 sentenceBegin,
519 sentenceEnd );
520#ifdef DEBUG_LP
521 LDEBUG << "ConllDumper::process predicates for sentence between"
522 << sentenceBegin << "and" << sentenceEnd << "are:" << m_d->predicates;
523#endif
524 auto keys = m_d->predicates.keys();
525
526 v = sentenceBegin;
527 bool firstTime = true;
528 uint64_t pStart = 0;
529 uint64_t pEnd = 0;
530 while (v != sentenceEnd)
531 {
532 auto vIn = v;
533 //as long as there are vertices in the sentence
534 auto ft = get(vertex_token,*posGraph,v);
535 if( ft != nullptr && v != sentenceBegin )
536 {
537 if(firstTime)
538 {
539 pStart = ft->position();
540 firstTime = false;
541 }
542 pEnd = ft->position() + ft->length();
543 }
544 LinguisticGraphOutEdgeIt outIter,outIterEnd;
545 for (boost::tie(outIter,outIterEnd) = boost::out_edges(v,*posGraph);
546 outIter!=outIterEnd; outIter++)
547 {
548 v = boost::target(*outIter,*posGraph);
549 }
550 if (v == vIn)
551 {
553 LERROR << "ConllDumper::process sentence traversal loop stalls on" << v << "; breaking out.";
554 break;
555 }
556 }
557 auto curSentenceText = originalText->mid(pStart-1, pEnd-pStart+1);
558
559 if (m_d->m_format == "CoNLL-U")
560 {
561 // The text below is mandatory for CONLL-U format
562 dstream->out() << "# text = "
563 << curSentenceText.replace("\r\n"," ").replace("\n"," ").toStdString()
564 << std::endl;
565 }
566 else if (m_d->m_format == "CoNLL-03") {}
567 else
568 {
570 QString errorMessage;
571 QTextStream qts(&errorMessage);
572 qts << "ConllDumper::process unknown format" << m_d->m_format;
573 LERROR << errorMessage;
574 return UNKNOWN_FORMAT;
575 }
576
577 toVisit.enqueue(sentenceBegin);
578 int tokenId = 1;
579 v = 0;
580 while (!toVisit.empty() && v!=sentenceEnd)
581 { //as long as there are vertices in the sentence
582 v = toVisit.dequeue();
583
584 if (transducedTokens.find(tokenId) != transducedTokens.end())
585 {
586 int firstTokenId = tokenId;
587 int lastTokenId = tokenId + transducedTokens[tokenId] - 1;
588 LimaString tokenForm;
589 auto ft = get(vertex_token,*anaGraph,v);
590 if (ft == 0) {
592 LWARN << "Empty token for vertex" << vertex_token << "in graph" << anaGraphData->getGraphId();
593 }
594 else if (ft->orthographicAlternatives().size() > 0)
595 {
596 StringsPoolIndex idx = *(ft->orthographicAlternatives().begin());
597 tokenForm = (*m_d->sp)[idx];
598
599 if (m_d->m_format == "CoNLL-U")
600 dstream->out() << firstTokenId << "-" << lastTokenId
601 << "\t" << tokenForm.toStdString() // FORM
602 << "\t" << "_" // LEMMA
603 << "\t" << "_" // UPOS
604 << "\t" << "_" // XPOS
605 << "\t" << "_" // FEATS
606 << "\t" << "_" // HEAD
607 << "\t" << "_" // DEPREL
608 << "\t" << "_" // DEPS
609 << "\t" << "_" // MISC
610 << std::endl;
611 }
612 }
613
614 m_d->dumpPosGraphVertex(dstream,
615 v,
616 tokenId,
617 vEndDone,
618 segmentationMapping,
619 "",
620 false);
621
622
623#ifdef DEBUG_LP
624 LDEBUG << "ConllDumper::process look at out edges of" << v;
625#endif
626 LinguisticGraphOutEdgeIt outIter,outIterEnd;
627 for (boost::tie(outIter,outIterEnd) = boost::out_edges(v,*posGraph);
628 outIter != outIterEnd; outIter++)
629 {
630 LinguisticGraphVertex next = boost::target(*outIter,*posGraph);
631#ifdef DEBUG_LP
632 LDEBUG << "ConllDumper::process looking out vertex" << next;
633#endif
634 if (!visited.contains(next))
635 {
636#ifdef DEBUG_LP
637 LDEBUG << "ConllDumper::process enqueuing" << next;
638#endif
639 visited.insert(next);
640 toVisit.enqueue(next);
641 }
642 }
643 if (v == sentenceEnd)
644 {
645 vEndDone = v;
646 continue;
647 }
648 }
649 limaConllTokenIdMapping->insert(std::make_pair(sentenceNb,
650 segmentationMappingReverse));
651
652 sbItr++;
653 if (sbItr != sd->getSegments().end())
654 {
655 dstream->out() << std::endl;
656 }
657 }
658
659 return SUCCESS_ID;
660}
661
662const SpecificEntityAnnotation* ConllDumperPrivate::getSpecificEntityAnnotation(
663 LinguisticGraphVertex v) const
664{
665 // check only entity found in current graph (not previous graph such as AnalysisGraph)
666
667 for (const auto& vx: annotationData->matches("PosGraph", v, "annot"))
668 {
669 if (annotationData->hasAnnotation(vx, QString::fromUtf8("SpecificEntity")))
670 {
671 //BoWToken* se = createSpecificEntity(v,*it, annotationData, anagraph, posgraph, offsetBegin);
672 auto se = annotationData->annotation(
673 vx,
674 QString::fromUtf8("SpecificEntity")).pointerValue<SpecificEntityAnnotation>();
675 if (se != nullptr)
676 {
677 return se;
678 }
679 }
680 }
681 return nullptr;
682
683}
684
685LimaStatusCode ConllDumperPrivate::dumpPosGraphVertex(
686 std::shared_ptr<DumperStream>& dstream,
688 int& tokenId,
689 LinguisticGraphVertex vEndDone,
690 std::map<LinguisticGraphVertex,int>& segmentationMapping,
691 const QString& parentNeType,
692 bool first)
693{
694#ifdef DEBUG_LP
696 LDEBUG << "ConllDumperPrivate::dumpPosGraphVertex IN" << v;
697#endif
698 if (anaGraph == nullptr || posGraph == nullptr || depGraph == nullptr
699 || annotationData == nullptr)
700 {
702 LERROR << "ConllDumperPrivate::dumpPosGraphVertex missing data";
703 return MISSING_DATA;
704 }
705 bool notDone(true);
706 if( v == vEndDone )
707 notDone = false;
708
709 auto ft = get(vertex_token, *posGraph, v);
710 auto morphoData = get(vertex_data, *posGraph, v);
711 if( morphoData != 0 && ft != 0
712 && ((!morphoData->empty()) || ft->length() > 0) && notDone )
713 {
714#ifdef DEBUG_LP
715 LDEBUG << "ConllDumperPrivate::dumpPosGraphVertex PosGraph nb different LinguisticCode"
716 << morphoData->size();
717#endif
718
719 auto micro = getMicro(morphoData);
720#ifdef DEBUG_LP
721 LDEBUG << "ConllDumperPrivate::dumpPosGraphVertex graphTag:" << micro;
722#endif
723
724 auto feats = getFeats(morphoData);
725#ifdef DEBUG_LP
726 LDEBUG << "ConllDumperPrivate::dumpPosGraphVertex feats:" << feats;
727#endif
728
729 auto inflectedToken = ft->stringForm().toStdString();
730 if (inflectedToken.find_first_of("\r\n\t") != std::string::npos)
731 boost::find_format_all(inflectedToken,
732 boost::token_finder(!boost::is_print()),
734
735 QString lemmatizedToken;
736 if (morphoData != 0 && !morphoData->empty())
737 {
738 lemmatizedToken = (*sp)[(*morphoData)[0].lemma];
739 }
740 // @TODO Should follow instructions here to output all MWE:
741 // https://universaldependencies.org/format.html#words-tokens-and-empty-nodes
742 QString neType = getNeType(v);
743
744 // Collect NE vertices and output them instead of a single line for
745 // current v. NE vertices can not only be PosGraph
746 // vertices (and thus can just call dumpPosGraphVertex
747 // recursively) but also AnalysisGraph vertices. In the latter case, data
748 // come partly from the AnalysisGraph and partly from the PosGraph
749 // Furthermore, named entities can be recursive...
750 if (neType != "_")
751 {
752 dumpNamedEntity(dstream, v, tokenId, vEndDone, segmentationMapping, neType);
753 }
754 else
755 {
756 if (!parentNeType.isEmpty())
757 {
758 neType = parentNeType;
759 }
760
761 QString conllRelName;
762 QString targetConllIdString;
763 std::tie(conllRelName,
764 targetConllIdString) = getConllRelName(v, segmentationMapping);
765
766 QStringList miscField;
767 if (neType != "_")
768 {
769#ifdef DEBUG_LP
770 LDEBUG << "ConllDumperPrivate::dumpPosGraphVertex specific entity type is"
771 << neType;
772#endif
773 miscField << (QString("NE=") + (first?"B-":"I-") + neType);
774 const auto annot = getSpecificEntityAnnotation(v);
775#ifdef DEBUG_LP
776 LDEBUG << "ConllDumperPrivate::dumpPosGraphVertex specific entity annotation is"
777 << annot;
778#endif
779 if (annot != nullptr)
780 {
781 const auto& features = annot->getFeatures();
782 for (const auto& feature: features)
783 {
784 QString featureString;
785 QTextStream qts(&featureString);
786 if(feature.getPosition() == UNDEFPOSITION
787 && !feature.getName().empty()
788 && !feature.getValueString().empty())
789 {
790 qts << "NE-" << QString::fromStdString(feature.getName()) << "=";
792 QString::fromStdString(feature.getValueString()));
793 }
794 miscField << featureString;
795 }
796 }
797 }
798
799 miscField << (QString("Pos=") + QString::number(ft->position()) );
800 miscField << (QString("Len=") + QString::number(ft->length()) );
801
802 if(!hasSpaceAfter(v, posGraph))
803 {
804 miscField << QString("SpaceAfter=No");
805 }
806
807 miscField << getPredicate(v);
808
809 if (miscField.empty())
810 {
811 miscField << "_";
812 }
813
814 dumpToken(dstream,
815 tokenId++, // ID
816 QString::fromStdString(inflectedToken), // FORM
817 lemmatizedToken, // LEMMA
818 micro, // UPOS
819 "_", // XPOS
820 feats, // FEATS
821 targetConllIdString, // HEAD
822 conllRelName, // DEPREL
823 "_", // DEPS @TODO
824 miscField, // MISC
825 neType,
826 previousNeType
827 );
828 previousNeType = neType;
829 }
830 }
831 return SUCCESS_ID;
832}
833
834void ConllDumperPrivate::collectPredicateTokens(Lima::AnalysisContent& analysis,
835 LinguisticGraphVertex sentenceBegin,
836 LinguisticGraphVertex sentenceEnd)
837{
838#ifdef DEBUG_LP
840#endif
841 QMultiMap<LinguisticGraphVertex, AnnotationGraphVertex> result;
842
843 auto annotationData = std::dynamic_pointer_cast<AnnotationData>(analysis.getData("AnnotationData"));
844 if (annotationData == nullptr)
845 {
847 LERROR << "No annotation data exists: check pipeline";
848 predicates = result;
849 return;
850 }
851
852 auto tokenList = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData(m_graph.toStdString()));
853 if (tokenList == nullptr)
854 {
856 LERROR << "graph" << m_graph << "has not been produced: check pipeline";
857 predicates = result;
858 return;
859 }
860 auto graph = tokenList->getGraph();
861
862 QQueue<LinguisticGraphVertex> toVisit;
863 QSet<LinguisticGraphVertex> visited;
864 toVisit.enqueue(sentenceBegin);
866 while (v != sentenceEnd && !toVisit.empty())
867 {
868 v = toVisit.dequeue();
869#ifdef DEBUG_LP
870 LDEBUG << "ConllDumperPrivate::collectPredicateTokens vertex:" << v;
871#endif
872 visited.insert(v);
873
874 auto vMatches = annotationData->matches(m_graph.toStdString(), v, "annot");
875 for (const auto& vMatch : vMatches)
876 {
877 if (annotationData->hasStringAnnotation(vMatch, "Predicate"))
878 {
879#ifdef DEBUG_LP
880 LDEBUG << "ConllDumperPrivate::collectPredicateTokens insert"
881 << v << vMatch;
882#endif
883 result.insert(v, vMatch);
884 }
885 }
886 LinguisticGraphOutEdgeIt outItr,outItrEnd;
887 for (boost::tie(outItr,outItrEnd)=boost::out_edges(v,*graph);
888 outItr!=outItrEnd; outItr++)
889 {
890 auto next = boost::target(*outItr, *graph);
891 if (!visited.contains(next) && next != tokenList->lastVertex())
892 {
893 toVisit.enqueue(next);
894 }
895 }
896 }
897 predicates = result;
898}
899
900QString ConllDumperPrivate::getNeType(LinguisticGraphVertex posGraphVertex)
901{
902 auto neType = QString::fromUtf8("_") ;
903 if (annotationData != nullptr)
904 {
905 // Check if the PosGraph vertex holds a specific entity
906 auto matches = annotationData->matches(m_graph.toStdString(), posGraphVertex, "annot");
907 for (const auto& vx: matches)
908 {
909 if (annotationData->hasAnnotation(
910 vx, QString::fromUtf8("SpecificEntity")))
911 {
912 auto se = annotationData->annotation(vx, QString::fromUtf8("SpecificEntity")).
913 pointerValue<SpecificEntityAnnotation>();
914 neType = MedData::single().getEntityName(se->getType());
915 break;
916 }
917 }
918 if (neType == "_")
919 {
920 // The PosGraph vertex did not hold a specific entity,
921 // check if the AnalysisGraph vertex does
922 auto anaVertices = annotationData->matches(m_graph.toStdString(), posGraphVertex,
923 "AnalysisGraph");
924 // note: anaVertices size should be 0 or 1
925 for (const auto& anaVertex: anaVertices)
926 {
927 auto matches = annotationData->matches("AnalysisGraph",
928 anaVertex,
929 "annot");
930 for (const auto& vx: matches)
931 {
932 if (annotationData->hasAnnotation(
933 vx, QString::fromUtf8("SpecificEntity")))
934 {
935 auto se = annotationData->annotation(
936 vx, QString::fromUtf8("SpecificEntity"))
938 neType = MedData::single().getEntityName(se->getType());
939 break;
940 }
941 }
942 if (neType != "_") break;
943 }
944 }
945 }
946 return neType;
947}
948
949std::pair<QString, QString> ConllDumperPrivate::getConllRelName(
951 std::map<LinguisticGraphVertex,int>& segmentationMapping)
952{
953#ifdef DEBUG_LP
955 LDEBUG << "ConllDumperPrivate::getConllRelName" << v;
956#endif
957 QString conllRelName = "_";
958 int targetConllId = 0;
959 if (vertexDependencyInformations.count(v) != 0)
960 {
961 auto target = vertexDependencyInformations.find(v)->second.first;
962#ifdef DEBUG_LP
963 LDEBUG << "ConllDumper::process target saved for"
964 << v << "is" << target;
965#endif
966 if (segmentationMapping.find(target) != segmentationMapping.end())
967 {
968 targetConllId = segmentationMapping.find(target)->second;
969 }
970 else
971 {
973 LERROR << "ConllDumper::process target" << target
974 << "not found in segmentation mapping";
975 }
976#ifdef DEBUG_LP
977 LDEBUG << "ConllDumper::process conll target saved for "
978 << v << " is " << targetConllId;
979#endif
980 auto relName = QString::fromUtf8(
981 vertexDependencyInformations.find(v)->second.second.c_str());
982#ifdef DEBUG_LP
983 LDEBUG << "ConllDumper::process the lima dependency tag for "
984 << v << " is " << relName;
985#endif
986 if (m_conllLimaDepMapping.contains(relName))
987 {
988 conllRelName = m_conllLimaDepMapping[relName];
989 }
990 else
991 {
992 conllRelName = relName;
993// LERROR << "ConllDumper::process" << relName << "not found in mapping";
994 }
995
996 // There is no way for vertex to have 0 as head.
997 if (conllRelName == "root")
998 {
999 targetConllId = 0;
1000 }
1001 }
1002 QString targetConllIdString = QString(QLatin1String("%1")).arg(targetConllId);
1003
1004 return { conllRelName, targetConllIdString };
1005}
1006
1007QStringList ConllDumperPrivate::getPredicate(LinguisticGraphVertex v)
1008{
1009#ifdef DEBUG_LP
1011 LDEBUG << "ConllDumperPrivate::getPredicate" << v;
1012#endif
1013 QStringList miscField;
1014 if (annotationData != nullptr && predicates.contains(v))
1015 {
1016 auto keys = predicates.keys();
1017 auto predicate = annotationData->stringAnnotation(predicates.value(v),
1018 "Predicate");
1019
1020 // Now output the roles supported by the current PoS graph token
1021#ifdef DEBUG_LP
1022 LDEBUG << "ConllDumper::process output the roles for the"
1023 << keys.size() << "predicates";
1024#endif
1025 for (int i = 0; i < keys.size(); i++)
1026 {
1027 auto predicateVertex = predicates.value(keys[keys.size()-1-i]);
1028
1029 auto vMatches = annotationData->matches(m_graph.toStdString(), v, "annot");
1030 if (!vMatches.empty())
1031 {
1032#ifdef DEBUG_LP
1033 LDEBUG << "ConllDumper::process there is" << vMatches.size()
1034 << "nodes matching PoS graph vertex" << v
1035 << "in the annotation graph.";
1036#endif
1037 QString roleAnnotation;
1038 for (auto it = vMatches.begin(); it != vMatches.end(); it++)
1039 {
1040 auto vMatch = *it;
1041 AnnotationGraphInEdgeIt vMatchInEdgesIt, vMatchInEdgesIt_end;
1042 boost::tie(vMatchInEdgesIt, vMatchInEdgesIt_end) =
1043 boost::in_edges(vMatch,annotationData->getGraph());
1044 for (; vMatchInEdgesIt != vMatchInEdgesIt_end; vMatchInEdgesIt++)
1045 {
1046 auto inVertex = boost::source(*vMatchInEdgesIt,
1047 annotationData->getGraph());
1048 auto inVertexAnnotPosGraphMatches = annotationData->matches(
1049 "annot",inVertex,m_graph.toStdString());
1050 if (inVertex == predicateVertex
1051 && !inVertexAnnotPosGraphMatches.empty())
1052 {
1053 // Current edge is holding a role of the current predicate
1054 roleAnnotation =
1055 annotationData->stringAnnotation(*vMatchInEdgesIt,
1056 "SemanticRole");
1057 break;
1058 }
1059 }
1060 }
1061 if (!roleAnnotation.isEmpty() )
1062 predicate = roleAnnotation + ":" + predicate;
1063 }
1064 }
1065 if (!predicate.isEmpty())
1066 {
1067 miscField << predicate;
1068 }
1069 }
1070 return miscField;
1071}
1072
1073bool ConllDumperPrivate::hasSpaceAfter(LinguisticGraphVertex v,
1074 LinguisticGraph* graph)
1075{
1076 auto ft = get(vertex_token, *graph, v);
1077 bool SpaceAfter = true;
1078 LinguisticGraphOutEdgeIt outIter, outIterEnd;
1079 for (boost::tie(outIter,outIterEnd) = boost::out_edges(v, *graph);
1080 outIter!=outIterEnd; outIter++)
1081 {
1082 auto next = boost::target(*outIter, *graph);
1083 auto nt = get(vertex_token, *graph, next);
1084 if( nt != nullptr
1085 && (nt->position() == ft->position()+ft->length()) )
1086 {
1087 SpaceAfter = false;
1088 break;
1089 }
1090 }
1091 return SpaceAfter;
1092}
1093
1094QString ConllDumperPrivate::getMicro(MorphoSyntacticData* morphoData)
1095{
1096 return QString::fromUtf8(static_cast<const LangData&>(
1097 MedData::single().mediaData(m_language)).getPropertyCodeManager()
1098 .getPropertyManager("MICRO")
1099 .getPropertySymbolicValue(morphoData->firstValue(
1100 *m_propertyAccessor)).c_str());
1101}
1102
1103QString ConllDumperPrivate::getFeats(MorphoSyntacticData* morphoData)
1104{
1105#ifdef DEBUG_LP
1107#endif
1108
1109 QStringList featuresList;
1110 for (auto i = managers->cbegin(); i != managers->cend(); i++)
1111 {
1112 auto key = QString::fromUtf8(i->first.c_str());
1113 if (key != "MACRO" && key != "MICRO")
1114 {
1115 const auto& pa = propertyCodeManager->getPropertyAccessor(key.toStdString());
1116 LinguisticCode lc = morphoData->firstValue(pa);
1117 auto value = QString::fromUtf8(i->second.getPropertySymbolicValue(lc).c_str());
1118 if (value != "NONE")
1119 {
1120 featuresList << QString("%1=%2").arg(key).arg(value);
1121 }
1122 }
1123 }
1124
1125 featuresList.sort();
1126 QString features;
1127 QTextStream featuresStream(&features);
1128 if (featuresList.isEmpty())
1129 {
1130 features = "_";
1131 }
1132 else
1133 {
1134 for (auto featuresListIt = featuresList.cbegin(); featuresListIt != featuresList.cend(); featuresListIt++)
1135 {
1136 if (featuresListIt != featuresList.cbegin())
1137 {
1138 featuresStream << "|";
1139 }
1140 featuresStream << *featuresListIt;
1141 }
1142 }
1143#ifdef DEBUG_LP
1144 LDEBUG << "ConllDumper::process features:" << features;
1145#endif
1146
1147 return features;
1148}
1149
1150void ConllDumperPrivate::getTransducedTokens(const LinguisticGraph& graph,
1153 std::map<int, size_t>& out)
1154{
1155 QQueue<LinguisticGraphVertex> toVisit;
1156 QSet<LinguisticGraphVertex> visited;
1157
1158 toVisit.enqueue(begin);
1159 int tokenId = 0;
1161 int firstTransducedTokenId = -1;
1162 uint64_t prev_pos = 0;
1163
1164 while (!toVisit.empty() && v != end)
1165 {
1166 //as long as there are vertices in the sentence
1167 v = toVisit.dequeue();
1168
1169 const Token *ft = get(vertex_token, graph, v);
1170
1171 if (ft != nullptr && prev_pos > 0 && prev_pos == ft->position())
1172 {
1173 if (firstTransducedTokenId >= 0)
1174 {
1175 out[firstTransducedTokenId] += 1;
1176 }
1177 else
1178 {
1179 if (tokenId == 0)
1180 throw std::runtime_error("token id should not be null");
1181 out[tokenId - 1] = 2;
1182 firstTransducedTokenId = tokenId - 1;
1183 }
1184 }
1185 else
1186 {
1187 firstTransducedTokenId = -1;
1188 }
1189
1190 if (ft != nullptr)
1191 prev_pos = ft->position();
1192
1193 if (v == end)
1194 {
1195 continue;
1196 }
1197
1198 LinguisticGraphOutEdgeIt outIter,outIterEnd;
1199 for (boost::tie(outIter,outIterEnd) = boost::out_edges(v,graph);
1200 outIter != outIterEnd; outIter++)
1201 {
1202 auto next = boost::target(*outIter, graph);
1203 if (!visited.contains(next))
1204 {
1205 visited.insert(next);
1206 toVisit.enqueue(next);
1207 }
1208 }
1209 tokenId++;
1210 }
1211}
1212
1213QString matchesS(const std::set<AnnotationGraphVertex>& s)
1214{
1215 QString result;
1216 QTextStream qts(&result);
1217 for (auto i: s) {qts << i << ",";}
1218 return result;
1219}
1220
1221void ConllDumperPrivate::dumpNamedEntity(std::shared_ptr<DumperStream>& dstream,
1223 int& tokenId,
1224 LinguisticGraphVertex vEndDone,
1225 std::map<LinguisticGraphVertex,int>& segmentationMapping,
1226 const QString& neType)
1227{
1228#ifdef DEBUG_LP
1230 LDEBUG << "ConllDumperPrivate::dumpNamedEntity" << v << tokenId << vEndDone
1231 << neType;
1232#endif
1233 // Check if the named entity is on AnalysisGraph.
1234 // If so, then we have to recursively get all analysis graph tokens and
1235 // collect the information about them, chosing randomly the "right" category
1236 // Otherwise, will retrieve the pos graph tokens and recursively do the same.
1237 // For final tokens that are on pos graph, the category will be unique.
1238
1239 if (annotationData != nullptr)
1240 {
1241 // Check if the PosGraph vertex holds a specific entity
1242 auto matches = annotationData->matches(m_graph.toStdString(), v, "annot");
1243#ifdef DEBUG_LP
1244 LDEBUG << "ConllDumperPrivate::dumpNamedEntity matches PosGraph" << v
1245 << "annot:" << matchesS(matches);
1246#endif
1247 for (const auto& vx: matches)
1248 {
1249 if (annotationData->hasAnnotation(
1250 vx, QString::fromUtf8("SpecificEntity")))
1251 {
1252 auto se = annotationData->annotation(vx,
1253 QString::fromUtf8("SpecificEntity"))
1255 previousNeType = "O";
1256 bool first = true;
1257 for (const auto& vse : se->vertices())
1258 {
1259 dumpPosGraphVertex(dstream, vse, tokenId, vEndDone,
1260 segmentationMapping, neType, first);
1261 first = false;
1262 }
1263#ifdef DEBUG_LP
1264 LDEBUG << "ConllDumperPrivate::dumpNamedEntity return after SpecificEntity annotation on PosGraph";
1265#endif
1266 return;
1267 }
1268 }
1269 auto anaVertices = annotationData->matches(m_graph.toStdString(), v, "AnalysisGraph");
1270#ifdef DEBUG_LP
1271 LDEBUG << "ConllDumperPrivate::dumpNamedEntity anaVertices for" << v
1272 << ":" << matchesS(anaVertices);
1273#endif
1274
1275 assert(anaVertices.size() == 1);
1276 auto anaVertex = *anaVertices.begin();
1277#ifdef DEBUG_LP
1278 LDEBUG << "ConllDumperPrivate::dumpNamedEntity anaVertex is" << anaVertex;
1279#endif
1280 if (annotationData->hasAnnotation(
1281 anaVertex, QString::fromUtf8("SpecificEntity")))
1282 {
1283 auto se = annotationData->annotation(anaVertex,
1284 QString::fromUtf8("SpecificEntity"))
1286#ifdef DEBUG_LP
1287 LDEBUG << "ConllDumperPrivate::dumpNamedEntity anaVertex se ("
1288 << (*sp)[se->getString()]
1289 << ") annotation vertices are" << se->vertices()
1290 << "and normalized form:" << (*sp)[se->getNormalizedForm()]
1291 << "and features:" << se->getFeatures();
1292#endif
1293 // All retrieved lines/tokens have the same netype. Depending on the
1294 // output style (CoNLL 2003, CoNLL-U, …), the generated line is different
1295 // and the ne-Type includes or not BIO information using in this case the
1296 // previousNeType member.
1297 previousNeType = "O";
1298 bool first = true;
1299 for (const auto& vse : se->vertices())
1300 {
1301 dumpAnalysisGraphVertex(dstream, vse, v, tokenId, vEndDone, neType,
1302 first, se->getFeatures());
1303 first = false;
1304 }
1305 previousNeType = neType;
1306 }
1307 }
1308}
1309
1310// TODO Split idiomatic alternative tokens and compound tokens
1311LimaStatusCode ConllDumperPrivate::dumpAnalysisGraphVertex(
1312 std::shared_ptr<DumperStream>& dstream,
1314 LinguisticGraphVertex posGraphVertex,
1315 int& tokenId,
1316 LinguisticGraphVertex vEndDone,
1317 const QString& neType,
1318 bool first,
1319 const Automaton::EntityFeatures& features)
1320{
1321 LIMA_UNUSED(posGraphVertex);
1322#ifdef DEBUG_LP
1324 LDEBUG << "ConllDumperPrivate::dumpAnalysisGraphVertex" << v << posGraphVertex
1325 << neType;
1326#endif
1327 if (anaGraph == nullptr || posGraph == nullptr || depGraph == nullptr
1328 || annotationData == nullptr)
1329 {
1331 LERROR << "ConllDumperPrivate::dumpAnalysisGraphVertex missing data";
1332 return MISSING_DATA;
1333 }
1334 bool notDone(true);
1335 if( v == vEndDone )
1336 notDone = false;
1337
1338 auto ft = get(vertex_token, *anaGraph, v);
1339 auto morphoData = get(vertex_data, *anaGraph, v);
1340#ifdef DEBUG_LP
1341 LDEBUG << "ConllDumperPrivate::dumpAnalysisGraphVertex PosGraph token" << v;
1342#endif
1343 if( morphoData != 0 && ft != 0
1344 && ((!morphoData->empty()) || ft->length() > 0) && notDone )
1345 {
1346#ifdef DEBUG_LP
1347 LDEBUG << "ConllDumperPrivate::dumpAnalysisGraphVertex PosGraph nb different LinguisticCode"
1348 << morphoData->size();
1349#endif
1350
1351 auto micro = getMicro(morphoData);
1352#ifdef DEBUG_LP
1353 LDEBUG << "ConllDumperPrivate::dumpAnalysisGraphVertex micro:" << micro;
1354#endif
1355
1356 auto feats = getFeats(morphoData);
1357#ifdef DEBUG_LP
1358 LDEBUG << "ConllDumperPrivate::dumpAnalysisGraphVertex feats:" << feats;
1359#endif
1360
1361 auto inflectedToken = ft->stringForm().toStdString();
1362 if (inflectedToken.find_first_of("\r\n\t") != std::string::npos)
1363 boost::find_format_all(inflectedToken,
1364 boost::token_finder(!boost::is_print()),
1366
1367 QString lemmatizedToken;
1368 if (morphoData != 0 && !morphoData->empty())
1369 {
1370 lemmatizedToken = (*sp)[(*morphoData)[0].lemma];
1371 }
1372 // @TODO Should follow instructions here to output all MWE:
1373 // https://universaldependencies.org/format.html#words-tokens-and-empty-nodes
1374
1375 // TODO Get correct UD dep relation for relations inside the named entity
1376 // and for the token that must be linked to the outside. For this one, the
1377 // relation is the one which links to posGraphVertex to the rest of the pos
1378 // graph.
1379 QString conllRelName = "_";
1380 QString targetConllIdString = "_";
1381
1382 QStringList miscField;
1383 if (neType != "_")
1384 {
1385#ifdef DEBUG_LP
1386 LDEBUG << "ConllDumperPrivate::dumpAnalysisGraphVertex specific entity type is" << neType;
1387 LDEBUG << "ConllDumperPrivate::dumpAnalysisGraphVertex posGraphVertex is" << posGraphVertex;
1388#endif
1389 miscField << (QString("NE=") + (first?"B-":"I-") + neType);
1390 for (const auto& feature: features)
1391 {
1392 if(!feature.getName().empty()
1393 && !feature.getValueString().empty())
1394 {
1395 QString featureString;
1396 QTextStream qts(&featureString);
1397 qts << "NE-" << QString::fromStdString(feature.getName()) << "=";
1399 QString::fromStdString(feature.getValueString()));
1400 miscField << featureString;
1401 }
1402 }
1403 }
1404
1405 miscField << (QString("Pos=") + QString::number(ft->position()) );
1406 miscField << (QString("Len=") + QString::number(ft->length()) );
1407
1408 if(!hasSpaceAfter(v, anaGraph))
1409 {
1410 miscField << QString("SpaceAfter=No");
1411 }
1412
1413 miscField << getPredicate(v);
1414
1415 if (miscField.empty())
1416 {
1417 miscField << "_";
1418 }
1419 dumpToken(dstream,
1420 tokenId++, // ID
1421 QString::fromStdString(inflectedToken), // FORM
1422 lemmatizedToken, // LEMMA
1423 micro, // UPOS
1424 "_", // XPOS
1425 feats, // FEATS
1426 targetConllIdString, // HEAD
1427 conllRelName, // DEPREL
1428 "_", // DEPS @TODO
1429 miscField, // MISC
1430 neType,
1431 previousNeType
1432 );
1433 }
1434 return SUCCESS_ID;
1435}
1436
1437void ConllDumperPrivate::dumpToken(
1438 std::shared_ptr<DumperStream>& dstream,
1439 int tokenId, // ID
1440 const QString& inflectedToken, // FORM
1441 const QString& lemmatizedToken, // LEMMA
1442 const QString& micro, // UPOS
1443 const QString& xpos, // XPOS
1444 const QString& features,// FEATS @TODO
1445 const QString& targetConllIdString, // HEAD
1446 const QString& conllRelName, // DEPREL
1447 const QString& deps, // DEPS @TODO
1448 const QStringList& miscField,
1449 const QString& neType,
1450 const QString& previousNeType)
1451{
1452#ifdef DEBUG_LP
1454 LDEBUG << "ConllDumperPrivate::dumpToken" << tokenId;
1455#endif
1456 if (m_format == "CoNLL-U")
1457 {
1458 // CONLL-U format
1459 // https://universaldependencies.org/format.html
1460 //
1461 // ID: Word index, integer starting at 1 for each new sentence; may be a
1462 // range for multiword tokens; may be a decimal number for empty
1463 // nodes (decimal numbers can be lower than 1 but must be greater
1464 // than 0).
1465 // FORM: Word form or punctuation symbol.
1466 // LEMMA: Lemma or stem of word form.
1467 // UPOS: Universal part-of-speech tag.
1468 // XPOS: Language-specific part-of-speech tag; underscore if not
1469 // available.
1470 // FEATS: List of morphological features from the universal feature
1471 // inventory or from a defined language-specific extension;
1472 // underscore if not available (this is the case currently in
1473 // LIMA).
1474 // HEAD: Head of the current word, which is either a value of ID or
1475 // zero (0).
1476 // DEPREL: Universal dependency relation to the HEAD (root iff HEAD = 0)
1477 // or a defined language-specific subtype of one.
1478 // DEPS: Enhanced dependency graph in the form of a list of head-deprel
1479 // pairs. Currently unavailable in LIMA, thus underscore.
1480 // MISC: Any other annotation. In LIMA, named entities and SRL
1481 // information.
1482
1483 dstream->out() << tokenId++ // ID
1484 << "\t" << inflectedToken.toStdString() // FORM
1485 << "\t" << lemmatizedToken.toStdString() // LEMMA
1486 << "\t" << micro.toStdString() // UPOS
1487 << "\t" << xpos.toStdString() // XPOS
1488 << "\t" << features.toStdString() // FEATS @TODO
1489 << "\t" << targetConllIdString.toStdString() // HEAD
1490 << "\t" << conllRelName.toStdString() // DEPREL
1491 << "\t" << deps.toStdString() // DEPS @TODO
1492 << "\t" << miscField.join('|').toStdString(); // MISC
1493 dstream->out() << std::endl;
1494 }
1495 else if (m_format == "CoNLL-03")
1496 {
1497 // CONLL 2003 format
1498 //
1499 // -DOCSTART- -X- O O
1500 //
1501 // CRICKET NNP I-NP O
1502 // - : O O
1503 // LEICESTERSHIRE NNP I-NP I-ORG
1504 // TAKE NNP I-NP O
1505
1506
1507#ifdef DEBUG_LP
1508 LDEBUG << "ConllDumperPrivate::dumpToken" << tokenId
1509 << inflectedToken.toStdString() << neType << previousNeType;
1510#endif
1511 QString inflectedTokenEscaped = inflectedToken;
1512 inflectedTokenEscaped.replace(" ", "_");
1513 dstream->out() << inflectedTokenEscaped.toStdString() // FORM
1514 << " " << micro.toStdString() // UPOS
1515 << " " << "I-NP";
1516 if (neType.isEmpty() || neType == "_")
1517 {
1518 dstream->out() << " " << "O";
1519 }
1520 else
1521 {
1522 if (neType == previousNeType)
1523 {
1524 dstream->out() << " " << "I-";
1525 }
1526 else
1527 {
1528 dstream->out() << " " << "B-";
1529 }
1530 dstream->out() << neType.toStdString();
1531 }
1532 dstream->out() << std::endl;
1533 }
1534 else
1535 {
1537 QString errorMessage;
1538 QTextStream qts(&errorMessage);
1539 qts << "ConllDumper::dumpToken unknown format" << m_format;
1540 LERROR << errorMessage;
1541 throw std::runtime_error(errorMessage.toStdString());
1542 }
1543}
1544
1545} // end namespace
1546} // end namespace
1547} // end namespace
This file is the main header file for the data related to annotation graphs.
A graph that stores any data (annotations) referencing primarily nodes of a text anlaysis.
#define CONLLDUMPER_CLASSID
Definition ConllDumper.h:19
DependencyGraph::out_edge_iterator DependencyGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, DepVertexProperties, DepEdgeProperties > DependencyGraph
The dependency graph class.
@ edge_deprel_type
#define UNDEFPOSITION
#define LWARN
Definition LimaCommon.h:160
#define LIMA_UNUSED(x)
Definition LimaCommon.h:224
#define LDEBUG
Definition LimaCommon.h:157
#define LINFO
Definition LimaCommon.h:158
#define LERROR
Definition LimaCommon.h:161
A graph structure for linguistic analysis.
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
@ vertex_data
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
#define DUMPERLOGINIT
Defines a Factory to create Object of type Base.
Data used for the syntactic analyzis of texts.
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Holds an annotation graph and gives an API to manipulate it.
std::set< AnnotationGraphVertex > matches(const std::string &first, AnnotationGraphVertex firstVx, const std::string &second) const
Gets the set of vertices matched in the second graph by the given vertex of the first graph.
const GenericAnnotation & annotation(AnnotationGraphVertex v1, AnnotationGraphVertex v2, const LimaString &annot) const
const LimaString & stringAnnotation(AnnotationGraphVertex v1, AnnotationGraphVertex v2, const LimaString &annot) const
Holds linguistic data for one language.
const PropertyCode::PropertyCodeManager & getPropertyCodeManager() const
const std::string & getSyntacticRelationName(SyntacticRelationId id) const
holds data about codes and names for grammatical categories, etc.
const FsaStringsPool & stringsPool(MediaId med) const
const std::string & getResourcesPath() const
const MediaData & mediaData(MediaId media) const
LimaString getEntityName(const EntityType &type) const
Provide function to read write and check a property.
Provide tools to parse a property file, and deal with the property coding system.
const std::map< std::string, PropertyManager > & getPropertyManagers() const
Get the map of all PropertyManagers.
const PropertyManager & getPropertyManager(const std::string &propertyName) const
Get the PropertyManager associated to a property.
const PropertyAccessor & getPropertyAccessor(const std::string &propertyName) const
Get the PropertyAccessor associated to a property.
Provide tools to manage a specific property.
return a message when a 'param' was not found
Manage initialization of InitializableObjects using configuration module and parameters.
const InitializationParameters & getInitializationParameters() const
get Initialization Parameters
Use this exception to signal an error in one of the configuration files.
Definition LimaCommon.h:345
std::shared_ptr< DumperStream > initialize(AnalysisContent &analysis) const
virtual void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
This dumper outputs analysis result in various CoNLL formats.
Definition ConllDumper.h:30
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
a list of generic features: each feature is unique (only one feature for a name)
LinguisticCode firstValue(const Common::PropertyCode::PropertyAccessor &propertyAccessor) const
Return the first non empty value for the given accessor.
A representation of a specific entity to store in the annotation graph.
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
AnnotationGraph::in_edge_iterator AnnotationGraphInEdgeIt
bool hasStringAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
bool hasAnnotation(AnnotationGraphVertex v, const LimaString &annot) const
std::string readLine(std::istream &inputFile)
LimaString transcodeToXmlEntities(const LimaString &str)
QString findFileInPaths(const QString &paths, const QString &fileName, const QChar &separator)
Find the given file in the given paths.
SimpleFactory< MediaProcessUnit, ConllDumper > conllDumperFactory(CONLLDUMPER_CLASSID)
QString matchesS(const std::set< AnnotationGraphVertex > &s)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ UNKNOWN_FORMAT
Definition LimaCommon.h:244
@ SUCCESS_ID
Definition LimaCommon.h:237
@ MISSING_DATA
Definition LimaCommon.h:243
QString LimaString
Definition LimaString.h:33
launch exception related to the configuration file parsing