LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
EntityTracker.h
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6#ifndef ENTITYTRACKER_H
7#define ENTITYTRACKER_H
8
16
17#include "CoreferenceEngine.h"
18
19#include <vector>
20#include <string.h>
21
22namespace Lima
23{
24namespace LinguisticProcessing
25{
26namespace EntityTracking
27{
28
29#define ENTITYTRACKER_CLASSID "EntityTracker"
30
31class LIMA_ENTITYTRACKING_EXPORT EntityTracker : public MediaProcessUnit, std::vector< LinguisticGraphVertex>
32{
33public:
34
37 FsaStringsPool& sp);
38 virtual ~EntityTracker();
39
41 Manager* manager) override;
42
43 LimaStatusCode process(AnalysisContent& analysis) const override;
44
45
47 void dump(std::ostream& os);
48
49 /* test if the two string are equals */
50 inline bool isEqual(const std::string a,const std::string b) const {if (strcmp(a.c_str(),b.c_str())==0) return true; else return false;}
51 /* return true if the current word is included in the original word */
52 bool isInclude(const std::string original, const std::string currentWord) const;
53 /* return true if the current word is an acronym of the original word */
54 bool isAcronym(const std::string original, const std::string currentWord) const;
55
56 void addNewForm(const std::string original, const std::string currentWord);
57 bool exist(const std::string original, const std::vector< std::vector<std::string> > Acronyms) const;
58
59 void searchCoreference(const std::string text);
60
61 //bool checkCoreference(const Token& tok, CoreferenceEngine ref) const;
63/* void storeSpecificEntity (const Lima::LinguisticProcessing::
64 SpecificEntities::SpecificEntityAnnotation * se) const; */
65
66private:
67
68 std::vector< std::vector<std::string> > Acronyms; // the vector of acronyms
69 std::vector<LinguisticAnalysisStructure::Token> storedAnnotations;
70 std::vector<LinguisticAnalysisStructure::Token> allToken;
71 std::vector< std::vector<LinguisticAnalysisStructure::Token> > findedToken; /* contient une structure de toutes les coréferences */
72
73// LinguisticGraphVertex m_head;
74// Common::MediaticData::EntityType m_type; /**< the type of the entity */
75// Automaton::EntityFeatures m_features;
76// StringsPoolIndex m_string;
77// StringsPoolIndex m_normalizedString;
78// StringsPoolIndex m_normalizedForm;
79// uint64_t m_position;
80// uint64_t m_length;
81
82 // Linguistic properties and normalized form are given by the normalized
83 // form of the vertex in the morphological graph
84 //
85 // LinguisticCode m_linguisticProperties; /**< associated ling prop */
86 // StringsPoolIndex m_normalizedForm; /**< the normalized form of the
87 // recognized entity*/
88
89};
90
91} // SpecificEntities
92} // LinguisticProcessing
93} // Lima
94
95#endif // ENTITYTRACKER_H
This file is the main header file for the data related to annotation graphs.
#define LIMA_ENTITYTRACKING_EXPORT
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
Manage initialization of InitializableObjects using configuration module and parameters.
EntityTracker(const Automaton::RecognizerMatch &entity, FsaStringsPool &sp)
void addNewForm(const std::string original, const std::string currentWord)
bool isInclude(const std::string original, const std::string currentWord) const
bool isEqual(const std::string a, const std::string b) const
bool isAcronym(const std::string original, const std::string currentWord) const
void storeAllToken(const LinguisticAnalysisStructure::Token *tok)
bool exist(const std::string original, const std::vector< std::vector< std::string > > Acronyms) const
void dump(std::ostream &os)
The functions that dumps a SpecificEntityAnnotation on an output stream.
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236