LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
Text.h
Go to the documentation of this file.
1// Copyright 2002-2013 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6// NAUTITIA
7//
8// jys 24-JUL-2002
9//
10// Text is the class which reads original text and does its
11// 1st transformation into characters classes string.
12
13#ifndef LIMA_LINGUISTICPROCESSING_FLATTOKENIZER_TEXT_H
14#define LIMA_LINGUISTICPROCESSING_FLATTOKENIZER_TEXT_H
15
16
17#include "FlatTokenizerExport.h"
22#include <wchar.h>
23
24#include "CharChart.h"
25
26
27namespace Lima
28{
29namespace LinguisticProcessing
30{
31namespace FlatTokenizer
32{
33
34class TextPrivate;
36{
37friend class TextPrivate;
38
39public:
40 Text(MediaId lang, std::shared_ptr<CharChart> charChart);
41 virtual ~Text();
42 Text(const Text&) = delete;
43 Text& operator=(const Text&) = delete;
44
45 void setText(const Lima::LimaString& text);
46
47 // Clear the entirely class and structure to accept new text
48 void clear();
49
50 // set graph in which insert token
51 void setGraph(LinguisticGraphVertex position,LinguisticGraph* graph);
52 void finalizeAndUnsetGraph();
53
54 // gives the current character class
55 const CharClass* currentClass() const;
56
57 // gives the current character
58 Lima::LimaChar currentChar() const;
59
60 //
61 // returns the character class with signed offset i
62 Lima::LimaChar operator[] (int i) const;
63
64 // increments text pointer
65 Lima::LimaChar advance();
66
67 // flushes current token
68 void flush();
69
70 // takes a token and add it to the result graph
71 Lima::LimaString token();
72
73 // performs a trace
74 void trace();
75
76 // sets token status
77 void setAlphaCapital(const LinguisticAnalysisStructure::AlphaCapitalType alphaCapital);
78 void setAlphaRoman(const LinguisticAnalysisStructure::AlphaRomanType alphaRoman);
79 void setAlphaHyphen(const unsigned char isAlphaHyphen);
80 void setAlphaPossessive(const unsigned char isAlphaPossessive);
81 void setAlphaConcatAbbrev(const unsigned char isConcatAbbreviation);
82 void setTwitter(const unsigned char isTwitter);
83 void setNumeric(const LinguisticAnalysisStructure::NumericType numeric);
84 void setStatus(const LinguisticAnalysisStructure::StatusType status);
85 void setDefaultKey(const Lima::LimaString& defaultKey);
86
87 int position() const;
88 int size() const;
89
90 void setStatus(const LinguisticAnalysisStructure::TStatus& status);
91
92 void computeDefaultStatus();
93
94private:
95 TextPrivate* m_d;
96};
97
98} //namespace FlatTokenizer
99} // namespace LinguisticProcessing
100} // namespace Lima
101
102#endif
#define LIMA_FLATTOKENIZER_EXPORT
A graph structure for linguistic analysis.
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
NAUTITIA.
QChar LimaChar
Definition LimaString.h:30
QString LimaString
Definition LimaString.h:33