20namespace LinguisticProcessing
23namespace MorphologicAnalysis
29 bool tryUncapitalized,
30 bool tryDesaccentedForm,
31 std::shared_ptr<FlatTokenizer::CharChart> charChart,
33 m_confidentMode(confidentMode),
34 m_tryDirect(tryDirect),
35 m_tryUncapitalized(tryUncapitalized),
36 m_tryDesaccentedForm(tryDesaccentedForm),
37 m_charChart(charChart),
55 LDEBUG <<
"AlternativesReader::readAlternatives (direct: " << m_tryDirect
56 <<
" ; uncap: " << m_tryUncapitalized <<
" ; desac: "
57 << m_tryDesaccentedForm <<
")" << str;
62 LDEBUG <<
"AlternativesReader::readAlternatives trying direct";
70 LDEBUG <<
"AlternativesReader::readAlternatives direct, hasLingInfos";
75 if (m_confidentMode)
return;
80 LDEBUG <<
"AlternativesReader::readAlternatives direct, hasConcatenated";
83 if (m_confidentMode)
return;
88 LDEBUG <<
"AlternativesReader::readAlternatives direct, hasAccentedForms";
91 if (m_confidentMode)
return;
95 if (m_tryUncapitalized && m_charChart)
97 LimaString lowerWord = m_charChart->toLower(str);
99 LDEBUG <<
"AlternativesReader::readAlternatives trying lower:" << lowerWord;
101 if (!(lowerWord.isEmpty()) && (lowerWord!=str))
105 <<
"<marked>" << str <<
"</marked>"
106 <<
"<lower>" << lowerWord <<
"</lower>";
108 StringsPoolIndex idx=(*m_sp)[lowerWord];
119 LDEBUG <<
"lower word" << lowerWord <<
"found";
121 if (m_confidentMode)
return;
127 LDEBUG <<
"lower word" << lowerWord <<
"concatenated found";
129 if (m_confidentMode)
return;
135 LDEBUG <<
"lower word" << lowerWord <<
"accented found";
137 if (m_confidentMode)
return;
142 if (m_tryDesaccentedForm && m_charChart)
146 LDEBUG <<
"AlternativesReader::readAlternatives trying unmarked:" << unmarked;
148 if ((unmarked!=
LimaString()) && (unmarked!=str))
154 <<
" to stringpool " << m_sp;
156 StringsPoolIndex idx=(*m_sp)[unmarked];
158 LDEBUG <<
"-> StringPool returned index " << idx;
169 LDEBUG <<
"confident mode: " << m_confidentMode;
170 LDEBUG <<
"lingInfosHandler: " << (
void*)accentedHandler <<
" entry.hasLingInfos:" << entry.
hasLingInfos();
177 if (m_confidentMode)
return;
180 LDEBUG <<
"concatHandler: " << (
void*)accentedHandler <<
" entry.hasConcatenated:" << entry.
hasConcatenated();
185 if (m_confidentMode)
return;
188 LDEBUG <<
"accentedHandler: " << (
void*)accentedHandler <<
" entry.hasAccentedForms:" << entry.
hasAccentedForms();
193 if (m_confidentMode)
return;
199 StringsPoolIndex idx=(*m_sp)[unmarked];
202 LDEBUG <<
"AlternativesReader::readAlternatives is an acronym; using simpler form" << unmarked;
207 LDEBUG <<
"AlternativesReader::readAlternatives no alternative found;";
interface for analysis dictionary
virtual DictionaryEntry getEntry(const Lima::LimaString &word) const =0
get dictionary entry for a word
bool hasLingInfos() const
void parseAccentedForms(AbstractDictionaryEntryHandler *handler) const
bool hasAccentedForms() const
void parseConcatenated(AbstractDictionaryEntryHandler *handler) const
void parseLingInfos(AbstractDictionaryEntryHandler *handler) const
bool hasConcatenated() const
AlphaCapitalType getAlphaCapital() const
holds surface data of a token
uint64_t position() const
void addOrthographicAlternatives(StringsPoolIndex alt)
const TStatus & status() const
const LimaString & stringForm() const
StringsPoolIndex form() const
AlternativesReader(bool confidentMode, bool tryDirect, bool tryUncapitalized, bool tryDesaccentedForm, std::shared_ptr< FlatTokenizer::CharChart > charChart, FsaStringsPool *sp)
virtual ~AlternativesReader()
void readAlternatives(LinguisticAnalysisStructure::Token &token, const AnalysisDict::AbstractAnalysisDictionary &dico, AnalysisDict::AbstractDictionaryEntryHandler *lingInfosHandler=0, AnalysisDict::AbstractDictionaryEntryHandler *concatHandler=0, AnalysisDict::AbstractDictionaryEntryHandler *accentedHandler=0) const
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.