LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
EnhancedAnalysisDictionaryEntry.cpp
Go to the documentation of this file.
1// Copyright 2002-2020 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
7
8#include <iostream>
9#include <cassert>
11
12namespace Lima
13{
14
15namespace LinguisticProcessing
16{
17
18namespace AnalysisDict
19{
20
22 StringsPoolIndex formId,
23 bool isFinal,
24 bool isEmpty,
25 bool hasLingInfos,
26 bool hasConcatenated,
27 bool hasAccentedForm,
28 unsigned char* startEntryData,
29 unsigned char* endEntryData,
30 const DictionaryData* dicoData,
31 bool isMainKeys,
32 std::shared_ptr<Lima::Common::AbstractAccessByString> access,
34 AbstractDictionaryEntry(formId,isFinal,isEmpty,hasLingInfos,
35 hasConcatenated,hasAccentedForm),
36 m_startEntryData(startEntryData),
37 m_endEntryData(endEntryData),
38 m_dicoData(dicoData),
39 m_notMainKeysHandler(0),
40 m_stringsPool(sp)
41{
43 LDEBUG << "EnhancedAnalysisDictionaryEntry::EnhancedAnalysisDictionaryEntry"
44 << m_stringsPool;
45 if (!isMainKeys) {
46 m_notMainKeysHandler=new NotMainKeysDictionaryEntryHandler(access,sp);
47 }
48}
49
53 m_startEntryData(eade.m_startEntryData),
54 m_endEntryData(eade.m_endEntryData),
55 m_dicoData(eade.m_dicoData),
56 m_notMainKeysHandler(0)
57{
58 if (eade.m_notMainKeysHandler) {
59 m_notMainKeysHandler=new NotMainKeysDictionaryEntryHandler(*(eade.m_notMainKeysHandler));
60 }
61}
62
64{
65 if (m_notMainKeysHandler) {
66 delete m_notMainKeysHandler;
67 m_notMainKeysHandler=0;
68 }
69}
70
72 AbstractDictionaryEntryHandler* targetHandler) const
73{
74 auto handler = targetHandler;
75 if (m_notMainKeysHandler)
76 {
77 m_notMainKeysHandler->setDelegate(targetHandler);
78 handler = m_notMainKeysHandler;
79 }
80 handler->startEntry(m_entryId);
81 parseLingInfos(m_startEntryData, m_endEntryData, m_dicoData, handler);
82 handler->endEntry();
83}
84
86 AbstractDictionaryEntryHandler* targetHandler) const
87{
88 auto handler = targetHandler;
89 if (m_notMainKeysHandler) {
90 m_notMainKeysHandler->setDelegate(targetHandler);
91 handler = m_notMainKeysHandler;
92 }
93 handler->startEntry(m_entryId);
94 parseConcatenated(m_startEntryData,m_endEntryData,m_dicoData,handler);
95 handler->endEntry();
96}
97
99 AbstractDictionaryEntryHandler* targetHandler) const
100{
101#ifdef DEBUG_LP
103 LDEBUG << "EnhancedAnalysisDictionaryEntry::parseAccentedForms";
104#endif
105 auto handler = targetHandler;
106 if (m_notMainKeysHandler) {
107 m_notMainKeysHandler->setDelegate(targetHandler);
108 handler = m_notMainKeysHandler;
109 }
110 handler->startEntry(m_entryId);
111 unsigned char* p=m_startEntryData;
112 assert(p != m_endEntryData);
113 // skip ling infos
114 auto read = static_cast<StringsPoolIndex>(DictionaryData::readCodedInt(p));
115 p+=read;
116 if (p != m_endEntryData)
117 {
118 // parse accented data
119 read = static_cast<StringsPoolIndex>(DictionaryData::readCodedInt(p));
120 unsigned char* end=p+read;
121 while (p!=end)
122 {
123 read = static_cast<StringsPoolIndex>(DictionaryData::readCodedInt(p));
124 if (read == static_cast<StringsPoolIndex>(0))
125 {
126 read = static_cast<StringsPoolIndex>(DictionaryData::readCodedInt(p));
127 handler->deleteAccentedForm(read);
128 }
129 else
130 {
131#ifdef DEBUG_LP
132 LDEBUG << " found accented form:" << (*m_stringsPool)[read];
133#endif
134 handler->foundAccentedForm(read);
135 // parse accented form
136 unsigned char* acc=m_dicoData->getEntryAddr(read);
137 uint64_t tmp=DictionaryData::readCodedInt(acc);
138 if (tmp == 1)
139 {
141 LWARN << "WARNING ! should never accentuate to a delete entry !";
143 }
144 // tmp contains length
145 if (tmp == 0)
146 {
148 LWARN << "WARNING ! should never accentuate to an empty entry !";
149 }
150 parseLingInfos(acc,acc+tmp,m_dicoData,handler);
151 parseConcatenated(acc,acc+tmp,m_dicoData,handler);
152 handler->endAccentedForm();
153 }
154 }
155 }
156 handler->endEntry();
157}
158
160 unsigned char* startEntry,
161 unsigned char* endEntry,
162 const DictionaryData* dicoData,
164{
165 LIMA_UNUSED(endEntry);
166
167#ifdef DEBUG_LP
169 LDEBUG << "EnhancedAnalysisDictionaryEntry::parseLingInfos : "
170 << (uint64_t)startEntry << " , " << (uint64_t)endEntry;
171#endif
172
173 unsigned char* p = startEntry;
174 assert(p != endEntry);
175 uint64_t read = DictionaryData::readCodedInt(p);
176#ifdef DEBUG_LP
177 LDEBUG << "read linginfo length = " << read;
178#endif
179 unsigned char* end = p + read;
180#ifdef DEBUG_LP
181 LDEBUG << "end = " << (uint64_t)(end);
182#endif
183 while ( p != end)
184 {
185#ifdef DEBUG_LP
186 LDEBUG << "read linginfo p = " << (uint64_t)p;
187#endif
188 bool toDelete = false;
189 auto lemma = static_cast<StringsPoolIndex>(DictionaryData::readCodedInt(p));
190
191 if (lemma == static_cast<StringsPoolIndex>(0))
192 {
193#ifdef DEBUG_LP
194 LDEBUG << "read delete flag (p =" << (uint64_t)p << ")";
195#endif
196 toDelete = true;
197 lemma = static_cast<StringsPoolIndex>(DictionaryData::readCodedInt(p));
198 }
199#ifdef DEBUG_LP
200 LDEBUG << "read lemma " << lemma << " (p =" << (uint64_t)p << ")";
201#endif
202 auto norm = static_cast<StringsPoolIndex>(DictionaryData::readCodedInt(p));
203 if (norm == static_cast<StringsPoolIndex>(0))
204 {
205 norm = lemma;
206 }
207#ifdef DEBUG_LP
208 LDEBUG << "read norm " << norm << " (p =" << (uint64_t)p << ")" ;
209#endif
210 if (toDelete)
211 {
212 handler->deleteLingInfos(lemma, norm);
213 }
214 uint64_t lingOffset = DictionaryData::readCodedInt(p);
215#ifdef DEBUG_LP
216 LDEBUG << "read lingOffset = " << lingOffset << " (p =" << (uint64_t)p << ")";
217#endif
218 // lingOffset=0 means there is no ling properties
219 if (lingOffset != 0)
220 {
221 handler->foundLingInfos(lemma, norm);
222 auto props = dicoData->getLingPropertiesAddr(lingOffset);
223 read = DictionaryData::readCodedInt(props);
224 auto propsEnd = props + read;
225 while (props != propsEnd)
226 {
228 }
229 handler->endLingInfos();
230 }
231 }
232}
233
235 unsigned char* startEntry,
236 unsigned char* endEntry,
237 const DictionaryData* dicoData,
238 AbstractDictionaryEntryHandler* handler)
239{
240// ANALYSISDICTLOGINIT;
241// LDEBUG << "parse concatenated " << (uint64_t)startEntry << " , "
242// << (uint64_t)endEntry;
243
244 unsigned char* p=startEntry;
245 assert(p != endEntry);
246 // skip linginfos
247 uint64_t read=DictionaryData::readCodedInt(p);
248 p+=read;
249// LDEBUG << "skip ling info of length " << read;
250 if (p != endEntry)
251 {
252 // skip accented
254 p+=read;
255// LDEBUG << "skip accented of length " << read;
256 if (p != endEntry)
257 {
258 // read concat
260 unsigned char* end=p+read;
261// LDEBUG << "read concat of length " << read;
262 while (p!=end)
263 {
265 if (read == 0)
266 {
268// LDEBUG << "has delete info";
269 // parse concat to provide delete infos
270 handler->deleteConcatenated();
271 bool hasInfo=false;
272 unsigned char* pp=p;
273 uint64_t nb=read;
274// LDEBUG << "has " << nb << " components";
275 while (nb-- > 0)
276 {
277 auto str = static_cast<StringsPoolIndex>(DictionaryData::readCodedInt(pp));
278// LDEBUG << "read str=" << str;
279 uint64_t pos=DictionaryData::readCodedInt(pp);
280// LDEBUG << "read pos=" << pos;
281 uint64_t len=DictionaryData::readCodedInt(pp);
282// LDEBUG << "read len=" << len;
283 handler->foundComponent(pos,len,str);
284 uint64_t lilength=DictionaryData::readCodedInt(pp);
285// LDEBUG << "lingInfo length = " << lilength;
286 unsigned char* pp_end=pp+lilength;
287 while (pp != pp_end)
288 {
289 hasInfo=true;
290 DictionaryData::readCodedInt(pp); // read lemma
291 DictionaryData::readCodedInt(pp); // read norm
292 DictionaryData::readCodedInt(pp); // read props
293 }
294 }
295 handler->endConcatenated();
296 if (!hasInfo)
297 {
298 p=pp;
299 continue;
300 }
301 }
302 uint64_t nbComponents=read;
303// LDEBUG << "has " << nbComponents << " components";
304 // parse concat infos
305 handler->foundConcatenated();
306 while (nbComponents-- > 0)
307 {
308 auto str = static_cast<StringsPoolIndex>(DictionaryData::readCodedInt(p));
309// LDEBUG << "read string " << str;
310 uint64_t pos=DictionaryData::readCodedInt(p);
311// LDEBUG << "read pos=" << pos;
312 uint64_t len=DictionaryData::readCodedInt(p);
313// LDEBUG << "read len=" << len;
314 uint64_t lilength=DictionaryData::readCodedInt(p);
315// LDEBUG << "read LIlength=" << lilength;
316 handler->foundComponent(pos,len,str);
317 unsigned char* liend=p+lilength;
318 while (p != liend)
319 {
320 auto lemma = static_cast<StringsPoolIndex>(DictionaryData::readCodedInt(p)); // read lemma
321// LDEBUG << "read lemma=" << lemma;
322 auto norm =static_cast<StringsPoolIndex>(DictionaryData::readCodedInt(p)); // read norm
323// LDEBUG << "read norm=" << norm;
324 if (norm == static_cast<StringsPoolIndex>(0))
325 {
326 norm=lemma;
327 }
328 handler->foundLingInfos(lemma,norm);
329 uint64_t lingOffset=DictionaryData::readCodedInt(p); // read props
330// LDEBUG << "read ling offset = " << lingOffset;
331 unsigned char* props = dicoData->getLingPropertiesAddr(lingOffset);
332 read = DictionaryData::readCodedInt(props);
333// LDEBUG << "lingprops length = " << read;
334 unsigned char* propsEnd = props + read;
335 while (props!=propsEnd)
336 {
337 handler->foundProperties(LinguisticCode::decodeFromBinary(props));
338 }
339 handler->endLingInfos();
340 }
341 handler->endComponent();
342 }
343 handler->endConcatenated();
344 }
345 }
346 }
347}
348
349
350void NotMainKeysDictionaryEntryHandler::foundLingInfos(StringsPoolIndex lemma,StringsPoolIndex norm)
351{
352 StringsPoolIndex splemma=(*m_sp)[m_access->getSpelling(lemma)];
353 StringsPoolIndex spnorm=splemma;
354 if (norm != lemma) {
355 spnorm=(*m_sp)[m_access->getSpelling(norm)];
356 }
357 m_delegate->foundLingInfos(splemma,spnorm);
358}
359
361 StringsPoolIndex norm)
362{
363 auto splemma=(*m_sp)[m_access->getSpelling(lemma)];
364 auto spnorm=splemma;
365 if (norm != lemma)
366 {
367 spnorm=(*m_sp)[m_access->getSpelling(norm)];
368 }
369 m_delegate->deleteLingInfos(splemma,spnorm);
370}
371
372}
373
374}
375
376}
#define LWARN
Definition LimaCommon.h:160
#define LIMA_UNUSED(x)
Definition LimaCommon.h:224
#define LDEBUG
Definition LimaCommon.h:157
#define ANALYSISDICTLOGINIT
static LinguisticCode decodeFromBinary(std::istream &is)
Definition StdBitset.h:197
virtual void deleteLingInfos(StringsPoolIndex lemma, StringsPoolIndex norm)
virtual void foundLingInfos(StringsPoolIndex lemma, StringsPoolIndex norm)
unsigned char * getLingPropertiesAddr(uint64_t index) const
unsigned char * getEntryAddr(uint64_t index) const
virtual void parseAccentedForms(AbstractDictionaryEntryHandler *handler) const override
virtual void parseLingInfos(AbstractDictionaryEntryHandler *handler) const override
EnhancedAnalysisDictionaryEntry(StringsPoolIndex formId, bool isFinal, bool isEmpty, bool hasLingInfos, bool hasConcatenated, bool hasAccentedForm, unsigned char *startEntryData, unsigned char *endEntryData, const DictionaryData *dicoData, bool isMainKeys, std::shared_ptr< Lima::Common::AbstractAccessByString > access, Lima::FsaStringsPool *sp)
virtual void parseConcatenated(AbstractDictionaryEntryHandler *handler) const override
DictionaryEntryHandler used to convert dico ids into stringPool ids when dictionary keys are differen...
NAUTITIA.