LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
MultiLevelAnalysisDictionaryEntry.cpp
Go to the documentation of this file.
1// Copyright 2002-2020 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/***************************************************************************
7 * Copyright (C) 2004-2020 by CEA LIST *
8 * *
9 ***************************************************************************/
11
14
16
17#include <vector>
18#include <limits>
19#include <iostream>
20#include <cassert>
21#include <set>
22
23namespace Lima {
24namespace LinguisticProcessing {
25namespace AnalysisDict {
26
28{
29public:
34
35 // position attributes
36 unsigned char* pos;
37 unsigned char* posEnd;
38
39 // state attributes
40 StringsPoolIndex currentLemma;
42 StringsPoolIndex currentNorm;
45 bool final;
46
47 // data attributes
49 std::shared_ptr<Lima::Common::AbstractAccessByString> keys;
51
52 void next(FsaStringsPool& sp);
53 bool end() const;
54 bool operator<(const LingInfoLevelState& lis) const;
55 bool operator==(const LingInfoLevelState& lis) const;
56
57};
58
60{
61public:
62
64 {
65 public:
66 StringsPoolIndex form;
68 uint64_t pos;
69 uint64_t len;
71
72 Component() = default;
73 ~Component() = default;
74 Component(const Component&) = default;
75 Component& operator=(const Component&) = default;
76
77 bool operator<(const Component& c) const { return formStr<c.formStr;};
78 bool operator==(const Component& c) const { return form==c.form;};
79 };
80
85
86 // position attributes
87 unsigned char* pos;
88 unsigned char* posEnd;
89
90 // state attributes
91 bool final;
92 std::vector<Component> components;
93
94 // data attributes
96 std::shared_ptr<Lima::Common::AbstractAccessByString> keys;
98
99
100 bool operator<(const ConcatenatedLevelState& cls) const
101 {
102 return (cls.components.empty()
103 || ( !components.empty() && (components < cls.components) ) );
104 };
105 bool operator==(const ConcatenatedLevelState& cls) const
106 {
107 return (components == cls.components);
108 };
109
110 void next(FsaStringsPool& sp);
111 bool end() const;
112
113};
114
116{
117public:
122
123 // position attributes
124 unsigned char* pos;
125 unsigned char* posEnd;
126
127 // state attributes
128 StringsPoolIndex accentedForm;
130 StringsPoolIndex accentedEntry;
131 bool final;
132
133 // data attributes
135 std::shared_ptr<Lima::Common::AbstractAccessByString> keys;
137
138 void next(FsaStringsPool& sp);
139 bool end() const;
140 bool operator<(const AccentedLevelState& lis) const;
141 bool operator==(const AccentedLevelState& lis) const;
142};
143
145 pos(0),
146 posEnd(0),
147 accentedForm(STRINGS_POOL_INDEX_MAX_VALUE),
148 accentedFormStr(),
149 accentedEntry(STRINGS_POOL_INDEX_MAX_VALUE),
150 final(false),
151 mainKeys(false),
152 keys(0),
153 dicoData(0)
154{}
155
156
157void parseLingInfos(std::vector<LingInfoLevelState>& data,Lima::FsaStringsPool* sp,
159void parseConcatenated(std::vector<ConcatenatedLevelState>& data,Lima::FsaStringsPool* sp,
161
162
164{
166
168 const std::vector<MultiLevelAnalysisDictionaryEntry::LevelData>& data,
170 m_data(data),
171 m_sp(sp) {};
172
173
175
178
179 std::vector<MultiLevelAnalysisDictionaryEntry::LevelData> m_data;
181
182};
183
184
186 StringsPoolIndex formId,
187 bool isFinal,
188 bool isEmpty,
189 bool hasLingInfos,
190 bool hasConcatenated,
191 bool hasAccentedForm,
192 const std::vector<LevelData>& data,
195 isFinal,
196 isEmpty,
197 hasLingInfos,
198 hasConcatenated,
199 hasAccentedForm),
201{
202}
203
210
217
218
223
225 AbstractDictionaryEntryHandler* handler) const
226{
227#ifdef DEBUG_LP
229 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseAccentedForms";
230#endif
231 handler->startEntry(m_entryId);
232
233 // initialize states
234 std::vector<AccentedLevelState> state;
235 for (const auto& ld : m_d->m_data)
236 {
237 AccentedLevelState curState;
238
239 // set position attributes
240 curState.pos = ld.startEntryData;
241 assert(curState.pos != ld.endEntryData);
242 // skip linginfos
243 auto read = DictionaryData::readCodedInt(curState.pos);
244 curState.pos += read;
245 if (curState.pos != ld.endEntryData)
246 {
247 // read accented
248 read = DictionaryData::readCodedInt(curState.pos);
249 curState.posEnd= curState.pos + read;
250
251 // set data attributes
252 curState.mainKeys = ld.mainKeys;
253 curState.keys = ld.keys;
254 curState.dicoData = ld.dicoData;
255
256 // set state attributes
257 curState.next(*m_d->m_sp);
258
259 state.push_back(curState);
260 }
261 }
262
263 // parse accented forms
264 while (true)
265 {
266 // find lower state
267 auto* lowerState = &state.front();
268 auto final = lowerState->final;
269 auto hasInfo = !lowerState->final;
270 for (auto& als : state)
271 {
272 if (als < *lowerState)
273 {
274 lowerState = &als;
275 final=als.final;
276 }
277 else if (als == *lowerState)
278 {
279 if (als.final)
280 {
281 final=true;
282 }
283 }
284 }
285
286 if (lowerState->end())
287 {
288 break;
289 }
290
291 if (final)
292 {
293 handler->deleteAccentedForm(lowerState->accentedForm);
294 }
295 if (hasInfo)
296 {
297 handler->foundAccentedForm(lowerState->accentedForm);
298 std::vector<LingInfoLevelState> liStates;
299 std::vector<ConcatenatedLevelState> concatStates;
300 auto finalReached = false;
301 for (const auto& als : state)
302 {
303 if (als == *lowerState)
304 {
305 finalReached = finalReached || als.final;
306 if (!finalReached)
307 {
308
309 auto acc = als.dicoData->getEntryAddr(als.accentedEntry);
310 auto tmp = DictionaryData::readCodedInt(acc);
311 if (tmp == 1)
312 {
314 LWARN << "WARNING ! should never accentuate to a delete entry !";
316 }
317 // tmp contains length
318 if (tmp == 0)
319 {
321 LWARN << "WARNING ! should never accentuate to a empty entry !";
322 }
323 auto accEnd = acc+tmp;
324 // read linginfo
326 if (tmp > 0)
327 {
328 LingInfoLevelState curState;
329 curState.pos = acc;
330 curState.posEnd = acc + tmp;
331
332 // set data attributes
333 curState.mainKeys = als.mainKeys;
334 curState.keys = als.keys;
335 curState.dicoData = als.dicoData;
336
337 // set state attributes
338 curState.next(*m_d->m_sp);
339
340 liStates.push_back(curState);
341
342 acc += tmp;
343 }
344 if (acc != accEnd)
345 {
346 // skip accented
348 acc += tmp;
349 if (acc != accEnd)
350 {
351 // read concat
353
354 ConcatenatedLevelState curState;
355 curState.pos = acc;
356 curState.posEnd = acc+tmp;
357
358 // set data attributes
359 curState.mainKeys = als.mainKeys;
360 curState.keys = als.keys;
361 curState.dicoData = als.dicoData;
362
363 // set state attributes
364 curState.next(*m_d->m_sp);
365
366 concatStates.push_back(curState);
367
368 acc += tmp;
369 assert(acc == accEnd);
370 }
371 }
372 }
373
374 AnalysisDict::parseLingInfos(liStates, m_d->m_sp, handler);
375 AnalysisDict::parseConcatenated(concatStates, m_d->m_sp, handler);
376
377 }
378 }
379 handler->endAccentedForm();
380 }
381 // advance states
382 for (auto& als : state)
383 {
384 if (als == *lowerState)
385 {
386 als.next(*m_d->m_sp);
387 }
388 }
389 }
390 handler->endEntry();
391}
392
394 AbstractDictionaryEntryHandler* handler) const
395{
396
397 handler->startEntry(m_entryId);
398
399 // initialize states
400 std::vector<ConcatenatedLevelState> state;
401
402 for (const auto& ld : m_d->m_data)
403 {
404 ConcatenatedLevelState curState;
405
406 // set position attributes
407 curState.pos = ld.startEntryData;
408 assert(curState.pos != ld.endEntryData);
409
410 // skip linginfos
411 uint64_t read = DictionaryData::readCodedInt(curState.pos);
412 curState.pos += read;
413 if (curState.pos != ld.endEntryData)
414 {
415 // skip accented
416 read = DictionaryData::readCodedInt(curState.pos);
417 curState.pos += read;
418 if (curState.pos != ld.endEntryData)
419 {
420 // read concat
421 read = DictionaryData::readCodedInt(curState.pos);
422 curState.posEnd = curState.pos+read;
423
424 // set data attributes
425 curState.mainKeys = ld.mainKeys;
426 curState.keys = ld.keys;
427 curState.dicoData = ld.dicoData;
428
429 // set state attributes
430 curState.next(*m_d->m_sp);
431
432 state.push_back(curState);
433 }
434 }
435 }
436
437 AnalysisDict::parseConcatenated(state, m_d->m_sp, handler);
438
439 handler->endEntry();
440
441}
442
444 AbstractDictionaryEntryHandler* handler) const
445{
446#ifdef DEBUG_LP
448 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseLingInfos initialize entry reading";
449#endif
450
451 handler->startEntry(m_entryId);
452
453 // initialize states
454 std::vector<LingInfoLevelState> states;
455 for (const auto& ld : m_d->m_data)
456 {
457 LingInfoLevelState curState;
458
459 // set position attributes
460 curState.pos = ld.startEntryData;
461 auto read = DictionaryData::readCodedInt(curState.pos);
462 curState.posEnd = curState.pos + read;
463
464 // set data attributes
465 curState.mainKeys = ld.mainKeys;
466 curState.keys = ld.keys;
467 curState.dicoData = ld.dicoData;
468
469 // read current level first state;
470 curState.next(*m_d->m_sp);
471
472 states.push_back(curState);
473 }
474 AnalysisDict::parseLingInfos(states, m_d->m_sp, handler);
475
476 handler->endEntry();
477}
478
480 pos(0),
481 posEnd(0),
482 currentLemma(STRINGS_POOL_INDEX_MAX_VALUE),
483 lemmaStr(),
484 currentNorm(STRINGS_POOL_INDEX_MAX_VALUE),
485 normStr(),
486 lingInfoOffset(0),
487 final(false),
488 mainKeys(false),
489 keys(0),
490 dicoData(0)
491{}
492
494{
495#ifdef DEBUG_LP
497 LDEBUG << "LingInfoLevelState::next" << (void*)pos << (void*)posEnd
498 << "lemma : " << lemmaStr << ", norm : " << normStr;
499#endif
500 if (pos != posEnd)
501 {
503 if (currentLemma == static_cast<StringsPoolIndex>(0))
504 {
505 final = true;
507 }
508 else
509 {
510 final = false;
511 }
512 lemmaStr = keys->getSpelling(currentLemma);
513 if (!mainKeys)
514 {
516 }
518 if (currentNorm == static_cast<StringsPoolIndex>(0))
519 {
522 }
523 else
524 {
525 normStr = keys->getSpelling(currentNorm);
526 if (!mainKeys)
527 {
528 currentNorm = sp[normStr];
529 }
530 }
532 }
533 else
534 {
536 lemmaStr.clear();
538 normStr.clear();
539 final = false;
540 lingInfoOffset = 0;
541 }
542#ifdef DEBUG_LP
543 LDEBUG << "LingInfoLevelState::next on OUT:" << (void*)pos << (void*)posEnd
544 << "lemma : " << lemmaStr << ", norm : " << normStr;
545#endif
546}
547
549{
550#ifdef DEBUG_LP
552 LDEBUG << "LingInfoLevelState::end" << pos << posEnd << currentLemma
554#endif
556}
557
559{
560 if (lis.end()) return true;
561 if (end()) return false;
562 if (currentLemma == lis.currentLemma) return (normStr < lis.normStr);
563 return lemmaStr < lis.lemmaStr;
564}
565
567{
568 return ((currentLemma == lis.currentLemma) && (currentNorm == lis.currentNorm));
569}
570
572 std::vector<LingInfoLevelState>& states,
575{
576 if (states.empty()) return;
577
578#ifdef DEBUG_LP
580 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseLingInfos IN";
581#endif
582 // read entry
583 while (true)
584 {
585 // find lower lemma,norm
586 auto* lowerState = &states.front();
587 auto final = false;
588 auto hasInfos = false;
589#ifdef DEBUG_LP
590 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseLingInfos find lower state. state size="
591 << states.size();
592#endif
593 for (auto& state : states)
594 {
595#ifdef DEBUG_LP
596 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseLingInfos next level is at state lemma="
597 << state.lemmaStr;
598#endif
599 if (state < *lowerState)
600 {
601#ifdef DEBUG_LP
602 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseLingInfos is lower !";
603#endif
604 lowerState = &state;
605 hasInfos = (lowerState->lingInfoOffset !=0);
606 final = lowerState->final;
607 }
608 else if (state == *lowerState)
609 {
610#ifdef DEBUG_LP
611 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseLingInfos stateItr == lowerState";
612#endif
613 if (!final && state.lingInfoOffset != 0)
614 {
615 hasInfos = true;
616 }
617 final = state.final;
618 }
619 }
620 if (lowerState->end())
621 {
622 // nothing more to read, exit the while(true)
623#ifdef DEBUG_LP
624 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseLingInfos nothing more to read, exit";
625#endif
626 break;
627 }
628#ifdef DEBUG_LP
629 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseLingInfos lowerState found lemma="
630 << lowerState->lemmaStr << ", norm=" << lowerState->normStr;
631#endif
632 // if final then call delete
633 if (final)
634 {
635 handler->deleteLingInfos(lowerState->currentLemma, lowerState->currentNorm);
636 }
637 // if has info, should have some properties
638 if (hasInfos)
639 {
640 handler->foundLingInfos(lowerState->currentLemma, lowerState->currentNorm);
641 }
642 bool finalReached = false;
643 std::set<LinguisticCode> propsRead;
644 for (auto& state : states)
645 {
646#ifdef DEBUG_LP
647 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseLingInfos in second states loop. lemma="
648 << state.lemmaStr << ", norm=" << state.normStr;
649#endif
650 if (state == *lowerState)
651 {
652#ifdef DEBUG_LP
653 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseLingInfos stateItr == lowerState !";
654#endif
655 // read this level
656 if (!finalReached)
657 {
658 finalReached = state.final;
659 if (state.lingInfoOffset != 0)
660 {
661#ifdef DEBUG_LP
662 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseLingInfos read level info";
663#endif
664 if (state.lingInfoOffset != 0)
665 {
666 auto props = state.dicoData->getLingPropertiesAddr(state.lingInfoOffset);
667 auto read = DictionaryData::readCodedInt(props);
668 auto propsEnd = props + read;
669 while (props!=propsEnd)
670 {
671 auto l = LinguisticCode::decodeFromBinary(props);
672#ifdef DEBUG_LP
673 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseLingInfos got linguistic code" << l.toString();
674#endif
675 if (propsRead.find(l) == propsRead.end())
676 {
677 handler->foundProperties(l);
678 propsRead.insert(l);
679 }
680 else
681 {
682#ifdef DEBUG_LP
683 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseLingInfos ling properties already read in previous dictionary !";
684#endif
685 }
686 }
687 }
688 }
689 }
690 // next state
691 state.next(*sp);
692 }
693 }
694 if (hasInfos)
695 {
696 handler->endLingInfos();
697 }
698 }
699}
700
702{
703#ifdef DEBUG_LP
705 LDEBUG << "ConcatenatedLevelState::next" << (void*)pos << (void*)posEnd;
706#endif
707
708 components.clear();
709 if (pos != posEnd)
710 {
712 if (read == 0)
713 {
714 final = true;
716 }
717 else
718 {
719 final = false;
720 }
721 // read the components
722 for (uint64_t nb = read; nb > 0; nb--)
723 {
724 components.push_back(Component());
725 auto& c = components.back();
726
727 // set component attributes
729 c.formStr = keys->getSpelling(c.form);
730 if (!mainKeys)
731 {
732 c.form = sp[c.formStr];
733 }
736
737 // set ling info state position attribute
739 c.liState.pos = pos;
740 pos += read;
741 c.liState.posEnd = pos;
742
743 // set ling info state data attributes
744 c.liState.mainKeys = mainKeys;
745 c.liState.keys = keys;
746 c.liState.dicoData = dicoData;
747
748 // set ling info state attributes
749 c.liState.next(sp);
750 }
751 }
752}
753
755{
756 return (pos==posEnd) && components.empty();
757}
758
759
761 std::vector<ConcatenatedLevelState>& state,
764{
765#ifdef DEBUG_LP
767 LDEBUG << "parseConcatenated";
768#endif
769 if (state.empty()) return;
770
771 // read entry
772 while (true)
773 {
774 // find lower state
775#ifdef DEBUG_LP
776 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseConcatenated find lower concat state";
777#endif
778 auto* lowerState = &state.front();
779 auto final = false;
780 auto hasInfos = false;
781 for (auto& cls : state)
782 {
783#ifdef DEBUG_LP
784 LDEBUG << "state has " << cls.components.size() << " components";
785#endif
786 if (cls < *lowerState)
787 {
788#ifdef DEBUG_LP
789 LDEBUG << "is lower !";
790#endif
791 lowerState = &cls;
792 final = cls.final;
793 hasInfos = (cls.components.front().liState.lingInfoOffset != 0);
794 }
795 else if (cls == *lowerState)
796 {
797#ifdef DEBUG_LP
798 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseConcatenated stateItr == lowerState";
799#endif
800 if (!final)
801 {
802 if (!cls.components.empty()
803 && cls.components.front().liState.lingInfoOffset != 0)
804 {
805 hasInfos = true;
806 }
807 if (cls.final)
808 {
809 final = true;
810 }
811 }
812 }
813 }
814 // check if info to read
815 if (lowerState->end())
816 {
817 break;
818 }
819#ifdef DEBUG_LP
820 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseConcatenated read infos";
821#endif
822 // read info
823 if (final)
824 {
825 handler->deleteConcatenated();
826 for (const auto& component : lowerState->components)
827 {
828 handler->foundComponent(component.pos, component.len, component.form);
829 }
830 handler->endConcatenated();
831 }
832 auto finalReached = false;
833 if (hasInfos)
834 {
835 // retrieve all components iterators
836 std::vector<std::pair<std::vector<ConcatenatedLevelState::Component>::iterator,
837 std::vector<ConcatenatedLevelState::Component>::iterator> > componentsIt;
838 for (auto& cls : state)
839 {
840 if (cls == *lowerState)
841 {
842#ifdef DEBUG_LP
843 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseConcatenated stateItr == lowerState";
844#endif
845 if (!finalReached)
846 {
847 componentsIt.push_back(std::make_pair(cls.components.begin(),
848 cls.components.end()));
849 if (cls.final)
850 {
851 finalReached = true;
852 }
853 }
854 }
855 }
856 // read all components
857#ifdef DEBUG_LP
858 LDEBUG << "MultiLevelAnalysisDictionaryEntry::parseConcatenated read the "
859 << componentsIt.size() << " level iterators";
860#endif
861 handler->foundConcatenated();
862 while (componentsIt.front().first != componentsIt.front().second)
863 {
864 handler->foundComponent(componentsIt.front().first->pos,
865 componentsIt.front().first->len,
866 componentsIt.front().first->form);
867 std::vector<LingInfoLevelState> lingInfoState;
868 for (auto& component : componentsIt)
869 {
870 assert(component.first != component.second);
871 lingInfoState.push_back(component.first->liState);
872 component.first.operator++();
873 }
874 parseLingInfos(lingInfoState, sp, handler);
875 handler->endComponent();
876 }
877 handler->endConcatenated();
878 }
879 // next states
880 for (auto& cls : state)
881 {
882 if (cls == *lowerState)
883 {
884 cls.next(*sp);
885 }
886 }
887
888 }
889}
890
892{
893#ifdef DEBUG_LP
895 LDEBUG << "AccentedLevelState::next" << (void*)pos << (void*)posEnd;
896#endif
897
898 if (pos != posEnd)
899 {
901 if (accentedEntry == static_cast<StringsPoolIndex>(0))
902 {
903 final = true;
905 }
906 else
907 {
908 final = false;
909 }
910 accentedFormStr = keys->getSpelling(accentedEntry);
911 if (!mainKeys)
912 {
914 }
915 else
916 {
918 }
919 }
920 else
921 {
922 final = false;
925 }
926
927}
928
933
935{
936 if (as.end()) return true;
937 if (end()) return false;
939}
940
942{
943 return accentedForm == as.accentedForm;
944}
945
946
947} // namespace
948} // namespace
949} // namespace
#define LWARN
Definition LimaCommon.h:160
#define LDEBUG
Definition LimaCommon.h:157
#define ANALYSISDICTLOGINIT
static LinguisticCode decodeFromBinary(std::istream &is)
Definition StdBitset.h:197
virtual void foundComponent(uint64_t position, uint64_t length, StringsPoolIndex form)
virtual void deleteLingInfos(StringsPoolIndex lemma, StringsPoolIndex norm)
virtual void foundLingInfos(StringsPoolIndex lemma, StringsPoolIndex norm)
AccentedLevelState(const AccentedLevelState &)=default
AccentedLevelState & operator=(const AccentedLevelState &)=default
ConcatenatedLevelState(const ConcatenatedLevelState &)=default
ConcatenatedLevelState & operator=(const ConcatenatedLevelState &)=default
std::shared_ptr< Lima::Common::AbstractAccessByString > keys
LingInfoLevelState(const LingInfoLevelState &)=default
LingInfoLevelState & operator=(const LingInfoLevelState &)=default
MultiLevelAnalysisDictionaryEntry(StringsPoolIndex formId, bool isFinal, bool isEmpty, bool hasLingInfos, bool hasConcatenated, bool hasAccentedForm, const std::vector< LevelData > &data, Lima::FsaStringsPool *sp)
virtual void parseConcatenated(AbstractDictionaryEntryHandler *handler) const override
virtual void parseLingInfos(AbstractDictionaryEntryHandler *handler) const override
MultiLevelAnalysisDictionaryEntry & operator=(const MultiLevelAnalysisDictionaryEntry &)
virtual void parseAccentedForms(AbstractDictionaryEntryHandler *handler) const override
void parseConcatenated(std::vector< ConcatenatedLevelState > &data, Lima::FsaStringsPool *sp, AbstractDictionaryEntryHandler *handler)
void parseLingInfos(std::vector< LingInfoLevelState > &data, Lima::FsaStringsPool *sp, AbstractDictionaryEntryHandler *handler)
NAUTITIA.
QString LimaString
Definition LimaString.h:33
#define STRINGS_POOL_INDEX_MAX_VALUE
Definition stringspool.h:40