LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
utf8_reader.h
Go to the documentation of this file.
1// Copyright 2021 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6#ifndef DEEPLIMA_APPS_UTF8_DECODER_H
7#define DEEPLIMA_APPS_UTF8_DECODER_H
8
9#include <cassert>
10#include <limits>
11#include <stdexcept>
12
13#include <string.h>
14#include <unicode/uchar.h>
15#include <unicode/uscript.h>
16
17namespace deeplima
18{
19namespace segmentation
20{
21namespace impl
22{
23
24#define UTF8_DECODER_CHAR_HISTORY_MASK 0x7FFFFFFFFFFFFFFFull
25// 21 * 3 characters = 63 bits accepted
26
34
35const uint8_t bits_per_position_table[] = {
36 21, // Unicode code point
37 8, // Unicode class
38 3, // UTF-8 character len in bytes (1 .. 6)
39 1 // Script changed since prev code point (Yes / No)
40};
41
42template <class TBufferType = uint64_t, int NUM_CHANNELS = 4, int BYTES_BUFFER_SIZE = 8>
44{
45protected:
46 typedef TBufferType buffer_t;
47
48private:
49 uint8_t m_bytes_buffer[BYTES_BUFFER_SIZE];
50 int32_t m_bytes_left_to_parse;
51
52 buffer_t m_buffer[NUM_CHANNELS];
53
54 uint8_t m_prev_script_code;
55
56 inline void push_value(uc_channel_t ch, uint32_t value)
57 {
58 m_buffer[ch] = (m_buffer[ch] << bits_per_position(ch)) | buffer_t(value);
59 }
60
61public:
62
64 : m_bytes_left_to_parse(0),
65 m_prev_script_code(USCRIPT_COMMON)
66 {
67 assert(USCRIPT_CODE_LIMIT < std::numeric_limits<uint8_t>::max());
68
69 reset();
70 }
71
72 void reset()
73 {
74 m_bytes_left_to_parse = 0;
75 m_prev_script_code = USCRIPT_COMMON;
76 memset(m_bytes_buffer, 0, BYTES_BUFFER_SIZE);
77 memset(m_buffer, 0, sizeof(buffer_t) * NUM_CHANNELS);
78 }
79
80 inline uint32_t parse_start(const uint8_t* str, int32_t* pos, int32_t len)
81 {
82 assert(nullptr != str);
83 assert(0 == *pos);
84 assert(len > 0);
85 if (m_bytes_left_to_parse > 0)
86 {
87 uint8_t bytes_to_copy = len > 4 ? 4 : len;
88 memcpy(m_bytes_buffer + m_bytes_left_to_parse, str, bytes_to_copy);
89
90 [[maybe_unused]] uint8_t char_len = parse(m_bytes_buffer, pos, m_bytes_left_to_parse + bytes_to_copy);
91 assert(char_len > 0);
92
93 *pos = *pos - m_bytes_left_to_parse;
94 uint32_t rv = m_bytes_left_to_parse;
95 m_bytes_left_to_parse = 0;
96
97 return rv;
98 }
99
100 return 0;
101 }
102
103 inline uint8_t parse(const uint8_t* str, int32_t* pos, int32_t len)
104 {
105 UChar32 uch = 0;
106 int32_t prev_pos = *pos;
107 // * @param s const uint8_t * string
108 // * @param i int32_t string offset, must be i<length
109 // * @param length int32_t string length
110 // * @param c output UChar32 variable, set to <0 in case of an error
111 // uint32_t uuch;
112 // int32_t offset = *pos;
113 // int32_t length = len;
114 // U8_NEXT(str, offset, length, uuch);
115 // U8_NEXT(str, int32_t(*pos), int32_t(len), uuch);
116 // U8_NEXT(str, *pos, len, uint32_t(uch));
117 U8_NEXT(str, *pos, len, uch);
118 // uch = uuch;
119 if (uch < 0)
120 {
121 uint32_t bytes_left = len - prev_pos;
122 if (bytes_left < 4)
123 {
124 m_bytes_left_to_parse = bytes_left;
125 memcpy(m_bytes_buffer, str + prev_pos, m_bytes_left_to_parse);
126 return 0;
127 // incomplete sequence
128 }
129 throw std::runtime_error("Incorrect UTF-8 sequence.");
130 }
131
132 push_value(channel_char, uch);
133
134 push_value(channel_type, u_charType(uch));
135
136 UErrorCode err = U_ZERO_ERROR;
137#ifdef NDEBUG
138 uint8_t script_code = uscript_getScript(uch, &err);
139#else
140 UScriptCode raw_script_code = uscript_getScript(uch, &err);
141 assert(err == U_ZERO_ERROR);
142 assert(raw_script_code >= 0);
143 assert(raw_script_code < USCRIPT_CODE_LIMIT);
144 uint8_t script_code = raw_script_code;
145#endif
146 push_value(channel_script_change, m_prev_script_code == script_code ? 0x00 : 0x01);
147 m_prev_script_code = script_code;
148
149 uint8_t char_len = *pos - prev_pos;
150 assert(char_len > 0);
151 assert(char_len <= 6);
152 push_value(channel_len, char_len);
153
154 return char_len;
155 }
156
157 inline const buffer_t& get_buffer(uint8_t channel_idx) const
158 {
159 assert(channel_idx < NUM_CHANNELS);
160 return m_buffer[channel_idx];
161 }
162
163 inline uint8_t get_len(uint8_t idx) const
164 {
165 return (m_buffer[channel_len] >> (bits_per_position(channel_len) * idx)) & 0x07;
166 }
167
168 inline static uint8_t bits_per_position(uint8_t channel_idx)
169 {
170 assert(channel_idx < NUM_CHANNELS);
171 return bits_per_position_table[channel_idx];
172 }
173};
174
175} // namespace impl
176} // namespace segmentation
177} // namespace deeplima
178
179#endif
uint8_t get_len(uint8_t idx) const
uint32_t parse_start(const uint8_t *str, int32_t *pos, int32_t len)
Definition utf8_reader.h:80
static uint8_t bits_per_position(uint8_t channel_idx)
const buffer_t & get_buffer(uint8_t channel_idx) const
uint8_t parse(const uint8_t *str, int32_t *pos, int32_t len)
const uint8_t bits_per_position_table[]
Definition utf8_reader.h:35