6#ifndef DEEPLIMA_SEGMENTATION_IMPL_H
7#define DEEPLIMA_SEGMENTATION_IMPL_H
34 typedef std::function < bool (uint8_t* buffer,
83 virtual void load(
const std::string& fn);
85 void init(
size_t threads,
size_t buffer_size_per_thread);
100 const auto& dicts = this->get_output_str_dicts();
virtual void finalize()=0
Cleanup all remaining locks if any.
virtual void register_handler(const segmentation_callback_t fn)=0
std::function< bool(uint8_t *buffer, int32_t &read, int32_t max) > read_callback_t
virtual void parse_from_stream(const read_callback_t fn)=0
virtual bool predicts_mwt() const
Whether this segmenter predicts multiword-token surfaces (sets the token_flags_t::multiword flag).
The implementation of the segmenter, a SegmentationClassifier, itself a RnnSequenceClassifier.
std::vector< uint8_t > m_char_len
locked_buffer_set_t m_buff_set
void send_results(int32_t slot_idx)
InputEncoder m_input_encoder
virtual void parse_from_stream(const read_callback_t fn) override
void vectorize_timepoint(uint64_t timepoint)
virtual ~SegmentationImpl()=default
virtual bool predicts_mwt() const override
MWT-aware iff the model has more than the base segmentation tag classes (i.e.
virtual void register_handler(const segmentation_callback_t fn) override
uint32_t m_current_slot_timepoints
virtual void finalize() override
Cleanup all remaining locks if any.
int32_t m_current_slot_no
void increment_timepoint(uint64_t &timepoint)
void init(size_t threads, size_t buffer_size_per_thread)
virtual void load(const std::string &fn)
uint64_t m_current_timepoint
int32_t m_last_completed_slot
DictEmbdVectorizer< EmbdUInt64FloatHolder, EmbdUInt64Float, eigen_wrp::EigenMatrixXf > EmbdVectorizer
impl::SegmentationInferenceWrapper< BiRnnEigenInferenceForSegmentation > Model
CharNgramEncoder< Utf8Reader<> > CharNgramEncoderFromUtf8
std::function< void(const std::vector< token_pos > &tokens, uint32_t len) > segmentation_callback_t