6#ifndef DEEPLIMA_SRC_INFERENCE_EIGEN_BILSTM_H
7#define DEEPLIMA_SRC_INFERENCE_EIGEN_BILSTM_H
11#include <eigen3/Eigen/Dense>
20template<
class M=Eigen::MatrixXf,
class V=Eigen::VectorXf>
46template<
class M=Eigen::MatrixXf,
class V=Eigen::VectorXf>
62 out <<
fw << std::endl;
63 out <<
bw << std::endl;
68template<
class M=Eigen::MatrixXf,
class V=Eigen::VectorXf>
84 layers[0].fw.init_from_double(arg.
fw);
85 layers[0].bw.init_from_double(arg.
bw);
90 for (
size_t i = 0; i <
layers.size(); ++i)
92 out <<
layers[i].fw << std::endl;
93 out <<
layers[i].bw << std::endl;
99template<
class M,
class V,
class T>
108 :
temps(p.layers.size()),
110 fw_h(p.layers.size()),
111 fw_c(p.layers.size()),
112 bw_h(p.layers.size()),
113 bw_c(p.layers.size()),
114 zeros(p.layers.size()),
117 for (
size_t i = 0; i < p.
layers.size(); ++i)
119 uint32_t hidden_size = p.
layers[i].fw.weight_ih.rows() / 4;
120 temps[i] = M::Zero((0 == i && precomputed_input) ? 0 : hidden_size * 8, input_size);
121 outputs[i] = M::Zero(hidden_size * 2, input_size);
122 zeros[i] = V::Zero(hidden_size);
123 fw_h[i] = V::Zero(hidden_size);
124 fw_c[i] = V::Zero(hidden_size);
125 bw_h[i] = V::Zero(hidden_size);
126 bw_c[i] = V::Zero(hidden_size);
147 virtual std::shared_ptr<Op_Base::workbench_t>
create_workbench(uint32_t input_size,
const std::shared_ptr<param_base_t> params,
bool precomputed_input=
false)
const override
149 assert(input_size > 0);
150 assert(
nullptr != params);
151 auto p = std::dynamic_pointer_cast<const params_t>(params);
153 return std::make_shared<workbench_t>(*p, input_size, precomputed_input);
161 virtual void precompute_inputs(
const std::shared_ptr<param_base_t> params,
const M& inputs, M& outputs, int64_t first_column)
163 assert(
nullptr != params);
164 auto p = std::dynamic_pointer_cast<const params_t>(params);
165 const auto& layer = p->layers[0];
167 size_t hidden_size = layer.fw.weight_ih.rows() / 4;
169 auto output_block = outputs.block(0, first_column, outputs.rows(), inputs.cols());
171 output_block.topRows(hidden_size * 4) = (layer.fw.weight_ih * inputs).colwise() + layer.fw.bias_ih;
173 output_block.bottomRows(hidden_size * 4) = (layer.bw.weight_ih * inputs).colwise() + layer.bw.bias_ih;
176 virtual size_t execute(std::shared_ptr<Op_Base::workbench_t> pwb,
177 const M& input_matrix,
178 const std::shared_ptr<param_base_t> params,
186 assert(
nullptr != pwb);
187 assert(
nullptr != params);
188 const auto& p = *std::dynamic_pointer_cast<const params_t>(params);
191 auto wb = std::dynamic_pointer_cast<workbench_t>(pwb);
193 M& output = wb->outputs[0];
197 size_t hidden_size = layer.
fw.
weight_ih.rows() / 4;
201 M input = input_matrix.block(0, input_begin, input_matrix.rows(),
202 input_end - input_begin);
206 M temp = M::Zero(hidden_size * 8, input_end - input_begin);
208 if (wb->m_precomputed_input)
216 temp.topLeftCorner(hidden_size * 4, input_end - input_begin) = (layer.
fw.
weight_ih * input).colwise() + layer.
fw.
bias_ih;
219 temp.bottomLeftCorner(hidden_size * 4, input_end - input_begin) = (layer.
bw.
weight_ih * input).colwise() + layer.
bw.
bias_ih;
224 step_fw(hidden_size, 0, s, g_u, g_o, g_if, c, output);
226 for (
int t = 1; t < input.cols(); t++)
228 s = temp.col(t).topRows(hidden_size * 4) + layer.
fw.
bias_hh + layer.
fw.
weight_hh * output.col(t-1).topRows(hidden_size);
229 step_fw(hidden_size, t, s, g_u, g_o, g_if, c, output);
232 int t = input.cols() - 1;
235 wb->fw_h[0] = output.col(t).topRows(hidden_size);
241 step_bw(hidden_size, t, s, g_u, g_o, g_if, c, output);
243 for (t = input.cols() - 2; t >= 0; t--)
245 s = temp.col(t).bottomRows(hidden_size * 4) + layer.
bw.
bias_hh + layer.
bw.
weight_hh * output.col(t+1).bottomRows(hidden_size);
246 step_bw(hidden_size, t, s, g_u, g_o, g_if, c, output);
249 wb->bw_h[0] = output.col(0).bottomRows(hidden_size);
259 virtual size_t execute(std::shared_ptr<Op_Base::workbench_t> pwb,
260 const M& input_matrix,
261 const std::shared_ptr<param_base_t> params,
262 const size_t input_begin,
263 const size_t input_end)
265 assert(
nullptr != pwb);
266 assert(
nullptr != params);
267 const auto& p = *std::dynamic_pointer_cast<const params_t>(params);
268 auto wb = std::dynamic_pointer_cast<workbench_t>(pwb);
270 for (
size_t i = 0; i < p.layers.size(); ++i)
273 M& temp = wb->temps[i];
274 M& output = wb->outputs[i];
277 size_t hidden_size = layer.
fw.
weight_ih.rows() / 4;
278 V c = V::Zero(hidden_size);
279 const V& zero = wb->zeros[i];
281 if (0 == i && wb->m_precomputed_input)
287 temp = input_matrix.block(0, input_begin, input_matrix.rows(),
288 input_end - input_begin);
301 const M input = input_matrix.block(0, input_begin, input_matrix.rows(),
302 input_end - input_begin);
304 temp.topLeftCorner(hidden_size * 4, input_end - input_begin)
309 temp.bottomLeftCorner(hidden_size * 4, input_end - input_begin)
317 temp.topLeftCorner(hidden_size * 4, input_end - input_begin)
318 = (layer.
fw.
weight_ih * wb->outputs[i-1].leftCols(input_end - input_begin)).colwise()
321 temp.bottomLeftCorner(hidden_size * 4, input_end - input_begin)
322 = (layer.
bw.
weight_ih * wb->outputs[i-1].leftCols(input_end - input_begin)).colwise()
328 forward_pass(hidden_size, layer.
fw, temp, s, g_u, g_o, g_if, c, output, zero, 0, input_end - input_begin);
330 wb->fw_h[i] = output.col(input_end - input_begin - 1).topRows(hidden_size);
333 c = V::Zero(hidden_size);
334 backward_pass(hidden_size, layer.
bw, temp, s, g_u, g_o, g_if, c, output, zero, 0, input_end - input_begin);
340 wb->bw_h[i] = output.col(0).bottomRows(hidden_size);
363 s = input.col(begin).topRows(hidden_size * 4) + fw.
bias_hh + fw.
weight_hh * initial_h;
364 step_fw(hidden_size, 0, s, g_u, g_o, g_if, c, output);
366 for (
auto t = begin + 1; t < end; t++)
368 s = input.col(t).topRows(hidden_size * 4) + fw.
bias_hh + fw.
weight_hh * output.col(t-1).topRows(hidden_size);
369 step_fw(hidden_size, t, s, g_u, g_o, g_if, c, output);
389 s = input.col(t).bottomRows(hidden_size * 4) + bw.
bias_hh + bw.
weight_hh * initial_h;
390 step_bw(hidden_size, t, s, g_u, g_o, g_if, c, output);
392 for (t = end - 2; t >= begin; t--)
394 s = input.col(t).bottomRows(hidden_size * 4) + bw.
bias_hh + bw.
weight_hh * output.col(t+1).bottomRows(hidden_size);
395 step_bw(hidden_size, t, s, g_u, g_o, g_if, c, output);
410 g_if = 1 / (1 + Eigen::exp( 0 - s.segment(0, hidden_size * 2).array() ) );
411 g_u = 2 / (1 + Eigen::exp( 0 - 2 * s.segment(hidden_size * 2, hidden_size).array() ) ) - 1;
412 g_o = 1 / (1 + Eigen::exp( 0 - s.segment(hidden_size * 3, hidden_size).array() ) );
414 c = g_if.segment(0, hidden_size).cwiseProduct(g_u) + g_if.segment(hidden_size, hidden_size).cwiseProduct(c);
416 output.col(t).topRows(hidden_size) = g_o.cwiseProduct(c.unaryExpr( [](
float x) { return my_tanh(x); } ));
430 g_if = 1 / (1 + Eigen::exp( 0 - s.segment(0, hidden_size * 2).array() ) );
431 g_u = 2 / (1 + Eigen::exp( 0 - 2 * s.segment(hidden_size * 2, hidden_size).array() ) ) - 1;
432 g_o = 1 / (1 + Eigen::exp( 0 - s.segment(hidden_size * 3, hidden_size).array() ) );
434 c = g_if.segment(0, hidden_size).cwiseProduct(g_u) + g_if.segment(hidden_size, hidden_size).cwiseProduct(c);
436 output.col(t).bottomRows(hidden_size) = g_o.cwiseProduct(c.unaryExpr( [](
float x) { return my_tanh(x); } ));
void step_fw(size_t hidden_size, size_t t, const V &s, V &g_u, V &g_o, V &g_if, V &c, M &output)
static float my_tanh(float x)
void forward_pass(size_t hidden_size, const params_lstm_t< M, V > &fw, M &input, V &s, V &g_u, V &g_o, V &g_if, V &c, M &output, const V &initial_h, int begin, int end)
virtual std::shared_ptr< Op_Base::workbench_t > create_workbench(uint32_t input_size, const std::shared_ptr< param_base_t > params, bool precomputed_input=false) const override
virtual size_t execute(std::shared_ptr< Op_Base::workbench_t > pwb, const M &input_matrix, const std::shared_ptr< param_base_t > params, const size_t input_begin, const size_t input_end)
virtual bool supports_precomputing() const
params_multilayer_bilstm_t< M, V > params_t
void step_bw(size_t hidden_size, size_t t, const V &s, V &g_u, V &g_o, V &g_if, V &c, M &output)
virtual size_t execute(std::shared_ptr< Op_Base::workbench_t > pwb, const M &input_matrix, const std::shared_ptr< param_base_t > params, size_t input_begin, size_t input_end, Vector &fw_h, Vector &fw_c, Vector &bw_h, Vector &bw_c)
void backward_pass(size_t hidden_size, const params_lstm_t< M, V > &bw, M &input, V &s, V &g_u, V &g_o, V &g_if, V &c, M &output, const V &initial_h, int begin, int end)
virtual void precompute_inputs(const std::shared_ptr< param_base_t > params, const M &inputs, M &outputs, int64_t first_column)
const M & get_last_output()
workbench_t(const params_t &p, uint32_t input_size, bool precomputed_input)
std::ostream & operator<<(std::ostream &out) const
void init_from_double(const params_bilstm_t< Eigen::MatrixXd, Eigen::VectorXd > &arg)
std::ostream & operator<<(std::ostream &out) const
void init_from_double(const params_lstm_t< Eigen::MatrixXd, Eigen::VectorXd > &arg)
std::ostream & operator<<(std::ostream &out) const
void init_from_double(const params_bilstm_t< Eigen::MatrixXd, Eigen::VectorXd > &arg)
params_multilayer_bilstm_t(const std::vector< layer_params_t > &arg)
params_multilayer_bilstm_t()
params_bilstm_t< M, V > layer_params_t
std::vector< layer_params_t > layers