From 77189825094729067123b6ecc4d0317fb2f5047a Mon Sep 17 00:00:00 2001 From: Lin Yuan Date: Tue, 18 Sep 2018 16:56:59 -0700 Subject: [PATCH 01/17] Implement ctc_loss as a normal operator --- src/operator/nn/ctc_loss-inl.h | 386 +++++++++++++++++++++++++++++++++ src/operator/nn/ctc_loss.cc | 126 +++++++++++ src/operator/nn/ctc_loss.cu | 36 +++ 3 files changed, 548 insertions(+) create mode 100644 src/operator/nn/ctc_loss-inl.h create mode 100644 src/operator/nn/ctc_loss.cc create mode 100644 src/operator/nn/ctc_loss.cu diff --git a/src/operator/nn/ctc_loss-inl.h b/src/operator/nn/ctc_loss-inl.h new file mode 100644 index 000000000000..f855efe144ff --- /dev/null +++ b/src/operator/nn/ctc_loss-inl.h @@ -0,0 +1,386 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +/*! + * Copyright (c) 2017 by Contributors + * \file ctc_loss-inl.h + * \brief +*/ + +#ifndef MXNET_OPERATOR_CTC_LOSS_INL_H_ +#define MXNET_OPERATOR_CTC_LOSS_INL_H_ + +#include +#include +#include "./sequence_mask-inl.h" +#include "../sequence_op_common.h" +#include "../operator_common.h" +#include "../elemwise_op_common.h" + +namespace mxnet { +namespace op { + +namespace ctc_loss { +enum CTCLossOpInputs { kData, kLabel }; +enum CTCLossOpOutputs { kOut, kGrad }; +enum CTCLossOpForwardResource { kTempSpace }; +} + +template +inline void get_workspace_size(std::vector *label_lengths, + std::vector *data_lengths, + int alphabet_size, int minibatch, bool gpu, + size_t *size_bytes) { + // This is the max of all S and T for all examples in the minibatch. + int maxL = *std::max_element(label_lengths->data(), + label_lengths->data() + minibatch); + int maxT = *std::max_element(data_lengths->data(), + data_lengths->data() + minibatch); + + const int S = 2 * maxL + 1; + + *size_bytes = 0; + + if (gpu) { + // GPU storage + // nll_forward, nll_backward + *size_bytes += 2 * sizeof(T) * minibatch; + + // repeats + *size_bytes += sizeof(int) * minibatch; + + // label offsets + *size_bytes += sizeof(int) * minibatch; + + // utt_length + *size_bytes += sizeof(int) * minibatch; + + // label lengths + *size_bytes += sizeof(int) * minibatch; + + // labels without blanks - overallocate for now + *size_bytes += sizeof(int) * maxL * minibatch; + + // labels with blanks + *size_bytes += sizeof(int) * S * minibatch; + + // alphas + *size_bytes += sizeof(T) * S * maxT * minibatch; + + // denoms + *size_bytes += sizeof(T) * maxT * minibatch; + + // probs (since we will pass in activations) + *size_bytes += sizeof(T) * alphabet_size * maxT * minibatch; + + } else { + // cpu can eventually replace all minibatch with + // max number of concurrent threads if memory is + // really tight + + // per minibatch memory + size_t per_minibatch_bytes = 0; + + // output + per_minibatch_bytes += sizeof(T) * alphabet_size; + + // alphas + per_minibatch_bytes += sizeof(T) * S * maxT; + + // betas + per_minibatch_bytes += sizeof(T) * S; + + // labels w/blanks, e_inc, s_inc + per_minibatch_bytes += 3 * sizeof(int) * S; + + *size_bytes = per_minibatch_bytes * minibatch; + + // probs + *size_bytes += sizeof(T) * alphabet_size * maxT * minibatch; + } +} + +// Takes a tensor of labels, and interprets 0-elements at the end of the vector +// as padding. The tensor is packed into an std::vector without padding +// characters. The label sequence lengths are also inferred from the padding chars. +// When cudnn is enabled, the return value signifies whether the cudnn length limit is exceeded. +template +inline bool LabelTensorToPackedVector(mshadow::Tensor labels, + int padding_mask, + std::vector *packed_labels, + std::vector *label_lengths) { + int batch = labels.size(0); + int max_num_labels = labels.size(1); + bool exceed_limit = false; + + std::vector cpu_labels(max_num_labels*batch); + mshadow::Tensor flat_labels = labels.FlatTo1D(); + IndexTensorToVector(flat_labels, &cpu_labels); + + for (int b = 0; b < batch; ++b) { + auto start = cpu_labels.data()+b*max_num_labels; + auto res = std::find(start, start+max_num_labels, padding_mask); + int len = std::distance(start, res); +#if defined(__CUDACC__) && MXNET_USE_CUDNN == 1 && CUDNN_MAJOR >= 7 + exceed_limit = exceed_limit || len > CUDNN_LABEL_LENGTH_LIMIT; +#endif + std::copy(start, start + len, + std::back_inserter(*packed_labels)); + label_lengths->at(b) = len; + } + return exceed_limit; +} + +// Takes a tensor of labels, and a vector which specifies the actual length of each label +// The tensor is packed into an std::vector without padding characters. +// The label length vector is copied into an std::vector. +// When cudnn is enabled, the return value signifies whether the cudnn length limit is exceeded. +template +inline bool PackLabelByLength(mshadow::Tensor labels, + mshadow::Tensor in_label_lengths, + std::vector *packed_labels, + std::vector *label_lengths) { + int batch = labels.size(0); + int max_num_labels = labels.size(1); + bool exceed_limit = false; + + IndexTensorToVector(in_label_lengths, label_lengths); + + std::vector cpu_labels(max_num_labels*batch); + mshadow::Tensor flat_labels = labels.FlatTo1D(); + IndexTensorToVector(flat_labels, &cpu_labels); + + for (int b = 0; b < batch; ++b) { + auto start = cpu_labels.data()+b*max_num_labels; + int len = label_lengths->at(b); +#if defined(__CUDACC__) && MXNET_USE_CUDNN == 1 && CUDNN_MAJOR >= 7 + exceed_limit = exceed_limit || len > CUDNN_LABEL_LENGTH_LIMIT; +#endif + std::copy(start, start + len, + std::back_inserter(*packed_labels)); + } + return exceed_limit; +} + +struct CTCLossOpParam : public dmlc::Parameter { + bool use_data_lengths; + bool use_label_lengths; + int blank_label; + DMLC_DECLARE_PARAMETER(CTCLossOpParam) { + DMLC_DECLARE_FIELD(use_data_lengths).set_default(false) + .describe("Whether the data lenghts are decided by `data_lengths`. " + "If false, the lengths are equal to the max sequence length."); + DMLC_DECLARE_FIELD(use_label_lengths).set_default(false) + .describe("Whether the label lenghts are decided by " + "`label_lengths`, or derived from `padding_mask`. " + "If false, the lengths are derived from the " + "first occurrence of the value of `padding_mask`. " + "The value of `padding_mask` is ``0`` when first CTC label is reserved for blank, " + "and ``-1`` when last label is reserved for blank. See `blank_label`."); + DMLC_DECLARE_FIELD(blank_label) + .add_enum("first", 0) + .add_enum("last", 1) + .set_default(0) + .describe("Set the label that is reserved for blank label." + "If \"first\", 0-th label is reserved, and " + "label values for tokens in the vocabulary are " + "between ``1`` and ``alphabet_size-1``, and the padding mask is ``-1``. " + "If \"last\", last label value ``alphabet_size-1`` " + "is reserved for blank label instead, " + "and label values for tokens in the vocabulary are " + "between ``0`` and ``alphabet_size-2``, and the padding mask is ``0``."); + } +}; + +inline bool CTCLossOpShape(const nnvm::NodeAttrs &attrs, + std::vector* in_attrs, + std::vector* out_attrs) { + CHECK_EQ(in_attrs->size(), 2U); + CHECK_EQ(out_attrs->size(), 2U); + + const TShape &dshape = (*in_attrs)[ctc_loss::kData]; + const TShape &lshape = (*in_attrs)[ctc_loss::kLabel]; + CHECK_EQ(dshape.ndim(), 3U) << "The data array must be of rank 3."; + CHECK_EQ(lshape.ndim(), 2U) << "The labels array must be of rank 2."; + CHECK_EQ(dshape[1], lshape[0]) + << "The batch size for the labels and data arrays must be the same."; + + CHECK_GE(dshape[0], lshape[1]) << "The max number of labels cannot exceed " + "the maximum sequence length of the " + "data."; + + TShape oshape(1); + oshape[0] = dshape[1]; // batch size + out_attrs->clear(); + out_attrs->push_back(oshape); + out_attrs->push_back(dshape); + //SHAPE_ASSIGN_CHECK(*out_attrs, 0, oshape); // forward output + //SHAPE_ASSIGN_CHECK(*out_attrs, 1, dshape); // grad output + return true; +} + +inline bool CTCLossOpType(const nnvm::NodeAttrs& attrs, + std::vector* in_attrs, + std::vector* out_attrs) { + CHECK_EQ(in_attrs->size(), 2U); + CHECK_EQ(out_attrs->size(), 2U); + int dtype = (*in_attrs)[ctc_loss::kData]; + CHECK_NE(dtype, -1) << "Input data must have specified type"; + + out_attrs->clear(); + out_attrs->push_back(dtype); + out_attrs->push_back(dtype); + //TYPE_ASSIGN_CHECK(*out_attrs, 0, dtype); // forward output + //TYPE_ASSIGN_CHECK(*out_attrs, 1, dtype); // grad output + return true; +} + +inline bool CTCLossOpStorageType(const nnvm::NodeAttrs& attrs, + const int dev_mask, + DispatchMode* dispatch_mode, + std::vector* in_attrs, + std::vector* out_attrs) { + CHECK_EQ(in_attrs->size(), 2U); + CHECK_EQ(out_attrs->size(), 2U); + const int in_stype = in_attrs->at(0); + bool dispatched = false; + if (!dispatched && in_stype == kDefaultStorage) { + // dns -> dns + dispatched = storage_type_assign(out_attrs, kDefaultStorage, + dispatch_mode, DispatchMode::kFCompute); + } + if (!dispatched) { + dispatched = dispatch_fallback(out_attrs, dispatch_mode); + } + return dispatched; +} + +inline std::vector CTCLossOpResource(const std::vector &in_shape) { + return std::vector{ResourceRequest::kTempSpace}; +} + +template +void CTCLossOpForward(const nnvm::NodeAttrs& attrs, + const OpContext& ctx, + const std::vector& inputs, + const std::vector& req, + const std::vector& outputs) { + CHECK_EQ(inputs.size(), 2U); + CHECK_EQ(outputs.size(), 2U); + CHECK_EQ(req.size(), 2U); + using namespace mshadow; + using namespace mshadow::expr; + Stream *s = ctx.get_stream(); + const CTCLossOpParam& param = nnvm::get(attrs.parsed); + + MSHADOW_TYPE_SWITCH(inputs[ctc_loss::kLabel].type_flag_, DType, { + Tensor data = + inputs[ctc_loss::kData].get(s); + Tensor labels = + inputs[ctc_loss::kLabel].get(s); + + Tensor costs = + outputs[ctc_loss::kOut].get(s); + Tensor grad = + outputs[ctc_loss::kGrad].get(s); + + int max_seq_len = data.size(0); + int batch_size = data.size(1); + int alphabet_size = data.size(2); + + // data_lengths + std::vector data_lengths(batch_size, max_seq_len); + if (param.use_data_lengths) { + int kInputLength = 2; + IndexTensorToVector(inputs[kInputLength].get(s), &data_lengths); + } + + // label_lengths + std::vector packed_labels; + std::vector label_lengths(batch_size); + + if (param.use_label_lengths) { + int kLabelLength = 2 + param.use_data_lengths; + PackLabelByLength(labels, inputs[kLabelLength].get(s), + &packed_labels, &label_lengths); + } else { + LabelTensorToPackedVector(labels, param.blank_label == 0 ? 0 : -1, + &packed_labels, &label_lengths); + } + + size_t size_bytes; + bool gpu = data.kDevCPU ? false : true; + get_workspace_size(&label_lengths, &data_lengths, alphabet_size, + batch_size, gpu, &size_bytes); + + // round-up so there are enough elems in memory + int num_tmp_elems = (size_bytes + sizeof(real_t) - 1) / sizeof(real_t); + Tensor workspace = + ctx.requested[0].get_space_typed(Shape1(num_tmp_elems), s); + + compute_ctc_cost(data, costs.dptr_, grad.dptr_, packed_labels.data(), + label_lengths.data(), data_lengths.data(), + workspace.dptr_, req[ctc_loss::kGrad] != mxnet::kNullOp, + param.blank_label == 0 ? 0 : (alphabet_size-1)); +/* + baidu_forward(ctx, s, data, costs, grad, + &data_lengths, &label_lengths, &packed_labels, + batch_size, alphabet_size, req[ctc_loss::kGrad] != mxnet::kNullOp); +*/ + if (param.use_data_lengths) { + // baidu warp CTC implementation sometimes includes undefined gradients + // for data outside of length mask. Setting to 0 to make it consistent + // with CPU implementation. + int kInputLength = 2; + mxnet_op::SequenceMask(grad, inputs[kInputLength].get(s), + static_cast(0)); + } + }); +} + +template +void CTCLossOpBackward(const nnvm::NodeAttrs& attrs, + const OpContext& ctx, + const std::vector& inputs, + const std::vector& req, + const std::vector& outputs) { + CHECK_EQ(inputs.size(), 2U); + CHECK_EQ(outputs.size(), 1U); + CHECK_EQ(req.size(), 1U); + using namespace mshadow; + using namespace mxnet_op; + + Stream *s = ctx.get_stream(); + + Tensor data_grad = + inputs[ctc_loss::kData].get(s); + Tensor output_grad = + outputs[ctc_loss::kOut].get(s); + + Tensor data_grad_computed = + outputs[ctc_loss::kGrad].get(s); + + Assign(data_grad, req[ctc_loss::kData], + mshadow::expr::broadcast<1>(output_grad, data_grad.shape_) * data_grad_computed); +} + +} // namespace op +} // namespace mxnet + +#endif // MXNET_OPERATOR_CTC_LOSS_INL_H_ \ No newline at end of file diff --git a/src/operator/nn/ctc_loss.cc b/src/operator/nn/ctc_loss.cc new file mode 100644 index 000000000000..8c80ef85dc8d --- /dev/null +++ b/src/operator/nn/ctc_loss.cc @@ -0,0 +1,126 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +/*! + * \file ctc_loss.cc + * \brief CPU Implementation of CTC Loss op + */ +#include "./ctc_loss-inl.h" +#include "../contrib/ctc_include/detail/cpu_ctc.h" + +namespace mshadow { +template +ctcStatus_t compute_ctc_cost(const Tensor activations, + DType *costs, DType *grads, int *labels, + int *label_lengths, int *data_lengths, + void *workspace, int train, int blank_label) { + int minibatch = static_cast(activations.size(1)); + int alphabet_size = static_cast(activations.size(2)); + mxnet_warpctc::CpuCTC ctc(alphabet_size, minibatch, workspace, blank_label); + if (train) { + return ctc.cost_and_grad(activations.dptr_, grads, costs, labels, + label_lengths, data_lengths); + } else { + return ctc.score_forward(activations.dptr_, costs, labels, label_lengths, + data_lengths); + } +} +} // namespace mshadow + +namespace mxnet { +namespace op { + +DMLC_REGISTER_PARAMETER(CTCLossOpParam); + +NNVM_REGISTER_OP(ctc_loss) +.describe(R"code(Connectionist Temporal Classification Loss. +The shapes of the inputs and outputs: + +- **data**: `(sequence_length, batch_size, alphabet_size)` +- **label**: `(batch_size, label_sequence_length)` +- **out**: `(batch_size)` + +The `data` tensor consists of sequences of activation vectors (without applying softmax), +with i-th channel in the last dimension corresponding to i-th label +for i between 0 and alphabet_size-1 (i.e always 0-indexed). +Alphabet size should include one additional value reserved for blank label. +When `blank_label` is ``"first"``, the ``0``-th channel is be reserved for +activation of blank label, or otherwise if it is "last", ``(alphabet_size-1)``-th channel should be +reserved for blank label. + +``label`` is an index matrix of integers. When `blank_label` is ``"first"``, +the value 0 is then reserved for blank label, and should not be passed in this matrix. Otherwise, +when `blank_label` is ``"last"``, the value `(alphabet_size-1)` is reserved for blank label. + +If a sequence of labels is shorter than *label_sequence_length*, use the special +padding value at the end of the sequence to conform it to the correct +length. The padding value is `0` when `blank_label` is ``"first"``, and `-1` otherwise. + +For example, suppose the vocabulary is `[a, b, c]`, and in one batch we have three sequences +'ba', 'cbb', and 'abac'. When `blank_label` is ``"first"``, we can index the labels as +`{'a': 1, 'b': 2, 'c': 3}`, and we reserve the 0-th channel for blank label in data tensor. +The resulting `label` tensor should be padded to be:: + + [[2, 1, 0, 0], [3, 2, 2, 0], [1, 2, 1, 3]] + +When `blank_label` is ``"last"``, we can index the labels as +`{'a': 0, 'b': 1, 'c': 2}`, and we reserve the channel index 3 for blank label in data tensor. +The resulting `label` tensor should be padded to be:: + + [[1, 0, -1, -1], [2, 1, 1, -1], [0, 1, 0, 2]] + +``out`` is a list of CTC loss values, one per example in the batch. + +See *Connectionist Temporal Classification: Labelling Unsegmented +Sequence Data with Recurrent Neural Networks*, A. Graves *et al*. for more +information on the definition and the algorithm. + +)code" ADD_FILELINE) +.set_attr_parser(ParamParser) +.set_num_inputs(2) +.set_num_outputs(2) +.set_attr("FListInputNames", + [](const NodeAttrs& attrs) { + return std::vector{"data", "label"}; + }) +.set_attr("FListOutputNAmes", + [](const NodeAttrs& attrs) { + return std::vector{"out", "grad"}; + }) +.set_attr("FInferShape", CTCLossOpShape) +.set_attr("FInferType", CTCLossOpType) +.set_attr("FInferStorageType", CTCLossOpStorageType) +.set_attr("FResourceRequest", [](const NodeAttrs& attrs) + { return std::vector{ResourceRequest::kTempSpace}; }) +//.set_attr("FResourceRequest", CTCLossOpResource) +.set_attr("FCompute", CTCLossOpForward) +.set_attr("FGradient", ElemwiseGradUseIn{"_backward_ctc_loss"}) +.add_argument("data", "NDArray-or-Symbol", "Input ndarray") +.add_argument("label", "NDArray-or-Symbol", "Ground-truth labels for the loss.") +.add_arguments(CTCLossOpParam::__FIELDS__()); + +NNVM_REGISTER_OP(_backward_ctc_loss) +.set_attr_parser(ParamParser) +.set_num_inputs(2) +.set_num_outputs(1) +.set_attr("TIsBackward", true) +.set_attr("FCompute", CTCLossOpBackward); + +} // namespace op +} // namespace mxnet \ No newline at end of file diff --git a/src/operator/nn/ctc_loss.cu b/src/operator/nn/ctc_loss.cu new file mode 100644 index 000000000000..5ab7d17b85ab --- /dev/null +++ b/src/operator/nn/ctc_loss.cu @@ -0,0 +1,36 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +/*! + * \file ctc_loss.cu + * \brief GPU Implementation of ctc_loss op + */ + #include "./ctc_loss-inl.h" + + namespace mxnet { + namespace op { + + NNVM_REGISTER_OP(ctc_loss) + .set_attr("FCompute", CTCLossOpForward); + + NNVM_REGISTER_OP(_backward_ctc_loss) + .set_attr("FCompute", CTCLossOpBackward); + + } // namespace op + } // namespace mxnet \ No newline at end of file From 3d7ca6868075e9bfcda4bf9b3f3160d5f4cd9bd0 Mon Sep 17 00:00:00 2001 From: Lin Yuan Date: Tue, 18 Sep 2018 21:28:22 -0700 Subject: [PATCH 02/17] Update unit test --- src/operator/nn/ctc_loss-inl.h | 14 ++++---------- tests/python/unittest/test_operator.py | 8 ++++---- 2 files changed, 8 insertions(+), 14 deletions(-) diff --git a/src/operator/nn/ctc_loss-inl.h b/src/operator/nn/ctc_loss-inl.h index f855efe144ff..b731df77f12e 100644 --- a/src/operator/nn/ctc_loss-inl.h +++ b/src/operator/nn/ctc_loss-inl.h @@ -227,11 +227,8 @@ inline bool CTCLossOpShape(const nnvm::NodeAttrs &attrs, TShape oshape(1); oshape[0] = dshape[1]; // batch size - out_attrs->clear(); - out_attrs->push_back(oshape); - out_attrs->push_back(dshape); - //SHAPE_ASSIGN_CHECK(*out_attrs, 0, oshape); // forward output - //SHAPE_ASSIGN_CHECK(*out_attrs, 1, dshape); // grad output + SHAPE_ASSIGN_CHECK(*out_attrs, 0, oshape); // forward output + SHAPE_ASSIGN_CHECK(*out_attrs, 1, dshape); // grad output return true; } @@ -243,11 +240,8 @@ inline bool CTCLossOpType(const nnvm::NodeAttrs& attrs, int dtype = (*in_attrs)[ctc_loss::kData]; CHECK_NE(dtype, -1) << "Input data must have specified type"; - out_attrs->clear(); - out_attrs->push_back(dtype); - out_attrs->push_back(dtype); - //TYPE_ASSIGN_CHECK(*out_attrs, 0, dtype); // forward output - //TYPE_ASSIGN_CHECK(*out_attrs, 1, dtype); // grad output + TYPE_ASSIGN_CHECK(*out_attrs, 0, in_attrs->at(0)); // forward output + TYPE_ASSIGN_CHECK(*out_attrs, 1, in_attrs->at(0)); // grad output return true; } diff --git a/tests/python/unittest/test_operator.py b/tests/python/unittest/test_operator.py index 2bf7e848850a..fd3a8ab2542f 100644 --- a/tests/python/unittest/test_operator.py +++ b/tests/python/unittest/test_operator.py @@ -4339,7 +4339,7 @@ def test_invalid_shape(): x = mx.sym.Variable('x') y = mx.sym.Variable('y') where_sym = mx.sym.where(condition, x, y) - + assert_exception(lambda: where_sym.eval(x=mx.nd.array([[2,3],[4,5],[6,7]]), y=mx.nd.array([[8,9],[10,11],[12,13]]), condition=mx.nd.array([1,0])), MXNetError) @@ -4478,7 +4478,7 @@ def test_pick_helper(index_type=np.int32): def check_ctc_loss(acts, labels, loss_truth): in_var = mx.sym.Variable('input') labels_var = mx.sym.Variable('labels') - ctc = mx.sym.contrib.ctc_loss(in_var, labels_var) + ctc = mx.sym.ctc_loss(in_var, labels_var) acts_nd = mx.nd.array(acts, ctx=default_context()) labels_nd = mx.nd.array(labels, ctx=default_context()) exe = ctc.bind(ctx=default_context(), args=[acts_nd, labels_nd]) @@ -4537,7 +4537,7 @@ def test_ctc_loss_with_large_classes(): [1000, 2000, 3000, 4000, 0, 5000, 0, 0]], dtype=np.int32) nd_data = mx.nd.array(data) nd_label = mx.nd.array(label) - loss = mx.nd.contrib.ctc_loss(data=nd_data, label=nd_label) + loss = mx.nd.ctc_loss(data=nd_data, label=nd_label) expected_loss = np.array([688.02826, 145.34462]) assert_almost_equal(loss.asnumpy(), expected_loss) @@ -4982,7 +4982,7 @@ def _validate_sample_location(input_rois, input_offset, spatial_scale, pooled_w, trans_x = input_offset[roi_idx, class_id * 2, part_h, part_w] * trans_std trans_y = input_offset[roi_idx, class_id * 2 + 1, part_h, part_w] * trans_std bin_h_start, bin_w_start = ph * bin_size_h + roi_start_h, pw * bin_size_w + roi_start_w - + need_check = True while need_check: pass_check = True From d4194c9df2dd99d338bf9b9c343c3b6a9c340f9e Mon Sep 17 00:00:00 2001 From: Lin Yuan Date: Fri, 21 Sep 2018 10:51:58 -0700 Subject: [PATCH 03/17] Update unit test and fix bug in backward --- src/operator/nn/ctc_loss-inl.h | 71 ++++++++++++++++++-------- src/operator/nn/ctc_loss.cc | 25 +++++---- tests/python/unittest/test_operator.py | 14 ++--- 3 files changed, 73 insertions(+), 37 deletions(-) diff --git a/src/operator/nn/ctc_loss-inl.h b/src/operator/nn/ctc_loss-inl.h index b731df77f12e..03fe3a791287 100644 --- a/src/operator/nn/ctc_loss-inl.h +++ b/src/operator/nn/ctc_loss-inl.h @@ -211,7 +211,9 @@ struct CTCLossOpParam : public dmlc::Parameter { inline bool CTCLossOpShape(const nnvm::NodeAttrs &attrs, std::vector* in_attrs, std::vector* out_attrs) { - CHECK_EQ(in_attrs->size(), 2U); + + const CTCLossOpParam& param = nnvm::get(attrs.parsed); + CHECK_EQ(in_attrs->size(), 2U + param.use_data_lengths + param.use_label_lengths); CHECK_EQ(out_attrs->size(), 2U); const TShape &dshape = (*in_attrs)[ctc_loss::kData]; @@ -221,6 +223,20 @@ inline bool CTCLossOpShape(const nnvm::NodeAttrs &attrs, CHECK_EQ(dshape[1], lshape[0]) << "The batch size for the labels and data arrays must be the same."; + if (param.use_data_lengths) { + int kInputLength = 2; + const TShape &dlshape = (*in_attrs)[kInputLength]; + CHECK_EQ(dlshape.ndim(), 1U) << "Data length array must be a vector."; + CHECK_EQ(dlshape[0], dshape[1]) + << "The batch size for the data and data lengths must be the same."; + } + if (param.use_label_lengths) { + int kLabelLength = 2 + param.use_data_lengths; + const TShape &llshape = (*in_attrs)[kLabelLength]; + CHECK_EQ(llshape.ndim(), 1U) << "Label length array must be a vector."; + CHECK_EQ(llshape[0], lshape[0]) + << "The batch size for the labels and label lengths must be the same."; + } CHECK_GE(dshape[0], lshape[1]) << "The max number of labels cannot exceed " "the maximum sequence length of the " "data."; @@ -235,7 +251,7 @@ inline bool CTCLossOpShape(const nnvm::NodeAttrs &attrs, inline bool CTCLossOpType(const nnvm::NodeAttrs& attrs, std::vector* in_attrs, std::vector* out_attrs) { - CHECK_EQ(in_attrs->size(), 2U); + CHECK_GE(in_attrs->size(), 2U); CHECK_EQ(out_attrs->size(), 2U); int dtype = (*in_attrs)[ctc_loss::kData]; CHECK_NE(dtype, -1) << "Input data must have specified type"; @@ -250,7 +266,7 @@ inline bool CTCLossOpStorageType(const nnvm::NodeAttrs& attrs, DispatchMode* dispatch_mode, std::vector* in_attrs, std::vector* out_attrs) { - CHECK_EQ(in_attrs->size(), 2U); + CHECK_GE(in_attrs->size(), 2U); CHECK_EQ(out_attrs->size(), 2U); const int in_stype = in_attrs->at(0); bool dispatched = false; @@ -269,20 +285,39 @@ inline std::vector CTCLossOpResource(const std::vector return std::vector{ResourceRequest::kTempSpace}; } +inline int CTCLossOpNumInputs(const NodeAttrs& attrs) { + const CTCLossOpParam& param = nnvm::get(attrs.parsed); + return 2 + param.use_data_lengths + param.use_label_lengths; +} + +inline std::vector CTCLossOpListInputNames(const NodeAttrs& attrs) { + const CTCLossOpParam& param = nnvm::get(attrs.parsed); + if (param.use_data_lengths && param.use_label_lengths) { + return {"data", "label", "data_lengths", "label_lengths"}; + } else if (param.use_data_lengths) { + return {"data", "label", "data_lengths"}; + } else if (param.use_label_lengths) { + return {"data", "label", "label_lengths"}; + } else { + return {"data", "label"}; + } +} + template void CTCLossOpForward(const nnvm::NodeAttrs& attrs, const OpContext& ctx, const std::vector& inputs, const std::vector& req, const std::vector& outputs) { - CHECK_EQ(inputs.size(), 2U); - CHECK_EQ(outputs.size(), 2U); - CHECK_EQ(req.size(), 2U); using namespace mshadow; using namespace mshadow::expr; - Stream *s = ctx.get_stream(); + const CTCLossOpParam& param = nnvm::get(attrs.parsed); - + CHECK_EQ(inputs.size(), 2U + param.use_data_lengths + param.use_label_lengths); + CHECK_EQ(outputs.size(), 2U); + CHECK_EQ(req.size(), 2U); + + Stream *s = ctx.get_stream(); MSHADOW_TYPE_SWITCH(inputs[ctc_loss::kLabel].type_flag_, DType, { Tensor data = inputs[ctc_loss::kData].get(s); @@ -332,11 +367,7 @@ void CTCLossOpForward(const nnvm::NodeAttrs& attrs, label_lengths.data(), data_lengths.data(), workspace.dptr_, req[ctc_loss::kGrad] != mxnet::kNullOp, param.blank_label == 0 ? 0 : (alphabet_size-1)); -/* - baidu_forward(ctx, s, data, costs, grad, - &data_lengths, &label_lengths, &packed_labels, - batch_size, alphabet_size, req[ctc_loss::kGrad] != mxnet::kNullOp); -*/ + if (param.use_data_lengths) { // baidu warp CTC implementation sometimes includes undefined gradients // for data outside of length mask. Setting to 0 to make it consistent @@ -354,23 +385,21 @@ void CTCLossOpBackward(const nnvm::NodeAttrs& attrs, const std::vector& inputs, const std::vector& req, const std::vector& outputs) { - CHECK_EQ(inputs.size(), 2U); - CHECK_EQ(outputs.size(), 1U); - CHECK_EQ(req.size(), 1U); using namespace mshadow; using namespace mxnet_op; Stream *s = ctx.get_stream(); Tensor data_grad = - inputs[ctc_loss::kData].get(s); + outputs[0].get(s); + Tensor output_grad = - outputs[ctc_loss::kOut].get(s); + inputs[0].get(s); Tensor data_grad_computed = - outputs[ctc_loss::kGrad].get(s); - - Assign(data_grad, req[ctc_loss::kData], + inputs[3].get(s); + + Assign(data_grad, req[0], mshadow::expr::broadcast<1>(output_grad, data_grad.shape_) * data_grad_computed); } diff --git a/src/operator/nn/ctc_loss.cc b/src/operator/nn/ctc_loss.cc index 8c80ef85dc8d..e56810d12512 100644 --- a/src/operator/nn/ctc_loss.cc +++ b/src/operator/nn/ctc_loss.cc @@ -93,16 +93,17 @@ information on the definition and the algorithm. )code" ADD_FILELINE) .set_attr_parser(ParamParser) -.set_num_inputs(2) +.set_num_inputs(CTCLossOpNumInputs) .set_num_outputs(2) -.set_attr("FListInputNames", - [](const NodeAttrs& attrs) { - return std::vector{"data", "label"}; - }) +.set_attr("FListInputNames", CTCLossOpListInputNames) .set_attr("FListOutputNAmes", [](const NodeAttrs& attrs) { return std::vector{"out", "grad"}; }) +.set_attr("FNumVisibleOutputs", + [](const NodeAttrs& attrs) { + return 1; + }) .set_attr("FInferShape", CTCLossOpShape) .set_attr("FInferType", CTCLossOpType) .set_attr("FInferStorageType", CTCLossOpStorageType) @@ -110,17 +111,23 @@ information on the definition and the algorithm. { return std::vector{ResourceRequest::kTempSpace}; }) //.set_attr("FResourceRequest", CTCLossOpResource) .set_attr("FCompute", CTCLossOpForward) -.set_attr("FGradient", ElemwiseGradUseIn{"_backward_ctc_loss"}) +.set_attr("FGradient", ElemwiseGradUseOut{"_backward_ctc_loss"}) .add_argument("data", "NDArray-or-Symbol", "Input ndarray") .add_argument("label", "NDArray-or-Symbol", "Ground-truth labels for the loss.") +.add_argument("data_lengths", "NDArray-or-Symbol", + "Lengths of data for each of the samples. Only required " + "when use_data_lengths is true.") +.add_argument("label_lengths", "NDArray-or-Symbol", + "Lengths of labels for each of the samples. Only required " + "when use_label_lengths is true.") .add_arguments(CTCLossOpParam::__FIELDS__()); NNVM_REGISTER_OP(_backward_ctc_loss) .set_attr_parser(ParamParser) -.set_num_inputs(2) -.set_num_outputs(1) +.set_num_inputs(1) +.set_num_outputs(CTCLossOpNumInputs) .set_attr("TIsBackward", true) -.set_attr("FCompute", CTCLossOpBackward); +.set_attr("FCompute", CTCLossOpBackward); } // namespace op } // namespace mxnet \ No newline at end of file diff --git a/tests/python/unittest/test_operator.py b/tests/python/unittest/test_operator.py index 0ef1432aa79c..6e382ebd9b96 100644 --- a/tests/python/unittest/test_operator.py +++ b/tests/python/unittest/test_operator.py @@ -4482,7 +4482,7 @@ def check_ctc_loss(acts, labels, loss_truth): acts_nd = mx.nd.array(acts, ctx=default_context()) labels_nd = mx.nd.array(labels, ctx=default_context()) exe = ctc.bind(ctx=default_context(), args=[acts_nd, labels_nd]) - # test forward without grad calc + # test forward with grad calc exe.forward(is_train=True) outTest = exe.outputs[0] # test forward without grad calc @@ -4611,12 +4611,12 @@ def check_ctc_loss_grad(blank_label): # from tf label = mx.nd.array(labels) data.attach_grad() with mx.autograd.record(): - l = mx.contrib.ndarray.CTCLoss(data, label, - use_data_lengths=True, - use_label_lengths=True, - data_lengths=mx.nd.array(seq_lens), - label_lengths=mx.nd.array(label_lens), - blank_label=blank_label) + l = mx.ndarray.ctc_loss(data, label, + use_data_lengths=True, + use_label_lengths=True, + data_lengths=mx.nd.array(seq_lens), + label_lengths=mx.nd.array(label_lens), + blank_label=blank_label) l.backward() assert_almost_equal(l.asnumpy(), loss_truth, atol=1e-5, rtol=1e-5) assert_almost_equal(data.grad.asnumpy(), grad_truth, atol=1e-5, rtol=1e-5) From 736ec1d3fcbe17c6ab9c782685aa36426a62a858 Mon Sep 17 00:00:00 2001 From: Lin Yuan Date: Fri, 21 Sep 2018 13:25:16 -0700 Subject: [PATCH 04/17] fix lint error --- src/operator/nn/ctc_loss-inl.h | 55 +++++++++++++--------------------- src/operator/nn/ctc_loss.cc | 5 ++-- src/operator/nn/ctc_loss.cu | 25 ++++++++-------- 3 files changed, 35 insertions(+), 50 deletions(-) diff --git a/src/operator/nn/ctc_loss-inl.h b/src/operator/nn/ctc_loss-inl.h index 03fe3a791287..23623acdcdff 100644 --- a/src/operator/nn/ctc_loss-inl.h +++ b/src/operator/nn/ctc_loss-inl.h @@ -20,14 +20,16 @@ /*! * Copyright (c) 2017 by Contributors * \file ctc_loss-inl.h - * \brief + * \brief CTC Loss operator */ -#ifndef MXNET_OPERATOR_CTC_LOSS_INL_H_ -#define MXNET_OPERATOR_CTC_LOSS_INL_H_ +#ifndef MXNET_OPERATOR_NN_CTC_LOSS_INL_H_ +#define MXNET_OPERATOR_NN_CTC_LOSS_INL_H_ #include #include +#include +#include #include "./sequence_mask-inl.h" #include "../sequence_op_common.h" #include "../operator_common.h" @@ -39,7 +41,6 @@ namespace op { namespace ctc_loss { enum CTCLossOpInputs { kData, kLabel }; enum CTCLossOpOutputs { kOut, kGrad }; -enum CTCLossOpForwardResource { kTempSpace }; } template @@ -119,15 +120,13 @@ inline void get_workspace_size(std::vector *label_lengths, // Takes a tensor of labels, and interprets 0-elements at the end of the vector // as padding. The tensor is packed into an std::vector without padding // characters. The label sequence lengths are also inferred from the padding chars. -// When cudnn is enabled, the return value signifies whether the cudnn length limit is exceeded. template -inline bool LabelTensorToPackedVector(mshadow::Tensor labels, +inline void LabelTensorToPackedVector(mshadow::Tensor labels, int padding_mask, std::vector *packed_labels, std::vector *label_lengths) { int batch = labels.size(0); int max_num_labels = labels.size(1); - bool exceed_limit = false; std::vector cpu_labels(max_num_labels*batch); mshadow::Tensor flat_labels = labels.FlatTo1D(); @@ -137,28 +136,22 @@ inline bool LabelTensorToPackedVector(mshadow::Tensor labels, auto start = cpu_labels.data()+b*max_num_labels; auto res = std::find(start, start+max_num_labels, padding_mask); int len = std::distance(start, res); -#if defined(__CUDACC__) && MXNET_USE_CUDNN == 1 && CUDNN_MAJOR >= 7 - exceed_limit = exceed_limit || len > CUDNN_LABEL_LENGTH_LIMIT; -#endif std::copy(start, start + len, std::back_inserter(*packed_labels)); label_lengths->at(b) = len; } - return exceed_limit; } // Takes a tensor of labels, and a vector which specifies the actual length of each label // The tensor is packed into an std::vector without padding characters. // The label length vector is copied into an std::vector. -// When cudnn is enabled, the return value signifies whether the cudnn length limit is exceeded. template -inline bool PackLabelByLength(mshadow::Tensor labels, +inline void PackLabelByLength(mshadow::Tensor labels, mshadow::Tensor in_label_lengths, std::vector *packed_labels, std::vector *label_lengths) { int batch = labels.size(0); int max_num_labels = labels.size(1); - bool exceed_limit = false; IndexTensorToVector(in_label_lengths, label_lengths); @@ -169,13 +162,9 @@ inline bool PackLabelByLength(mshadow::Tensor labels, for (int b = 0; b < batch; ++b) { auto start = cpu_labels.data()+b*max_num_labels; int len = label_lengths->at(b); -#if defined(__CUDACC__) && MXNET_USE_CUDNN == 1 && CUDNN_MAJOR >= 7 - exceed_limit = exceed_limit || len > CUDNN_LABEL_LENGTH_LIMIT; -#endif std::copy(start, start + len, std::back_inserter(*packed_labels)); } - return exceed_limit; } struct CTCLossOpParam : public dmlc::Parameter { @@ -205,13 +194,12 @@ struct CTCLossOpParam : public dmlc::Parameter { "is reserved for blank label instead, " "and label values for tokens in the vocabulary are " "between ``0`` and ``alphabet_size-2``, and the padding mask is ``0``."); - } + } }; inline bool CTCLossOpShape(const nnvm::NodeAttrs &attrs, std::vector* in_attrs, std::vector* out_attrs) { - const CTCLossOpParam& param = nnvm::get(attrs.parsed); CHECK_EQ(in_attrs->size(), 2U + param.use_data_lengths + param.use_label_lengths); CHECK_EQ(out_attrs->size(), 2U); @@ -243,8 +231,8 @@ inline bool CTCLossOpShape(const nnvm::NodeAttrs &attrs, TShape oshape(1); oshape[0] = dshape[1]; // batch size - SHAPE_ASSIGN_CHECK(*out_attrs, 0, oshape); // forward output - SHAPE_ASSIGN_CHECK(*out_attrs, 1, dshape); // grad output + SHAPE_ASSIGN_CHECK(*out_attrs, 0, oshape); // forward output + SHAPE_ASSIGN_CHECK(*out_attrs, 1, dshape); // grad output return true; } @@ -256,8 +244,8 @@ inline bool CTCLossOpType(const nnvm::NodeAttrs& attrs, int dtype = (*in_attrs)[ctc_loss::kData]; CHECK_NE(dtype, -1) << "Input data must have specified type"; - TYPE_ASSIGN_CHECK(*out_attrs, 0, in_attrs->at(0)); // forward output - TYPE_ASSIGN_CHECK(*out_attrs, 1, in_attrs->at(0)); // grad output + TYPE_ASSIGN_CHECK(*out_attrs, 0, in_attrs->at(0)); // forward output + TYPE_ASSIGN_CHECK(*out_attrs, 1, in_attrs->at(0)); // grad output return true; } @@ -281,10 +269,6 @@ inline bool CTCLossOpStorageType(const nnvm::NodeAttrs& attrs, return dispatched; } -inline std::vector CTCLossOpResource(const std::vector &in_shape) { - return std::vector{ResourceRequest::kTempSpace}; -} - inline int CTCLossOpNumInputs(const NodeAttrs& attrs) { const CTCLossOpParam& param = nnvm::get(attrs.parsed); return 2 + param.use_data_lengths + param.use_label_lengths; @@ -311,12 +295,12 @@ void CTCLossOpForward(const nnvm::NodeAttrs& attrs, const std::vector& outputs) { using namespace mshadow; using namespace mshadow::expr; - + const CTCLossOpParam& param = nnvm::get(attrs.parsed); CHECK_EQ(inputs.size(), 2U + param.use_data_lengths + param.use_label_lengths); CHECK_EQ(outputs.size(), 2U); CHECK_EQ(req.size(), 2U); - + Stream *s = ctx.get_stream(); MSHADOW_TYPE_SWITCH(inputs[ctc_loss::kLabel].type_flag_, DType, { Tensor data = @@ -366,7 +350,7 @@ void CTCLossOpForward(const nnvm::NodeAttrs& attrs, compute_ctc_cost(data, costs.dptr_, grad.dptr_, packed_labels.data(), label_lengths.data(), data_lengths.data(), workspace.dptr_, req[ctc_loss::kGrad] != mxnet::kNullOp, - param.blank_label == 0 ? 0 : (alphabet_size-1)); + param.blank_label == 0 ? 0 : (alphabet_size-1)); if (param.use_data_lengths) { // baidu warp CTC implementation sometimes includes undefined gradients @@ -387,7 +371,7 @@ void CTCLossOpBackward(const nnvm::NodeAttrs& attrs, const std::vector& outputs) { using namespace mshadow; using namespace mxnet_op; - + Stream *s = ctx.get_stream(); Tensor data_grad = @@ -398,12 +382,13 @@ void CTCLossOpBackward(const nnvm::NodeAttrs& attrs, Tensor data_grad_computed = inputs[3].get(s); - + Assign(data_grad, req[0], - mshadow::expr::broadcast<1>(output_grad, data_grad.shape_) * data_grad_computed); + mshadow::expr::broadcast<1>(output_grad, data_grad.shape_) * data_grad_computed); } } // namespace op } // namespace mxnet -#endif // MXNET_OPERATOR_CTC_LOSS_INL_H_ \ No newline at end of file +#endif // MXNET_OPERATOR_NN_CTC_LOSS_INL_H_ + diff --git a/src/operator/nn/ctc_loss.cc b/src/operator/nn/ctc_loss.cc index e56810d12512..d4bc194b94d0 100644 --- a/src/operator/nn/ctc_loss.cc +++ b/src/operator/nn/ctc_loss.cc @@ -41,7 +41,7 @@ ctcStatus_t compute_ctc_cost(const Tensor activations, data_lengths); } } -} // namespace mshadow +} // namespace mshadow namespace mxnet { namespace op { @@ -109,7 +109,6 @@ information on the definition and the algorithm. .set_attr("FInferStorageType", CTCLossOpStorageType) .set_attr("FResourceRequest", [](const NodeAttrs& attrs) { return std::vector{ResourceRequest::kTempSpace}; }) -//.set_attr("FResourceRequest", CTCLossOpResource) .set_attr("FCompute", CTCLossOpForward) .set_attr("FGradient", ElemwiseGradUseOut{"_backward_ctc_loss"}) .add_argument("data", "NDArray-or-Symbol", "Input ndarray") @@ -130,4 +129,4 @@ NNVM_REGISTER_OP(_backward_ctc_loss) .set_attr("FCompute", CTCLossOpBackward); } // namespace op -} // namespace mxnet \ No newline at end of file +} // namespace mxnet diff --git a/src/operator/nn/ctc_loss.cu b/src/operator/nn/ctc_loss.cu index 5ab7d17b85ab..e9e4c058ec94 100644 --- a/src/operator/nn/ctc_loss.cu +++ b/src/operator/nn/ctc_loss.cu @@ -21,16 +21,17 @@ * \file ctc_loss.cu * \brief GPU Implementation of ctc_loss op */ - #include "./ctc_loss-inl.h" - namespace mxnet { - namespace op { - - NNVM_REGISTER_OP(ctc_loss) - .set_attr("FCompute", CTCLossOpForward); - - NNVM_REGISTER_OP(_backward_ctc_loss) - .set_attr("FCompute", CTCLossOpBackward); - - } // namespace op - } // namespace mxnet \ No newline at end of file +#include "./ctc_loss-inl.h" + +namespace mxnet { +namespace op { + +NNVM_REGISTER_OP(ctc_loss) +.set_attr("FCompute", CTCLossOpForward); + +NNVM_REGISTER_OP(_backward_ctc_loss) +.set_attr("FCompute", CTCLossOpBackward); + +} // namespace op +} // namespace mxnet From 923a8c1fb0183e88916ffb4fd1d604207fb80390 Mon Sep 17 00:00:00 2001 From: Lin Yuan Date: Fri, 21 Sep 2018 16:06:23 -0700 Subject: [PATCH 05/17] refactoring --- src/operator/nn/ctc_loss-inl.h | 59 ++++++++++++++++------------------ 1 file changed, 28 insertions(+), 31 deletions(-) diff --git a/src/operator/nn/ctc_loss-inl.h b/src/operator/nn/ctc_loss-inl.h index 23623acdcdff..5cca38544c37 100644 --- a/src/operator/nn/ctc_loss-inl.h +++ b/src/operator/nn/ctc_loss-inl.h @@ -44,9 +44,9 @@ enum CTCLossOpOutputs { kOut, kGrad }; } template -inline void get_workspace_size(std::vector *label_lengths, - std::vector *data_lengths, - int alphabet_size, int minibatch, bool gpu, +inline void get_workspace_size(const std::vector *label_lengths, + const std::vector *data_lengths, + int alphabet_size, int minibatch, bool isGPU, size_t *size_bytes) { // This is the max of all S and T for all examples in the minibatch. int maxL = *std::max_element(label_lengths->data(), @@ -58,7 +58,7 @@ inline void get_workspace_size(std::vector *label_lengths, *size_bytes = 0; - if (gpu) { + if (isGPU) { // GPU storage // nll_forward, nll_backward *size_bytes += 2 * sizeof(T) * minibatch; @@ -128,13 +128,13 @@ inline void LabelTensorToPackedVector(mshadow::Tensor labels, int batch = labels.size(0); int max_num_labels = labels.size(1); - std::vector cpu_labels(max_num_labels*batch); + std::vector cpu_labels(max_num_labels * batch); mshadow::Tensor flat_labels = labels.FlatTo1D(); IndexTensorToVector(flat_labels, &cpu_labels); for (int b = 0; b < batch; ++b) { - auto start = cpu_labels.data()+b*max_num_labels; - auto res = std::find(start, start+max_num_labels, padding_mask); + auto start = cpu_labels.data() + b * max_num_labels; + auto res = std::find(start, start + max_num_labels, padding_mask); int len = std::distance(start, res); std::copy(start, start + len, std::back_inserter(*packed_labels)); @@ -155,12 +155,12 @@ inline void PackLabelByLength(mshadow::Tensor labels, IndexTensorToVector(in_label_lengths, label_lengths); - std::vector cpu_labels(max_num_labels*batch); + std::vector cpu_labels(max_num_labels * batch); mshadow::Tensor flat_labels = labels.FlatTo1D(); IndexTensorToVector(flat_labels, &cpu_labels); for (int b = 0; b < batch; ++b) { - auto start = cpu_labels.data()+b*max_num_labels; + auto start = cpu_labels.data() + b * max_num_labels; int len = label_lengths->at(b); std::copy(start, start + len, std::back_inserter(*packed_labels)); @@ -301,17 +301,17 @@ void CTCLossOpForward(const nnvm::NodeAttrs& attrs, CHECK_EQ(outputs.size(), 2U); CHECK_EQ(req.size(), 2U); + const TBlob& in_data = inputs[ctc_loss::kData]; + const TBlob& in_label = inputs[ctc_loss::kLabel]; + const TBlob& out_data = outputs[ctc_loss::kOut]; + const TBlob& out_grad = outputs[ctc_loss::kGrad]; + Stream *s = ctx.get_stream(); MSHADOW_TYPE_SWITCH(inputs[ctc_loss::kLabel].type_flag_, DType, { - Tensor data = - inputs[ctc_loss::kData].get(s); - Tensor labels = - inputs[ctc_loss::kLabel].get(s); - - Tensor costs = - outputs[ctc_loss::kOut].get(s); - Tensor grad = - outputs[ctc_loss::kGrad].get(s); + Tensor data = in_data.get(s); + Tensor labels = in_label.get(s); + Tensor costs = out_data.get(s); + Tensor grad = out_grad.get(s); int max_seq_len = data.size(0); int batch_size = data.size(1); @@ -338,9 +338,8 @@ void CTCLossOpForward(const nnvm::NodeAttrs& attrs, } size_t size_bytes; - bool gpu = data.kDevCPU ? false : true; get_workspace_size(&label_lengths, &data_lengths, alphabet_size, - batch_size, gpu, &size_bytes); + batch_size, data.kDevCPU ? true : false, &size_bytes); // round-up so there are enough elems in memory int num_tmp_elems = (size_bytes + sizeof(real_t) - 1) / sizeof(real_t); @@ -350,7 +349,7 @@ void CTCLossOpForward(const nnvm::NodeAttrs& attrs, compute_ctc_cost(data, costs.dptr_, grad.dptr_, packed_labels.data(), label_lengths.data(), data_lengths.data(), workspace.dptr_, req[ctc_loss::kGrad] != mxnet::kNullOp, - param.blank_label == 0 ? 0 : (alphabet_size-1)); + param.blank_label == 0 ? 0 : (alphabet_size - 1)); if (param.use_data_lengths) { // baidu warp CTC implementation sometimes includes undefined gradients @@ -373,18 +372,16 @@ void CTCLossOpBackward(const nnvm::NodeAttrs& attrs, using namespace mxnet_op; Stream *s = ctx.get_stream(); + const TBlob& in_grad = outputs[0]; + const TBlob& out_grad = inputs[0]; + const TBlob& grad_computed = inputs[3]; // grad computed in the forward step - Tensor data_grad = - outputs[0].get(s); - - Tensor output_grad = - inputs[0].get(s); - - Tensor data_grad_computed = - inputs[3].get(s); + Tensor igrad_data = in_grad.get(s); + Tensor ograd_data = out_grad.get(s); + Tensor computed_grad_data = grad_computed.get(s); - Assign(data_grad, req[0], - mshadow::expr::broadcast<1>(output_grad, data_grad.shape_) * data_grad_computed); + Assign(igrad_data, req[0], + mshadow::expr::broadcast<1>(ograd_data, computed_grad_data.shape_) * computed_grad_data); } } // namespace op From de3e830e9af93dab245d346a3283df1e68a7fc01 Mon Sep 17 00:00:00 2001 From: Lin Yuan Date: Mon, 24 Sep 2018 10:32:52 -0700 Subject: [PATCH 06/17] Fix compilation error in CUDA --- src/operator/nn/ctc_loss.cu | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/src/operator/nn/ctc_loss.cu b/src/operator/nn/ctc_loss.cu index e9e4c058ec94..550c4c476a44 100644 --- a/src/operator/nn/ctc_loss.cu +++ b/src/operator/nn/ctc_loss.cu @@ -18,11 +18,33 @@ */ /*! + * Copyright (c) 2018 by Contributors * \file ctc_loss.cu * \brief GPU Implementation of ctc_loss op */ #include "./ctc_loss-inl.h" +#include "../contrib/ctc_include/detail/cpu_ctc.h" + +namespace mshadow { + +template +ctcStatus_t compute_ctc_cost(const Tensor activations, + DType *costs, DType *grads, int *labels, + int *label_lengths, int *input_lengths, + void *workspace, int train, int blank_label) { + int minibatch = static_cast(activations.size(1)); + int alphabet_size = static_cast(activations.size(2)); + mxnet_warpctc::GpuCTC ctc(alphabet_size, minibatch, workspace, + activations.stream_->stream_, blank_label); + if (train) + return ctc.cost_and_grad(activations.dptr_, grads, costs, labels, + label_lengths, input_lengths); + else + return ctc.score_forward(activations.dptr_, costs, labels, + label_lengths, input_lengths); +} +} // namespace mshadow namespace mxnet { namespace op { From c37a9b53d3afc2670b5f1586b91488523d9e596d Mon Sep 17 00:00:00 2001 From: Lin Yuan Date: Mon, 24 Sep 2018 18:23:03 +0000 Subject: [PATCH 07/17] Fix CPU compilation error --- src/operator/nn/ctc_loss-inl.h | 1 + src/operator/nn/ctc_loss.cu | 2 +- 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/src/operator/nn/ctc_loss-inl.h b/src/operator/nn/ctc_loss-inl.h index 5cca38544c37..fed3453ebb4d 100644 --- a/src/operator/nn/ctc_loss-inl.h +++ b/src/operator/nn/ctc_loss-inl.h @@ -30,6 +30,7 @@ #include #include #include +#include "../mshadow_op.h" #include "./sequence_mask-inl.h" #include "../sequence_op_common.h" #include "../operator_common.h" diff --git a/src/operator/nn/ctc_loss.cu b/src/operator/nn/ctc_loss.cu index 550c4c476a44..11d054ffd953 100644 --- a/src/operator/nn/ctc_loss.cu +++ b/src/operator/nn/ctc_loss.cu @@ -24,7 +24,7 @@ */ #include "./ctc_loss-inl.h" -#include "../contrib/ctc_include/detail/cpu_ctc.h" +#include "../contrib/ctc_include/detail/gpu_ctc.h" namespace mshadow { From fe7b33a1fa9dac5f2492cfa1c4364192cef0bd80 Mon Sep 17 00:00:00 2001 From: Lin Yuan Date: Thu, 27 Sep 2018 13:22:53 -0700 Subject: [PATCH 08/17] Move ctc_include to nn folder and refactor --- Makefile | 2 +- src/operator/nn/ctc_include/LICENSE | 205 ++++++ .../nn/ctc_include/contrib/moderngpu/LICENSE | 26 + .../include/device/ctaloadbalance.cuh | 136 ++++ .../moderngpu/include/device/ctamerge.cuh | 333 +++++++++ .../moderngpu/include/device/ctascan.cuh | 308 ++++++++ .../moderngpu/include/device/ctasearch.cuh | 207 ++++++ .../moderngpu/include/device/ctasegreduce.cuh | 238 +++++++ .../moderngpu/include/device/ctasegscan.cuh | 137 ++++ .../moderngpu/include/device/ctasegsort.cuh | 443 ++++++++++++ .../include/device/ctasortedsearch.cuh | 208 ++++++ .../moderngpu/include/device/devicetypes.cuh | 363 ++++++++++ .../moderngpu/include/device/deviceutil.cuh | 143 ++++ .../moderngpu/include/device/intrinsics.cuh | 421 +++++++++++ .../moderngpu/include/device/loadstore.cuh | 674 ++++++++++++++++++ .../moderngpu/include/device/serialsets.cuh | 235 ++++++ .../moderngpu/include/device/sortnetwork.cuh | 168 +++++ .../contrib/moderngpu/include/mgpudevice.cuh | 289 ++++++++ .../contrib/moderngpu/include/mgpuenums.h | 70 ++ .../contrib/moderngpu/include/util/static.h | 183 +++++ src/operator/nn/ctc_include/detail/cpu_ctc.h | 509 +++++++++++++ .../nn/ctc_include/detail/ctc_helper.h | 93 +++ src/operator/nn/ctc_include/detail/gpu_ctc.h | 505 +++++++++++++ .../nn/ctc_include/detail/gpu_ctc_kernels.h | 507 +++++++++++++ .../nn/ctc_include/detail/hostdevice.h | 27 + src/operator/nn/ctc_loss-inl.h | 21 +- src/operator/nn/ctc_loss.cc | 6 +- src/operator/nn/ctc_loss.cu | 2 +- 28 files changed, 6446 insertions(+), 13 deletions(-) create mode 100644 src/operator/nn/ctc_include/LICENSE create mode 100644 src/operator/nn/ctc_include/contrib/moderngpu/LICENSE create mode 100644 src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctaloadbalance.cuh create mode 100644 src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctamerge.cuh create mode 100644 src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctascan.cuh create mode 100644 src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasearch.cuh create mode 100644 src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasegreduce.cuh create mode 100644 src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasegscan.cuh create mode 100644 src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasegsort.cuh create mode 100644 src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasortedsearch.cuh create mode 100644 src/operator/nn/ctc_include/contrib/moderngpu/include/device/devicetypes.cuh create mode 100644 src/operator/nn/ctc_include/contrib/moderngpu/include/device/deviceutil.cuh create mode 100644 src/operator/nn/ctc_include/contrib/moderngpu/include/device/intrinsics.cuh create mode 100644 src/operator/nn/ctc_include/contrib/moderngpu/include/device/loadstore.cuh create mode 100644 src/operator/nn/ctc_include/contrib/moderngpu/include/device/serialsets.cuh create mode 100644 src/operator/nn/ctc_include/contrib/moderngpu/include/device/sortnetwork.cuh create mode 100644 src/operator/nn/ctc_include/contrib/moderngpu/include/mgpudevice.cuh create mode 100644 src/operator/nn/ctc_include/contrib/moderngpu/include/mgpuenums.h create mode 100644 src/operator/nn/ctc_include/contrib/moderngpu/include/util/static.h create mode 100644 src/operator/nn/ctc_include/detail/cpu_ctc.h create mode 100644 src/operator/nn/ctc_include/detail/ctc_helper.h create mode 100644 src/operator/nn/ctc_include/detail/gpu_ctc.h create mode 100644 src/operator/nn/ctc_include/detail/gpu_ctc_kernels.h create mode 100644 src/operator/nn/ctc_include/detail/hostdevice.h diff --git a/Makefile b/Makefile index 69b25daa9d1c..a37694019e7a 100644 --- a/Makefile +++ b/Makefile @@ -568,7 +568,7 @@ lint: cpplint rcpplint jnilint pylint cpplint: 3rdparty/dmlc-core/scripts/lint.py mxnet cpp include src plugin cpp-package tests \ - --exclude_path src/operator/contrib/ctc_include + --exclude_path src/operator/contrib/ctc_include src/operator/nn/ctc_include pylint: pylint --rcfile=$(ROOTDIR)/ci/other/pylintrc --ignore-patterns=".*\.so$$,.*\.dll$$,.*\.dylib$$" python/mxnet tools/caffe_converter/*.py diff --git a/src/operator/nn/ctc_include/LICENSE b/src/operator/nn/ctc_include/LICENSE new file mode 100644 index 000000000000..4946875860dd --- /dev/null +++ b/src/operator/nn/ctc_include/LICENSE @@ -0,0 +1,205 @@ + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright [yyyy] [name of copyright owner] + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. + + ---- + + Copyright 2015-2016, Baidu USA LLC. \ No newline at end of file diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/LICENSE b/src/operator/nn/ctc_include/contrib/moderngpu/LICENSE new file mode 100644 index 000000000000..98128870cf41 --- /dev/null +++ b/src/operator/nn/ctc_include/contrib/moderngpu/LICENSE @@ -0,0 +1,26 @@ +/****************************************************************************** +* Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. +* +* Redistribution and use in source and binary forms, with or without +* modification, are permitted provided that the following conditions are met: +* * Redistributions of source code must retain the above copyright +* notice, this list of conditions and the following disclaimer. +* * Redistributions in binary form must reproduce the above copyright +* notice, this list of conditions and the following disclaimer in the +* documentation and/or other materials provided with the distribution. +* * Neither the name of the NVIDIA CORPORATION nor the +* names of its contributors may be used to endorse or promote products +* derived from this software without specific prior written permission. +* +* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" +* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE +* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE +* ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY +* DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES +* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; +* LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND +* ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT +* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS +* SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +* +******************************************************************************/ diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctaloadbalance.cuh b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctaloadbalance.cuh new file mode 100644 index 000000000000..69f6d96d1fa4 --- /dev/null +++ b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctaloadbalance.cuh @@ -0,0 +1,136 @@ +/****************************************************************************** + * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions are met: + * * Redistributions of source code must retain the above copyright + * notice, this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright + * notice, this list of conditions and the following disclaimer in the + * documentation and/or other materials provided with the distribution. + * * Neither the name of the NVIDIA CORPORATION nor the + * names of its contributors may be used to endorse or promote products + * derived from this software without specific prior written permission. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY + * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND + * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + * + ******************************************************************************/ + +/****************************************************************************** + * + * Code and text by Sean Baxter, NVIDIA Research + * See http://nvlabs.github.io/moderngpu for repository and documentation. + * + ******************************************************************************/ + +#pragma once + +#include "ctasearch.cuh" +#include "loadstore.cuh" + +namespace mgpu { + +//////////////////////////////////////////////////////////////////////////////// +// DeviceLoadBalancingSearch +// Upper Bound search from A (needles) into B (haystack). The A values are +// natural numbers from aBegin to aEnd. bFirst is the index of the B value at +// bBegin in shared memory. + +template +MGPU_DEVICE void DeviceSerialLoadBalanceSearch(const int* b_shared, int aBegin, + int aEnd, int bFirst, int bBegin, int bEnd, int* a_shared) { + + int bKey = b_shared[bBegin]; + + #pragma unroll + for(int i = 0; i < VT; ++i) { + bool p; + if(RangeCheck) + p = (aBegin < aEnd) && ((bBegin >= bEnd) || (aBegin < bKey)); + else + p = aBegin < bKey; + + if(p) + // Advance A (the needle). + a_shared[aBegin++] = bFirst + bBegin; + else + // Advance B (the haystack). + bKey = b_shared[++bBegin]; + } +} + +//////////////////////////////////////////////////////////////////////////////// +// CTALoadBalance +// Computes upper_bound(counting_iterator(first), b_global) - 1. + +// Unlike most other CTA* functions, CTALoadBalance loads from global memory. +// This returns the loaded B elements at the beginning or end of shared memory +// depending on the aFirst argument. + +// CTALoadBalance requires NT * VT + 2 slots of shared memory. +template +MGPU_DEVICE int4 CTALoadBalance(int destCount, InputIt b_global, + int sourceCount, int block, int tid, const int* mp_global, + int* indices_shared, bool loadPrecedingB) { + + int4 range = ComputeMergeRange(destCount, sourceCount, block, 0, NT * VT, + mp_global); + + int a0 = range.x; + int a1 = range.y; + int b0 = range.z; + int b1 = range.w; + if(!b0) loadPrecedingB = false; + + // Load one trailing term from B. If we're already at the end, fill the + // end of the buffer with destCount. + int aCount = a1 - a0; + int bCount = b1 - b0; + int extended = b1 < sourceCount; + int loadCount = bCount + extended; + int fillCount = NT * VT + 1 - loadCount - aCount; + + int* a_shared = indices_shared; + int* b_shared = indices_shared + aCount + (int)loadPrecedingB; + + // Load the B values. +// DeviceMemToMemLoop(bCount + extended + (int)loadPrecedingB, +// b_global + b0 - (int)loadPrecedingB, tid, +// b_shared - (int)loadPrecedingB); + + for(int i = tid - (int)loadPrecedingB; i < bCount + extended; i += NT) + b_shared[i] = b_global[b0 + i]; + + // Fill the end of the array with destCount. + for(int i = tid + extended; i < fillCount; i += NT) + b_shared[bCount + i] = destCount; + __syncthreads(); + + // Run a merge path to find the start of the serial merge for each thread. + int diag = VT * tid; + int mp = MergePath(mgpu::counting_iterator(a0), + aCount, b_shared, bCount, diag, mgpu::less()); + + int a0tid = a0 + mp; + int b0tid = diag - mp; + + // Subtract 1 from b0 because we want to return upper_bound - 1. + DeviceSerialLoadBalanceSearch(b_shared, a0tid, a1, b0 - 1, + b0tid, bCount, a_shared - a0); + __syncthreads(); + + b0 -= (int)loadPrecedingB; + return make_int4(a0, a1, b0, b1); +} + + +} // namespace mgpu diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctamerge.cuh b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctamerge.cuh new file mode 100644 index 000000000000..bb702c460455 --- /dev/null +++ b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctamerge.cuh @@ -0,0 +1,333 @@ +/****************************************************************************** + * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions are met: + * * Redistributions of source code must retain the above copyright + * notice, this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright + * notice, this list of conditions and the following disclaimer in the + * documentation and/or other materials provided with the distribution. + * * Neither the name of the NVIDIA CORPORATION nor the + * names of its contributors may be used to endorse or promote products + * derived from this software without specific prior written permission. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY + * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND + * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + * + ******************************************************************************/ + +/****************************************************************************** + * + * Code and text by Sean Baxter, NVIDIA Research + * See http://nvlabs.github.io/moderngpu for repository and documentation. + * + ******************************************************************************/ + +#pragma once + +#include "ctasearch.cuh" +#include "loadstore.cuh" +#include "sortnetwork.cuh" + +namespace mgpu { + +//////////////////////////////////////////////////////////////////////////////// +// SerialMerge + +template +MGPU_DEVICE void SerialMerge(const T* keys_shared, int aBegin, int aEnd, + int bBegin, int bEnd, T* results, int* indices, Comp comp) { + + T aKey = keys_shared[aBegin]; + T bKey = keys_shared[bBegin]; + + #pragma unroll + for(int i = 0; i < VT; ++i) { + bool p; + if(RangeCheck) + p = (bBegin >= bEnd) || ((aBegin < aEnd) && !comp(bKey, aKey)); + else + p = !comp(bKey, aKey); + + results[i] = p ? aKey : bKey; + indices[i] = p ? aBegin : bBegin - !RangeCheck; + + if(p) aKey = keys_shared[++aBegin]; + else bKey = keys_shared[++bBegin]; + } + __syncthreads(); +} + +//////////////////////////////////////////////////////////////////////////////// +// FindMergeFrame and FindMergesortInterval help mergesort (both CTA and global +// merge pass levels) locate lists within the single source array. + +// Returns (offset of a, offset of b, length of list). +MGPU_HOST_DEVICE int3 FindMergesortFrame(int coop, int block, int nv) { + // coop is the number of CTAs or threads cooperating to merge two lists into + // one. We round block down to the first CTA's ID that is working on this + // merge. + int start = ~(coop - 1) & block; + int size = nv * (coop>> 1); + return make_int3(nv * start, nv * start + size, size); +} + +// Returns (a0, a1, b0, b1) into mergesort input lists between mp0 and mp1. +MGPU_HOST_DEVICE int4 FindMergesortInterval(int3 frame, int coop, int block, + int nv, int count, int mp0, int mp1) { + + // Locate diag from the start of the A sublist. + int diag = nv * block - frame.x; + int a0 = frame.x + mp0; + int a1 = min(count, frame.x + mp1); + int b0 = min(count, frame.y + diag - mp0); + int b1 = min(count, frame.y + diag + nv - mp1); + + // The end partition of the last block for each merge operation is computed + // and stored as the begin partition for the subsequent merge. i.e. it is + // the same partition but in the wrong coordinate system, so its 0 when it + // should be listSize. Correct that by checking if this is the last block + // in this merge operation. + if(coop - 1 == ((coop - 1) & block)) { + a1 = min(count, frame.x + frame.z); + b1 = min(count, frame.y + frame.z); + } + return make_int4(a0, a1, b0, b1); +} + +//////////////////////////////////////////////////////////////////////////////// +// ComputeMergeRange + +MGPU_HOST_DEVICE int4 ComputeMergeRange(int aCount, int bCount, int block, + int coop, int NV, const int* mp_global) { + + // Load the merge paths computed by the partitioning kernel. + int mp0 = mp_global[block]; + int mp1 = mp_global[block + 1]; + int gid = NV * block; + + // Compute the ranges of the sources in global memory. + int4 range; + if(coop) { + int3 frame = FindMergesortFrame(coop, block, NV); + range = FindMergesortInterval(frame, coop, block, NV, aCount, mp0, + mp1); + } else { + range.x = mp0; // a0 + range.y = mp1; // a1 + range.z = gid - range.x; // b0 + range.w = min(aCount + bCount, gid + NV) - range.y; // b1 + } + return range; +} + +//////////////////////////////////////////////////////////////////////////////// +// CTA mergesort support + +template +MGPU_DEVICE void CTABlocksortPass(T* keys_shared, int tid, int count, + int coop, T* keys, int* indices, Comp comp) { + + int list = ~(coop - 1) & tid; + int diag = min(count, VT * ((coop - 1) & tid)); + int start = VT * list; + int a0 = min(count, start); + int b0 = min(count, start + VT * (coop / 2)); + int b1 = min(count, start + VT * coop); + + int p = MergePath(keys_shared + a0, b0 - a0, + keys_shared + b0, b1 - b0, diag, comp); + + SerialMerge(keys_shared, a0 + p, b0, b0 + diag - p, b1, keys, + indices, comp); +} + +template +MGPU_DEVICE void CTABlocksortLoop(ValType threadValues[VT], + KeyType* keys_shared, ValType* values_shared, int tid, int count, + Comp comp) { + + #pragma unroll + for(int coop = 2; coop <= NT; coop *= 2) { + int indices[VT]; + KeyType keys[VT]; + CTABlocksortPass(keys_shared, tid, count, coop, keys, + indices, comp); + + if(HasValues) { + // Exchange the values through shared memory. + DeviceThreadToShared(threadValues, tid, values_shared); + DeviceGather(NT * VT, values_shared, indices, tid, + threadValues); + } + + // Store results in shared memory in sorted order. + DeviceThreadToShared(keys, tid, keys_shared); + } +} + +//////////////////////////////////////////////////////////////////////////////// +// CTAMergesort +// Caller provides the keys in shared memory. This functions sorts the first +// count elements. + +template +MGPU_DEVICE void CTAMergesort(KeyType threadKeys[VT], ValType threadValues[VT], + KeyType* keys_shared, ValType* values_shared, int count, int tid, + Comp comp) { + + // Stable sort the keys in the thread. + if(VT * tid < count) { + if(Stable) + OddEvenTransposeSort(threadKeys, threadValues, comp); + else + OddEvenMergesort(threadKeys, threadValues, comp); + } + + // Store the locally sorted keys into shared memory. + DeviceThreadToShared(threadKeys, tid, keys_shared); + + // Recursively merge lists until the entire CTA is sorted. + CTABlocksortLoop(threadValues, keys_shared, + values_shared, tid, count, comp); +} + +template +MGPU_DEVICE void CTAMergesortKeys(KeyType threadKeys[VT], + KeyType* keys_shared, int count, int tid, Comp comp) { + + int valuesTemp[VT]; + CTAMergesort(threadKeys, valuesTemp, keys_shared, + (int*)keys_shared, count, tid, comp); +} + +template +MGPU_DEVICE void CTAMergesortPairs(KeyType threadKeys[VT], + ValType threadValues[VT], KeyType* keys_shared, ValType* values_shared, + int count, int tid, Comp comp) { + + CTAMergesort(threadKeys, threadValues, keys_shared, + values_shared, count, tid, comp); +} + +//////////////////////////////////////////////////////////////////////////////// +// DeviceMergeKeysIndices + +template +MGPU_DEVICE void DeviceMergeKeysIndices(It1 a_global, int aCount, It2 b_global, + int bCount, int4 range, int tid, T* keys_shared, T* results, int* indices, + Comp comp) { + + int a0 = range.x; + int a1 = range.y; + int b0 = range.z; + int b1 = range.w; + + if(LoadExtended) { + bool extended = (a1 < aCount) && (b1 < bCount); + aCount = a1 - a0; + bCount = b1 - b0; + int aCount2 = aCount + (int)extended; + int bCount2 = bCount + (int)extended; + + // Load one element past the end of each input to avoid having to use + // range checking in the merge loop. + DeviceLoad2ToShared(a_global + a0, aCount2, + b_global + b0, bCount2, tid, keys_shared); + + // Run a Merge Path search for each thread's starting point. + int diag = VT * tid; + int mp = MergePath(keys_shared, aCount, + keys_shared + aCount2, bCount, diag, comp); + + // Compute the ranges of the sources in shared memory. + int a0tid = mp; + int b0tid = aCount2 + diag - mp; + if(extended) { + SerialMerge(keys_shared, a0tid, 0, b0tid, 0, results, + indices, comp); + } else { + int a1tid = aCount; + int b1tid = aCount2 + bCount; + SerialMerge(keys_shared, a0tid, a1tid, b0tid, b1tid, + results, indices, comp); + } + } else { + // Use the input intervals from the ranges between the merge path + // intersections. + aCount = a1 - a0; + bCount = b1 - b0; + + // Load the data into shared memory. + DeviceLoad2ToShared(a_global + a0, aCount, b_global + b0, + bCount, tid, keys_shared); + + // Run a merge path to find the start of the serial merge for each + // thread. + int diag = VT * tid; + int mp = MergePath(keys_shared, aCount, + keys_shared + aCount, bCount, diag, comp); + + // Compute the ranges of the sources in shared memory. + int a0tid = mp; + int a1tid = aCount; + int b0tid = aCount + diag - mp; + int b1tid = aCount + bCount; + + // Serial merge into register. + SerialMerge(keys_shared, a0tid, a1tid, b0tid, b1tid, results, + indices, comp); + } +} + +//////////////////////////////////////////////////////////////////////////////// +// DeviceMerge +// Merge pairs from global memory into global memory. Useful factorization to +// enable calling from merge, mergesort, and locality sort. + +template +MGPU_DEVICE void DeviceMerge(KeysIt1 aKeys_global, ValsIt1 aVals_global, + int aCount, KeysIt2 bKeys_global, ValsIt2 bVals_global, int bCount, + int tid, int block, int4 range, KeyType* keys_shared, int* indices_shared, + KeysIt3 keys_global, ValsIt3 vals_global, Comp comp) { + + KeyType results[VT]; + int indices[VT]; + DeviceMergeKeysIndices(aKeys_global, aCount, + bKeys_global, bCount, range, tid, keys_shared, results, indices, comp); + + // Store merge results back to shared memory. + DeviceThreadToShared(results, tid, keys_shared); + + // Store merged keys to global memory. + aCount = range.y - range.x; + bCount = range.w - range.z; + DeviceSharedToGlobal(aCount + bCount, keys_shared, tid, + keys_global + NT * VT * block); + + // Copy the values. + if(HasValues) { + DeviceThreadToShared(indices, tid, indices_shared); + + DeviceTransferMergeValuesShared(aCount + bCount, + aVals_global + range.x, bVals_global + range.z, aCount, + indices_shared, tid, vals_global + NT * VT * block); + } +} + +} // namespace mgpu diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctascan.cuh b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctascan.cuh new file mode 100644 index 000000000000..91ed8c4e04c7 --- /dev/null +++ b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctascan.cuh @@ -0,0 +1,308 @@ +/****************************************************************************** + * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions are met: + * * Redistributions of source code must retain the above copyright + * notice, this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright + * notice, this list of conditions and the following disclaimer in the + * documentation and/or other materials provided with the distribution. + * * Neither the name of the NVIDIA CORPORATION nor the + * names of its contributors may be used to endorse or promote products + * derived from this software without specific prior written permission. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY + * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND + * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + * + ******************************************************************************/ + +/****************************************************************************** + * + * Code and text by Sean Baxter, NVIDIA Research + * See http://nvlabs.github.io/moderngpu for repository and documentation. + * + ******************************************************************************/ + +#pragma once + +#include "../mgpuenums.h" +#include "deviceutil.cuh" +#include "intrinsics.cuh" + +namespace mgpu { + +//////////////////////////////////////////////////////////////////////////////// +// CTAReduce + +template > +struct CTAReduce { + typedef typename Op::first_argument_type T; + enum { Size = NT, Capacity = NT }; + struct Storage { T shared[Capacity]; }; + + MGPU_DEVICE static T Reduce(int tid, T x, Storage& storage, Op op = Op()) { + storage.shared[tid] = x; + __syncthreads(); + + // Fold the data in half with each pass. + #pragma unroll + for(int destCount = NT / 2; destCount >= 1; destCount /= 2) { + if(tid < destCount) { + // Read from the right half and store to the left half. + x = op(x, storage.shared[destCount + tid]); + storage.shared[tid] = x; + } + __syncthreads(); + } + T total = storage.shared[0]; + __syncthreads(); + return total; + } +}; + +#if __CUDA_ARCH__ >= 300 + +template +struct CTAReduce > { + typedef mgpu::plus Op; + typedef int T; + enum { Size = NT, Capacity = WARP_SIZE }; + struct Storage { int shared[Capacity]; }; + + MGPU_DEVICE static int Reduce(int tid, int x, Storage& storage, + Op op = Op()) { + + const int NumSections = WARP_SIZE; + const int SecSize = NT / NumSections; + int lane = (SecSize - 1) & tid; + int sec = tid / SecSize; + + // In the first phase, threads cooperatively find the reduction within + // their segment. The segments are SecSize threads (NT / WARP_SIZE) + // wide. + #pragma unroll + for(int offset = 1; offset < SecSize; offset *= 2) + x = shfl_add(x, offset, SecSize); + + // The last thread in each segment stores the local reduction to shared + // memory. + if(SecSize - 1 == lane) storage.shared[sec] = x; + __syncthreads(); + + // Reduce the totals of each input segment. The spine is WARP_SIZE + // threads wide. + if(tid < NumSections) { + x = storage.shared[tid]; + #pragma unroll + for(int offset = 1; offset < NumSections; offset *= 2) + x = shfl_add(x, offset, NumSections); + storage.shared[tid] = x; + } + __syncthreads(); + + int reduction = storage.shared[NumSections - 1]; + __syncthreads(); + + return reduction; + } +}; + +template +struct CTAReduce > { + typedef mgpu::maximum Op; + enum { Size = NT, Capacity = WARP_SIZE }; + struct Storage { int shared[Capacity]; }; + + MGPU_DEVICE static int Reduce(int tid, int x, Storage& storage, + Op op = Op()) { + + const int NumSections = WARP_SIZE; + const int SecSize = NT / NumSections; + int lane = (SecSize - 1) & tid; + int sec = tid / SecSize; + + #pragma unroll + for(int offset = 1; offset < SecSize; offset *= 2) + x = shfl_max(x, offset, SecSize); + + if(SecSize - 1 == lane) storage.shared[sec] = x; + __syncthreads(); + + if(tid < NumSections) { + x = storage.shared[tid]; + #pragma unroll + for(int offset = 1; offset < NumSections; offset *= 2) + x = shfl_max(x, offset, NumSections); + storage.shared[tid] = x; + } + __syncthreads(); + + int reduction = storage.shared[NumSections - 1]; + __syncthreads(); + + return reduction; + } +}; + +#endif // __CUDA_ARCH__ >= 300 + +//////////////////////////////////////////////////////////////////////////////// +// CTAScan + +template > +struct CTAScan { + typedef typename Op::result_type T; + enum { Size = NT, Capacity = 2 * NT + 1 }; + struct Storage { T shared[Capacity]; }; + + MGPU_DEVICE static T Scan(int tid, T x, Storage& storage, T* total, + MgpuScanType type = MgpuScanTypeExc, T identity = (T)0, Op op = Op()) { + + storage.shared[tid] = x; + int first = 0; + __syncthreads(); + + #pragma unroll + for(int offset = 1; offset < NT; offset += offset) { + if(tid >= offset) + x = op(storage.shared[first + tid - offset], x); + first = NT - first; + storage.shared[first + tid] = x; + __syncthreads(); + } + *total = storage.shared[first + NT - 1]; + + if(MgpuScanTypeExc == type) + x = tid ? storage.shared[first + tid - 1] : identity; + + __syncthreads(); + return x; + } + MGPU_DEVICE static T Scan(int tid, T x, Storage& storage) { + T total; + return Scan(tid, x, storage, &total, MgpuScanTypeExc, (T)0, Op()); + } +}; + +//////////////////////////////////////////////////////////////////////////////// +// Special partial specialization for CTAScan on Kepler. +// This uses the shfl intrinsic to reduce scan latency. + +#if __CUDA_ARCH__ >= 300 + +template +struct CTAScan > { + typedef mgpu::plus Op; + enum { Size = NT, NumSegments = WARP_SIZE, SegSize = NT / NumSegments }; + enum { Capacity = NumSegments + 1 }; + struct Storage { int shared[Capacity + 1]; }; + + MGPU_DEVICE static int Scan(int tid, int x, Storage& storage, int* total, + MgpuScanType type = MgpuScanTypeExc, int identity = 0, Op op = Op()) { + + // Define WARP_SIZE segments that are NT / WARP_SIZE large. + // Each warp makes log(SegSize) shfl_add calls. + // The spine makes log(WARP_SIZE) shfl_add calls. + int lane = (SegSize - 1) & tid; + int segment = tid / SegSize; + + // Scan each segment using shfl_add. + int scan = x; + #pragma unroll + for(int offset = 1; offset < SegSize; offset *= 2) + scan = shfl_add(scan, offset, SegSize); + + // Store the reduction (last element) of each segment into storage. + if(SegSize - 1 == lane) storage.shared[segment] = scan; + __syncthreads(); + + // Warp 0 does a full shfl warp scan on the partials. The total is + // stored to shared[NumSegments]. (NumSegments = WARP_SIZE) + if(tid < NumSegments) { + int y = storage.shared[tid]; + int scan = y; + #pragma unroll + for(int offset = 1; offset < NumSegments; offset *= 2) + scan = shfl_add(scan, offset, NumSegments); + storage.shared[tid] = scan - y; + if(NumSegments - 1 == tid) storage.shared[NumSegments] = scan; + } + __syncthreads(); + + // Add the scanned partials back in and convert to exclusive scan. + scan += storage.shared[segment]; + if(MgpuScanTypeExc == type) { + scan -= x; + if(identity && !tid) scan = identity; + } + *total = storage.shared[NumSegments]; + __syncthreads(); + + return scan; + } + MGPU_DEVICE static int Scan(int tid, int x, Storage& storage) { + int total; + return Scan(tid, x, storage, &total, MgpuScanTypeExc, 0); + } +}; + +#endif // __CUDA_ARCH__ >= 300 + +//////////////////////////////////////////////////////////////////////////////// +// CTABinaryScan + +template +MGPU_DEVICE int CTABinaryScan(int tid, bool x, int* shared, int* total) { + const int NumWarps = NT / WARP_SIZE; + int warp = tid / WARP_SIZE; + int lane = (WARP_SIZE - 1); + + // Store the bit totals for each warp. + uint bits = __ballot(x); + shared[warp] = popc(bits); + __syncthreads(); + +#if __CUDA_ARCH__ >= 300 + if(tid < NumWarps) { + int x = shared[tid]; + int scan = x; + #pragma unroll + for(int offset = 1; offset < NumWarps; offset *= 2) + scan = shfl_add(scan, offset, NumWarps); + shared[tid] = scan - x; + } + __syncthreads(); + +#else + // Thread 0 scans warp totals. + if(!tid) { + int scan = 0; + #pragma unroll + for(int i = 0; i < NumWarps; ++i) { + int y = shared[i]; + shared[i] = scan; + scan += y; + } + shared[NumWarps] = scan; + } + __syncthreads(); + +#endif // __CUDA_ARCH__ >= 300 + + // Add the warp scan back into the partials. + int scan = shared[warp] + __popc(bfe(bits, 0, lane)); + *total = shared[NumWarps]; + __syncthreads(); + return scan; +} + +} // namespace mgpu diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasearch.cuh b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasearch.cuh new file mode 100644 index 000000000000..1797d4b51a42 --- /dev/null +++ b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasearch.cuh @@ -0,0 +1,207 @@ +/****************************************************************************** + * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions are met: + * * Redistributions of source code must retain the above copyright + * notice, this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright + * notice, this list of conditions and the following disclaimer in the + * documentation and/or other materials provided with the distribution. + * * Neither the name of the NVIDIA CORPORATION nor the + * names of its contributors may be used to endorse or promote products + * derived from this software without specific prior written permission. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY + * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND + * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + * + ******************************************************************************/ + +/****************************************************************************** + * + * Code and text by Sean Baxter, NVIDIA Research + * See http://nvlabs.github.io/moderngpu for repository and documentation. + * + ******************************************************************************/ + +#pragma once + +#include "deviceutil.cuh" +#include "../mgpudevice.cuh" + +namespace mgpu { + +template +MGPU_HOST_DEVICE void BinarySearchIt(It data, int& begin, int& end, T key, + int shift, Comp comp) { + + IntT scale = (1<< shift) - 1; + int mid = (int)((begin + scale * end)>> shift); + + T key2 = data[mid]; + bool pred = (MgpuBoundsUpper == Bounds) ? + !comp(key, key2) : + comp(key2, key); + if(pred) begin = mid + 1; + else end = mid; +} + +template +MGPU_HOST_DEVICE int BiasedBinarySearch(It data, int count, T key, int levels, + Comp comp) { + + int begin = 0; + int end = count; + + if(levels >= 4 && begin < end) + BinarySearchIt(data, begin, end, key, 9, comp); + if(levels >= 3 && begin < end) + BinarySearchIt(data, begin, end, key, 7, comp); + if(levels >= 2 && begin < end) + BinarySearchIt(data, begin, end, key, 5, comp); + if(levels >= 1 && begin < end) + BinarySearchIt(data, begin, end, key, 4, comp); + + while(begin < end) + BinarySearchIt(data, begin, end, key, 1, comp); + return begin; +} + +template +MGPU_HOST_DEVICE int BinarySearch(It data, int count, T key, Comp comp) { + int begin = 0; + int end = count; + while(begin < end) + BinarySearchIt(data, begin, end, key, 1, comp); + return begin; +} + +//////////////////////////////////////////////////////////////////////////////// +// MergePath search + +template +MGPU_HOST_DEVICE int MergePath(It1 a, int aCount, It2 b, int bCount, int diag, + Comp comp) { + + typedef typename std::iterator_traits::value_type T; + int begin = max(0, diag - bCount); + int end = min(diag, aCount); + + while(begin < end) { + int mid = (begin + end)>> 1; + T aKey = a[mid]; + T bKey = b[diag - 1 - mid]; + bool pred = (MgpuBoundsUpper == Bounds) ? + comp(aKey, bKey) : + !comp(bKey, aKey); + if(pred) begin = mid + 1; + else end = mid; + } + return begin; +} + + +//////////////////////////////////////////////////////////////////////////////// +// SegmentedMergePath search + +template +MGPU_HOST_DEVICE int SegmentedMergePath(InputIt keys, int aOffset, int aCount, + int bOffset, int bCount, int leftEnd, int rightStart, int diag, Comp comp) { + + // leftEnd and rightStart are defined from the origin, and diag is defined + // from aOffset. + // We only need to run a Merge Path search if the diagonal intersects the + // segment that strides the left and right halves (i.e. is between leftEnd + // and rightStart). + if(aOffset + diag <= leftEnd) return diag; + if(aOffset + diag >= rightStart) return aCount; + + bCount = min(bCount, rightStart - bOffset); + int begin = max(max(leftEnd - aOffset, 0), diag - bCount); + int end = min(diag, aCount); + + while(begin < end) { + int mid = (begin + end)>> 1; + int ai = aOffset + mid; + int bi = bOffset + diag - 1 - mid; + + bool pred = !comp(keys[bi], keys[ai]); + if(pred) begin = mid + 1; + else end = mid; + } + return begin; +} + +//////////////////////////////////////////////////////////////////////////////// +// BalancedPath search + +template +MGPU_HOST_DEVICE int2 BalancedPath(InputIt1 a, int aCount, InputIt2 b, + int bCount, int diag, int levels, Comp comp) { + + typedef typename std::iterator_traits::value_type T; + + int p = MergePath(a, aCount, b, bCount, diag, comp); + int aIndex = p; + int bIndex = diag - p; + + bool star = false; + if(bIndex < bCount) { + if(Duplicates) { + T x = b[bIndex]; + + // Search for the beginning of the duplicate run in both A and B. + // Because + int aStart = BiasedBinarySearch(a, aIndex, x, + levels, comp); + int bStart = BiasedBinarySearch(b, bIndex, x, + levels, comp); + + // The distance between the merge path and the lower_bound is the + // 'run'. We add up the a- and b- runs and evenly distribute them to + // get a stairstep path. + int aRun = aIndex - aStart; + int bRun = bIndex - bStart; + int xCount = aRun + bRun; + + // Attempt to advance b and regress a. + int bAdvance = max(xCount>> 1, bRun); + int bEnd = min(bCount, bStart + bAdvance + 1); + int bRunEnd = BinarySearch(b + bIndex, + bEnd - bIndex, x, comp) + bIndex; + bRun = bRunEnd - bStart; + + bAdvance = min(bAdvance, bRun); + int aAdvance = xCount - bAdvance; + + bool roundUp = (aAdvance == bAdvance + 1) && (bAdvance < bRun); + aIndex = aStart + aAdvance; + + if(roundUp) star = true; + } else { + if(aIndex && aCount) { + T aKey = a[aIndex - 1]; + T bKey = b[bIndex]; + + // If the last consumed element in A (aIndex - 1) is the same as + // the next element in B (bIndex), we're sitting at a starred + // partition. + if(!comp(aKey, bKey)) star = true; + } + } + } + return make_int2(aIndex, star); +} + +} // namespace mgpu diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasegreduce.cuh b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasegreduce.cuh new file mode 100644 index 000000000000..959748fddb96 --- /dev/null +++ b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasegreduce.cuh @@ -0,0 +1,238 @@ +/****************************************************************************** + * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions are met: + * * Redistributions of source code must retain the above copyright + * notice, this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright + * notice, this list of conditions and the following disclaimer in the + * documentation and/or other materials provided with the distribution. + * * Neither the name of the NVIDIA CORPORATION nor the + * names of its contributors may be used to endorse or promote products + * derived from this software without specific prior written permission. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY + * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND + * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + * + ******************************************************************************/ + +/****************************************************************************** + * + * Code and text by Sean Baxter, NVIDIA Research + * See http://nvlabs.github.io/moderngpu for repository and documentation. + * + ******************************************************************************/ + +#pragma once + +#include "ctasegscan.cuh" +#include "ctasearch.cuh" + +namespace mgpu { + +//////////////////////////////////////////////////////////////////////////////// +// Segmented reduce utility functions. + +// Extract the upper-bound indices from the coded ranges. Decrement to include +// the first addressed row/segment. + +struct SegReduceRange { + int begin; + int end; + int total; + bool flushLast; +}; + +MGPU_DEVICE SegReduceRange DeviceShiftRange(int limit0, int limit1) { + SegReduceRange range; + range.begin = 0x7fffffff & limit0; + range.end = 0x7fffffff & limit1; + range.total = range.end - range.begin; + range.flushLast = 0 == (0x80000000 & limit1); + range.end += !range.flushLast; + return range; +} + +// Reconstitute row/segment indices from a starting row index and packed end +// flags. Used for pre-processed versions of interval reduce and interval Spmv. +template +MGPU_DEVICE void DeviceExpandFlagsToRows(int first, int endFlags, + int rows[VT + 1]) { + + rows[0] = first; + #pragma unroll + for(int i = 0; i < VT; ++i) { + if((1<< i) & endFlags) ++first; + rows[i + 1] = first; + } +} + +//////////////////////////////////////////////////////////////////////////////// +// After loading CSR terms into shared memory, each thread binary searches +// (upper-bound) to find its starting point. Each thread then walks forward, +// emitting the csr0-relative row indices to register. + +template +MGPU_DEVICE int DeviceExpandCsrRows(int tidOffset, int* csr_shared, + int numRows, int end, int rows[VT + 1], int rowStarts[VT]) { + + // Each thread binary searches for its starting row. + int row = BinarySearch(csr_shared, numRows, tidOffset, + mgpu::less()) - 1; + + // Each thread starts at row and scans forward, emitting row IDs into + // register. Store the CTA-local row index (starts at 0) to rows and the + // start of the row (globally) to rowStarts. + int curOffset = csr_shared[row]; + int nextOffset = (row + 1 < numRows) ? csr_shared[row + 1] : end; + + rows[0] = row; + rowStarts[0] = curOffset; + int endFlags = 0; + + #pragma unroll + for(int i = 1; i <= VT; ++i) { + // Advance the row cursor when the iterator hits the next row offset. + if(tidOffset + i == nextOffset) { + // Set an end flag when the cursor advances to the next row. + endFlags |= 1<< (i - 1); + + // Advance the cursor and load the next row offset. + ++row; + curOffset = nextOffset; + nextOffset = (row + 1 < numRows) ? csr_shared[row + 1] : end; + } + rows[i] = row; + if(i < VT) rowStarts[i] = curOffset; + } + __syncthreads(); + + return endFlags; +} + +//////////////////////////////////////////////////////////////////////////////// +// DeviceSegReducePrepare +// Expand non-empty interval of CSR elements into row indices. Compute end-flags +// by comparing adjacent row IDs. + +// DeviceSegReducePrepare may be called either by a pre-processing kernel or by +// the kernel that actually evaluates the segmented reduction if no preprocesing +// is desired. +struct SegReduceTerms { + int endFlags; + int tidDelta; +}; + +template +MGPU_DEVICE SegReduceTerms DeviceSegReducePrepare(int* csr_shared, int numRows, + int tid, int gid, bool flushLast, int rows[VT + 1], int rowStarts[VT]) { + + // Pass a sentinel (end) to point to the next segment start. If we flush, + // this is the end of this tile. Otherwise it is INT_MAX + int endFlags = DeviceExpandCsrRows(gid + VT * tid, csr_shared, + numRows, flushLast ? (gid + NT * VT) : INT_MAX, rows, rowStarts); + + // Find the distance to to scan to compute carry-in for each thread. Use the + // existance of an end flag anywhere in the thread to determine if carry-out + // values from the left should propagate through to the right. + int tidDelta = DeviceFindSegScanDelta(tid, rows[0] != rows[VT], + csr_shared); + + SegReduceTerms terms = { endFlags, tidDelta }; + return terms; +} + +//////////////////////////////////////////////////////////////////////////////// +// CTASegReduce +// Core segmented reduction code. Supports fast-path and slow-path for intra-CTA +// segmented reduction. Stores partials to global memory. +// Callers feed CTASegReduce::ReduceToGlobal values in thread order. +template +struct CTASegReduce { + typedef CTASegScan SegScan; + + enum { + NV = NT * VT, + Capacity = HalfCapacity ? (NV / 2) : NV + }; + + union Storage { + typename SegScan::Storage segScanStorage; + T values[Capacity]; + }; + + template + MGPU_DEVICE static void ReduceToGlobal(const int rows[VT + 1], int total, + int tidDelta, int startRow, int block, int tid, T data[VT], + DestIt dest_global, T* carryOut_global, T identity, Op op, + Storage& storage) { + + // Run a segmented scan within the thread. + T x, localScan[VT]; + #pragma unroll + for(int i = 0; i < VT; ++i) { + x = i ? op(x, data[i]) : data[i]; + localScan[i] = x; + if(rows[i] != rows[i + 1]) x = identity; + } + + // Run a parallel segmented scan over the carry-out values to compute + // carry-in. + T carryOut; + T carryIn = SegScan::SegScanDelta(tid, tidDelta, x, + storage.segScanStorage, &carryOut, identity, op); + + // Store the carry-out for the entire CTA to global memory. + if(!tid) carryOut_global[block] = carryOut; + + dest_global += startRow; + if(HalfCapacity && total > Capacity) { + // Add carry-in to each thread-local scan value. Store directly + // to global. + #pragma unroll + for(int i = 0; i < VT; ++i) { + // Add the carry-in to the local scan. + T x2 = op(carryIn, localScan[i]); + + // Store on the end flag and clear the carry-in. + if(rows[i] != rows[i + 1]) { + carryIn = identity; + dest_global[rows[i]] = x2; + } + } + } else { + // All partials fit in shared memory. Add carry-in to each thread- + // local scan value. + #pragma unroll + for(int i = 0; i < VT; ++i) { + // Add the carry-in to the local scan. + T x2 = op(carryIn, localScan[i]); + + // Store reduction when the segment changes and clear the + // carry-in. + if(rows[i] != rows[i + 1]) { + storage.values[rows[i]] = x2; + carryIn = identity; + } + } + __syncthreads(); + + // Cooperatively store reductions to global memory. + for(int index = tid; index < total; index += NT) + dest_global[index] = storage.values[index]; + __syncthreads(); + } + } +}; + +} // namespace mgpu + diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasegscan.cuh b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasegscan.cuh new file mode 100644 index 000000000000..492d73a54750 --- /dev/null +++ b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasegscan.cuh @@ -0,0 +1,137 @@ +/****************************************************************************** + * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions are met: + * * Redistributions of source code must retain the above copyright + * notice, this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright + * notice, this list of conditions and the following disclaimer in the + * documentation and/or other materials provided with the distribution. + * * Neither the name of the NVIDIA CORPORATION nor the + * names of its contributors may be used to endorse or promote products + * derived from this software without specific prior written permission. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY + * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND + * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + * + ******************************************************************************/ + +/****************************************************************************** + * + * Code and text by Sean Baxter, NVIDIA Research + * See http://nvlabs.github.io/moderngpu for repository and documentation. + * + ******************************************************************************/ + +#pragma once + +#include "ctascan.cuh" + +namespace mgpu { + +//////////////////////////////////////////////////////////////////////////////// +// DeviceFindSegScanDelta +// Runs an inclusive max-index scan over binary inputs. + +template +MGPU_DEVICE int DeviceFindSegScanDelta(int tid, bool flag, int* delta_shared) { + const int NumWarps = NT / 32; + + int warp = tid / 32; + int lane = 31 & tid; + uint warpMask = 0xffffffff>> (31 - lane); // inclusive search + uint ctaMask = 0x7fffffff>> (31 - lane); // exclusive search + + uint warpBits = __ballot(flag); + delta_shared[warp] = warpBits; + __syncthreads(); + + if(tid < NumWarps) { + uint ctaBits = __ballot(0 != delta_shared[tid]); + int warpSegment = 31 - clz(ctaMask & ctaBits); + int start = (-1 != warpSegment) ? + (31 - clz(delta_shared[warpSegment]) + 32 * warpSegment) : 0; + delta_shared[NumWarps + tid] = start; + } + __syncthreads(); + + // Find the closest flag to the left of this thread within the warp. + // Include the flag for this thread. + int start = 31 - clz(warpMask & warpBits); + if(-1 != start) start += ~31 & tid; + else start = delta_shared[NumWarps + warp]; + __syncthreads(); + + return tid - start; +} + +//////////////////////////////////////////////////////////////////////////////// +// CTASegScan + +template > +struct CTASegScan { + typedef _Op Op; + typedef typename Op::result_type T; + enum { NumWarps = NT / 32, Size = NT, Capacity = 2 * NT }; + union Storage { + int delta[NumWarps]; + T values[Capacity]; + }; + + // Each thread passes the reduction of the LAST SEGMENT that it covers. + // flag is set to true if there's at least one segment flag in the thread. + // SegScan returns the reduction of values for the first segment in this + // thread over the preceding threads. + // Return the value init for the first thread. + + // When scanning single elements per thread, interpret the flag as a BEGIN + // FLAG. If tid's flag is set, its value belongs to thread tid + 1, not + // thread tid. + + // The function returns the reduction of the last segment in the CTA. + + MGPU_DEVICE static T SegScanDelta(int tid, int tidDelta, T x, + Storage& storage, T* carryOut, T identity = (T)0, Op op = Op()) { + + // Run an inclusive scan + int first = 0; + storage.values[first + tid] = x; + __syncthreads(); + + #pragma unroll + for(int offset = 1; offset < NT; offset += offset) { + if(tidDelta >= offset) + x = op(storage.values[first + tid - offset], x); + first = NT - first; + storage.values[first + tid] = x; + __syncthreads(); + } + + // Get the exclusive scan. + x = tid ? storage.values[first + tid - 1] : identity; + *carryOut = storage.values[first + NT - 1]; + __syncthreads(); + return x; + } + + MGPU_DEVICE static T SegScan(int tid, T x, bool flag, Storage& storage, + T* carryOut, T identity = (T)0, Op op = Op()) { + + // Find the left-most thread that covers the first segment of this + // thread. + int tidDelta = DeviceFindSegScanDelta(tid, flag, storage.delta); + + return SegScanDelta(tid, tidDelta, x, storage, carryOut, identity, op); + } +}; + +} // namespace mgpu diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasegsort.cuh b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasegsort.cuh new file mode 100644 index 000000000000..6a9516ae446e --- /dev/null +++ b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasegsort.cuh @@ -0,0 +1,443 @@ +/****************************************************************************** + * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions are met: + * * Redistributions of source code must retain the above copyright + * notice, this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright + * notice, this list of conditions and the following disclaimer in the + * documentation and/or other materials provided with the distribution. + * * Neither the name of the NVIDIA CORPORATION nor the + * names of its contributors may be used to endorse or promote products + * derived from this software without specific prior written permission. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY + * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND + * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + * + ******************************************************************************/ + +/****************************************************************************** + * + * Code and text by Sean Baxter, NVIDIA Research + * See http://nvlabs.github.io/moderngpu for repository and documentation. + * + ******************************************************************************/ + +#pragma once + +#include "ctascan.cuh" +#include "ctasearch.cuh" +#include "loadstore.cuh" +#include "sortnetwork.cuh" + +namespace mgpu { + +template +MGPU_DEVICE void SegmentedSerialMerge(const T* keys_shared, int aBegin, + int aEnd, int bBegin, int bEnd, T results[VT], int indices[VT], + int leftEnd, int rightStart, Comp comp, bool sync = true) { + + bEnd = min(rightStart, bEnd); + T aKey = keys_shared[aBegin]; + T bKey = keys_shared[bBegin]; + + #pragma unroll + for(int i = 0; i < VT; ++i) { + bool p; + + // If A has run out of inputs, emit B. + if(aBegin >= aEnd) + p = false; + else if(bBegin >= bEnd || aBegin < leftEnd) + // B has hit the end of the middle segment. + // Emit A if A has inputs remaining in the middle segment. + p = true; + else + // Emit the smaller element in the middle segment. + p = !comp(bKey, aKey); + + results[i] = p ? aKey : bKey; + indices[i] = p ? aBegin : bBegin; + if(p) aKey = keys_shared[++aBegin]; + else bKey = keys_shared[++bBegin]; + } + if(sync) { __syncthreads(); } +} + +//////////////////////////////////////////////////////////////////////////////// +// CTASegsortPass + +template +MGPU_DEVICE void CTASegsortPass(T* keys_shared, int* ranges_shared, int tid, + int pass, T results[VT], int indices[VT], int2& activeRange, Comp comp) { + + // Locate the intervals of the input lists. + int3 frame = FindMergesortFrame(2<< pass, tid, VT); + int a0 = frame.x; + int b0 = frame.y; + int listLen = frame.z; + int list = tid>> pass; + int listParity = 1 & list; + int diag = VT * tid - frame.x; + + // Fetch the active range for the list this thread's list is merging with. + int siblingRange = ranges_shared[1 ^ list]; + int siblingStart = 0x0000ffff & siblingRange; + int siblingEnd = siblingRange>> 16; + + // Create a new active range for the merge. + int leftEnd = listParity ? siblingEnd : activeRange.y; + int rightStart = listParity ? activeRange.x : siblingStart; + activeRange.x = min(activeRange.x, siblingStart); + activeRange.y = max(activeRange.y, siblingEnd); + + int p = SegmentedMergePath(keys_shared, a0, listLen, b0, listLen, leftEnd, + rightStart, diag, comp); + + int a0tid = a0 + p; + int b0tid = b0 + diag - p; + SegmentedSerialMerge(keys_shared, a0tid, b0, b0tid, b0 + listLen, + results, indices, leftEnd, rightStart, comp); + + // Store the ranges to shared memory. + if(0 == diag) + ranges_shared[list>> 1] = + (int)bfi(activeRange.y, activeRange.x, 16, 16); +} + +//////////////////////////////////////////////////////////////////////////////// +// CTASegsortLoop + +template +MGPU_DEVICE int2 CTASegsortLoop(KeyType threadKeys[VT], + ValType threadValues[VT], KeyType* keys_shared, ValType* values_shared, + int* ranges_shared, int tid, int2 activeRange, Comp comp) { + + const int NumPasses = sLogPow2::value; + #pragma unroll + for(int pass = 0; pass < NumPasses; ++pass) { + int indices[VT]; + CTASegsortPass(keys_shared, ranges_shared, tid, pass, + threadKeys, indices, activeRange, comp); + + if(HasValues) { + // Exchange values through shared memory. + DeviceThreadToShared(threadValues, tid, values_shared); + DeviceGather(NT * VT, values_shared, indices, tid, + threadValues); + } + + // Store results in shared memory in sorted order. + DeviceThreadToShared(threadKeys, tid, keys_shared); + } + return activeRange; +} + +//////////////////////////////////////////////////////////////////////////////// +// CTASegsort +// Pass keys and values in register. On return, values are returned in register +// and keys returned in shared memory. + +template +MGPU_DEVICE int2 CTASegsort(KeyType threadKeys[VT], ValType threadValues[VT], + int tid, int headFlags, KeyType* keys_shared, ValType* values_shared, + int* ranges_shared, Comp comp) { + + if(Stable) + // Odd-even transpose sort. + OddEvenTransposeSortFlags(threadKeys, threadValues, headFlags, + comp); + else + // Batcher's odd-even mergesort. + OddEvenMergesortFlags(threadKeys, threadValues, headFlags, comp); + + // Record the first and last occurrence of head flags in this segment. + int blockEnd = 31 - clz(headFlags); + if(-1 != blockEnd) blockEnd += VT * tid; + + int blockStart = ffs(headFlags); + blockStart = blockStart ? (VT * tid - 1 + blockStart) : (NT * VT); + + ranges_shared[tid] = (int)bfi(blockEnd, blockStart, 16, 16); + + // Store back to shared mem. The values are in VT-length sorted lists. + // These are merged recursively. + DeviceThreadToShared(threadKeys, tid, keys_shared); + + int2 activeRange = CTASegsortLoop(threadKeys, + threadValues, keys_shared, values_shared, ranges_shared, tid, + make_int2(blockStart, blockEnd), comp); + return activeRange; +} + + +template +MGPU_DEVICE int2 CTASegsortKeys(KeyType threadKeys[VT], int tid, int headFlags, + KeyType* keys_shared, int* ranges_shared, Comp comp) { + + int valuesTemp[VT]; + return CTASegsort(threadKeys, valuesTemp, tid, + headFlags, keys_shared, (int*)keys_shared, ranges_shared, comp); +} + +template +MGPU_DEVICE int2 CTASegsortPairs(KeyType threadKeys[VT], + ValType threadValues[VT], int tid, int headFlags, KeyType* keys_shared, + ValType* values_shared, int* ranges_shared, Comp comp) { + + return CTASegsort(threadKeys, threadValues, tid, + headFlags, keys_shared, values_shared, ranges_shared, comp); +} + +//////////////////////////////////////////////////////////////////////////////// +// DeviceSegBlocksort +// Load keys and values from global memory, sort in shared memory, and store +// back to global memory. Store the left-most and right-most encountered +// headflag locations to ranges_global to prepare for the next pass. +// This function is factored out of the blocksort kernel to allow easier +// customization of that kernel - we have two implementations currently: +// sort over indices and sort over bitfield. + +template +MGPU_DEVICE void DeviceSegBlocksort(InputIt1 keys_global, + InputIt2 values_global, int count2, KeyType* keys_shared, + ValType* values_shared, int* ranges_shared, int headFlags, int tid, + int block, OutputIt1 keysDest_global, OutputIt2 valsDest_global, + int* ranges_global, Comp comp) { + + // Load keys into register in thread order. + int gid = NT * VT * block; + KeyType threadKeys[VT]; + DeviceGlobalToShared(count2, keys_global + gid, tid, keys_shared); + DeviceSharedToThread(keys_shared, tid, threadKeys); + + // Load the values from global memory and into register in thread order. + ValType threadValues[VT]; + if(HasValues) { + DeviceGlobalToShared(count2, values_global + gid, tid, + values_shared); + DeviceSharedToThread(values_shared, tid, threadValues); + } + + // Run the CTA segmented blocksort. + int2 activeRange = CTASegsort(threadKeys, + threadValues, tid, headFlags, keys_shared, values_shared, ranges_shared, + comp); + + // Store the keys to global memory. + DeviceSharedToGlobal(count2, keys_shared, tid, + keysDest_global + gid); + + if(HasValues) { + // Store the values to global memory.xk b + DeviceThreadToShared(threadValues, tid, values_shared); + DeviceSharedToGlobal(count2, values_shared, tid, + valsDest_global + gid, false); + } + + // Store the 16-bit packed ranges. These are used by all merge kernels and + // the first level of global segmented merge path partitioning. + if(!tid) + ranges_global[block] = bfi(activeRange.y, activeRange.x, 16, 16); +} + +//////////////////////////////////////////////////////////////////////////////// +// DeviceIndicesToHeadFlags +// Load indices from an array and cooperatively turn into a head flag bitfield +// for each thread. + +template +MGPU_DEVICE int DeviceIndicesToHeadFlags(const int* indices_global, + const int* partitions_global, int tid, int block, int count2, + int* words_shared, byte* flags_shared) { + + const int FlagWordsPerThread = MGPU_DIV_UP(VT, 4); + int gid = NT * VT * block; + int p0 = partitions_global[block]; + int p1 = partitions_global[block + 1]; + + int headFlags = 0; + if(p1 > p0 || count2 < NT * VT) { + + // Clear the flag bytes, then loop through the indices and poke in flag + // values. + #pragma unroll + for(int i = 0; i < FlagWordsPerThread; ++i) + words_shared[NT * i + tid] = 0; + __syncthreads(); + + for(int index = p0 + tid; index < p1; index += NT) { + int headFlag = indices_global[index]; + flags_shared[headFlag - gid] = 1; + } + __syncthreads(); + + // Combine all the head flags for this thread. + int first = VT * tid; + int offset = first / 4; + int prev = words_shared[offset]; + int mask = 0x3210 + 0x1111 * (3 & first); + #pragma unroll + for(int i = 0; i < FlagWordsPerThread; ++i) { + // Gather the next four flags. + int next = words_shared[offset + 1 + i]; + int x = prmt(prev, next, mask); + prev = next; + + // Set the head flag bits. + if(0x00000001 & x) headFlags |= 1<< (4 * i); + if(0x00000100 & x) headFlags |= 1<< (4 * i + 1); + if(0x00010000 & x) headFlags |= 1<< (4 * i + 2); + if(0x01000000 & x) headFlags |= 1<< (4 * i + 3); + } + __syncthreads(); + + // Set head flags for out-of-range keys. + int outOfRange = min(VT, first + VT - count2); + if(outOfRange > 0) + headFlags = bfi(0xffffffff, headFlags, VT - outOfRange, outOfRange); + + // Clear head flags above VT. + headFlags &= (1<< VT) - 1; + } + return headFlags; +} + +//////////////////////////////////////////////////////////////////////////////// +// SegSortSupport + +struct SegSortSupport { + int* ranges_global; + int2* ranges2_global; + + int4* mergeList_global; + int* copyList_global; + int2* queueCounters_global; + int2* nextCounters_global; + + byte* copyStatus_global; +}; + +//////////////////////////////////////////////////////////////////////////////// +// DeviceSegSortMerge + +template +MGPU_DEVICE void DeviceSegSortMerge(const KeyType* keys_global, + const ValueType* values_global, int2 segmentRange, int tid, + int block, int4 range, int pass, KeyType* keys_shared, + int* indices_shared, KeyType* keysDest_global, ValueType* valsDest_global, + Comp comp) { + + const int NV = NT * VT; + int gid = NV * block; + + // Load the local compressed segment indices. + int a0 = range.x; + int aCount = range.y - range.x; + int b0 = range.z; + int bCount = range.w - range.z; + + DeviceLoad2ToShared(keys_global + a0, aCount, keys_global + b0, + bCount, tid, keys_shared); + + //////////////////////////////////////////////////////////////////////////// + // Run a merge path to find the starting point for each thread to merge. + // If the entire warp fits into the already-sorted segments, we can skip + // sorting it and leave its keys in shared memory. Doing this on the warp + // level rather than thread level (also legal) gives slightly better + // performance. + + int segStart = segmentRange.x; + int segEnd = segmentRange.y; + int listParity = 1 & (block>> pass); + + int warpOffset = VT * (~31 & tid); + bool sortWarp = listParity ? + // The spliced segment is to the left (segStart). + (warpOffset < segStart) : + // The spliced segment is to the right (segEnd). + (warpOffset + 32 * VT > segEnd); + + KeyType threadKeys[VT]; + int indices[VT]; + if(sortWarp) { + int diag = VT * tid; + int mp = SegmentedMergePath(keys_shared, 0, aCount, aCount, bCount, + listParity ? 0 : segEnd, listParity ? segStart : NV, diag, comp); + int a0tid = mp; + int a1tid = aCount; + int b0tid = aCount + diag - mp; + int b1tid = aCount + bCount; + + // Serial merge into register. All threads in the CTA so we hoist the + // check for list parity outside the function call to simplify the + // logic. Unlike in the blocksort, this does not cause warp divergence. + SegmentedSerialMerge(keys_shared, a0tid, a1tid, b0tid, b1tid, + threadKeys, indices, listParity ? 0 : segEnd, + listParity ? segStart : NV, comp, false); + } + __syncthreads(); + + // Store sorted data in register back to shared memory. Then copy to global. + if(sortWarp) + DeviceThreadToShared(threadKeys, tid, keys_shared, false); + __syncthreads(); + + DeviceSharedToGlobal(aCount + bCount, keys_shared, tid, + keysDest_global + gid); + + //////////////////////////////////////////////////////////////////////////// + // Use the merge indices to gather values from global memory. Store directly + // to valsDest_global. + + if(HasValues) { + // Transpose the gather indices to help coalesce loads. + if(sortWarp) + DeviceThreadToShared(indices, tid, indices_shared, false); + else { + #pragma unroll + for(int i = 0; i < VT; ++i) + indices_shared[VT * tid + i] = VT * tid + i; + } + __syncthreads(); + + DeviceTransferMergeValuesShared(aCount + bCount, + values_global + a0, values_global + b0, aCount, indices_shared, + tid, valsDest_global + NV * block); + } +} + +//////////////////////////////////////////////////////////////////////////////// +// DeviceSegSortCopy + +template +MGPU_DEVICE void DeviceSegSortCopy(const KeyType* keys_global, + const ValueType* values_global, int tid, int block, int count, + KeyType* keysDest_global, ValueType* valsDest_global) { + + int gid = NT * VT * block; + int count2 = min(NT * VT, count - gid); + + DeviceGlobalToGlobal(count2, keys_global + gid, tid, + keysDest_global + gid); + if(HasValues) + DeviceGlobalToGlobal(count2, values_global + gid, tid, + valsDest_global + gid); +} + +} // namespace mgpu diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasortedsearch.cuh b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasortedsearch.cuh new file mode 100644 index 000000000000..48b7f3a8fe10 --- /dev/null +++ b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasortedsearch.cuh @@ -0,0 +1,208 @@ +/****************************************************************************** + * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions are met: + * * Redistributions of source code must retain the above copyright + * notice, this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright + * notice, this list of conditions and the following disclaimer in the + * documentation and/or other materials provided with the distribution. + * * Neither the name of the NVIDIA CORPORATION nor the + * names of its contributors may be used to endorse or promote products + * derived from this software without specific prior written permission. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY + * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND + * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + * + ******************************************************************************/ + +/****************************************************************************** + * + * Code and text by Sean Baxter, NVIDIA Research + * See http://nvlabs.github.io/moderngpu for repository and documentation. + * + ******************************************************************************/ + +#pragma once + +#include "../mgpudevice.cuh" +#include "ctasearch.cuh" + +namespace mgpu { + + +//////////////////////////////////////////////////////////////////////////////// +// DeviceSerialSearch + +template +MGPU_DEVICE int3 DeviceSerialSearch(const T* keys_shared, int aBegin, + int aEnd, int bBegin, int bEnd, int aOffset, int bOffset, int* indices, + Comp comp) { + + const int FlagA = IndexA ? 0x80000000 : 1; + const int FlagB = IndexB ? 0x80000000 : 1; + + T aKey = keys_shared[aBegin]; + T bKey = keys_shared[bBegin]; + T aPrev, bPrev; + if(aBegin > 0) aPrev = keys_shared[aBegin - 1]; + if(bBegin > 0) bPrev = keys_shared[bBegin - 1]; + int decisions = 0; + int matchCountA = 0; + int matchCountB = 0; + + #pragma unroll + for(int i = 0; i < VT; ++i) { + bool p; + if(RangeCheck && aBegin >= aEnd) p = false; + else if(RangeCheck && bBegin >= bEnd) p = true; + else p = (MgpuBoundsUpper == Bounds) ? + comp(aKey, bKey) : + !comp(bKey, aKey); + + if(p) { + // aKey is smaller than bKey, so it is inserted before bKey. + // Save bKey's index (bBegin + first) as the result of the search + // and advance to the next needle in A. + bool match = false; + if(MatchA) { + // Test if there is an element in B that matches aKey. + if(MgpuBoundsUpper == Bounds) { + // Upper Bound: We're inserting aKey after bKey. If there + // is a match for aKey it must be bPrev. Check that bPrev + // is in range and equal to aKey. + // The predicate test result !comp(aKey, bPrev) was + // established on the previous A-advancing iteration (it + // failed the comp(aKey, bKey) test to get us to this + // point). Check the other half of the equality condition + // with a second comparison. + bool inRange = !RangeCheck || (bBegin > aEnd); + match = inRange && !comp(bPrev, aKey); + } else { + // Lower Bound: We're inserting aKey before bKey. If there + // is a match for aKey, it must be bKey. Check that bKey + // is in range and equal to aKey. + // The predicate test !comp(bKey, aKey) has established one + // half of the equality condition. We establish the other + // half with a second comparison. + bool inRange = !RangeCheck || (bBegin < bEnd); + match = inRange && !comp(aKey, bKey); + } + } + + int index = 0; + if(IndexA) index = bOffset + bBegin; + if(match) index |= FlagA; + if(IndexA || MatchA) indices[i] = index; + matchCountA += match; + + // Mark the decision bit to indicate that this iteration has + // progressed A (the needles). + decisions |= 1<< i; + aPrev = aKey; + aKey = keys_shared[++aBegin]; + } else { + // aKey is larger than bKey, so it is inserted after bKey (but we + // don't know where yet). Advance the B index to the next element in + // the haystack to continue the search for the current needle. + bool match = false; + if(MatchB) { + if(MgpuBoundsUpper == Bounds) { + // Upper Bound: aKey is not smaller than bKey. We advance to + // the next haystack element in B. If there is a match in A + // for bKey it must be aKey. By entering this branch we've + // verified that !comp(aKey, bKey). Making the reciprocal + // comparison !comp(bKey, aKey) establishes aKey == bKey. + bool inRange = !RangeCheck || + ((bBegin < bEnd) && (aBegin < aEnd)); + match = inRange && !comp(bKey, aKey); + } else { + // Lower Bound: bKey is smaller than aKey. We advance to the + // next element in B. If there is a match for bKey, it must + // be aPrev. The previous A-advancing iteration proved that + // !comp(bKey, aPrev). We test !comp(aPrev, bKey) for the + // other half of the equality condition. + bool inRange = !RangeCheck || + ((bBegin < bEnd) && (aBegin > 0)); + match = inRange && !comp(aPrev, bKey); + } + } + + int index = 0; + if(IndexB) index = aOffset + aBegin; + if(match) index |= FlagB; + if(IndexB || MatchB) indices[i] = index; + matchCountB += match; + + // Keep the decision bit cleared to indicate that this iteration + // has progressed B (the haystack). + bPrev = bKey; + bKey = keys_shared[++bBegin]; + } + } + return make_int3(decisions, matchCountA, matchCountB); +} + +//////////////////////////////////////////////////////////////////////////////// +// CTASortedSearch +// Take keys in shared memory and return indices and b-match flags in shared +// memory. +// NOTE: This function doesn't do any strided-to-thread order transposes so +// using an even number of values per thread will incur no additional bank +// conflicts. + +template +MGPU_DEVICE int2 CTASortedSearch(T* keys_shared, int aStart, int aCount, + int aEnd, int a0, int bStart, int bCount, int bEnd, int b0, bool extended, + int tid, int* indices_shared, Comp comp) { + + // Run a merge path to find the start of the serial search for each thread. + int diag = VT * tid; + int mp = MergePath(keys_shared + aStart, aCount, + keys_shared + bStart, bCount, diag, comp); + int a0tid = mp; + int b0tid = diag - mp; + + // Serial search into register. + int3 results; + int indices[VT]; + if(extended) + results = DeviceSerialSearch(keys_shared, a0tid + aStart, aEnd, b0tid + bStart, bEnd, + a0 - aStart, b0 - bStart, indices, comp); + else + results = DeviceSerialSearch(keys_shared, a0tid + aStart, aEnd, b0tid + bStart, bEnd, + a0 - aStart, b0 - bStart, indices, comp); + __syncthreads(); + + // Compact the indices into shared memory. Use the decision bits (set is A, + // cleared is B) to select the destination. + int decisions = results.x; + b0tid += aCount; + #pragma unroll + for(int i = 0; i < VT; ++i) { + if((1<< i) & decisions) { + if(IndexA || MatchA) indices_shared[a0tid++] = indices[i]; + } else { + if(IndexB || MatchB) indices_shared[b0tid++] = indices[i]; + } + } + __syncthreads(); + + // Return the match counts for A and B keys. + return make_int2(results.y, results.z); +} + +} // namespace mgpu diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/devicetypes.cuh b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/devicetypes.cuh new file mode 100644 index 000000000000..282bedd5ab23 --- /dev/null +++ b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/devicetypes.cuh @@ -0,0 +1,363 @@ +/****************************************************************************** + * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions are met: + * * Redistributions of source code must retain the above copyright + * notice, this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright + * notice, this list of conditions and the following disclaimer in the + * documentation and/or other materials provided with the distribution. + * * Neither the name of the NVIDIA CORPORATION nor the + * names of its contributors may be used to endorse or promote products + * derived from this software without specific prior written permission. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY + * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND + * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + * + ******************************************************************************/ + +/****************************************************************************** + * + * Code and text by Sean Baxter, NVIDIA Research + * See http://nvlabs.github.io/moderngpu for repository and documentation. + * + ******************************************************************************/ + +#pragma once + +#if __CUDA_ARCH__ == 100 + #error "COMPUTE CAPABILITY 1.0 NOT SUPPORTED BY MPGU. TRY 2.0!" +#endif + +#include +#include "../util/static.h" + +#ifdef _MSC_VER +#define INLINESYMBOL __forceinline__ +#else +#define INLINESYMBOL inline +#endif + +namespace mgpu { + +#define MGPU_HOST __host__ INLINESYMBOL +#define MGPU_DEVICE __device__ INLINESYMBOL +#define MGPU_HOST_DEVICE __host__ __device__ INLINESYMBOL + +const int WARP_SIZE = 32; +const int LOG_WARP_SIZE = 5; + +//////////////////////////////////////////////////////////////////////////////// +// Device-side comparison operators + +template +struct less : public std::binary_function { + MGPU_HOST_DEVICE bool operator()(T a, T b) { return a < b; } +}; +template +struct less_equal : public std::binary_function { + MGPU_HOST_DEVICE bool operator()(T a, T b) { return a <= b; } +}; +template +struct greater : public std::binary_function { + MGPU_HOST_DEVICE bool operator()(T a, T b) { return a > b; } +}; +template +struct greater_equal : public std::binary_function { + MGPU_HOST_DEVICE bool operator()(T a, T b) { return a >= b; } +}; +template +struct equal_to : public std::binary_function { + MGPU_HOST_DEVICE bool operator()(T a, T b) { return a == b; } +}; +template +struct not_equal_to : public std::binary_function { + MGPU_HOST_DEVICE bool operator()(T a, T b) { return a != b; } +}; + +//////////////////////////////////////////////////////////////////////////////// +// Device-side arithmetic operators + +template +struct plus : public std::binary_function { + MGPU_HOST_DEVICE T operator()(T a, T b) { return a + b; } +}; + +template +struct minus : public std::binary_function { + MGPU_HOST_DEVICE T operator()(T a, T b) { return a - b; } +}; + +template +struct multiplies : public std::binary_function { + MGPU_HOST_DEVICE T operator()(T a, T b) { return a * b; } +}; + +template +struct modulus : public std::binary_function { + MGPU_HOST_DEVICE T operator()(T a, T b) { return a % b; } +}; + +template +struct bit_or : public std::binary_function { + MGPU_HOST_DEVICE T operator()(T a, T b) { return a | b; } +}; + +template +struct bit_and : public std::binary_function { + MGPU_HOST_DEVICE T operator()(T a, T b) { return a & b; } +}; + +template +struct bit_xor : public std::binary_function { + MGPU_HOST_DEVICE T operator()(T a, T b) { return a ^ b; } +}; + +template +struct maximum : public std::binary_function { + MGPU_HOST_DEVICE T operator()(T a, T b) { return max(a, b); } +}; + +template +struct minimum : public std::binary_function { + MGPU_HOST_DEVICE T operator()(T a, T b) { return min(a, b); } +}; + +//////////////////////////////////////////////////////////////////////////////// + +template +MGPU_HOST_DEVICE void swap(T& a, T& b) { + T c = a; + a = b; + b = c; +} + +template +struct DevicePair { + T x, y; +}; + +template +MGPU_HOST_DEVICE DevicePair MakeDevicePair(T x, T y) { + DevicePair p = { x, y }; + return p; +} + +template struct numeric_limits; +template<> struct numeric_limits { + MGPU_HOST_DEVICE static int min() { return INT_MIN; } + MGPU_HOST_DEVICE static int max() { return INT_MAX; } + MGPU_HOST_DEVICE static int lowest() { return INT_MIN; } + MGPU_HOST_DEVICE static int AddIdent() { return 0; } + MGPU_HOST_DEVICE static int MulIdent() { return 1; } +}; +template<> struct numeric_limits { + MGPU_HOST_DEVICE static long long min() { return LLONG_MIN; } + MGPU_HOST_DEVICE static long long max() { return LLONG_MAX; } + MGPU_HOST_DEVICE static long long lowest() { return LLONG_MIN; } + MGPU_HOST_DEVICE static long long AddIdent() { return 0; } + MGPU_HOST_DEVICE static long long MulIdent() { return 1; } +}; +template<> struct numeric_limits { + MGPU_HOST_DEVICE static uint min() { return 0; } + MGPU_HOST_DEVICE static uint max() { return UINT_MAX; } + MGPU_HOST_DEVICE static uint lowest() { return 0; } + MGPU_HOST_DEVICE static uint AddIdent() { return 0; } + MGPU_HOST_DEVICE static uint MulIdent() { return 1; } +}; +template<> struct numeric_limits { + MGPU_HOST_DEVICE static unsigned long long min() { return 0; } + MGPU_HOST_DEVICE static unsigned long long max() { return ULLONG_MAX; } + MGPU_HOST_DEVICE static unsigned long long lowest() { return 0; } + MGPU_HOST_DEVICE static unsigned long long AddIdent() { return 0; } + MGPU_HOST_DEVICE static unsigned long long MulIdent() { return 1; } +}; +template<> struct numeric_limits { + MGPU_HOST_DEVICE static float min() { return FLT_MIN; } + MGPU_HOST_DEVICE static float max() { return FLT_MAX; } + MGPU_HOST_DEVICE static float lowest() { return -FLT_MAX; } + MGPU_HOST_DEVICE static float AddIdent() { return 0; } + MGPU_HOST_DEVICE static float MulIdent() { return 1; } +}; +template<> struct numeric_limits { + MGPU_HOST_DEVICE static double min() { return DBL_MIN; } + MGPU_HOST_DEVICE static double max() { return DBL_MAX; } + MGPU_HOST_DEVICE static double lowest() { return -DBL_MAX; } + MGPU_HOST_DEVICE static double AddIdent() { return 0; } + MGPU_HOST_DEVICE static double MulIdent() { return 1; } +}; + + +MGPU_HOST_DEVICE int2 operator+(int2 a, int2 b) { + return make_int2(a.x + b.x, a.y + b.y); +} +MGPU_HOST_DEVICE int2& operator+=(int2& a, int2 b) { + a = a + b; + return a; +} +MGPU_HOST_DEVICE int2 operator*(int2 a, int2 b) { + return make_int2(a.x * b.x, a.y * b.y); +} +MGPU_HOST_DEVICE int2& operator*=(int2& a, int2 b) { + a = a * b; + return a; +} + +template +MGPU_HOST_DEVICE T max(T a, T b) { +#if !defined(__CUDA_ARCH__) || (__CUDA_ARCH__ < 100) + return std::max(a, b); +#else + return (a < b) ? b : a; +#endif +} +template +MGPU_HOST_DEVICE T min(T a, T b) { +#if !defined(__CUDA_ARCH__) || (__CUDA_ARCH__ < 100) + return std::min(a, b); +#else + return (b < a) ? b : a; +#endif +} + +MGPU_HOST_DEVICE int2 max(int2 a, int2 b) { + return make_int2(max(a.x, b.x), max(a.y, b.y)); +} + +MGPU_HOST_DEVICE int2 min(int2 a, int2 b) { + return make_int2(min(a.x, b.x), min(a.y, b.y)); +} + +template<> struct numeric_limits { + MGPU_HOST_DEVICE static int2 min() { return make_int2(INT_MIN, INT_MIN); } + MGPU_HOST_DEVICE static int2 max() { return make_int2(INT_MAX, INT_MAX); } + MGPU_HOST_DEVICE static int2 lowest() { + return make_int2(INT_MIN, INT_MIN); + } + MGPU_HOST_DEVICE static int2 AddIdent() { return make_int2(0, 0); } + MGPU_HOST_DEVICE static int2 MulIdent() { return make_int2(1, 1); } +}; + +template +class constant_iterator : public std::iterator_traits { +public: + MGPU_HOST_DEVICE constant_iterator(T value) : _value(value) { } + + MGPU_HOST_DEVICE T operator[](ptrdiff_t i) const { + return _value; + } + MGPU_HOST_DEVICE T operator*() const { + return _value; + } + MGPU_HOST_DEVICE constant_iterator operator+(ptrdiff_t diff) const { + return constant_iterator(_value); + } + MGPU_HOST_DEVICE constant_iterator operator-(ptrdiff_t diff) const { + return constant_iterator(_value); + } + MGPU_HOST_DEVICE constant_iterator& operator+=(ptrdiff_t diff) { + return *this; + } + MGPU_HOST_DEVICE constant_iterator& operator-=(ptrdiff_t diff) { + return *this; + } +private: + T _value; +}; + +template +class counting_iterator : public std::iterator_traits { +public: + MGPU_HOST_DEVICE counting_iterator(T value) : _value(value) { } + + MGPU_HOST_DEVICE T operator[](ptrdiff_t i) { + return _value + i; + } + MGPU_HOST_DEVICE T operator*() { + return _value; + } + MGPU_HOST_DEVICE counting_iterator operator+(ptrdiff_t diff) { + return counting_iterator(_value + diff); + } + MGPU_HOST_DEVICE counting_iterator operator-(ptrdiff_t diff) { + return counting_iterator(_value - diff); + } + MGPU_HOST_DEVICE counting_iterator& operator+=(ptrdiff_t diff) { + _value += diff; + return *this; + } + MGPU_HOST_DEVICE counting_iterator& operator-=(ptrdiff_t diff) { + _value -= diff; + return *this; + } +private: + T _value; +}; + +template +class step_iterator : public std::iterator_traits { +public: + MGPU_HOST_DEVICE step_iterator(T base, T step) : + _base(base), _step(step), _offset(0) { } + + MGPU_HOST_DEVICE T operator[](ptrdiff_t i) { + return _base + (_offset + i) * _step; + } + MGPU_HOST_DEVICE T operator*() { + return _base + _offset * _step; + } + MGPU_HOST_DEVICE step_iterator operator+(ptrdiff_t diff) { + step_iterator it = *this; + it._offset += diff; + return it; + } + MGPU_HOST_DEVICE step_iterator operator-(ptrdiff_t diff) { + step_iterator it = *this; + it._offset -= diff; + return it; + } + MGPU_HOST_DEVICE step_iterator& operator+=(ptrdiff_t diff) { + _offset += diff; + return *this; + } + MGPU_HOST_DEVICE step_iterator& operator-=(ptrdiff_t diff) { + _offset -= diff; + return *this; + } +private: + ptrdiff_t _offset; + T _base, _step; +}; + +} // namespace mgpu + + +template +MGPU_HOST_DEVICE mgpu::counting_iterator operator+(ptrdiff_t diff, + mgpu::counting_iterator it) { + return it + diff; +} +template +MGPU_HOST_DEVICE mgpu::counting_iterator operator-(ptrdiff_t diff, + mgpu::counting_iterator it) { + return it + (-diff); +} +template +MGPU_HOST_DEVICE mgpu::step_iterator operator+(ptrdiff_t diff, + mgpu::step_iterator it) { + return it + diff; +} +template +MGPU_HOST_DEVICE mgpu::step_iterator operator-(ptrdiff_t diff, + mgpu::step_iterator it) { + return it + (-diff); +} diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/deviceutil.cuh b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/deviceutil.cuh new file mode 100644 index 000000000000..e18807f38496 --- /dev/null +++ b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/deviceutil.cuh @@ -0,0 +1,143 @@ +/****************************************************************************** + * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions are met: + * * Redistributions of source code must retain the above copyright + * notice, this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright + * notice, this list of conditions and the following disclaimer in the + * documentation and/or other materials provided with the distribution. + * * Neither the name of the NVIDIA CORPORATION nor the + * names of its contributors may be used to endorse or promote products + * derived from this software without specific prior written permission. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY + * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND + * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + * + ******************************************************************************/ + +/****************************************************************************** + * + * Code and text by Sean Baxter, NVIDIA Research + * See http://nvlabs.github.io/moderngpu for repository and documentation. + * + ******************************************************************************/ + +#pragma once + +#include "intrinsics.cuh" + +namespace mgpu { + +// Get the difference between two pointers in bytes. +MGPU_HOST_DEVICE ptrdiff_t PtrDiff(const void* a, const void* b) { + return (const byte*)b - (const byte*)a; +} + +// Offset a pointer by i bytes. +template +MGPU_HOST_DEVICE const T* PtrOffset(const T* p, ptrdiff_t i) { + return (const T*)((const byte*)p + i); +} +template +MGPU_HOST_DEVICE T* PtrOffset(T* p, ptrdiff_t i) { + return (T*)((byte*)p + i); +} + +//////////////////////////////////////////////////////////////////////////////// +// Task range support +// Evenly distributes variable-length arrays over a fixed number of CTAs. + +MGPU_HOST int2 DivideTaskRange(int numItems, int numWorkers) { + div_t d = div(numItems, numWorkers); + return make_int2(d.quot, d.rem); +} + +MGPU_HOST_DEVICE int2 ComputeTaskRange(int block, int2 task) { + int2 range; + range.x = task.x * block; + range.x += min(block, task.y); + range.y = range.x + task.x + (block < task.y); + return range; +} + +MGPU_HOST_DEVICE int2 ComputeTaskRange(int block, int2 task, int blockSize, + int count) { + int2 range = ComputeTaskRange(block, task); + range.x *= blockSize; + range.y = min(count, range.y * blockSize); + return range; +} + +//////////////////////////////////////////////////////////////////////////////// +// DeviceExtractHeadFlags +// Input array flags is a bit array with 32 head flags per word. +// ExtractThreadHeadFlags returns numBits flags starting at bit index. + +MGPU_HOST_DEVICE uint DeviceExtractHeadFlags(const uint* flags, int index, + int numBits) { + + int index2 = index>> 5; + int shift = 31 & index; + uint headFlags = flags[index2]>> shift; + int shifted = 32 - shift; + + if(shifted < numBits) + // We also need to shift in the next set of bits. + headFlags = bfi(flags[index2 + 1], headFlags, shifted, shift); + headFlags &= (1<< numBits) - 1; + return headFlags; +} + +//////////////////////////////////////////////////////////////////////////////// +// DevicePackHeadFlags +// Pack VT bits per thread at 32 bits/thread. Will consume an integer number of +// words, because CTA size is a multiple of 32. The first NT * VT / 32 threads +// return packed words. + +template +MGPU_DEVICE uint DevicePackHeadFlags(uint threadBits, int tid, + uint* flags_shared) { + + const int WordCount = NT * VT / 32; + + // Each thread stores its thread bits to flags_shared[tid]. + flags_shared[tid] = threadBits; + __syncthreads(); + + uint packed = 0; + if(tid < WordCount) { + const int Items = MGPU_DIV_UP(32, VT); + int index = 32 * tid; + int first = index / VT; + int bit = 0; + + int rem = index - VT * first; + packed = flags_shared[first]>> rem; + bit = VT - rem; + ++first; + + #pragma unroll + for(int i = 0; i < Items; ++i) { + if(i < Items - 1 || bit < 32) { + uint x = flags_shared[first + i]; + if(bit < 32) packed |= x<< bit; + bit += VT; + } + } + } + __syncthreads(); + + return packed; +} + +} // namespace mgpu diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/intrinsics.cuh b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/intrinsics.cuh new file mode 100644 index 000000000000..afcfc00e6617 --- /dev/null +++ b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/intrinsics.cuh @@ -0,0 +1,421 @@ +/****************************************************************************** + * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions are met: + * * Redistributions of source code must retain the above copyright + * notice, this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright + * notice, this list of conditions and the following disclaimer in the + * documentation and/or other materials provided with the distribution. + * * Neither the name of the NVIDIA CORPORATION nor the + * names of its contributors may be used to endorse or promote products + * derived from this software without specific prior written permission. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY + * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND + * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + * + ******************************************************************************/ + +/****************************************************************************** + * + * Code and text by Sean Baxter, NVIDIA Research + * See http://nvlabs.github.io/moderngpu for repository and documentation. + * + ******************************************************************************/ + +#include "devicetypes.cuh" + +#pragma once + +#pragma GCC diagnostic push +#pragma GCC diagnostic ignored "-Wstrict-aliasing" + +namespace mgpu { + +MGPU_HOST_DEVICE uint2 ulonglong_as_uint2(uint64 x) { + return *reinterpret_cast(&x); +} +MGPU_HOST_DEVICE uint64 uint2_as_ulonglong(uint2 x) { + return *reinterpret_cast(&x); +} + +MGPU_HOST_DEVICE int2 longlong_as_int2(int64 x) { + return *reinterpret_cast(&x); +} +MGPU_HOST_DEVICE int64 int2_as_longlong(int2 x) { + return *reinterpret_cast(&x); +} + +MGPU_HOST_DEVICE int2 double_as_int2(double x) { + return *reinterpret_cast(&x); +} +MGPU_HOST_DEVICE double int2_as_double(int2 x) { + return *reinterpret_cast(&x); +} + +MGPU_HOST_DEVICE void SetDoubleX(double& d, int x) { + reinterpret_cast(&d)[0] = x; +} +MGPU_HOST_DEVICE int GetDoubleX(double d) { + return double_as_int2(d).x; +} +MGPU_HOST_DEVICE void SetDoubleY(double& d, int y) { + reinterpret_cast(&d)[1] = y; +} +MGPU_HOST_DEVICE int GetDoubleY(double d) { + return double_as_int2(d).y; +} + + +//////////////////////////////////////////////////////////////////////////////// +// PTX for bfe and bfi + +#if __CUDA_ARCH__ >= 200 + +MGPU_DEVICE uint bfe_ptx(uint x, uint bit, uint numBits) { + uint result; + asm("bfe.u32 %0, %1, %2, %3;" : + "=r"(result) : "r"(x), "r"(bit), "r"(numBits)); + return result; +} + + +MGPU_DEVICE uint bfi_ptx(uint x, uint y, uint bit, uint numBits) { + uint result; + asm("bfi.b32 %0, %1, %2, %3, %4;" : + "=r"(result) : "r"(x), "r"(y), "r"(bit), "r"(numBits)); + return result; +} + +MGPU_DEVICE uint prmt_ptx(uint a, uint b, uint index) { + uint ret; + asm("prmt.b32 %0, %1, %2, %3;" : "=r"(ret) : "r"(a), "r"(b), "r"(index)); + return ret; +} + +#endif // __CUDA_ARCH__ >= 200 + +#if CUDA_VERSION >= 9000 +//////////////////////////////////////////////////////////////////////////////// +// shfl_add + +MGPU_DEVICE int shfl_add(int x, int offset, int width = WARP_SIZE, unsigned int threadmask = 0xFFFFFFFF) { + int result = 0; +#if __CUDA_ARCH__ >= 300 + int mask = (WARP_SIZE - width)<< 8; + asm( + "{.reg .s32 r0;" + ".reg .pred p;" + "shfl.sync.up.b32 r0|p, %1, %2, %3, %4;" + "@p add.s32 r0, r0, %5;" + "mov.s32 %0, r0; }" + : "=r"(result) : "r"(x), "r"(offset), "r"(mask), "r"(threadmask), "r"(x)); +#endif + return result; +} + +MGPU_DEVICE int shfl_max(int x, int offset, int width = WARP_SIZE, unsigned int threadmask = 0xFFFFFFFF) { + int result = 0; +#if __CUDA_ARCH__ >= 300 + int mask = (WARP_SIZE - width)<< 8; + asm( + "{.reg .s32 r0;" + ".reg .pred p;" + "shfl.sync.up.b32 r0|p, %1, %2, %3, %4;" + "@p max.s32 r0, r0, %5;" + "mov.s32 %0, r0; }" + : "=r"(result) : "r"(x), "r"(offset), "r"(mask), "r"(threadmask), "r"(x)); +#endif + return result; +} +#else +//////////////////////////////////////////////////////////////////////////////// +// shfl_add + +MGPU_DEVICE int shfl_add(int x, int offset, int width = WARP_SIZE) { + int result = 0; +#if __CUDA_ARCH__ >= 300 + int mask = (WARP_SIZE - width)<< 8; + asm( + "{.reg .s32 r0;" + ".reg .pred p;" + "shfl.up.b32 r0|p, %1, %2, %3;" + "@p add.s32 r0, r0, %4;" + "mov.s32 %0, r0; }" + : "=r"(result) : "r"(x), "r"(offset), "r"(mask), "r"(x)); +#endif + return result; +} + +MGPU_DEVICE int shfl_max(int x, int offset, int width = WARP_SIZE) { + int result = 0; +#if __CUDA_ARCH__ >= 300 + int mask = (WARP_SIZE - width)<< 8; + asm( + "{.reg .s32 r0;" + ".reg .pred p;" + "shfl.up.b32 r0|p, %1, %2, %3;" + "@p max.s32 r0, r0, %4;" + "mov.s32 %0, r0; }" + : "=r"(result) : "r"(x), "r"(offset), "r"(mask), "r"(x)); +#endif + return result; +} +#endif + +//////////////////////////////////////////////////////////////////////////////// +// brev, popc, clz, bfe, bfi, prmt + +// Reverse the bits in an integer. +MGPU_HOST_DEVICE uint brev(uint x) { +#if __CUDA_ARCH__ >= 200 + uint y = __brev(x); +#else + uint y = 0; + for(int i = 0; i < 32; ++i) + y |= (1 & (x>> i))<< (31 - i); +#endif + return y; +} + +// Count number of bits in a register. +MGPU_HOST_DEVICE int popc(uint x) { +#if __CUDA_ARCH__ >= 200 + return __popc(x); +#else + int c; + for(c = 0; x; ++c) + x &= x - 1; + return c; +#endif +} + +// Count leading zeros - start from most significant bit. +MGPU_HOST_DEVICE int clz(int x) { +#if __CUDA_ARCH__ >= 200 + return __clz(x); +#else + for(int i = 31; i >= 0; --i) + if((1<< i) & x) return 31 - i; + return 32; +#endif +} + +// Find first set - start from least significant bit. LSB is 1. ffs(0) is 0. +MGPU_HOST_DEVICE int ffs(int x) { +#if __CUDA_ARCH__ >= 200 + return __ffs(x); +#else + for(int i = 0; i < 32; ++i) + if((1<< i) & x) return i + 1; + return 0; +#endif +} + +MGPU_HOST_DEVICE uint bfe(uint x, uint bit, uint numBits) { +#if __CUDA_ARCH__ >= 200 + return bfe_ptx(x, bit, numBits); +#else + return ((1<< numBits) - 1) & (x>> bit); +#endif +} + +MGPU_HOST_DEVICE uint bfi(uint x, uint y, uint bit, uint numBits) { + uint result; +#if __CUDA_ARCH__ >= 200 + result = bfi_ptx(x, y, bit, numBits); +#else + if(bit + numBits > 32) numBits = 32 - bit; + uint mask = ((1<< numBits) - 1)<< bit; + result = y & ~mask; + result |= mask & (x<< bit); +#endif + return result; +} + +MGPU_HOST_DEVICE uint prmt(uint a, uint b, uint index) { + uint result; +#if __CUDA_ARCH__ >= 200 + result = prmt_ptx(a, b, index); +#else + result = 0; + for(int i = 0; i < 4; ++i) { + uint sel = 0xf & (index>> (4 * i)); + uint x = ((7 & sel) > 3) ? b : a; + x = 0xff & (x>> (8 * (3 & sel))); + if(8 & sel) x = (128 & x) ? 0xff : 0; + result |= x<< (8 * i); + } +#endif + return result; +} + +// Find log2(x) and optionally round up to the next integer logarithm. +MGPU_HOST_DEVICE int FindLog2(int x, bool roundUp = false) { + int a = 31 - clz(x); + if(roundUp) a += !MGPU_IS_POW_2(x); + return a; +} + +//////////////////////////////////////////////////////////////////////////////// +// vset4 + +#if __CUDA_ARCH__ >= 300 + +// Performs four byte-wise comparisons and returns 1 for each byte that +// satisfies the conditional, and zero otherwise. +MGPU_DEVICE uint vset4_lt_add_ptx(uint a, uint b, uint c) { + uint result; + asm("vset4.u32.u32.lt.add %0, %1, %2, %3;" : + "=r"(result) : "r"(a), "r"(b), "r"(c)); + return result; +} +MGPU_DEVICE uint vset4_eq_ptx(uint a, uint b) { + uint result; + asm("vset4.u32.u32.eq %0, %1, %2, %3;" : + "=r"(result) : "r"(a), "r"(b), "r"(0)); + return result; +} +#endif // __CUDA_ARCH__ >= 300 + +MGPU_HOST_DEVICE uint vset4_lt_add(uint a, uint b, uint c) { + uint result; +#if __CUDA_ARCH__ >= 300 + result = vset4_lt_add_ptx(a, b, c); +#else + result = c; + if((0x000000ff & a) < (0x000000ff & b)) result += 0x00000001; + if((0x0000ff00 & a) < (0x0000ff00 & b)) result += 0x00000100; + if((0x00ff0000 & a) < (0x00ff0000 & b)) result += 0x00010000; + if((0xff000000 & a) < (0xff000000 & b)) result += 0x01000000; +#endif + return result; +} + +MGPU_HOST_DEVICE uint vset4_eq(uint a, uint b) { + uint result; +#if __CUDA_ARCH__ >= 300 + result = vset4_eq_ptx(a, b); +#else + result = 0; + if((0x000000ff & a) == (0x000000ff & b)) result = 0x00000001; + if((0x0000ff00 & a) == (0x0000ff00 & b)) result += 0x00000100; + if((0x00ff0000 & a) == (0x00ff0000 & b)) result += 0x00010000; + if((0xff000000 & a) == (0xff000000 & b)) result += 0x01000000; +#endif + return result; +} + +//////////////////////////////////////////////////////////////////////////////// +// + +MGPU_HOST_DEVICE uint umulhi(uint x, uint y) { +#if __CUDA_ARCH__ >= 100 + return __umulhi(x, y); +#else + uint64 product = (uint64)x * y; + return (uint)(product>> 32); +#endif +} + +//////////////////////////////////////////////////////////////////////////////// +// ldg() function defined for all devices and all types. Only compiles to __ldg +// intrinsic for __CUDA_ARCH__ >= 320 && __CUDA_ARCH__ < 400 for types supported +// by __ldg in sm_32_intrinsics.h + +template +struct IsLdgType { + enum { value = false }; +}; +#define DEFINE_LDG_TYPE(T) \ + template<> struct IsLdgType { enum { value = true }; }; + +template::value> +struct LdgShim { + MGPU_DEVICE static T Ldg(const T* p) { + return *p; + } +}; + +#if __CUDA_ARCH__ >= 320 && __CUDA_ARCH__ < 400 + + // List of __ldg-compatible types from sm_32_intrinsics.h. + DEFINE_LDG_TYPE(char) + DEFINE_LDG_TYPE(short) + DEFINE_LDG_TYPE(int) + DEFINE_LDG_TYPE(long long) + DEFINE_LDG_TYPE(char2) + DEFINE_LDG_TYPE(char4) + DEFINE_LDG_TYPE(short2) + DEFINE_LDG_TYPE(short4) + DEFINE_LDG_TYPE(int2) + DEFINE_LDG_TYPE(int4) + DEFINE_LDG_TYPE(longlong2) + + DEFINE_LDG_TYPE(unsigned char) + DEFINE_LDG_TYPE(unsigned short) + DEFINE_LDG_TYPE(unsigned int) + DEFINE_LDG_TYPE(unsigned long long) + DEFINE_LDG_TYPE(uchar2) + DEFINE_LDG_TYPE(uchar4) + DEFINE_LDG_TYPE(ushort2) + DEFINE_LDG_TYPE(ushort4) + DEFINE_LDG_TYPE(uint2) + DEFINE_LDG_TYPE(uint4) + DEFINE_LDG_TYPE(ulonglong2) + + DEFINE_LDG_TYPE(float) + DEFINE_LDG_TYPE(double) + DEFINE_LDG_TYPE(float2) + DEFINE_LDG_TYPE(float4) + DEFINE_LDG_TYPE(double2) + + template struct LdgShim { + MGPU_DEVICE static T Ldg(const T* p) { + return __ldg(p); + } + }; +#endif + +template +MGPU_DEVICE T ldg(const T* p) { + return LdgShim::Ldg(p); +} + +//////////////////////////////////////////////////////////////////////////////// + +// Fast division for 31-bit integers. +// Uses the method in Hacker's Delight (2nd edition) page 228. +// Evaluates for denom > 1 and x < 2^31. +struct FastDivide { + uint denom; + uint coef; + uint shift; + + MGPU_HOST_DEVICE uint Divide(uint x) { + return umulhi(x, coef)>> shift; + } + MGPU_HOST_DEVICE uint Modulus(uint x) { + return x - Divide(x) * denom; + } + + explicit FastDivide(uint denom_) { + denom = denom_; + uint p = 31 + FindLog2(denom, true); + coef = (uint)(((1ull<< p) + denom - 1) / denom); + shift = p - 32; + } +}; + +#pragma GCC diagnostic pop + +} // namespace mgpu diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/loadstore.cuh b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/loadstore.cuh new file mode 100644 index 000000000000..aae17d8490a4 --- /dev/null +++ b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/loadstore.cuh @@ -0,0 +1,674 @@ +/****************************************************************************** + * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions are met: + * * Redistributions of source code must retain the above copyright + * notice, this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright + * notice, this list of conditions and the following disclaimer in the + * documentation and/or other materials provided with the distribution. + * * Neither the name of the NVIDIA CORPORATION nor the + * names of its contributors may be used to endorse or promote products + * derived from this software without specific prior written permission. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY + * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND + * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + * + ******************************************************************************/ + +/****************************************************************************** + * + * Code and text by Sean Baxter, NVIDIA Research + * See http://nvlabs.github.io/moderngpu for repository and documentation. + * + ******************************************************************************/ + +#pragma once + +#include "../mgpudevice.cuh" +#include "deviceutil.cuh" +#include "intrinsics.cuh" + +namespace mgpu { + +//////////////////////////////////////////////////////////////////////////////// +// Cooperative load functions. + +template +MGPU_DEVICE void DeviceSharedToReg(InputIt data, int tid, T* reg, + bool sync) { + + #pragma unroll + for(int i = 0; i < VT; ++i) + reg[i] = data[NT * i + tid]; + + if(sync) __syncthreads(); +} + +template +MGPU_DEVICE void DeviceGlobalToRegPred(int count, InputIt data, int tid, + T* reg, bool sync) { + + // TODO: Attempt to issue 4 loads at a time. + #pragma unroll + for(int i = 0; i < VT; ++i) { + int index = NT * i + tid; + if(index < count) reg[i] = data[index]; + } + if(sync) __syncthreads(); +} + +template +MGPU_DEVICE void DeviceGlobalToReg(int count, InputIt data, int tid, + T* reg, bool sync) { + + if(count >= NT * VT) { + #pragma unroll + for(int i = 0; i < VT; ++i) + reg[i] = data[NT * i + tid]; + } else + DeviceGlobalToRegPred(count, data, tid, reg, false); + if(sync) __syncthreads(); +} +template +MGPU_DEVICE void DeviceGlobalToReg2(int count, InputIt data, int tid, + T* reg, bool sync) { + + DeviceGlobalToReg(count, data, tid, reg, false); + #pragma unroll + for(int i = VT0; i < VT1; ++i) { + int index = NT * i + tid; + if(index < count) reg[i] = data[index]; + } + if(sync) __syncthreads(); +} + +template +MGPU_DEVICE void DeviceGlobalToRegDefault(int count, InputIt data, int tid, + T* reg, T init, bool sync) { + + if(count >= NT * VT) { + #pragma unroll + for(int i = 0; i < VT; ++i) + reg[i] = data[NT * i + tid]; + } else { + #pragma unroll + for(int i = 0; i < VT; ++i) { + int index = NT * i + tid; + reg[i] = init; + if(index < count) reg[i] = data[index]; + } + } + if(sync) __syncthreads(); +} +template +MGPU_DEVICE void DeviceGlobalToRegDefault2(int count, InputIt data, int tid, + T* reg, T init, bool sync) { + + DeviceGlobalToRegDefault(count, data, tid, reg, init, false); + #pragma unroll + for(int i = VT0; i < VT1; ++i) { + int index = NT * i + tid; + reg[i] = init; + if(index < count) reg[i] = data[index]; + } + if(sync) __syncthreads(); +} + +//////////////////////////////////////////////////////////////////////////////// + +template +MGPU_DEVICE void DeviceGlobalToThread(int count, InputIt data, int tid, + T* reg) { + + data += VT * tid; + if(count >= NT * VT) { + #pragma unroll + for(int i = 0; i < VT; ++i) + reg[i] = ldg(data + i); + } else { + count -= VT * tid; + #pragma unroll + for(int i = 0; i < VT; ++i) + if(i < count) reg[i] = ldg(data + i); + } +} + +template +MGPU_DEVICE void DeviceGlobalToThreadDefault(int count, InputIt data, int tid, + T* reg, T init) { + + data += VT * tid; + if(count >= NT * VT) { + #pragma unroll + for(int i = 0; i < VT; ++i) + reg[i] = ldg(data + i); + } else { + count -= VT * tid; + #pragma unroll + for(int i = 0; i < VT; ++i) + reg[i] = (i < count) ? ldg(data + i) : init; + } +} + + +//////////////////////////////////////////////////////////////////////////////// +// Cooperative store functions. + +template +MGPU_DEVICE void DeviceRegToShared(const T* reg, int tid, + OutputIt dest, bool sync) { + + typedef typename std::iterator_traits::value_type T2; + #pragma unroll + for(int i = 0; i < VT; ++i) + dest[NT * i + tid] = (T2)reg[i]; + + if(sync) __syncthreads(); +} + +template +MGPU_DEVICE void DeviceRegToGlobal(int count, const T* reg, int tid, + OutputIt dest, bool sync) { + + #pragma unroll + for(int i = 0; i < VT; ++i) { + int index = NT * i + tid; + if(index < count) + dest[index] = reg[i]; + } + if(sync) __syncthreads(); +} + +//////////////////////////////////////////////////////////////////////////////// +// DeviceMemToMemLoop +// Transfer from shared memory to global, or global to shared, for transfers +// that are smaller than NT * VT in the average case. The goal is to reduce +// unnecessary comparison logic. + +template +MGPU_DEVICE void DeviceMemToMem4(int count, InputIt source, int tid, + OutputIt dest, bool sync) { + + typedef typename std::iterator_traits::value_type T; + + T x[VT]; + const int Count = (VT < 4) ? VT : 4; + if(count >= NT * VT) { + #pragma unroll + for(int i = 0; i < Count; ++i) + x[i] = source[NT * i + tid]; + #pragma unroll + for(int i = 0; i < Count; ++i) + dest[NT * i + tid] = x[i]; + } else { + #pragma unroll + for(int i = 0; i < Count; ++i) { + int index = NT * i + tid; + if(index < count) + x[i] = source[NT * i + tid]; + } + #pragma unroll + for(int i = 0; i < Count; ++i) { + int index = NT * i + tid; + if(index < count) + dest[index] = x[i]; + } + } + if(sync) __syncthreads(); +} +template +MGPU_DEVICE void DeviceMemToMemLoop(int count, InputIt source, int tid, + OutputIt dest, bool sync) { + + for(int i = 0; i < count; i += 4 * NT) + DeviceMemToMem4(count - i, source + i, tid, dest + i, + false); + if(sync) __syncthreads(); +} + + +//////////////////////////////////////////////////////////////////////////////// +// Functions to copy between shared and global memory where the average case is +// to transfer NT * VT elements. + +template +MGPU_DEVICE void DeviceSharedToGlobal(int count, const T* source, int tid, + OutputIt dest, bool sync) { + + typedef typename std::iterator_traits::value_type T2; + #pragma unroll + for(int i = 0; i < VT; ++i) { + int index = NT * i + tid; + if(index < count) dest[index] = (T2)source[index]; + } + if(sync) __syncthreads(); +} + +template +MGPU_DEVICE void DeviceGlobalToShared(int count, InputIt source, int tid, + T* dest, bool sync) { + + T reg[VT]; + DeviceGlobalToReg(count, source, tid, reg, false); + DeviceRegToShared(reg, tid, dest, sync); +} + +template +MGPU_DEVICE void DeviceGlobalToShared2(int count, InputIt source, int tid, + T* dest, bool sync) { + + T reg[VT1]; + DeviceGlobalToReg2(count, source, tid, reg, false); + DeviceRegToShared(reg, tid, dest, sync); +} + + +template +MGPU_DEVICE void DeviceGlobalToSharedDefault(int count, InputIt source, int tid, + T* dest, T init, bool sync) { + + T reg[VT]; + DeviceGlobalToRegDefault(count, source, tid, reg, init, false); + DeviceRegToShared(reg, tid, dest, sync); +} + +template +MGPU_DEVICE void DeviceGlobalToSharedDefault2(int count, InputIt data, int tid, + T* dest, T init, bool sync) { + + T reg[VT1]; + DeviceGlobalToRegDefault2(count, data, tid, reg, init, false); + DeviceRegToShared(reg, tid, dest, sync); +} + + +//////////////////////////////////////////////////////////////////////////////// + +template +MGPU_DEVICE void DeviceGlobalToSharedLoop(int count, InputIt source, int tid, + T* dest, bool sync) { + + const int Granularity = MGPU_MIN(VT, 3); + DeviceGlobalToShared(count, source, tid, dest, false); + + int offset = Granularity * NT; + if(count > offset) + DeviceGlobalToShared(count - offset, + source + offset, tid, dest + offset, false); + + if(sync) __syncthreads(); + + /* + source += tid; + while(count > 0) { + T reg[Granularity]; + #pragma unroll + for(int i = 0; i < Granularity; ++i) { + int index = NT * i + tid; + if(index < count) + reg[i] = source[NT * i]; + } + DeviceRegToShared(reg, tid, dest, false); + source += Granularity * NT; + dest += Granularity * NT; + count -= Granularity * NT; + } + if(sync) __syncthreads();*/ +} + +template +MGPU_DEVICE void DeviceGlobalToGlobal(int count, InputIt source, int tid, + OutputIt dest, bool sync) { + + typedef typename std::iterator_traits::value_type T; + T values[VT]; + DeviceGlobalToReg(count, source, tid, values, false); + DeviceRegToGlobal(count, values, tid, dest, sync); +} + +//////////////////////////////////////////////////////////////////////////////// +// Transponse VT elements in NT threads (x) into thread-order registers (y) +// using only NT * VT / 2 elements of shared memory. + +//This function definitely has a bug, don't use!!! fix TODO(erich) +template +MGPU_DEVICE void HalfSmemTranspose(const T* x, int tid, T* shared, T* y) { + printf("HalfSmemTranspose has a bug, use WAR SmemTranpose or find bug before using in production"); + // Transpose the first half values (tid < NT / 2) + #pragma unroll + for(int i = 0; i <= VT / 2; ++i) + if(i < VT / 2 || tid < NT / 2) + shared[NT * i + tid] = x[i]; + __syncthreads(); + + if(tid < NT / 2) { + #pragma unroll + for(int i = 0; i < VT; ++i) + y[i] = shared[VT * tid + i]; + } + __syncthreads(); + + // Transpose the second half values (tid >= NT / 2) + #pragma unroll + for(int i = VT / 2; i < VT; ++i) + if(i > VT / 2 || tid >= NT / 2) + shared[NT * i - NT * VT / 2 + tid] = x[i]; + __syncthreads(); + + if(tid >= NT / 2) { + #pragma unroll + for(int i = 0; i < VT; ++i) + y[i] = shared[VT * tid + i - NT * VT / 2]; + } + __syncthreads(); +} + +//////////////////////////////////////////////////////////////////////////////// +// Gather/scatter functions + +template +MGPU_DEVICE void DeviceGather(int count, InputIt data, int indices[VT], + int tid, T* reg, bool sync) { + + if(count >= NT * VT) { + #pragma unroll + for(int i = 0; i < VT; ++i) + reg[i] = data[indices[i]]; + } else { + #pragma unroll + for(int i = 0; i < VT; ++i) { + int index = NT * i + tid; + if(index < count) + reg[i] = data[indices[i]]; + } + } + if(sync) __syncthreads(); +} + +template +MGPU_DEVICE void DeviceGatherDefault(int count, InputIt data, int indices[VT], + int tid, T* reg, T identity, bool sync) { + + if(count >= NT * VT) { + #pragma unroll + for(int i = 0; i < VT; ++i) + reg[i] = data[indices[i]]; + } else { + #pragma unroll + for(int i = 0; i < VT; ++i) { + int index = NT * i + tid; + reg[i] = (index < count) ? data[indices[i]] : identity; + } + } + if(sync) __syncthreads(); +} + +template +MGPU_DEVICE void DeviceScatter(int count, const T* reg, int tid, + int indices[VT], OutputIt data, bool sync) { + + if(count >= NT * VT) { + #pragma unroll + for(int i = 0; i < VT; ++i) + data[indices[i]] = reg[i]; + } else { + #pragma unroll + for(int i = 0; i < VT; ++i) { + int index = NT * i + tid; + if(index < count) + data[indices[i]] = reg[i]; + } + } + if(sync) __syncthreads(); +} + +//////////////////////////////////////////////////////////////////////////////// +// Cooperative transpose functions (strided to thread order) + +template +MGPU_DEVICE void DeviceThreadToShared(const T* threadReg, int tid, T* shared, + bool sync) { + + if(1 & VT) { + // Odd grain size. Store as type T. + #pragma unroll + for(int i = 0; i < VT; ++i) + shared[VT * tid + i] = threadReg[i]; + } else { + // Even grain size. Store as DevicePair. This lets us exploit the + // 8-byte shared memory mode on Kepler. + DevicePair* dest = (DevicePair*)(shared + VT * tid); + #pragma unroll + for(int i = 0; i < VT / 2; ++i) + dest[i] = MakeDevicePair(threadReg[2 * i], threadReg[2 * i + 1]); + } + if(sync) __syncthreads(); +} + +template +MGPU_DEVICE void DeviceSharedToThread(const T* shared, int tid, T* threadReg, + bool sync) { + + if(1 & VT) { + #pragma unroll + for(int i = 0; i < VT; ++i) + threadReg[i] = shared[VT * tid + i]; + } else { + const DevicePair* source = (const DevicePair*)(shared + VT * tid); + #pragma unroll + for(int i = 0; i < VT / 2; ++i) { + DevicePair p = source[i]; + threadReg[2 * i] = p.x; + threadReg[2 * i + 1] = p.y; + } + } + if(sync) __syncthreads(); +} + +//////////////////////////////////////////////////////////////////////////////// +// DeviceLoad2 - load from pointers of the same type. Optimize for a single LD +// statement. + +template +MGPU_DEVICE void DeviceLoad2ToReg(const T* a_global, int aCount, + const T* b_global, int bCount, int tid, T* reg, bool sync) { + + int b0 = b_global - a_global - aCount; + int total = aCount + bCount; + if(total >= NT * VT0) { + #pragma unroll + for(int i = 0; i < VT0; ++i) { + int index = NT * i + tid; + reg[i] = a_global[index + ((index >= aCount) ? b0 : 0)]; + } + } else { + #pragma unroll + for(int i = 0; i < VT0; ++i) { + int index = NT * i + tid; + if(index < total) + reg[i] = a_global[index + ((index >= aCount) ? b0 : 0)]; + } + } + #pragma unroll + for(int i = VT0; i < VT1; ++i) { + int index = NT * i + tid; + if(index < total) + reg[i] = a_global[index + ((index >= aCount) ? b0 : 0)]; + } +} + +template +MGPU_DEVICE void DeviceLoad2ToShared(const T* a_global, int aCount, + const T* b_global, int bCount, int tid, T* shared, bool sync) { + + T reg[VT1]; + DeviceLoad2ToReg(a_global, aCount, b_global, bCount, tid, + reg, false); + DeviceRegToShared(reg, tid, shared, sync); +} + +//////////////////////////////////////////////////////////////////////////////// +// DeviceLoad2 - load from pointers of different types. Uses two LD statements. + +template +MGPU_DEVICE void DeviceLoad2ToReg(InputIt1 a_global, int aCount, + InputIt2 b_global, int bCount, int tid, T* reg, bool sync) { + + b_global -= aCount; + int total = aCount + bCount; + if(total >= NT * VT0) { + #pragma unroll + for(int i = 0; i < VT0; ++i) { + int index = NT * i + tid; + if(index < aCount) reg[i] = a_global[index]; + else reg[i] = b_global[index]; + } + } else { + #pragma unroll + for(int i = 0; i < VT0; ++i) { + int index = NT * i + tid; + if(index < aCount) reg[i] = a_global[index]; + else if(index < total) reg[i] = b_global[index]; + } + } + #pragma unroll + for(int i = VT0; i < VT1; ++i) { + int index = NT * i + tid; + if(index < aCount) reg[i] = a_global[index]; + else if(index < total) reg[i] = b_global[index]; + } +} + +template +MGPU_DEVICE void DeviceLoad2ToShared(InputIt1 a_global, int aCount, + InputIt2 b_global, int bCount, int tid, T* shared, bool sync) { + + T reg[VT1]; + DeviceLoad2ToReg(a_global, aCount, b_global, bCount, tid, + reg, false); + DeviceRegToShared(reg, tid, shared, sync); +} + + +//////////////////////////////////////////////////////////////////////////////// +// DeviceGatherGlobalToGlobal + +template +MGPU_DEVICE void DeviceGatherGlobalToGlobal(int count, InputIt data_global, + const int* indices_shared, int tid, OutputIt dest_global, bool sync) { + + typedef typename std::iterator_traits::value_type ValType; + ValType values[VT]; + + #pragma unroll + for(int i = 0; i < VT; ++i) { + int index = NT * i + tid; + if(index < count) { + int gather = indices_shared[index]; + values[i] = data_global[gather]; + } + } + if(sync) __syncthreads(); + DeviceRegToGlobal(count, values, tid, dest_global, false); +} + +//////////////////////////////////////////////////////////////////////////////// +// DeviceTransferMergeValues +// Gather in a merge-like value from two input arrays and store to a single +// output. Like DeviceGatherGlobalToGlobal, but for two arrays at once. + +template +MGPU_DEVICE void DeviceTransferMergeValuesReg(int count, InputIt1 a_global, + InputIt2 b_global, int bStart, const int* indices, int tid, + T* reg, bool sync) { + + b_global -= bStart; + if(count >= NT * VT) { + #pragma unroll + for(int i = 0; i < VT; ++i) { + reg[i] = (indices[i] < bStart) ? a_global[indices[i]] : + b_global[indices[i]]; + } + } else { + #pragma unroll + for(int i = 0; i < VT; ++i) { + int index = NT * i + tid; + if(index < count) + reg[i] = (indices[i] < bStart) ? a_global[indices[i]] : + b_global[indices[i]]; + } + } + if(sync) __syncthreads(); +} + +template +MGPU_DEVICE void DeviceTransferMergeValuesShared(int count, InputIt1 a_global, + InputIt2 b_global, int bStart, const int* indices_shared, int tid, + OutputIt dest_global, bool sync) { + + int indices[VT]; + DeviceSharedToReg(indices_shared, tid, indices); + + typedef typename std::iterator_traits::value_type ValType; + ValType reg[VT]; + DeviceTransferMergeValuesReg(count, a_global, b_global, bStart, + indices, tid, reg, sync); + DeviceRegToGlobal(count, reg, tid, dest_global, sync); +} + +template +MGPU_DEVICE void DeviceTransferMergeValuesReg(int count, const T* a_global, + const T* b_global, int bStart, const int* indices, int tid, T* reg, + bool sync) { + + int bOffset = (int)(b_global - a_global - bStart); + + if(count >= NT * VT) { + #pragma unroll + for(int i = 0; i < VT; ++i) { + int gather = indices[i]; + if(gather >= bStart) gather += bOffset; + reg[i] = a_global[gather]; + } + } else { + #pragma unroll + for(int i = 0; i < VT; ++i) { + int index = NT * i + tid; + int gather = indices[i]; + if(gather >= bStart) gather += bOffset; + if(index < count) + reg[i] = a_global[gather]; + } + } + if(sync) __syncthreads(); +} + +template +MGPU_DEVICE void DeviceTransferMergeValuesShared(int count, const T* a_global, + const T* b_global, int bStart, const int* indices_shared, int tid, + OutputIt dest_global, bool sync) { + + int indices[VT]; + DeviceSharedToReg(indices_shared, tid, indices); + + T reg[VT]; + DeviceTransferMergeValuesReg(count, a_global, b_global, bStart, + indices, tid, reg, sync); + DeviceRegToGlobal(count, reg, tid, dest_global, sync); +} + +} // namespace mgpu diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/serialsets.cuh b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/serialsets.cuh new file mode 100644 index 000000000000..d0e7a3a81722 --- /dev/null +++ b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/serialsets.cuh @@ -0,0 +1,235 @@ +/****************************************************************************** + * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions are met: + * * Redistributions of source code must retain the above copyright + * notice, this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright + * notice, this list of conditions and the following disclaimer in the + * documentation and/or other materials provided with the distribution. + * * Neither the name of the NVIDIA CORPORATION nor the + * names of its contributors may be used to endorse or promote products + * derived from this software without specific prior written permission. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY + * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND + * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + * + ******************************************************************************/ + +/****************************************************************************** + * + * Code and text by Sean Baxter, NVIDIA Research + * See http://nvlabs.github.io/moderngpu for repository and documentation. + * + ******************************************************************************/ + +#pragma once + +#include "deviceutil.cuh" + +namespace mgpu { + +//////////////////////////////////////////////////////////////////////////////// +// SerialSetIntersection +// Emit A if A and B are in range and equal. + +template +MGPU_DEVICE int SerialSetIntersection(const T* data, int aBegin, int aEnd, + int bBegin, int bEnd, int end, T* results, int* indices, Comp comp) { + + const int MinIterations = VT / 2; + int commit = 0; + + #pragma unroll + for(int i = 0; i < VT; ++i) { + bool test = RangeCheck ? + ((aBegin + bBegin < end) && (aBegin < aEnd) && (bBegin < bEnd)) : + (i < MinIterations || (aBegin + bBegin < end)); + + if(test) { + T aKey = data[aBegin]; + T bKey = data[bBegin]; + + bool pA = comp(aKey, bKey); + bool pB = comp(bKey, aKey); + + // The outputs must come from A by definition of set interection. + results[i] = aKey; + indices[i] = aBegin; + + if(!pB) ++aBegin; + if(!pA) ++bBegin; + if(pA == pB) commit |= 1<< i; + } + } + return commit; +} + +//////////////////////////////////////////////////////////////////////////////// +// SerialSetUnion +// Emit A if A <= B. Emit B if B < A. + +template +MGPU_DEVICE int SerialSetUnion(const T* data, int aBegin, int aEnd, + int bBegin, int bEnd, int end, T* results, int* indices, Comp comp) { + + const int MinIterations = VT / 2; + int commit = 0; + + #pragma unroll + for(int i = 0; i < VT; ++i) { + bool test = RangeCheck ? + (aBegin + bBegin < end) : + (i < MinIterations || (aBegin + bBegin < end)); + + if(test) { + T aKey = data[aBegin]; + T bKey = data[bBegin]; + + bool pA = false, pB = false; + if(RangeCheck && aBegin >= aEnd) + pB = true; + else if(RangeCheck && bBegin >= bEnd) + pA = true; + else { + // Both are in range. + pA = comp(aKey, bKey); + pB = comp(bKey, aKey); + } + + // Output A in case of a tie, so check if b < a. + results[i] = pB ? bKey : aKey; + indices[i] = pB ? bBegin : aBegin; + if(!pB) ++aBegin; + if(!pA) ++bBegin; + commit |= 1<< i; + } + } + return commit; +} + +//////////////////////////////////////////////////////////////////////////////// +// SerialSetDifference +// Emit A if A < B. + +template +MGPU_DEVICE int SerialSetDifference(const T* data, int aBegin, int aEnd, + int bBegin, int bEnd, int end, T* results, int* indices, Comp comp) { + + const int MinIterations = VT / 2; + int commit = 0; + + #pragma unroll + for(int i = 0; i < VT; ++i) { + bool test = RangeCheck ? + (aBegin + bBegin < end) : + (i < MinIterations || (aBegin + bBegin < end)); + if(test) { + T aKey = data[aBegin]; + T bKey = data[bBegin]; + + bool pA = false, pB = false; + if(RangeCheck && aBegin >= aEnd) + pB = true; + else if(RangeCheck && bBegin >= bEnd) + pA = true; + else { + pA = comp(aKey, bKey); + pB = comp(bKey, aKey); + } + + // The outputs must come from A by definition of set difference. + results[i] = aKey; + indices[i] = aBegin; + if(!pB) ++aBegin; + if(!pA) ++bBegin; + if(pA) commit |= 1<< i; + } + } + return commit; +} + +//////////////////////////////////////////////////////////////////////////////// +// SerialSetSymDiff +// Emit A if A < B and emit B if B < A. + +template +MGPU_DEVICE int SerialSetSymDiff(const T* data, int aBegin, int aEnd, + int bBegin, int bEnd, int end, T* results, int* indices, Comp comp) { + + const int MinIterations = VT / 2; + int commit = 0; + + #pragma unroll + for(int i = 0; i < VT; ++i) { + bool test = RangeCheck ? + (aBegin + bBegin < end) : + (i < MinIterations || (aBegin + bBegin < end)); + if(test) { + T aKey = data[aBegin]; + T bKey = data[bBegin]; + + bool pA = false, pB = false; + if(RangeCheck && (bBegin >= bEnd)) + pA = true; + else if(RangeCheck && (aBegin >= aEnd)) + pB = true; + else { + pA = comp(aKey, bKey); + pB = comp(bKey, aKey); + } + + results[i] = pA ? aKey : bKey; + indices[i] = pA ? aBegin : bBegin; + if(!pA) ++bBegin; + if(!pB) ++aBegin; + if(pA != pB) commit |= 1<< i; + } + } + return commit; +} + +//////////////////////////////////////////////////////////////////////////////// +// SerialSetOp +// Uses the MgpuSetOp enum to statically select one of the four serial ops +// above. + +template +MGPU_DEVICE int SerialSetOp(const T* data, int aBegin, int aEnd, + int bBegin, int bEnd, int star, T* results, int* indices, Comp comp) { + + int end = aBegin + bBegin + VT - star; + if(RangeCheck) end = min(end, aEnd + bEnd); + int commit; + switch(Op) { + case MgpuSetOpIntersection: + commit = SerialSetIntersection(data, aBegin, + aEnd, bBegin, bEnd, end, results, indices, comp); + break; + case MgpuSetOpUnion: + commit = SerialSetUnion(data, aBegin, aEnd, + bBegin, bEnd, end, results, indices, comp); + break; + case MgpuSetOpDiff: + commit = SerialSetDifference(data, aBegin, aEnd, + bBegin, bEnd, end, results, indices, comp); + break; + case MgpuSetOpSymDiff: + commit = SerialSetSymDiff(data, aBegin, aEnd, + bBegin, bEnd, end, results, indices, comp); + break; + } + __syncthreads(); + return commit; +} + +} // namespace mgpu diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/sortnetwork.cuh b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/sortnetwork.cuh new file mode 100644 index 000000000000..94c88f71d93d --- /dev/null +++ b/src/operator/nn/ctc_include/contrib/moderngpu/include/device/sortnetwork.cuh @@ -0,0 +1,168 @@ +/****************************************************************************** + * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions are met: + * * Redistributions of source code must retain the above copyright + * notice, this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright + * notice, this list of conditions and the following disclaimer in the + * documentation and/or other materials provided with the distribution. + * * Neither the name of the NVIDIA CORPORATION nor the + * names of its contributors may be used to endorse or promote products + * derived from this software without specific prior written permission. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY + * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND + * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + * + ******************************************************************************/ + +/****************************************************************************** + * + * Code and text by Sean Baxter, NVIDIA Research + * See http://nvlabs.github.io/moderngpu for repository and documentation. + * + ******************************************************************************/ + +#pragma once + +#include "deviceutil.cuh" + +namespace mgpu { + +//////////////////////////////////////////////////////////////////////////////// +// Odd-even transposition sorting network. Sorts keys and values in-place in +// register. +// http://en.wikipedia.org/wiki/Odd%E2%80%93even_sort + +// CUDA Compiler does not currently unroll these loops correctly. Write using +// template loop unrolling. +/* +template +MGPU_DEVICE void OddEvenTransposeSort(T* keys, V* values, Comp comp) { + #pragma unroll + for(int level = 0; level < VT; ++level) { + + #pragma unroll + for(int i = 1 & level; i < VT - 1; i += 2) { + if(comp(keys[i + 1], keys[i])) { + mgpu::swap(keys[i], keys[i + 1]); + mgpu::swap(values[i], values[i + 1]); + } + } + } +}*/ + +template +struct OddEvenTransposeSortT { + // Sort segments marked by head flags. If the head flag between i and i + 1 + // is set (so that (2<< i) & flags is true), the values belong to different + // segments and are not swapped. + template + static MGPU_DEVICE void Sort(K* keys, V* values, int flags, Comp comp) { + #pragma unroll + for(int i = 1 & I; i < VT - 1; i += 2) + if((0 == ((2<< i) & flags)) && comp(keys[i + 1], keys[i])) { + mgpu::swap(keys[i], keys[i + 1]); + mgpu::swap(values[i], values[i + 1]); + } + OddEvenTransposeSortT::Sort(keys, values, flags, comp); + } +}; +template struct OddEvenTransposeSortT { + template + static MGPU_DEVICE void Sort(K* keys, V* values, int flags, Comp comp) { } +}; + +template +MGPU_DEVICE void OddEvenTransposeSort(K* keys, V* values, Comp comp) { + OddEvenTransposeSortT<0, VT>::Sort(keys, values, 0, comp); +} +template +MGPU_DEVICE void OddEvenTransposeSortFlags(K* keys, V* values, int flags, + Comp comp) { + OddEvenTransposeSortT<0, VT>::Sort(keys, values, flags, comp); +} + +//////////////////////////////////////////////////////////////////////////////// +// Batcher Odd-Even Mergesort network +// Unstable but executes much faster than the transposition sort. +// http://en.wikipedia.org/wiki/Batcher_odd%E2%80%93even_mergesort + +template +struct OddEvenMergesortT { + template + MGPU_DEVICE static void CompareAndSwap(K* keys, V* values, int flags, + int a, int b, Comp comp) { + if(b < Count) { + // Mask the bits between a and b. Any head flags in this interval + // means the keys are in different segments and must not be swapped. + const int Mask = ((2<< b) - 1) ^ ((2<< a) - 1); + if(!(Mask & flags) && comp(keys[b], keys[a])) { + mgpu::swap(keys[b], keys[a]); + mgpu::swap(values[b], values[a]); + } + } + } + + template + struct OddEvenMerge { + template + MGPU_DEVICE static void Merge(K* keys, V* values, int flags, + Comp comp) { + // Compare and swap + const int M = 2 * R; + OddEvenMerge::Merge(keys, values, flags, comp); + OddEvenMerge::Merge(keys, values, flags, comp); + + #pragma unroll + for(int i = Low2 + R; i + R < Low2 + Width; i += M) + CompareAndSwap(keys, values, flags, i, i + R, comp); + } + }; + template + struct OddEvenMerge { + template + MGPU_DEVICE static void Merge(K* keys, V* values, int flags, + Comp comp) { + CompareAndSwap(keys, values, flags, Low2, Low2 + R, comp); + } + }; + + template + MGPU_DEVICE static void Sort(K* keys, V* values, int flags, + Comp comp) { + + const int M = Width / 2; + OddEvenMergesortT::Sort(keys, values, flags, comp); + OddEvenMergesortT::Sort(keys, values, flags, comp); + OddEvenMerge<1, Low>::Merge(keys, values, flags, comp); + } +}; +template struct OddEvenMergesortT<1, Low, Count> { + template + MGPU_DEVICE static void Sort(K* keys, V* values, int flags, + Comp comp) { } +}; + +template +MGPU_DEVICE void OddEvenMergesort(K* keys, V* values, Comp comp) { + const int Width = 1<< sLogPow2::value; + OddEvenMergesortT::Sort(keys, values, 0, comp); +} +template +MGPU_DEVICE void OddEvenMergesortFlags(K* keys, V* values, int flags, + Comp comp) { + const int Width = 1<< sLogPow2::value; + OddEvenMergesortT::Sort(keys, values, flags, comp); +} + +} // namespace mgpu diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/mgpudevice.cuh b/src/operator/nn/ctc_include/contrib/moderngpu/include/mgpudevice.cuh new file mode 100644 index 000000000000..ad8742d46db6 --- /dev/null +++ b/src/operator/nn/ctc_include/contrib/moderngpu/include/mgpudevice.cuh @@ -0,0 +1,289 @@ +/****************************************************************************** + * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions are met: + * * Redistributions of source code must retain the above copyright + * notice, this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright + * notice, this list of conditions and the following disclaimer in the + * documentation and/or other materials provided with the distribution. + * * Neither the name of the NVIDIA CORPORATION nor the + * names of its contributors may be used to endorse or promote products + * derived from this software without specific prior written permission. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY + * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND + * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + * + ******************************************************************************/ + +/****************************************************************************** + * + * Code and text by Sean Baxter, NVIDIA Research + * See http://nvlabs.github.io/moderngpu for repository and documentation. + * + ******************************************************************************/ + +#pragma once + +#include "mgpuenums.h" +#include "device/deviceutil.cuh" + +namespace mgpu { + +//////////////////////////////////////////////////////////////////////////////// +// device/loadstore.cuh + +// For 0 <= i < VT: +// index = NT * i + tid; +// reg[i] = data[index]; +// Synchronize after load. +template +MGPU_DEVICE void DeviceSharedToReg(InputIt data, int tid, T* reg, + bool sync = true); + +// For 0 <= i < VT: +// index = NT * i + tid; +// if(index < count) reg[i] = data[index]; +// No synchronize after load. +template +MGPU_DEVICE void DeviceGlobalToReg(int count, InputIt data, int tid, + T* reg, bool sync = false); + +template +MGPU_DEVICE void DeviceGlobalToRegDefault(int count, InputIt data, int tid, + T* reg, T init, bool sync = false); + +// For 0 <= i < VT: +// index = NT * i + tid; +// if(index < count) reg[i] = data[index]; +// No synchronize after load. +template +MGPU_DEVICE void DeviceGlobalToReg(int count, InputIt data, int tid, + T* reg, bool sync = false); + +// For 0 <= i < VT: +// index = NT * i + tid; +// if(index < count) reg[i] = data[index]; +// No synchronize after load. +template +MGPU_DEVICE void DeviceGlobalToRegDefault2(int count, InputIt data, int tid, + T* reg, T init, bool sync = false); + +// For 0 <= i < VT: +// index = NT * i + tid; +// if(index < count) reg[i] = data[index]; +// No synchronize after load. +// No optimized code path for count < NV (smaller generated code). +template +MGPU_DEVICE void DeviceGlobalToRegLoop(int count, InputIt data, int tid, + T* reg, bool sync = false); + + +// For 0 <= i < VT: +// index = VT * tid + i. +// if(index < count) reg[i] = data[index]; +// No synchronize after load. +template +MGPU_DEVICE void DeviceGlobalToThread(int count, InputIt data, int tid, + T* reg); + +template +MGPU_DEVICE void DeviceGlobalToThreadDefault(int count, InputIt data, int tid, + T* reg, T init); + +// For 0 <= i < VT: +// index = NT * i + tid; +// if(index < count) data[index] = reg[i]; +// Synchronize after load. +template +MGPU_DEVICE void DeviceRegToShared(const T* reg, int tid, OutputIt dest, + bool sync = true); + +// For 0 <= i < VT: +// index = NT * i + tid; +// if(index < count) data[index] = reg[i]; +// No synchronize after load. +template +MGPU_DEVICE void DeviceRegToGlobal(int count, const T* reg, int tid, + OutputIt dest, bool sync = false); + +// For 0 <= index < count: +// dest[index] = source[index]; +// This function is intended to replace DeviceGlobalToShared in cases where +// count is much less than NT * VT. +template +MGPU_DEVICE void DeviceMemToMemLoop(int count, InputIt source, int tid, + OutputIt dest, bool sync = true); + +// For 0 <= index < count: +// dest[index] = source[index]; +// Synchronize after store. +template +MGPU_DEVICE void DeviceSharedToGlobal(int count, const T* source, int tid, + OutputIt dest, bool sync = true); + +// For 0 <= index < count: +// dest[index] = source[index]; +// Synchronize after store. +template +MGPU_DEVICE void DeviceGlobalToShared(int count, InputIt source, int tid, + T* dest, bool sync = true); + +template +MGPU_DEVICE void DeviceGlobalToShared2(int count, InputIt source, int tid, + T* dest, bool sync = true); + +// For 0 <= index < count: +// dest[index] = source[index]; +// Synchronize after store. +// No optimized code path for count < NV (smaller generated code). +template +MGPU_DEVICE void DeviceGlobalToSharedLoop(int count, InputIt source, int tid, + T* dest, bool sync = true); + +template +MGPU_DEVICE void DeviceGlobalToSharedDefault(int count, InputIt source, int tid, + T* dest, T init, bool sync = true); + +template +MGPU_DEVICE void DeviceGlobalToSharedDefault2(int count, InputIt source, + int tid, T* dest, T init, bool sync = true); + +// For 0 <= index < count: +// dest[index] = source[index]; +// No synchronize. +template +MGPU_DEVICE void DeviceGlobalToGlobal(int count, InputIt source, int tid, + OutputIt dest, bool sync = false); + +// Transponse VT elements in NT threads (x) into thread-order registers (y) +// using only NT * VT / 2 elements of shared memory. +template +MGPU_DEVICE void HalfSmemTranspose(const T* x, int tid, T* shared, T* y); + +// For 0 <= i < VT: +// index = NT * i + tid; +// if(index < count) +// gather = indices[index]; +// reg[i] = data[gather]; +// Synchronize after load. +template +MGPU_DEVICE void DeviceGather(int count, InputIt data, int indices[VT], + int tid, T* reg, bool sync = true); + +template +MGPU_DEVICE void DeviceGatherDefault(int count, InputIt data, int indices[VT], + int tid, T* reg, T identity, bool sync = true); + +// For 0 <= i < VT: +// index = NT * i + tid; +// if(index < count) +// scatter = indices[index]; +// data[scatter] = reg[i]; +// Synchronize after store. +template +MGPU_DEVICE void DeviceScatter(int count, const T* reg, int tid, + int indices[VT], OutputIt data, bool sync = true); + +// For 0 <= i < VT: +// shared[VT * tid + i] = threadReg[i]; +// Synchronize after store. +// Note this function moves data in THREAD ORDER. +// (DeviceRegToShared moves data in STRIDED ORDER). +template +MGPU_DEVICE void DeviceThreadToShared(const T* threadReg, int tid, T* shared, + bool sync = true); + +// For 0 <= i < VT: +// threadReg[i] = shared[VT * tid + i]; +// Synchronize after load. +// Note this function moves data in THREAD ORDER. +// (DeviceSharedToReg moves data in STRIDED ORDER). +template +MGPU_DEVICE void DeviceSharedToThread(const T* shared, int tid, T* threadReg, + bool sync = true); + +// For 0 <= index < aCount: +// shared[index] = a_global[index]; +// For 0 <= index < bCount: +// shared[aCount + index] = b_global[index]; +// VT0 is the lower-bound for predication-free execution: +// If count >= NT * VT0, a predication-free branch is taken. +// VT1 is the upper-bound for loads: +// NT * VT1 must >= aCount + bCount. + +template +MGPU_DEVICE void DeviceLoad2ToReg(const T* a_global, int aCount, + const T* b_global, int bCount, int tid, T* reg, bool sync = false); + +template +MGPU_DEVICE void DeviceLoad2ToShared(const T* a_global, int aCount, + const T* b_global, int bCount, int tid, T* shared, bool sync = true); + +template +MGPU_DEVICE void DeviceLoad2ToReg(InputIt1 a_global, int aCount, + InputIt2 b_global, int bCount, int tid, T* reg, bool sync = false); + +template +MGPU_DEVICE void DeviceLoad2ToShared(InputIt1 a_global, int aCount, + InputIt2 b_global, int bCount, int tid, T* shared, bool sync = true); + +// For 0 <= i < VT +// index = NT * i + tid; +// if(index < count) +// gather = indices_shared[index]; +// dest_global[index] = data_global[gather]; +// Synchronize after load. +template +MGPU_DEVICE void DeviceGatherGlobalToGlobal(int count, InputIt data_global, + const int* indices_shared, int tid, OutputIt dest_global, + bool sync = true); + +// For 0 <= i < VT +// index = NT * i + tid +// if(index < count) +// gather = indices[index]; +// if(gather < aCount) data = a_global[gather]; +// else data = b_global[gather - aCount]; +// dest_global[index] = data; +// Synchronize after load. +template +MGPU_DEVICE void DeviceTransferMergeValuesReg(int count, InputIt1 a_global, + InputIt2 b_global, int bStart, const int* indices, int tid, + T* reg, bool sync = false); + +template +MGPU_DEVICE void DeviceTransferMergeValuesShared(int count, InputIt1 a_global, + InputIt2 b_global, int bStart, const int* indices_shared, int tid, + OutputIt dest_global, bool sync = true); + +template +MGPU_DEVICE void DeviceTransferMergeValuesReg(int count, const T* a_global, + const T* b_global, int bStart, const int* indices, int tid, + T* reg, bool sync = false); + +template +MGPU_DEVICE void DeviceTransferMergeValuesShared(int count, const T* a_global, + const T* b_global, int bStart, const int* indices_shared, int tid, + OutputIt dest_global, bool sync = true); + + + +} // namespace mgpu + + +#include "device/loadstore.cuh" +#include "device/ctasegscan.cuh" diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/mgpuenums.h b/src/operator/nn/ctc_include/contrib/moderngpu/include/mgpuenums.h new file mode 100644 index 000000000000..be2b8314a8ad --- /dev/null +++ b/src/operator/nn/ctc_include/contrib/moderngpu/include/mgpuenums.h @@ -0,0 +1,70 @@ +/****************************************************************************** + * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions are met: + * * Redistributions of source code must retain the above copyright + * notice, this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright + * notice, this list of conditions and the following disclaimer in the + * documentation and/or other materials provided with the distribution. + * * Neither the name of the NVIDIA CORPORATION nor the + * names of its contributors may be used to endorse or promote products + * derived from this software without specific prior written permission. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY + * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND + * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + * + ******************************************************************************/ + +/****************************************************************************** + * + * Code and text by Sean Baxter, NVIDIA Research + * See http://nvlabs.github.io/moderngpu for repository and documentation. + * + ******************************************************************************/ + +#pragma once + +namespace mgpu { + +enum MgpuBounds { + MgpuBoundsLower, + MgpuBoundsUpper +}; + +enum MgpuScanType { + MgpuScanTypeExc, + MgpuScanTypeInc +}; + +enum MgpuSearchType { + MgpuSearchTypeNone, + MgpuSearchTypeIndex, + MgpuSearchTypeMatch, + MgpuSearchTypeIndexMatch +}; + +enum MgpuJoinKind { + MgpuJoinKindInner, + MgpuJoinKindLeft, + MgpuJoinKindRight, + MgpuJoinKindOuter +}; + +enum MgpuSetOp { + MgpuSetOpIntersection, + MgpuSetOpUnion, + MgpuSetOpDiff, + MgpuSetOpSymDiff +}; + +} // namespace mgpu diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/util/static.h b/src/operator/nn/ctc_include/contrib/moderngpu/include/util/static.h new file mode 100644 index 000000000000..c7209077506a --- /dev/null +++ b/src/operator/nn/ctc_include/contrib/moderngpu/include/util/static.h @@ -0,0 +1,183 @@ +/****************************************************************************** + * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the following conditions are met: + * * Redistributions of source code must retain the above copyright + * notice, this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright + * notice, this list of conditions and the following disclaimer in the + * documentation and/or other materials provided with the distribution. + * * Neither the name of the NVIDIA CORPORATION nor the + * names of its contributors may be used to endorse or promote products + * derived from this software without specific prior written permission. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY + * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND + * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + * + ******************************************************************************/ + +/****************************************************************************** + * + * Code and text by Sean Baxter, NVIDIA Research + * See http://nvlabs.github.io/moderngpu for repository and documentation. + * + ******************************************************************************/ + +#pragma once + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#ifndef MGPU_MIN +#define MGPU_MIN(x, y) (((x) <= (y)) ? (x) : (y)) +#define MGPU_MAX(x, y) (((x) >= (y)) ? (x) : (y)) +#define MGPU_MAX0(x) (((x) >= 0) ? (x) : 0) +#define MGPU_ABS(x) (((x) >= 0) ? (x) : (-x)) + +#define MGPU_DIV_UP(x, y) (((x) + (y) - 1) / (y)) +#define MGPU_DIV_ROUND(x, y) (((x) + (y) / 2) / (y)) +#define MGPU_ROUND_UP(x, y) ((y) * MGPU_DIV_UP(x, y)) +#define MGPU_SHIFT_DIV_UP(x, y) (((x) + ((1<< (y)) - 1))>> y) +#define MGPU_ROUND_UP_POW2(x, y) (((x) + (y) - 1) & ~((y) - 1)) +#define MGPU_ROUND_DOWN_POW2(x, y) ((x) & ~((y) - 1)) +#define MGPU_IS_POW_2(x) (0 == ((x) & ((x) - 1))) + +#endif // MGPU_MIN + +namespace mgpu { + + +typedef unsigned char byte; + +typedef unsigned int uint; +typedef signed short int16; + +typedef unsigned short ushort; +typedef unsigned short uint16; + +typedef long long int64; +typedef unsigned long long uint64; + +// IsPow2::value is true if X is a power of 2. +template struct sIsPow2 { + enum { value = 0 == (X & (X - 1)) }; +}; + +// Finds the base-2 logarithm of X. value is -1 if X is not a power of 2. +template struct sLogPow2 { + enum { extra = sIsPow2::value ? 0 : (roundUp ? 1 : 0) }; + enum { inner = sLogPow2::inner + 1 }; + enum { value = inner + extra }; +}; +template struct sLogPow2<0, roundUp> { + enum { inner = 0 }; + enum { value = 0 }; +}; +template struct sLogPow2<1, roundUp> { + enum { inner = 0 }; + enum { value = 0 }; +}; + +template +struct sDivUp { + enum { value = (X + Y - 1) / Y }; +}; + +template struct sDiv2RoundUp { + enum { value = sDiv2RoundUp::value, levels - 1>::value }; +}; +template struct sDiv2RoundUp { + enum { value = count }; +}; + +template +struct sDivSafe { + enum { value = X / Y }; +}; +template +struct sDivSafe { + enum { value = 0 }; +}; + +template +struct sRoundUp { + enum { rem = X % Y }; + enum { value = X + (rem ? (Y - rem) : 0) }; +}; + +template +struct sRoundDown { + enum { rem = X % Y }; + enum { value = X - rem }; +}; + +// IntegerDiv is a template for avoiding divisions by zero in template +// evaluation. Templates always evaluate both b and c in an expression like +// a ? b : c, and will error if either rhs contains an illegal expression, +// even if the ternary is explictly designed to guard against that. +template +struct sIntegerDiv { + enum { value = X / (Y ? Y : (X + 1)) }; +}; + +template +struct sMax { + enum { value = (X >= Y) ? X : Y }; +}; +template +struct sMin { + enum { value = (X <= Y) ? X : Y }; +}; + +template +struct sAbs { + enum { value = (X >= 0) ? X : -X }; +}; + + +// Finds the number of powers of 2 in the prime factorization of X. +template struct sNumFactorsOf2 { + enum { shifted = X >> 1 }; + enum { value = 1 + sNumFactorsOf2::value }; +}; +template struct sNumFactorsOf2 { + enum { value = 0 }; +}; + +// Returns the divisor for a conflict-free transpose. +template struct sBankConflictDivisor { + enum { value = + (1 & X) ? 0 : + (sIsPow2::value ? NumBanks : + (1<< sNumFactorsOf2::value)) }; + enum { log_value = sLogPow2::value }; +}; + +template struct sConflictFreeStorage { + enum { count = NT * X }; + enum { divisor = sBankConflictDivisor::value }; + enum { padding = sDivSafe::value }; + enum { value = count + padding }; +}; + +} // namespace mgpu diff --git a/src/operator/nn/ctc_include/detail/cpu_ctc.h b/src/operator/nn/ctc_include/detail/cpu_ctc.h new file mode 100644 index 000000000000..005b956343d4 --- /dev/null +++ b/src/operator/nn/ctc_include/detail/cpu_ctc.h @@ -0,0 +1,509 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +#pragma once + +#include +#include +#include +#include +#include + +#include + +#include "ctc_helper.h" + +namespace mxnet_warpctc { + +template +class CpuCTC { +public: + // Noncopyable + CpuCTC(int alphabet_size, int minibatch, void* workspace, + int blank_label) : + alphabet_size_(alphabet_size), minibatch_(minibatch), + workspace_(workspace), blank_label_(blank_label) { + + }; + + CpuCTC(const CpuCTC&) = delete; + CpuCTC& operator=(const CpuCTC&) = delete; + + ctcStatus_t cost_and_grad(const ProbT* const activations, + ProbT *grads, + ProbT* costs, + const int* const flat_labels, + const int* const label_lengths, + const int* const input_lengths); + + + ctcStatus_t score_forward(const ProbT* const activations, + ProbT* costs, + const int* const flat_labels, + const int* const label_lengths, + const int* const input_lengths); + +private: + + class CpuCTC_metadata { + + private: + int setup_labels(const int* const labels, int blank_label, int L, int S); + + public: + CpuCTC_metadata(int L, int S, int T, int mb, int alphabet_size, + void* workspace, size_t bytes_used, int blank_label, + const int* const labels); + + ProbT* alphas; + ProbT* betas; + int* labels_w_blanks; + int* e_inc; + int* s_inc; + ProbT* output; + int repeats; + }; + + int alphabet_size_; // Number of characters plus blank + int minibatch_; + void* workspace_; + int blank_label_; + + void log_softmax(const ProbT* const activations, ProbT* log_probs, + const int* const input_lengths); + + std::tuple + cost_and_grad_kernel(ProbT *grad, const ProbT* const log_probs, + const int* const labels, int T, int L, + int mb, size_t bytes_used); + + ProbT compute_alphas(const ProbT* log_probs, int repeats, int S, int T, + const int* const e_inc, + const int* const s_inc, + const int* const labels, + ProbT* alphas); + + ProbT compute_betas_and_grad(ProbT* grad, const ProbT* const log_probs, + ProbT log_partition, int repeats, + int S, int T, const int* const e_inc, + const int* const s_inc, + const int* const labels, + ProbT* alphas, + ProbT* betas, + ProbT* output); +}; + +template +CpuCTC::CpuCTC_metadata::CpuCTC_metadata(int L, int S, int T, int mb, + int alphabet_size, + void* workspace, size_t bytes_used, + int blank_label, + const int* const labels) { + + alphas = reinterpret_cast(static_cast(workspace) + bytes_used); + bytes_used += sizeof(ProbT) * S * T; + std::fill(alphas, alphas + S * T, ctc_helper::neg_inf()); + betas = reinterpret_cast(static_cast(workspace) + bytes_used); + bytes_used += sizeof(ProbT) * S; + std::fill(betas, betas + S, ctc_helper::neg_inf()); + labels_w_blanks = reinterpret_cast(static_cast(workspace) + bytes_used); + bytes_used += sizeof(int) * S; + e_inc = reinterpret_cast(static_cast(workspace) + bytes_used); + bytes_used += sizeof(int) * S; + s_inc = reinterpret_cast(static_cast(workspace) + bytes_used); + bytes_used += sizeof(int) * S; + output = reinterpret_cast(static_cast(workspace) + bytes_used); + bytes_used += sizeof(ProbT) * alphabet_size; + + repeats = setup_labels(labels, blank_label, L, S); +} + +template +int CpuCTC::CpuCTC_metadata::setup_labels(const int* const labels, + int blank_label, int L, int S) { + int e_counter = 0; + int s_counter = 0; + + s_inc[s_counter++] = 1; + + int repeats = 0; + + for (int i = 1; i < L; ++i) { + if (labels[i-1] == labels[i]) { + s_inc[s_counter++] = 1; + s_inc[s_counter++] = 1; + e_inc[e_counter++] = 1; + e_inc[e_counter++] = 1; + ++repeats; + } + else { + s_inc[s_counter++] = 2; + e_inc[e_counter++] = 2; + } + } + e_inc[e_counter++] = 1; + + for (int i = 0; i < L; ++i) { + labels_w_blanks[2 * i] = blank_label; + labels_w_blanks[2 * i + 1] = labels[i]; + } + labels_w_blanks[S - 1] = blank_label; + + return repeats; +} + +template +void +CpuCTC::log_softmax(const ProbT* const activations, ProbT* log_probs, + const int* const input_lengths) { +#pragma omp parallel for + for (int mb = 0; mb < minibatch_; ++mb) { + for(int c = 0; c < input_lengths[mb]; ++c) { + int col_offset = (mb + minibatch_ * c) * alphabet_size_; + ProbT max_activation = -std::numeric_limits::infinity(); + for(int r = 0; r < alphabet_size_; ++r) + max_activation = std::max(max_activation, activations[r + col_offset]); + + ProbT denom = ProbT(0.); + for(int r = 0; r < alphabet_size_; ++r) { + denom += std::exp(activations[r + col_offset] - max_activation); + } + + for(int r = 0; r < alphabet_size_; ++r) { + log_probs[r + col_offset] = activations[r + col_offset] + - max_activation - std::log(denom); + } + } + } +} + +template +std::tuple +CpuCTC::cost_and_grad_kernel(ProbT *grad, const ProbT* const log_probs, + const int* const labels, + int T, int L, int mb, size_t bytes_used) { + + const int S = 2*L + 1; // Number of labels with blanks + + CpuCTC_metadata ctcm(L, S, T, mb, alphabet_size_, workspace_, bytes_used, blank_label_, labels); + + bool over_threshold = false; + + if (L + ctcm.repeats > T) { + return std::make_tuple(ProbT(0), over_threshold); // TODO, not right to return 0 + } + + ProbT llForward = compute_alphas(log_probs, ctcm.repeats, S, T, ctcm.e_inc, + ctcm.s_inc, ctcm.labels_w_blanks, + ctcm.alphas); + + ProbT llBackward = compute_betas_and_grad(grad, log_probs, llForward, ctcm.repeats, + S, T, ctcm.e_inc, ctcm.s_inc, + ctcm.labels_w_blanks, + ctcm.alphas, + ctcm.betas, + ctcm.output); + + ProbT diff = std::abs(llForward - llBackward); + if (diff > ctc_helper::threshold) { + over_threshold = true; + } + + return std::make_tuple(-llForward, over_threshold); +} + +// Computes forward probabilities +template +ProbT CpuCTC::compute_alphas(const ProbT* log_probs, int repeats, int S, int T, + const int* const e_inc, + const int* const s_inc, + const int* const labels, + ProbT* alphas) { + + int start = (((S /2) + repeats - T) < 0) ? 0 : 1, + end = S > 1 ? 2 : 1; + + for (int i = start; i < end; ++i) { + alphas[i] = log_probs[labels[i]]; + } + + for(int t = 1; t < T; ++t) { + int remain = (S / 2) + repeats - (T - t); + if(remain >= 0) + start += s_inc[remain]; + if(t <= (S / 2) + repeats) + end += e_inc[t - 1]; + int startloop = start; + int idx1 = t * S, idx2 = (t - 1) * S, idx3 = t * (alphabet_size_ * minibatch_); + + if (start == 0) { + alphas[idx1] = alphas[idx2] + log_probs[blank_label_ + idx3]; + startloop += 1; + } + + for(int i = startloop; i < end; ++i) { + ProbT prev_sum = ctc_helper::log_plus()(alphas[i + idx2], alphas[(i-1) + idx2]); + + // Skip two if not on blank and not on repeat. + if (labels[i] != blank_label_ && i != 1 && labels[i] != labels[i-2]) + prev_sum = ctc_helper::log_plus()(prev_sum, alphas[(i-2) + idx2]); + + alphas[i + idx1] = prev_sum + log_probs[labels[i] + idx3]; + } + } + + ProbT loglike = ctc_helper::neg_inf(); + for(int i = start; i < end; ++i) { + loglike = ctc_helper::log_plus()(loglike, alphas[i + (T - 1) * S]); + } + + return loglike; +} + +// Starting from T, we sweep backward over the alpha array computing one column +// of betas as we go. At each position we can update product alpha * beta and then +// sum into the gradient associated with each label. +// NOTE computes gradient w.r.t UNNORMALIZED final layer activations. +// Assumed passed in grads are already zeroed! +template +ProbT CpuCTC::compute_betas_and_grad(ProbT* grad, const ProbT* const log_probs, + ProbT log_partition, int repeats, + int S, int T, const int* const e_inc, + const int* const s_inc, + const int* const labels, + ProbT* alphas, + ProbT* betas, + ProbT* output) { + int start = S > 1 ? (S - 2) : 0, + end = (T > (S / 2) + repeats) ? S : S-1; + + std::fill(output, output + alphabet_size_, ctc_helper::neg_inf()); + + //set the starting values in the beta column at the very right edge + for (int i = start; i < end; ++i) { + betas[i] = log_probs[labels[i] + (T - 1) * (alphabet_size_ * minibatch_)]; + + //compute alpha * beta in log space at this position in (S, T) space + alphas[i + (T - 1) * S] += betas[i]; + + //update the gradient associated with this label + //essentially performing a reduce-by-key in a sequential manner + output[labels[i]] = + ctc_helper::log_plus()(alphas[i + (T - 1) * S], output[labels[i]]); + } + + //update the gradient wrt to each unique label + for (int i = 0; i < alphabet_size_; ++i) { + int idx3 = (T - 1) * alphabet_size_ * minibatch_ + i; + + if (output[i] == 0.0 || output[i] == ctc_helper::neg_inf() || + log_probs[idx3] == ctc_helper::neg_inf()) { + grad[idx3] = std::exp(log_probs[idx3]); + } else { + grad[idx3] = std::exp(log_probs[idx3]) + - std::exp(output[i] - log_probs[idx3] - log_partition); + } + } + + //loop from the second to last column all the way to the left + for(int t = T - 2; t >= 0; --t) { + int remain = (S / 2) + repeats - (T - t); + if(remain >= -1) + start -= s_inc[remain + 1]; + if(t < (S / 2) + repeats) + end -= e_inc[t]; + + int endloop = end == S ? end - 1 : end; + int idx1 = t * S, idx3 = t * (alphabet_size_ * minibatch_); + + std::fill(output, output + alphabet_size_, ctc_helper::neg_inf()); + + for(int i = start; i < endloop; ++i) { + ProbT next_sum = ctc_helper::log_plus()(betas[i], betas[(i+1)]); + // Skip two if not on blank and not on repeat. + if (labels[i] != blank_label_ && i != (S-2) && labels[i] != labels[i+2]){ + next_sum = ctc_helper::log_plus()(next_sum, betas[(i+2)]); + } + betas[i] = next_sum + log_probs[labels[i] + idx3]; + + //compute alpha * beta in log space + alphas[i + idx1] += betas[i]; + + //update the gradient associated with this label + output[labels[i]] = + ctc_helper::log_plus()(alphas[i + idx1], output[labels[i]]); + } + + if (end == S) { + betas[(S-1)] = betas[(S-1)] + log_probs[blank_label_ + idx3]; + alphas[(S-1) + idx1] += betas[(S-1)]; + + output[labels[S-1]] = + ctc_helper::log_plus()(alphas[S-1 + idx1], output[labels[S-1]]); + } + + //go over the unique labels and compute the final grad + // wrt to each one at this time step + for (int i = 0; i < alphabet_size_; ++i) { + + if (output[i] == 0.0 || output[i] == ctc_helper::neg_inf() || + log_probs[idx3] == ctc_helper::neg_inf()) { + grad[idx3] = std::exp(log_probs[idx3]); + } else { + grad[idx3] = std::exp(log_probs[idx3]) + - std::exp(output[i] - log_probs[idx3] - log_partition); + } + ++idx3; + } + } + + ProbT loglike = ctc_helper::neg_inf(); + for(int i = start; i < end; ++i) { + loglike = ctc_helper::log_plus()(loglike, betas[i]); + } + + return loglike; +} + +template +ctcStatus_t +CpuCTC::cost_and_grad(const ProbT* const activations, + ProbT *grads, + ProbT *costs, + const int* const flat_labels, + const int* const label_lengths, + const int* const input_lengths) { + if (activations == nullptr || + grads == nullptr || + costs == nullptr || + flat_labels == nullptr || + label_lengths == nullptr || + input_lengths == nullptr + ) + return CTC_STATUS_INVALID_VALUE; + + ProbT* log_probs = static_cast(workspace_); + + int maxT = *std::max_element(input_lengths, input_lengths + minibatch_); + + size_t bytes_used = sizeof(ProbT) * minibatch_ * alphabet_size_ * maxT; + + //per minibatch memory + size_t per_minibatch_bytes = 0; + + int maxL = *std::max_element(label_lengths, label_lengths + minibatch_);; + int maxS = 2 * maxL + 1; + + //output + per_minibatch_bytes += sizeof(float) * alphabet_size_; + + //alphas + per_minibatch_bytes += sizeof(float) * maxS * maxT; + + //betas + per_minibatch_bytes += sizeof(float) * maxS; + + //labels w/blanks, e_inc, s_inc + per_minibatch_bytes += 3 * sizeof(int) * maxS; + + log_softmax(activations, log_probs, input_lengths); + +#pragma omp parallel for + for (int mb = 0; mb < minibatch_; ++mb) { + const int T = input_lengths[mb]; // Length of utterance (time) + const int L = label_lengths[mb]; // Number of labels in transcription + + bool mb_status; + + std::tie(costs[mb], mb_status) = + cost_and_grad_kernel(grads + mb * alphabet_size_, + log_probs + mb * alphabet_size_, + flat_labels + std::accumulate(label_lengths, label_lengths + mb, 0), + T, L, mb, + bytes_used + mb * per_minibatch_bytes); + } + + return CTC_STATUS_SUCCESS; +} + +template +ctcStatus_t CpuCTC::score_forward(const ProbT* const activations, + ProbT* costs, + const int* const flat_labels, + const int* const label_lengths, + const int* const input_lengths) { + if (activations == nullptr || + costs == nullptr || + flat_labels == nullptr || + label_lengths == nullptr || + input_lengths == nullptr + ) + return CTC_STATUS_INVALID_VALUE; + + ProbT* log_probs = static_cast(workspace_); + + int maxT = *std::max_element(input_lengths, input_lengths + minibatch_); + + size_t bytes_used = sizeof(ProbT) * minibatch_ * alphabet_size_ * maxT; + + //per minibatch memory + size_t per_minibatch_bytes = 0; + + int maxL = *std::max_element(label_lengths, label_lengths + minibatch_); + int maxS = 2 * maxL + 1; + + //output + per_minibatch_bytes += sizeof(float) * alphabet_size_; + + //alphas + per_minibatch_bytes += sizeof(float) * maxS * maxT; + + //betas + per_minibatch_bytes += sizeof(float) * maxS; + + //labels w/blanks, e_inc, s_inc + per_minibatch_bytes += 3 * sizeof(int) * maxS; + + log_softmax(activations, log_probs, input_lengths); + +#pragma omp parallel for + for (int mb = 0; mb < minibatch_; ++mb) { + const int T = input_lengths[mb]; // Length of utterance (time) + const int L = label_lengths[mb]; // Number of labels in transcription + const int S = 2*L + 1; // Number of labels with blanks + + CpuCTC_metadata ctcm(L, S, T, mb, alphabet_size_, workspace_, + bytes_used + mb * per_minibatch_bytes, blank_label_, + flat_labels + std::accumulate(label_lengths, label_lengths + mb, 0)); + + + if (L + ctcm.repeats > T) + costs[mb] = ProbT(0); + else { + costs[mb] = -compute_alphas(log_probs + mb * alphabet_size_, ctcm.repeats, S, T, + ctcm.e_inc, ctcm.s_inc, ctcm.labels_w_blanks, + ctcm.alphas); + } + + } + + return CTC_STATUS_SUCCESS; +} + +} // mxnet_warpctc diff --git a/src/operator/nn/ctc_include/detail/ctc_helper.h b/src/operator/nn/ctc_include/detail/ctc_helper.h new file mode 100644 index 000000000000..250188c697c6 --- /dev/null +++ b/src/operator/nn/ctc_include/detail/ctc_helper.h @@ -0,0 +1,93 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +#pragma once + +#include +#include +#include + +#include "hostdevice.h" + +typedef enum { + CTC_STATUS_SUCCESS = 0, + CTC_STATUS_MEMOPS_FAILED = 1, + CTC_STATUS_INVALID_VALUE = 2, + CTC_STATUS_EXECUTION_FAILED = 3, + CTC_STATUS_UNKNOWN_ERROR = 4 +} ctcStatus_t; + +typedef enum { + CTC_CPU = 0, + CTC_GPU = 1 +} ctcComputeLocation; + +namespace ctc_helper { + +static const float threshold = 1e-1; + +template +HOSTDEVICE +T neg_inf() { return -T(INFINITY); } + +inline int div_up(int x, int y) { + return (x + y - 1) / y; +} + +template struct maximum { + HOSTDEVICE + Res operator()(const Arg& x, const Arg& y) const { + return x < y ? y : x; + } +}; + +template struct add { + HOSTDEVICE + Res operator()(const Arg& x, const Arg& y) const { + return x + y; + } +}; + +template struct identity { + HOSTDEVICE Res operator()(const Arg& x) const {return Res(x);} +}; + +template struct negate { + HOSTDEVICE Res operator()(const Arg& x) const {return Res(-x);} +}; + +template struct exponential { + HOSTDEVICE Res operator()(const Arg& x) const {return std::exp(x);} +}; + +template +struct log_plus { + typedef Res result_type; + HOSTDEVICE + Res operator()(const Arg1& p1, const Arg2& p2) { + if (p1 == neg_inf()) + return p2; + if (p2 == neg_inf()) + return p1; + Res result = log1p(exp(-fabs(p1 - p2))) + maximum()(p1, p2); + return result; + } +}; + +} diff --git a/src/operator/nn/ctc_include/detail/gpu_ctc.h b/src/operator/nn/ctc_include/detail/gpu_ctc.h new file mode 100644 index 000000000000..2c521b5abb5d --- /dev/null +++ b/src/operator/nn/ctc_include/detail/gpu_ctc.h @@ -0,0 +1,505 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +#pragma once + + +#include "ctc_helper.h" +#include "gpu_ctc_kernels.h" + +namespace mxnet_warpctc { + +template +class GpuCTC { + public: + GpuCTC(int alphabet_size, + int minibatch, + void *workspace, + CUstream stream, + int blank_label) : + out_dim_(alphabet_size), minibatch_(minibatch), + gpu_workspace_(workspace), stream_(stream), + blank_label_(blank_label) {}; + + // Noncopyable + GpuCTC(const GpuCTC&) = delete; + GpuCTC& operator=(const GpuCTC&) = delete; + + ctcStatus_t + cost_and_grad(const ProbT* const activations, + ProbT* grads, + ProbT* costs, + const int* const flat_labels, + const int* const label_lengths, + const int* const input_lengths); + + ctcStatus_t + score_forward(const ProbT* const activations, + ProbT* costs, + const int* const flat_labels, + const int* const label_lengths, + const int* const input_lengths); + + private: + + template + ctcStatus_t launch_alpha_beta_kernels(const ProbT* const log_probs, + ProbT *grads, + bool compute_alpha, + bool compute_beta); + + ctcStatus_t + launch_gpu_kernels(const ProbT* const log_probs, + ProbT *grads, + size_t config, + bool launch_alpha, + bool launch_beta); + + ctcStatus_t + setup_gpu_metadata(const int* const flat_labels, + const int* const label_lengths, + const int* const input_lengths); + + ctcStatus_t + create_metadata_and_choose_config(const int* const label_lengths, + const int* const flat_labels, + const int* const input_lengths, + size_t& best_config); + + ctcStatus_t + compute_log_probs(const ProbT* const activations); + + ctcStatus_t + compute_cost_and_score(const ProbT* const activations, + ProbT* grads, + ProbT* costs, + const int* const flat_labels, + const int* const label_lengths, + const int* const input_lengths, + bool compute_alpha, + bool compute_betas_and_grad); + + + int out_dim_; // Number of characters plus blank + int minibatch_; + + int S_; + int T_; + + int activation_cols_; // Number of columns in activations + + void *gpu_workspace_; // Buffer for all temporary GPU memory + CUstream stream_; + int blank_label_; + + int *utt_length_; // T + int *label_sizes_; // L + int *repeats_; // repeats_ + int *label_offsets_; + int *labels_without_blanks_; + int *labels_with_blanks_; + ProbT *alphas_; + ProbT *nll_forward_; + ProbT *nll_backward_; + ProbT *denoms_; // Temporary storage for denoms for softmax + ProbT *log_probs_; // Temporary storage for probabilities (log softmax output) +}; + +template +ctcStatus_t +GpuCTC::setup_gpu_metadata(const int* const flat_labels, + const int* const label_lengths, + const int* const input_lengths) +{ + size_t gpu_bytes_used = 0; + + nll_forward_ = + reinterpret_cast(static_cast(gpu_workspace_) + + gpu_bytes_used); + gpu_bytes_used += minibatch_ * sizeof(ProbT); + + + nll_backward_ = + reinterpret_cast(static_cast(gpu_workspace_) + + gpu_bytes_used); + gpu_bytes_used += minibatch_ * sizeof(ProbT); + + + repeats_ = + reinterpret_cast(static_cast(gpu_workspace_) + + gpu_bytes_used); + gpu_bytes_used += minibatch_ * sizeof(int); + + label_offsets_ = + reinterpret_cast(static_cast(gpu_workspace_) + + gpu_bytes_used); + gpu_bytes_used += minibatch_ * sizeof(int); + + + // This is the max of all S and T for all valid examples in the minibatch. + // A valid example is one for which L + repeats <= T + S_ = 0; + T_ = 0; + + // This is the max of all timesteps, valid or not. Needed to compute offsets + int Tmax = 0; + + // This is the max of all labels, valid or not. Needed to compute offsets + int Lmax = 0; + int total_label_length = 0; + + constexpr int cpu_buffer_size = 64; + int repeats[cpu_buffer_size]; + int label_offsets[cpu_buffer_size]; + + const int num_passes = ctc_helper::div_up(minibatch_, cpu_buffer_size); + + cudaError_t cuda_status; + + for (int pass = 0; pass < num_passes; ++pass) { + + const int start_idx = pass * cpu_buffer_size; + const int end_idx = std::min(minibatch_, (pass+1) * cpu_buffer_size); + + for (int j = start_idx; j < end_idx; ++j) { + const int L = label_lengths[j]; + const int local_T = input_lengths[j]; + const int *label_ptr = &(flat_labels[total_label_length]); + + label_offsets[j % cpu_buffer_size] = total_label_length; + total_label_length += L; + + int repeat_counter = 0; + + for (int i = 1; i < L; ++i) + repeat_counter += (label_ptr[i] == label_ptr[i-1]); + + repeats[j % cpu_buffer_size] = repeat_counter; + const bool valid_label = ((L + repeat_counter) <= local_T); + + // Only update S and T if label is valid + S_ = (valid_label) ? std::max(S_, L) : S_; + T_ = (valid_label) ? std::max(T_, local_T) : T_; + + Tmax = std::max(Tmax, local_T); + Lmax = std::max(Lmax, L); + } + + cuda_status = cudaMemcpyAsync(&(repeats_[start_idx]), repeats, + (end_idx - start_idx) * sizeof(int), + cudaMemcpyHostToDevice, stream_); + if (cuda_status != cudaSuccess) + return CTC_STATUS_MEMOPS_FAILED; + + + cuda_status = cudaMemcpyAsync(&(label_offsets_[start_idx]), label_offsets, + (end_idx - start_idx) * sizeof(int), + cudaMemcpyHostToDevice, stream_); + if (cuda_status != cudaSuccess) + return CTC_STATUS_MEMOPS_FAILED; + } + + S_ = 2 * S_ + 1; + const int Smax = 2 * Lmax + 1; + + activation_cols_ = minibatch_ * Tmax; + + // Allocate memory for T + utt_length_ = + reinterpret_cast(static_cast(gpu_workspace_) + + gpu_bytes_used); + gpu_bytes_used += minibatch_ * sizeof(int); + + cuda_status = cudaMemcpyAsync(utt_length_, input_lengths, + minibatch_ * sizeof(int), + cudaMemcpyHostToDevice, stream_); + if (cuda_status != cudaSuccess) + return CTC_STATUS_MEMOPS_FAILED; + + label_sizes_ = + reinterpret_cast(static_cast(gpu_workspace_) + + gpu_bytes_used); + gpu_bytes_used += minibatch_ * sizeof(int); + cuda_status = cudaMemcpyAsync(label_sizes_, label_lengths, + minibatch_ * sizeof(int), + cudaMemcpyHostToDevice, stream_); + if (cuda_status != cudaSuccess) + return CTC_STATUS_MEMOPS_FAILED; + + labels_without_blanks_ = + reinterpret_cast(static_cast(gpu_workspace_) + + gpu_bytes_used); + gpu_bytes_used += Lmax * minibatch_ * sizeof(int); + cuda_status = cudaMemcpyAsync(labels_without_blanks_, flat_labels, + total_label_length * sizeof(int), + cudaMemcpyHostToDevice, stream_); + if (cuda_status != cudaSuccess) + return CTC_STATUS_MEMOPS_FAILED; + + labels_with_blanks_ = + reinterpret_cast(static_cast(gpu_workspace_) + + gpu_bytes_used); + gpu_bytes_used += Smax * minibatch_ * sizeof(int); + + alphas_ = + reinterpret_cast(static_cast(gpu_workspace_) + + gpu_bytes_used); + gpu_bytes_used += (S_ * T_) * minibatch_ * sizeof(ProbT); + + + denoms_ = + reinterpret_cast(static_cast(gpu_workspace_) + + gpu_bytes_used); + gpu_bytes_used += activation_cols_ * sizeof(ProbT); + + log_probs_ = + reinterpret_cast(static_cast(gpu_workspace_) + + gpu_bytes_used); + gpu_bytes_used += out_dim_ * activation_cols_ * sizeof(ProbT); + + return CTC_STATUS_SUCCESS; +} + +template +template +ctcStatus_t GpuCTC::launch_alpha_beta_kernels(const ProbT* const log_probs, + ProbT* grads, + bool compute_alpha, + bool compute_beta ) { + + // One thread block per utterance + const int grid_size = minibatch_; + + // The data is laid out so that the next timestep is minibatch entries + // away + const int stride = minibatch_; + + if (compute_alpha) + compute_alpha_kernel<<>> + (log_probs, label_sizes_, utt_length_, + repeats_, labels_without_blanks_, label_offsets_, + labels_with_blanks_, alphas_, nll_forward_, + stride, out_dim_, S_, T_, blank_label_); + + + if (compute_beta) { + compute_betas_and_grad_kernel<<>> + (log_probs, label_sizes_, utt_length_, repeats_, + labels_with_blanks_, alphas_, nll_forward_, nll_backward_, + grads, stride, out_dim_, S_, T_, blank_label_); + + cudaStreamSynchronize(stream_); + } + + cudaError_t err = cudaGetLastError(); + if (err != cudaSuccess) + return CTC_STATUS_EXECUTION_FAILED; + + return CTC_STATUS_SUCCESS; +} + +template +ctcStatus_t +GpuCTC::create_metadata_and_choose_config(const int* const flat_labels, + const int* const label_lengths, + const int* const input_lengths, + size_t& best_config) { + + // Setup the metadata for GPU + ctcStatus_t status = setup_gpu_metadata(flat_labels, label_lengths, input_lengths); + if (status != CTC_STATUS_SUCCESS) + return status; + + constexpr int num_configs = 12; + + int config_NT[num_configs] = + {32, 64, 128, 64, 128, 32, 64, 128, 64, 128, 128, 128}; + int config_VT[num_configs] = + { 1, 1, 1, 3, 2, 9, 6, 4, 9, 6, 9, 10}; + + best_config = 0; + + for (int i = 0; i < num_configs; ++i) { + if ((config_NT[i]* config_VT[i]) >= S_) + break; + else + best_config++; + } + + if (best_config >= num_configs) + return CTC_STATUS_UNKNOWN_ERROR; + + return CTC_STATUS_SUCCESS; +} + +template +ctcStatus_t +GpuCTC::launch_gpu_kernels(const ProbT* const log_probs, + ProbT* grads, + size_t config, + bool l_a, + bool l_b) { + + switch(config) { + case 0: {return launch_alpha_beta_kernels<32, 1>(log_probs, grads, l_a, l_b);} + case 1: {return launch_alpha_beta_kernels<64, 1>(log_probs, grads, l_a, l_b);} + case 2: {return launch_alpha_beta_kernels<128, 1>(log_probs, grads, l_a, l_b);} + case 3: {return launch_alpha_beta_kernels<64, 3>(log_probs, grads, l_a, l_b);} + case 4: {return launch_alpha_beta_kernels<128, 2>(log_probs, grads, l_a, l_b);} + case 5: {return launch_alpha_beta_kernels<32, 9>(log_probs, grads, l_a, l_b);} + case 6: {return launch_alpha_beta_kernels<64, 6>(log_probs, grads, l_a, l_b);} + case 7: {return launch_alpha_beta_kernels<128, 4>(log_probs, grads, l_a, l_b);} + case 8: {return launch_alpha_beta_kernels<64, 9>(log_probs, grads, l_a, l_b);} + case 9: {return launch_alpha_beta_kernels<128, 6>(log_probs, grads, l_a, l_b);} + case 10: {return launch_alpha_beta_kernels<128, 9>(log_probs, grads, l_a, l_b);} + case 11: {return launch_alpha_beta_kernels<128, 10>(log_probs, grads, l_a, l_b);} + } + + return CTC_STATUS_EXECUTION_FAILED; +} + +template +ctcStatus_t +GpuCTC::compute_log_probs(const ProbT* const activations) { + + cudaError_t cuda_status; + cuda_status = + cudaMemcpyAsync(log_probs_, activations, + activation_cols_ * out_dim_ *sizeof(ProbT), + cudaMemcpyDeviceToDevice, stream_); + if (cuda_status != cudaSuccess) + return CTC_STATUS_MEMOPS_FAILED; + + + + // create mshadow handles to data + using namespace mshadow; + using namespace mshadow::expr; + Stream mxstream; + mxstream.stream_ = stream_; + Tensor log_probs_handle(log_probs_, mshadow::Shape2(activation_cols_, out_dim_), &mxstream); + Tensor denoms_handle(denoms_, mshadow::Shape1(activation_cols_), &mxstream); + denoms_handle = reduce_with_axis(log_probs_handle, 1); + + + + // Kernel launch to subtract maximum + const int NT = 128; + const int VT = 1; + const int NV = NT * VT; + const int num_elements = out_dim_ * activation_cols_; + const int grid_size = ctc_helper::div_up(num_elements, NV); + + prepare_stable_LSM_kernel <<< grid_size, NT, 0, stream_>>> + (ctc_helper::identity(), log_probs_, + denoms_, out_dim_, num_elements); + + // compute denominators for softmax + denoms_handle = reduce_with_axis(F(log_probs_handle), 1); + + // Kernel launch to calculate probabilities + compute_log_probs_kernel<<>> + (ctc_helper::identity(), log_probs_, + denoms_, out_dim_, num_elements); + + cuda_status = cudaGetLastError(); + if (cuda_status != cudaSuccess) + return CTC_STATUS_EXECUTION_FAILED; + + return CTC_STATUS_SUCCESS; +} + +template +ctcStatus_t +GpuCTC::compute_cost_and_score(const ProbT* const activations, + ProbT* grads, + ProbT* costs, + const int* const flat_labels, + const int* const label_lengths, + const int* const input_lengths, + bool compute_alpha, + bool compute_betas_and_grad) { + + size_t best_config; + ctcStatus_t status = create_metadata_and_choose_config(flat_labels, + label_lengths, + input_lengths, + best_config); + if (status != CTC_STATUS_SUCCESS) + return status; + + status = compute_log_probs(activations); + if (status != CTC_STATUS_SUCCESS) + return status; + + launch_gpu_kernels(log_probs_, grads, best_config, + compute_alpha, compute_betas_and_grad); + + cudaError_t cuda_status_mem, cuda_status_sync; + cuda_status_mem = cudaMemcpyAsync(costs, nll_forward_, + sizeof(ProbT) * minibatch_, + cudaMemcpyDeviceToHost, stream_); + cuda_status_sync = cudaStreamSynchronize(stream_); + if (cuda_status_mem != cudaSuccess || cuda_status_sync != cudaSuccess) + return CTC_STATUS_MEMOPS_FAILED; + + return CTC_STATUS_SUCCESS; +} + +template +ctcStatus_t +GpuCTC::cost_and_grad(const ProbT* const activations, + ProbT* grads, + ProbT* costs, + const int* const flat_labels, + const int* const label_lengths, + const int* const input_lengths) { + if (activations == nullptr || + grads == nullptr || + costs == nullptr || + flat_labels == nullptr || + label_lengths == nullptr || + input_lengths == nullptr + ) + return CTC_STATUS_INVALID_VALUE; + + return compute_cost_and_score(activations, grads, costs, flat_labels, + label_lengths, input_lengths, true, true); +} + +template +ctcStatus_t +GpuCTC::score_forward(const ProbT* const activations, + ProbT* costs, + const int* const flat_labels, + const int* const label_lengths, + const int* const input_lengths) { + if (activations == nullptr || + costs == nullptr || + flat_labels == nullptr || + label_lengths == nullptr || + input_lengths == nullptr + ) + return CTC_STATUS_INVALID_VALUE; + + return compute_cost_and_score(activations, nullptr, costs, flat_labels, + label_lengths, input_lengths, true, false); +} + +} // mxnet_warpctc diff --git a/src/operator/nn/ctc_include/detail/gpu_ctc_kernels.h b/src/operator/nn/ctc_include/detail/gpu_ctc_kernels.h new file mode 100644 index 000000000000..c9bc2026efb5 --- /dev/null +++ b/src/operator/nn/ctc_include/detail/gpu_ctc_kernels.h @@ -0,0 +1,507 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +#pragma once + +#include "../contrib/moderngpu/include/device/ctascan.cuh" +#include "../contrib/moderngpu/include/device/ctamerge.cuh" + +#include "ctc_helper.h" + +using namespace mgpu; + +template +struct CTASegReduce { + + enum {NV = NT * VT}; + + union Storage { + typename CTAScan::Storage scanStorage; + int indices[NV]; + }; + + //adapted from global kernel KernelReduceByKeyPreprocess + __device__ static void preprocessKeys(KeyT *keys, int count, + int *numUniqueLabels, int seg_start[VT], + int seg_end[VT], int *scanout) { + __shared__ Storage shared; + + const int tid = threadIdx.x; + // Compare adjacent keys within each thread and mark discontinuities + int endFlags = 0; + T key = keys[VT * tid]; + #pragma unroll + for (int i = 0; i < VT; ++i) { + int index = VT * tid + 1 + i; + T next = keys[index]; + if(index == count || (index < count && key != next)) { + endFlags |= 1 << i; + } + key = next; + } + + __syncthreads(); + + //Count the number of encountered end flags + int scan = CTAScan::Scan(tid, popc(endFlags), shared.scanStorage, numUniqueLabels); + + __syncthreads(); + + //output the unique keys + //use indices as scratch space + int outputPos = scan; + #pragma unroll + for (int i = 0; i < VT; ++i) { + + if ( (endFlags >> i) & 1) { + shared.indices[outputPos] = keys[VT * tid + i]; + scanout[outputPos] = VT * tid + i; + outputPos++; + } + } + + __syncthreads(); + + // Create start and end + for (int idx = tid, j = 0; idx < (*numUniqueLabels); idx += blockDim.x, ++j) { + seg_start[j] = (idx == 0) ? 0 : (scanout[idx-1] + 1); + seg_end[j] = scanout[idx]; + } + + __syncthreads(); + + //copy from the scratch space back into the keys + #pragma unroll + for (int i = 0; i < VT; ++i) { + keys[i * NT + tid] = shared.indices[i * NT + tid]; + } + + __syncthreads(); + } +}; + +// Computes forward probabilities. This fills in a T * S matrix. +// The computation starts at t=1 (2nd row) and ends at t=T-1 (last row). Each row has +// S elements where S = 2L + 1. +// +// We only need to read in probabilities corresponding to the labels, thus a sparse +// set of values are read from the log probs matrix since the character set is much smaller +// than the labels. This is much more true for Mandarin than English. +template +__global__ +void compute_alpha_kernel (const ProbT* log_probs, const int *label_sizes, + const int *utt_length, const int *repeats_in_labels, + const int *labels_without_blanks, const int *label_offsets, + int *labels_with_blanks, ProbT *alphas, + ProbT* nll_forward, int stride, int out_dim, + int S_memoffset, int T_memoffset, int blank_label) { + + ctc_helper::log_plus log_plus_f; + + const int tid = threadIdx.x; + const int L = label_sizes[blockIdx.x]; + const int T = utt_length[blockIdx.x]; + const int S = 2*L + 1; + const int prob_offset = out_dim * blockIdx.x; + const int repeats = repeats_in_labels[blockIdx.x]; + + const int NV = NT * VT; + __shared__ int label[NV]; + + if ((L + repeats) > T) + return; + + // Generate labels with blanks from labels without blanks + { + const int label_start_offset = label_offsets[blockIdx.x]; + for (int idx = tid; idx < L; idx += blockDim.x) { + const int offset = (blockIdx.x * S_memoffset) + 2 * idx; + labels_with_blanks[offset] = blank_label; + labels_with_blanks[offset+1] = labels_without_blanks[label_start_offset + idx]; + } + if (tid == 0) { + labels_with_blanks[(blockIdx.x * S_memoffset) + 2 * L] = blank_label; + } + } + __syncthreads(); + + const int *labels = labels_with_blanks; + const int* label_global = &labels[blockIdx.x * S_memoffset]; + ProbT* alpha = &alphas[blockIdx.x * (S_memoffset * T_memoffset)]; + + // Set the first row of alpha neg_inf - it is much more efficient to do it + // here than outside + #pragma unroll + for (int idx = tid; idx < min(S, NV); idx += blockDim.x) { + alpha[idx] = ctc_helper::neg_inf(); + } + + // Load labels into shared memory + #pragma unroll + for (int i = tid; i < S; i += NT) { + label[i] = label_global[i]; + } + + __syncthreads(); + + int start = (L + repeats < T) ? 0 : 1; + int end = S > 1 ? 2 : 1; + + // Initialize the first row corresponding to t=0; + for(int i = tid; i < (end-start); i += blockDim.x) + alpha[i + start] = log_probs[prob_offset + label[i + start]]; + + __syncthreads(); + + // Fill in the rest of matrix, one row at a time (outer loop). + for(int t = 1; t < T; ++t) { + + // Start offsets into the current and previous row + const int start_cur_row = t * S; + const int start_prev_row = (t - 1) * S; + + // The prob is a 2D column major array, with probabilites for each t strided + // by (out_dim * stride), where stride is the minibatch size + const int start_prob_col = t * (out_dim * stride); + + // This is the first column and in this case there is nothing left of it + if (tid == 0) { + if (start == 0) { + alpha[start_cur_row] = alpha[start_prev_row] + + log_probs[prob_offset + start_prob_col + blank_label]; + } + else if (start == 1) { + alpha[start_cur_row] = alpha[start_prev_row]; + } + } + + __syncthreads(); + + // Fill in the elements in each row. There is no loop dependence here since our + // input is the row above. We sum either two or three adjacent values from the + // row above depending on whether we have a blank or repeated characters. Finally + // we add the probability corresponding to this label at time t + #pragma unroll + for (int idx = (tid+1); idx < S; idx += blockDim.x) { + + ProbT prev_sum = log_plus_f(alpha[idx + start_prev_row], alpha[(idx-1) + start_prev_row]); + + // Skip two if not on blank and not on repeat. + if ((label[idx] != blank_label) && + (idx != 1) && (label[idx] != label[idx-2])) + prev_sum = log_plus_f(prev_sum, alpha[(idx-2) + start_prev_row]); + + alpha[idx + start_cur_row] = + prev_sum + log_probs[prob_offset + start_prob_col + label[idx]]; + } + + __syncthreads(); + } + + if (tid == 0) { + // Add and return the rightmost two/one element(s) in the last row. + ProbT loglike = ctc_helper::neg_inf(); + + // This is the total increment for s_inc and e_inc through the loop + const int val = 2 * (L-1) + 1 - (((L + repeats) == T) ? 1 : 0); + + start = (val * (L!=0) + start); + end = (val * (L!=0) + end); + + for(int i = start; i < end; ++i) + loglike = log_plus_f(loglike, alpha[i + (T - 1) * S]); + + nll_forward[blockIdx.x] = -loglike; + } +} + +// Computes backward probabilities. This also fills in a T * S matrix +// +// See comments above compute_alphas for more context. +template +__global__ +void compute_betas_and_grad_kernel (const ProbT* log_probs, const int *label_sizes, + const int *utt_length, const int *repeats_in_labels, + const int *labels_with_blanks, ProbT *alphas, + const ProbT* nll_forward, ProbT *nll_backward, + ProbT *grads, int stride, int out_dim, + int S_memoffset, int T_memoffset, int blank_label) { + + ctc_helper::log_plus log_plus_f; + typedef CTASegReduce> SegReduce; + + const int tid = threadIdx.x; + const int L = label_sizes[blockIdx.x]; + const int T = utt_length[blockIdx.x]; + const int S = 2*L + 1; + const int prob_offset = out_dim * blockIdx.x; + const int repeats = repeats_in_labels[blockIdx.x]; + const ProbT log_partition = -nll_forward[blockIdx.x]; + + const int* labels = labels_with_blanks; + const int* label_global = &labels[blockIdx.x * S_memoffset]; + ProbT* alpha = &alphas[blockIdx.x * (S_memoffset * T_memoffset)]; + + const int NV = NT * VT; + + union TempStorage { + ProbT beta[NV]; + int result[NV]; + }; + + __shared__ TempStorage temp_buffer; + + __shared__ int label[NV]; + + // Temporaries needed for segmented reduce + // TODO: see if we can combine the shared memory requirements + __shared__ int keys_shared[NV]; + __shared__ int gather_indices[NV]; + __shared__ ProbT output[NV]; + + ProbT beta_val[VT]; + + if ((L + repeats) > T) + return; + + int start = S > 1 ? (S - 2) : 0; + int end = (L + repeats < T) ? S : S-1; + + // Setup shared memory buffers + #pragma unroll + for (int idx = tid; idx < NV; idx += NT) { + label[idx] = (idx < S) ? label_global[idx] : INT_MAX; + } + + __syncthreads(); + + // int flags; + int uniquelabels; + int seg_start[VT]; + int seg_end[VT]; + + // Sort labels and record indices from which to gather from + { + int key[VT]; + int gather_val[VT]; + + #pragma unroll + for (int i = 0; i < VT; ++i) { + const int idx = tid * VT + i; + gather_val[i] = idx; + key[i] = label[idx]; + } + + __syncthreads(); + + CTAMergesort> + (key, gather_val, keys_shared, gather_indices, S, tid, mgpu::less()); + + __syncthreads(); + + for (int i = 0; i < VT; ++i) { + const int idx = tid * VT + i; + gather_indices[idx] = gather_val[i]; + } + + __syncthreads(); + + SegReduce::preprocessKeys(keys_shared, S, &uniquelabels, seg_start, seg_end, + temp_buffer.result); + __syncthreads(); + } + + // TODO: probably not necessary + __syncthreads(); + + // Load labels back + #pragma unroll + for (int idx = tid; idx < NV; idx += NT) { + temp_buffer.beta[idx] = ctc_helper::neg_inf(); + } + __syncthreads(); + + // Initialize the two rightmost values in the last row (assuming L non-zero) + for(int i = tid; i < (end-start); i += blockDim.x) + temp_buffer.beta[i + start] = + log_probs[prob_offset + (T - 1) * (out_dim * stride) + label[i + start]]; + + __syncthreads(); + + // Load output data in registers through the transpose trick - should really be a function + #pragma unroll + for (int idx = tid; idx < S; idx += NT) { + output[idx] = alpha[idx + (T - 1) * S] + temp_buffer.beta[idx]; + } + + __syncthreads(); + + // Start at the second to last row and backward in time + for(int t = T - 1; t >= 0; --t) { + + // Start offsets into the current and next row + const int start_cur_row = t * S; + + // Starting offset of column that we read from the log probs array + const int start_prob_col = t * (out_dim * stride); + + if (t < T-1) { + + // Filling up one row at at time but going back in time from the last row + // to the first. As in the forward pass, there is no loop dependence and we + // do a variable length filter of maximum filter size of 3 + #pragma unroll + for(int idx = tid, i = 0; idx < (S-1); idx += NT, i++) { + ProbT next_sum = log_plus_f(temp_buffer.beta[idx], temp_buffer.beta[idx+1]); + + // Skip two if not on blank and not on repeat. + if ((label[idx] != blank_label) && + (idx != (S-2)) && (label[idx] != label[idx+2])) + next_sum = log_plus_f(next_sum, temp_buffer.beta[idx+2]); + + beta_val[i] = next_sum + log_probs[prob_offset + start_prob_col + label[idx]]; + } + + __syncthreads(); + + // Initialize values for the rightmost column since there is nothing to the right + // Update input buffer for next iteration + if ((tid == 0) && (end == S)) + temp_buffer.beta[(S-1)] = temp_buffer.beta[(S-1)] + + log_probs[prob_offset + start_prob_col + blank_label]; + + #pragma unroll + for(int idx = tid, i = 0; idx < (S-1); idx += NT, i++) { + temp_buffer.beta[idx] = beta_val[i]; + } + + __syncthreads(); + + // Beta Computation done - add to alpha and update the gradient. Reload + // the gradient back for segmented reduce later on + #pragma unroll + for(int idx = tid; idx < S; idx += NT) { + output[idx] = alpha[idx + start_cur_row] + temp_buffer.beta[idx]; + } + + __syncthreads(); + + } + + __syncthreads(); + + // Compute segmented reduction of output by using label as key + { + // Somewhat faster key value reduce + ProbT accum[VT]; + + for (int idx = tid, j = 0; idx < uniquelabels; idx += blockDim.x, ++j) { + + accum[j] = ctc_helper::neg_inf(); + for (int i = seg_start[j]; i <= seg_end[j]; ++i) { + accum[j] = log_plus_f(accum[j], output[gather_indices[i]]); + } + } + __syncthreads(); + + // Write accumulated value into output since that is not used + for (int idx = tid, j = 0; idx < uniquelabels; idx += blockDim.x, ++j) { + output[idx] = accum[j]; + } + __syncthreads(); + + for (int idx = tid; idx < out_dim; idx += blockDim.x) { + const int grads_offset = prob_offset + start_prob_col + idx; + grads[grads_offset] = exp(log_probs[grads_offset]); + } + + __syncthreads(); + + for (int idx = tid; idx < uniquelabels; idx += blockDim.x) { + const int grads_offset = prob_offset + start_prob_col + keys_shared[idx]; + + ProbT grad = output[idx]; + + if ((grad == 0.0) || (log_probs[grads_offset] == ctc_helper::neg_inf()) || + (grad == ctc_helper::neg_inf())) { + } else { + grads[grads_offset] = + exp(log_probs[grads_offset]) - exp(grad - log_probs[grads_offset] - log_partition); + } + } + + __syncthreads(); + } + + // Output backward log likelihood + if ((t == 0) && (tid == 0)) { + ProbT loglike = ctc_helper::neg_inf(); + + const int val = 2 * (L-1) + 1 - (((L + repeats) == T) ? 1 : 0); + + start = (-val * (L != 0) + start); + end = (-val * (L != 0) + end); + + // Sum and return the leftmost one/two value(s) in first row + for(int i = start; i < end; ++i) + loglike = log_plus_f(loglike, temp_buffer.beta[i]); + + nll_backward[blockIdx.x] = -loglike; + } + + // For some reason this is important + __syncthreads(); + } +} + +template +__global__ void compute_log_probs_kernel(Op f, ProbT* log_probs, + const ProbT* const denom, + int alphabet_size, + int count) { + + int idx = blockDim.x * blockIdx.x + threadIdx.x; + int stride = blockDim.x * gridDim.x; +#pragma unroll + for(int i = 0; i < VT; i++) { + if (idx < count) { + const int column_idx = idx / alphabet_size; + log_probs[idx] = log_probs[idx] - log(denom[column_idx]); + } + idx += stride; + } +} + +template +__global__ void prepare_stable_LSM_kernel(Op f, ProbT* log_probs, + const ProbT* const col_max, + int alphabet_size, + int count) { + + int idx = blockDim.x * blockIdx.x + threadIdx.x; + int stride = blockDim.x * gridDim.x; +#pragma unroll + for(int i = 0; i < VT; i++) { + if (idx < count) { + const int column_idx = idx / alphabet_size; + log_probs[idx] = f(log_probs[idx] - col_max[column_idx]); + } + idx += stride; + } +} diff --git a/src/operator/nn/ctc_include/detail/hostdevice.h b/src/operator/nn/ctc_include/detail/hostdevice.h new file mode 100644 index 000000000000..f7f0425bf26d --- /dev/null +++ b/src/operator/nn/ctc_include/detail/hostdevice.h @@ -0,0 +1,27 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + + +#pragma once + +#ifdef __CUDACC__ + #define HOSTDEVICE __host__ __device__ +#else + #define HOSTDEVICE +#endif diff --git a/src/operator/nn/ctc_loss-inl.h b/src/operator/nn/ctc_loss-inl.h index fed3453ebb4d..b1699e0536e6 100644 --- a/src/operator/nn/ctc_loss-inl.h +++ b/src/operator/nn/ctc_loss-inl.h @@ -198,17 +198,26 @@ struct CTCLossOpParam : public dmlc::Parameter { } }; +// By default, the inputs must include data array and label array +// if use_data_lengths parameter is set, user should also supply +// data_lengths array; if use_label_lengths parameter is set, user +// should also specify label_lengths array. +inline uint32_t CTCLossOpNumInputs(const NodeAttrs& attrs) { + const CTCLossOpParam& param = nnvm::get(attrs.parsed); + return 2U + param.use_data_lengths + param.use_label_lengths; +} + inline bool CTCLossOpShape(const nnvm::NodeAttrs &attrs, std::vector* in_attrs, std::vector* out_attrs) { const CTCLossOpParam& param = nnvm::get(attrs.parsed); - CHECK_EQ(in_attrs->size(), 2U + param.use_data_lengths + param.use_label_lengths); + CHECK_EQ(in_attrs->size(), CTCLossOpNumInputs(attrs)); CHECK_EQ(out_attrs->size(), 2U); const TShape &dshape = (*in_attrs)[ctc_loss::kData]; const TShape &lshape = (*in_attrs)[ctc_loss::kLabel]; - CHECK_EQ(dshape.ndim(), 3U) << "The data array must be of rank 3."; - CHECK_EQ(lshape.ndim(), 2U) << "The labels array must be of rank 2."; + CHECK_EQ(dshape.ndim(), 3U) << "The number of dimensions of data array must be 3."; + CHECK_EQ(lshape.ndim(), 2U) << "The number of dimensions of labels array must be 2."; CHECK_EQ(dshape[1], lshape[0]) << "The batch size for the labels and data arrays must be the same."; @@ -270,10 +279,6 @@ inline bool CTCLossOpStorageType(const nnvm::NodeAttrs& attrs, return dispatched; } -inline int CTCLossOpNumInputs(const NodeAttrs& attrs) { - const CTCLossOpParam& param = nnvm::get(attrs.parsed); - return 2 + param.use_data_lengths + param.use_label_lengths; -} inline std::vector CTCLossOpListInputNames(const NodeAttrs& attrs) { const CTCLossOpParam& param = nnvm::get(attrs.parsed); @@ -298,7 +303,7 @@ void CTCLossOpForward(const nnvm::NodeAttrs& attrs, using namespace mshadow::expr; const CTCLossOpParam& param = nnvm::get(attrs.parsed); - CHECK_EQ(inputs.size(), 2U + param.use_data_lengths + param.use_label_lengths); + CHECK_EQ(inputs.size(), CTCLossOpNumInputs(attrs)); CHECK_EQ(outputs.size(), 2U); CHECK_EQ(req.size(), 2U); diff --git a/src/operator/nn/ctc_loss.cc b/src/operator/nn/ctc_loss.cc index d4bc194b94d0..946ba90b6667 100644 --- a/src/operator/nn/ctc_loss.cc +++ b/src/operator/nn/ctc_loss.cc @@ -22,18 +22,18 @@ * \brief CPU Implementation of CTC Loss op */ #include "./ctc_loss-inl.h" -#include "../contrib/ctc_include/detail/cpu_ctc.h" +#include "./ctc_include/detail/cpu_ctc.h" namespace mshadow { template ctcStatus_t compute_ctc_cost(const Tensor activations, DType *costs, DType *grads, int *labels, int *label_lengths, int *data_lengths, - void *workspace, int train, int blank_label) { + void *workspace, bool isTraining, int blank_label) { int minibatch = static_cast(activations.size(1)); int alphabet_size = static_cast(activations.size(2)); mxnet_warpctc::CpuCTC ctc(alphabet_size, minibatch, workspace, blank_label); - if (train) { + if (isTraining) { return ctc.cost_and_grad(activations.dptr_, grads, costs, labels, label_lengths, data_lengths); } else { diff --git a/src/operator/nn/ctc_loss.cu b/src/operator/nn/ctc_loss.cu index 11d054ffd953..0c0851f037ed 100644 --- a/src/operator/nn/ctc_loss.cu +++ b/src/operator/nn/ctc_loss.cu @@ -24,7 +24,7 @@ */ #include "./ctc_loss-inl.h" -#include "../contrib/ctc_include/detail/gpu_ctc.h" +#include "./ctc_include/detail/gpu_ctc.h" namespace mshadow { From 7a5526217859cedc689854d0cd2bc6412b0e2876 Mon Sep 17 00:00:00 2001 From: Lin Yuan Date: Thu, 27 Sep 2018 13:30:55 -0700 Subject: [PATCH 09/17] temporarily disable lint on 3rd party includes --- src/operator/nn/ctc_loss.cc | 2 +- src/operator/nn/ctc_loss.cu | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/src/operator/nn/ctc_loss.cc b/src/operator/nn/ctc_loss.cc index 946ba90b6667..8db9dfe55afb 100644 --- a/src/operator/nn/ctc_loss.cc +++ b/src/operator/nn/ctc_loss.cc @@ -22,7 +22,7 @@ * \brief CPU Implementation of CTC Loss op */ #include "./ctc_loss-inl.h" -#include "./ctc_include/detail/cpu_ctc.h" +#include "./ctc_include/detail/cpu_ctc.h" // NOLINT(*) namespace mshadow { template diff --git a/src/operator/nn/ctc_loss.cu b/src/operator/nn/ctc_loss.cu index 0c0851f037ed..7144904cadb7 100644 --- a/src/operator/nn/ctc_loss.cu +++ b/src/operator/nn/ctc_loss.cu @@ -24,7 +24,7 @@ */ #include "./ctc_loss-inl.h" -#include "./ctc_include/detail/gpu_ctc.h" +#include "./ctc_include/detail/gpu_ctc.h" // NOLINT(*) namespace mshadow { From 6fa4a13716b9c1942eab146292c85b61b10b9a0a Mon Sep 17 00:00:00 2001 From: Lin Yuan Date: Mon, 1 Oct 2018 14:59:53 -0700 Subject: [PATCH 10/17] move ctc_include to 3rdparty --- {src/operator/nn => 3rdparty}/ctc_include/LICENSE | 0 .../nn => 3rdparty}/ctc_include/contrib/moderngpu/LICENSE | 0 .../contrib/moderngpu/include/device/ctaloadbalance.cuh | 0 .../ctc_include/contrib/moderngpu/include/device/ctamerge.cuh | 0 .../ctc_include/contrib/moderngpu/include/device/ctascan.cuh | 0 .../ctc_include/contrib/moderngpu/include/device/ctasearch.cuh | 0 .../contrib/moderngpu/include/device/ctasegreduce.cuh | 0 .../ctc_include/contrib/moderngpu/include/device/ctasegscan.cuh | 0 .../ctc_include/contrib/moderngpu/include/device/ctasegsort.cuh | 0 .../contrib/moderngpu/include/device/ctasortedsearch.cuh | 0 .../contrib/moderngpu/include/device/devicetypes.cuh | 0 .../ctc_include/contrib/moderngpu/include/device/deviceutil.cuh | 0 .../ctc_include/contrib/moderngpu/include/device/intrinsics.cuh | 0 .../ctc_include/contrib/moderngpu/include/device/loadstore.cuh | 0 .../ctc_include/contrib/moderngpu/include/device/serialsets.cuh | 0 .../contrib/moderngpu/include/device/sortnetwork.cuh | 0 .../ctc_include/contrib/moderngpu/include/mgpudevice.cuh | 0 .../ctc_include/contrib/moderngpu/include/mgpuenums.h | 0 .../ctc_include/contrib/moderngpu/include/util/static.h | 0 {src/operator/nn => 3rdparty}/ctc_include/detail/cpu_ctc.h | 0 {src/operator/nn => 3rdparty}/ctc_include/detail/ctc_helper.h | 0 {src/operator/nn => 3rdparty}/ctc_include/detail/gpu_ctc.h | 0 .../nn => 3rdparty}/ctc_include/detail/gpu_ctc_kernels.h | 0 {src/operator/nn => 3rdparty}/ctc_include/detail/hostdevice.h | 0 Makefile | 2 +- src/operator/nn/ctc_loss.cc | 2 +- src/operator/nn/ctc_loss.cu | 2 +- 27 files changed, 3 insertions(+), 3 deletions(-) rename {src/operator/nn => 3rdparty}/ctc_include/LICENSE (100%) rename {src/operator/nn => 3rdparty}/ctc_include/contrib/moderngpu/LICENSE (100%) rename {src/operator/nn => 3rdparty}/ctc_include/contrib/moderngpu/include/device/ctaloadbalance.cuh (100%) rename {src/operator/nn => 3rdparty}/ctc_include/contrib/moderngpu/include/device/ctamerge.cuh (100%) rename {src/operator/nn => 3rdparty}/ctc_include/contrib/moderngpu/include/device/ctascan.cuh (100%) rename {src/operator/nn => 3rdparty}/ctc_include/contrib/moderngpu/include/device/ctasearch.cuh (100%) rename {src/operator/nn => 3rdparty}/ctc_include/contrib/moderngpu/include/device/ctasegreduce.cuh (100%) rename {src/operator/nn => 3rdparty}/ctc_include/contrib/moderngpu/include/device/ctasegscan.cuh (100%) rename {src/operator/nn => 3rdparty}/ctc_include/contrib/moderngpu/include/device/ctasegsort.cuh (100%) rename {src/operator/nn => 3rdparty}/ctc_include/contrib/moderngpu/include/device/ctasortedsearch.cuh (100%) rename {src/operator/nn => 3rdparty}/ctc_include/contrib/moderngpu/include/device/devicetypes.cuh (100%) rename {src/operator/nn => 3rdparty}/ctc_include/contrib/moderngpu/include/device/deviceutil.cuh (100%) rename {src/operator/nn => 3rdparty}/ctc_include/contrib/moderngpu/include/device/intrinsics.cuh (100%) rename {src/operator/nn => 3rdparty}/ctc_include/contrib/moderngpu/include/device/loadstore.cuh (100%) rename {src/operator/nn => 3rdparty}/ctc_include/contrib/moderngpu/include/device/serialsets.cuh (100%) rename {src/operator/nn => 3rdparty}/ctc_include/contrib/moderngpu/include/device/sortnetwork.cuh (100%) rename {src/operator/nn => 3rdparty}/ctc_include/contrib/moderngpu/include/mgpudevice.cuh (100%) rename {src/operator/nn => 3rdparty}/ctc_include/contrib/moderngpu/include/mgpuenums.h (100%) rename {src/operator/nn => 3rdparty}/ctc_include/contrib/moderngpu/include/util/static.h (100%) rename {src/operator/nn => 3rdparty}/ctc_include/detail/cpu_ctc.h (100%) rename {src/operator/nn => 3rdparty}/ctc_include/detail/ctc_helper.h (100%) rename {src/operator/nn => 3rdparty}/ctc_include/detail/gpu_ctc.h (100%) rename {src/operator/nn => 3rdparty}/ctc_include/detail/gpu_ctc_kernels.h (100%) rename {src/operator/nn => 3rdparty}/ctc_include/detail/hostdevice.h (100%) diff --git a/src/operator/nn/ctc_include/LICENSE b/3rdparty/ctc_include/LICENSE similarity index 100% rename from src/operator/nn/ctc_include/LICENSE rename to 3rdparty/ctc_include/LICENSE diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/LICENSE b/3rdparty/ctc_include/contrib/moderngpu/LICENSE similarity index 100% rename from src/operator/nn/ctc_include/contrib/moderngpu/LICENSE rename to 3rdparty/ctc_include/contrib/moderngpu/LICENSE diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctaloadbalance.cuh b/3rdparty/ctc_include/contrib/moderngpu/include/device/ctaloadbalance.cuh similarity index 100% rename from src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctaloadbalance.cuh rename to 3rdparty/ctc_include/contrib/moderngpu/include/device/ctaloadbalance.cuh diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctamerge.cuh b/3rdparty/ctc_include/contrib/moderngpu/include/device/ctamerge.cuh similarity index 100% rename from src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctamerge.cuh rename to 3rdparty/ctc_include/contrib/moderngpu/include/device/ctamerge.cuh diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctascan.cuh b/3rdparty/ctc_include/contrib/moderngpu/include/device/ctascan.cuh similarity index 100% rename from src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctascan.cuh rename to 3rdparty/ctc_include/contrib/moderngpu/include/device/ctascan.cuh diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasearch.cuh b/3rdparty/ctc_include/contrib/moderngpu/include/device/ctasearch.cuh similarity index 100% rename from src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasearch.cuh rename to 3rdparty/ctc_include/contrib/moderngpu/include/device/ctasearch.cuh diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasegreduce.cuh b/3rdparty/ctc_include/contrib/moderngpu/include/device/ctasegreduce.cuh similarity index 100% rename from src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasegreduce.cuh rename to 3rdparty/ctc_include/contrib/moderngpu/include/device/ctasegreduce.cuh diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasegscan.cuh b/3rdparty/ctc_include/contrib/moderngpu/include/device/ctasegscan.cuh similarity index 100% rename from src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasegscan.cuh rename to 3rdparty/ctc_include/contrib/moderngpu/include/device/ctasegscan.cuh diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasegsort.cuh b/3rdparty/ctc_include/contrib/moderngpu/include/device/ctasegsort.cuh similarity index 100% rename from src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasegsort.cuh rename to 3rdparty/ctc_include/contrib/moderngpu/include/device/ctasegsort.cuh diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasortedsearch.cuh b/3rdparty/ctc_include/contrib/moderngpu/include/device/ctasortedsearch.cuh similarity index 100% rename from src/operator/nn/ctc_include/contrib/moderngpu/include/device/ctasortedsearch.cuh rename to 3rdparty/ctc_include/contrib/moderngpu/include/device/ctasortedsearch.cuh diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/devicetypes.cuh b/3rdparty/ctc_include/contrib/moderngpu/include/device/devicetypes.cuh similarity index 100% rename from src/operator/nn/ctc_include/contrib/moderngpu/include/device/devicetypes.cuh rename to 3rdparty/ctc_include/contrib/moderngpu/include/device/devicetypes.cuh diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/deviceutil.cuh b/3rdparty/ctc_include/contrib/moderngpu/include/device/deviceutil.cuh similarity index 100% rename from src/operator/nn/ctc_include/contrib/moderngpu/include/device/deviceutil.cuh rename to 3rdparty/ctc_include/contrib/moderngpu/include/device/deviceutil.cuh diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/intrinsics.cuh b/3rdparty/ctc_include/contrib/moderngpu/include/device/intrinsics.cuh similarity index 100% rename from src/operator/nn/ctc_include/contrib/moderngpu/include/device/intrinsics.cuh rename to 3rdparty/ctc_include/contrib/moderngpu/include/device/intrinsics.cuh diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/loadstore.cuh b/3rdparty/ctc_include/contrib/moderngpu/include/device/loadstore.cuh similarity index 100% rename from src/operator/nn/ctc_include/contrib/moderngpu/include/device/loadstore.cuh rename to 3rdparty/ctc_include/contrib/moderngpu/include/device/loadstore.cuh diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/serialsets.cuh b/3rdparty/ctc_include/contrib/moderngpu/include/device/serialsets.cuh similarity index 100% rename from src/operator/nn/ctc_include/contrib/moderngpu/include/device/serialsets.cuh rename to 3rdparty/ctc_include/contrib/moderngpu/include/device/serialsets.cuh diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/device/sortnetwork.cuh b/3rdparty/ctc_include/contrib/moderngpu/include/device/sortnetwork.cuh similarity index 100% rename from src/operator/nn/ctc_include/contrib/moderngpu/include/device/sortnetwork.cuh rename to 3rdparty/ctc_include/contrib/moderngpu/include/device/sortnetwork.cuh diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/mgpudevice.cuh b/3rdparty/ctc_include/contrib/moderngpu/include/mgpudevice.cuh similarity index 100% rename from src/operator/nn/ctc_include/contrib/moderngpu/include/mgpudevice.cuh rename to 3rdparty/ctc_include/contrib/moderngpu/include/mgpudevice.cuh diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/mgpuenums.h b/3rdparty/ctc_include/contrib/moderngpu/include/mgpuenums.h similarity index 100% rename from src/operator/nn/ctc_include/contrib/moderngpu/include/mgpuenums.h rename to 3rdparty/ctc_include/contrib/moderngpu/include/mgpuenums.h diff --git a/src/operator/nn/ctc_include/contrib/moderngpu/include/util/static.h b/3rdparty/ctc_include/contrib/moderngpu/include/util/static.h similarity index 100% rename from src/operator/nn/ctc_include/contrib/moderngpu/include/util/static.h rename to 3rdparty/ctc_include/contrib/moderngpu/include/util/static.h diff --git a/src/operator/nn/ctc_include/detail/cpu_ctc.h b/3rdparty/ctc_include/detail/cpu_ctc.h similarity index 100% rename from src/operator/nn/ctc_include/detail/cpu_ctc.h rename to 3rdparty/ctc_include/detail/cpu_ctc.h diff --git a/src/operator/nn/ctc_include/detail/ctc_helper.h b/3rdparty/ctc_include/detail/ctc_helper.h similarity index 100% rename from src/operator/nn/ctc_include/detail/ctc_helper.h rename to 3rdparty/ctc_include/detail/ctc_helper.h diff --git a/src/operator/nn/ctc_include/detail/gpu_ctc.h b/3rdparty/ctc_include/detail/gpu_ctc.h similarity index 100% rename from src/operator/nn/ctc_include/detail/gpu_ctc.h rename to 3rdparty/ctc_include/detail/gpu_ctc.h diff --git a/src/operator/nn/ctc_include/detail/gpu_ctc_kernels.h b/3rdparty/ctc_include/detail/gpu_ctc_kernels.h similarity index 100% rename from src/operator/nn/ctc_include/detail/gpu_ctc_kernels.h rename to 3rdparty/ctc_include/detail/gpu_ctc_kernels.h diff --git a/src/operator/nn/ctc_include/detail/hostdevice.h b/3rdparty/ctc_include/detail/hostdevice.h similarity index 100% rename from src/operator/nn/ctc_include/detail/hostdevice.h rename to 3rdparty/ctc_include/detail/hostdevice.h diff --git a/Makefile b/Makefile index a37694019e7a..69b25daa9d1c 100644 --- a/Makefile +++ b/Makefile @@ -568,7 +568,7 @@ lint: cpplint rcpplint jnilint pylint cpplint: 3rdparty/dmlc-core/scripts/lint.py mxnet cpp include src plugin cpp-package tests \ - --exclude_path src/operator/contrib/ctc_include src/operator/nn/ctc_include + --exclude_path src/operator/contrib/ctc_include pylint: pylint --rcfile=$(ROOTDIR)/ci/other/pylintrc --ignore-patterns=".*\.so$$,.*\.dll$$,.*\.dylib$$" python/mxnet tools/caffe_converter/*.py diff --git a/src/operator/nn/ctc_loss.cc b/src/operator/nn/ctc_loss.cc index 8db9dfe55afb..d3996a27677b 100644 --- a/src/operator/nn/ctc_loss.cc +++ b/src/operator/nn/ctc_loss.cc @@ -22,7 +22,7 @@ * \brief CPU Implementation of CTC Loss op */ #include "./ctc_loss-inl.h" -#include "./ctc_include/detail/cpu_ctc.h" // NOLINT(*) +#include "../../../3rdparty/ctc_include/detail/cpu_ctc.h" namespace mshadow { template diff --git a/src/operator/nn/ctc_loss.cu b/src/operator/nn/ctc_loss.cu index 7144904cadb7..a83337b6ee2f 100644 --- a/src/operator/nn/ctc_loss.cu +++ b/src/operator/nn/ctc_loss.cu @@ -24,7 +24,7 @@ */ #include "./ctc_loss-inl.h" -#include "./ctc_include/detail/gpu_ctc.h" // NOLINT(*) +#include "../../../3rdparty/ctc_include/detail/gpu_ctc.h" namespace mshadow { From c13197a7e3505a98920ed424ee049e88f39da350 Mon Sep 17 00:00:00 2001 From: Lin Yuan Date: Tue, 2 Oct 2018 09:08:21 -0700 Subject: [PATCH 11/17] remove contrib ctc_loss operator --- python/mxnet/gluon/loss.py | 8 +- src/operator/contrib/ctc_include/LICENSE | 205 ------ .../ctc_include/contrib/moderngpu/LICENSE | 26 - .../include/device/ctaloadbalance.cuh | 136 ---- .../moderngpu/include/device/ctamerge.cuh | 333 --------- .../moderngpu/include/device/ctascan.cuh | 308 -------- .../moderngpu/include/device/ctasearch.cuh | 207 ------ .../moderngpu/include/device/ctasegreduce.cuh | 238 ------- .../moderngpu/include/device/ctasegscan.cuh | 137 ---- .../moderngpu/include/device/ctasegsort.cuh | 443 ------------ .../include/device/ctasortedsearch.cuh | 208 ------ .../moderngpu/include/device/devicetypes.cuh | 363 ---------- .../moderngpu/include/device/deviceutil.cuh | 143 ---- .../moderngpu/include/device/intrinsics.cuh | 421 ----------- .../moderngpu/include/device/loadstore.cuh | 674 ------------------ .../moderngpu/include/device/serialsets.cuh | 235 ------ .../moderngpu/include/device/sortnetwork.cuh | 168 ----- .../contrib/moderngpu/include/mgpudevice.cuh | 289 -------- .../contrib/moderngpu/include/mgpuenums.h | 70 -- .../contrib/moderngpu/include/util/static.h | 183 ----- .../contrib/ctc_include/detail/cpu_ctc.h | 509 ------------- .../contrib/ctc_include/detail/ctc_helper.h | 93 --- .../contrib/ctc_include/detail/gpu_ctc.h | 505 ------------- .../ctc_include/detail/gpu_ctc_kernels.h | 507 ------------- .../contrib/ctc_include/detail/hostdevice.h | 27 - src/operator/contrib/ctc_loss-inl.h | 591 --------------- src/operator/contrib/ctc_loss.cc | 130 ---- src/operator/contrib/ctc_loss.cu | 61 -- src/operator/nn/ctc_loss.cc | 3 +- src/operator/nn/ctc_loss.cu | 3 +- tests/python/unittest/test_loss.py | 7 +- 31 files changed, 13 insertions(+), 7218 deletions(-) delete mode 100644 src/operator/contrib/ctc_include/LICENSE delete mode 100644 src/operator/contrib/ctc_include/contrib/moderngpu/LICENSE delete mode 100644 src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctaloadbalance.cuh delete mode 100644 src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctamerge.cuh delete mode 100644 src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctascan.cuh delete mode 100644 src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasearch.cuh delete mode 100644 src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasegreduce.cuh delete mode 100644 src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasegscan.cuh delete mode 100644 src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasegsort.cuh delete mode 100644 src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasortedsearch.cuh delete mode 100644 src/operator/contrib/ctc_include/contrib/moderngpu/include/device/devicetypes.cuh delete mode 100644 src/operator/contrib/ctc_include/contrib/moderngpu/include/device/deviceutil.cuh delete mode 100644 src/operator/contrib/ctc_include/contrib/moderngpu/include/device/intrinsics.cuh delete mode 100644 src/operator/contrib/ctc_include/contrib/moderngpu/include/device/loadstore.cuh delete mode 100644 src/operator/contrib/ctc_include/contrib/moderngpu/include/device/serialsets.cuh delete mode 100644 src/operator/contrib/ctc_include/contrib/moderngpu/include/device/sortnetwork.cuh delete mode 100644 src/operator/contrib/ctc_include/contrib/moderngpu/include/mgpudevice.cuh delete mode 100644 src/operator/contrib/ctc_include/contrib/moderngpu/include/mgpuenums.h delete mode 100644 src/operator/contrib/ctc_include/contrib/moderngpu/include/util/static.h delete mode 100644 src/operator/contrib/ctc_include/detail/cpu_ctc.h delete mode 100644 src/operator/contrib/ctc_include/detail/ctc_helper.h delete mode 100644 src/operator/contrib/ctc_include/detail/gpu_ctc.h delete mode 100644 src/operator/contrib/ctc_include/detail/gpu_ctc_kernels.h delete mode 100644 src/operator/contrib/ctc_include/detail/hostdevice.h delete mode 100644 src/operator/contrib/ctc_loss-inl.h delete mode 100644 src/operator/contrib/ctc_loss.cc delete mode 100644 src/operator/contrib/ctc_loss.cu diff --git a/python/mxnet/gluon/loss.py b/python/mxnet/gluon/loss.py index 2be43981a64c..7e4d34577635 100644 --- a/python/mxnet/gluon/loss.py +++ b/python/mxnet/gluon/loss.py @@ -468,10 +468,10 @@ def hybrid_forward(self, F, pred, label, pred = F.swapaxes(pred, 0, 1) if self._batch_axis == 1: label = F.swapaxes(label, 0, 1) - loss = F.contrib.CTCLoss(pred, label, pred_lengths, label_lengths, - use_data_lengths=pred_lengths is not None, - use_label_lengths=label_lengths is not None, - blank_label='last') + loss = F.CTCLoss(pred, label, pred_lengths, label_lengths, + use_data_lengths=pred_lengths is not None, + use_label_lengths=label_lengths is not None, + blank_label='last') return _apply_weighting(F, loss, self._weight, sample_weight) diff --git a/src/operator/contrib/ctc_include/LICENSE b/src/operator/contrib/ctc_include/LICENSE deleted file mode 100644 index 4946875860dd..000000000000 --- a/src/operator/contrib/ctc_include/LICENSE +++ /dev/null @@ -1,205 +0,0 @@ - Apache License - Version 2.0, January 2004 - http://www.apache.org/licenses/ - - TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION - - 1. Definitions. - - "License" shall mean the terms and conditions for use, reproduction, - and distribution as defined by Sections 1 through 9 of this document. - - "Licensor" shall mean the copyright owner or entity authorized by - the copyright owner that is granting the License. - - "Legal Entity" shall mean the union of the acting entity and all - other entities that control, are controlled by, or are under common - control with that entity. For the purposes of this definition, - "control" means (i) the power, direct or indirect, to cause the - direction or management of such entity, whether by contract or - otherwise, or (ii) ownership of fifty percent (50%) or more of the - outstanding shares, or (iii) beneficial ownership of such entity. - - "You" (or "Your") shall mean an individual or Legal Entity - exercising permissions granted by this License. - - "Source" form shall mean the preferred form for making modifications, - including but not limited to software source code, documentation - source, and configuration files. - - "Object" form shall mean any form resulting from mechanical - transformation or translation of a Source form, including but - not limited to compiled object code, generated documentation, - and conversions to other media types. - - "Work" shall mean the work of authorship, whether in Source or - Object form, made available under the License, as indicated by a - copyright notice that is included in or attached to the work - (an example is provided in the Appendix below). - - "Derivative Works" shall mean any work, whether in Source or Object - form, that is based on (or derived from) the Work and for which the - editorial revisions, annotations, elaborations, or other modifications - represent, as a whole, an original work of authorship. For the purposes - of this License, Derivative Works shall not include works that remain - separable from, or merely link (or bind by name) to the interfaces of, - the Work and Derivative Works thereof. - - "Contribution" shall mean any work of authorship, including - the original version of the Work and any modifications or additions - to that Work or Derivative Works thereof, that is intentionally - submitted to Licensor for inclusion in the Work by the copyright owner - or by an individual or Legal Entity authorized to submit on behalf of - the copyright owner. For the purposes of this definition, "submitted" - means any form of electronic, verbal, or written communication sent - to the Licensor or its representatives, including but not limited to - communication on electronic mailing lists, source code control systems, - and issue tracking systems that are managed by, or on behalf of, the - Licensor for the purpose of discussing and improving the Work, but - excluding communication that is conspicuously marked or otherwise - designated in writing by the copyright owner as "Not a Contribution." - - "Contributor" shall mean Licensor and any individual or Legal Entity - on behalf of whom a Contribution has been received by Licensor and - subsequently incorporated within the Work. - - 2. Grant of Copyright License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - copyright license to reproduce, prepare Derivative Works of, - publicly display, publicly perform, sublicense, and distribute the - Work and such Derivative Works in Source or Object form. - - 3. Grant of Patent License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - (except as stated in this section) patent license to make, have made, - use, offer to sell, sell, import, and otherwise transfer the Work, - where such license applies only to those patent claims licensable - by such Contributor that are necessarily infringed by their - Contribution(s) alone or by combination of their Contribution(s) - with the Work to which such Contribution(s) was submitted. If You - institute patent litigation against any entity (including a - cross-claim or counterclaim in a lawsuit) alleging that the Work - or a Contribution incorporated within the Work constitutes direct - or contributory patent infringement, then any patent licenses - granted to You under this License for that Work shall terminate - as of the date such litigation is filed. - - 4. Redistribution. You may reproduce and distribute copies of the - Work or Derivative Works thereof in any medium, with or without - modifications, and in Source or Object form, provided that You - meet the following conditions: - - (a) You must give any other recipients of the Work or - Derivative Works a copy of this License; and - - (b) You must cause any modified files to carry prominent notices - stating that You changed the files; and - - (c) You must retain, in the Source form of any Derivative Works - that You distribute, all copyright, patent, trademark, and - attribution notices from the Source form of the Work, - excluding those notices that do not pertain to any part of - the Derivative Works; and - - (d) If the Work includes a "NOTICE" text file as part of its - distribution, then any Derivative Works that You distribute must - include a readable copy of the attribution notices contained - within such NOTICE file, excluding those notices that do not - pertain to any part of the Derivative Works, in at least one - of the following places: within a NOTICE text file distributed - as part of the Derivative Works; within the Source form or - documentation, if provided along with the Derivative Works; or, - within a display generated by the Derivative Works, if and - wherever such third-party notices normally appear. The contents - of the NOTICE file are for informational purposes only and - do not modify the License. You may add Your own attribution - notices within Derivative Works that You distribute, alongside - or as an addendum to the NOTICE text from the Work, provided - that such additional attribution notices cannot be construed - as modifying the License. - - You may add Your own copyright statement to Your modifications and - may provide additional or different license terms and conditions - for use, reproduction, or distribution of Your modifications, or - for any such Derivative Works as a whole, provided Your use, - reproduction, and distribution of the Work otherwise complies with - the conditions stated in this License. - - 5. Submission of Contributions. Unless You explicitly state otherwise, - any Contribution intentionally submitted for inclusion in the Work - by You to the Licensor shall be under the terms and conditions of - this License, without any additional terms or conditions. - Notwithstanding the above, nothing herein shall supersede or modify - the terms of any separate license agreement you may have executed - with Licensor regarding such Contributions. - - 6. Trademarks. This License does not grant permission to use the trade - names, trademarks, service marks, or product names of the Licensor, - except as required for reasonable and customary use in describing the - origin of the Work and reproducing the content of the NOTICE file. - - 7. Disclaimer of Warranty. Unless required by applicable law or - agreed to in writing, Licensor provides the Work (and each - Contributor provides its Contributions) on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or - implied, including, without limitation, any warranties or conditions - of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A - PARTICULAR PURPOSE. You are solely responsible for determining the - appropriateness of using or redistributing the Work and assume any - risks associated with Your exercise of permissions under this License. - - 8. Limitation of Liability. In no event and under no legal theory, - whether in tort (including negligence), contract, or otherwise, - unless required by applicable law (such as deliberate and grossly - negligent acts) or agreed to in writing, shall any Contributor be - liable to You for damages, including any direct, indirect, special, - incidental, or consequential damages of any character arising as a - result of this License or out of the use or inability to use the - Work (including but not limited to damages for loss of goodwill, - work stoppage, computer failure or malfunction, or any and all - other commercial damages or losses), even if such Contributor - has been advised of the possibility of such damages. - - 9. Accepting Warranty or Additional Liability. While redistributing - the Work or Derivative Works thereof, You may choose to offer, - and charge a fee for, acceptance of support, warranty, indemnity, - or other liability obligations and/or rights consistent with this - License. However, in accepting such obligations, You may act only - on Your own behalf and on Your sole responsibility, not on behalf - of any other Contributor, and only if You agree to indemnify, - defend, and hold each Contributor harmless for any liability - incurred by, or claims asserted against, such Contributor by reason - of your accepting any such warranty or additional liability. - - END OF TERMS AND CONDITIONS - - APPENDIX: How to apply the Apache License to your work. - - To apply the Apache License to your work, attach the following - boilerplate notice, with the fields enclosed by brackets "[]" - replaced with your own identifying information. (Don't include - the brackets!) The text should be enclosed in the appropriate - comment syntax for the file format. We also recommend that a - file or class name and description of purpose be included on the - same "printed page" as the copyright notice for easier - identification within third-party archives. - - Copyright [yyyy] [name of copyright owner] - - Licensed under the Apache License, Version 2.0 (the "License"); - you may not use this file except in compliance with the License. - You may obtain a copy of the License at - - http://www.apache.org/licenses/LICENSE-2.0 - - Unless required by applicable law or agreed to in writing, software - distributed under the License is distributed on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - See the License for the specific language governing permissions and - limitations under the License. - - ---- - - Copyright 2015-2016, Baidu USA LLC. \ No newline at end of file diff --git a/src/operator/contrib/ctc_include/contrib/moderngpu/LICENSE b/src/operator/contrib/ctc_include/contrib/moderngpu/LICENSE deleted file mode 100644 index 98128870cf41..000000000000 --- a/src/operator/contrib/ctc_include/contrib/moderngpu/LICENSE +++ /dev/null @@ -1,26 +0,0 @@ -/****************************************************************************** -* Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. -* -* Redistribution and use in source and binary forms, with or without -* modification, are permitted provided that the following conditions are met: -* * Redistributions of source code must retain the above copyright -* notice, this list of conditions and the following disclaimer. -* * Redistributions in binary form must reproduce the above copyright -* notice, this list of conditions and the following disclaimer in the -* documentation and/or other materials provided with the distribution. -* * Neither the name of the NVIDIA CORPORATION nor the -* names of its contributors may be used to endorse or promote products -* derived from this software without specific prior written permission. -* -* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" -* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE -* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE -* ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY -* DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES -* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; -* LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND -* ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT -* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS -* SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. -* -******************************************************************************/ diff --git a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctaloadbalance.cuh b/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctaloadbalance.cuh deleted file mode 100644 index 69f6d96d1fa4..000000000000 --- a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctaloadbalance.cuh +++ /dev/null @@ -1,136 +0,0 @@ -/****************************************************************************** - * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions are met: - * * Redistributions of source code must retain the above copyright - * notice, this list of conditions and the following disclaimer. - * * Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * * Neither the name of the NVIDIA CORPORATION nor the - * names of its contributors may be used to endorse or promote products - * derived from this software without specific prior written permission. - * - * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" - * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE - * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE - * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY - * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES - * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; - * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND - * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS - * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - * - ******************************************************************************/ - -/****************************************************************************** - * - * Code and text by Sean Baxter, NVIDIA Research - * See http://nvlabs.github.io/moderngpu for repository and documentation. - * - ******************************************************************************/ - -#pragma once - -#include "ctasearch.cuh" -#include "loadstore.cuh" - -namespace mgpu { - -//////////////////////////////////////////////////////////////////////////////// -// DeviceLoadBalancingSearch -// Upper Bound search from A (needles) into B (haystack). The A values are -// natural numbers from aBegin to aEnd. bFirst is the index of the B value at -// bBegin in shared memory. - -template -MGPU_DEVICE void DeviceSerialLoadBalanceSearch(const int* b_shared, int aBegin, - int aEnd, int bFirst, int bBegin, int bEnd, int* a_shared) { - - int bKey = b_shared[bBegin]; - - #pragma unroll - for(int i = 0; i < VT; ++i) { - bool p; - if(RangeCheck) - p = (aBegin < aEnd) && ((bBegin >= bEnd) || (aBegin < bKey)); - else - p = aBegin < bKey; - - if(p) - // Advance A (the needle). - a_shared[aBegin++] = bFirst + bBegin; - else - // Advance B (the haystack). - bKey = b_shared[++bBegin]; - } -} - -//////////////////////////////////////////////////////////////////////////////// -// CTALoadBalance -// Computes upper_bound(counting_iterator(first), b_global) - 1. - -// Unlike most other CTA* functions, CTALoadBalance loads from global memory. -// This returns the loaded B elements at the beginning or end of shared memory -// depending on the aFirst argument. - -// CTALoadBalance requires NT * VT + 2 slots of shared memory. -template -MGPU_DEVICE int4 CTALoadBalance(int destCount, InputIt b_global, - int sourceCount, int block, int tid, const int* mp_global, - int* indices_shared, bool loadPrecedingB) { - - int4 range = ComputeMergeRange(destCount, sourceCount, block, 0, NT * VT, - mp_global); - - int a0 = range.x; - int a1 = range.y; - int b0 = range.z; - int b1 = range.w; - if(!b0) loadPrecedingB = false; - - // Load one trailing term from B. If we're already at the end, fill the - // end of the buffer with destCount. - int aCount = a1 - a0; - int bCount = b1 - b0; - int extended = b1 < sourceCount; - int loadCount = bCount + extended; - int fillCount = NT * VT + 1 - loadCount - aCount; - - int* a_shared = indices_shared; - int* b_shared = indices_shared + aCount + (int)loadPrecedingB; - - // Load the B values. -// DeviceMemToMemLoop(bCount + extended + (int)loadPrecedingB, -// b_global + b0 - (int)loadPrecedingB, tid, -// b_shared - (int)loadPrecedingB); - - for(int i = tid - (int)loadPrecedingB; i < bCount + extended; i += NT) - b_shared[i] = b_global[b0 + i]; - - // Fill the end of the array with destCount. - for(int i = tid + extended; i < fillCount; i += NT) - b_shared[bCount + i] = destCount; - __syncthreads(); - - // Run a merge path to find the start of the serial merge for each thread. - int diag = VT * tid; - int mp = MergePath(mgpu::counting_iterator(a0), - aCount, b_shared, bCount, diag, mgpu::less()); - - int a0tid = a0 + mp; - int b0tid = diag - mp; - - // Subtract 1 from b0 because we want to return upper_bound - 1. - DeviceSerialLoadBalanceSearch(b_shared, a0tid, a1, b0 - 1, - b0tid, bCount, a_shared - a0); - __syncthreads(); - - b0 -= (int)loadPrecedingB; - return make_int4(a0, a1, b0, b1); -} - - -} // namespace mgpu diff --git a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctamerge.cuh b/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctamerge.cuh deleted file mode 100644 index bb702c460455..000000000000 --- a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctamerge.cuh +++ /dev/null @@ -1,333 +0,0 @@ -/****************************************************************************** - * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions are met: - * * Redistributions of source code must retain the above copyright - * notice, this list of conditions and the following disclaimer. - * * Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * * Neither the name of the NVIDIA CORPORATION nor the - * names of its contributors may be used to endorse or promote products - * derived from this software without specific prior written permission. - * - * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" - * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE - * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE - * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY - * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES - * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; - * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND - * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS - * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - * - ******************************************************************************/ - -/****************************************************************************** - * - * Code and text by Sean Baxter, NVIDIA Research - * See http://nvlabs.github.io/moderngpu for repository and documentation. - * - ******************************************************************************/ - -#pragma once - -#include "ctasearch.cuh" -#include "loadstore.cuh" -#include "sortnetwork.cuh" - -namespace mgpu { - -//////////////////////////////////////////////////////////////////////////////// -// SerialMerge - -template -MGPU_DEVICE void SerialMerge(const T* keys_shared, int aBegin, int aEnd, - int bBegin, int bEnd, T* results, int* indices, Comp comp) { - - T aKey = keys_shared[aBegin]; - T bKey = keys_shared[bBegin]; - - #pragma unroll - for(int i = 0; i < VT; ++i) { - bool p; - if(RangeCheck) - p = (bBegin >= bEnd) || ((aBegin < aEnd) && !comp(bKey, aKey)); - else - p = !comp(bKey, aKey); - - results[i] = p ? aKey : bKey; - indices[i] = p ? aBegin : bBegin - !RangeCheck; - - if(p) aKey = keys_shared[++aBegin]; - else bKey = keys_shared[++bBegin]; - } - __syncthreads(); -} - -//////////////////////////////////////////////////////////////////////////////// -// FindMergeFrame and FindMergesortInterval help mergesort (both CTA and global -// merge pass levels) locate lists within the single source array. - -// Returns (offset of a, offset of b, length of list). -MGPU_HOST_DEVICE int3 FindMergesortFrame(int coop, int block, int nv) { - // coop is the number of CTAs or threads cooperating to merge two lists into - // one. We round block down to the first CTA's ID that is working on this - // merge. - int start = ~(coop - 1) & block; - int size = nv * (coop>> 1); - return make_int3(nv * start, nv * start + size, size); -} - -// Returns (a0, a1, b0, b1) into mergesort input lists between mp0 and mp1. -MGPU_HOST_DEVICE int4 FindMergesortInterval(int3 frame, int coop, int block, - int nv, int count, int mp0, int mp1) { - - // Locate diag from the start of the A sublist. - int diag = nv * block - frame.x; - int a0 = frame.x + mp0; - int a1 = min(count, frame.x + mp1); - int b0 = min(count, frame.y + diag - mp0); - int b1 = min(count, frame.y + diag + nv - mp1); - - // The end partition of the last block for each merge operation is computed - // and stored as the begin partition for the subsequent merge. i.e. it is - // the same partition but in the wrong coordinate system, so its 0 when it - // should be listSize. Correct that by checking if this is the last block - // in this merge operation. - if(coop - 1 == ((coop - 1) & block)) { - a1 = min(count, frame.x + frame.z); - b1 = min(count, frame.y + frame.z); - } - return make_int4(a0, a1, b0, b1); -} - -//////////////////////////////////////////////////////////////////////////////// -// ComputeMergeRange - -MGPU_HOST_DEVICE int4 ComputeMergeRange(int aCount, int bCount, int block, - int coop, int NV, const int* mp_global) { - - // Load the merge paths computed by the partitioning kernel. - int mp0 = mp_global[block]; - int mp1 = mp_global[block + 1]; - int gid = NV * block; - - // Compute the ranges of the sources in global memory. - int4 range; - if(coop) { - int3 frame = FindMergesortFrame(coop, block, NV); - range = FindMergesortInterval(frame, coop, block, NV, aCount, mp0, - mp1); - } else { - range.x = mp0; // a0 - range.y = mp1; // a1 - range.z = gid - range.x; // b0 - range.w = min(aCount + bCount, gid + NV) - range.y; // b1 - } - return range; -} - -//////////////////////////////////////////////////////////////////////////////// -// CTA mergesort support - -template -MGPU_DEVICE void CTABlocksortPass(T* keys_shared, int tid, int count, - int coop, T* keys, int* indices, Comp comp) { - - int list = ~(coop - 1) & tid; - int diag = min(count, VT * ((coop - 1) & tid)); - int start = VT * list; - int a0 = min(count, start); - int b0 = min(count, start + VT * (coop / 2)); - int b1 = min(count, start + VT * coop); - - int p = MergePath(keys_shared + a0, b0 - a0, - keys_shared + b0, b1 - b0, diag, comp); - - SerialMerge(keys_shared, a0 + p, b0, b0 + diag - p, b1, keys, - indices, comp); -} - -template -MGPU_DEVICE void CTABlocksortLoop(ValType threadValues[VT], - KeyType* keys_shared, ValType* values_shared, int tid, int count, - Comp comp) { - - #pragma unroll - for(int coop = 2; coop <= NT; coop *= 2) { - int indices[VT]; - KeyType keys[VT]; - CTABlocksortPass(keys_shared, tid, count, coop, keys, - indices, comp); - - if(HasValues) { - // Exchange the values through shared memory. - DeviceThreadToShared(threadValues, tid, values_shared); - DeviceGather(NT * VT, values_shared, indices, tid, - threadValues); - } - - // Store results in shared memory in sorted order. - DeviceThreadToShared(keys, tid, keys_shared); - } -} - -//////////////////////////////////////////////////////////////////////////////// -// CTAMergesort -// Caller provides the keys in shared memory. This functions sorts the first -// count elements. - -template -MGPU_DEVICE void CTAMergesort(KeyType threadKeys[VT], ValType threadValues[VT], - KeyType* keys_shared, ValType* values_shared, int count, int tid, - Comp comp) { - - // Stable sort the keys in the thread. - if(VT * tid < count) { - if(Stable) - OddEvenTransposeSort(threadKeys, threadValues, comp); - else - OddEvenMergesort(threadKeys, threadValues, comp); - } - - // Store the locally sorted keys into shared memory. - DeviceThreadToShared(threadKeys, tid, keys_shared); - - // Recursively merge lists until the entire CTA is sorted. - CTABlocksortLoop(threadValues, keys_shared, - values_shared, tid, count, comp); -} - -template -MGPU_DEVICE void CTAMergesortKeys(KeyType threadKeys[VT], - KeyType* keys_shared, int count, int tid, Comp comp) { - - int valuesTemp[VT]; - CTAMergesort(threadKeys, valuesTemp, keys_shared, - (int*)keys_shared, count, tid, comp); -} - -template -MGPU_DEVICE void CTAMergesortPairs(KeyType threadKeys[VT], - ValType threadValues[VT], KeyType* keys_shared, ValType* values_shared, - int count, int tid, Comp comp) { - - CTAMergesort(threadKeys, threadValues, keys_shared, - values_shared, count, tid, comp); -} - -//////////////////////////////////////////////////////////////////////////////// -// DeviceMergeKeysIndices - -template -MGPU_DEVICE void DeviceMergeKeysIndices(It1 a_global, int aCount, It2 b_global, - int bCount, int4 range, int tid, T* keys_shared, T* results, int* indices, - Comp comp) { - - int a0 = range.x; - int a1 = range.y; - int b0 = range.z; - int b1 = range.w; - - if(LoadExtended) { - bool extended = (a1 < aCount) && (b1 < bCount); - aCount = a1 - a0; - bCount = b1 - b0; - int aCount2 = aCount + (int)extended; - int bCount2 = bCount + (int)extended; - - // Load one element past the end of each input to avoid having to use - // range checking in the merge loop. - DeviceLoad2ToShared(a_global + a0, aCount2, - b_global + b0, bCount2, tid, keys_shared); - - // Run a Merge Path search for each thread's starting point. - int diag = VT * tid; - int mp = MergePath(keys_shared, aCount, - keys_shared + aCount2, bCount, diag, comp); - - // Compute the ranges of the sources in shared memory. - int a0tid = mp; - int b0tid = aCount2 + diag - mp; - if(extended) { - SerialMerge(keys_shared, a0tid, 0, b0tid, 0, results, - indices, comp); - } else { - int a1tid = aCount; - int b1tid = aCount2 + bCount; - SerialMerge(keys_shared, a0tid, a1tid, b0tid, b1tid, - results, indices, comp); - } - } else { - // Use the input intervals from the ranges between the merge path - // intersections. - aCount = a1 - a0; - bCount = b1 - b0; - - // Load the data into shared memory. - DeviceLoad2ToShared(a_global + a0, aCount, b_global + b0, - bCount, tid, keys_shared); - - // Run a merge path to find the start of the serial merge for each - // thread. - int diag = VT * tid; - int mp = MergePath(keys_shared, aCount, - keys_shared + aCount, bCount, diag, comp); - - // Compute the ranges of the sources in shared memory. - int a0tid = mp; - int a1tid = aCount; - int b0tid = aCount + diag - mp; - int b1tid = aCount + bCount; - - // Serial merge into register. - SerialMerge(keys_shared, a0tid, a1tid, b0tid, b1tid, results, - indices, comp); - } -} - -//////////////////////////////////////////////////////////////////////////////// -// DeviceMerge -// Merge pairs from global memory into global memory. Useful factorization to -// enable calling from merge, mergesort, and locality sort. - -template -MGPU_DEVICE void DeviceMerge(KeysIt1 aKeys_global, ValsIt1 aVals_global, - int aCount, KeysIt2 bKeys_global, ValsIt2 bVals_global, int bCount, - int tid, int block, int4 range, KeyType* keys_shared, int* indices_shared, - KeysIt3 keys_global, ValsIt3 vals_global, Comp comp) { - - KeyType results[VT]; - int indices[VT]; - DeviceMergeKeysIndices(aKeys_global, aCount, - bKeys_global, bCount, range, tid, keys_shared, results, indices, comp); - - // Store merge results back to shared memory. - DeviceThreadToShared(results, tid, keys_shared); - - // Store merged keys to global memory. - aCount = range.y - range.x; - bCount = range.w - range.z; - DeviceSharedToGlobal(aCount + bCount, keys_shared, tid, - keys_global + NT * VT * block); - - // Copy the values. - if(HasValues) { - DeviceThreadToShared(indices, tid, indices_shared); - - DeviceTransferMergeValuesShared(aCount + bCount, - aVals_global + range.x, bVals_global + range.z, aCount, - indices_shared, tid, vals_global + NT * VT * block); - } -} - -} // namespace mgpu diff --git a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctascan.cuh b/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctascan.cuh deleted file mode 100644 index 91ed8c4e04c7..000000000000 --- a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctascan.cuh +++ /dev/null @@ -1,308 +0,0 @@ -/****************************************************************************** - * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions are met: - * * Redistributions of source code must retain the above copyright - * notice, this list of conditions and the following disclaimer. - * * Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * * Neither the name of the NVIDIA CORPORATION nor the - * names of its contributors may be used to endorse or promote products - * derived from this software without specific prior written permission. - * - * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" - * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE - * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE - * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY - * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES - * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; - * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND - * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS - * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - * - ******************************************************************************/ - -/****************************************************************************** - * - * Code and text by Sean Baxter, NVIDIA Research - * See http://nvlabs.github.io/moderngpu for repository and documentation. - * - ******************************************************************************/ - -#pragma once - -#include "../mgpuenums.h" -#include "deviceutil.cuh" -#include "intrinsics.cuh" - -namespace mgpu { - -//////////////////////////////////////////////////////////////////////////////// -// CTAReduce - -template > -struct CTAReduce { - typedef typename Op::first_argument_type T; - enum { Size = NT, Capacity = NT }; - struct Storage { T shared[Capacity]; }; - - MGPU_DEVICE static T Reduce(int tid, T x, Storage& storage, Op op = Op()) { - storage.shared[tid] = x; - __syncthreads(); - - // Fold the data in half with each pass. - #pragma unroll - for(int destCount = NT / 2; destCount >= 1; destCount /= 2) { - if(tid < destCount) { - // Read from the right half and store to the left half. - x = op(x, storage.shared[destCount + tid]); - storage.shared[tid] = x; - } - __syncthreads(); - } - T total = storage.shared[0]; - __syncthreads(); - return total; - } -}; - -#if __CUDA_ARCH__ >= 300 - -template -struct CTAReduce > { - typedef mgpu::plus Op; - typedef int T; - enum { Size = NT, Capacity = WARP_SIZE }; - struct Storage { int shared[Capacity]; }; - - MGPU_DEVICE static int Reduce(int tid, int x, Storage& storage, - Op op = Op()) { - - const int NumSections = WARP_SIZE; - const int SecSize = NT / NumSections; - int lane = (SecSize - 1) & tid; - int sec = tid / SecSize; - - // In the first phase, threads cooperatively find the reduction within - // their segment. The segments are SecSize threads (NT / WARP_SIZE) - // wide. - #pragma unroll - for(int offset = 1; offset < SecSize; offset *= 2) - x = shfl_add(x, offset, SecSize); - - // The last thread in each segment stores the local reduction to shared - // memory. - if(SecSize - 1 == lane) storage.shared[sec] = x; - __syncthreads(); - - // Reduce the totals of each input segment. The spine is WARP_SIZE - // threads wide. - if(tid < NumSections) { - x = storage.shared[tid]; - #pragma unroll - for(int offset = 1; offset < NumSections; offset *= 2) - x = shfl_add(x, offset, NumSections); - storage.shared[tid] = x; - } - __syncthreads(); - - int reduction = storage.shared[NumSections - 1]; - __syncthreads(); - - return reduction; - } -}; - -template -struct CTAReduce > { - typedef mgpu::maximum Op; - enum { Size = NT, Capacity = WARP_SIZE }; - struct Storage { int shared[Capacity]; }; - - MGPU_DEVICE static int Reduce(int tid, int x, Storage& storage, - Op op = Op()) { - - const int NumSections = WARP_SIZE; - const int SecSize = NT / NumSections; - int lane = (SecSize - 1) & tid; - int sec = tid / SecSize; - - #pragma unroll - for(int offset = 1; offset < SecSize; offset *= 2) - x = shfl_max(x, offset, SecSize); - - if(SecSize - 1 == lane) storage.shared[sec] = x; - __syncthreads(); - - if(tid < NumSections) { - x = storage.shared[tid]; - #pragma unroll - for(int offset = 1; offset < NumSections; offset *= 2) - x = shfl_max(x, offset, NumSections); - storage.shared[tid] = x; - } - __syncthreads(); - - int reduction = storage.shared[NumSections - 1]; - __syncthreads(); - - return reduction; - } -}; - -#endif // __CUDA_ARCH__ >= 300 - -//////////////////////////////////////////////////////////////////////////////// -// CTAScan - -template > -struct CTAScan { - typedef typename Op::result_type T; - enum { Size = NT, Capacity = 2 * NT + 1 }; - struct Storage { T shared[Capacity]; }; - - MGPU_DEVICE static T Scan(int tid, T x, Storage& storage, T* total, - MgpuScanType type = MgpuScanTypeExc, T identity = (T)0, Op op = Op()) { - - storage.shared[tid] = x; - int first = 0; - __syncthreads(); - - #pragma unroll - for(int offset = 1; offset < NT; offset += offset) { - if(tid >= offset) - x = op(storage.shared[first + tid - offset], x); - first = NT - first; - storage.shared[first + tid] = x; - __syncthreads(); - } - *total = storage.shared[first + NT - 1]; - - if(MgpuScanTypeExc == type) - x = tid ? storage.shared[first + tid - 1] : identity; - - __syncthreads(); - return x; - } - MGPU_DEVICE static T Scan(int tid, T x, Storage& storage) { - T total; - return Scan(tid, x, storage, &total, MgpuScanTypeExc, (T)0, Op()); - } -}; - -//////////////////////////////////////////////////////////////////////////////// -// Special partial specialization for CTAScan on Kepler. -// This uses the shfl intrinsic to reduce scan latency. - -#if __CUDA_ARCH__ >= 300 - -template -struct CTAScan > { - typedef mgpu::plus Op; - enum { Size = NT, NumSegments = WARP_SIZE, SegSize = NT / NumSegments }; - enum { Capacity = NumSegments + 1 }; - struct Storage { int shared[Capacity + 1]; }; - - MGPU_DEVICE static int Scan(int tid, int x, Storage& storage, int* total, - MgpuScanType type = MgpuScanTypeExc, int identity = 0, Op op = Op()) { - - // Define WARP_SIZE segments that are NT / WARP_SIZE large. - // Each warp makes log(SegSize) shfl_add calls. - // The spine makes log(WARP_SIZE) shfl_add calls. - int lane = (SegSize - 1) & tid; - int segment = tid / SegSize; - - // Scan each segment using shfl_add. - int scan = x; - #pragma unroll - for(int offset = 1; offset < SegSize; offset *= 2) - scan = shfl_add(scan, offset, SegSize); - - // Store the reduction (last element) of each segment into storage. - if(SegSize - 1 == lane) storage.shared[segment] = scan; - __syncthreads(); - - // Warp 0 does a full shfl warp scan on the partials. The total is - // stored to shared[NumSegments]. (NumSegments = WARP_SIZE) - if(tid < NumSegments) { - int y = storage.shared[tid]; - int scan = y; - #pragma unroll - for(int offset = 1; offset < NumSegments; offset *= 2) - scan = shfl_add(scan, offset, NumSegments); - storage.shared[tid] = scan - y; - if(NumSegments - 1 == tid) storage.shared[NumSegments] = scan; - } - __syncthreads(); - - // Add the scanned partials back in and convert to exclusive scan. - scan += storage.shared[segment]; - if(MgpuScanTypeExc == type) { - scan -= x; - if(identity && !tid) scan = identity; - } - *total = storage.shared[NumSegments]; - __syncthreads(); - - return scan; - } - MGPU_DEVICE static int Scan(int tid, int x, Storage& storage) { - int total; - return Scan(tid, x, storage, &total, MgpuScanTypeExc, 0); - } -}; - -#endif // __CUDA_ARCH__ >= 300 - -//////////////////////////////////////////////////////////////////////////////// -// CTABinaryScan - -template -MGPU_DEVICE int CTABinaryScan(int tid, bool x, int* shared, int* total) { - const int NumWarps = NT / WARP_SIZE; - int warp = tid / WARP_SIZE; - int lane = (WARP_SIZE - 1); - - // Store the bit totals for each warp. - uint bits = __ballot(x); - shared[warp] = popc(bits); - __syncthreads(); - -#if __CUDA_ARCH__ >= 300 - if(tid < NumWarps) { - int x = shared[tid]; - int scan = x; - #pragma unroll - for(int offset = 1; offset < NumWarps; offset *= 2) - scan = shfl_add(scan, offset, NumWarps); - shared[tid] = scan - x; - } - __syncthreads(); - -#else - // Thread 0 scans warp totals. - if(!tid) { - int scan = 0; - #pragma unroll - for(int i = 0; i < NumWarps; ++i) { - int y = shared[i]; - shared[i] = scan; - scan += y; - } - shared[NumWarps] = scan; - } - __syncthreads(); - -#endif // __CUDA_ARCH__ >= 300 - - // Add the warp scan back into the partials. - int scan = shared[warp] + __popc(bfe(bits, 0, lane)); - *total = shared[NumWarps]; - __syncthreads(); - return scan; -} - -} // namespace mgpu diff --git a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasearch.cuh b/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasearch.cuh deleted file mode 100644 index 1797d4b51a42..000000000000 --- a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasearch.cuh +++ /dev/null @@ -1,207 +0,0 @@ -/****************************************************************************** - * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions are met: - * * Redistributions of source code must retain the above copyright - * notice, this list of conditions and the following disclaimer. - * * Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * * Neither the name of the NVIDIA CORPORATION nor the - * names of its contributors may be used to endorse or promote products - * derived from this software without specific prior written permission. - * - * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" - * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE - * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE - * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY - * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES - * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; - * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND - * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS - * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - * - ******************************************************************************/ - -/****************************************************************************** - * - * Code and text by Sean Baxter, NVIDIA Research - * See http://nvlabs.github.io/moderngpu for repository and documentation. - * - ******************************************************************************/ - -#pragma once - -#include "deviceutil.cuh" -#include "../mgpudevice.cuh" - -namespace mgpu { - -template -MGPU_HOST_DEVICE void BinarySearchIt(It data, int& begin, int& end, T key, - int shift, Comp comp) { - - IntT scale = (1<< shift) - 1; - int mid = (int)((begin + scale * end)>> shift); - - T key2 = data[mid]; - bool pred = (MgpuBoundsUpper == Bounds) ? - !comp(key, key2) : - comp(key2, key); - if(pred) begin = mid + 1; - else end = mid; -} - -template -MGPU_HOST_DEVICE int BiasedBinarySearch(It data, int count, T key, int levels, - Comp comp) { - - int begin = 0; - int end = count; - - if(levels >= 4 && begin < end) - BinarySearchIt(data, begin, end, key, 9, comp); - if(levels >= 3 && begin < end) - BinarySearchIt(data, begin, end, key, 7, comp); - if(levels >= 2 && begin < end) - BinarySearchIt(data, begin, end, key, 5, comp); - if(levels >= 1 && begin < end) - BinarySearchIt(data, begin, end, key, 4, comp); - - while(begin < end) - BinarySearchIt(data, begin, end, key, 1, comp); - return begin; -} - -template -MGPU_HOST_DEVICE int BinarySearch(It data, int count, T key, Comp comp) { - int begin = 0; - int end = count; - while(begin < end) - BinarySearchIt(data, begin, end, key, 1, comp); - return begin; -} - -//////////////////////////////////////////////////////////////////////////////// -// MergePath search - -template -MGPU_HOST_DEVICE int MergePath(It1 a, int aCount, It2 b, int bCount, int diag, - Comp comp) { - - typedef typename std::iterator_traits::value_type T; - int begin = max(0, diag - bCount); - int end = min(diag, aCount); - - while(begin < end) { - int mid = (begin + end)>> 1; - T aKey = a[mid]; - T bKey = b[diag - 1 - mid]; - bool pred = (MgpuBoundsUpper == Bounds) ? - comp(aKey, bKey) : - !comp(bKey, aKey); - if(pred) begin = mid + 1; - else end = mid; - } - return begin; -} - - -//////////////////////////////////////////////////////////////////////////////// -// SegmentedMergePath search - -template -MGPU_HOST_DEVICE int SegmentedMergePath(InputIt keys, int aOffset, int aCount, - int bOffset, int bCount, int leftEnd, int rightStart, int diag, Comp comp) { - - // leftEnd and rightStart are defined from the origin, and diag is defined - // from aOffset. - // We only need to run a Merge Path search if the diagonal intersects the - // segment that strides the left and right halves (i.e. is between leftEnd - // and rightStart). - if(aOffset + diag <= leftEnd) return diag; - if(aOffset + diag >= rightStart) return aCount; - - bCount = min(bCount, rightStart - bOffset); - int begin = max(max(leftEnd - aOffset, 0), diag - bCount); - int end = min(diag, aCount); - - while(begin < end) { - int mid = (begin + end)>> 1; - int ai = aOffset + mid; - int bi = bOffset + diag - 1 - mid; - - bool pred = !comp(keys[bi], keys[ai]); - if(pred) begin = mid + 1; - else end = mid; - } - return begin; -} - -//////////////////////////////////////////////////////////////////////////////// -// BalancedPath search - -template -MGPU_HOST_DEVICE int2 BalancedPath(InputIt1 a, int aCount, InputIt2 b, - int bCount, int diag, int levels, Comp comp) { - - typedef typename std::iterator_traits::value_type T; - - int p = MergePath(a, aCount, b, bCount, diag, comp); - int aIndex = p; - int bIndex = diag - p; - - bool star = false; - if(bIndex < bCount) { - if(Duplicates) { - T x = b[bIndex]; - - // Search for the beginning of the duplicate run in both A and B. - // Because - int aStart = BiasedBinarySearch(a, aIndex, x, - levels, comp); - int bStart = BiasedBinarySearch(b, bIndex, x, - levels, comp); - - // The distance between the merge path and the lower_bound is the - // 'run'. We add up the a- and b- runs and evenly distribute them to - // get a stairstep path. - int aRun = aIndex - aStart; - int bRun = bIndex - bStart; - int xCount = aRun + bRun; - - // Attempt to advance b and regress a. - int bAdvance = max(xCount>> 1, bRun); - int bEnd = min(bCount, bStart + bAdvance + 1); - int bRunEnd = BinarySearch(b + bIndex, - bEnd - bIndex, x, comp) + bIndex; - bRun = bRunEnd - bStart; - - bAdvance = min(bAdvance, bRun); - int aAdvance = xCount - bAdvance; - - bool roundUp = (aAdvance == bAdvance + 1) && (bAdvance < bRun); - aIndex = aStart + aAdvance; - - if(roundUp) star = true; - } else { - if(aIndex && aCount) { - T aKey = a[aIndex - 1]; - T bKey = b[bIndex]; - - // If the last consumed element in A (aIndex - 1) is the same as - // the next element in B (bIndex), we're sitting at a starred - // partition. - if(!comp(aKey, bKey)) star = true; - } - } - } - return make_int2(aIndex, star); -} - -} // namespace mgpu diff --git a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasegreduce.cuh b/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasegreduce.cuh deleted file mode 100644 index 959748fddb96..000000000000 --- a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasegreduce.cuh +++ /dev/null @@ -1,238 +0,0 @@ -/****************************************************************************** - * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions are met: - * * Redistributions of source code must retain the above copyright - * notice, this list of conditions and the following disclaimer. - * * Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * * Neither the name of the NVIDIA CORPORATION nor the - * names of its contributors may be used to endorse or promote products - * derived from this software without specific prior written permission. - * - * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" - * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE - * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE - * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY - * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES - * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; - * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND - * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS - * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - * - ******************************************************************************/ - -/****************************************************************************** - * - * Code and text by Sean Baxter, NVIDIA Research - * See http://nvlabs.github.io/moderngpu for repository and documentation. - * - ******************************************************************************/ - -#pragma once - -#include "ctasegscan.cuh" -#include "ctasearch.cuh" - -namespace mgpu { - -//////////////////////////////////////////////////////////////////////////////// -// Segmented reduce utility functions. - -// Extract the upper-bound indices from the coded ranges. Decrement to include -// the first addressed row/segment. - -struct SegReduceRange { - int begin; - int end; - int total; - bool flushLast; -}; - -MGPU_DEVICE SegReduceRange DeviceShiftRange(int limit0, int limit1) { - SegReduceRange range; - range.begin = 0x7fffffff & limit0; - range.end = 0x7fffffff & limit1; - range.total = range.end - range.begin; - range.flushLast = 0 == (0x80000000 & limit1); - range.end += !range.flushLast; - return range; -} - -// Reconstitute row/segment indices from a starting row index and packed end -// flags. Used for pre-processed versions of interval reduce and interval Spmv. -template -MGPU_DEVICE void DeviceExpandFlagsToRows(int first, int endFlags, - int rows[VT + 1]) { - - rows[0] = first; - #pragma unroll - for(int i = 0; i < VT; ++i) { - if((1<< i) & endFlags) ++first; - rows[i + 1] = first; - } -} - -//////////////////////////////////////////////////////////////////////////////// -// After loading CSR terms into shared memory, each thread binary searches -// (upper-bound) to find its starting point. Each thread then walks forward, -// emitting the csr0-relative row indices to register. - -template -MGPU_DEVICE int DeviceExpandCsrRows(int tidOffset, int* csr_shared, - int numRows, int end, int rows[VT + 1], int rowStarts[VT]) { - - // Each thread binary searches for its starting row. - int row = BinarySearch(csr_shared, numRows, tidOffset, - mgpu::less()) - 1; - - // Each thread starts at row and scans forward, emitting row IDs into - // register. Store the CTA-local row index (starts at 0) to rows and the - // start of the row (globally) to rowStarts. - int curOffset = csr_shared[row]; - int nextOffset = (row + 1 < numRows) ? csr_shared[row + 1] : end; - - rows[0] = row; - rowStarts[0] = curOffset; - int endFlags = 0; - - #pragma unroll - for(int i = 1; i <= VT; ++i) { - // Advance the row cursor when the iterator hits the next row offset. - if(tidOffset + i == nextOffset) { - // Set an end flag when the cursor advances to the next row. - endFlags |= 1<< (i - 1); - - // Advance the cursor and load the next row offset. - ++row; - curOffset = nextOffset; - nextOffset = (row + 1 < numRows) ? csr_shared[row + 1] : end; - } - rows[i] = row; - if(i < VT) rowStarts[i] = curOffset; - } - __syncthreads(); - - return endFlags; -} - -//////////////////////////////////////////////////////////////////////////////// -// DeviceSegReducePrepare -// Expand non-empty interval of CSR elements into row indices. Compute end-flags -// by comparing adjacent row IDs. - -// DeviceSegReducePrepare may be called either by a pre-processing kernel or by -// the kernel that actually evaluates the segmented reduction if no preprocesing -// is desired. -struct SegReduceTerms { - int endFlags; - int tidDelta; -}; - -template -MGPU_DEVICE SegReduceTerms DeviceSegReducePrepare(int* csr_shared, int numRows, - int tid, int gid, bool flushLast, int rows[VT + 1], int rowStarts[VT]) { - - // Pass a sentinel (end) to point to the next segment start. If we flush, - // this is the end of this tile. Otherwise it is INT_MAX - int endFlags = DeviceExpandCsrRows(gid + VT * tid, csr_shared, - numRows, flushLast ? (gid + NT * VT) : INT_MAX, rows, rowStarts); - - // Find the distance to to scan to compute carry-in for each thread. Use the - // existance of an end flag anywhere in the thread to determine if carry-out - // values from the left should propagate through to the right. - int tidDelta = DeviceFindSegScanDelta(tid, rows[0] != rows[VT], - csr_shared); - - SegReduceTerms terms = { endFlags, tidDelta }; - return terms; -} - -//////////////////////////////////////////////////////////////////////////////// -// CTASegReduce -// Core segmented reduction code. Supports fast-path and slow-path for intra-CTA -// segmented reduction. Stores partials to global memory. -// Callers feed CTASegReduce::ReduceToGlobal values in thread order. -template -struct CTASegReduce { - typedef CTASegScan SegScan; - - enum { - NV = NT * VT, - Capacity = HalfCapacity ? (NV / 2) : NV - }; - - union Storage { - typename SegScan::Storage segScanStorage; - T values[Capacity]; - }; - - template - MGPU_DEVICE static void ReduceToGlobal(const int rows[VT + 1], int total, - int tidDelta, int startRow, int block, int tid, T data[VT], - DestIt dest_global, T* carryOut_global, T identity, Op op, - Storage& storage) { - - // Run a segmented scan within the thread. - T x, localScan[VT]; - #pragma unroll - for(int i = 0; i < VT; ++i) { - x = i ? op(x, data[i]) : data[i]; - localScan[i] = x; - if(rows[i] != rows[i + 1]) x = identity; - } - - // Run a parallel segmented scan over the carry-out values to compute - // carry-in. - T carryOut; - T carryIn = SegScan::SegScanDelta(tid, tidDelta, x, - storage.segScanStorage, &carryOut, identity, op); - - // Store the carry-out for the entire CTA to global memory. - if(!tid) carryOut_global[block] = carryOut; - - dest_global += startRow; - if(HalfCapacity && total > Capacity) { - // Add carry-in to each thread-local scan value. Store directly - // to global. - #pragma unroll - for(int i = 0; i < VT; ++i) { - // Add the carry-in to the local scan. - T x2 = op(carryIn, localScan[i]); - - // Store on the end flag and clear the carry-in. - if(rows[i] != rows[i + 1]) { - carryIn = identity; - dest_global[rows[i]] = x2; - } - } - } else { - // All partials fit in shared memory. Add carry-in to each thread- - // local scan value. - #pragma unroll - for(int i = 0; i < VT; ++i) { - // Add the carry-in to the local scan. - T x2 = op(carryIn, localScan[i]); - - // Store reduction when the segment changes and clear the - // carry-in. - if(rows[i] != rows[i + 1]) { - storage.values[rows[i]] = x2; - carryIn = identity; - } - } - __syncthreads(); - - // Cooperatively store reductions to global memory. - for(int index = tid; index < total; index += NT) - dest_global[index] = storage.values[index]; - __syncthreads(); - } - } -}; - -} // namespace mgpu - diff --git a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasegscan.cuh b/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasegscan.cuh deleted file mode 100644 index 492d73a54750..000000000000 --- a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasegscan.cuh +++ /dev/null @@ -1,137 +0,0 @@ -/****************************************************************************** - * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions are met: - * * Redistributions of source code must retain the above copyright - * notice, this list of conditions and the following disclaimer. - * * Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * * Neither the name of the NVIDIA CORPORATION nor the - * names of its contributors may be used to endorse or promote products - * derived from this software without specific prior written permission. - * - * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" - * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE - * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE - * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY - * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES - * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; - * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND - * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS - * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - * - ******************************************************************************/ - -/****************************************************************************** - * - * Code and text by Sean Baxter, NVIDIA Research - * See http://nvlabs.github.io/moderngpu for repository and documentation. - * - ******************************************************************************/ - -#pragma once - -#include "ctascan.cuh" - -namespace mgpu { - -//////////////////////////////////////////////////////////////////////////////// -// DeviceFindSegScanDelta -// Runs an inclusive max-index scan over binary inputs. - -template -MGPU_DEVICE int DeviceFindSegScanDelta(int tid, bool flag, int* delta_shared) { - const int NumWarps = NT / 32; - - int warp = tid / 32; - int lane = 31 & tid; - uint warpMask = 0xffffffff>> (31 - lane); // inclusive search - uint ctaMask = 0x7fffffff>> (31 - lane); // exclusive search - - uint warpBits = __ballot(flag); - delta_shared[warp] = warpBits; - __syncthreads(); - - if(tid < NumWarps) { - uint ctaBits = __ballot(0 != delta_shared[tid]); - int warpSegment = 31 - clz(ctaMask & ctaBits); - int start = (-1 != warpSegment) ? - (31 - clz(delta_shared[warpSegment]) + 32 * warpSegment) : 0; - delta_shared[NumWarps + tid] = start; - } - __syncthreads(); - - // Find the closest flag to the left of this thread within the warp. - // Include the flag for this thread. - int start = 31 - clz(warpMask & warpBits); - if(-1 != start) start += ~31 & tid; - else start = delta_shared[NumWarps + warp]; - __syncthreads(); - - return tid - start; -} - -//////////////////////////////////////////////////////////////////////////////// -// CTASegScan - -template > -struct CTASegScan { - typedef _Op Op; - typedef typename Op::result_type T; - enum { NumWarps = NT / 32, Size = NT, Capacity = 2 * NT }; - union Storage { - int delta[NumWarps]; - T values[Capacity]; - }; - - // Each thread passes the reduction of the LAST SEGMENT that it covers. - // flag is set to true if there's at least one segment flag in the thread. - // SegScan returns the reduction of values for the first segment in this - // thread over the preceding threads. - // Return the value init for the first thread. - - // When scanning single elements per thread, interpret the flag as a BEGIN - // FLAG. If tid's flag is set, its value belongs to thread tid + 1, not - // thread tid. - - // The function returns the reduction of the last segment in the CTA. - - MGPU_DEVICE static T SegScanDelta(int tid, int tidDelta, T x, - Storage& storage, T* carryOut, T identity = (T)0, Op op = Op()) { - - // Run an inclusive scan - int first = 0; - storage.values[first + tid] = x; - __syncthreads(); - - #pragma unroll - for(int offset = 1; offset < NT; offset += offset) { - if(tidDelta >= offset) - x = op(storage.values[first + tid - offset], x); - first = NT - first; - storage.values[first + tid] = x; - __syncthreads(); - } - - // Get the exclusive scan. - x = tid ? storage.values[first + tid - 1] : identity; - *carryOut = storage.values[first + NT - 1]; - __syncthreads(); - return x; - } - - MGPU_DEVICE static T SegScan(int tid, T x, bool flag, Storage& storage, - T* carryOut, T identity = (T)0, Op op = Op()) { - - // Find the left-most thread that covers the first segment of this - // thread. - int tidDelta = DeviceFindSegScanDelta(tid, flag, storage.delta); - - return SegScanDelta(tid, tidDelta, x, storage, carryOut, identity, op); - } -}; - -} // namespace mgpu diff --git a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasegsort.cuh b/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasegsort.cuh deleted file mode 100644 index 6a9516ae446e..000000000000 --- a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasegsort.cuh +++ /dev/null @@ -1,443 +0,0 @@ -/****************************************************************************** - * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions are met: - * * Redistributions of source code must retain the above copyright - * notice, this list of conditions and the following disclaimer. - * * Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * * Neither the name of the NVIDIA CORPORATION nor the - * names of its contributors may be used to endorse or promote products - * derived from this software without specific prior written permission. - * - * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" - * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE - * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE - * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY - * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES - * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; - * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND - * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS - * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - * - ******************************************************************************/ - -/****************************************************************************** - * - * Code and text by Sean Baxter, NVIDIA Research - * See http://nvlabs.github.io/moderngpu for repository and documentation. - * - ******************************************************************************/ - -#pragma once - -#include "ctascan.cuh" -#include "ctasearch.cuh" -#include "loadstore.cuh" -#include "sortnetwork.cuh" - -namespace mgpu { - -template -MGPU_DEVICE void SegmentedSerialMerge(const T* keys_shared, int aBegin, - int aEnd, int bBegin, int bEnd, T results[VT], int indices[VT], - int leftEnd, int rightStart, Comp comp, bool sync = true) { - - bEnd = min(rightStart, bEnd); - T aKey = keys_shared[aBegin]; - T bKey = keys_shared[bBegin]; - - #pragma unroll - for(int i = 0; i < VT; ++i) { - bool p; - - // If A has run out of inputs, emit B. - if(aBegin >= aEnd) - p = false; - else if(bBegin >= bEnd || aBegin < leftEnd) - // B has hit the end of the middle segment. - // Emit A if A has inputs remaining in the middle segment. - p = true; - else - // Emit the smaller element in the middle segment. - p = !comp(bKey, aKey); - - results[i] = p ? aKey : bKey; - indices[i] = p ? aBegin : bBegin; - if(p) aKey = keys_shared[++aBegin]; - else bKey = keys_shared[++bBegin]; - } - if(sync) { __syncthreads(); } -} - -//////////////////////////////////////////////////////////////////////////////// -// CTASegsortPass - -template -MGPU_DEVICE void CTASegsortPass(T* keys_shared, int* ranges_shared, int tid, - int pass, T results[VT], int indices[VT], int2& activeRange, Comp comp) { - - // Locate the intervals of the input lists. - int3 frame = FindMergesortFrame(2<< pass, tid, VT); - int a0 = frame.x; - int b0 = frame.y; - int listLen = frame.z; - int list = tid>> pass; - int listParity = 1 & list; - int diag = VT * tid - frame.x; - - // Fetch the active range for the list this thread's list is merging with. - int siblingRange = ranges_shared[1 ^ list]; - int siblingStart = 0x0000ffff & siblingRange; - int siblingEnd = siblingRange>> 16; - - // Create a new active range for the merge. - int leftEnd = listParity ? siblingEnd : activeRange.y; - int rightStart = listParity ? activeRange.x : siblingStart; - activeRange.x = min(activeRange.x, siblingStart); - activeRange.y = max(activeRange.y, siblingEnd); - - int p = SegmentedMergePath(keys_shared, a0, listLen, b0, listLen, leftEnd, - rightStart, diag, comp); - - int a0tid = a0 + p; - int b0tid = b0 + diag - p; - SegmentedSerialMerge(keys_shared, a0tid, b0, b0tid, b0 + listLen, - results, indices, leftEnd, rightStart, comp); - - // Store the ranges to shared memory. - if(0 == diag) - ranges_shared[list>> 1] = - (int)bfi(activeRange.y, activeRange.x, 16, 16); -} - -//////////////////////////////////////////////////////////////////////////////// -// CTASegsortLoop - -template -MGPU_DEVICE int2 CTASegsortLoop(KeyType threadKeys[VT], - ValType threadValues[VT], KeyType* keys_shared, ValType* values_shared, - int* ranges_shared, int tid, int2 activeRange, Comp comp) { - - const int NumPasses = sLogPow2::value; - #pragma unroll - for(int pass = 0; pass < NumPasses; ++pass) { - int indices[VT]; - CTASegsortPass(keys_shared, ranges_shared, tid, pass, - threadKeys, indices, activeRange, comp); - - if(HasValues) { - // Exchange values through shared memory. - DeviceThreadToShared(threadValues, tid, values_shared); - DeviceGather(NT * VT, values_shared, indices, tid, - threadValues); - } - - // Store results in shared memory in sorted order. - DeviceThreadToShared(threadKeys, tid, keys_shared); - } - return activeRange; -} - -//////////////////////////////////////////////////////////////////////////////// -// CTASegsort -// Pass keys and values in register. On return, values are returned in register -// and keys returned in shared memory. - -template -MGPU_DEVICE int2 CTASegsort(KeyType threadKeys[VT], ValType threadValues[VT], - int tid, int headFlags, KeyType* keys_shared, ValType* values_shared, - int* ranges_shared, Comp comp) { - - if(Stable) - // Odd-even transpose sort. - OddEvenTransposeSortFlags(threadKeys, threadValues, headFlags, - comp); - else - // Batcher's odd-even mergesort. - OddEvenMergesortFlags(threadKeys, threadValues, headFlags, comp); - - // Record the first and last occurrence of head flags in this segment. - int blockEnd = 31 - clz(headFlags); - if(-1 != blockEnd) blockEnd += VT * tid; - - int blockStart = ffs(headFlags); - blockStart = blockStart ? (VT * tid - 1 + blockStart) : (NT * VT); - - ranges_shared[tid] = (int)bfi(blockEnd, blockStart, 16, 16); - - // Store back to shared mem. The values are in VT-length sorted lists. - // These are merged recursively. - DeviceThreadToShared(threadKeys, tid, keys_shared); - - int2 activeRange = CTASegsortLoop(threadKeys, - threadValues, keys_shared, values_shared, ranges_shared, tid, - make_int2(blockStart, blockEnd), comp); - return activeRange; -} - - -template -MGPU_DEVICE int2 CTASegsortKeys(KeyType threadKeys[VT], int tid, int headFlags, - KeyType* keys_shared, int* ranges_shared, Comp comp) { - - int valuesTemp[VT]; - return CTASegsort(threadKeys, valuesTemp, tid, - headFlags, keys_shared, (int*)keys_shared, ranges_shared, comp); -} - -template -MGPU_DEVICE int2 CTASegsortPairs(KeyType threadKeys[VT], - ValType threadValues[VT], int tid, int headFlags, KeyType* keys_shared, - ValType* values_shared, int* ranges_shared, Comp comp) { - - return CTASegsort(threadKeys, threadValues, tid, - headFlags, keys_shared, values_shared, ranges_shared, comp); -} - -//////////////////////////////////////////////////////////////////////////////// -// DeviceSegBlocksort -// Load keys and values from global memory, sort in shared memory, and store -// back to global memory. Store the left-most and right-most encountered -// headflag locations to ranges_global to prepare for the next pass. -// This function is factored out of the blocksort kernel to allow easier -// customization of that kernel - we have two implementations currently: -// sort over indices and sort over bitfield. - -template -MGPU_DEVICE void DeviceSegBlocksort(InputIt1 keys_global, - InputIt2 values_global, int count2, KeyType* keys_shared, - ValType* values_shared, int* ranges_shared, int headFlags, int tid, - int block, OutputIt1 keysDest_global, OutputIt2 valsDest_global, - int* ranges_global, Comp comp) { - - // Load keys into register in thread order. - int gid = NT * VT * block; - KeyType threadKeys[VT]; - DeviceGlobalToShared(count2, keys_global + gid, tid, keys_shared); - DeviceSharedToThread(keys_shared, tid, threadKeys); - - // Load the values from global memory and into register in thread order. - ValType threadValues[VT]; - if(HasValues) { - DeviceGlobalToShared(count2, values_global + gid, tid, - values_shared); - DeviceSharedToThread(values_shared, tid, threadValues); - } - - // Run the CTA segmented blocksort. - int2 activeRange = CTASegsort(threadKeys, - threadValues, tid, headFlags, keys_shared, values_shared, ranges_shared, - comp); - - // Store the keys to global memory. - DeviceSharedToGlobal(count2, keys_shared, tid, - keysDest_global + gid); - - if(HasValues) { - // Store the values to global memory.xk b - DeviceThreadToShared(threadValues, tid, values_shared); - DeviceSharedToGlobal(count2, values_shared, tid, - valsDest_global + gid, false); - } - - // Store the 16-bit packed ranges. These are used by all merge kernels and - // the first level of global segmented merge path partitioning. - if(!tid) - ranges_global[block] = bfi(activeRange.y, activeRange.x, 16, 16); -} - -//////////////////////////////////////////////////////////////////////////////// -// DeviceIndicesToHeadFlags -// Load indices from an array and cooperatively turn into a head flag bitfield -// for each thread. - -template -MGPU_DEVICE int DeviceIndicesToHeadFlags(const int* indices_global, - const int* partitions_global, int tid, int block, int count2, - int* words_shared, byte* flags_shared) { - - const int FlagWordsPerThread = MGPU_DIV_UP(VT, 4); - int gid = NT * VT * block; - int p0 = partitions_global[block]; - int p1 = partitions_global[block + 1]; - - int headFlags = 0; - if(p1 > p0 || count2 < NT * VT) { - - // Clear the flag bytes, then loop through the indices and poke in flag - // values. - #pragma unroll - for(int i = 0; i < FlagWordsPerThread; ++i) - words_shared[NT * i + tid] = 0; - __syncthreads(); - - for(int index = p0 + tid; index < p1; index += NT) { - int headFlag = indices_global[index]; - flags_shared[headFlag - gid] = 1; - } - __syncthreads(); - - // Combine all the head flags for this thread. - int first = VT * tid; - int offset = first / 4; - int prev = words_shared[offset]; - int mask = 0x3210 + 0x1111 * (3 & first); - #pragma unroll - for(int i = 0; i < FlagWordsPerThread; ++i) { - // Gather the next four flags. - int next = words_shared[offset + 1 + i]; - int x = prmt(prev, next, mask); - prev = next; - - // Set the head flag bits. - if(0x00000001 & x) headFlags |= 1<< (4 * i); - if(0x00000100 & x) headFlags |= 1<< (4 * i + 1); - if(0x00010000 & x) headFlags |= 1<< (4 * i + 2); - if(0x01000000 & x) headFlags |= 1<< (4 * i + 3); - } - __syncthreads(); - - // Set head flags for out-of-range keys. - int outOfRange = min(VT, first + VT - count2); - if(outOfRange > 0) - headFlags = bfi(0xffffffff, headFlags, VT - outOfRange, outOfRange); - - // Clear head flags above VT. - headFlags &= (1<< VT) - 1; - } - return headFlags; -} - -//////////////////////////////////////////////////////////////////////////////// -// SegSortSupport - -struct SegSortSupport { - int* ranges_global; - int2* ranges2_global; - - int4* mergeList_global; - int* copyList_global; - int2* queueCounters_global; - int2* nextCounters_global; - - byte* copyStatus_global; -}; - -//////////////////////////////////////////////////////////////////////////////// -// DeviceSegSortMerge - -template -MGPU_DEVICE void DeviceSegSortMerge(const KeyType* keys_global, - const ValueType* values_global, int2 segmentRange, int tid, - int block, int4 range, int pass, KeyType* keys_shared, - int* indices_shared, KeyType* keysDest_global, ValueType* valsDest_global, - Comp comp) { - - const int NV = NT * VT; - int gid = NV * block; - - // Load the local compressed segment indices. - int a0 = range.x; - int aCount = range.y - range.x; - int b0 = range.z; - int bCount = range.w - range.z; - - DeviceLoad2ToShared(keys_global + a0, aCount, keys_global + b0, - bCount, tid, keys_shared); - - //////////////////////////////////////////////////////////////////////////// - // Run a merge path to find the starting point for each thread to merge. - // If the entire warp fits into the already-sorted segments, we can skip - // sorting it and leave its keys in shared memory. Doing this on the warp - // level rather than thread level (also legal) gives slightly better - // performance. - - int segStart = segmentRange.x; - int segEnd = segmentRange.y; - int listParity = 1 & (block>> pass); - - int warpOffset = VT * (~31 & tid); - bool sortWarp = listParity ? - // The spliced segment is to the left (segStart). - (warpOffset < segStart) : - // The spliced segment is to the right (segEnd). - (warpOffset + 32 * VT > segEnd); - - KeyType threadKeys[VT]; - int indices[VT]; - if(sortWarp) { - int diag = VT * tid; - int mp = SegmentedMergePath(keys_shared, 0, aCount, aCount, bCount, - listParity ? 0 : segEnd, listParity ? segStart : NV, diag, comp); - int a0tid = mp; - int a1tid = aCount; - int b0tid = aCount + diag - mp; - int b1tid = aCount + bCount; - - // Serial merge into register. All threads in the CTA so we hoist the - // check for list parity outside the function call to simplify the - // logic. Unlike in the blocksort, this does not cause warp divergence. - SegmentedSerialMerge(keys_shared, a0tid, a1tid, b0tid, b1tid, - threadKeys, indices, listParity ? 0 : segEnd, - listParity ? segStart : NV, comp, false); - } - __syncthreads(); - - // Store sorted data in register back to shared memory. Then copy to global. - if(sortWarp) - DeviceThreadToShared(threadKeys, tid, keys_shared, false); - __syncthreads(); - - DeviceSharedToGlobal(aCount + bCount, keys_shared, tid, - keysDest_global + gid); - - //////////////////////////////////////////////////////////////////////////// - // Use the merge indices to gather values from global memory. Store directly - // to valsDest_global. - - if(HasValues) { - // Transpose the gather indices to help coalesce loads. - if(sortWarp) - DeviceThreadToShared(indices, tid, indices_shared, false); - else { - #pragma unroll - for(int i = 0; i < VT; ++i) - indices_shared[VT * tid + i] = VT * tid + i; - } - __syncthreads(); - - DeviceTransferMergeValuesShared(aCount + bCount, - values_global + a0, values_global + b0, aCount, indices_shared, - tid, valsDest_global + NV * block); - } -} - -//////////////////////////////////////////////////////////////////////////////// -// DeviceSegSortCopy - -template -MGPU_DEVICE void DeviceSegSortCopy(const KeyType* keys_global, - const ValueType* values_global, int tid, int block, int count, - KeyType* keysDest_global, ValueType* valsDest_global) { - - int gid = NT * VT * block; - int count2 = min(NT * VT, count - gid); - - DeviceGlobalToGlobal(count2, keys_global + gid, tid, - keysDest_global + gid); - if(HasValues) - DeviceGlobalToGlobal(count2, values_global + gid, tid, - valsDest_global + gid); -} - -} // namespace mgpu diff --git a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasortedsearch.cuh b/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasortedsearch.cuh deleted file mode 100644 index 48b7f3a8fe10..000000000000 --- a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/ctasortedsearch.cuh +++ /dev/null @@ -1,208 +0,0 @@ -/****************************************************************************** - * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions are met: - * * Redistributions of source code must retain the above copyright - * notice, this list of conditions and the following disclaimer. - * * Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * * Neither the name of the NVIDIA CORPORATION nor the - * names of its contributors may be used to endorse or promote products - * derived from this software without specific prior written permission. - * - * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" - * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE - * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE - * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY - * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES - * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; - * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND - * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS - * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - * - ******************************************************************************/ - -/****************************************************************************** - * - * Code and text by Sean Baxter, NVIDIA Research - * See http://nvlabs.github.io/moderngpu for repository and documentation. - * - ******************************************************************************/ - -#pragma once - -#include "../mgpudevice.cuh" -#include "ctasearch.cuh" - -namespace mgpu { - - -//////////////////////////////////////////////////////////////////////////////// -// DeviceSerialSearch - -template -MGPU_DEVICE int3 DeviceSerialSearch(const T* keys_shared, int aBegin, - int aEnd, int bBegin, int bEnd, int aOffset, int bOffset, int* indices, - Comp comp) { - - const int FlagA = IndexA ? 0x80000000 : 1; - const int FlagB = IndexB ? 0x80000000 : 1; - - T aKey = keys_shared[aBegin]; - T bKey = keys_shared[bBegin]; - T aPrev, bPrev; - if(aBegin > 0) aPrev = keys_shared[aBegin - 1]; - if(bBegin > 0) bPrev = keys_shared[bBegin - 1]; - int decisions = 0; - int matchCountA = 0; - int matchCountB = 0; - - #pragma unroll - for(int i = 0; i < VT; ++i) { - bool p; - if(RangeCheck && aBegin >= aEnd) p = false; - else if(RangeCheck && bBegin >= bEnd) p = true; - else p = (MgpuBoundsUpper == Bounds) ? - comp(aKey, bKey) : - !comp(bKey, aKey); - - if(p) { - // aKey is smaller than bKey, so it is inserted before bKey. - // Save bKey's index (bBegin + first) as the result of the search - // and advance to the next needle in A. - bool match = false; - if(MatchA) { - // Test if there is an element in B that matches aKey. - if(MgpuBoundsUpper == Bounds) { - // Upper Bound: We're inserting aKey after bKey. If there - // is a match for aKey it must be bPrev. Check that bPrev - // is in range and equal to aKey. - // The predicate test result !comp(aKey, bPrev) was - // established on the previous A-advancing iteration (it - // failed the comp(aKey, bKey) test to get us to this - // point). Check the other half of the equality condition - // with a second comparison. - bool inRange = !RangeCheck || (bBegin > aEnd); - match = inRange && !comp(bPrev, aKey); - } else { - // Lower Bound: We're inserting aKey before bKey. If there - // is a match for aKey, it must be bKey. Check that bKey - // is in range and equal to aKey. - // The predicate test !comp(bKey, aKey) has established one - // half of the equality condition. We establish the other - // half with a second comparison. - bool inRange = !RangeCheck || (bBegin < bEnd); - match = inRange && !comp(aKey, bKey); - } - } - - int index = 0; - if(IndexA) index = bOffset + bBegin; - if(match) index |= FlagA; - if(IndexA || MatchA) indices[i] = index; - matchCountA += match; - - // Mark the decision bit to indicate that this iteration has - // progressed A (the needles). - decisions |= 1<< i; - aPrev = aKey; - aKey = keys_shared[++aBegin]; - } else { - // aKey is larger than bKey, so it is inserted after bKey (but we - // don't know where yet). Advance the B index to the next element in - // the haystack to continue the search for the current needle. - bool match = false; - if(MatchB) { - if(MgpuBoundsUpper == Bounds) { - // Upper Bound: aKey is not smaller than bKey. We advance to - // the next haystack element in B. If there is a match in A - // for bKey it must be aKey. By entering this branch we've - // verified that !comp(aKey, bKey). Making the reciprocal - // comparison !comp(bKey, aKey) establishes aKey == bKey. - bool inRange = !RangeCheck || - ((bBegin < bEnd) && (aBegin < aEnd)); - match = inRange && !comp(bKey, aKey); - } else { - // Lower Bound: bKey is smaller than aKey. We advance to the - // next element in B. If there is a match for bKey, it must - // be aPrev. The previous A-advancing iteration proved that - // !comp(bKey, aPrev). We test !comp(aPrev, bKey) for the - // other half of the equality condition. - bool inRange = !RangeCheck || - ((bBegin < bEnd) && (aBegin > 0)); - match = inRange && !comp(aPrev, bKey); - } - } - - int index = 0; - if(IndexB) index = aOffset + aBegin; - if(match) index |= FlagB; - if(IndexB || MatchB) indices[i] = index; - matchCountB += match; - - // Keep the decision bit cleared to indicate that this iteration - // has progressed B (the haystack). - bPrev = bKey; - bKey = keys_shared[++bBegin]; - } - } - return make_int3(decisions, matchCountA, matchCountB); -} - -//////////////////////////////////////////////////////////////////////////////// -// CTASortedSearch -// Take keys in shared memory and return indices and b-match flags in shared -// memory. -// NOTE: This function doesn't do any strided-to-thread order transposes so -// using an even number of values per thread will incur no additional bank -// conflicts. - -template -MGPU_DEVICE int2 CTASortedSearch(T* keys_shared, int aStart, int aCount, - int aEnd, int a0, int bStart, int bCount, int bEnd, int b0, bool extended, - int tid, int* indices_shared, Comp comp) { - - // Run a merge path to find the start of the serial search for each thread. - int diag = VT * tid; - int mp = MergePath(keys_shared + aStart, aCount, - keys_shared + bStart, bCount, diag, comp); - int a0tid = mp; - int b0tid = diag - mp; - - // Serial search into register. - int3 results; - int indices[VT]; - if(extended) - results = DeviceSerialSearch(keys_shared, a0tid + aStart, aEnd, b0tid + bStart, bEnd, - a0 - aStart, b0 - bStart, indices, comp); - else - results = DeviceSerialSearch(keys_shared, a0tid + aStart, aEnd, b0tid + bStart, bEnd, - a0 - aStart, b0 - bStart, indices, comp); - __syncthreads(); - - // Compact the indices into shared memory. Use the decision bits (set is A, - // cleared is B) to select the destination. - int decisions = results.x; - b0tid += aCount; - #pragma unroll - for(int i = 0; i < VT; ++i) { - if((1<< i) & decisions) { - if(IndexA || MatchA) indices_shared[a0tid++] = indices[i]; - } else { - if(IndexB || MatchB) indices_shared[b0tid++] = indices[i]; - } - } - __syncthreads(); - - // Return the match counts for A and B keys. - return make_int2(results.y, results.z); -} - -} // namespace mgpu diff --git a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/devicetypes.cuh b/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/devicetypes.cuh deleted file mode 100644 index 282bedd5ab23..000000000000 --- a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/devicetypes.cuh +++ /dev/null @@ -1,363 +0,0 @@ -/****************************************************************************** - * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions are met: - * * Redistributions of source code must retain the above copyright - * notice, this list of conditions and the following disclaimer. - * * Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * * Neither the name of the NVIDIA CORPORATION nor the - * names of its contributors may be used to endorse or promote products - * derived from this software without specific prior written permission. - * - * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" - * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE - * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE - * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY - * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES - * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; - * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND - * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS - * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - * - ******************************************************************************/ - -/****************************************************************************** - * - * Code and text by Sean Baxter, NVIDIA Research - * See http://nvlabs.github.io/moderngpu for repository and documentation. - * - ******************************************************************************/ - -#pragma once - -#if __CUDA_ARCH__ == 100 - #error "COMPUTE CAPABILITY 1.0 NOT SUPPORTED BY MPGU. TRY 2.0!" -#endif - -#include -#include "../util/static.h" - -#ifdef _MSC_VER -#define INLINESYMBOL __forceinline__ -#else -#define INLINESYMBOL inline -#endif - -namespace mgpu { - -#define MGPU_HOST __host__ INLINESYMBOL -#define MGPU_DEVICE __device__ INLINESYMBOL -#define MGPU_HOST_DEVICE __host__ __device__ INLINESYMBOL - -const int WARP_SIZE = 32; -const int LOG_WARP_SIZE = 5; - -//////////////////////////////////////////////////////////////////////////////// -// Device-side comparison operators - -template -struct less : public std::binary_function { - MGPU_HOST_DEVICE bool operator()(T a, T b) { return a < b; } -}; -template -struct less_equal : public std::binary_function { - MGPU_HOST_DEVICE bool operator()(T a, T b) { return a <= b; } -}; -template -struct greater : public std::binary_function { - MGPU_HOST_DEVICE bool operator()(T a, T b) { return a > b; } -}; -template -struct greater_equal : public std::binary_function { - MGPU_HOST_DEVICE bool operator()(T a, T b) { return a >= b; } -}; -template -struct equal_to : public std::binary_function { - MGPU_HOST_DEVICE bool operator()(T a, T b) { return a == b; } -}; -template -struct not_equal_to : public std::binary_function { - MGPU_HOST_DEVICE bool operator()(T a, T b) { return a != b; } -}; - -//////////////////////////////////////////////////////////////////////////////// -// Device-side arithmetic operators - -template -struct plus : public std::binary_function { - MGPU_HOST_DEVICE T operator()(T a, T b) { return a + b; } -}; - -template -struct minus : public std::binary_function { - MGPU_HOST_DEVICE T operator()(T a, T b) { return a - b; } -}; - -template -struct multiplies : public std::binary_function { - MGPU_HOST_DEVICE T operator()(T a, T b) { return a * b; } -}; - -template -struct modulus : public std::binary_function { - MGPU_HOST_DEVICE T operator()(T a, T b) { return a % b; } -}; - -template -struct bit_or : public std::binary_function { - MGPU_HOST_DEVICE T operator()(T a, T b) { return a | b; } -}; - -template -struct bit_and : public std::binary_function { - MGPU_HOST_DEVICE T operator()(T a, T b) { return a & b; } -}; - -template -struct bit_xor : public std::binary_function { - MGPU_HOST_DEVICE T operator()(T a, T b) { return a ^ b; } -}; - -template -struct maximum : public std::binary_function { - MGPU_HOST_DEVICE T operator()(T a, T b) { return max(a, b); } -}; - -template -struct minimum : public std::binary_function { - MGPU_HOST_DEVICE T operator()(T a, T b) { return min(a, b); } -}; - -//////////////////////////////////////////////////////////////////////////////// - -template -MGPU_HOST_DEVICE void swap(T& a, T& b) { - T c = a; - a = b; - b = c; -} - -template -struct DevicePair { - T x, y; -}; - -template -MGPU_HOST_DEVICE DevicePair MakeDevicePair(T x, T y) { - DevicePair p = { x, y }; - return p; -} - -template struct numeric_limits; -template<> struct numeric_limits { - MGPU_HOST_DEVICE static int min() { return INT_MIN; } - MGPU_HOST_DEVICE static int max() { return INT_MAX; } - MGPU_HOST_DEVICE static int lowest() { return INT_MIN; } - MGPU_HOST_DEVICE static int AddIdent() { return 0; } - MGPU_HOST_DEVICE static int MulIdent() { return 1; } -}; -template<> struct numeric_limits { - MGPU_HOST_DEVICE static long long min() { return LLONG_MIN; } - MGPU_HOST_DEVICE static long long max() { return LLONG_MAX; } - MGPU_HOST_DEVICE static long long lowest() { return LLONG_MIN; } - MGPU_HOST_DEVICE static long long AddIdent() { return 0; } - MGPU_HOST_DEVICE static long long MulIdent() { return 1; } -}; -template<> struct numeric_limits { - MGPU_HOST_DEVICE static uint min() { return 0; } - MGPU_HOST_DEVICE static uint max() { return UINT_MAX; } - MGPU_HOST_DEVICE static uint lowest() { return 0; } - MGPU_HOST_DEVICE static uint AddIdent() { return 0; } - MGPU_HOST_DEVICE static uint MulIdent() { return 1; } -}; -template<> struct numeric_limits { - MGPU_HOST_DEVICE static unsigned long long min() { return 0; } - MGPU_HOST_DEVICE static unsigned long long max() { return ULLONG_MAX; } - MGPU_HOST_DEVICE static unsigned long long lowest() { return 0; } - MGPU_HOST_DEVICE static unsigned long long AddIdent() { return 0; } - MGPU_HOST_DEVICE static unsigned long long MulIdent() { return 1; } -}; -template<> struct numeric_limits { - MGPU_HOST_DEVICE static float min() { return FLT_MIN; } - MGPU_HOST_DEVICE static float max() { return FLT_MAX; } - MGPU_HOST_DEVICE static float lowest() { return -FLT_MAX; } - MGPU_HOST_DEVICE static float AddIdent() { return 0; } - MGPU_HOST_DEVICE static float MulIdent() { return 1; } -}; -template<> struct numeric_limits { - MGPU_HOST_DEVICE static double min() { return DBL_MIN; } - MGPU_HOST_DEVICE static double max() { return DBL_MAX; } - MGPU_HOST_DEVICE static double lowest() { return -DBL_MAX; } - MGPU_HOST_DEVICE static double AddIdent() { return 0; } - MGPU_HOST_DEVICE static double MulIdent() { return 1; } -}; - - -MGPU_HOST_DEVICE int2 operator+(int2 a, int2 b) { - return make_int2(a.x + b.x, a.y + b.y); -} -MGPU_HOST_DEVICE int2& operator+=(int2& a, int2 b) { - a = a + b; - return a; -} -MGPU_HOST_DEVICE int2 operator*(int2 a, int2 b) { - return make_int2(a.x * b.x, a.y * b.y); -} -MGPU_HOST_DEVICE int2& operator*=(int2& a, int2 b) { - a = a * b; - return a; -} - -template -MGPU_HOST_DEVICE T max(T a, T b) { -#if !defined(__CUDA_ARCH__) || (__CUDA_ARCH__ < 100) - return std::max(a, b); -#else - return (a < b) ? b : a; -#endif -} -template -MGPU_HOST_DEVICE T min(T a, T b) { -#if !defined(__CUDA_ARCH__) || (__CUDA_ARCH__ < 100) - return std::min(a, b); -#else - return (b < a) ? b : a; -#endif -} - -MGPU_HOST_DEVICE int2 max(int2 a, int2 b) { - return make_int2(max(a.x, b.x), max(a.y, b.y)); -} - -MGPU_HOST_DEVICE int2 min(int2 a, int2 b) { - return make_int2(min(a.x, b.x), min(a.y, b.y)); -} - -template<> struct numeric_limits { - MGPU_HOST_DEVICE static int2 min() { return make_int2(INT_MIN, INT_MIN); } - MGPU_HOST_DEVICE static int2 max() { return make_int2(INT_MAX, INT_MAX); } - MGPU_HOST_DEVICE static int2 lowest() { - return make_int2(INT_MIN, INT_MIN); - } - MGPU_HOST_DEVICE static int2 AddIdent() { return make_int2(0, 0); } - MGPU_HOST_DEVICE static int2 MulIdent() { return make_int2(1, 1); } -}; - -template -class constant_iterator : public std::iterator_traits { -public: - MGPU_HOST_DEVICE constant_iterator(T value) : _value(value) { } - - MGPU_HOST_DEVICE T operator[](ptrdiff_t i) const { - return _value; - } - MGPU_HOST_DEVICE T operator*() const { - return _value; - } - MGPU_HOST_DEVICE constant_iterator operator+(ptrdiff_t diff) const { - return constant_iterator(_value); - } - MGPU_HOST_DEVICE constant_iterator operator-(ptrdiff_t diff) const { - return constant_iterator(_value); - } - MGPU_HOST_DEVICE constant_iterator& operator+=(ptrdiff_t diff) { - return *this; - } - MGPU_HOST_DEVICE constant_iterator& operator-=(ptrdiff_t diff) { - return *this; - } -private: - T _value; -}; - -template -class counting_iterator : public std::iterator_traits { -public: - MGPU_HOST_DEVICE counting_iterator(T value) : _value(value) { } - - MGPU_HOST_DEVICE T operator[](ptrdiff_t i) { - return _value + i; - } - MGPU_HOST_DEVICE T operator*() { - return _value; - } - MGPU_HOST_DEVICE counting_iterator operator+(ptrdiff_t diff) { - return counting_iterator(_value + diff); - } - MGPU_HOST_DEVICE counting_iterator operator-(ptrdiff_t diff) { - return counting_iterator(_value - diff); - } - MGPU_HOST_DEVICE counting_iterator& operator+=(ptrdiff_t diff) { - _value += diff; - return *this; - } - MGPU_HOST_DEVICE counting_iterator& operator-=(ptrdiff_t diff) { - _value -= diff; - return *this; - } -private: - T _value; -}; - -template -class step_iterator : public std::iterator_traits { -public: - MGPU_HOST_DEVICE step_iterator(T base, T step) : - _base(base), _step(step), _offset(0) { } - - MGPU_HOST_DEVICE T operator[](ptrdiff_t i) { - return _base + (_offset + i) * _step; - } - MGPU_HOST_DEVICE T operator*() { - return _base + _offset * _step; - } - MGPU_HOST_DEVICE step_iterator operator+(ptrdiff_t diff) { - step_iterator it = *this; - it._offset += diff; - return it; - } - MGPU_HOST_DEVICE step_iterator operator-(ptrdiff_t diff) { - step_iterator it = *this; - it._offset -= diff; - return it; - } - MGPU_HOST_DEVICE step_iterator& operator+=(ptrdiff_t diff) { - _offset += diff; - return *this; - } - MGPU_HOST_DEVICE step_iterator& operator-=(ptrdiff_t diff) { - _offset -= diff; - return *this; - } -private: - ptrdiff_t _offset; - T _base, _step; -}; - -} // namespace mgpu - - -template -MGPU_HOST_DEVICE mgpu::counting_iterator operator+(ptrdiff_t diff, - mgpu::counting_iterator it) { - return it + diff; -} -template -MGPU_HOST_DEVICE mgpu::counting_iterator operator-(ptrdiff_t diff, - mgpu::counting_iterator it) { - return it + (-diff); -} -template -MGPU_HOST_DEVICE mgpu::step_iterator operator+(ptrdiff_t diff, - mgpu::step_iterator it) { - return it + diff; -} -template -MGPU_HOST_DEVICE mgpu::step_iterator operator-(ptrdiff_t diff, - mgpu::step_iterator it) { - return it + (-diff); -} diff --git a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/deviceutil.cuh b/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/deviceutil.cuh deleted file mode 100644 index e18807f38496..000000000000 --- a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/deviceutil.cuh +++ /dev/null @@ -1,143 +0,0 @@ -/****************************************************************************** - * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions are met: - * * Redistributions of source code must retain the above copyright - * notice, this list of conditions and the following disclaimer. - * * Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * * Neither the name of the NVIDIA CORPORATION nor the - * names of its contributors may be used to endorse or promote products - * derived from this software without specific prior written permission. - * - * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" - * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE - * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE - * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY - * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES - * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; - * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND - * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS - * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - * - ******************************************************************************/ - -/****************************************************************************** - * - * Code and text by Sean Baxter, NVIDIA Research - * See http://nvlabs.github.io/moderngpu for repository and documentation. - * - ******************************************************************************/ - -#pragma once - -#include "intrinsics.cuh" - -namespace mgpu { - -// Get the difference between two pointers in bytes. -MGPU_HOST_DEVICE ptrdiff_t PtrDiff(const void* a, const void* b) { - return (const byte*)b - (const byte*)a; -} - -// Offset a pointer by i bytes. -template -MGPU_HOST_DEVICE const T* PtrOffset(const T* p, ptrdiff_t i) { - return (const T*)((const byte*)p + i); -} -template -MGPU_HOST_DEVICE T* PtrOffset(T* p, ptrdiff_t i) { - return (T*)((byte*)p + i); -} - -//////////////////////////////////////////////////////////////////////////////// -// Task range support -// Evenly distributes variable-length arrays over a fixed number of CTAs. - -MGPU_HOST int2 DivideTaskRange(int numItems, int numWorkers) { - div_t d = div(numItems, numWorkers); - return make_int2(d.quot, d.rem); -} - -MGPU_HOST_DEVICE int2 ComputeTaskRange(int block, int2 task) { - int2 range; - range.x = task.x * block; - range.x += min(block, task.y); - range.y = range.x + task.x + (block < task.y); - return range; -} - -MGPU_HOST_DEVICE int2 ComputeTaskRange(int block, int2 task, int blockSize, - int count) { - int2 range = ComputeTaskRange(block, task); - range.x *= blockSize; - range.y = min(count, range.y * blockSize); - return range; -} - -//////////////////////////////////////////////////////////////////////////////// -// DeviceExtractHeadFlags -// Input array flags is a bit array with 32 head flags per word. -// ExtractThreadHeadFlags returns numBits flags starting at bit index. - -MGPU_HOST_DEVICE uint DeviceExtractHeadFlags(const uint* flags, int index, - int numBits) { - - int index2 = index>> 5; - int shift = 31 & index; - uint headFlags = flags[index2]>> shift; - int shifted = 32 - shift; - - if(shifted < numBits) - // We also need to shift in the next set of bits. - headFlags = bfi(flags[index2 + 1], headFlags, shifted, shift); - headFlags &= (1<< numBits) - 1; - return headFlags; -} - -//////////////////////////////////////////////////////////////////////////////// -// DevicePackHeadFlags -// Pack VT bits per thread at 32 bits/thread. Will consume an integer number of -// words, because CTA size is a multiple of 32. The first NT * VT / 32 threads -// return packed words. - -template -MGPU_DEVICE uint DevicePackHeadFlags(uint threadBits, int tid, - uint* flags_shared) { - - const int WordCount = NT * VT / 32; - - // Each thread stores its thread bits to flags_shared[tid]. - flags_shared[tid] = threadBits; - __syncthreads(); - - uint packed = 0; - if(tid < WordCount) { - const int Items = MGPU_DIV_UP(32, VT); - int index = 32 * tid; - int first = index / VT; - int bit = 0; - - int rem = index - VT * first; - packed = flags_shared[first]>> rem; - bit = VT - rem; - ++first; - - #pragma unroll - for(int i = 0; i < Items; ++i) { - if(i < Items - 1 || bit < 32) { - uint x = flags_shared[first + i]; - if(bit < 32) packed |= x<< bit; - bit += VT; - } - } - } - __syncthreads(); - - return packed; -} - -} // namespace mgpu diff --git a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/intrinsics.cuh b/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/intrinsics.cuh deleted file mode 100644 index afcfc00e6617..000000000000 --- a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/intrinsics.cuh +++ /dev/null @@ -1,421 +0,0 @@ -/****************************************************************************** - * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions are met: - * * Redistributions of source code must retain the above copyright - * notice, this list of conditions and the following disclaimer. - * * Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * * Neither the name of the NVIDIA CORPORATION nor the - * names of its contributors may be used to endorse or promote products - * derived from this software without specific prior written permission. - * - * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" - * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE - * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE - * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY - * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES - * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; - * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND - * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS - * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - * - ******************************************************************************/ - -/****************************************************************************** - * - * Code and text by Sean Baxter, NVIDIA Research - * See http://nvlabs.github.io/moderngpu for repository and documentation. - * - ******************************************************************************/ - -#include "devicetypes.cuh" - -#pragma once - -#pragma GCC diagnostic push -#pragma GCC diagnostic ignored "-Wstrict-aliasing" - -namespace mgpu { - -MGPU_HOST_DEVICE uint2 ulonglong_as_uint2(uint64 x) { - return *reinterpret_cast(&x); -} -MGPU_HOST_DEVICE uint64 uint2_as_ulonglong(uint2 x) { - return *reinterpret_cast(&x); -} - -MGPU_HOST_DEVICE int2 longlong_as_int2(int64 x) { - return *reinterpret_cast(&x); -} -MGPU_HOST_DEVICE int64 int2_as_longlong(int2 x) { - return *reinterpret_cast(&x); -} - -MGPU_HOST_DEVICE int2 double_as_int2(double x) { - return *reinterpret_cast(&x); -} -MGPU_HOST_DEVICE double int2_as_double(int2 x) { - return *reinterpret_cast(&x); -} - -MGPU_HOST_DEVICE void SetDoubleX(double& d, int x) { - reinterpret_cast(&d)[0] = x; -} -MGPU_HOST_DEVICE int GetDoubleX(double d) { - return double_as_int2(d).x; -} -MGPU_HOST_DEVICE void SetDoubleY(double& d, int y) { - reinterpret_cast(&d)[1] = y; -} -MGPU_HOST_DEVICE int GetDoubleY(double d) { - return double_as_int2(d).y; -} - - -//////////////////////////////////////////////////////////////////////////////// -// PTX for bfe and bfi - -#if __CUDA_ARCH__ >= 200 - -MGPU_DEVICE uint bfe_ptx(uint x, uint bit, uint numBits) { - uint result; - asm("bfe.u32 %0, %1, %2, %3;" : - "=r"(result) : "r"(x), "r"(bit), "r"(numBits)); - return result; -} - - -MGPU_DEVICE uint bfi_ptx(uint x, uint y, uint bit, uint numBits) { - uint result; - asm("bfi.b32 %0, %1, %2, %3, %4;" : - "=r"(result) : "r"(x), "r"(y), "r"(bit), "r"(numBits)); - return result; -} - -MGPU_DEVICE uint prmt_ptx(uint a, uint b, uint index) { - uint ret; - asm("prmt.b32 %0, %1, %2, %3;" : "=r"(ret) : "r"(a), "r"(b), "r"(index)); - return ret; -} - -#endif // __CUDA_ARCH__ >= 200 - -#if CUDA_VERSION >= 9000 -//////////////////////////////////////////////////////////////////////////////// -// shfl_add - -MGPU_DEVICE int shfl_add(int x, int offset, int width = WARP_SIZE, unsigned int threadmask = 0xFFFFFFFF) { - int result = 0; -#if __CUDA_ARCH__ >= 300 - int mask = (WARP_SIZE - width)<< 8; - asm( - "{.reg .s32 r0;" - ".reg .pred p;" - "shfl.sync.up.b32 r0|p, %1, %2, %3, %4;" - "@p add.s32 r0, r0, %5;" - "mov.s32 %0, r0; }" - : "=r"(result) : "r"(x), "r"(offset), "r"(mask), "r"(threadmask), "r"(x)); -#endif - return result; -} - -MGPU_DEVICE int shfl_max(int x, int offset, int width = WARP_SIZE, unsigned int threadmask = 0xFFFFFFFF) { - int result = 0; -#if __CUDA_ARCH__ >= 300 - int mask = (WARP_SIZE - width)<< 8; - asm( - "{.reg .s32 r0;" - ".reg .pred p;" - "shfl.sync.up.b32 r0|p, %1, %2, %3, %4;" - "@p max.s32 r0, r0, %5;" - "mov.s32 %0, r0; }" - : "=r"(result) : "r"(x), "r"(offset), "r"(mask), "r"(threadmask), "r"(x)); -#endif - return result; -} -#else -//////////////////////////////////////////////////////////////////////////////// -// shfl_add - -MGPU_DEVICE int shfl_add(int x, int offset, int width = WARP_SIZE) { - int result = 0; -#if __CUDA_ARCH__ >= 300 - int mask = (WARP_SIZE - width)<< 8; - asm( - "{.reg .s32 r0;" - ".reg .pred p;" - "shfl.up.b32 r0|p, %1, %2, %3;" - "@p add.s32 r0, r0, %4;" - "mov.s32 %0, r0; }" - : "=r"(result) : "r"(x), "r"(offset), "r"(mask), "r"(x)); -#endif - return result; -} - -MGPU_DEVICE int shfl_max(int x, int offset, int width = WARP_SIZE) { - int result = 0; -#if __CUDA_ARCH__ >= 300 - int mask = (WARP_SIZE - width)<< 8; - asm( - "{.reg .s32 r0;" - ".reg .pred p;" - "shfl.up.b32 r0|p, %1, %2, %3;" - "@p max.s32 r0, r0, %4;" - "mov.s32 %0, r0; }" - : "=r"(result) : "r"(x), "r"(offset), "r"(mask), "r"(x)); -#endif - return result; -} -#endif - -//////////////////////////////////////////////////////////////////////////////// -// brev, popc, clz, bfe, bfi, prmt - -// Reverse the bits in an integer. -MGPU_HOST_DEVICE uint brev(uint x) { -#if __CUDA_ARCH__ >= 200 - uint y = __brev(x); -#else - uint y = 0; - for(int i = 0; i < 32; ++i) - y |= (1 & (x>> i))<< (31 - i); -#endif - return y; -} - -// Count number of bits in a register. -MGPU_HOST_DEVICE int popc(uint x) { -#if __CUDA_ARCH__ >= 200 - return __popc(x); -#else - int c; - for(c = 0; x; ++c) - x &= x - 1; - return c; -#endif -} - -// Count leading zeros - start from most significant bit. -MGPU_HOST_DEVICE int clz(int x) { -#if __CUDA_ARCH__ >= 200 - return __clz(x); -#else - for(int i = 31; i >= 0; --i) - if((1<< i) & x) return 31 - i; - return 32; -#endif -} - -// Find first set - start from least significant bit. LSB is 1. ffs(0) is 0. -MGPU_HOST_DEVICE int ffs(int x) { -#if __CUDA_ARCH__ >= 200 - return __ffs(x); -#else - for(int i = 0; i < 32; ++i) - if((1<< i) & x) return i + 1; - return 0; -#endif -} - -MGPU_HOST_DEVICE uint bfe(uint x, uint bit, uint numBits) { -#if __CUDA_ARCH__ >= 200 - return bfe_ptx(x, bit, numBits); -#else - return ((1<< numBits) - 1) & (x>> bit); -#endif -} - -MGPU_HOST_DEVICE uint bfi(uint x, uint y, uint bit, uint numBits) { - uint result; -#if __CUDA_ARCH__ >= 200 - result = bfi_ptx(x, y, bit, numBits); -#else - if(bit + numBits > 32) numBits = 32 - bit; - uint mask = ((1<< numBits) - 1)<< bit; - result = y & ~mask; - result |= mask & (x<< bit); -#endif - return result; -} - -MGPU_HOST_DEVICE uint prmt(uint a, uint b, uint index) { - uint result; -#if __CUDA_ARCH__ >= 200 - result = prmt_ptx(a, b, index); -#else - result = 0; - for(int i = 0; i < 4; ++i) { - uint sel = 0xf & (index>> (4 * i)); - uint x = ((7 & sel) > 3) ? b : a; - x = 0xff & (x>> (8 * (3 & sel))); - if(8 & sel) x = (128 & x) ? 0xff : 0; - result |= x<< (8 * i); - } -#endif - return result; -} - -// Find log2(x) and optionally round up to the next integer logarithm. -MGPU_HOST_DEVICE int FindLog2(int x, bool roundUp = false) { - int a = 31 - clz(x); - if(roundUp) a += !MGPU_IS_POW_2(x); - return a; -} - -//////////////////////////////////////////////////////////////////////////////// -// vset4 - -#if __CUDA_ARCH__ >= 300 - -// Performs four byte-wise comparisons and returns 1 for each byte that -// satisfies the conditional, and zero otherwise. -MGPU_DEVICE uint vset4_lt_add_ptx(uint a, uint b, uint c) { - uint result; - asm("vset4.u32.u32.lt.add %0, %1, %2, %3;" : - "=r"(result) : "r"(a), "r"(b), "r"(c)); - return result; -} -MGPU_DEVICE uint vset4_eq_ptx(uint a, uint b) { - uint result; - asm("vset4.u32.u32.eq %0, %1, %2, %3;" : - "=r"(result) : "r"(a), "r"(b), "r"(0)); - return result; -} -#endif // __CUDA_ARCH__ >= 300 - -MGPU_HOST_DEVICE uint vset4_lt_add(uint a, uint b, uint c) { - uint result; -#if __CUDA_ARCH__ >= 300 - result = vset4_lt_add_ptx(a, b, c); -#else - result = c; - if((0x000000ff & a) < (0x000000ff & b)) result += 0x00000001; - if((0x0000ff00 & a) < (0x0000ff00 & b)) result += 0x00000100; - if((0x00ff0000 & a) < (0x00ff0000 & b)) result += 0x00010000; - if((0xff000000 & a) < (0xff000000 & b)) result += 0x01000000; -#endif - return result; -} - -MGPU_HOST_DEVICE uint vset4_eq(uint a, uint b) { - uint result; -#if __CUDA_ARCH__ >= 300 - result = vset4_eq_ptx(a, b); -#else - result = 0; - if((0x000000ff & a) == (0x000000ff & b)) result = 0x00000001; - if((0x0000ff00 & a) == (0x0000ff00 & b)) result += 0x00000100; - if((0x00ff0000 & a) == (0x00ff0000 & b)) result += 0x00010000; - if((0xff000000 & a) == (0xff000000 & b)) result += 0x01000000; -#endif - return result; -} - -//////////////////////////////////////////////////////////////////////////////// -// - -MGPU_HOST_DEVICE uint umulhi(uint x, uint y) { -#if __CUDA_ARCH__ >= 100 - return __umulhi(x, y); -#else - uint64 product = (uint64)x * y; - return (uint)(product>> 32); -#endif -} - -//////////////////////////////////////////////////////////////////////////////// -// ldg() function defined for all devices and all types. Only compiles to __ldg -// intrinsic for __CUDA_ARCH__ >= 320 && __CUDA_ARCH__ < 400 for types supported -// by __ldg in sm_32_intrinsics.h - -template -struct IsLdgType { - enum { value = false }; -}; -#define DEFINE_LDG_TYPE(T) \ - template<> struct IsLdgType { enum { value = true }; }; - -template::value> -struct LdgShim { - MGPU_DEVICE static T Ldg(const T* p) { - return *p; - } -}; - -#if __CUDA_ARCH__ >= 320 && __CUDA_ARCH__ < 400 - - // List of __ldg-compatible types from sm_32_intrinsics.h. - DEFINE_LDG_TYPE(char) - DEFINE_LDG_TYPE(short) - DEFINE_LDG_TYPE(int) - DEFINE_LDG_TYPE(long long) - DEFINE_LDG_TYPE(char2) - DEFINE_LDG_TYPE(char4) - DEFINE_LDG_TYPE(short2) - DEFINE_LDG_TYPE(short4) - DEFINE_LDG_TYPE(int2) - DEFINE_LDG_TYPE(int4) - DEFINE_LDG_TYPE(longlong2) - - DEFINE_LDG_TYPE(unsigned char) - DEFINE_LDG_TYPE(unsigned short) - DEFINE_LDG_TYPE(unsigned int) - DEFINE_LDG_TYPE(unsigned long long) - DEFINE_LDG_TYPE(uchar2) - DEFINE_LDG_TYPE(uchar4) - DEFINE_LDG_TYPE(ushort2) - DEFINE_LDG_TYPE(ushort4) - DEFINE_LDG_TYPE(uint2) - DEFINE_LDG_TYPE(uint4) - DEFINE_LDG_TYPE(ulonglong2) - - DEFINE_LDG_TYPE(float) - DEFINE_LDG_TYPE(double) - DEFINE_LDG_TYPE(float2) - DEFINE_LDG_TYPE(float4) - DEFINE_LDG_TYPE(double2) - - template struct LdgShim { - MGPU_DEVICE static T Ldg(const T* p) { - return __ldg(p); - } - }; -#endif - -template -MGPU_DEVICE T ldg(const T* p) { - return LdgShim::Ldg(p); -} - -//////////////////////////////////////////////////////////////////////////////// - -// Fast division for 31-bit integers. -// Uses the method in Hacker's Delight (2nd edition) page 228. -// Evaluates for denom > 1 and x < 2^31. -struct FastDivide { - uint denom; - uint coef; - uint shift; - - MGPU_HOST_DEVICE uint Divide(uint x) { - return umulhi(x, coef)>> shift; - } - MGPU_HOST_DEVICE uint Modulus(uint x) { - return x - Divide(x) * denom; - } - - explicit FastDivide(uint denom_) { - denom = denom_; - uint p = 31 + FindLog2(denom, true); - coef = (uint)(((1ull<< p) + denom - 1) / denom); - shift = p - 32; - } -}; - -#pragma GCC diagnostic pop - -} // namespace mgpu diff --git a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/loadstore.cuh b/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/loadstore.cuh deleted file mode 100644 index aae17d8490a4..000000000000 --- a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/loadstore.cuh +++ /dev/null @@ -1,674 +0,0 @@ -/****************************************************************************** - * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions are met: - * * Redistributions of source code must retain the above copyright - * notice, this list of conditions and the following disclaimer. - * * Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * * Neither the name of the NVIDIA CORPORATION nor the - * names of its contributors may be used to endorse or promote products - * derived from this software without specific prior written permission. - * - * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" - * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE - * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE - * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY - * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES - * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; - * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND - * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS - * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - * - ******************************************************************************/ - -/****************************************************************************** - * - * Code and text by Sean Baxter, NVIDIA Research - * See http://nvlabs.github.io/moderngpu for repository and documentation. - * - ******************************************************************************/ - -#pragma once - -#include "../mgpudevice.cuh" -#include "deviceutil.cuh" -#include "intrinsics.cuh" - -namespace mgpu { - -//////////////////////////////////////////////////////////////////////////////// -// Cooperative load functions. - -template -MGPU_DEVICE void DeviceSharedToReg(InputIt data, int tid, T* reg, - bool sync) { - - #pragma unroll - for(int i = 0; i < VT; ++i) - reg[i] = data[NT * i + tid]; - - if(sync) __syncthreads(); -} - -template -MGPU_DEVICE void DeviceGlobalToRegPred(int count, InputIt data, int tid, - T* reg, bool sync) { - - // TODO: Attempt to issue 4 loads at a time. - #pragma unroll - for(int i = 0; i < VT; ++i) { - int index = NT * i + tid; - if(index < count) reg[i] = data[index]; - } - if(sync) __syncthreads(); -} - -template -MGPU_DEVICE void DeviceGlobalToReg(int count, InputIt data, int tid, - T* reg, bool sync) { - - if(count >= NT * VT) { - #pragma unroll - for(int i = 0; i < VT; ++i) - reg[i] = data[NT * i + tid]; - } else - DeviceGlobalToRegPred(count, data, tid, reg, false); - if(sync) __syncthreads(); -} -template -MGPU_DEVICE void DeviceGlobalToReg2(int count, InputIt data, int tid, - T* reg, bool sync) { - - DeviceGlobalToReg(count, data, tid, reg, false); - #pragma unroll - for(int i = VT0; i < VT1; ++i) { - int index = NT * i + tid; - if(index < count) reg[i] = data[index]; - } - if(sync) __syncthreads(); -} - -template -MGPU_DEVICE void DeviceGlobalToRegDefault(int count, InputIt data, int tid, - T* reg, T init, bool sync) { - - if(count >= NT * VT) { - #pragma unroll - for(int i = 0; i < VT; ++i) - reg[i] = data[NT * i + tid]; - } else { - #pragma unroll - for(int i = 0; i < VT; ++i) { - int index = NT * i + tid; - reg[i] = init; - if(index < count) reg[i] = data[index]; - } - } - if(sync) __syncthreads(); -} -template -MGPU_DEVICE void DeviceGlobalToRegDefault2(int count, InputIt data, int tid, - T* reg, T init, bool sync) { - - DeviceGlobalToRegDefault(count, data, tid, reg, init, false); - #pragma unroll - for(int i = VT0; i < VT1; ++i) { - int index = NT * i + tid; - reg[i] = init; - if(index < count) reg[i] = data[index]; - } - if(sync) __syncthreads(); -} - -//////////////////////////////////////////////////////////////////////////////// - -template -MGPU_DEVICE void DeviceGlobalToThread(int count, InputIt data, int tid, - T* reg) { - - data += VT * tid; - if(count >= NT * VT) { - #pragma unroll - for(int i = 0; i < VT; ++i) - reg[i] = ldg(data + i); - } else { - count -= VT * tid; - #pragma unroll - for(int i = 0; i < VT; ++i) - if(i < count) reg[i] = ldg(data + i); - } -} - -template -MGPU_DEVICE void DeviceGlobalToThreadDefault(int count, InputIt data, int tid, - T* reg, T init) { - - data += VT * tid; - if(count >= NT * VT) { - #pragma unroll - for(int i = 0; i < VT; ++i) - reg[i] = ldg(data + i); - } else { - count -= VT * tid; - #pragma unroll - for(int i = 0; i < VT; ++i) - reg[i] = (i < count) ? ldg(data + i) : init; - } -} - - -//////////////////////////////////////////////////////////////////////////////// -// Cooperative store functions. - -template -MGPU_DEVICE void DeviceRegToShared(const T* reg, int tid, - OutputIt dest, bool sync) { - - typedef typename std::iterator_traits::value_type T2; - #pragma unroll - for(int i = 0; i < VT; ++i) - dest[NT * i + tid] = (T2)reg[i]; - - if(sync) __syncthreads(); -} - -template -MGPU_DEVICE void DeviceRegToGlobal(int count, const T* reg, int tid, - OutputIt dest, bool sync) { - - #pragma unroll - for(int i = 0; i < VT; ++i) { - int index = NT * i + tid; - if(index < count) - dest[index] = reg[i]; - } - if(sync) __syncthreads(); -} - -//////////////////////////////////////////////////////////////////////////////// -// DeviceMemToMemLoop -// Transfer from shared memory to global, or global to shared, for transfers -// that are smaller than NT * VT in the average case. The goal is to reduce -// unnecessary comparison logic. - -template -MGPU_DEVICE void DeviceMemToMem4(int count, InputIt source, int tid, - OutputIt dest, bool sync) { - - typedef typename std::iterator_traits::value_type T; - - T x[VT]; - const int Count = (VT < 4) ? VT : 4; - if(count >= NT * VT) { - #pragma unroll - for(int i = 0; i < Count; ++i) - x[i] = source[NT * i + tid]; - #pragma unroll - for(int i = 0; i < Count; ++i) - dest[NT * i + tid] = x[i]; - } else { - #pragma unroll - for(int i = 0; i < Count; ++i) { - int index = NT * i + tid; - if(index < count) - x[i] = source[NT * i + tid]; - } - #pragma unroll - for(int i = 0; i < Count; ++i) { - int index = NT * i + tid; - if(index < count) - dest[index] = x[i]; - } - } - if(sync) __syncthreads(); -} -template -MGPU_DEVICE void DeviceMemToMemLoop(int count, InputIt source, int tid, - OutputIt dest, bool sync) { - - for(int i = 0; i < count; i += 4 * NT) - DeviceMemToMem4(count - i, source + i, tid, dest + i, - false); - if(sync) __syncthreads(); -} - - -//////////////////////////////////////////////////////////////////////////////// -// Functions to copy between shared and global memory where the average case is -// to transfer NT * VT elements. - -template -MGPU_DEVICE void DeviceSharedToGlobal(int count, const T* source, int tid, - OutputIt dest, bool sync) { - - typedef typename std::iterator_traits::value_type T2; - #pragma unroll - for(int i = 0; i < VT; ++i) { - int index = NT * i + tid; - if(index < count) dest[index] = (T2)source[index]; - } - if(sync) __syncthreads(); -} - -template -MGPU_DEVICE void DeviceGlobalToShared(int count, InputIt source, int tid, - T* dest, bool sync) { - - T reg[VT]; - DeviceGlobalToReg(count, source, tid, reg, false); - DeviceRegToShared(reg, tid, dest, sync); -} - -template -MGPU_DEVICE void DeviceGlobalToShared2(int count, InputIt source, int tid, - T* dest, bool sync) { - - T reg[VT1]; - DeviceGlobalToReg2(count, source, tid, reg, false); - DeviceRegToShared(reg, tid, dest, sync); -} - - -template -MGPU_DEVICE void DeviceGlobalToSharedDefault(int count, InputIt source, int tid, - T* dest, T init, bool sync) { - - T reg[VT]; - DeviceGlobalToRegDefault(count, source, tid, reg, init, false); - DeviceRegToShared(reg, tid, dest, sync); -} - -template -MGPU_DEVICE void DeviceGlobalToSharedDefault2(int count, InputIt data, int tid, - T* dest, T init, bool sync) { - - T reg[VT1]; - DeviceGlobalToRegDefault2(count, data, tid, reg, init, false); - DeviceRegToShared(reg, tid, dest, sync); -} - - -//////////////////////////////////////////////////////////////////////////////// - -template -MGPU_DEVICE void DeviceGlobalToSharedLoop(int count, InputIt source, int tid, - T* dest, bool sync) { - - const int Granularity = MGPU_MIN(VT, 3); - DeviceGlobalToShared(count, source, tid, dest, false); - - int offset = Granularity * NT; - if(count > offset) - DeviceGlobalToShared(count - offset, - source + offset, tid, dest + offset, false); - - if(sync) __syncthreads(); - - /* - source += tid; - while(count > 0) { - T reg[Granularity]; - #pragma unroll - for(int i = 0; i < Granularity; ++i) { - int index = NT * i + tid; - if(index < count) - reg[i] = source[NT * i]; - } - DeviceRegToShared(reg, tid, dest, false); - source += Granularity * NT; - dest += Granularity * NT; - count -= Granularity * NT; - } - if(sync) __syncthreads();*/ -} - -template -MGPU_DEVICE void DeviceGlobalToGlobal(int count, InputIt source, int tid, - OutputIt dest, bool sync) { - - typedef typename std::iterator_traits::value_type T; - T values[VT]; - DeviceGlobalToReg(count, source, tid, values, false); - DeviceRegToGlobal(count, values, tid, dest, sync); -} - -//////////////////////////////////////////////////////////////////////////////// -// Transponse VT elements in NT threads (x) into thread-order registers (y) -// using only NT * VT / 2 elements of shared memory. - -//This function definitely has a bug, don't use!!! fix TODO(erich) -template -MGPU_DEVICE void HalfSmemTranspose(const T* x, int tid, T* shared, T* y) { - printf("HalfSmemTranspose has a bug, use WAR SmemTranpose or find bug before using in production"); - // Transpose the first half values (tid < NT / 2) - #pragma unroll - for(int i = 0; i <= VT / 2; ++i) - if(i < VT / 2 || tid < NT / 2) - shared[NT * i + tid] = x[i]; - __syncthreads(); - - if(tid < NT / 2) { - #pragma unroll - for(int i = 0; i < VT; ++i) - y[i] = shared[VT * tid + i]; - } - __syncthreads(); - - // Transpose the second half values (tid >= NT / 2) - #pragma unroll - for(int i = VT / 2; i < VT; ++i) - if(i > VT / 2 || tid >= NT / 2) - shared[NT * i - NT * VT / 2 + tid] = x[i]; - __syncthreads(); - - if(tid >= NT / 2) { - #pragma unroll - for(int i = 0; i < VT; ++i) - y[i] = shared[VT * tid + i - NT * VT / 2]; - } - __syncthreads(); -} - -//////////////////////////////////////////////////////////////////////////////// -// Gather/scatter functions - -template -MGPU_DEVICE void DeviceGather(int count, InputIt data, int indices[VT], - int tid, T* reg, bool sync) { - - if(count >= NT * VT) { - #pragma unroll - for(int i = 0; i < VT; ++i) - reg[i] = data[indices[i]]; - } else { - #pragma unroll - for(int i = 0; i < VT; ++i) { - int index = NT * i + tid; - if(index < count) - reg[i] = data[indices[i]]; - } - } - if(sync) __syncthreads(); -} - -template -MGPU_DEVICE void DeviceGatherDefault(int count, InputIt data, int indices[VT], - int tid, T* reg, T identity, bool sync) { - - if(count >= NT * VT) { - #pragma unroll - for(int i = 0; i < VT; ++i) - reg[i] = data[indices[i]]; - } else { - #pragma unroll - for(int i = 0; i < VT; ++i) { - int index = NT * i + tid; - reg[i] = (index < count) ? data[indices[i]] : identity; - } - } - if(sync) __syncthreads(); -} - -template -MGPU_DEVICE void DeviceScatter(int count, const T* reg, int tid, - int indices[VT], OutputIt data, bool sync) { - - if(count >= NT * VT) { - #pragma unroll - for(int i = 0; i < VT; ++i) - data[indices[i]] = reg[i]; - } else { - #pragma unroll - for(int i = 0; i < VT; ++i) { - int index = NT * i + tid; - if(index < count) - data[indices[i]] = reg[i]; - } - } - if(sync) __syncthreads(); -} - -//////////////////////////////////////////////////////////////////////////////// -// Cooperative transpose functions (strided to thread order) - -template -MGPU_DEVICE void DeviceThreadToShared(const T* threadReg, int tid, T* shared, - bool sync) { - - if(1 & VT) { - // Odd grain size. Store as type T. - #pragma unroll - for(int i = 0; i < VT; ++i) - shared[VT * tid + i] = threadReg[i]; - } else { - // Even grain size. Store as DevicePair. This lets us exploit the - // 8-byte shared memory mode on Kepler. - DevicePair* dest = (DevicePair*)(shared + VT * tid); - #pragma unroll - for(int i = 0; i < VT / 2; ++i) - dest[i] = MakeDevicePair(threadReg[2 * i], threadReg[2 * i + 1]); - } - if(sync) __syncthreads(); -} - -template -MGPU_DEVICE void DeviceSharedToThread(const T* shared, int tid, T* threadReg, - bool sync) { - - if(1 & VT) { - #pragma unroll - for(int i = 0; i < VT; ++i) - threadReg[i] = shared[VT * tid + i]; - } else { - const DevicePair* source = (const DevicePair*)(shared + VT * tid); - #pragma unroll - for(int i = 0; i < VT / 2; ++i) { - DevicePair p = source[i]; - threadReg[2 * i] = p.x; - threadReg[2 * i + 1] = p.y; - } - } - if(sync) __syncthreads(); -} - -//////////////////////////////////////////////////////////////////////////////// -// DeviceLoad2 - load from pointers of the same type. Optimize for a single LD -// statement. - -template -MGPU_DEVICE void DeviceLoad2ToReg(const T* a_global, int aCount, - const T* b_global, int bCount, int tid, T* reg, bool sync) { - - int b0 = b_global - a_global - aCount; - int total = aCount + bCount; - if(total >= NT * VT0) { - #pragma unroll - for(int i = 0; i < VT0; ++i) { - int index = NT * i + tid; - reg[i] = a_global[index + ((index >= aCount) ? b0 : 0)]; - } - } else { - #pragma unroll - for(int i = 0; i < VT0; ++i) { - int index = NT * i + tid; - if(index < total) - reg[i] = a_global[index + ((index >= aCount) ? b0 : 0)]; - } - } - #pragma unroll - for(int i = VT0; i < VT1; ++i) { - int index = NT * i + tid; - if(index < total) - reg[i] = a_global[index + ((index >= aCount) ? b0 : 0)]; - } -} - -template -MGPU_DEVICE void DeviceLoad2ToShared(const T* a_global, int aCount, - const T* b_global, int bCount, int tid, T* shared, bool sync) { - - T reg[VT1]; - DeviceLoad2ToReg(a_global, aCount, b_global, bCount, tid, - reg, false); - DeviceRegToShared(reg, tid, shared, sync); -} - -//////////////////////////////////////////////////////////////////////////////// -// DeviceLoad2 - load from pointers of different types. Uses two LD statements. - -template -MGPU_DEVICE void DeviceLoad2ToReg(InputIt1 a_global, int aCount, - InputIt2 b_global, int bCount, int tid, T* reg, bool sync) { - - b_global -= aCount; - int total = aCount + bCount; - if(total >= NT * VT0) { - #pragma unroll - for(int i = 0; i < VT0; ++i) { - int index = NT * i + tid; - if(index < aCount) reg[i] = a_global[index]; - else reg[i] = b_global[index]; - } - } else { - #pragma unroll - for(int i = 0; i < VT0; ++i) { - int index = NT * i + tid; - if(index < aCount) reg[i] = a_global[index]; - else if(index < total) reg[i] = b_global[index]; - } - } - #pragma unroll - for(int i = VT0; i < VT1; ++i) { - int index = NT * i + tid; - if(index < aCount) reg[i] = a_global[index]; - else if(index < total) reg[i] = b_global[index]; - } -} - -template -MGPU_DEVICE void DeviceLoad2ToShared(InputIt1 a_global, int aCount, - InputIt2 b_global, int bCount, int tid, T* shared, bool sync) { - - T reg[VT1]; - DeviceLoad2ToReg(a_global, aCount, b_global, bCount, tid, - reg, false); - DeviceRegToShared(reg, tid, shared, sync); -} - - -//////////////////////////////////////////////////////////////////////////////// -// DeviceGatherGlobalToGlobal - -template -MGPU_DEVICE void DeviceGatherGlobalToGlobal(int count, InputIt data_global, - const int* indices_shared, int tid, OutputIt dest_global, bool sync) { - - typedef typename std::iterator_traits::value_type ValType; - ValType values[VT]; - - #pragma unroll - for(int i = 0; i < VT; ++i) { - int index = NT * i + tid; - if(index < count) { - int gather = indices_shared[index]; - values[i] = data_global[gather]; - } - } - if(sync) __syncthreads(); - DeviceRegToGlobal(count, values, tid, dest_global, false); -} - -//////////////////////////////////////////////////////////////////////////////// -// DeviceTransferMergeValues -// Gather in a merge-like value from two input arrays and store to a single -// output. Like DeviceGatherGlobalToGlobal, but for two arrays at once. - -template -MGPU_DEVICE void DeviceTransferMergeValuesReg(int count, InputIt1 a_global, - InputIt2 b_global, int bStart, const int* indices, int tid, - T* reg, bool sync) { - - b_global -= bStart; - if(count >= NT * VT) { - #pragma unroll - for(int i = 0; i < VT; ++i) { - reg[i] = (indices[i] < bStart) ? a_global[indices[i]] : - b_global[indices[i]]; - } - } else { - #pragma unroll - for(int i = 0; i < VT; ++i) { - int index = NT * i + tid; - if(index < count) - reg[i] = (indices[i] < bStart) ? a_global[indices[i]] : - b_global[indices[i]]; - } - } - if(sync) __syncthreads(); -} - -template -MGPU_DEVICE void DeviceTransferMergeValuesShared(int count, InputIt1 a_global, - InputIt2 b_global, int bStart, const int* indices_shared, int tid, - OutputIt dest_global, bool sync) { - - int indices[VT]; - DeviceSharedToReg(indices_shared, tid, indices); - - typedef typename std::iterator_traits::value_type ValType; - ValType reg[VT]; - DeviceTransferMergeValuesReg(count, a_global, b_global, bStart, - indices, tid, reg, sync); - DeviceRegToGlobal(count, reg, tid, dest_global, sync); -} - -template -MGPU_DEVICE void DeviceTransferMergeValuesReg(int count, const T* a_global, - const T* b_global, int bStart, const int* indices, int tid, T* reg, - bool sync) { - - int bOffset = (int)(b_global - a_global - bStart); - - if(count >= NT * VT) { - #pragma unroll - for(int i = 0; i < VT; ++i) { - int gather = indices[i]; - if(gather >= bStart) gather += bOffset; - reg[i] = a_global[gather]; - } - } else { - #pragma unroll - for(int i = 0; i < VT; ++i) { - int index = NT * i + tid; - int gather = indices[i]; - if(gather >= bStart) gather += bOffset; - if(index < count) - reg[i] = a_global[gather]; - } - } - if(sync) __syncthreads(); -} - -template -MGPU_DEVICE void DeviceTransferMergeValuesShared(int count, const T* a_global, - const T* b_global, int bStart, const int* indices_shared, int tid, - OutputIt dest_global, bool sync) { - - int indices[VT]; - DeviceSharedToReg(indices_shared, tid, indices); - - T reg[VT]; - DeviceTransferMergeValuesReg(count, a_global, b_global, bStart, - indices, tid, reg, sync); - DeviceRegToGlobal(count, reg, tid, dest_global, sync); -} - -} // namespace mgpu diff --git a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/serialsets.cuh b/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/serialsets.cuh deleted file mode 100644 index d0e7a3a81722..000000000000 --- a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/serialsets.cuh +++ /dev/null @@ -1,235 +0,0 @@ -/****************************************************************************** - * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions are met: - * * Redistributions of source code must retain the above copyright - * notice, this list of conditions and the following disclaimer. - * * Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * * Neither the name of the NVIDIA CORPORATION nor the - * names of its contributors may be used to endorse or promote products - * derived from this software without specific prior written permission. - * - * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" - * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE - * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE - * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY - * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES - * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; - * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND - * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS - * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - * - ******************************************************************************/ - -/****************************************************************************** - * - * Code and text by Sean Baxter, NVIDIA Research - * See http://nvlabs.github.io/moderngpu for repository and documentation. - * - ******************************************************************************/ - -#pragma once - -#include "deviceutil.cuh" - -namespace mgpu { - -//////////////////////////////////////////////////////////////////////////////// -// SerialSetIntersection -// Emit A if A and B are in range and equal. - -template -MGPU_DEVICE int SerialSetIntersection(const T* data, int aBegin, int aEnd, - int bBegin, int bEnd, int end, T* results, int* indices, Comp comp) { - - const int MinIterations = VT / 2; - int commit = 0; - - #pragma unroll - for(int i = 0; i < VT; ++i) { - bool test = RangeCheck ? - ((aBegin + bBegin < end) && (aBegin < aEnd) && (bBegin < bEnd)) : - (i < MinIterations || (aBegin + bBegin < end)); - - if(test) { - T aKey = data[aBegin]; - T bKey = data[bBegin]; - - bool pA = comp(aKey, bKey); - bool pB = comp(bKey, aKey); - - // The outputs must come from A by definition of set interection. - results[i] = aKey; - indices[i] = aBegin; - - if(!pB) ++aBegin; - if(!pA) ++bBegin; - if(pA == pB) commit |= 1<< i; - } - } - return commit; -} - -//////////////////////////////////////////////////////////////////////////////// -// SerialSetUnion -// Emit A if A <= B. Emit B if B < A. - -template -MGPU_DEVICE int SerialSetUnion(const T* data, int aBegin, int aEnd, - int bBegin, int bEnd, int end, T* results, int* indices, Comp comp) { - - const int MinIterations = VT / 2; - int commit = 0; - - #pragma unroll - for(int i = 0; i < VT; ++i) { - bool test = RangeCheck ? - (aBegin + bBegin < end) : - (i < MinIterations || (aBegin + bBegin < end)); - - if(test) { - T aKey = data[aBegin]; - T bKey = data[bBegin]; - - bool pA = false, pB = false; - if(RangeCheck && aBegin >= aEnd) - pB = true; - else if(RangeCheck && bBegin >= bEnd) - pA = true; - else { - // Both are in range. - pA = comp(aKey, bKey); - pB = comp(bKey, aKey); - } - - // Output A in case of a tie, so check if b < a. - results[i] = pB ? bKey : aKey; - indices[i] = pB ? bBegin : aBegin; - if(!pB) ++aBegin; - if(!pA) ++bBegin; - commit |= 1<< i; - } - } - return commit; -} - -//////////////////////////////////////////////////////////////////////////////// -// SerialSetDifference -// Emit A if A < B. - -template -MGPU_DEVICE int SerialSetDifference(const T* data, int aBegin, int aEnd, - int bBegin, int bEnd, int end, T* results, int* indices, Comp comp) { - - const int MinIterations = VT / 2; - int commit = 0; - - #pragma unroll - for(int i = 0; i < VT; ++i) { - bool test = RangeCheck ? - (aBegin + bBegin < end) : - (i < MinIterations || (aBegin + bBegin < end)); - if(test) { - T aKey = data[aBegin]; - T bKey = data[bBegin]; - - bool pA = false, pB = false; - if(RangeCheck && aBegin >= aEnd) - pB = true; - else if(RangeCheck && bBegin >= bEnd) - pA = true; - else { - pA = comp(aKey, bKey); - pB = comp(bKey, aKey); - } - - // The outputs must come from A by definition of set difference. - results[i] = aKey; - indices[i] = aBegin; - if(!pB) ++aBegin; - if(!pA) ++bBegin; - if(pA) commit |= 1<< i; - } - } - return commit; -} - -//////////////////////////////////////////////////////////////////////////////// -// SerialSetSymDiff -// Emit A if A < B and emit B if B < A. - -template -MGPU_DEVICE int SerialSetSymDiff(const T* data, int aBegin, int aEnd, - int bBegin, int bEnd, int end, T* results, int* indices, Comp comp) { - - const int MinIterations = VT / 2; - int commit = 0; - - #pragma unroll - for(int i = 0; i < VT; ++i) { - bool test = RangeCheck ? - (aBegin + bBegin < end) : - (i < MinIterations || (aBegin + bBegin < end)); - if(test) { - T aKey = data[aBegin]; - T bKey = data[bBegin]; - - bool pA = false, pB = false; - if(RangeCheck && (bBegin >= bEnd)) - pA = true; - else if(RangeCheck && (aBegin >= aEnd)) - pB = true; - else { - pA = comp(aKey, bKey); - pB = comp(bKey, aKey); - } - - results[i] = pA ? aKey : bKey; - indices[i] = pA ? aBegin : bBegin; - if(!pA) ++bBegin; - if(!pB) ++aBegin; - if(pA != pB) commit |= 1<< i; - } - } - return commit; -} - -//////////////////////////////////////////////////////////////////////////////// -// SerialSetOp -// Uses the MgpuSetOp enum to statically select one of the four serial ops -// above. - -template -MGPU_DEVICE int SerialSetOp(const T* data, int aBegin, int aEnd, - int bBegin, int bEnd, int star, T* results, int* indices, Comp comp) { - - int end = aBegin + bBegin + VT - star; - if(RangeCheck) end = min(end, aEnd + bEnd); - int commit; - switch(Op) { - case MgpuSetOpIntersection: - commit = SerialSetIntersection(data, aBegin, - aEnd, bBegin, bEnd, end, results, indices, comp); - break; - case MgpuSetOpUnion: - commit = SerialSetUnion(data, aBegin, aEnd, - bBegin, bEnd, end, results, indices, comp); - break; - case MgpuSetOpDiff: - commit = SerialSetDifference(data, aBegin, aEnd, - bBegin, bEnd, end, results, indices, comp); - break; - case MgpuSetOpSymDiff: - commit = SerialSetSymDiff(data, aBegin, aEnd, - bBegin, bEnd, end, results, indices, comp); - break; - } - __syncthreads(); - return commit; -} - -} // namespace mgpu diff --git a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/sortnetwork.cuh b/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/sortnetwork.cuh deleted file mode 100644 index 94c88f71d93d..000000000000 --- a/src/operator/contrib/ctc_include/contrib/moderngpu/include/device/sortnetwork.cuh +++ /dev/null @@ -1,168 +0,0 @@ -/****************************************************************************** - * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions are met: - * * Redistributions of source code must retain the above copyright - * notice, this list of conditions and the following disclaimer. - * * Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * * Neither the name of the NVIDIA CORPORATION nor the - * names of its contributors may be used to endorse or promote products - * derived from this software without specific prior written permission. - * - * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" - * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE - * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE - * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY - * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES - * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; - * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND - * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS - * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - * - ******************************************************************************/ - -/****************************************************************************** - * - * Code and text by Sean Baxter, NVIDIA Research - * See http://nvlabs.github.io/moderngpu for repository and documentation. - * - ******************************************************************************/ - -#pragma once - -#include "deviceutil.cuh" - -namespace mgpu { - -//////////////////////////////////////////////////////////////////////////////// -// Odd-even transposition sorting network. Sorts keys and values in-place in -// register. -// http://en.wikipedia.org/wiki/Odd%E2%80%93even_sort - -// CUDA Compiler does not currently unroll these loops correctly. Write using -// template loop unrolling. -/* -template -MGPU_DEVICE void OddEvenTransposeSort(T* keys, V* values, Comp comp) { - #pragma unroll - for(int level = 0; level < VT; ++level) { - - #pragma unroll - for(int i = 1 & level; i < VT - 1; i += 2) { - if(comp(keys[i + 1], keys[i])) { - mgpu::swap(keys[i], keys[i + 1]); - mgpu::swap(values[i], values[i + 1]); - } - } - } -}*/ - -template -struct OddEvenTransposeSortT { - // Sort segments marked by head flags. If the head flag between i and i + 1 - // is set (so that (2<< i) & flags is true), the values belong to different - // segments and are not swapped. - template - static MGPU_DEVICE void Sort(K* keys, V* values, int flags, Comp comp) { - #pragma unroll - for(int i = 1 & I; i < VT - 1; i += 2) - if((0 == ((2<< i) & flags)) && comp(keys[i + 1], keys[i])) { - mgpu::swap(keys[i], keys[i + 1]); - mgpu::swap(values[i], values[i + 1]); - } - OddEvenTransposeSortT::Sort(keys, values, flags, comp); - } -}; -template struct OddEvenTransposeSortT { - template - static MGPU_DEVICE void Sort(K* keys, V* values, int flags, Comp comp) { } -}; - -template -MGPU_DEVICE void OddEvenTransposeSort(K* keys, V* values, Comp comp) { - OddEvenTransposeSortT<0, VT>::Sort(keys, values, 0, comp); -} -template -MGPU_DEVICE void OddEvenTransposeSortFlags(K* keys, V* values, int flags, - Comp comp) { - OddEvenTransposeSortT<0, VT>::Sort(keys, values, flags, comp); -} - -//////////////////////////////////////////////////////////////////////////////// -// Batcher Odd-Even Mergesort network -// Unstable but executes much faster than the transposition sort. -// http://en.wikipedia.org/wiki/Batcher_odd%E2%80%93even_mergesort - -template -struct OddEvenMergesortT { - template - MGPU_DEVICE static void CompareAndSwap(K* keys, V* values, int flags, - int a, int b, Comp comp) { - if(b < Count) { - // Mask the bits between a and b. Any head flags in this interval - // means the keys are in different segments and must not be swapped. - const int Mask = ((2<< b) - 1) ^ ((2<< a) - 1); - if(!(Mask & flags) && comp(keys[b], keys[a])) { - mgpu::swap(keys[b], keys[a]); - mgpu::swap(values[b], values[a]); - } - } - } - - template - struct OddEvenMerge { - template - MGPU_DEVICE static void Merge(K* keys, V* values, int flags, - Comp comp) { - // Compare and swap - const int M = 2 * R; - OddEvenMerge::Merge(keys, values, flags, comp); - OddEvenMerge::Merge(keys, values, flags, comp); - - #pragma unroll - for(int i = Low2 + R; i + R < Low2 + Width; i += M) - CompareAndSwap(keys, values, flags, i, i + R, comp); - } - }; - template - struct OddEvenMerge { - template - MGPU_DEVICE static void Merge(K* keys, V* values, int flags, - Comp comp) { - CompareAndSwap(keys, values, flags, Low2, Low2 + R, comp); - } - }; - - template - MGPU_DEVICE static void Sort(K* keys, V* values, int flags, - Comp comp) { - - const int M = Width / 2; - OddEvenMergesortT::Sort(keys, values, flags, comp); - OddEvenMergesortT::Sort(keys, values, flags, comp); - OddEvenMerge<1, Low>::Merge(keys, values, flags, comp); - } -}; -template struct OddEvenMergesortT<1, Low, Count> { - template - MGPU_DEVICE static void Sort(K* keys, V* values, int flags, - Comp comp) { } -}; - -template -MGPU_DEVICE void OddEvenMergesort(K* keys, V* values, Comp comp) { - const int Width = 1<< sLogPow2::value; - OddEvenMergesortT::Sort(keys, values, 0, comp); -} -template -MGPU_DEVICE void OddEvenMergesortFlags(K* keys, V* values, int flags, - Comp comp) { - const int Width = 1<< sLogPow2::value; - OddEvenMergesortT::Sort(keys, values, flags, comp); -} - -} // namespace mgpu diff --git a/src/operator/contrib/ctc_include/contrib/moderngpu/include/mgpudevice.cuh b/src/operator/contrib/ctc_include/contrib/moderngpu/include/mgpudevice.cuh deleted file mode 100644 index ad8742d46db6..000000000000 --- a/src/operator/contrib/ctc_include/contrib/moderngpu/include/mgpudevice.cuh +++ /dev/null @@ -1,289 +0,0 @@ -/****************************************************************************** - * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions are met: - * * Redistributions of source code must retain the above copyright - * notice, this list of conditions and the following disclaimer. - * * Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * * Neither the name of the NVIDIA CORPORATION nor the - * names of its contributors may be used to endorse or promote products - * derived from this software without specific prior written permission. - * - * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" - * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE - * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE - * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY - * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES - * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; - * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND - * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS - * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - * - ******************************************************************************/ - -/****************************************************************************** - * - * Code and text by Sean Baxter, NVIDIA Research - * See http://nvlabs.github.io/moderngpu for repository and documentation. - * - ******************************************************************************/ - -#pragma once - -#include "mgpuenums.h" -#include "device/deviceutil.cuh" - -namespace mgpu { - -//////////////////////////////////////////////////////////////////////////////// -// device/loadstore.cuh - -// For 0 <= i < VT: -// index = NT * i + tid; -// reg[i] = data[index]; -// Synchronize after load. -template -MGPU_DEVICE void DeviceSharedToReg(InputIt data, int tid, T* reg, - bool sync = true); - -// For 0 <= i < VT: -// index = NT * i + tid; -// if(index < count) reg[i] = data[index]; -// No synchronize after load. -template -MGPU_DEVICE void DeviceGlobalToReg(int count, InputIt data, int tid, - T* reg, bool sync = false); - -template -MGPU_DEVICE void DeviceGlobalToRegDefault(int count, InputIt data, int tid, - T* reg, T init, bool sync = false); - -// For 0 <= i < VT: -// index = NT * i + tid; -// if(index < count) reg[i] = data[index]; -// No synchronize after load. -template -MGPU_DEVICE void DeviceGlobalToReg(int count, InputIt data, int tid, - T* reg, bool sync = false); - -// For 0 <= i < VT: -// index = NT * i + tid; -// if(index < count) reg[i] = data[index]; -// No synchronize after load. -template -MGPU_DEVICE void DeviceGlobalToRegDefault2(int count, InputIt data, int tid, - T* reg, T init, bool sync = false); - -// For 0 <= i < VT: -// index = NT * i + tid; -// if(index < count) reg[i] = data[index]; -// No synchronize after load. -// No optimized code path for count < NV (smaller generated code). -template -MGPU_DEVICE void DeviceGlobalToRegLoop(int count, InputIt data, int tid, - T* reg, bool sync = false); - - -// For 0 <= i < VT: -// index = VT * tid + i. -// if(index < count) reg[i] = data[index]; -// No synchronize after load. -template -MGPU_DEVICE void DeviceGlobalToThread(int count, InputIt data, int tid, - T* reg); - -template -MGPU_DEVICE void DeviceGlobalToThreadDefault(int count, InputIt data, int tid, - T* reg, T init); - -// For 0 <= i < VT: -// index = NT * i + tid; -// if(index < count) data[index] = reg[i]; -// Synchronize after load. -template -MGPU_DEVICE void DeviceRegToShared(const T* reg, int tid, OutputIt dest, - bool sync = true); - -// For 0 <= i < VT: -// index = NT * i + tid; -// if(index < count) data[index] = reg[i]; -// No synchronize after load. -template -MGPU_DEVICE void DeviceRegToGlobal(int count, const T* reg, int tid, - OutputIt dest, bool sync = false); - -// For 0 <= index < count: -// dest[index] = source[index]; -// This function is intended to replace DeviceGlobalToShared in cases where -// count is much less than NT * VT. -template -MGPU_DEVICE void DeviceMemToMemLoop(int count, InputIt source, int tid, - OutputIt dest, bool sync = true); - -// For 0 <= index < count: -// dest[index] = source[index]; -// Synchronize after store. -template -MGPU_DEVICE void DeviceSharedToGlobal(int count, const T* source, int tid, - OutputIt dest, bool sync = true); - -// For 0 <= index < count: -// dest[index] = source[index]; -// Synchronize after store. -template -MGPU_DEVICE void DeviceGlobalToShared(int count, InputIt source, int tid, - T* dest, bool sync = true); - -template -MGPU_DEVICE void DeviceGlobalToShared2(int count, InputIt source, int tid, - T* dest, bool sync = true); - -// For 0 <= index < count: -// dest[index] = source[index]; -// Synchronize after store. -// No optimized code path for count < NV (smaller generated code). -template -MGPU_DEVICE void DeviceGlobalToSharedLoop(int count, InputIt source, int tid, - T* dest, bool sync = true); - -template -MGPU_DEVICE void DeviceGlobalToSharedDefault(int count, InputIt source, int tid, - T* dest, T init, bool sync = true); - -template -MGPU_DEVICE void DeviceGlobalToSharedDefault2(int count, InputIt source, - int tid, T* dest, T init, bool sync = true); - -// For 0 <= index < count: -// dest[index] = source[index]; -// No synchronize. -template -MGPU_DEVICE void DeviceGlobalToGlobal(int count, InputIt source, int tid, - OutputIt dest, bool sync = false); - -// Transponse VT elements in NT threads (x) into thread-order registers (y) -// using only NT * VT / 2 elements of shared memory. -template -MGPU_DEVICE void HalfSmemTranspose(const T* x, int tid, T* shared, T* y); - -// For 0 <= i < VT: -// index = NT * i + tid; -// if(index < count) -// gather = indices[index]; -// reg[i] = data[gather]; -// Synchronize after load. -template -MGPU_DEVICE void DeviceGather(int count, InputIt data, int indices[VT], - int tid, T* reg, bool sync = true); - -template -MGPU_DEVICE void DeviceGatherDefault(int count, InputIt data, int indices[VT], - int tid, T* reg, T identity, bool sync = true); - -// For 0 <= i < VT: -// index = NT * i + tid; -// if(index < count) -// scatter = indices[index]; -// data[scatter] = reg[i]; -// Synchronize after store. -template -MGPU_DEVICE void DeviceScatter(int count, const T* reg, int tid, - int indices[VT], OutputIt data, bool sync = true); - -// For 0 <= i < VT: -// shared[VT * tid + i] = threadReg[i]; -// Synchronize after store. -// Note this function moves data in THREAD ORDER. -// (DeviceRegToShared moves data in STRIDED ORDER). -template -MGPU_DEVICE void DeviceThreadToShared(const T* threadReg, int tid, T* shared, - bool sync = true); - -// For 0 <= i < VT: -// threadReg[i] = shared[VT * tid + i]; -// Synchronize after load. -// Note this function moves data in THREAD ORDER. -// (DeviceSharedToReg moves data in STRIDED ORDER). -template -MGPU_DEVICE void DeviceSharedToThread(const T* shared, int tid, T* threadReg, - bool sync = true); - -// For 0 <= index < aCount: -// shared[index] = a_global[index]; -// For 0 <= index < bCount: -// shared[aCount + index] = b_global[index]; -// VT0 is the lower-bound for predication-free execution: -// If count >= NT * VT0, a predication-free branch is taken. -// VT1 is the upper-bound for loads: -// NT * VT1 must >= aCount + bCount. - -template -MGPU_DEVICE void DeviceLoad2ToReg(const T* a_global, int aCount, - const T* b_global, int bCount, int tid, T* reg, bool sync = false); - -template -MGPU_DEVICE void DeviceLoad2ToShared(const T* a_global, int aCount, - const T* b_global, int bCount, int tid, T* shared, bool sync = true); - -template -MGPU_DEVICE void DeviceLoad2ToReg(InputIt1 a_global, int aCount, - InputIt2 b_global, int bCount, int tid, T* reg, bool sync = false); - -template -MGPU_DEVICE void DeviceLoad2ToShared(InputIt1 a_global, int aCount, - InputIt2 b_global, int bCount, int tid, T* shared, bool sync = true); - -// For 0 <= i < VT -// index = NT * i + tid; -// if(index < count) -// gather = indices_shared[index]; -// dest_global[index] = data_global[gather]; -// Synchronize after load. -template -MGPU_DEVICE void DeviceGatherGlobalToGlobal(int count, InputIt data_global, - const int* indices_shared, int tid, OutputIt dest_global, - bool sync = true); - -// For 0 <= i < VT -// index = NT * i + tid -// if(index < count) -// gather = indices[index]; -// if(gather < aCount) data = a_global[gather]; -// else data = b_global[gather - aCount]; -// dest_global[index] = data; -// Synchronize after load. -template -MGPU_DEVICE void DeviceTransferMergeValuesReg(int count, InputIt1 a_global, - InputIt2 b_global, int bStart, const int* indices, int tid, - T* reg, bool sync = false); - -template -MGPU_DEVICE void DeviceTransferMergeValuesShared(int count, InputIt1 a_global, - InputIt2 b_global, int bStart, const int* indices_shared, int tid, - OutputIt dest_global, bool sync = true); - -template -MGPU_DEVICE void DeviceTransferMergeValuesReg(int count, const T* a_global, - const T* b_global, int bStart, const int* indices, int tid, - T* reg, bool sync = false); - -template -MGPU_DEVICE void DeviceTransferMergeValuesShared(int count, const T* a_global, - const T* b_global, int bStart, const int* indices_shared, int tid, - OutputIt dest_global, bool sync = true); - - - -} // namespace mgpu - - -#include "device/loadstore.cuh" -#include "device/ctasegscan.cuh" diff --git a/src/operator/contrib/ctc_include/contrib/moderngpu/include/mgpuenums.h b/src/operator/contrib/ctc_include/contrib/moderngpu/include/mgpuenums.h deleted file mode 100644 index be2b8314a8ad..000000000000 --- a/src/operator/contrib/ctc_include/contrib/moderngpu/include/mgpuenums.h +++ /dev/null @@ -1,70 +0,0 @@ -/****************************************************************************** - * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions are met: - * * Redistributions of source code must retain the above copyright - * notice, this list of conditions and the following disclaimer. - * * Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * * Neither the name of the NVIDIA CORPORATION nor the - * names of its contributors may be used to endorse or promote products - * derived from this software without specific prior written permission. - * - * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" - * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE - * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE - * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY - * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES - * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; - * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND - * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS - * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - * - ******************************************************************************/ - -/****************************************************************************** - * - * Code and text by Sean Baxter, NVIDIA Research - * See http://nvlabs.github.io/moderngpu for repository and documentation. - * - ******************************************************************************/ - -#pragma once - -namespace mgpu { - -enum MgpuBounds { - MgpuBoundsLower, - MgpuBoundsUpper -}; - -enum MgpuScanType { - MgpuScanTypeExc, - MgpuScanTypeInc -}; - -enum MgpuSearchType { - MgpuSearchTypeNone, - MgpuSearchTypeIndex, - MgpuSearchTypeMatch, - MgpuSearchTypeIndexMatch -}; - -enum MgpuJoinKind { - MgpuJoinKindInner, - MgpuJoinKindLeft, - MgpuJoinKindRight, - MgpuJoinKindOuter -}; - -enum MgpuSetOp { - MgpuSetOpIntersection, - MgpuSetOpUnion, - MgpuSetOpDiff, - MgpuSetOpSymDiff -}; - -} // namespace mgpu diff --git a/src/operator/contrib/ctc_include/contrib/moderngpu/include/util/static.h b/src/operator/contrib/ctc_include/contrib/moderngpu/include/util/static.h deleted file mode 100644 index c7209077506a..000000000000 --- a/src/operator/contrib/ctc_include/contrib/moderngpu/include/util/static.h +++ /dev/null @@ -1,183 +0,0 @@ -/****************************************************************************** - * Copyright (c) 2013, NVIDIA CORPORATION. All rights reserved. - * - * Redistribution and use in source and binary forms, with or without - * modification, are permitted provided that the following conditions are met: - * * Redistributions of source code must retain the above copyright - * notice, this list of conditions and the following disclaimer. - * * Redistributions in binary form must reproduce the above copyright - * notice, this list of conditions and the following disclaimer in the - * documentation and/or other materials provided with the distribution. - * * Neither the name of the NVIDIA CORPORATION nor the - * names of its contributors may be used to endorse or promote products - * derived from this software without specific prior written permission. - * - * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" - * AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE - * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE - * ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE FOR ANY - * DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES - * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; - * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND - * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT - * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS - * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - * - ******************************************************************************/ - -/****************************************************************************** - * - * Code and text by Sean Baxter, NVIDIA Research - * See http://nvlabs.github.io/moderngpu for repository and documentation. - * - ******************************************************************************/ - -#pragma once - -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include - -#ifndef MGPU_MIN -#define MGPU_MIN(x, y) (((x) <= (y)) ? (x) : (y)) -#define MGPU_MAX(x, y) (((x) >= (y)) ? (x) : (y)) -#define MGPU_MAX0(x) (((x) >= 0) ? (x) : 0) -#define MGPU_ABS(x) (((x) >= 0) ? (x) : (-x)) - -#define MGPU_DIV_UP(x, y) (((x) + (y) - 1) / (y)) -#define MGPU_DIV_ROUND(x, y) (((x) + (y) / 2) / (y)) -#define MGPU_ROUND_UP(x, y) ((y) * MGPU_DIV_UP(x, y)) -#define MGPU_SHIFT_DIV_UP(x, y) (((x) + ((1<< (y)) - 1))>> y) -#define MGPU_ROUND_UP_POW2(x, y) (((x) + (y) - 1) & ~((y) - 1)) -#define MGPU_ROUND_DOWN_POW2(x, y) ((x) & ~((y) - 1)) -#define MGPU_IS_POW_2(x) (0 == ((x) & ((x) - 1))) - -#endif // MGPU_MIN - -namespace mgpu { - - -typedef unsigned char byte; - -typedef unsigned int uint; -typedef signed short int16; - -typedef unsigned short ushort; -typedef unsigned short uint16; - -typedef long long int64; -typedef unsigned long long uint64; - -// IsPow2::value is true if X is a power of 2. -template struct sIsPow2 { - enum { value = 0 == (X & (X - 1)) }; -}; - -// Finds the base-2 logarithm of X. value is -1 if X is not a power of 2. -template struct sLogPow2 { - enum { extra = sIsPow2::value ? 0 : (roundUp ? 1 : 0) }; - enum { inner = sLogPow2::inner + 1 }; - enum { value = inner + extra }; -}; -template struct sLogPow2<0, roundUp> { - enum { inner = 0 }; - enum { value = 0 }; -}; -template struct sLogPow2<1, roundUp> { - enum { inner = 0 }; - enum { value = 0 }; -}; - -template -struct sDivUp { - enum { value = (X + Y - 1) / Y }; -}; - -template struct sDiv2RoundUp { - enum { value = sDiv2RoundUp::value, levels - 1>::value }; -}; -template struct sDiv2RoundUp { - enum { value = count }; -}; - -template -struct sDivSafe { - enum { value = X / Y }; -}; -template -struct sDivSafe { - enum { value = 0 }; -}; - -template -struct sRoundUp { - enum { rem = X % Y }; - enum { value = X + (rem ? (Y - rem) : 0) }; -}; - -template -struct sRoundDown { - enum { rem = X % Y }; - enum { value = X - rem }; -}; - -// IntegerDiv is a template for avoiding divisions by zero in template -// evaluation. Templates always evaluate both b and c in an expression like -// a ? b : c, and will error if either rhs contains an illegal expression, -// even if the ternary is explictly designed to guard against that. -template -struct sIntegerDiv { - enum { value = X / (Y ? Y : (X + 1)) }; -}; - -template -struct sMax { - enum { value = (X >= Y) ? X : Y }; -}; -template -struct sMin { - enum { value = (X <= Y) ? X : Y }; -}; - -template -struct sAbs { - enum { value = (X >= 0) ? X : -X }; -}; - - -// Finds the number of powers of 2 in the prime factorization of X. -template struct sNumFactorsOf2 { - enum { shifted = X >> 1 }; - enum { value = 1 + sNumFactorsOf2::value }; -}; -template struct sNumFactorsOf2 { - enum { value = 0 }; -}; - -// Returns the divisor for a conflict-free transpose. -template struct sBankConflictDivisor { - enum { value = - (1 & X) ? 0 : - (sIsPow2::value ? NumBanks : - (1<< sNumFactorsOf2::value)) }; - enum { log_value = sLogPow2::value }; -}; - -template struct sConflictFreeStorage { - enum { count = NT * X }; - enum { divisor = sBankConflictDivisor::value }; - enum { padding = sDivSafe::value }; - enum { value = count + padding }; -}; - -} // namespace mgpu diff --git a/src/operator/contrib/ctc_include/detail/cpu_ctc.h b/src/operator/contrib/ctc_include/detail/cpu_ctc.h deleted file mode 100644 index 005b956343d4..000000000000 --- a/src/operator/contrib/ctc_include/detail/cpu_ctc.h +++ /dev/null @@ -1,509 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, - * software distributed under the License is distributed on an - * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY - * KIND, either express or implied. See the License for the - * specific language governing permissions and limitations - * under the License. - */ - -#pragma once - -#include -#include -#include -#include -#include - -#include - -#include "ctc_helper.h" - -namespace mxnet_warpctc { - -template -class CpuCTC { -public: - // Noncopyable - CpuCTC(int alphabet_size, int minibatch, void* workspace, - int blank_label) : - alphabet_size_(alphabet_size), minibatch_(minibatch), - workspace_(workspace), blank_label_(blank_label) { - - }; - - CpuCTC(const CpuCTC&) = delete; - CpuCTC& operator=(const CpuCTC&) = delete; - - ctcStatus_t cost_and_grad(const ProbT* const activations, - ProbT *grads, - ProbT* costs, - const int* const flat_labels, - const int* const label_lengths, - const int* const input_lengths); - - - ctcStatus_t score_forward(const ProbT* const activations, - ProbT* costs, - const int* const flat_labels, - const int* const label_lengths, - const int* const input_lengths); - -private: - - class CpuCTC_metadata { - - private: - int setup_labels(const int* const labels, int blank_label, int L, int S); - - public: - CpuCTC_metadata(int L, int S, int T, int mb, int alphabet_size, - void* workspace, size_t bytes_used, int blank_label, - const int* const labels); - - ProbT* alphas; - ProbT* betas; - int* labels_w_blanks; - int* e_inc; - int* s_inc; - ProbT* output; - int repeats; - }; - - int alphabet_size_; // Number of characters plus blank - int minibatch_; - void* workspace_; - int blank_label_; - - void log_softmax(const ProbT* const activations, ProbT* log_probs, - const int* const input_lengths); - - std::tuple - cost_and_grad_kernel(ProbT *grad, const ProbT* const log_probs, - const int* const labels, int T, int L, - int mb, size_t bytes_used); - - ProbT compute_alphas(const ProbT* log_probs, int repeats, int S, int T, - const int* const e_inc, - const int* const s_inc, - const int* const labels, - ProbT* alphas); - - ProbT compute_betas_and_grad(ProbT* grad, const ProbT* const log_probs, - ProbT log_partition, int repeats, - int S, int T, const int* const e_inc, - const int* const s_inc, - const int* const labels, - ProbT* alphas, - ProbT* betas, - ProbT* output); -}; - -template -CpuCTC::CpuCTC_metadata::CpuCTC_metadata(int L, int S, int T, int mb, - int alphabet_size, - void* workspace, size_t bytes_used, - int blank_label, - const int* const labels) { - - alphas = reinterpret_cast(static_cast(workspace) + bytes_used); - bytes_used += sizeof(ProbT) * S * T; - std::fill(alphas, alphas + S * T, ctc_helper::neg_inf()); - betas = reinterpret_cast(static_cast(workspace) + bytes_used); - bytes_used += sizeof(ProbT) * S; - std::fill(betas, betas + S, ctc_helper::neg_inf()); - labels_w_blanks = reinterpret_cast(static_cast(workspace) + bytes_used); - bytes_used += sizeof(int) * S; - e_inc = reinterpret_cast(static_cast(workspace) + bytes_used); - bytes_used += sizeof(int) * S; - s_inc = reinterpret_cast(static_cast(workspace) + bytes_used); - bytes_used += sizeof(int) * S; - output = reinterpret_cast(static_cast(workspace) + bytes_used); - bytes_used += sizeof(ProbT) * alphabet_size; - - repeats = setup_labels(labels, blank_label, L, S); -} - -template -int CpuCTC::CpuCTC_metadata::setup_labels(const int* const labels, - int blank_label, int L, int S) { - int e_counter = 0; - int s_counter = 0; - - s_inc[s_counter++] = 1; - - int repeats = 0; - - for (int i = 1; i < L; ++i) { - if (labels[i-1] == labels[i]) { - s_inc[s_counter++] = 1; - s_inc[s_counter++] = 1; - e_inc[e_counter++] = 1; - e_inc[e_counter++] = 1; - ++repeats; - } - else { - s_inc[s_counter++] = 2; - e_inc[e_counter++] = 2; - } - } - e_inc[e_counter++] = 1; - - for (int i = 0; i < L; ++i) { - labels_w_blanks[2 * i] = blank_label; - labels_w_blanks[2 * i + 1] = labels[i]; - } - labels_w_blanks[S - 1] = blank_label; - - return repeats; -} - -template -void -CpuCTC::log_softmax(const ProbT* const activations, ProbT* log_probs, - const int* const input_lengths) { -#pragma omp parallel for - for (int mb = 0; mb < minibatch_; ++mb) { - for(int c = 0; c < input_lengths[mb]; ++c) { - int col_offset = (mb + minibatch_ * c) * alphabet_size_; - ProbT max_activation = -std::numeric_limits::infinity(); - for(int r = 0; r < alphabet_size_; ++r) - max_activation = std::max(max_activation, activations[r + col_offset]); - - ProbT denom = ProbT(0.); - for(int r = 0; r < alphabet_size_; ++r) { - denom += std::exp(activations[r + col_offset] - max_activation); - } - - for(int r = 0; r < alphabet_size_; ++r) { - log_probs[r + col_offset] = activations[r + col_offset] - - max_activation - std::log(denom); - } - } - } -} - -template -std::tuple -CpuCTC::cost_and_grad_kernel(ProbT *grad, const ProbT* const log_probs, - const int* const labels, - int T, int L, int mb, size_t bytes_used) { - - const int S = 2*L + 1; // Number of labels with blanks - - CpuCTC_metadata ctcm(L, S, T, mb, alphabet_size_, workspace_, bytes_used, blank_label_, labels); - - bool over_threshold = false; - - if (L + ctcm.repeats > T) { - return std::make_tuple(ProbT(0), over_threshold); // TODO, not right to return 0 - } - - ProbT llForward = compute_alphas(log_probs, ctcm.repeats, S, T, ctcm.e_inc, - ctcm.s_inc, ctcm.labels_w_blanks, - ctcm.alphas); - - ProbT llBackward = compute_betas_and_grad(grad, log_probs, llForward, ctcm.repeats, - S, T, ctcm.e_inc, ctcm.s_inc, - ctcm.labels_w_blanks, - ctcm.alphas, - ctcm.betas, - ctcm.output); - - ProbT diff = std::abs(llForward - llBackward); - if (diff > ctc_helper::threshold) { - over_threshold = true; - } - - return std::make_tuple(-llForward, over_threshold); -} - -// Computes forward probabilities -template -ProbT CpuCTC::compute_alphas(const ProbT* log_probs, int repeats, int S, int T, - const int* const e_inc, - const int* const s_inc, - const int* const labels, - ProbT* alphas) { - - int start = (((S /2) + repeats - T) < 0) ? 0 : 1, - end = S > 1 ? 2 : 1; - - for (int i = start; i < end; ++i) { - alphas[i] = log_probs[labels[i]]; - } - - for(int t = 1; t < T; ++t) { - int remain = (S / 2) + repeats - (T - t); - if(remain >= 0) - start += s_inc[remain]; - if(t <= (S / 2) + repeats) - end += e_inc[t - 1]; - int startloop = start; - int idx1 = t * S, idx2 = (t - 1) * S, idx3 = t * (alphabet_size_ * minibatch_); - - if (start == 0) { - alphas[idx1] = alphas[idx2] + log_probs[blank_label_ + idx3]; - startloop += 1; - } - - for(int i = startloop; i < end; ++i) { - ProbT prev_sum = ctc_helper::log_plus()(alphas[i + idx2], alphas[(i-1) + idx2]); - - // Skip two if not on blank and not on repeat. - if (labels[i] != blank_label_ && i != 1 && labels[i] != labels[i-2]) - prev_sum = ctc_helper::log_plus()(prev_sum, alphas[(i-2) + idx2]); - - alphas[i + idx1] = prev_sum + log_probs[labels[i] + idx3]; - } - } - - ProbT loglike = ctc_helper::neg_inf(); - for(int i = start; i < end; ++i) { - loglike = ctc_helper::log_plus()(loglike, alphas[i + (T - 1) * S]); - } - - return loglike; -} - -// Starting from T, we sweep backward over the alpha array computing one column -// of betas as we go. At each position we can update product alpha * beta and then -// sum into the gradient associated with each label. -// NOTE computes gradient w.r.t UNNORMALIZED final layer activations. -// Assumed passed in grads are already zeroed! -template -ProbT CpuCTC::compute_betas_and_grad(ProbT* grad, const ProbT* const log_probs, - ProbT log_partition, int repeats, - int S, int T, const int* const e_inc, - const int* const s_inc, - const int* const labels, - ProbT* alphas, - ProbT* betas, - ProbT* output) { - int start = S > 1 ? (S - 2) : 0, - end = (T > (S / 2) + repeats) ? S : S-1; - - std::fill(output, output + alphabet_size_, ctc_helper::neg_inf()); - - //set the starting values in the beta column at the very right edge - for (int i = start; i < end; ++i) { - betas[i] = log_probs[labels[i] + (T - 1) * (alphabet_size_ * minibatch_)]; - - //compute alpha * beta in log space at this position in (S, T) space - alphas[i + (T - 1) * S] += betas[i]; - - //update the gradient associated with this label - //essentially performing a reduce-by-key in a sequential manner - output[labels[i]] = - ctc_helper::log_plus()(alphas[i + (T - 1) * S], output[labels[i]]); - } - - //update the gradient wrt to each unique label - for (int i = 0; i < alphabet_size_; ++i) { - int idx3 = (T - 1) * alphabet_size_ * minibatch_ + i; - - if (output[i] == 0.0 || output[i] == ctc_helper::neg_inf() || - log_probs[idx3] == ctc_helper::neg_inf()) { - grad[idx3] = std::exp(log_probs[idx3]); - } else { - grad[idx3] = std::exp(log_probs[idx3]) - - std::exp(output[i] - log_probs[idx3] - log_partition); - } - } - - //loop from the second to last column all the way to the left - for(int t = T - 2; t >= 0; --t) { - int remain = (S / 2) + repeats - (T - t); - if(remain >= -1) - start -= s_inc[remain + 1]; - if(t < (S / 2) + repeats) - end -= e_inc[t]; - - int endloop = end == S ? end - 1 : end; - int idx1 = t * S, idx3 = t * (alphabet_size_ * minibatch_); - - std::fill(output, output + alphabet_size_, ctc_helper::neg_inf()); - - for(int i = start; i < endloop; ++i) { - ProbT next_sum = ctc_helper::log_plus()(betas[i], betas[(i+1)]); - // Skip two if not on blank and not on repeat. - if (labels[i] != blank_label_ && i != (S-2) && labels[i] != labels[i+2]){ - next_sum = ctc_helper::log_plus()(next_sum, betas[(i+2)]); - } - betas[i] = next_sum + log_probs[labels[i] + idx3]; - - //compute alpha * beta in log space - alphas[i + idx1] += betas[i]; - - //update the gradient associated with this label - output[labels[i]] = - ctc_helper::log_plus()(alphas[i + idx1], output[labels[i]]); - } - - if (end == S) { - betas[(S-1)] = betas[(S-1)] + log_probs[blank_label_ + idx3]; - alphas[(S-1) + idx1] += betas[(S-1)]; - - output[labels[S-1]] = - ctc_helper::log_plus()(alphas[S-1 + idx1], output[labels[S-1]]); - } - - //go over the unique labels and compute the final grad - // wrt to each one at this time step - for (int i = 0; i < alphabet_size_; ++i) { - - if (output[i] == 0.0 || output[i] == ctc_helper::neg_inf() || - log_probs[idx3] == ctc_helper::neg_inf()) { - grad[idx3] = std::exp(log_probs[idx3]); - } else { - grad[idx3] = std::exp(log_probs[idx3]) - - std::exp(output[i] - log_probs[idx3] - log_partition); - } - ++idx3; - } - } - - ProbT loglike = ctc_helper::neg_inf(); - for(int i = start; i < end; ++i) { - loglike = ctc_helper::log_plus()(loglike, betas[i]); - } - - return loglike; -} - -template -ctcStatus_t -CpuCTC::cost_and_grad(const ProbT* const activations, - ProbT *grads, - ProbT *costs, - const int* const flat_labels, - const int* const label_lengths, - const int* const input_lengths) { - if (activations == nullptr || - grads == nullptr || - costs == nullptr || - flat_labels == nullptr || - label_lengths == nullptr || - input_lengths == nullptr - ) - return CTC_STATUS_INVALID_VALUE; - - ProbT* log_probs = static_cast(workspace_); - - int maxT = *std::max_element(input_lengths, input_lengths + minibatch_); - - size_t bytes_used = sizeof(ProbT) * minibatch_ * alphabet_size_ * maxT; - - //per minibatch memory - size_t per_minibatch_bytes = 0; - - int maxL = *std::max_element(label_lengths, label_lengths + minibatch_);; - int maxS = 2 * maxL + 1; - - //output - per_minibatch_bytes += sizeof(float) * alphabet_size_; - - //alphas - per_minibatch_bytes += sizeof(float) * maxS * maxT; - - //betas - per_minibatch_bytes += sizeof(float) * maxS; - - //labels w/blanks, e_inc, s_inc - per_minibatch_bytes += 3 * sizeof(int) * maxS; - - log_softmax(activations, log_probs, input_lengths); - -#pragma omp parallel for - for (int mb = 0; mb < minibatch_; ++mb) { - const int T = input_lengths[mb]; // Length of utterance (time) - const int L = label_lengths[mb]; // Number of labels in transcription - - bool mb_status; - - std::tie(costs[mb], mb_status) = - cost_and_grad_kernel(grads + mb * alphabet_size_, - log_probs + mb * alphabet_size_, - flat_labels + std::accumulate(label_lengths, label_lengths + mb, 0), - T, L, mb, - bytes_used + mb * per_minibatch_bytes); - } - - return CTC_STATUS_SUCCESS; -} - -template -ctcStatus_t CpuCTC::score_forward(const ProbT* const activations, - ProbT* costs, - const int* const flat_labels, - const int* const label_lengths, - const int* const input_lengths) { - if (activations == nullptr || - costs == nullptr || - flat_labels == nullptr || - label_lengths == nullptr || - input_lengths == nullptr - ) - return CTC_STATUS_INVALID_VALUE; - - ProbT* log_probs = static_cast(workspace_); - - int maxT = *std::max_element(input_lengths, input_lengths + minibatch_); - - size_t bytes_used = sizeof(ProbT) * minibatch_ * alphabet_size_ * maxT; - - //per minibatch memory - size_t per_minibatch_bytes = 0; - - int maxL = *std::max_element(label_lengths, label_lengths + minibatch_); - int maxS = 2 * maxL + 1; - - //output - per_minibatch_bytes += sizeof(float) * alphabet_size_; - - //alphas - per_minibatch_bytes += sizeof(float) * maxS * maxT; - - //betas - per_minibatch_bytes += sizeof(float) * maxS; - - //labels w/blanks, e_inc, s_inc - per_minibatch_bytes += 3 * sizeof(int) * maxS; - - log_softmax(activations, log_probs, input_lengths); - -#pragma omp parallel for - for (int mb = 0; mb < minibatch_; ++mb) { - const int T = input_lengths[mb]; // Length of utterance (time) - const int L = label_lengths[mb]; // Number of labels in transcription - const int S = 2*L + 1; // Number of labels with blanks - - CpuCTC_metadata ctcm(L, S, T, mb, alphabet_size_, workspace_, - bytes_used + mb * per_minibatch_bytes, blank_label_, - flat_labels + std::accumulate(label_lengths, label_lengths + mb, 0)); - - - if (L + ctcm.repeats > T) - costs[mb] = ProbT(0); - else { - costs[mb] = -compute_alphas(log_probs + mb * alphabet_size_, ctcm.repeats, S, T, - ctcm.e_inc, ctcm.s_inc, ctcm.labels_w_blanks, - ctcm.alphas); - } - - } - - return CTC_STATUS_SUCCESS; -} - -} // mxnet_warpctc diff --git a/src/operator/contrib/ctc_include/detail/ctc_helper.h b/src/operator/contrib/ctc_include/detail/ctc_helper.h deleted file mode 100644 index 250188c697c6..000000000000 --- a/src/operator/contrib/ctc_include/detail/ctc_helper.h +++ /dev/null @@ -1,93 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, - * software distributed under the License is distributed on an - * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY - * KIND, either express or implied. See the License for the - * specific language governing permissions and limitations - * under the License. - */ - -#pragma once - -#include -#include -#include - -#include "hostdevice.h" - -typedef enum { - CTC_STATUS_SUCCESS = 0, - CTC_STATUS_MEMOPS_FAILED = 1, - CTC_STATUS_INVALID_VALUE = 2, - CTC_STATUS_EXECUTION_FAILED = 3, - CTC_STATUS_UNKNOWN_ERROR = 4 -} ctcStatus_t; - -typedef enum { - CTC_CPU = 0, - CTC_GPU = 1 -} ctcComputeLocation; - -namespace ctc_helper { - -static const float threshold = 1e-1; - -template -HOSTDEVICE -T neg_inf() { return -T(INFINITY); } - -inline int div_up(int x, int y) { - return (x + y - 1) / y; -} - -template struct maximum { - HOSTDEVICE - Res operator()(const Arg& x, const Arg& y) const { - return x < y ? y : x; - } -}; - -template struct add { - HOSTDEVICE - Res operator()(const Arg& x, const Arg& y) const { - return x + y; - } -}; - -template struct identity { - HOSTDEVICE Res operator()(const Arg& x) const {return Res(x);} -}; - -template struct negate { - HOSTDEVICE Res operator()(const Arg& x) const {return Res(-x);} -}; - -template struct exponential { - HOSTDEVICE Res operator()(const Arg& x) const {return std::exp(x);} -}; - -template -struct log_plus { - typedef Res result_type; - HOSTDEVICE - Res operator()(const Arg1& p1, const Arg2& p2) { - if (p1 == neg_inf()) - return p2; - if (p2 == neg_inf()) - return p1; - Res result = log1p(exp(-fabs(p1 - p2))) + maximum()(p1, p2); - return result; - } -}; - -} diff --git a/src/operator/contrib/ctc_include/detail/gpu_ctc.h b/src/operator/contrib/ctc_include/detail/gpu_ctc.h deleted file mode 100644 index 2c521b5abb5d..000000000000 --- a/src/operator/contrib/ctc_include/detail/gpu_ctc.h +++ /dev/null @@ -1,505 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, - * software distributed under the License is distributed on an - * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY - * KIND, either express or implied. See the License for the - * specific language governing permissions and limitations - * under the License. - */ - -#pragma once - - -#include "ctc_helper.h" -#include "gpu_ctc_kernels.h" - -namespace mxnet_warpctc { - -template -class GpuCTC { - public: - GpuCTC(int alphabet_size, - int minibatch, - void *workspace, - CUstream stream, - int blank_label) : - out_dim_(alphabet_size), minibatch_(minibatch), - gpu_workspace_(workspace), stream_(stream), - blank_label_(blank_label) {}; - - // Noncopyable - GpuCTC(const GpuCTC&) = delete; - GpuCTC& operator=(const GpuCTC&) = delete; - - ctcStatus_t - cost_and_grad(const ProbT* const activations, - ProbT* grads, - ProbT* costs, - const int* const flat_labels, - const int* const label_lengths, - const int* const input_lengths); - - ctcStatus_t - score_forward(const ProbT* const activations, - ProbT* costs, - const int* const flat_labels, - const int* const label_lengths, - const int* const input_lengths); - - private: - - template - ctcStatus_t launch_alpha_beta_kernels(const ProbT* const log_probs, - ProbT *grads, - bool compute_alpha, - bool compute_beta); - - ctcStatus_t - launch_gpu_kernels(const ProbT* const log_probs, - ProbT *grads, - size_t config, - bool launch_alpha, - bool launch_beta); - - ctcStatus_t - setup_gpu_metadata(const int* const flat_labels, - const int* const label_lengths, - const int* const input_lengths); - - ctcStatus_t - create_metadata_and_choose_config(const int* const label_lengths, - const int* const flat_labels, - const int* const input_lengths, - size_t& best_config); - - ctcStatus_t - compute_log_probs(const ProbT* const activations); - - ctcStatus_t - compute_cost_and_score(const ProbT* const activations, - ProbT* grads, - ProbT* costs, - const int* const flat_labels, - const int* const label_lengths, - const int* const input_lengths, - bool compute_alpha, - bool compute_betas_and_grad); - - - int out_dim_; // Number of characters plus blank - int minibatch_; - - int S_; - int T_; - - int activation_cols_; // Number of columns in activations - - void *gpu_workspace_; // Buffer for all temporary GPU memory - CUstream stream_; - int blank_label_; - - int *utt_length_; // T - int *label_sizes_; // L - int *repeats_; // repeats_ - int *label_offsets_; - int *labels_without_blanks_; - int *labels_with_blanks_; - ProbT *alphas_; - ProbT *nll_forward_; - ProbT *nll_backward_; - ProbT *denoms_; // Temporary storage for denoms for softmax - ProbT *log_probs_; // Temporary storage for probabilities (log softmax output) -}; - -template -ctcStatus_t -GpuCTC::setup_gpu_metadata(const int* const flat_labels, - const int* const label_lengths, - const int* const input_lengths) -{ - size_t gpu_bytes_used = 0; - - nll_forward_ = - reinterpret_cast(static_cast(gpu_workspace_) + - gpu_bytes_used); - gpu_bytes_used += minibatch_ * sizeof(ProbT); - - - nll_backward_ = - reinterpret_cast(static_cast(gpu_workspace_) + - gpu_bytes_used); - gpu_bytes_used += minibatch_ * sizeof(ProbT); - - - repeats_ = - reinterpret_cast(static_cast(gpu_workspace_) + - gpu_bytes_used); - gpu_bytes_used += minibatch_ * sizeof(int); - - label_offsets_ = - reinterpret_cast(static_cast(gpu_workspace_) + - gpu_bytes_used); - gpu_bytes_used += minibatch_ * sizeof(int); - - - // This is the max of all S and T for all valid examples in the minibatch. - // A valid example is one for which L + repeats <= T - S_ = 0; - T_ = 0; - - // This is the max of all timesteps, valid or not. Needed to compute offsets - int Tmax = 0; - - // This is the max of all labels, valid or not. Needed to compute offsets - int Lmax = 0; - int total_label_length = 0; - - constexpr int cpu_buffer_size = 64; - int repeats[cpu_buffer_size]; - int label_offsets[cpu_buffer_size]; - - const int num_passes = ctc_helper::div_up(minibatch_, cpu_buffer_size); - - cudaError_t cuda_status; - - for (int pass = 0; pass < num_passes; ++pass) { - - const int start_idx = pass * cpu_buffer_size; - const int end_idx = std::min(minibatch_, (pass+1) * cpu_buffer_size); - - for (int j = start_idx; j < end_idx; ++j) { - const int L = label_lengths[j]; - const int local_T = input_lengths[j]; - const int *label_ptr = &(flat_labels[total_label_length]); - - label_offsets[j % cpu_buffer_size] = total_label_length; - total_label_length += L; - - int repeat_counter = 0; - - for (int i = 1; i < L; ++i) - repeat_counter += (label_ptr[i] == label_ptr[i-1]); - - repeats[j % cpu_buffer_size] = repeat_counter; - const bool valid_label = ((L + repeat_counter) <= local_T); - - // Only update S and T if label is valid - S_ = (valid_label) ? std::max(S_, L) : S_; - T_ = (valid_label) ? std::max(T_, local_T) : T_; - - Tmax = std::max(Tmax, local_T); - Lmax = std::max(Lmax, L); - } - - cuda_status = cudaMemcpyAsync(&(repeats_[start_idx]), repeats, - (end_idx - start_idx) * sizeof(int), - cudaMemcpyHostToDevice, stream_); - if (cuda_status != cudaSuccess) - return CTC_STATUS_MEMOPS_FAILED; - - - cuda_status = cudaMemcpyAsync(&(label_offsets_[start_idx]), label_offsets, - (end_idx - start_idx) * sizeof(int), - cudaMemcpyHostToDevice, stream_); - if (cuda_status != cudaSuccess) - return CTC_STATUS_MEMOPS_FAILED; - } - - S_ = 2 * S_ + 1; - const int Smax = 2 * Lmax + 1; - - activation_cols_ = minibatch_ * Tmax; - - // Allocate memory for T - utt_length_ = - reinterpret_cast(static_cast(gpu_workspace_) + - gpu_bytes_used); - gpu_bytes_used += minibatch_ * sizeof(int); - - cuda_status = cudaMemcpyAsync(utt_length_, input_lengths, - minibatch_ * sizeof(int), - cudaMemcpyHostToDevice, stream_); - if (cuda_status != cudaSuccess) - return CTC_STATUS_MEMOPS_FAILED; - - label_sizes_ = - reinterpret_cast(static_cast(gpu_workspace_) + - gpu_bytes_used); - gpu_bytes_used += minibatch_ * sizeof(int); - cuda_status = cudaMemcpyAsync(label_sizes_, label_lengths, - minibatch_ * sizeof(int), - cudaMemcpyHostToDevice, stream_); - if (cuda_status != cudaSuccess) - return CTC_STATUS_MEMOPS_FAILED; - - labels_without_blanks_ = - reinterpret_cast(static_cast(gpu_workspace_) + - gpu_bytes_used); - gpu_bytes_used += Lmax * minibatch_ * sizeof(int); - cuda_status = cudaMemcpyAsync(labels_without_blanks_, flat_labels, - total_label_length * sizeof(int), - cudaMemcpyHostToDevice, stream_); - if (cuda_status != cudaSuccess) - return CTC_STATUS_MEMOPS_FAILED; - - labels_with_blanks_ = - reinterpret_cast(static_cast(gpu_workspace_) + - gpu_bytes_used); - gpu_bytes_used += Smax * minibatch_ * sizeof(int); - - alphas_ = - reinterpret_cast(static_cast(gpu_workspace_) + - gpu_bytes_used); - gpu_bytes_used += (S_ * T_) * minibatch_ * sizeof(ProbT); - - - denoms_ = - reinterpret_cast(static_cast(gpu_workspace_) + - gpu_bytes_used); - gpu_bytes_used += activation_cols_ * sizeof(ProbT); - - log_probs_ = - reinterpret_cast(static_cast(gpu_workspace_) + - gpu_bytes_used); - gpu_bytes_used += out_dim_ * activation_cols_ * sizeof(ProbT); - - return CTC_STATUS_SUCCESS; -} - -template -template -ctcStatus_t GpuCTC::launch_alpha_beta_kernels(const ProbT* const log_probs, - ProbT* grads, - bool compute_alpha, - bool compute_beta ) { - - // One thread block per utterance - const int grid_size = minibatch_; - - // The data is laid out so that the next timestep is minibatch entries - // away - const int stride = minibatch_; - - if (compute_alpha) - compute_alpha_kernel<<>> - (log_probs, label_sizes_, utt_length_, - repeats_, labels_without_blanks_, label_offsets_, - labels_with_blanks_, alphas_, nll_forward_, - stride, out_dim_, S_, T_, blank_label_); - - - if (compute_beta) { - compute_betas_and_grad_kernel<<>> - (log_probs, label_sizes_, utt_length_, repeats_, - labels_with_blanks_, alphas_, nll_forward_, nll_backward_, - grads, stride, out_dim_, S_, T_, blank_label_); - - cudaStreamSynchronize(stream_); - } - - cudaError_t err = cudaGetLastError(); - if (err != cudaSuccess) - return CTC_STATUS_EXECUTION_FAILED; - - return CTC_STATUS_SUCCESS; -} - -template -ctcStatus_t -GpuCTC::create_metadata_and_choose_config(const int* const flat_labels, - const int* const label_lengths, - const int* const input_lengths, - size_t& best_config) { - - // Setup the metadata for GPU - ctcStatus_t status = setup_gpu_metadata(flat_labels, label_lengths, input_lengths); - if (status != CTC_STATUS_SUCCESS) - return status; - - constexpr int num_configs = 12; - - int config_NT[num_configs] = - {32, 64, 128, 64, 128, 32, 64, 128, 64, 128, 128, 128}; - int config_VT[num_configs] = - { 1, 1, 1, 3, 2, 9, 6, 4, 9, 6, 9, 10}; - - best_config = 0; - - for (int i = 0; i < num_configs; ++i) { - if ((config_NT[i]* config_VT[i]) >= S_) - break; - else - best_config++; - } - - if (best_config >= num_configs) - return CTC_STATUS_UNKNOWN_ERROR; - - return CTC_STATUS_SUCCESS; -} - -template -ctcStatus_t -GpuCTC::launch_gpu_kernels(const ProbT* const log_probs, - ProbT* grads, - size_t config, - bool l_a, - bool l_b) { - - switch(config) { - case 0: {return launch_alpha_beta_kernels<32, 1>(log_probs, grads, l_a, l_b);} - case 1: {return launch_alpha_beta_kernels<64, 1>(log_probs, grads, l_a, l_b);} - case 2: {return launch_alpha_beta_kernels<128, 1>(log_probs, grads, l_a, l_b);} - case 3: {return launch_alpha_beta_kernels<64, 3>(log_probs, grads, l_a, l_b);} - case 4: {return launch_alpha_beta_kernels<128, 2>(log_probs, grads, l_a, l_b);} - case 5: {return launch_alpha_beta_kernels<32, 9>(log_probs, grads, l_a, l_b);} - case 6: {return launch_alpha_beta_kernels<64, 6>(log_probs, grads, l_a, l_b);} - case 7: {return launch_alpha_beta_kernels<128, 4>(log_probs, grads, l_a, l_b);} - case 8: {return launch_alpha_beta_kernels<64, 9>(log_probs, grads, l_a, l_b);} - case 9: {return launch_alpha_beta_kernels<128, 6>(log_probs, grads, l_a, l_b);} - case 10: {return launch_alpha_beta_kernels<128, 9>(log_probs, grads, l_a, l_b);} - case 11: {return launch_alpha_beta_kernels<128, 10>(log_probs, grads, l_a, l_b);} - } - - return CTC_STATUS_EXECUTION_FAILED; -} - -template -ctcStatus_t -GpuCTC::compute_log_probs(const ProbT* const activations) { - - cudaError_t cuda_status; - cuda_status = - cudaMemcpyAsync(log_probs_, activations, - activation_cols_ * out_dim_ *sizeof(ProbT), - cudaMemcpyDeviceToDevice, stream_); - if (cuda_status != cudaSuccess) - return CTC_STATUS_MEMOPS_FAILED; - - - - // create mshadow handles to data - using namespace mshadow; - using namespace mshadow::expr; - Stream mxstream; - mxstream.stream_ = stream_; - Tensor log_probs_handle(log_probs_, mshadow::Shape2(activation_cols_, out_dim_), &mxstream); - Tensor denoms_handle(denoms_, mshadow::Shape1(activation_cols_), &mxstream); - denoms_handle = reduce_with_axis(log_probs_handle, 1); - - - - // Kernel launch to subtract maximum - const int NT = 128; - const int VT = 1; - const int NV = NT * VT; - const int num_elements = out_dim_ * activation_cols_; - const int grid_size = ctc_helper::div_up(num_elements, NV); - - prepare_stable_LSM_kernel <<< grid_size, NT, 0, stream_>>> - (ctc_helper::identity(), log_probs_, - denoms_, out_dim_, num_elements); - - // compute denominators for softmax - denoms_handle = reduce_with_axis(F(log_probs_handle), 1); - - // Kernel launch to calculate probabilities - compute_log_probs_kernel<<>> - (ctc_helper::identity(), log_probs_, - denoms_, out_dim_, num_elements); - - cuda_status = cudaGetLastError(); - if (cuda_status != cudaSuccess) - return CTC_STATUS_EXECUTION_FAILED; - - return CTC_STATUS_SUCCESS; -} - -template -ctcStatus_t -GpuCTC::compute_cost_and_score(const ProbT* const activations, - ProbT* grads, - ProbT* costs, - const int* const flat_labels, - const int* const label_lengths, - const int* const input_lengths, - bool compute_alpha, - bool compute_betas_and_grad) { - - size_t best_config; - ctcStatus_t status = create_metadata_and_choose_config(flat_labels, - label_lengths, - input_lengths, - best_config); - if (status != CTC_STATUS_SUCCESS) - return status; - - status = compute_log_probs(activations); - if (status != CTC_STATUS_SUCCESS) - return status; - - launch_gpu_kernels(log_probs_, grads, best_config, - compute_alpha, compute_betas_and_grad); - - cudaError_t cuda_status_mem, cuda_status_sync; - cuda_status_mem = cudaMemcpyAsync(costs, nll_forward_, - sizeof(ProbT) * minibatch_, - cudaMemcpyDeviceToHost, stream_); - cuda_status_sync = cudaStreamSynchronize(stream_); - if (cuda_status_mem != cudaSuccess || cuda_status_sync != cudaSuccess) - return CTC_STATUS_MEMOPS_FAILED; - - return CTC_STATUS_SUCCESS; -} - -template -ctcStatus_t -GpuCTC::cost_and_grad(const ProbT* const activations, - ProbT* grads, - ProbT* costs, - const int* const flat_labels, - const int* const label_lengths, - const int* const input_lengths) { - if (activations == nullptr || - grads == nullptr || - costs == nullptr || - flat_labels == nullptr || - label_lengths == nullptr || - input_lengths == nullptr - ) - return CTC_STATUS_INVALID_VALUE; - - return compute_cost_and_score(activations, grads, costs, flat_labels, - label_lengths, input_lengths, true, true); -} - -template -ctcStatus_t -GpuCTC::score_forward(const ProbT* const activations, - ProbT* costs, - const int* const flat_labels, - const int* const label_lengths, - const int* const input_lengths) { - if (activations == nullptr || - costs == nullptr || - flat_labels == nullptr || - label_lengths == nullptr || - input_lengths == nullptr - ) - return CTC_STATUS_INVALID_VALUE; - - return compute_cost_and_score(activations, nullptr, costs, flat_labels, - label_lengths, input_lengths, true, false); -} - -} // mxnet_warpctc diff --git a/src/operator/contrib/ctc_include/detail/gpu_ctc_kernels.h b/src/operator/contrib/ctc_include/detail/gpu_ctc_kernels.h deleted file mode 100644 index c9bc2026efb5..000000000000 --- a/src/operator/contrib/ctc_include/detail/gpu_ctc_kernels.h +++ /dev/null @@ -1,507 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, - * software distributed under the License is distributed on an - * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY - * KIND, either express or implied. See the License for the - * specific language governing permissions and limitations - * under the License. - */ - -#pragma once - -#include "../contrib/moderngpu/include/device/ctascan.cuh" -#include "../contrib/moderngpu/include/device/ctamerge.cuh" - -#include "ctc_helper.h" - -using namespace mgpu; - -template -struct CTASegReduce { - - enum {NV = NT * VT}; - - union Storage { - typename CTAScan::Storage scanStorage; - int indices[NV]; - }; - - //adapted from global kernel KernelReduceByKeyPreprocess - __device__ static void preprocessKeys(KeyT *keys, int count, - int *numUniqueLabels, int seg_start[VT], - int seg_end[VT], int *scanout) { - __shared__ Storage shared; - - const int tid = threadIdx.x; - // Compare adjacent keys within each thread and mark discontinuities - int endFlags = 0; - T key = keys[VT * tid]; - #pragma unroll - for (int i = 0; i < VT; ++i) { - int index = VT * tid + 1 + i; - T next = keys[index]; - if(index == count || (index < count && key != next)) { - endFlags |= 1 << i; - } - key = next; - } - - __syncthreads(); - - //Count the number of encountered end flags - int scan = CTAScan::Scan(tid, popc(endFlags), shared.scanStorage, numUniqueLabels); - - __syncthreads(); - - //output the unique keys - //use indices as scratch space - int outputPos = scan; - #pragma unroll - for (int i = 0; i < VT; ++i) { - - if ( (endFlags >> i) & 1) { - shared.indices[outputPos] = keys[VT * tid + i]; - scanout[outputPos] = VT * tid + i; - outputPos++; - } - } - - __syncthreads(); - - // Create start and end - for (int idx = tid, j = 0; idx < (*numUniqueLabels); idx += blockDim.x, ++j) { - seg_start[j] = (idx == 0) ? 0 : (scanout[idx-1] + 1); - seg_end[j] = scanout[idx]; - } - - __syncthreads(); - - //copy from the scratch space back into the keys - #pragma unroll - for (int i = 0; i < VT; ++i) { - keys[i * NT + tid] = shared.indices[i * NT + tid]; - } - - __syncthreads(); - } -}; - -// Computes forward probabilities. This fills in a T * S matrix. -// The computation starts at t=1 (2nd row) and ends at t=T-1 (last row). Each row has -// S elements where S = 2L + 1. -// -// We only need to read in probabilities corresponding to the labels, thus a sparse -// set of values are read from the log probs matrix since the character set is much smaller -// than the labels. This is much more true for Mandarin than English. -template -__global__ -void compute_alpha_kernel (const ProbT* log_probs, const int *label_sizes, - const int *utt_length, const int *repeats_in_labels, - const int *labels_without_blanks, const int *label_offsets, - int *labels_with_blanks, ProbT *alphas, - ProbT* nll_forward, int stride, int out_dim, - int S_memoffset, int T_memoffset, int blank_label) { - - ctc_helper::log_plus log_plus_f; - - const int tid = threadIdx.x; - const int L = label_sizes[blockIdx.x]; - const int T = utt_length[blockIdx.x]; - const int S = 2*L + 1; - const int prob_offset = out_dim * blockIdx.x; - const int repeats = repeats_in_labels[blockIdx.x]; - - const int NV = NT * VT; - __shared__ int label[NV]; - - if ((L + repeats) > T) - return; - - // Generate labels with blanks from labels without blanks - { - const int label_start_offset = label_offsets[blockIdx.x]; - for (int idx = tid; idx < L; idx += blockDim.x) { - const int offset = (blockIdx.x * S_memoffset) + 2 * idx; - labels_with_blanks[offset] = blank_label; - labels_with_blanks[offset+1] = labels_without_blanks[label_start_offset + idx]; - } - if (tid == 0) { - labels_with_blanks[(blockIdx.x * S_memoffset) + 2 * L] = blank_label; - } - } - __syncthreads(); - - const int *labels = labels_with_blanks; - const int* label_global = &labels[blockIdx.x * S_memoffset]; - ProbT* alpha = &alphas[blockIdx.x * (S_memoffset * T_memoffset)]; - - // Set the first row of alpha neg_inf - it is much more efficient to do it - // here than outside - #pragma unroll - for (int idx = tid; idx < min(S, NV); idx += blockDim.x) { - alpha[idx] = ctc_helper::neg_inf(); - } - - // Load labels into shared memory - #pragma unroll - for (int i = tid; i < S; i += NT) { - label[i] = label_global[i]; - } - - __syncthreads(); - - int start = (L + repeats < T) ? 0 : 1; - int end = S > 1 ? 2 : 1; - - // Initialize the first row corresponding to t=0; - for(int i = tid; i < (end-start); i += blockDim.x) - alpha[i + start] = log_probs[prob_offset + label[i + start]]; - - __syncthreads(); - - // Fill in the rest of matrix, one row at a time (outer loop). - for(int t = 1; t < T; ++t) { - - // Start offsets into the current and previous row - const int start_cur_row = t * S; - const int start_prev_row = (t - 1) * S; - - // The prob is a 2D column major array, with probabilites for each t strided - // by (out_dim * stride), where stride is the minibatch size - const int start_prob_col = t * (out_dim * stride); - - // This is the first column and in this case there is nothing left of it - if (tid == 0) { - if (start == 0) { - alpha[start_cur_row] = alpha[start_prev_row] + - log_probs[prob_offset + start_prob_col + blank_label]; - } - else if (start == 1) { - alpha[start_cur_row] = alpha[start_prev_row]; - } - } - - __syncthreads(); - - // Fill in the elements in each row. There is no loop dependence here since our - // input is the row above. We sum either two or three adjacent values from the - // row above depending on whether we have a blank or repeated characters. Finally - // we add the probability corresponding to this label at time t - #pragma unroll - for (int idx = (tid+1); idx < S; idx += blockDim.x) { - - ProbT prev_sum = log_plus_f(alpha[idx + start_prev_row], alpha[(idx-1) + start_prev_row]); - - // Skip two if not on blank and not on repeat. - if ((label[idx] != blank_label) && - (idx != 1) && (label[idx] != label[idx-2])) - prev_sum = log_plus_f(prev_sum, alpha[(idx-2) + start_prev_row]); - - alpha[idx + start_cur_row] = - prev_sum + log_probs[prob_offset + start_prob_col + label[idx]]; - } - - __syncthreads(); - } - - if (tid == 0) { - // Add and return the rightmost two/one element(s) in the last row. - ProbT loglike = ctc_helper::neg_inf(); - - // This is the total increment for s_inc and e_inc through the loop - const int val = 2 * (L-1) + 1 - (((L + repeats) == T) ? 1 : 0); - - start = (val * (L!=0) + start); - end = (val * (L!=0) + end); - - for(int i = start; i < end; ++i) - loglike = log_plus_f(loglike, alpha[i + (T - 1) * S]); - - nll_forward[blockIdx.x] = -loglike; - } -} - -// Computes backward probabilities. This also fills in a T * S matrix -// -// See comments above compute_alphas for more context. -template -__global__ -void compute_betas_and_grad_kernel (const ProbT* log_probs, const int *label_sizes, - const int *utt_length, const int *repeats_in_labels, - const int *labels_with_blanks, ProbT *alphas, - const ProbT* nll_forward, ProbT *nll_backward, - ProbT *grads, int stride, int out_dim, - int S_memoffset, int T_memoffset, int blank_label) { - - ctc_helper::log_plus log_plus_f; - typedef CTASegReduce> SegReduce; - - const int tid = threadIdx.x; - const int L = label_sizes[blockIdx.x]; - const int T = utt_length[blockIdx.x]; - const int S = 2*L + 1; - const int prob_offset = out_dim * blockIdx.x; - const int repeats = repeats_in_labels[blockIdx.x]; - const ProbT log_partition = -nll_forward[blockIdx.x]; - - const int* labels = labels_with_blanks; - const int* label_global = &labels[blockIdx.x * S_memoffset]; - ProbT* alpha = &alphas[blockIdx.x * (S_memoffset * T_memoffset)]; - - const int NV = NT * VT; - - union TempStorage { - ProbT beta[NV]; - int result[NV]; - }; - - __shared__ TempStorage temp_buffer; - - __shared__ int label[NV]; - - // Temporaries needed for segmented reduce - // TODO: see if we can combine the shared memory requirements - __shared__ int keys_shared[NV]; - __shared__ int gather_indices[NV]; - __shared__ ProbT output[NV]; - - ProbT beta_val[VT]; - - if ((L + repeats) > T) - return; - - int start = S > 1 ? (S - 2) : 0; - int end = (L + repeats < T) ? S : S-1; - - // Setup shared memory buffers - #pragma unroll - for (int idx = tid; idx < NV; idx += NT) { - label[idx] = (idx < S) ? label_global[idx] : INT_MAX; - } - - __syncthreads(); - - // int flags; - int uniquelabels; - int seg_start[VT]; - int seg_end[VT]; - - // Sort labels and record indices from which to gather from - { - int key[VT]; - int gather_val[VT]; - - #pragma unroll - for (int i = 0; i < VT; ++i) { - const int idx = tid * VT + i; - gather_val[i] = idx; - key[i] = label[idx]; - } - - __syncthreads(); - - CTAMergesort> - (key, gather_val, keys_shared, gather_indices, S, tid, mgpu::less()); - - __syncthreads(); - - for (int i = 0; i < VT; ++i) { - const int idx = tid * VT + i; - gather_indices[idx] = gather_val[i]; - } - - __syncthreads(); - - SegReduce::preprocessKeys(keys_shared, S, &uniquelabels, seg_start, seg_end, - temp_buffer.result); - __syncthreads(); - } - - // TODO: probably not necessary - __syncthreads(); - - // Load labels back - #pragma unroll - for (int idx = tid; idx < NV; idx += NT) { - temp_buffer.beta[idx] = ctc_helper::neg_inf(); - } - __syncthreads(); - - // Initialize the two rightmost values in the last row (assuming L non-zero) - for(int i = tid; i < (end-start); i += blockDim.x) - temp_buffer.beta[i + start] = - log_probs[prob_offset + (T - 1) * (out_dim * stride) + label[i + start]]; - - __syncthreads(); - - // Load output data in registers through the transpose trick - should really be a function - #pragma unroll - for (int idx = tid; idx < S; idx += NT) { - output[idx] = alpha[idx + (T - 1) * S] + temp_buffer.beta[idx]; - } - - __syncthreads(); - - // Start at the second to last row and backward in time - for(int t = T - 1; t >= 0; --t) { - - // Start offsets into the current and next row - const int start_cur_row = t * S; - - // Starting offset of column that we read from the log probs array - const int start_prob_col = t * (out_dim * stride); - - if (t < T-1) { - - // Filling up one row at at time but going back in time from the last row - // to the first. As in the forward pass, there is no loop dependence and we - // do a variable length filter of maximum filter size of 3 - #pragma unroll - for(int idx = tid, i = 0; idx < (S-1); idx += NT, i++) { - ProbT next_sum = log_plus_f(temp_buffer.beta[idx], temp_buffer.beta[idx+1]); - - // Skip two if not on blank and not on repeat. - if ((label[idx] != blank_label) && - (idx != (S-2)) && (label[idx] != label[idx+2])) - next_sum = log_plus_f(next_sum, temp_buffer.beta[idx+2]); - - beta_val[i] = next_sum + log_probs[prob_offset + start_prob_col + label[idx]]; - } - - __syncthreads(); - - // Initialize values for the rightmost column since there is nothing to the right - // Update input buffer for next iteration - if ((tid == 0) && (end == S)) - temp_buffer.beta[(S-1)] = temp_buffer.beta[(S-1)] + - log_probs[prob_offset + start_prob_col + blank_label]; - - #pragma unroll - for(int idx = tid, i = 0; idx < (S-1); idx += NT, i++) { - temp_buffer.beta[idx] = beta_val[i]; - } - - __syncthreads(); - - // Beta Computation done - add to alpha and update the gradient. Reload - // the gradient back for segmented reduce later on - #pragma unroll - for(int idx = tid; idx < S; idx += NT) { - output[idx] = alpha[idx + start_cur_row] + temp_buffer.beta[idx]; - } - - __syncthreads(); - - } - - __syncthreads(); - - // Compute segmented reduction of output by using label as key - { - // Somewhat faster key value reduce - ProbT accum[VT]; - - for (int idx = tid, j = 0; idx < uniquelabels; idx += blockDim.x, ++j) { - - accum[j] = ctc_helper::neg_inf(); - for (int i = seg_start[j]; i <= seg_end[j]; ++i) { - accum[j] = log_plus_f(accum[j], output[gather_indices[i]]); - } - } - __syncthreads(); - - // Write accumulated value into output since that is not used - for (int idx = tid, j = 0; idx < uniquelabels; idx += blockDim.x, ++j) { - output[idx] = accum[j]; - } - __syncthreads(); - - for (int idx = tid; idx < out_dim; idx += blockDim.x) { - const int grads_offset = prob_offset + start_prob_col + idx; - grads[grads_offset] = exp(log_probs[grads_offset]); - } - - __syncthreads(); - - for (int idx = tid; idx < uniquelabels; idx += blockDim.x) { - const int grads_offset = prob_offset + start_prob_col + keys_shared[idx]; - - ProbT grad = output[idx]; - - if ((grad == 0.0) || (log_probs[grads_offset] == ctc_helper::neg_inf()) || - (grad == ctc_helper::neg_inf())) { - } else { - grads[grads_offset] = - exp(log_probs[grads_offset]) - exp(grad - log_probs[grads_offset] - log_partition); - } - } - - __syncthreads(); - } - - // Output backward log likelihood - if ((t == 0) && (tid == 0)) { - ProbT loglike = ctc_helper::neg_inf(); - - const int val = 2 * (L-1) + 1 - (((L + repeats) == T) ? 1 : 0); - - start = (-val * (L != 0) + start); - end = (-val * (L != 0) + end); - - // Sum and return the leftmost one/two value(s) in first row - for(int i = start; i < end; ++i) - loglike = log_plus_f(loglike, temp_buffer.beta[i]); - - nll_backward[blockIdx.x] = -loglike; - } - - // For some reason this is important - __syncthreads(); - } -} - -template -__global__ void compute_log_probs_kernel(Op f, ProbT* log_probs, - const ProbT* const denom, - int alphabet_size, - int count) { - - int idx = blockDim.x * blockIdx.x + threadIdx.x; - int stride = blockDim.x * gridDim.x; -#pragma unroll - for(int i = 0; i < VT; i++) { - if (idx < count) { - const int column_idx = idx / alphabet_size; - log_probs[idx] = log_probs[idx] - log(denom[column_idx]); - } - idx += stride; - } -} - -template -__global__ void prepare_stable_LSM_kernel(Op f, ProbT* log_probs, - const ProbT* const col_max, - int alphabet_size, - int count) { - - int idx = blockDim.x * blockIdx.x + threadIdx.x; - int stride = blockDim.x * gridDim.x; -#pragma unroll - for(int i = 0; i < VT; i++) { - if (idx < count) { - const int column_idx = idx / alphabet_size; - log_probs[idx] = f(log_probs[idx] - col_max[column_idx]); - } - idx += stride; - } -} diff --git a/src/operator/contrib/ctc_include/detail/hostdevice.h b/src/operator/contrib/ctc_include/detail/hostdevice.h deleted file mode 100644 index f7f0425bf26d..000000000000 --- a/src/operator/contrib/ctc_include/detail/hostdevice.h +++ /dev/null @@ -1,27 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, - * software distributed under the License is distributed on an - * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY - * KIND, either express or implied. See the License for the - * specific language governing permissions and limitations - * under the License. - */ - - -#pragma once - -#ifdef __CUDACC__ - #define HOSTDEVICE __host__ __device__ -#else - #define HOSTDEVICE -#endif diff --git a/src/operator/contrib/ctc_loss-inl.h b/src/operator/contrib/ctc_loss-inl.h deleted file mode 100644 index c8a8b2637401..000000000000 --- a/src/operator/contrib/ctc_loss-inl.h +++ /dev/null @@ -1,591 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, - * software distributed under the License is distributed on an - * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY - * KIND, either express or implied. See the License for the - * specific language governing permissions and limitations - * under the License. - */ - -/*! - * Copyright (c) 2016 by Contributors - * \file ctc_loss-inl.h - * \brief - * \author Sebastian Bodenstien -*/ - -#ifndef MXNET_OPERATOR_CONTRIB_CTC_LOSS_INL_H_ -#define MXNET_OPERATOR_CONTRIB_CTC_LOSS_INL_H_ - -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include "../operator_common.h" -#include "../sequence_op_common.h" -#include "../mshadow_op.h" -#include "../nn/sequence_mask-inl.h" - -#if defined(__CUDACC__) && MXNET_USE_CUDNN == 1 && CUDNN_MAJOR >= 7 -#define CUDNN_LABEL_LENGTH_LIMIT 256 -#include "../nn/softmax-inl.h" -#endif // CUDNN - -namespace mxnet { -namespace op { - -namespace ctc_loss { -enum CTCLossOpInputs { kData, kLabel }; -enum CTCLossOpOutputs { kOut, kGrad }; -enum CTCLossOpForwardResource { kTempSpace }; -} - -template -inline void get_workspace_size(std::vector *label_lengths, - std::vector *data_lengths, - int alphabet_size, int minibatch, bool gpu, - size_t *size_bytes) { - // This is the max of all S and T for all examples in the minibatch. - int maxL = *std::max_element(label_lengths->data(), - label_lengths->data() + minibatch); - int maxT = *std::max_element(data_lengths->data(), - data_lengths->data() + minibatch); - - const int S = 2 * maxL + 1; - - *size_bytes = 0; - - if (gpu) { - // GPU storage - // nll_forward, nll_backward - *size_bytes += 2 * sizeof(T) * minibatch; - - // repeats - *size_bytes += sizeof(int) * minibatch; - - // label offsets - *size_bytes += sizeof(int) * minibatch; - - // utt_length - *size_bytes += sizeof(int) * minibatch; - - // label lengths - *size_bytes += sizeof(int) * minibatch; - - // labels without blanks - overallocate for now - *size_bytes += sizeof(int) * maxL * minibatch; - - // labels with blanks - *size_bytes += sizeof(int) * S * minibatch; - - // alphas - *size_bytes += sizeof(T) * S * maxT * minibatch; - - // denoms - *size_bytes += sizeof(T) * maxT * minibatch; - - // probs (since we will pass in activations) - *size_bytes += sizeof(T) * alphabet_size * maxT * minibatch; - - } else { - // cpu can eventually replace all minibatch with - // max number of concurrent threads if memory is - // really tight - - // per minibatch memory - size_t per_minibatch_bytes = 0; - - // output - per_minibatch_bytes += sizeof(T) * alphabet_size; - - // alphas - per_minibatch_bytes += sizeof(T) * S * maxT; - - // betas - per_minibatch_bytes += sizeof(T) * S; - - // labels w/blanks, e_inc, s_inc - per_minibatch_bytes += 3 * sizeof(int) * S; - - *size_bytes = per_minibatch_bytes * minibatch; - - // probs - *size_bytes += sizeof(T) * alphabet_size * maxT * minibatch; - } -} - -// Takes a tensor of labels, and interprets 0-elements at the end of the vector -// as padding. The tensor is packed into an std::vector without padding -// characters. The label sequence lengths are also inferred from the padding chars. -// When cudnn is enabled, the return value signifies whether the cudnn length limit is exceeded. -template -inline bool LabelTensorToPackedVector(mshadow::Tensor labels, - int padding_mask, - std::vector *packed_labels, - std::vector *label_lengths) { - int batch = labels.size(0); - int max_num_labels = labels.size(1); - bool exceed_limit = false; - - std::vector cpu_labels(max_num_labels*batch); - mshadow::Tensor flat_labels = labels.FlatTo1D(); - IndexTensorToVector(flat_labels, &cpu_labels); - - for (int b = 0; b < batch; ++b) { - auto start = cpu_labels.data()+b*max_num_labels; - auto res = std::find(start, start+max_num_labels, padding_mask); - int len = std::distance(start, res); -#if defined(__CUDACC__) && MXNET_USE_CUDNN == 1 && CUDNN_MAJOR >= 7 - exceed_limit = exceed_limit || len > CUDNN_LABEL_LENGTH_LIMIT; -#endif - std::copy(start, start + len, - std::back_inserter(*packed_labels)); - label_lengths->at(b) = len; - } - return exceed_limit; -} - -// Takes a tensor of labels, and a vector which specifies the actual length of each label -// The tensor is packed into an std::vector without padding characters. -// The label length vector is copied into an std::vector. -// When cudnn is enabled, the return value signifies whether the cudnn length limit is exceeded. -template -inline bool PackLabelByLength(mshadow::Tensor labels, - mshadow::Tensor in_label_lengths, - std::vector *packed_labels, - std::vector *label_lengths) { - int batch = labels.size(0); - int max_num_labels = labels.size(1); - bool exceed_limit = false; - - IndexTensorToVector(in_label_lengths, label_lengths); - - std::vector cpu_labels(max_num_labels*batch); - mshadow::Tensor flat_labels = labels.FlatTo1D(); - IndexTensorToVector(flat_labels, &cpu_labels); - - for (int b = 0; b < batch; ++b) { - auto start = cpu_labels.data()+b*max_num_labels; - int len = label_lengths->at(b); -#if defined(__CUDACC__) && MXNET_USE_CUDNN == 1 && CUDNN_MAJOR >= 7 - exceed_limit = exceed_limit || len > CUDNN_LABEL_LENGTH_LIMIT; -#endif - std::copy(start, start + len, - std::back_inserter(*packed_labels)); - } - return exceed_limit; -} - -struct CTCLossParam : public dmlc::Parameter { - bool use_data_lengths; - bool use_label_lengths; - int blank_label; - DMLC_DECLARE_PARAMETER(CTCLossParam) { - DMLC_DECLARE_FIELD(use_data_lengths).set_default(false) - .describe("Whether the data lenghts are decided by `data_lengths`. " - "If false, the lengths are equal to the max sequence length."); - DMLC_DECLARE_FIELD(use_label_lengths).set_default(false) - .describe("Whether the label lenghts are decided by " - "`label_lengths`, or derived from `padding_mask`. " - "If false, the lengths are derived from the " - "first occurrence of the value of `padding_mask`. " - "The value of `padding_mask` is ``0`` when first CTC label is reserved for blank, " - "and ``-1`` when last label is reserved for blank. See `blank_label`."); - DMLC_DECLARE_FIELD(blank_label) - .add_enum("first", 0) - .add_enum("last", 1) - .set_default(0) - .describe("Set the label that is reserved for blank label." - "If \"first\", 0-th label is reserved, and " - "label values for tokens in the vocabulary are " - "between ``1`` and ``alphabet_size-1``, and the padding mask is ``-1``. " - "If \"last\", last label value ``alphabet_size-1`` " - "is reserved for blank label instead, " - "and label values for tokens in the vocabulary are " - "between ``0`` and ``alphabet_size-2``, and the padding mask is ``0``."); - } -}; - -template -class CTCLossOp : public Operator { - public: - explicit CTCLossOp(CTCLossParam p) { - this->param_ = p; - exceed_cudnn_limit = false; -#if defined(__CUDACC__) && MXNET_USE_CUDNN == 1 && CUDNN_MAJOR >= 7 - CUDNN_CALL(cudnnCreateCTCLossDescriptor(&ctc_desc_)); - CUDNN_CALL(cudnnSetCTCLossDescriptor(ctc_desc_, CUDNN_DATA_FLOAT)); - CUDNN_CALL(cudnnCreateTensorDescriptor(&prob_desc_)); - CUDNN_CALL(cudnnCreateTensorDescriptor(&grad_desc_)); -#endif - } - - ~CTCLossOp() { -#if defined(__CUDACC__) && MXNET_USE_CUDNN == 1 && CUDNN_MAJOR >= 7 - CUDNN_CALL(cudnnDestroyCTCLossDescriptor(ctc_desc_)); - CUDNN_CALL(cudnnDestroyTensorDescriptor(prob_desc_)); - CUDNN_CALL(cudnnDestroyTensorDescriptor(grad_desc_)); -#endif - } - - virtual void Forward(const OpContext &ctx, const std::vector &in_data, - const std::vector &req, - const std::vector &out_data, - const std::vector &aux_args) { - using namespace mshadow; - using namespace mshadow::expr; - CHECK_EQ(in_data.size(), 2U+param_.use_data_lengths+param_.use_label_lengths); - CHECK_EQ(out_data.size(), 2U); - exceed_cudnn_limit = false; - Stream *s = ctx.get_stream(); - - MSHADOW_TYPE_SWITCH(in_data[ctc_loss::kLabel].type_flag_, DType, { - Tensor data = - in_data[ctc_loss::kData].get(s); - Tensor labels = - in_data[ctc_loss::kLabel].get(s); - - Tensor costs = - out_data[ctc_loss::kOut].get(s); - Tensor grad = - out_data[ctc_loss::kGrad].get(s); - - int max_seq_len = data.size(0); - int batch_size = data.size(1); - int alphabet_size = data.size(2); - - // data_lengths - std::vector data_lengths(batch_size, max_seq_len); - if (param_.use_data_lengths) { - int kInputLength = 2; - IndexTensorToVector(in_data[kInputLength].get(s), &data_lengths); - } - - // label_lengths - std::vector packed_labels; - std::vector label_lengths(batch_size); - - if (param_.use_label_lengths) { - int kLabelLength = 2 + param_.use_data_lengths; - exceed_cudnn_limit = - PackLabelByLength(labels, in_data[kLabelLength].get(s), - &packed_labels, &label_lengths); - } else { - exceed_cudnn_limit = LabelTensorToPackedVector(labels, param_.blank_label == 0 ? 0 : -1, - &packed_labels, &label_lengths); - } - - // CUDNN is disabled due to lack of support for input lengths - /* #if defined(__CUDACC__) && MXNET_USE_CUDNN == 1 && CUDNN_MAJOR >= 7 */ - /* if (!exceed_cudnn_limit) { */ - /* cudnn_forward(ctx, s, data, costs, grad, */ - /* &data_lengths, &label_lengths, &packed_labels, */ - /* max_seq_len, batch_size, alphabet_size, */ - /* req[ctc_loss::kGrad] != mxnet::kNullOp); */ - /* } else { */ - /* baidu_forward(ctx, s, data, costs, grad, */ - /* &data_lengths, &label_lengths, &packed_labels, */ - /* batch_size, alphabet_size, req[ctc_loss::kGrad] != mxnet::kNullOp);*/ - /* } */ - /* #else */ - - baidu_forward(ctx, s, data, costs, grad, - &data_lengths, &label_lengths, &packed_labels, - batch_size, alphabet_size, req[ctc_loss::kGrad] != mxnet::kNullOp); - - if (param_.use_data_lengths) { - // baidu warp CTC implementation sometimes includes undefined gradients - // for data outside of length mask. Setting to 0 to make it consistent - // with CPU implementation. - int kInputLength = 2; - mxnet_op::SequenceMask(grad, in_data[kInputLength].get(s), - static_cast(0)); - } - }); - } - - virtual void Backward(const OpContext &ctx, - const std::vector &out_grad, - const std::vector &in_data, - const std::vector &out_data, - const std::vector &req, - const std::vector &in_grad, - const std::vector &aux_args) { - using namespace mshadow; - using namespace mshadow::expr; - - Stream *s = ctx.get_stream(); - - Tensor data_grad = - in_grad[ctc_loss::kData].get(s); - Tensor output_grad = - out_grad[ctc_loss::kOut].get(s); - - Tensor data_grad_computed = - out_data[ctc_loss::kGrad].get(s); - - Assign(data_grad, req[ctc_loss::kData], - mshadow::expr::broadcast<1>(output_grad, data_grad.shape_) * data_grad_computed); - } - - private: - CTCLossParam param_; - bool exceed_cudnn_limit; - -#if defined(__CUDACC__) && MXNET_USE_CUDNN == 1 && CUDNN_MAJOR >= 7 - cudnnDataType_t dtype_; - cudnnCTCLossDescriptor_t ctc_desc_; - cudnnTensorDescriptor_t prob_desc_, grad_desc_; - - inline virtual void cudnn_forward(const OpContext &ctx, - mshadow::Stream* s, - mshadow::Tensor data, - mshadow::Tensor costs, - mshadow::Tensor grad, - std::vector* data_lengths, - std::vector* label_lengths, - std::vector* packed_labels, - int max_seq_len, - int batch_size, - int alphabet_size, - bool req_grad) { - using namespace mshadow; - - // call cudnn to calculate ctc loss - dtype_ = CUDNN_DATA_FLOAT; - int dims[3], strides[3]; - size_t workspace_bytes; - int workspace_size; - dims[0] = max_seq_len; - dims[1] = batch_size; - dims[2] = alphabet_size; - strides[0] = batch_size*alphabet_size; - strides[1] = alphabet_size; - strides[2] = 1; - cudnnCTCLossAlgo_t ctc_algo = CUDNN_CTC_LOSS_ALGO_DETERMINISTIC; - CUDNN_CALL(cudnnSetTensorNdDescriptor(prob_desc_, - dtype_, - 3, - dims, - strides)); - CUDNN_CALL(cudnnSetTensorNdDescriptor(grad_desc_, - dtype_, - 3, - dims, - strides)); - CUDNN_CALL(cudnnGetCTCLossWorkspaceSize(s->dnn_handle_, - prob_desc_, - req_grad?grad_desc_:NULL, - packed_labels->data(), - label_lengths->data(), - data_lengths->data(), - ctc_algo, - ctc_desc_, - &workspace_bytes)); - workspace_size = (workspace_bytes + sizeof(real_t) - 1)/sizeof(real_t); - - Tensor temp_space = - ctx.requested[ctc_loss::kTempSpace].get_space_typed( - mshadow::Shape1(workspace_size+data.shape_.FlatTo1D()[0]), s); - - Tensor work_space(temp_space.dptr_, - mshadow::Shape1(workspace_size), s); - Tensor prob(temp_space.dptr_+workspace_size, - data.shape_, s); - - // since the input is activation before softmax and cudnn ctc takes softmax - // apply softmax to inputs first. - mxnet_op::Softmax( - s, data.dptr_, prob.dptr_, data.shape_, 2, 1.0); - - CUDNN_CALL(cudnnCTCLoss(s->dnn_handle_, - prob_desc_, - prob.dptr_, - packed_labels->data(), - label_lengths->data(), - data_lengths->data(), - costs.dptr_, - req_grad?grad_desc_:NULL, - req_grad?grad.dptr_:NULL, - ctc_algo, - ctc_desc_, - work_space.dptr_, - workspace_bytes)); - - if (req_grad) { - mxnet_op::SoftmaxGrad( - s, prob.dptr_, grad.dptr_, grad.dptr_, data.shape_, 2, 1.0); - Assign(grad, mxnet::kWriteInplace, grad * alphabet_size); - } - } -#endif // __CUDACC__ && CUDNN - - inline void baidu_forward(const OpContext &ctx, - mshadow::Stream* s, - mshadow::Tensor data, - mshadow::Tensor costs, - mshadow::Tensor grad, - std::vector* data_lengths, - std::vector* label_lengths, - std::vector* packed_labels, - int batch_size, - int alphabet_size, - bool req_grad) { - using namespace mshadow; - // allocate temporary workspace - size_t size_bytes; - bool gpu = data.kDevCPU ? false : true; - get_workspace_size(label_lengths, data_lengths, alphabet_size, - batch_size, gpu, &size_bytes); - - // round-up so there are enough elems in memory - int num_tmp_elems = (size_bytes + sizeof(real_t) - 1) / sizeof(real_t); - Tensor workspace = - ctx.requested[ctc_loss::kTempSpace].get_space_typed( - Shape1(num_tmp_elems), s); - - compute_ctc_cost(data, costs.dptr_, grad.dptr_, packed_labels->data(), - label_lengths->data(), data_lengths->data(), - workspace.dptr_, req_grad, - param_.blank_label == 0 ? 0 : (alphabet_size-1)); - } -}; // class CTCLossOp - -template -Operator *CreateOp(CTCLossParam param, int dtype); - -#if DMLC_USE_CXX11 -class CTCLossProp : public OperatorProperty { - public: - int NumVisibleOutputs() const override { return 1; } - - int NumOutputs() const override { return 2; } - - std::vector ListArguments() const override { - if (param_.use_data_lengths && param_.use_label_lengths) { - return {"data", "label", "data_lengths", "label_lengths"}; - } else if (param_.use_data_lengths) { - return {"data", "label", "data_lengths"}; - } else if (param_.use_label_lengths) { - return {"data", "label", "label_lengths"}; - } else { - return {"data", "label"}; - } - } - - std::vector ListOutputs() const override { - return {"output", "grad"}; - } - - void Init(const std::vector> &kwargs) override { - param_.Init(kwargs); - } - - std::map GetParams() const override { - return param_.__DICT__(); - } - - bool InferShape(std::vector *in_shape, std::vector *out_shape, - std::vector *aux_shape) const override { - using namespace mshadow; - index_t expected_inputs = 2+param_.use_data_lengths+param_.use_label_lengths; - CHECK_EQ(in_shape->size(), expected_inputs) - << "Expect " << expected_inputs << " inputs to the symbol."; - - const TShape &dshape = (*in_shape)[ctc_loss::kData]; - const TShape &lshape = (*in_shape)[ctc_loss::kLabel]; - CHECK_EQ(dshape.ndim(), 3U) << "The data array must be of rank 3."; - CHECK_EQ(lshape.ndim(), 2U) << "The labels array must be of rank 2."; - CHECK_EQ(dshape[1], lshape[0]) - << "The batch size for the labels and data arrays must be the same."; - if (param_.use_data_lengths) { - int kInputLength = 2; - const TShape &dlshape = (*in_shape)[kInputLength]; - CHECK_EQ(dlshape.ndim(), 1U) << "Data length array must be a vector."; - CHECK_EQ(dlshape[0], dshape[1]) - << "The batch size for the data and data lengths must be the same."; - } - if (param_.use_label_lengths) { - int kLabelLength = 2+param_.use_data_lengths; - const TShape &llshape = (*in_shape)[kLabelLength]; - CHECK_EQ(llshape.ndim(), 1U) << "Label length array must be a vector."; - CHECK_EQ(llshape[0], lshape[0]) - << "The batch size for the labels and label lengths must be the same."; - } - - CHECK_GE(dshape[0], lshape[1]) << "The max number of labels cannot exceed " - "the maximum sequence length of the " - "data."; - - TShape oshape(1); - oshape[0] = dshape[1]; // batch size - out_shape->clear(); - out_shape->push_back(oshape); // forward output - out_shape->push_back(dshape); // grad output - return true; - } - - bool InferType(std::vector *in_type, - std::vector *out_type, - std::vector *aux_type) const override { - CHECK_LE(in_type->size(), this->ListArguments().size()); - int dtype = (*in_type)[ctc_loss::kData]; - CHECK_NE(dtype, -1) << "Input data must have specified type"; - - out_type->clear(); - out_type->push_back(dtype); // forward output - out_type->push_back(dtype); // grad output - return true; - } - - OperatorProperty *Copy() const override { - auto ptr = new CTCLossProp(); - ptr->param_ = param_; - return ptr; - } - - std::string TypeString() const override { return "_contrib_CTCLoss"; } - - std::vector ForwardResource( - const std::vector &in_shape) const override { - return {ResourceRequest::kTempSpace}; - } - - std::vector DeclareBackwardDependency( - const std::vector &out_grad, const std::vector &in_data, - const std::vector &out_data) const override { - return {out_grad[ctc_loss::kOut], out_data[ctc_loss::kGrad]}; - } - - Operator *CreateOperator(Context ctx) const override { - LOG(FATAL) << "Not Implemented."; - return NULL; - } - - Operator *CreateOperatorEx(Context ctx, std::vector *in_shape, - std::vector *in_type) const override; - - private: - CTCLossParam param_; -}; // class CTCLossProp -#endif // DMLC_USE_CXX11 -} // namespace op -} // namespace mxnet -#endif // MXNET_OPERATOR_CONTRIB_CTC_LOSS_INL_H_ diff --git a/src/operator/contrib/ctc_loss.cc b/src/operator/contrib/ctc_loss.cc deleted file mode 100644 index 32e8e629f090..000000000000 --- a/src/operator/contrib/ctc_loss.cc +++ /dev/null @@ -1,130 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, - * software distributed under the License is distributed on an - * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY - * KIND, either express or implied. See the License for the - * specific language governing permissions and limitations - * under the License. - */ - -/*! - * Copyright (c) 2015 by Contributors - * \file ctc_loss.cc - * \brief - * \author Sebastian Bodenstein -*/ - -#include "./ctc_loss-inl.h" -#include "./ctc_include/detail/cpu_ctc.h" - -namespace mshadow { - -template -ctcStatus_t compute_ctc_cost(const Tensor activations, - DType *costs, DType *grads, int *labels, - int *label_lengths, int *data_lengths, - void *workspace, int train, int blank_label) { - int minibatch = static_cast(activations.size(1)); - int alphabet_size = static_cast(activations.size(2)); - mxnet_warpctc::CpuCTC ctc(alphabet_size, minibatch, workspace, blank_label); - if (train) { - return ctc.cost_and_grad(activations.dptr_, grads, costs, labels, - label_lengths, data_lengths); - } else { - return ctc.score_forward(activations.dptr_, costs, labels, label_lengths, - data_lengths); - } -} - -} // namespace mshadow - -namespace mxnet { -namespace op { -template <> -Operator *CreateOp(CTCLossParam param, int dtype) { - return new CTCLossOp(param); -} - -// DO_BIND_DISPATCH comes from operator_common.h -Operator *CTCLossProp::CreateOperatorEx(Context ctx, - std::vector *in_shape, - std::vector *in_type) const { - std::vector out_shape, aux_shape; - std::vector out_type, aux_type; - CHECK(InferType(in_type, &out_type, &aux_type)); - CHECK(InferShape(in_shape, &out_shape, &aux_shape)); - DO_BIND_DISPATCH(CreateOp, param_, (*in_type)[0]); -} - -DMLC_REGISTER_PARAMETER(CTCLossParam); - -MXNET_REGISTER_OP_PROPERTY(_contrib_CTCLoss, CTCLossProp) - .describe(R"code(Connectionist Temporal Classification Loss. - -The shapes of the inputs and outputs: - -- **data**: `(sequence_length, batch_size, alphabet_size)` -- **label**: `(batch_size, label_sequence_length)` -- **out**: `(batch_size)` - -The `data` tensor consists of sequences of activation vectors (without applying softmax), -with i-th channel in the last dimension corresponding to i-th label -for i between 0 and alphabet_size-1 (i.e always 0-indexed). -Alphabet size should include one additional value reserved for blank label. -When `blank_label` is ``"first"``, the ``0``-th channel is be reserved for -activation of blank label, or otherwise if it is "last", ``(alphabet_size-1)``-th channel should be -reserved for blank label. - -``label`` is an index matrix of integers. When `blank_label` is ``"first"``, -the value 0 is then reserved for blank label, and should not be passed in this matrix. Otherwise, -when `blank_label` is ``"last"``, the value `(alphabet_size-1)` is reserved for blank label. - -If a sequence of labels is shorter than *label_sequence_length*, use the special -padding value at the end of the sequence to conform it to the correct -length. The padding value is `0` when `blank_label` is ``"first"``, and `-1` otherwise. - -For example, suppose the vocabulary is `[a, b, c]`, and in one batch we have three sequences -'ba', 'cbb', and 'abac'. When `blank_label` is ``"first"``, we can index the labels as -`{'a': 1, 'b': 2, 'c': 3}`, and we reserve the 0-th channel for blank label in data tensor. -The resulting `label` tensor should be padded to be:: - - [[2, 1, 0, 0], [3, 2, 2, 0], [1, 2, 1, 3]] - -When `blank_label` is ``"last"``, we can index the labels as -`{'a': 0, 'b': 1, 'c': 2}`, and we reserve the channel index 3 for blank label in data tensor. -The resulting `label` tensor should be padded to be:: - - [[1, 0, -1, -1], [2, 1, 1, -1], [0, 1, 0, 2]] - -``out`` is a list of CTC loss values, one per example in the batch. - -See *Connectionist Temporal Classification: Labelling Unsegmented -Sequence Data with Recurrent Neural Networks*, A. Graves *et al*. for more -information on the definition and the algorithm. - -)code" ADD_FILELINE) - .add_argument("data", "NDArray-or-Symbol", "Input data to the ctc_loss op.") - .add_argument("label", "NDArray-or-Symbol", - "Ground-truth labels for the loss.") - .add_argument("data_lengths", "NDArray-or-Symbol", - "Lengths of data for each of the samples. Only required " - "when use_data_lengths is true.") - .add_argument("label_lengths", "NDArray-or-Symbol", - "Lengths of labels for each of the samples. Only required " - "when use_label_lengths is true.") - .add_arguments(CTCLossParam::__FIELDS__()); - -NNVM_REGISTER_OP(_contrib_CTCLoss).add_alias("_contrib_ctc_loss"); - -} // namespace op -} // namespace mxnet diff --git a/src/operator/contrib/ctc_loss.cu b/src/operator/contrib/ctc_loss.cu deleted file mode 100644 index 3f5f12ca4394..000000000000 --- a/src/operator/contrib/ctc_loss.cu +++ /dev/null @@ -1,61 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements. See the NOTICE file - * distributed with this work for additional information - * regarding copyright ownership. The ASF licenses this file - * to you under the Apache License, Version 2.0 (the - * "License"); you may not use this file except in compliance - * with the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, - * software distributed under the License is distributed on an - * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY - * KIND, either express or implied. See the License for the - * specific language governing permissions and limitations - * under the License. - */ - -/*! - * Copyright (c) 2015 by Contributors - * \file ctc_loss.cu - * \brief - * \author Sebastian Bodenstein -*/ -#include -#include "./ctc_loss-inl.h" -#include "./ctc_include/detail/gpu_ctc.h" - -namespace mshadow { - -template -ctcStatus_t compute_ctc_cost(const Tensor activations, - DType *costs, DType *grads, int *labels, - int *label_lengths, int *input_lengths, - void *workspace, int train, int blank_label) { - int minibatch = static_cast(activations.size(1)); - int alphabet_size = static_cast(activations.size(2)); - mxnet_warpctc::GpuCTC ctc(alphabet_size, minibatch, workspace, - activations.stream_->stream_, blank_label); - if (train) - return ctc.cost_and_grad(activations.dptr_, grads, costs, labels, - label_lengths, input_lengths); - else - return ctc.score_forward(activations.dptr_, costs, labels, - label_lengths, input_lengths); -} - -} // namespace mshadow - -//////////////////////////////////////////////////////////////////////////////// - -namespace mxnet { -namespace op { -template <> -Operator *CreateOp(CTCLossParam param, int dtype) { - return new CTCLossOp(param); -} - -} // namespace op -} // namespace mxnet diff --git a/src/operator/nn/ctc_loss.cc b/src/operator/nn/ctc_loss.cc index d3996a27677b..5b243870d2ab 100644 --- a/src/operator/nn/ctc_loss.cc +++ b/src/operator/nn/ctc_loss.cc @@ -48,7 +48,8 @@ namespace op { DMLC_REGISTER_PARAMETER(CTCLossOpParam); -NNVM_REGISTER_OP(ctc_loss) +NNVM_REGISTER_OP(CTCLoss) +.add_alias("ctc_loss") .describe(R"code(Connectionist Temporal Classification Loss. The shapes of the inputs and outputs: diff --git a/src/operator/nn/ctc_loss.cu b/src/operator/nn/ctc_loss.cu index a83337b6ee2f..c4f721728a21 100644 --- a/src/operator/nn/ctc_loss.cu +++ b/src/operator/nn/ctc_loss.cu @@ -49,7 +49,8 @@ ctcStatus_t compute_ctc_cost(const Tensor activations, namespace mxnet { namespace op { -NNVM_REGISTER_OP(ctc_loss) +NNVM_REGISTER_OP(CTCLoss) +.add_alias("ctc_loss") .set_attr("FCompute", CTCLossOpForward); NNVM_REGISTER_OP(_backward_ctc_loss) diff --git a/tests/python/unittest/test_loss.py b/tests/python/unittest/test_loss.py index 24cc747a3087..8b713927677f 100644 --- a/tests/python/unittest/test_loss.py +++ b/tests/python/unittest/test_loss.py @@ -224,6 +224,8 @@ def test_ctc_loss_train(): mod.fit(data_iter, num_epoch=200, optimizer_params={'learning_rate': 0.01}, initializer=mx.init.Xavier(magnitude=2), eval_metric=mx.metric.Loss(), optimizer='adam') + score = mod.score(data_iter, eval_metric=mx.metric.Loss()) + print(score) assert mod.score(data_iter, eval_metric=mx.metric.Loss())[0][1] < 10 @@ -350,5 +352,6 @@ def test_triplet_loss(): if __name__ == '__main__': - import nose - nose.runmodule() + test_ctc_loss_train() + #import nose + #nose.runmodule() From b4b45b20269972a0eeb15393c790edc607e7ba44 Mon Sep 17 00:00:00 2001 From: Lin Yuan Date: Tue, 2 Oct 2018 09:29:40 -0700 Subject: [PATCH 12/17] revert a change by mistake --- tests/python/unittest/test_loss.py | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/tests/python/unittest/test_loss.py b/tests/python/unittest/test_loss.py index 8b713927677f..3312fd350fb7 100644 --- a/tests/python/unittest/test_loss.py +++ b/tests/python/unittest/test_loss.py @@ -225,7 +225,7 @@ def test_ctc_loss_train(): initializer=mx.init.Xavier(magnitude=2), eval_metric=mx.metric.Loss(), optimizer='adam') score = mod.score(data_iter, eval_metric=mx.metric.Loss()) - print(score) + print(score) assert mod.score(data_iter, eval_metric=mx.metric.Loss())[0][1] < 10 @@ -352,6 +352,5 @@ def test_triplet_loss(): if __name__ == '__main__': - test_ctc_loss_train() - #import nose - #nose.runmodule() + import nose + nose.runmodule() From 8b36f971934275f12e851f4883cd76d494837368 Mon Sep 17 00:00:00 2001 From: Lin Yuan Date: Tue, 2 Oct 2018 14:34:03 -0700 Subject: [PATCH 13/17] Fix a bug in kDevCPU --- src/operator/nn/ctc_loss-inl.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/operator/nn/ctc_loss-inl.h b/src/operator/nn/ctc_loss-inl.h index b1699e0536e6..754cf8471b5d 100644 --- a/src/operator/nn/ctc_loss-inl.h +++ b/src/operator/nn/ctc_loss-inl.h @@ -345,7 +345,7 @@ void CTCLossOpForward(const nnvm::NodeAttrs& attrs, size_t size_bytes; get_workspace_size(&label_lengths, &data_lengths, alphabet_size, - batch_size, data.kDevCPU ? true : false, &size_bytes); + batch_size, data.kDevCPU ? false : true, &size_bytes); // round-up so there are enough elems in memory int num_tmp_elems = (size_bytes + sizeof(real_t) - 1) / sizeof(real_t); From 8555743090ec3e9f0ef8cdc3aedfce5c74f3a435 Mon Sep 17 00:00:00 2001 From: Lin Yuan Date: Tue, 2 Oct 2018 14:37:58 -0700 Subject: [PATCH 14/17] revert change by mistake --- tests/python/unittest/test_loss.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/tests/python/unittest/test_loss.py b/tests/python/unittest/test_loss.py index 3312fd350fb7..24cc747a3087 100644 --- a/tests/python/unittest/test_loss.py +++ b/tests/python/unittest/test_loss.py @@ -224,8 +224,6 @@ def test_ctc_loss_train(): mod.fit(data_iter, num_epoch=200, optimizer_params={'learning_rate': 0.01}, initializer=mx.init.Xavier(magnitude=2), eval_metric=mx.metric.Loss(), optimizer='adam') - score = mod.score(data_iter, eval_metric=mx.metric.Loss()) - print(score) assert mod.score(data_iter, eval_metric=mx.metric.Loss())[0][1] < 10 From cf8c10768fbc7aba31915e572f0ef11e0d5dc615 Mon Sep 17 00:00:00 2001 From: Lin Yuan Date: Tue, 2 Oct 2018 15:19:44 -0700 Subject: [PATCH 15/17] add alias to make it backward compatible --- src/operator/nn/ctc_loss.cc | 2 ++ src/operator/nn/ctc_loss.cu | 2 ++ tests/python/unittest/test_operator.py | 12 ++++++------ 3 files changed, 10 insertions(+), 6 deletions(-) diff --git a/src/operator/nn/ctc_loss.cc b/src/operator/nn/ctc_loss.cc index 5b243870d2ab..c381677b3ce0 100644 --- a/src/operator/nn/ctc_loss.cc +++ b/src/operator/nn/ctc_loss.cc @@ -50,6 +50,8 @@ DMLC_REGISTER_PARAMETER(CTCLossOpParam); NNVM_REGISTER_OP(CTCLoss) .add_alias("ctc_loss") +.add_alias("_contrib_CTCLoss") +.add_alias("_contrib_ctc_loss") .describe(R"code(Connectionist Temporal Classification Loss. The shapes of the inputs and outputs: diff --git a/src/operator/nn/ctc_loss.cu b/src/operator/nn/ctc_loss.cu index c4f721728a21..a4491bf6986e 100644 --- a/src/operator/nn/ctc_loss.cu +++ b/src/operator/nn/ctc_loss.cu @@ -51,6 +51,8 @@ namespace op { NNVM_REGISTER_OP(CTCLoss) .add_alias("ctc_loss") +.add_alias("_contrib_ctc_loss") +.add_alias("_contrib_CTCLoss") .set_attr("FCompute", CTCLossOpForward); NNVM_REGISTER_OP(_backward_ctc_loss) diff --git a/tests/python/unittest/test_operator.py b/tests/python/unittest/test_operator.py index 6e382ebd9b96..f8e7b2baf471 100644 --- a/tests/python/unittest/test_operator.py +++ b/tests/python/unittest/test_operator.py @@ -4611,12 +4611,12 @@ def check_ctc_loss_grad(blank_label): # from tf label = mx.nd.array(labels) data.attach_grad() with mx.autograd.record(): - l = mx.ndarray.ctc_loss(data, label, - use_data_lengths=True, - use_label_lengths=True, - data_lengths=mx.nd.array(seq_lens), - label_lengths=mx.nd.array(label_lens), - blank_label=blank_label) + l = mx.ndarray.CTCLoss(data, label, + use_data_lengths=True, + use_label_lengths=True, + data_lengths=mx.nd.array(seq_lens), + label_lengths=mx.nd.array(label_lens), + blank_label=blank_label) l.backward() assert_almost_equal(l.asnumpy(), loss_truth, atol=1e-5, rtol=1e-5) assert_almost_equal(data.grad.asnumpy(), grad_truth, atol=1e-5, rtol=1e-5) From f53c390bb8656fd60130d6a480b415911535bdb3 Mon Sep 17 00:00:00 2001 From: Lin Yuan Date: Fri, 5 Oct 2018 14:54:36 -0700 Subject: [PATCH 16/17] add unit test for backward compatibility --- tests/python/unittest/test_operator.py | 108 +++++++++++++++++++++++++ 1 file changed, 108 insertions(+) diff --git a/tests/python/unittest/test_operator.py b/tests/python/unittest/test_operator.py index 3a669303e902..01fb96b744b8 100644 --- a/tests/python/unittest/test_operator.py +++ b/tests/python/unittest/test_operator.py @@ -4503,6 +4503,29 @@ def check_ctc_loss(acts, labels, loss_truth): outTrain = exe.outputs[0] # make sure losses calculated with both modes are the same assert_almost_equal(outTest.asnumpy(), outTrain.asnumpy()) + + # test against ground truth, if available + if loss_truth is not None: + assert_almost_equal(outTest.asnumpy(), loss_truth) + # test grad + check_numeric_gradient(ctc, [acts, labels], grad_nodes=['input'], rtol=0.05, atol=1e-3) + +def check_contrib_ctc_loss(acts, labels, loss_truth): + in_var = mx.sym.Variable('input') + labels_var = mx.sym.Variable('labels') + ctc = mx.sym.contrib.ctc_loss(in_var, labels_var) + acts_nd = mx.nd.array(acts, ctx=default_context()) + labels_nd = mx.nd.array(labels, ctx=default_context()) + exe = ctc.bind(ctx=default_context(), args=[acts_nd, labels_nd]) + # test forward with grad calc + exe.forward(is_train=True) + outTest = exe.outputs[0] + # test forward without grad calc + exe.forward(is_train=False) + outTrain = exe.outputs[0] + # make sure losses calculated with both modes are the same + assert_almost_equal(outTest.asnumpy(), outTrain.asnumpy()) + # test against ground truth, if available if loss_truth is not None: assert_almost_equal(outTest.asnumpy(), loss_truth) @@ -4520,6 +4543,8 @@ def test_ctc_loss(): labels = np.array([[2, 3, 0], [2, 3, 0]]) true_loss = np.array([4.04789, 4.04789], dtype=np.float32) # from Torch check_ctc_loss(acts, labels, true_loss) + check_contrib_ctc_loss(acts, labels, true_loss) + # Test 2: acts2 = np.array([ [[-5, -4, -3, -2, -1], [1.2, 3.4, 1.2, -0.1, -2.34]], @@ -4528,11 +4553,13 @@ def test_ctc_loss(): labels2 = np.array([[2, 3, 1], [2, 0, 0]], dtype=np.float32) true_loss = np.array([7.3557, 5.4091], dtype=np.float32) # from Torch check_ctc_loss(acts2, labels2, true_loss) + check_contrib_ctc_loss(acts2, labels2, true_loss) # Test 3: check use integer type as label labels3 = np.array([[2, 3, 1], [2, 0, 0]], dtype=np.int32) true_loss = np.array([7.3557, 5.4091], dtype=np.float32) # from Torch check_ctc_loss(acts2, labels3, true_loss) + check_contrib_ctc_loss(acts2, labels3, true_loss) @with_seed() def test_ctc_loss_with_large_classes(): @@ -4634,8 +4661,89 @@ def check_ctc_loss_grad(blank_label): # from tf assert_almost_equal(l.asnumpy(), loss_truth, atol=1e-5, rtol=1e-5) assert_almost_equal(data.grad.asnumpy(), grad_truth, atol=1e-5, rtol=1e-5) + def check_contrib_ctc_loss_grad(blank_label): # from tf + vocab_size = 5 + max_label_len = 5 + padding_mask = -1+ (blank_label=='first') + + targets_0 = [0, 1, 2, 1, 0] + loss_log_prob_0 = -3.34211 + input_prob_matrix_0 = np.asarray( + [[0.633766, 0.221185, 0.0917319, 0.0129757, 0.0142857, 0.0260553], + [0.111121, 0.588392, 0.278779, 0.0055756, 0.00569609, 0.010436], + [0.0357786, 0.633813, 0.321418, 0.00249248, 0.00272882, 0.0037688], + [0.0663296, 0.643849, 0.280111, 0.00283995, 0.0035545, 0.00331533], + [0.458235, 0.396634, 0.123377, 0.00648837, 0.00903441, 0.00623107]], + dtype=np.float32) + gradient_log_prob_0 = np.asarray( + [[-0.366234, 0.221185, 0.0917319, 0.0129757, 0.0142857, 0.0260553], + [0.111121, -0.411608, 0.278779, 0.0055756, 0.00569609, 0.010436], + [0.0357786, 0.633813, -0.678582, 0.00249248, 0.00272882, 0.0037688], + [0.0663296, -0.356151, 0.280111, 0.00283995, 0.0035545, 0.00331533], + [-0.541765, 0.396634, 0.123377, 0.00648837, 0.00903441, 0.00623107]], + dtype=np.float32) + + targets_1 = [0, 1, 1, 0] + loss_log_prob_1 = -5.42262 + input_prob_matrix_1 = np.asarray( + [[0.30176, 0.28562, 0.0831517, 0.0862751, 0.0816851, 0.161508], + [0.24082, 0.397533, 0.0557226, 0.0546814, 0.0557528, 0.19549], + [0.230246, 0.450868, 0.0389607, 0.038309, 0.0391602, 0.202456], + [0.280884, 0.429522, 0.0326593, 0.0339046, 0.0326856, 0.190345], + [0.423286, 0.315517, 0.0338439, 0.0393744, 0.0339315, 0.154046]], + dtype=np.float32) + gradient_log_prob_1 = np.asarray( + [[-0.69824, 0.28562, 0.0831517, 0.0862751, 0.0816851, 0.161508], + [0.24082, -0.602467, 0.0557226, 0.0546814, 0.0557528, 0.19549], + [0.230246, 0.450868, 0.0389607, 0.038309, 0.0391602, -0.797544], + [0.280884, -0.570478, 0.0326593, 0.0339046, 0.0326856, 0.190345], + [-0.576714, 0.315517, 0.0338439, 0.0393744, 0.0339315, 0.154046]], + dtype=np.float32) + + inputs = [ + np.vstack( + [input_prob_matrix_0[t, :], input_prob_matrix_1[t, :]]) + for t in range(5) + ] + 2 * [np.nan * np.ones((2, vocab_size+1), np.float32)] + inputs = np.log(np.asarray(inputs, dtype=np.float32)) + + grad_truth = np.array([ + np.vstack( + [gradient_log_prob_0[t, :], gradient_log_prob_1[t, :]]) + for t in range(5) + ] + 2 * [np.zeros((2, vocab_size+1), np.float32)]) + + if blank_label == 'first': + inputs = np.roll(inputs, 1, axis=2) + grad_truth = np.roll(grad_truth, 1, axis=2) + + labels = (np.asarray([x + [padding_mask]*(max_label_len-len(x)) + for x in [targets_0, targets_1]])+(blank_label == 'first')) + + seq_lens = np.array([5, 5], dtype=np.int32) + label_lens = np.array([5, 4], dtype=np.int32) + loss_truth = np.array([-loss_log_prob_0, -loss_log_prob_1], np.float32) + + with default_context(): + data = mx.nd.array(inputs) + label = mx.nd.array(labels) + data.attach_grad() + with mx.autograd.record(): + l = mx.contrib.ndarray.CTCLoss(data, label, + use_data_lengths=True, + use_label_lengths=True, + data_lengths=mx.nd.array(seq_lens), + label_lengths=mx.nd.array(label_lens), + blank_label=blank_label) + l.backward() + assert_almost_equal(l.asnumpy(), loss_truth, atol=1e-5, rtol=1e-5) + assert_almost_equal(data.grad.asnumpy(), grad_truth, atol=1e-5, rtol=1e-5) + + check_ctc_loss_grad('first') check_ctc_loss_grad('last') + check_contrib_ctc_loss_grad('first') + check_contrib_ctc_loss_grad('last') @with_seed() From 1f09bf85c85d31d084a93bd136542a164d02ca98 Mon Sep 17 00:00:00 2001 From: Lin Yuan Date: Fri, 5 Oct 2018 14:59:27 -0700 Subject: [PATCH 17/17] linting --- tests/python/unittest/test_operator.py | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/tests/python/unittest/test_operator.py b/tests/python/unittest/test_operator.py index 01fb96b744b8..5332517fa680 100644 --- a/tests/python/unittest/test_operator.py +++ b/tests/python/unittest/test_operator.py @@ -4510,6 +4510,7 @@ def check_ctc_loss(acts, labels, loss_truth): # test grad check_numeric_gradient(ctc, [acts, labels], grad_nodes=['input'], rtol=0.05, atol=1e-3) +# check contrib operator for backward compatibility def check_contrib_ctc_loss(acts, labels, loss_truth): in_var = mx.sym.Variable('input') labels_var = mx.sym.Variable('labels') @@ -4661,6 +4662,7 @@ def check_ctc_loss_grad(blank_label): # from tf assert_almost_equal(l.asnumpy(), loss_truth, atol=1e-5, rtol=1e-5) assert_almost_equal(data.grad.asnumpy(), grad_truth, atol=1e-5, rtol=1e-5) + # check contrib operator for backward compatibility def check_contrib_ctc_loss_grad(blank_label): # from tf vocab_size = 5 max_label_len = 5 @@ -4730,11 +4732,11 @@ def check_contrib_ctc_loss_grad(blank_label): # from tf data.attach_grad() with mx.autograd.record(): l = mx.contrib.ndarray.CTCLoss(data, label, - use_data_lengths=True, - use_label_lengths=True, - data_lengths=mx.nd.array(seq_lens), - label_lengths=mx.nd.array(label_lens), - blank_label=blank_label) + use_data_lengths=True, + use_label_lengths=True, + data_lengths=mx.nd.array(seq_lens), + label_lengths=mx.nd.array(label_lens), + blank_label=blank_label) l.backward() assert_almost_equal(l.asnumpy(), loss_truth, atol=1e-5, rtol=1e-5) assert_almost_equal(data.grad.asnumpy(), grad_truth, atol=1e-5, rtol=1e-5)