| 89 | |
| 90 | template <typename Dtype> |
| 91 | void CuDNNConvolutionLayer<Dtype>::Reshape( |
| 92 | const vector<Blob<Dtype>*>& bottom, const vector<Blob<Dtype>*>& top) { |
| 93 | ConvolutionLayer<Dtype>::Reshape(bottom, top); |
| 94 | CHECK_EQ(2, this->num_spatial_axes_) |
| 95 | << "CuDNNConvolution input must have 2 spatial axes " |
| 96 | << "(e.g., height and width). " |
| 97 | << "Use 'engine: CAFFE' for general ND convolution."; |
| 98 | bottom_offset_ = this->bottom_dim_ / this->group_; |
| 99 | top_offset_ = this->top_dim_ / this->group_; |
| 100 | const int height = bottom[0]->shape(this->channel_axis_ + 1); |
| 101 | const int width = bottom[0]->shape(this->channel_axis_ + 2); |
| 102 | const int height_out = top[0]->shape(this->channel_axis_ + 1); |
| 103 | const int width_out = top[0]->shape(this->channel_axis_ + 2); |
| 104 | const int* pad_data = this->pad_.cpu_data(); |
| 105 | const int pad_h = pad_data[0]; |
| 106 | const int pad_w = pad_data[1]; |
| 107 | const int* stride_data = this->stride_.cpu_data(); |
| 108 | const int stride_h = stride_data[0]; |
| 109 | const int stride_w = stride_data[1]; |
| 110 | |
| 111 | // Specify workspace limit for kernels directly until we have a |
| 112 | // planning strategy and a rewrite of Caffe's GPU memory mangagement |
| 113 | size_t workspace_limit_bytes = 8*1024*1024; |
| 114 | |
| 115 | for (int i = 0; i < bottom.size(); i++) { |
| 116 | cudnn::setTensor4dDesc<Dtype>(&bottom_descs_[i], |
| 117 | this->num_, |
| 118 | this->channels_ / this->group_, height, width, |
| 119 | this->channels_ * height * width, |
| 120 | height * width, width, 1); |
| 121 | cudnn::setTensor4dDesc<Dtype>(&top_descs_[i], |
| 122 | this->num_, |
| 123 | this->num_output_ / this->group_, height_out, width_out, |
| 124 | this->num_output_ * this->out_spatial_dim_, |
| 125 | this->out_spatial_dim_, width_out, 1); |
| 126 | cudnn::setConvolutionDesc<Dtype>(&conv_descs_[i], bottom_descs_[i], |
| 127 | filter_desc_, pad_h, pad_w, |
| 128 | stride_h, stride_w); |
| 129 | |
| 130 | // choose forward and backward algorithms + workspace(s) |
| 131 | CUDNN_CHECK(cudnnGetConvolutionForwardAlgorithm(handle_[0], |
| 132 | bottom_descs_[i], |
| 133 | filter_desc_, |
| 134 | conv_descs_[i], |
| 135 | top_descs_[i], |
| 136 | CUDNN_CONVOLUTION_FWD_SPECIFY_WORKSPACE_LIMIT, |
| 137 | workspace_limit_bytes, |
| 138 | &fwd_algo_[i])); |
| 139 | |
| 140 | CUDNN_CHECK(cudnnGetConvolutionForwardWorkspaceSize(handle_[0], |
| 141 | bottom_descs_[i], |
| 142 | filter_desc_, |
| 143 | conv_descs_[i], |
| 144 | top_descs_[i], |
| 145 | fwd_algo_[i], |
| 146 | &(workspace_fwd_sizes_[i]))); |
| 147 | |
| 148 | // choose backward algorithm for filter |