MCPcopy Create free account
hub / github.com/DeepRec-AI/DeepRec / DoFusedConvolveImpl

Method DoFusedConvolveImpl

tensorflow/stream_executor/cuda/cuda_dnn.cc:4239–4457  ·  view source on GitHub ↗

Source from the content-addressed store, hash-verified

4237template <typename ElementType, typename BiasType, typename ScaleType,
4238 typename OutputType>
4239port::Status CudnnSupport::DoFusedConvolveImpl(
4240 Stream* stream, const dnn::BatchDescriptor& conv_input_descriptor,
4241 const DeviceMemory<ElementType>& conv_input_data,
4242 ScaleType conv_input_scale, const dnn::FilterDescriptor& filter_descriptor,
4243 const DeviceMemory<ElementType>& filter_data,
4244 const dnn::ConvolutionDescriptor& convolution_descriptor,
4245 const DeviceMemory<OutputType>& side_input_data, ScaleType side_input_scale,
4246 const dnn::BatchDescriptor& bias_descriptor,
4247 const DeviceMemory<BiasType>& biases, dnn::ActivationMode activation_mode,
4248 const dnn::BatchDescriptor& output_descriptor,
4249 DeviceMemory<OutputType>* output_data, dnn::DataType accumulator_type,
4250 ScratchAllocator* scratch_allocator,
4251 const dnn::ExecutionPlanConfig &plan_config,
4252 dnn::ProfileExecutionPlanResult* output_profile_result) {
4253#if CUDNN_VERSION >= 8100
4254 auto cudnn = cudnn_->GetHandle(parent_, stream);
4255
4256 absl::optional<dnn::ExecutionPlanDesc> plan_or = plan_config.plan();
4257 absl::optional<dnn::ExecutionPlanDesc> plan_no_scratch_or =
4258 plan_config.plan_no_scratch();
4259
4260 std::unique_ptr<cudnn_frontend::ExecutionPlan> current_plan;
4261 if (!plan_or.has_value()) {
4262 // TODO(kaixih): the filtered_configs cannot be reused, since the engine
4263 // config's ownership will be moved to plan during build(). Therefore, we
4264 // need to create a separate filtered_configs here. This and the following
4265 // loop could be replaced with cudnn frontend api when it is available.
4266 SE_ASSIGN_OR_RETURN(
4267 std::unique_ptr<cudnn_frontend::OperationGraph> op_graph,
4268 GetCudnnFusedOperationGraph(
4269 dnn::ConvolutionKind::FORWARD, accumulator_type, stream,
4270 conv_input_descriptor, filter_descriptor, bias_descriptor,
4271 output_descriptor, convolution_descriptor, cudnn));
4272
4273 auto heuristics = cudnn_frontend::EngineHeuristicsBuilder()
4274 .setOperationGraph(*op_graph)
4275 .setHeurMode(CUDNN_HEUR_MODE_INSTANT)
4276 .build();
4277 RETURN_MSG_IF_CUDNN_ERROR(heuristics);
4278
4279 auto fallback =
4280 cudnn_frontend::EngineFallbackListBuilder()
4281 .setOperationGraph(*op_graph)
4282 .setOperation(
4283 CUDNN_BACKEND_OPERATION_CONVOLUTION_FORWARD_DESCRIPTOR)
4284 .build();
4285 RETURN_MSG_IF_CUDNN_ERROR(fallback);
4286
4287 auto engine_count = heuristics.getEngineConfigCount();
4288 auto &engine_config = heuristics.getEngineConfig(engine_count);
4289 auto &fallback_list = fallback.getFallbackList();
4290
4291 cudnn_frontend::EngineConfigList filtered_configs;
4292 if (tensorflow::tensor_float_32_execution_enabled()) {
4293 if (stream_executor::cuda::RequireCuDNNDeterminism()) {
4294 cudnn_frontend::filter(engine_config, filtered_configs,
4295 isNonDeterministicOrIsDownConverting);
4296 cudnn_frontend::filter(fallback_list, filtered_configs,

Callers

nothing calls this directly

Calls 15

RequireCuDNNDeterminismFunction · 0.85
parseFunction · 0.85
AsGpuStreamFunction · 0.85
ToCudnnDataTypeFunction · 0.85
IsTensorMathOpSetFunction · 0.85
plan_no_scratchMethod · 0.80
get_statusMethod · 0.80
exec_plan_idMethod · 0.80
exec_plan_descMethod · 0.80

Tested by

no test coverage detected