| 4237 | template <typename ElementType, typename BiasType, typename ScaleType, |
| 4238 | typename OutputType> |
| 4239 | port::Status CudnnSupport::DoFusedConvolveImpl( |
| 4240 | Stream* stream, const dnn::BatchDescriptor& conv_input_descriptor, |
| 4241 | const DeviceMemory<ElementType>& conv_input_data, |
| 4242 | ScaleType conv_input_scale, const dnn::FilterDescriptor& filter_descriptor, |
| 4243 | const DeviceMemory<ElementType>& filter_data, |
| 4244 | const dnn::ConvolutionDescriptor& convolution_descriptor, |
| 4245 | const DeviceMemory<OutputType>& side_input_data, ScaleType side_input_scale, |
| 4246 | const dnn::BatchDescriptor& bias_descriptor, |
| 4247 | const DeviceMemory<BiasType>& biases, dnn::ActivationMode activation_mode, |
| 4248 | const dnn::BatchDescriptor& output_descriptor, |
| 4249 | DeviceMemory<OutputType>* output_data, dnn::DataType accumulator_type, |
| 4250 | ScratchAllocator* scratch_allocator, |
| 4251 | const dnn::ExecutionPlanConfig &plan_config, |
| 4252 | dnn::ProfileExecutionPlanResult* output_profile_result) { |
| 4253 | #if CUDNN_VERSION >= 8100 |
| 4254 | auto cudnn = cudnn_->GetHandle(parent_, stream); |
| 4255 | |
| 4256 | absl::optional<dnn::ExecutionPlanDesc> plan_or = plan_config.plan(); |
| 4257 | absl::optional<dnn::ExecutionPlanDesc> plan_no_scratch_or = |
| 4258 | plan_config.plan_no_scratch(); |
| 4259 | |
| 4260 | std::unique_ptr<cudnn_frontend::ExecutionPlan> current_plan; |
| 4261 | if (!plan_or.has_value()) { |
| 4262 | // TODO(kaixih): the filtered_configs cannot be reused, since the engine |
| 4263 | // config's ownership will be moved to plan during build(). Therefore, we |
| 4264 | // need to create a separate filtered_configs here. This and the following |
| 4265 | // loop could be replaced with cudnn frontend api when it is available. |
| 4266 | SE_ASSIGN_OR_RETURN( |
| 4267 | std::unique_ptr<cudnn_frontend::OperationGraph> op_graph, |
| 4268 | GetCudnnFusedOperationGraph( |
| 4269 | dnn::ConvolutionKind::FORWARD, accumulator_type, stream, |
| 4270 | conv_input_descriptor, filter_descriptor, bias_descriptor, |
| 4271 | output_descriptor, convolution_descriptor, cudnn)); |
| 4272 | |
| 4273 | auto heuristics = cudnn_frontend::EngineHeuristicsBuilder() |
| 4274 | .setOperationGraph(*op_graph) |
| 4275 | .setHeurMode(CUDNN_HEUR_MODE_INSTANT) |
| 4276 | .build(); |
| 4277 | RETURN_MSG_IF_CUDNN_ERROR(heuristics); |
| 4278 | |
| 4279 | auto fallback = |
| 4280 | cudnn_frontend::EngineFallbackListBuilder() |
| 4281 | .setOperationGraph(*op_graph) |
| 4282 | .setOperation( |
| 4283 | CUDNN_BACKEND_OPERATION_CONVOLUTION_FORWARD_DESCRIPTOR) |
| 4284 | .build(); |
| 4285 | RETURN_MSG_IF_CUDNN_ERROR(fallback); |
| 4286 | |
| 4287 | auto engine_count = heuristics.getEngineConfigCount(); |
| 4288 | auto &engine_config = heuristics.getEngineConfig(engine_count); |
| 4289 | auto &fallback_list = fallback.getFallbackList(); |
| 4290 | |
| 4291 | cudnn_frontend::EngineConfigList filtered_configs; |
| 4292 | if (tensorflow::tensor_float_32_execution_enabled()) { |
| 4293 | if (stream_executor::cuda::RequireCuDNNDeterminism()) { |
| 4294 | cudnn_frontend::filter(engine_config, filtered_configs, |
| 4295 | isNonDeterministicOrIsDownConverting); |
| 4296 | cudnn_frontend::filter(fallback_list, filtered_configs, |
nothing calls this directly
no test coverage detected