| 23 | |
| 24 | template<typename T> |
| 25 | inline void Normalize(nvbench::state &state, nvbench::type_list<T>) |
| 26 | try |
| 27 | { |
| 28 | long3 srcShape = benchutils::GetShape<3>(state.get_string("shape")); |
| 29 | long varShape = state.get_int64("varShape"); |
| 30 | long3 dstShape = srcShape; |
| 31 | |
| 32 | float globalScale = 1.234f; |
| 33 | float globalShift = 2.345f; |
| 34 | float epsilon = 12.34f; |
| 35 | uint32_t flags = CVCUDA_NORMALIZE_SCALE_IS_STDDEV; |
| 36 | |
| 37 | long3 baseShape{srcShape.x, 1, 1}; |
| 38 | long3 scaleShape{srcShape.x, 1, 1}; |
| 39 | |
| 40 | state.add_global_memory_reads(srcShape.x * srcShape.y * srcShape.z * sizeof(T) |
| 41 | + baseShape.x * baseShape.y * baseShape.z * sizeof(float) |
| 42 | + scaleShape.x * scaleShape.y * scaleShape.z * sizeof(float)); |
| 43 | state.add_global_memory_writes(dstShape.x * dstShape.y * dstShape.z * sizeof(T)); |
| 44 | |
| 45 | cvcuda::Normalize op; |
| 46 | |
| 47 | // clang-format off |
| 48 | |
| 49 | nvcv::Tensor base({{baseShape.x, baseShape.y, baseShape.z, 1}, "NHWC"}, nvcv::TYPE_F32); |
| 50 | nvcv::Tensor scale({{scaleShape.x, scaleShape.y, scaleShape.z, 1}, "NHWC"}, nvcv::TYPE_F32); |
| 51 | |
| 52 | benchutils::FillTensor<float>(base, benchutils::RandomValues<T>()); |
| 53 | benchutils::FillTensor<float>(scale, benchutils::RandomValues<float>(0.f, 1.f)); |
| 54 | |
| 55 | if (varShape < 0) // negative var shape means use Tensor |
| 56 | { |
| 57 | nvcv::Tensor src({{srcShape.x, srcShape.y, srcShape.z, 1}, "NHWC"}, benchutils::GetDataType<T>()); |
| 58 | nvcv::Tensor dst({{dstShape.x, dstShape.y, dstShape.z, 1}, "NHWC"}, benchutils::GetDataType<T>()); |
| 59 | |
| 60 | benchutils::FillTensor<T>(src, benchutils::RandomValues<T>()); |
| 61 | |
| 62 | state.exec(nvbench::exec_tag::sync, |
| 63 | [&op, &src, &base, &scale, &dst, &globalScale, &globalShift, &epsilon, &flags] |
| 64 | (nvbench::launch &launch) |
| 65 | { |
| 66 | op(launch.get_stream(), src, base, scale, dst, globalScale, globalShift, epsilon, flags); |
| 67 | }); |
| 68 | } |
| 69 | else // zero and positive var shape means use ImageBatchVarShape |
| 70 | { |
| 71 | nvcv::ImageBatchVarShape src(srcShape.x); |
| 72 | nvcv::ImageBatchVarShape dst(dstShape.x); |
| 73 | |
| 74 | benchutils::FillImageBatch<T>(src, long2{srcShape.z, srcShape.y}, long2{varShape, varShape}, |
| 75 | benchutils::RandomValues<T>()); |
| 76 | benchutils::FillImageBatch<T>(dst, long2{dstShape.z, dstShape.y}, long2{varShape, varShape}, |
| 77 | benchutils::RandomValues<T>()); |
| 78 | |
| 79 | state.exec(nvbench::exec_tag::sync, |
| 80 | [&op, &src, &base, &scale, &dst, &globalScale, &globalShift, &epsilon, &flags] |
| 81 | (nvbench::launch &launch) |
| 82 | { |