| 254 | } |
| 255 | |
| 256 | void Requantize(OpKernelContext* tf_context, const qint32* input, int count, |
| 257 | float input_min, float input_max, float output_min, |
| 258 | float output_max, quint8* output) { |
| 259 | #ifdef TENSORFLOW_USE_META |
| 260 | mutex_lock library_lock(GetMutex()); |
| 261 | typedef gemmlowp::meta::Transform1DParams<int32_t, uint8_t, |
| 262 | gemmlowp::meta::Requantize> |
| 263 | Params; |
| 264 | |
| 265 | Params params; |
| 266 | params.input = reinterpret_cast<const int32_t*>(input); |
| 267 | params.output = reinterpret_cast<uint8_t*>(output); |
| 268 | params.kernel.count = count; |
| 269 | params.kernel.input_range_min = input_min; |
| 270 | params.kernel.output_range_min = output_min; |
| 271 | params.kernel.input_range_scale = |
| 272 | CalculateRangeScale<int32_t>(input_min, input_max); |
| 273 | params.kernel.one_over_output_range_scale = |
| 274 | CalculateOneOverRangeScale<uint8_t>(output_min, output_max); |
| 275 | params.kernel.input_range_offset = |
| 276 | static_cast<float>(std::numeric_limits<int32_t>::lowest()); |
| 277 | |
| 278 | // After adding the output_range_offset the value is cast from float to uint. |
| 279 | // The float to int/uint cast in NEON uses round toward 0. To keep the |
| 280 | // rounding consistent with Eigen, which uses round toward closest, we can |
| 281 | // add 0.5f and exploit the fact that we only operate on non negative values. |
| 282 | // TODO(maciekc): fix the actual kernel in gemmlowp/meta |
| 283 | params.kernel.output_range_offset = |
| 284 | static_cast<float>(std::numeric_limits<uint8_t>::lowest()) + 0.5f; |
| 285 | |
| 286 | MultiThreadTransform1D<Params, 16>(tf_context, params); |
| 287 | #else |
| 288 | LOG(FATAL) << "Requantize: Meta fastpath not supported."; |
| 289 | #endif |
| 290 | } |
| 291 | |
| 292 | void Dequantize(OpKernelContext* tf_context, const quint8* input, int count, |
| 293 | float range_min, float range_max, float* output) { |