\brief OCVDNNDetector::ParseOldYOLO \param crop \param detections \param tmpRegions
| 436 | /// \param tmpRegions |
| 437 | /// |
| 438 | void OCVDNNDetector::ParseOldYOLO(const cv::Rect& crop, const std::vector<cv::Mat>& detections, regions_t& tmpRegions) |
| 439 | { |
| 440 | if (m_outLayerTypes[0] == "DetectionOutput") |
| 441 | { |
| 442 | // Network produces output blob with a shape 1x1xNx7 where N is a number of detections and an every detection is a vector of values |
| 443 | // [batchId, classId, confidence, left, top, right, bottom] |
| 444 | CV_Assert(detections.size() > 0); |
| 445 | for (size_t k = 0; k < detections.size(); ++k) |
| 446 | { |
| 447 | const float* data = reinterpret_cast<float*>(detections[k].data); |
| 448 | for (size_t i = 0; i < detections[k].total(); i += 7) |
| 449 | { |
| 450 | float confidence = data[i + 2]; |
| 451 | if (confidence > m_confidenceThreshold) |
| 452 | { |
| 453 | int left = (int)data[i + 3]; |
| 454 | int top = (int)data[i + 4]; |
| 455 | int right = (int)data[i + 5]; |
| 456 | int bottom = (int)data[i + 6]; |
| 457 | int width = right - left + 1; |
| 458 | int height = bottom - top + 1; |
| 459 | if (width <= 2 || height <= 2) |
| 460 | { |
| 461 | left = cvRound(data[i + 3] * crop.width); |
| 462 | top = cvRound(data[i + 4] * crop.height); |
| 463 | right = cvRound(data[i + 5] * crop.width); |
| 464 | bottom = cvRound(data[i + 6] * crop.height); |
| 465 | width = right - left + 1; |
| 466 | height = bottom - top + 1; |
| 467 | } |
| 468 | size_t objectClass = (int)(data[i + 1]) - 1; |
| 469 | if (m_classesWhiteList.empty() || m_classesWhiteList.find(T2T(objectClass)) != std::end(m_classesWhiteList)) |
| 470 | tmpRegions.emplace_back(cv::Rect(left + crop.x, top + crop.y, width, height), T2T(objectClass), confidence); |
| 471 | } |
| 472 | } |
| 473 | } |
| 474 | } |
| 475 | else if (m_outLayerTypes[0] == "Region") |
| 476 | { |
| 477 | for (size_t i = 0; i < detections.size(); ++i) |
| 478 | { |
| 479 | // Network produces output blob with a shape NxC where N is a number of detected objects and C is a number of classes + 4 where the first 4 |
| 480 | // numbers are [center_x, center_y, width, height] |
| 481 | const float* data = reinterpret_cast<float*>(detections[i].data); |
| 482 | for (int j = 0; j < detections[i].rows; ++j, data += detections[i].cols) |
| 483 | { |
| 484 | cv::Mat scores = detections[i].row(j).colRange(5, detections[i].cols); |
| 485 | cv::Point classIdPoint; |
| 486 | double confidence = 0; |
| 487 | cv::minMaxLoc(scores, 0, &confidence, 0, &classIdPoint); |
| 488 | if (confidence > m_confidenceThreshold) |
| 489 | { |
| 490 | int centerX = cvRound(data[0] * crop.width); |
| 491 | int centerY = cvRound(data[1] * crop.height); |
| 492 | int width = cvRound(data[2] * crop.width); |
| 493 | int height = cvRound(data[3] * crop.height); |
| 494 | int left = centerX - width / 2; |
| 495 | int top = centerY - height / 2; |