Resegments the words by running the classifier in an attempt to find the correct segmentation that produces the required string.
| 507 | /// Resegments the words by running the classifier in an attempt to find the |
| 508 | /// correct segmentation that produces the required string. |
| 509 | void Tesseract::ReSegmentByClassification(PAGE_RES* page_res) { |
| 510 | PAGE_RES_IT pr_it(page_res); |
| 511 | WERD_RES* word_res; |
| 512 | for (; (word_res = pr_it.word()) != NULL; pr_it.forward()) { |
| 513 | WERD* word = word_res->word; |
| 514 | if (word->text() == NULL || word->text()[0] == '\0') |
| 515 | continue; // Ignore words that have no text. |
| 516 | // Convert the correct text to a vector of UNICHAR_ID |
| 517 | GenericVector<UNICHAR_ID> target_text; |
| 518 | if (!ConvertStringToUnichars(word->text(), &target_text)) { |
| 519 | tprintf("APPLY_BOX: FAILURE: can't find class_id for '%s'\n", |
| 520 | word->text()); |
| 521 | pr_it.DeleteCurrentWord(); |
| 522 | continue; |
| 523 | } |
| 524 | if (!FindSegmentation(target_text, word_res)) { |
| 525 | tprintf("APPLY_BOX: FAILURE: can't find segmentation for '%s'\n", |
| 526 | word->text()); |
| 527 | pr_it.DeleteCurrentWord(); |
| 528 | continue; |
| 529 | } |
| 530 | } |
| 531 | } |
| 532 | |
| 533 | /// Converts the space-delimited string of utf8 text to a vector of UNICHAR_ID. |
| 534 | /// @return false if an invalid UNICHAR_ID is encountered. |
no test coverage detected