Builds a PAGE_RES from the block_list in the way required for ApplyBoxes: All fuzzy spaces are removed, and all the words are maximally chopped.
| 215 | /// Builds a PAGE_RES from the block_list in the way required for ApplyBoxes: |
| 216 | /// All fuzzy spaces are removed, and all the words are maximally chopped. |
| 217 | PAGE_RES* Tesseract::SetupApplyBoxes(const GenericVector<TBOX>& boxes, |
| 218 | BLOCK_LIST *block_list) { |
| 219 | PreenXHeights(block_list); |
| 220 | // Strip all fuzzy space markers to simplify the PAGE_RES. |
| 221 | BLOCK_IT b_it(block_list); |
| 222 | for (b_it.mark_cycle_pt(); !b_it.cycled_list(); b_it.forward()) { |
| 223 | BLOCK* block = b_it.data(); |
| 224 | ROW_IT r_it(block->row_list()); |
| 225 | for (r_it.mark_cycle_pt(); !r_it.cycled_list(); r_it.forward ()) { |
| 226 | ROW* row = r_it.data(); |
| 227 | WERD_IT w_it(row->word_list()); |
| 228 | for (w_it.mark_cycle_pt(); !w_it.cycled_list(); w_it.forward()) { |
| 229 | WERD* word = w_it.data(); |
| 230 | if (word->cblob_list()->empty()) { |
| 231 | delete w_it.extract(); |
| 232 | } else { |
| 233 | word->set_flag(W_FUZZY_SP, false); |
| 234 | word->set_flag(W_FUZZY_NON, false); |
| 235 | } |
| 236 | } |
| 237 | } |
| 238 | } |
| 239 | PAGE_RES* page_res = new PAGE_RES(false, block_list, NULL); |
| 240 | PAGE_RES_IT pr_it(page_res); |
| 241 | WERD_RES* word_res; |
| 242 | while ((word_res = pr_it.word()) != NULL) { |
| 243 | MaximallyChopWord(boxes, pr_it.block()->block, |
| 244 | pr_it.row()->row, word_res); |
| 245 | pr_it.forward(); |
| 246 | } |
| 247 | return page_res; |
| 248 | } |
| 249 | |
| 250 | /// Tests the chopper by exhaustively running chop_one_blob. |
| 251 | /// The word_res will contain filled chopped_word, seam_array, denorm, |
no test coverage detected