* Applies the given word to the adaptive classifier if possible. * The word must be SPACE-DELIMITED UTF-8 - l i k e t h i s , so it can * tell the boundaries of the graphemes. * Assumes that SetImage/SetRectangle have been used to set the image * to the given word. The mode arg should be PSM_SINGLE_WORD or * PSM_CIRCLE_WORD, as that will be used to control layout analysis. * The currently se
| 2000 | * Returns false if adaption was not possible for some reason. |
| 2001 | */ |
| 2002 | bool TessBaseAPI::AdaptToWordStr(PageSegMode mode, const char* wordstr) { |
| 2003 | int debug = 0; |
| 2004 | GetIntVariable("applybox_debug", &debug); |
| 2005 | bool success = true; |
| 2006 | PageSegMode current_psm = GetPageSegMode(); |
| 2007 | SetPageSegMode(mode); |
| 2008 | SetVariable("classify_enable_learning", "0"); |
| 2009 | char* text = GetUTF8Text(); |
| 2010 | if (debug) { |
| 2011 | tprintf("Trying to adapt \"%s\" to \"%s\"\n", text, wordstr); |
| 2012 | } |
| 2013 | if (text != NULL) { |
| 2014 | PAGE_RES_IT it(page_res_); |
| 2015 | WERD_RES* word_res = it.word(); |
| 2016 | if (word_res != NULL) { |
| 2017 | word_res->word->set_text(wordstr); |
| 2018 | } else { |
| 2019 | success = false; |
| 2020 | } |
| 2021 | // Check to see if text matches wordstr. |
| 2022 | int w = 0; |
| 2023 | int t = 0; |
| 2024 | for (t = 0; text[t] != '\0'; ++t) { |
| 2025 | if (text[t] == '\n' || text[t] == ' ') |
| 2026 | continue; |
| 2027 | while (wordstr[w] != '\0' && wordstr[w] == ' ') |
| 2028 | ++w; |
| 2029 | if (text[t] != wordstr[w]) |
| 2030 | break; |
| 2031 | ++w; |
| 2032 | } |
| 2033 | if (text[t] != '\0' || wordstr[w] != '\0') { |
| 2034 | // No match. |
| 2035 | delete page_res_; |
| 2036 | GenericVector<TBOX> boxes; |
| 2037 | page_res_ = tesseract_->SetupApplyBoxes(boxes, block_list_); |
| 2038 | tesseract_->ReSegmentByClassification(page_res_); |
| 2039 | tesseract_->TidyUp(page_res_); |
| 2040 | PAGE_RES_IT pr_it(page_res_); |
| 2041 | if (pr_it.word() == NULL) |
| 2042 | success = false; |
| 2043 | else |
| 2044 | word_res = pr_it.word(); |
| 2045 | } else { |
| 2046 | word_res->BestChoiceToCorrectText(); |
| 2047 | } |
| 2048 | if (success) { |
| 2049 | tesseract_->EnableLearning = true; |
| 2050 | tesseract_->LearnWord(NULL, word_res); |
| 2051 | } |
| 2052 | delete [] text; |
| 2053 | } else { |
| 2054 | success = false; |
| 2055 | } |
| 2056 | SetPageSegMode(current_psm); |
| 2057 | return success; |
| 2058 | } |
| 2059 |
no test coverage detected