| 492 | } |
| 493 | |
| 494 | std::string BPETokenizer::decode(const std::vector<uint32_t>& tokens) const { |
| 495 | if (runtime_config_.decoder == TokenizerRuntimeConfig::Decoder::REPLACE_METASPACE) { |
| 496 | std::string result; |
| 497 | result.reserve(tokens.size() * 4); |
| 498 | |
| 499 | for (uint32_t token_id : tokens) { |
| 500 | if (token_id >= id_to_token_.size()) continue; |
| 501 | result += id_to_token_[token_id]; |
| 502 | } |
| 503 | |
| 504 | size_t pos = 0; |
| 505 | while ((pos = result.find(kMetaspace, pos)) != std::string::npos) { |
| 506 | result.replace(pos, 3, " "); |
| 507 | pos += 1; |
| 508 | } |
| 509 | return result; |
| 510 | } |
| 511 | |
| 512 | std::string unicode_result; |
| 513 | unicode_result.reserve(tokens.size() * 4); |
| 514 | |
| 515 | for (uint32_t token_id : tokens) { |
| 516 | if (token_id >= id_to_token_.size()) continue; |
| 517 | const std::string& tok = id_to_token_[token_id]; |
| 518 | |
| 519 | size_t pos = 0; |
| 520 | while (pos < tok.size()) { |
| 521 | if (pos + 3 <= tok.size() && |
| 522 | (unsigned char)tok[pos] == 0xE2 && |
| 523 | (unsigned char)tok[pos+1] == 0x96 && |
| 524 | (unsigned char)tok[pos+2] == 0x81) { |
| 525 | unicode_result.push_back(' '); |
| 526 | pos += 3; |
| 527 | } else { |
| 528 | unicode_result.push_back(tok[pos++]); |
| 529 | } |
| 530 | } |
| 531 | } |
| 532 | |
| 533 | return unicode_to_bytes(unicode_result); |
| 534 | } |
| 535 | |
| 536 | |
| 537 | } |