MCPcopy Create free account
hub / github.com/RHVoice/RHVoice / tokenize

Method tokenize

src/core/language.cpp:1356–1441  ·  view source on GitHub ↗

Source from the content-addressed store, hash-verified

1354 }
1355
1356 void language::tokenize(utterance& u) const
1357 {
1358 if(!lcfg.tok_sent)
1359 return;
1360 if(!u.has_relation("TokIn"))
1361 return;
1362 relation& in_rel=u.get_relation("TokIn");
1363 const std::string sb{"sb"};
1364 const std::string eb{"eb"};
1365 const std::string br{"br"};
1366 std::vector<std::string> input;
1367 for(auto it=in_rel.begin(); it!=in_rel.end(); ++it)
1368 {
1369 input.push_back(sb);
1370 const std::string& name=it->get("name").as<std::string>();
1371 str::utf8explode(name, std::back_inserter(input));
1372 input.push_back(eb);
1373 if(!it->has_next())
1374 break;
1375 if(it->has_feature("break"))
1376 {
1377 input.push_back(br);
1378 continue;
1379 }
1380 const std::string& whitespace=it->next().get("whitespace").as<std::string>();
1381 str::utf8explode(whitespace, std::back_inserter(input));
1382 }
1383 std::vector<std::string> output;
1384 if(!tok_fst.translate(input.begin(), input.end(), std::back_inserter(output)))
1385 throw tokenization_error("");
1386 auto in_it=input.cbegin();
1387 auto in_tok_it=in_rel.begin();
1388 auto out_tok_it=in_tok_it;
1389 std::string name;
1390 bool in_whitespace=false;
1391 bool first_sb=true;
1392 item* last_subtok{nullptr};
1393 for(const auto& out_sym: output)
1394 {
1395 if(in_it==input.end() || out_sym!=*in_it)
1396 {
1397 if(out_sym.size()>1 && out_sym[0]=='+' && last_subtok)
1398 {
1399 const std::string subtag=out_sym.substr(1);
1400 last_subtok->as("TokStructure").set<std::string>("tag_"+subtag, "1");
1401 continue;
1402 }
1403 if(out_tok_it==in_rel.end())
1404 throw tokenization_error(name);
1405 if(name.empty())
1406 throw tokenization_error("");
1407 auto tag=out_sym;
1408 append_subtoken(*out_tok_it, name, tag);
1409 name.clear();
1410 last_subtok=out_tok_it->as("TokStructure").last_child_ptr();
1411 out_tok_it=in_tok_it;
1412 continue;
1413 }

Callers 1

create_utteranceMethod · 0.80

Calls 15

utf8explodeFunction · 0.85
tokenization_errorClass · 0.85
has_relationMethod · 0.80
has_nextMethod · 0.80
has_featureMethod · 0.80
last_child_ptrMethod · 0.80
translateMethod · 0.65
beginMethod · 0.45
endMethod · 0.45
getMethod · 0.45
nextMethod · 0.45
sizeMethod · 0.45

Tested by

no test coverage detected