| 398 | } |
| 399 | |
| 400 | static inline |
| 401 | void _TokenizeToSegments(string const &src, char const *delimiters, |
| 402 | vector<pair<char const *, char const *> > &segments) |
| 403 | { |
| 404 | // Delimiter checking LUT. |
| 405 | // NOTE: For some reason, calling memset here is faster than doing the |
| 406 | // aggregate initialization. Beats me. Ask gcc. (10/07) |
| 407 | char _isDelim[256]; // = {0}; |
| 408 | memset(_isDelim, 0, sizeof(_isDelim)); |
| 409 | for (char const *p = delimiters; *p; ++p) |
| 410 | _isDelim[static_cast<unsigned char>(*p)] = 1; |
| 411 | |
| 412 | #define IS_DELIMITER(c) (_isDelim[static_cast<unsigned char>(c)]) |
| 413 | |
| 414 | // First build a vector of segments. A segment is a pair of pointers into |
| 415 | // \a src's data, the first indicating the start of a token, the second |
| 416 | // pointing one past the last character of a token (like a pair of |
| 417 | // iterators). |
| 418 | |
| 419 | // A small amount of reservation seems to help. |
| 420 | segments.reserve(8); |
| 421 | char const *end = src.data() + src.size(); |
| 422 | for (char const *c = src.data(); c < end; ++c) { |
| 423 | // skip delimiters |
| 424 | if (IS_DELIMITER(*c)) |
| 425 | continue; |
| 426 | // have a token until the next delimiter. |
| 427 | // push back a new segment, but we only know the begin point yet. |
| 428 | segments.push_back(make_pair(c, c)); |
| 429 | for (++c; c != end; ++c) |
| 430 | if (IS_DELIMITER(*c)) |
| 431 | break; |
| 432 | // complete the segment with the end point. |
| 433 | segments.back().second = c; |
| 434 | } |
| 435 | |
| 436 | #undef IS_DELIMITER |
| 437 | } |
| 438 | |
| 439 | vector<string> |
| 440 | TfStringSplit(string const &src, string const &separator) |