MCPcopy Create free account
hub / github.com/InteractiveComputerGraphics/SPlisHSPlasH / ImTextCharFromUtf8

Function ImTextCharFromUtf8

extern/imgui/imgui.cpp:2026–2076  ·  view source on GitHub ↗

Convert UTF-8 to 32-bit character, process single character input. A nearly-branchless UTF-8 decoder, based on work of Christopher Wellons (https://github.com/skeeto/branchless-utf8). We handle UTF-8 decoding error by skipping forward.

Source from the content-addressed store, hash-verified

2024// A nearly-branchless UTF-8 decoder, based on work of Christopher Wellons (https://github.com/skeeto/branchless-utf8).
2025// We handle UTF-8 decoding error by skipping forward.
2026int ImTextCharFromUtf8(unsigned int* out_char, const char* in_text, const char* in_text_end)
2027{
2028 static const char lengths[32] = { 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 2, 2, 2, 2, 3, 3, 4, 0 };
2029 static const int masks[] = { 0x00, 0x7f, 0x1f, 0x0f, 0x07 };
2030 static const uint32_t mins[] = { 0x400000, 0, 0x80, 0x800, 0x10000 };
2031 static const int shiftc[] = { 0, 18, 12, 6, 0 };
2032 static const int shifte[] = { 0, 6, 4, 2, 0 };
2033 int len = lengths[*(const unsigned char*)in_text >> 3];
2034 int wanted = len + !len;
2035
2036 if (in_text_end == NULL)
2037 in_text_end = in_text + wanted; // Max length, nulls will be taken into account.
2038
2039 // Copy at most 'len' bytes, stop copying at 0 or past in_text_end. Branch predictor does a good job here,
2040 // so it is fast even with excessive branching.
2041 unsigned char s[4];
2042 s[0] = in_text + 0 < in_text_end ? in_text[0] : 0;
2043 s[1] = in_text + 1 < in_text_end ? in_text[1] : 0;
2044 s[2] = in_text + 2 < in_text_end ? in_text[2] : 0;
2045 s[3] = in_text + 3 < in_text_end ? in_text[3] : 0;
2046
2047 // Assume a four-byte character and load four bytes. Unused bits are shifted out.
2048 *out_char = (uint32_t)(s[0] & masks[len]) << 18;
2049 *out_char |= (uint32_t)(s[1] & 0x3f) << 12;
2050 *out_char |= (uint32_t)(s[2] & 0x3f) << 6;
2051 *out_char |= (uint32_t)(s[3] & 0x3f) << 0;
2052 *out_char >>= shiftc[len];
2053
2054 // Accumulate the various error conditions.
2055 int e = 0;
2056 e = (*out_char < mins[len]) << 6; // non-canonical encoding
2057 e |= ((*out_char >> 11) == 0x1b) << 7; // surrogate half?
2058 e |= (*out_char > IM_UNICODE_CODEPOINT_MAX) << 8; // out of range?
2059 e |= (s[1] & 0xc0) >> 2;
2060 e |= (s[2] & 0xc0) >> 4;
2061 e |= (s[3] ) >> 6;
2062 e ^= 0x2a; // top two bits of each tail byte correct?
2063 e >>= shifte[len];
2064
2065 if (e)
2066 {
2067 // No bytes are consumed when *in_text == 0 || in_text == in_text_end.
2068 // One byte is consumed in case of invalid first byte of in_text.
2069 // All available bytes (at most `len` bytes) are consumed on incomplete/invalid second to last bytes.
2070 // Invalid or incomplete input may consume less bytes than wanted, therefore every byte has to be inspected in s.
2071 wanted = ImMin(wanted, !!s[0] + !!s[1] + !!s[2] + !!s[3]);
2072 *out_char = IM_UNICODE_CODEPOINT_INVALID;
2073 }
2074
2075 return wanted;
2076}
2077
2078int ImTextStrFromUtf8(ImWchar* buf, int buf_size, const char* in_text, const char* in_text_end, const char** in_text_remaining)
2079{

Callers 10

ImTextStrFromUtf8Function · 0.85
ImTextCountCharsFromUtf8Function · 0.85
DebugTextEncodingMethod · 0.85
AddTextMethod · 0.85
CalcWordWrapPositionAMethod · 0.85
CalcTextSizeAMethod · 0.85
RenderTextMethod · 0.85
InputTextExMethod · 0.85

Calls 1

ImMinFunction · 0.85

Tested by

no test coverage detected