MultibyteToWchar: correctly handle UTF-16 surrogate pairs.
* Whilst in WcharToMultibyte, we correctly convert our UTF-32 wchar characters to multibyte, the same wasn't done in MultibyteToWchar. Now, if we detect a leading surrogate, we'll re-read the multibyte sequence with space for a UTF-16 pair, which allows U16_GET to correctly convert the UTF-16 byte sequence into the needed UTF-32 codepoint. Fixes #13184.
This commit is contained in:
@@ -223,13 +223,26 @@ ICUCtypeData::MultibyteToWchar(wchar_t* wcOut, const char* mb, size_t mbLen,
|
|||||||
UErrorCode icuStatus = U_ZERO_ERROR;
|
UErrorCode icuStatus = U_ZERO_ERROR;
|
||||||
|
|
||||||
const char* buffer = mb;
|
const char* buffer = mb;
|
||||||
UChar targetBuffer[2];
|
UChar targetBuffer[3];
|
||||||
UChar* target = targetBuffer;
|
UChar* target = targetBuffer;
|
||||||
ucnv_toUnicode(converter, &target, target + 1, &buffer, buffer + mbLen,
|
ucnv_toUnicode(converter, &target, target + 1, &buffer, buffer + mbLen,
|
||||||
NULL, FALSE, &icuStatus);
|
NULL, FALSE, &icuStatus);
|
||||||
size_t sourceLengthUsed = buffer - mb;
|
size_t sourceLengthUsed = buffer - mb;
|
||||||
size_t targetLengthUsed = (size_t)(target - targetBuffer);
|
size_t targetLengthUsed = (size_t)(target - targetBuffer);
|
||||||
|
|
||||||
|
if (U16_IS_LEAD(targetBuffer[0])) {
|
||||||
|
// we have a surrogate pair, so re-read with enough space for a pair
|
||||||
|
// of characters instead
|
||||||
|
TRACE(("MultibyteToWchar(): have a surrogate pair\n"));
|
||||||
|
ucnv_resetToUnicode(converter);
|
||||||
|
buffer = mb;
|
||||||
|
target = targetBuffer;
|
||||||
|
ucnv_toUnicode(converter, &target, target + 2, &buffer, buffer + mbLen,
|
||||||
|
NULL, FALSE, &icuStatus);
|
||||||
|
sourceLengthUsed = buffer - mb;
|
||||||
|
targetLengthUsed = (size_t)(target - targetBuffer);
|
||||||
|
}
|
||||||
|
|
||||||
if (icuStatus == U_BUFFER_OVERFLOW_ERROR && targetLengthUsed > 0) {
|
if (icuStatus == U_BUFFER_OVERFLOW_ERROR && targetLengthUsed > 0) {
|
||||||
// we've got one character, which is all that we wanted
|
// we've got one character, which is all that we wanted
|
||||||
icuStatus = U_ZERO_ERROR;
|
icuStatus = U_ZERO_ERROR;
|
||||||
@@ -248,7 +261,7 @@ ICUCtypeData::MultibyteToWchar(wchar_t* wcOut, const char* mb, size_t mbLen,
|
|||||||
result = B_BAD_INDEX;
|
result = B_BAD_INDEX;
|
||||||
} else {
|
} else {
|
||||||
UChar32 unicodeChar = 0xBADBEEF;
|
UChar32 unicodeChar = 0xBADBEEF;
|
||||||
U16_GET(targetBuffer, 0, 0, 2, unicodeChar);
|
U16_GET(targetBuffer, 0, 0, targetLengthUsed, unicodeChar);
|
||||||
|
|
||||||
if (unicodeChar == 0) {
|
if (unicodeChar == 0) {
|
||||||
// reset to initial state
|
// reset to initial state
|
||||||
|
|||||||
Reference in New Issue
Block a user