322 lines
15 KiB
C++
322 lines
15 KiB
C++
// ----------------------------------------------------------------------------
|
|
// Unit test for the class core/UnicodeTools
|
|
// ----------------------------------------------------------------------------
|
|
#include "precompiled_unittests_core.h"
|
|
|
|
#include "core/UnicodeTools.h"
|
|
|
|
namespace ITF
|
|
{
|
|
|
|
namespace UnicodeTestTools
|
|
{
|
|
enum SampleUnicodeChars
|
|
{
|
|
// ASCII codes
|
|
SAMPLE_UNICODE_NULL = 0x0000,
|
|
SAMPLE_UNICODE_SPACE = 0x0020,
|
|
SAMPLE_UNICODE_A = 0x0041,
|
|
SAMPLE_UNICODE_TILDE = 0x007E,
|
|
SAMPLE_UNICODE_DEL = 0x007F,
|
|
|
|
// UTF8 in 2 bytes
|
|
SAMPLE_UNICODE_PAD = 0x0080,
|
|
SAMPLE_UNICODE_ECUTE = 0x00E9,
|
|
SAMPLE_UNICODE_7FF = 0x07FF,
|
|
|
|
// UTF8 in 3 bytes
|
|
SAMPLE_UNICODE_EURO = 0x20AC,
|
|
SAMPLE_UNICODE_CHINESE = 0xF900,
|
|
SAMPLE_UNICODE_FDF0 = 0xFDF0,
|
|
|
|
// UTF8 in 4 bytes
|
|
SAMPLE_UNICODE_20000 = 0x20000,
|
|
SAMPLE_UNICODE_2FFFD = 0x2FFFD,
|
|
};
|
|
|
|
struct UnicodeCharInfo
|
|
{
|
|
static const u32 MAX_UTF8_CHARS = 4 + 1; // 1 is for zero termination
|
|
SampleUnicodeChars m_unicode;
|
|
char m_utf8[MAX_UTF8_CHARS];
|
|
|
|
explicit UnicodeCharInfo(SampleUnicodeChars _unicode) : m_unicode(_unicode){ m_utf8[0] = m_utf8[1] = m_utf8[2] = m_utf8[3] = m_utf8[4] = 0; }
|
|
UnicodeCharInfo(SampleUnicodeChars _unicode, u8 _c0)
|
|
: m_unicode(_unicode) { m_utf8[0] = _c0; m_utf8[1] = 0; m_utf8[2] = 0; m_utf8[3] = 0; m_utf8[4] = 0; }
|
|
UnicodeCharInfo(SampleUnicodeChars _unicode, u8 _c0, u8 _c1)
|
|
: m_unicode(_unicode) { m_utf8[0] = _c0; m_utf8[1] = _c1; m_utf8[2] = 0; m_utf8[3] = 0; m_utf8[4] = 0; }
|
|
UnicodeCharInfo(SampleUnicodeChars _unicode, u8 _c0, u8 _c1, u8 _c2)
|
|
: m_unicode(_unicode) { m_utf8[0] = _c0; m_utf8[1] = _c1; m_utf8[2] = _c2; m_utf8[3] = 0; m_utf8[4] = 0; }
|
|
UnicodeCharInfo(SampleUnicodeChars _unicode, u8 _c0, u8 _c1, u8 _c2, u8 _c3)
|
|
: m_unicode(_unicode) { m_utf8[0] = _c0; m_utf8[1] = _c1; m_utf8[2] = _c2; m_utf8[3] = _c3; m_utf8[4] = 0; }
|
|
|
|
bool operator ==(const UnicodeCharInfo& _rhs) const { return m_unicode == _rhs.m_unicode; }
|
|
};
|
|
|
|
static UnicodeCharInfo sampleUnicodeChars[] =
|
|
{
|
|
UnicodeCharInfo(SAMPLE_UNICODE_NULL , 0x00),
|
|
UnicodeCharInfo(SAMPLE_UNICODE_SPACE , 0x20),
|
|
UnicodeCharInfo(SAMPLE_UNICODE_A , 0x41),
|
|
UnicodeCharInfo(SAMPLE_UNICODE_TILDE , 0x7E),
|
|
UnicodeCharInfo(SAMPLE_UNICODE_DEL , 0x7F),
|
|
|
|
UnicodeCharInfo(SAMPLE_UNICODE_PAD , 0xC2, 0x80),
|
|
UnicodeCharInfo(SAMPLE_UNICODE_ECUTE , 0xC3, 0xA9),
|
|
UnicodeCharInfo(SAMPLE_UNICODE_7FF , 0xDF, 0xBF),
|
|
|
|
UnicodeCharInfo(SAMPLE_UNICODE_EURO , 0xE2, 0x82, 0xAC),
|
|
UnicodeCharInfo(SAMPLE_UNICODE_CHINESE, 0xEF, 0xA4, 0x80),
|
|
UnicodeCharInfo(SAMPLE_UNICODE_FDF0 , 0xEF, 0xB7, 0xB0),
|
|
|
|
UnicodeCharInfo(SAMPLE_UNICODE_20000 , 0xF0, 0xA0, 0x80, 0x80),
|
|
UnicodeCharInfo(SAMPLE_UNICODE_2FFFD , 0xF0, 0xAF, 0xBF, 0xBD)
|
|
};
|
|
|
|
static void buildUTF8String(const SampleUnicodeChars* _first, const SampleUnicodeChars* _last, String8& _str)
|
|
{
|
|
const UnicodeCharInfo* first = sampleUnicodeChars;
|
|
const UnicodeCharInfo* last = sampleUnicodeChars + ITF_ARRAY_SIZE(sampleUnicodeChars);
|
|
for( ; _first != _last; ++_first)
|
|
{
|
|
const UnicodeCharInfo* uci = std::find(first, last, UnicodeCharInfo(*_first));
|
|
if (uci)
|
|
{
|
|
_str += uci->m_utf8;
|
|
}
|
|
}
|
|
}
|
|
|
|
struct SpecialSeqInfo
|
|
{
|
|
enum { SEQUENCE_MAX_SIZE = 7 };
|
|
u8 utf8[SEQUENCE_MAX_SIZE];
|
|
UnicodeChar unicodeSeqExpected[SEQUENCE_MAX_SIZE];
|
|
};
|
|
|
|
} // namespace UnicodeTestTools
|
|
|
|
// test decoding of a sequence of utf8 encoded characters in a string8
|
|
// using the functions getNbUnicodeChar() and getNextUnicodeChar()
|
|
TEST(UnicodeTools, test_utf8Decoding_1)
|
|
{
|
|
UnicodeTestTools::SampleUnicodeChars strCodes[] =
|
|
{
|
|
UnicodeTestTools::SAMPLE_UNICODE_SPACE,
|
|
UnicodeTestTools::SAMPLE_UNICODE_EURO,
|
|
UnicodeTestTools::SAMPLE_UNICODE_DEL,
|
|
UnicodeTestTools::SAMPLE_UNICODE_20000,
|
|
UnicodeTestTools::SAMPLE_UNICODE_A,
|
|
UnicodeTestTools::SAMPLE_UNICODE_ECUTE,
|
|
};
|
|
u32 expectedNbUnicodeChar = ITF_ARRAY_SIZE(strCodes);
|
|
|
|
String8 utf8String;
|
|
UnicodeTestTools::buildUTF8String(strCodes, strCodes + ITF_ARRAY_SIZE(strCodes), utf8String);
|
|
|
|
// check
|
|
const char* utfString8CurrentPos = utf8String.cStr();
|
|
EXPECT_EQ(expectedNbUnicodeChar, getNbUnicodeChar(utfString8CurrentPos));
|
|
for(u32 i = 0; i < expectedNbUnicodeChar; ++i)
|
|
{
|
|
ASSERT_NE(0, *utfString8CurrentPos); // unexpected end of string
|
|
|
|
u32 unicodeChar = getNextUnicodeChar(utfString8CurrentPos);
|
|
EXPECT_EQ(unicodeChar, u32(strCodes[i]));
|
|
}
|
|
EXPECT_EQ(0, *utfString8CurrentPos);
|
|
}
|
|
|
|
// test decoding of a sequence of utf8 encoded characters in a string8
|
|
// using the functions getNbUnicodeChar() and getNextUnicodeChar()
|
|
TEST(UnicodeTools, test_utf8Decoding_2)
|
|
{
|
|
UnicodeTestTools::SampleUnicodeChars strCodes[] =
|
|
{
|
|
UnicodeTestTools::SAMPLE_UNICODE_2FFFD,
|
|
UnicodeTestTools::SAMPLE_UNICODE_TILDE,
|
|
UnicodeTestTools::SAMPLE_UNICODE_CHINESE,
|
|
UnicodeTestTools::SAMPLE_UNICODE_PAD,
|
|
UnicodeTestTools::SAMPLE_UNICODE_A,
|
|
UnicodeTestTools::SAMPLE_UNICODE_7FF,
|
|
UnicodeTestTools::SAMPLE_UNICODE_FDF0,
|
|
};
|
|
u32 expectedNbUnicodeChar = ITF_ARRAY_SIZE(strCodes);
|
|
|
|
String8 utf8String;
|
|
UnicodeTestTools::buildUTF8String(strCodes, strCodes + ITF_ARRAY_SIZE(strCodes), utf8String);
|
|
|
|
// check
|
|
const char* utfString8CurrentPos = utf8String.cStr();
|
|
EXPECT_EQ(expectedNbUnicodeChar, getNbUnicodeChar(utfString8CurrentPos));
|
|
for(u32 i = 0; i < expectedNbUnicodeChar; ++i)
|
|
{
|
|
ASSERT_NE(0, *utfString8CurrentPos); // unexpected end of string
|
|
|
|
u32 unicodeChar = getNextUnicodeChar(utfString8CurrentPos);
|
|
EXPECT_EQ(unicodeChar, u32(strCodes[i]));
|
|
}
|
|
EXPECT_EQ(0, *utfString8CurrentPos);
|
|
}
|
|
|
|
// test decoding of a pure Ascii sequence of characters in a string8
|
|
// using the functions getNbUnicodeChar() and getNextUnicodeChar()
|
|
TEST(UnicodeTools, test_utf8Decoding_Ascii)
|
|
{
|
|
String8 strPureAscii = "Test\nAscii"; // containing only ascii chars
|
|
u32 expectedNbUnicodeChar = strPureAscii.getLen();
|
|
|
|
const char* utfString8CurrentPos = strPureAscii.cStr();
|
|
EXPECT_EQ(expectedNbUnicodeChar, getNbUnicodeChar(utfString8CurrentPos));
|
|
for(u32 i = 0; i < expectedNbUnicodeChar; ++i)
|
|
{
|
|
ASSERT_NE(0, *utfString8CurrentPos); // unexpected end of string
|
|
|
|
const char currentChar = *utfString8CurrentPos;
|
|
u32 unicodeChar = getNextUnicodeChar(utfString8CurrentPos);
|
|
EXPECT_EQ(unicodeChar, u32(currentChar));
|
|
}
|
|
EXPECT_EQ(0, *utfString8CurrentPos);
|
|
}
|
|
|
|
// test UTF16 <-> UTF8 conversion functions
|
|
TEST(UnicodeTools, test_utf16)
|
|
{
|
|
static const u8 unsignedKosme[] = { 'k', 'o', 's', 'm', 'e', // kosme in ascii
|
|
0xF0, 0xA4, 0xAD, 0xA2, // additionnal chinese character (to test UTF16 surrogate)
|
|
0xCE, 0xBA, 0xE1, 0xBD, 0xB9, 0xCF, 0x83, 0xCE, 0xBC, 0xCE, 0xB5, // kosme in greek
|
|
0x00};
|
|
|
|
const char * kosme = reinterpret_cast<const char *>(unsignedKosme);
|
|
|
|
u32 nbU16Word = getNbUTF16Word(kosme);
|
|
u16 * utf16Kosme = newAlloc(mId_Temporary, u16[nbU16Word]);
|
|
buildUTF16FromUTF8(kosme, utf16Kosme, nbU16Word);
|
|
|
|
u32 nbU8 = getNbUTF8Byte(utf16Kosme)+1u;
|
|
char * utf8Kosme = newAlloc(mId_Temporary, char[nbU8]);
|
|
|
|
buildUTF8FromUTF16(utf16Kosme, utf8Kosme, nbU8);
|
|
|
|
EXPECT_EQ(0, strcmp(utf8Kosme, kosme));
|
|
|
|
delete [] utf8Kosme;
|
|
delete [] utf16Kosme;
|
|
}
|
|
|
|
// test special UTF sequences
|
|
// cf http://www.cl.cam.ac.uk/~mgk25/ucs/examples/UTF-8-test.txt for invalid sequence sample
|
|
TEST(UnicodeTools, test_SpecialSequences)
|
|
{
|
|
UnicodeTestTools::SpecialSeqInfo specialSeqences[] =
|
|
{
|
|
{ /*SPECIAL_SEQ_BOUND_FIRST_1BYTE,*/ { 0x00 }, { 0x0000 } },
|
|
{ /*SPECIAL_SEQ_BOUND_FIRST_2BYTES,*/ { 0xC2, 0x80, 0x00}, { 0x0080, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_BOUND_FIRST_3BYTES,*/ { 0xE0, 0xA0, 0x80 , 0x00}, { 0x0800, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_BOUND_FIRST_4BYTES,*/ { 0xF0, 0x90, 0x80, 0x80 , 0x00}, { 0x00010000, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_BOUND_FIRST_5BYTES,*/ { 0xF8, 0x88, 0x80, 0x80, 0x80, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_BOUND_FIRST_6BYTES,*/ { 0xFC, 0x84, 0x80, 0x80, 0x80, 0x80, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
|
|
|
|
{ /*SPECIAL_SEQ_BOUND_LAST_1BYTE,*/ { 0x7F, 0x00}, { 0x007F, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_BOUND_LAST_2BYTES,*/ { 0xDF, 0xBF, 0x00}, { 0x07FF, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_BOUND_LAST_3BYTES,*/ { 0xEF, 0xBF, 0xBF, 0x00}, { 0xFFFF, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_BOUND_LAST_4BYTES,*/ { 0xF7, 0xBF, 0xBF, 0xBF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_BOUND_LAST_5BYTES,*/ { 0xFB, 0xBF, 0xBF, 0xBF, 0xBF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD,0x0000 } },
|
|
{ /*SPECIAL_SEQ_BOUND_LAST_6BYTES,*/ { 0xFD, 0xBF, 0xBF, 0xBF, 0xBF, 0xBF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
|
|
|
|
{ /*SPECIAL_SEQ_BOUND_OTHER_D7FF,*/ { 0xED, 0x9F, 0xBF, 0x00}, { 0xD7FF, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_BOUND_OTHER_E000,*/ { 0xEE, 0x80, 0x80, 0x00}, { 0xE000, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_BOUND_OTHER_FFFD,*/ { 0xEF, 0xBF, 0xBD, 0x00}, { 0xFFFD, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_BOUND_OTHER_10FFFF,*/ { 0xF4, 0x8F, 0xBF, 0xBF, 0x00}, { 0x10FFFF, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_BOUND_OTHER_110000,*/ { 0xF4, 0x90, 0x80, 0x80, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
|
|
|
|
{ /*SPECIAL_SEQ_CONT_FIRST,*/ { 0x80, 0x00}, { 0xFFFD, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_CONT_LAST,*/ { 0xBF, 0x00}, { 0xFFFD, 0x0000 } },
|
|
|
|
{ /*SPECIAL_SEQ_CONT_2BYTES,*/ { 0x80, 0xBF, 0x00}, { 0xFFFD, 0xFFFD, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_CONT_3BYTES,*/ { 0x80, 0xBF, 0x80, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_CONT_4BYTES,*/ { 0x80, 0xBF, 0x80, 0xBF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_CONT_5BYTES,*/ { 0x80, 0xBF, 0x80, 0xBF, 0x80, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_CONT_6BYTES,*/ { 0x80, 0xBF, 0x80, 0xBF, 0x80, 0xBF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
|
|
|
|
{ /*SPECIAL_SEQ_IMPOSSIBLE_FE,*/ { 0xFE, 0x00}, { 0xFFFD, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_IMPOSSIBLE_FF,*/ { 0xFF, 0x00}, { 0xFFFD, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_IMPOSSIBLE_FEFF,*/ { 0xFE, 0xFE, 0xFF, 0xFF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
|
|
|
|
{ /*SPECIAL_SEQ_OVERLONG_2,*/ { 0xC0, 0xAF, 0x00}, { 0xFFFD, 0xFFFD, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_OVERLONG_3,*/ { 0xE0, 0x80, 0xAF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_OVERLONG_4,*/ { 0xF0, 0x80, 0x80, 0xAF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_OVERLONG_5,*/ { 0xF8, 0x80, 0x80, 0x80, 0xAF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
|
|
{ /*SPECIAL_SEQ_OVERLONG_6,*/ { 0xFC, 0x80, 0x80, 0x80, 0x80, 0xAF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
|
|
};
|
|
|
|
const u32 specialSequenceCount = ITF_ARRAY_SIZE(specialSeqences);
|
|
for(u32 i = 0; i < specialSequenceCount; ++i)
|
|
{
|
|
const UnicodeTestTools::SpecialSeqInfo& specialSeq = specialSeqences[i];
|
|
|
|
const UnicodeChar* end = std::find(specialSeq.unicodeSeqExpected,
|
|
specialSeq.unicodeSeqExpected + UnicodeTestTools::SpecialSeqInfo::SEQUENCE_MAX_SIZE,
|
|
0);
|
|
std::vector<UnicodeChar> expected(specialSeq.unicodeSeqExpected, end);
|
|
std::vector<UnicodeChar> result;
|
|
|
|
const char* currentUTF8 = reinterpret_cast<const char*>(specialSeq.utf8);
|
|
while(*currentUTF8)
|
|
{
|
|
UnicodeChar uniChar = getNextUnicodeChar(currentUTF8);
|
|
result.push_back(uniChar);
|
|
};
|
|
|
|
EXPECT_EQ(expected, result);
|
|
}
|
|
}
|
|
|
|
TEST(UnicodeTools, test_UTF8ToUpper)
|
|
{
|
|
const u16 toBeConvertedAsUTF16[] = { 0x00E0, 0x00E1, 0x00E2, 0x00E3, 0x00FE, 0x00FF, 0x0101, 0x0103, 0x0000 };
|
|
const u16 expectedAsUTF16[] = { 0x00C0, 0x00C1, 0x00C2, 0x00C3, 0x00DE, 0x0178, 0x0100, 0x0102, 0x0000 };
|
|
|
|
String8 toBeConverted(UTF16ToUTF8(toBeConvertedAsUTF16).get());
|
|
String8 expected(UTF16ToUTF8(expectedAsUTF16).get());
|
|
|
|
UTF8ToUpper upperString(toBeConverted.cStr());
|
|
String8 converted(upperString.get());
|
|
EXPECT_EQ(expected, converted);
|
|
}
|
|
|
|
TEST(UnicodeTools, test_UTF8ToLower)
|
|
{
|
|
const u16 toBeConvertedAsUTF16[] = { 0x00C0, 0x00C1, 0x00C2, 0x00C3, 0x00DE, 0x0178, 0x0100, 0x0102, 0x0000 };
|
|
const u16 expectedAsUTF16[] = { 0x00E0, 0x00E1, 0x00E2, 0x00E3, 0x00FE, 0x00FF, 0x0101, 0x0103, 0x0000 };
|
|
|
|
String8 toBeConverted(UTF16ToUTF8(toBeConvertedAsUTF16).get());
|
|
String8 expected(UTF16ToUTF8(expectedAsUTF16).get());
|
|
|
|
UTF8ToLower lowerString(toBeConverted.cStr());
|
|
String8 converted(lowerString.get());
|
|
EXPECT_EQ(expected, converted);
|
|
}
|
|
|
|
TEST(UnicodeTools, test_UTF8ToUpper_Ascii)
|
|
{
|
|
String8 toBeConverted = "azertyuiopqsdfghjklmwxcvbn1234567890";
|
|
String8 expected = "AZERTYUIOPQSDFGHJKLMWXCVBN1234567890";
|
|
|
|
UTF8ToUpper upperString(toBeConverted.cStr());
|
|
String8 converted(upperString.get());
|
|
EXPECT_EQ(expected, converted);
|
|
}
|
|
|
|
TEST(UnicodeTools, test_UTF8ToLower_Ascii)
|
|
{
|
|
String8 toBeConverted = "AZERTYUIOPQSDFGHJKLMWXCVBN1234567890";
|
|
String8 expected = "azertyuiopqsdfghjklmwxcvbn1234567890";
|
|
|
|
UTF8ToLower lowerString(toBeConverted.cStr());
|
|
String8 converted(lowerString.get());
|
|
EXPECT_EQ(expected, converted);
|
|
}
|
|
|
|
|
|
} // namespace ITF
|