JD2022-TU1/main/unittests/core/UnicodeTools.cpp

322 lines
15 KiB
C++

// ----------------------------------------------------------------------------
// Unit test for the class core/UnicodeTools
// ----------------------------------------------------------------------------
#include "precompiled_unittests_core.h"
#include "core/UnicodeTools.h"
namespace ITF
{
namespace UnicodeTestTools
{
enum SampleUnicodeChars
{
// ASCII codes
SAMPLE_UNICODE_NULL = 0x0000,
SAMPLE_UNICODE_SPACE = 0x0020,
SAMPLE_UNICODE_A = 0x0041,
SAMPLE_UNICODE_TILDE = 0x007E,
SAMPLE_UNICODE_DEL = 0x007F,
// UTF8 in 2 bytes
SAMPLE_UNICODE_PAD = 0x0080,
SAMPLE_UNICODE_ECUTE = 0x00E9,
SAMPLE_UNICODE_7FF = 0x07FF,
// UTF8 in 3 bytes
SAMPLE_UNICODE_EURO = 0x20AC,
SAMPLE_UNICODE_CHINESE = 0xF900,
SAMPLE_UNICODE_FDF0 = 0xFDF0,
// UTF8 in 4 bytes
SAMPLE_UNICODE_20000 = 0x20000,
SAMPLE_UNICODE_2FFFD = 0x2FFFD,
};
struct UnicodeCharInfo
{
static const u32 MAX_UTF8_CHARS = 4 + 1; // 1 is for zero termination
SampleUnicodeChars m_unicode;
char m_utf8[MAX_UTF8_CHARS];
explicit UnicodeCharInfo(SampleUnicodeChars _unicode) : m_unicode(_unicode){ m_utf8[0] = m_utf8[1] = m_utf8[2] = m_utf8[3] = m_utf8[4] = 0; }
UnicodeCharInfo(SampleUnicodeChars _unicode, u8 _c0)
: m_unicode(_unicode) { m_utf8[0] = _c0; m_utf8[1] = 0; m_utf8[2] = 0; m_utf8[3] = 0; m_utf8[4] = 0; }
UnicodeCharInfo(SampleUnicodeChars _unicode, u8 _c0, u8 _c1)
: m_unicode(_unicode) { m_utf8[0] = _c0; m_utf8[1] = _c1; m_utf8[2] = 0; m_utf8[3] = 0; m_utf8[4] = 0; }
UnicodeCharInfo(SampleUnicodeChars _unicode, u8 _c0, u8 _c1, u8 _c2)
: m_unicode(_unicode) { m_utf8[0] = _c0; m_utf8[1] = _c1; m_utf8[2] = _c2; m_utf8[3] = 0; m_utf8[4] = 0; }
UnicodeCharInfo(SampleUnicodeChars _unicode, u8 _c0, u8 _c1, u8 _c2, u8 _c3)
: m_unicode(_unicode) { m_utf8[0] = _c0; m_utf8[1] = _c1; m_utf8[2] = _c2; m_utf8[3] = _c3; m_utf8[4] = 0; }
bool operator ==(const UnicodeCharInfo& _rhs) const { return m_unicode == _rhs.m_unicode; }
};
static UnicodeCharInfo sampleUnicodeChars[] =
{
UnicodeCharInfo(SAMPLE_UNICODE_NULL , 0x00),
UnicodeCharInfo(SAMPLE_UNICODE_SPACE , 0x20),
UnicodeCharInfo(SAMPLE_UNICODE_A , 0x41),
UnicodeCharInfo(SAMPLE_UNICODE_TILDE , 0x7E),
UnicodeCharInfo(SAMPLE_UNICODE_DEL , 0x7F),
UnicodeCharInfo(SAMPLE_UNICODE_PAD , 0xC2, 0x80),
UnicodeCharInfo(SAMPLE_UNICODE_ECUTE , 0xC3, 0xA9),
UnicodeCharInfo(SAMPLE_UNICODE_7FF , 0xDF, 0xBF),
UnicodeCharInfo(SAMPLE_UNICODE_EURO , 0xE2, 0x82, 0xAC),
UnicodeCharInfo(SAMPLE_UNICODE_CHINESE, 0xEF, 0xA4, 0x80),
UnicodeCharInfo(SAMPLE_UNICODE_FDF0 , 0xEF, 0xB7, 0xB0),
UnicodeCharInfo(SAMPLE_UNICODE_20000 , 0xF0, 0xA0, 0x80, 0x80),
UnicodeCharInfo(SAMPLE_UNICODE_2FFFD , 0xF0, 0xAF, 0xBF, 0xBD)
};
static void buildUTF8String(const SampleUnicodeChars* _first, const SampleUnicodeChars* _last, String8& _str)
{
const UnicodeCharInfo* first = sampleUnicodeChars;
const UnicodeCharInfo* last = sampleUnicodeChars + ITF_ARRAY_SIZE(sampleUnicodeChars);
for( ; _first != _last; ++_first)
{
const UnicodeCharInfo* uci = std::find(first, last, UnicodeCharInfo(*_first));
if (uci)
{
_str += uci->m_utf8;
}
}
}
struct SpecialSeqInfo
{
enum { SEQUENCE_MAX_SIZE = 7 };
u8 utf8[SEQUENCE_MAX_SIZE];
UnicodeChar unicodeSeqExpected[SEQUENCE_MAX_SIZE];
};
} // namespace UnicodeTestTools
// test decoding of a sequence of utf8 encoded characters in a string8
// using the functions getNbUnicodeChar() and getNextUnicodeChar()
TEST(UnicodeTools, test_utf8Decoding_1)
{
UnicodeTestTools::SampleUnicodeChars strCodes[] =
{
UnicodeTestTools::SAMPLE_UNICODE_SPACE,
UnicodeTestTools::SAMPLE_UNICODE_EURO,
UnicodeTestTools::SAMPLE_UNICODE_DEL,
UnicodeTestTools::SAMPLE_UNICODE_20000,
UnicodeTestTools::SAMPLE_UNICODE_A,
UnicodeTestTools::SAMPLE_UNICODE_ECUTE,
};
u32 expectedNbUnicodeChar = ITF_ARRAY_SIZE(strCodes);
String8 utf8String;
UnicodeTestTools::buildUTF8String(strCodes, strCodes + ITF_ARRAY_SIZE(strCodes), utf8String);
// check
const char* utfString8CurrentPos = utf8String.cStr();
EXPECT_EQ(expectedNbUnicodeChar, getNbUnicodeChar(utfString8CurrentPos));
for(u32 i = 0; i < expectedNbUnicodeChar; ++i)
{
ASSERT_NE(0, *utfString8CurrentPos); // unexpected end of string
u32 unicodeChar = getNextUnicodeChar(utfString8CurrentPos);
EXPECT_EQ(unicodeChar, u32(strCodes[i]));
}
EXPECT_EQ(0, *utfString8CurrentPos);
}
// test decoding of a sequence of utf8 encoded characters in a string8
// using the functions getNbUnicodeChar() and getNextUnicodeChar()
TEST(UnicodeTools, test_utf8Decoding_2)
{
UnicodeTestTools::SampleUnicodeChars strCodes[] =
{
UnicodeTestTools::SAMPLE_UNICODE_2FFFD,
UnicodeTestTools::SAMPLE_UNICODE_TILDE,
UnicodeTestTools::SAMPLE_UNICODE_CHINESE,
UnicodeTestTools::SAMPLE_UNICODE_PAD,
UnicodeTestTools::SAMPLE_UNICODE_A,
UnicodeTestTools::SAMPLE_UNICODE_7FF,
UnicodeTestTools::SAMPLE_UNICODE_FDF0,
};
u32 expectedNbUnicodeChar = ITF_ARRAY_SIZE(strCodes);
String8 utf8String;
UnicodeTestTools::buildUTF8String(strCodes, strCodes + ITF_ARRAY_SIZE(strCodes), utf8String);
// check
const char* utfString8CurrentPos = utf8String.cStr();
EXPECT_EQ(expectedNbUnicodeChar, getNbUnicodeChar(utfString8CurrentPos));
for(u32 i = 0; i < expectedNbUnicodeChar; ++i)
{
ASSERT_NE(0, *utfString8CurrentPos); // unexpected end of string
u32 unicodeChar = getNextUnicodeChar(utfString8CurrentPos);
EXPECT_EQ(unicodeChar, u32(strCodes[i]));
}
EXPECT_EQ(0, *utfString8CurrentPos);
}
// test decoding of a pure Ascii sequence of characters in a string8
// using the functions getNbUnicodeChar() and getNextUnicodeChar()
TEST(UnicodeTools, test_utf8Decoding_Ascii)
{
String8 strPureAscii = "Test\nAscii"; // containing only ascii chars
u32 expectedNbUnicodeChar = strPureAscii.getLen();
const char* utfString8CurrentPos = strPureAscii.cStr();
EXPECT_EQ(expectedNbUnicodeChar, getNbUnicodeChar(utfString8CurrentPos));
for(u32 i = 0; i < expectedNbUnicodeChar; ++i)
{
ASSERT_NE(0, *utfString8CurrentPos); // unexpected end of string
const char currentChar = *utfString8CurrentPos;
u32 unicodeChar = getNextUnicodeChar(utfString8CurrentPos);
EXPECT_EQ(unicodeChar, u32(currentChar));
}
EXPECT_EQ(0, *utfString8CurrentPos);
}
// test UTF16 <-> UTF8 conversion functions
TEST(UnicodeTools, test_utf16)
{
static const u8 unsignedKosme[] = { 'k', 'o', 's', 'm', 'e', // kosme in ascii
0xF0, 0xA4, 0xAD, 0xA2, // additionnal chinese character (to test UTF16 surrogate)
0xCE, 0xBA, 0xE1, 0xBD, 0xB9, 0xCF, 0x83, 0xCE, 0xBC, 0xCE, 0xB5, // kosme in greek
0x00};
const char * kosme = reinterpret_cast<const char *>(unsignedKosme);
u32 nbU16Word = getNbUTF16Word(kosme);
u16 * utf16Kosme = newAlloc(mId_Temporary, u16[nbU16Word]);
buildUTF16FromUTF8(kosme, utf16Kosme, nbU16Word);
u32 nbU8 = getNbUTF8Byte(utf16Kosme)+1u;
char * utf8Kosme = newAlloc(mId_Temporary, char[nbU8]);
buildUTF8FromUTF16(utf16Kosme, utf8Kosme, nbU8);
EXPECT_EQ(0, strcmp(utf8Kosme, kosme));
delete [] utf8Kosme;
delete [] utf16Kosme;
}
// test special UTF sequences
// cf http://www.cl.cam.ac.uk/~mgk25/ucs/examples/UTF-8-test.txt for invalid sequence sample
TEST(UnicodeTools, test_SpecialSequences)
{
UnicodeTestTools::SpecialSeqInfo specialSeqences[] =
{
{ /*SPECIAL_SEQ_BOUND_FIRST_1BYTE,*/ { 0x00 }, { 0x0000 } },
{ /*SPECIAL_SEQ_BOUND_FIRST_2BYTES,*/ { 0xC2, 0x80, 0x00}, { 0x0080, 0x0000 } },
{ /*SPECIAL_SEQ_BOUND_FIRST_3BYTES,*/ { 0xE0, 0xA0, 0x80 , 0x00}, { 0x0800, 0x0000 } },
{ /*SPECIAL_SEQ_BOUND_FIRST_4BYTES,*/ { 0xF0, 0x90, 0x80, 0x80 , 0x00}, { 0x00010000, 0x0000 } },
{ /*SPECIAL_SEQ_BOUND_FIRST_5BYTES,*/ { 0xF8, 0x88, 0x80, 0x80, 0x80, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_BOUND_FIRST_6BYTES,*/ { 0xFC, 0x84, 0x80, 0x80, 0x80, 0x80, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_BOUND_LAST_1BYTE,*/ { 0x7F, 0x00}, { 0x007F, 0x0000 } },
{ /*SPECIAL_SEQ_BOUND_LAST_2BYTES,*/ { 0xDF, 0xBF, 0x00}, { 0x07FF, 0x0000 } },
{ /*SPECIAL_SEQ_BOUND_LAST_3BYTES,*/ { 0xEF, 0xBF, 0xBF, 0x00}, { 0xFFFF, 0x0000 } },
{ /*SPECIAL_SEQ_BOUND_LAST_4BYTES,*/ { 0xF7, 0xBF, 0xBF, 0xBF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_BOUND_LAST_5BYTES,*/ { 0xFB, 0xBF, 0xBF, 0xBF, 0xBF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD,0x0000 } },
{ /*SPECIAL_SEQ_BOUND_LAST_6BYTES,*/ { 0xFD, 0xBF, 0xBF, 0xBF, 0xBF, 0xBF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_BOUND_OTHER_D7FF,*/ { 0xED, 0x9F, 0xBF, 0x00}, { 0xD7FF, 0x0000 } },
{ /*SPECIAL_SEQ_BOUND_OTHER_E000,*/ { 0xEE, 0x80, 0x80, 0x00}, { 0xE000, 0x0000 } },
{ /*SPECIAL_SEQ_BOUND_OTHER_FFFD,*/ { 0xEF, 0xBF, 0xBD, 0x00}, { 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_BOUND_OTHER_10FFFF,*/ { 0xF4, 0x8F, 0xBF, 0xBF, 0x00}, { 0x10FFFF, 0x0000 } },
{ /*SPECIAL_SEQ_BOUND_OTHER_110000,*/ { 0xF4, 0x90, 0x80, 0x80, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_CONT_FIRST,*/ { 0x80, 0x00}, { 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_CONT_LAST,*/ { 0xBF, 0x00}, { 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_CONT_2BYTES,*/ { 0x80, 0xBF, 0x00}, { 0xFFFD, 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_CONT_3BYTES,*/ { 0x80, 0xBF, 0x80, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_CONT_4BYTES,*/ { 0x80, 0xBF, 0x80, 0xBF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_CONT_5BYTES,*/ { 0x80, 0xBF, 0x80, 0xBF, 0x80, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_CONT_6BYTES,*/ { 0x80, 0xBF, 0x80, 0xBF, 0x80, 0xBF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_IMPOSSIBLE_FE,*/ { 0xFE, 0x00}, { 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_IMPOSSIBLE_FF,*/ { 0xFF, 0x00}, { 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_IMPOSSIBLE_FEFF,*/ { 0xFE, 0xFE, 0xFF, 0xFF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_OVERLONG_2,*/ { 0xC0, 0xAF, 0x00}, { 0xFFFD, 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_OVERLONG_3,*/ { 0xE0, 0x80, 0xAF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_OVERLONG_4,*/ { 0xF0, 0x80, 0x80, 0xAF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_OVERLONG_5,*/ { 0xF8, 0x80, 0x80, 0x80, 0xAF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
{ /*SPECIAL_SEQ_OVERLONG_6,*/ { 0xFC, 0x80, 0x80, 0x80, 0x80, 0xAF, 0x00}, { 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0x0000 } },
};
const u32 specialSequenceCount = ITF_ARRAY_SIZE(specialSeqences);
for(u32 i = 0; i < specialSequenceCount; ++i)
{
const UnicodeTestTools::SpecialSeqInfo& specialSeq = specialSeqences[i];
const UnicodeChar* end = std::find(specialSeq.unicodeSeqExpected,
specialSeq.unicodeSeqExpected + UnicodeTestTools::SpecialSeqInfo::SEQUENCE_MAX_SIZE,
0);
std::vector<UnicodeChar> expected(specialSeq.unicodeSeqExpected, end);
std::vector<UnicodeChar> result;
const char* currentUTF8 = reinterpret_cast<const char*>(specialSeq.utf8);
while(*currentUTF8)
{
UnicodeChar uniChar = getNextUnicodeChar(currentUTF8);
result.push_back(uniChar);
};
EXPECT_EQ(expected, result);
}
}
TEST(UnicodeTools, test_UTF8ToUpper)
{
const u16 toBeConvertedAsUTF16[] = { 0x00E0, 0x00E1, 0x00E2, 0x00E3, 0x00FE, 0x00FF, 0x0101, 0x0103, 0x0000 };
const u16 expectedAsUTF16[] = { 0x00C0, 0x00C1, 0x00C2, 0x00C3, 0x00DE, 0x0178, 0x0100, 0x0102, 0x0000 };
String8 toBeConverted(UTF16ToUTF8(toBeConvertedAsUTF16).get());
String8 expected(UTF16ToUTF8(expectedAsUTF16).get());
UTF8ToUpper upperString(toBeConverted.cStr());
String8 converted(upperString.get());
EXPECT_EQ(expected, converted);
}
TEST(UnicodeTools, test_UTF8ToLower)
{
const u16 toBeConvertedAsUTF16[] = { 0x00C0, 0x00C1, 0x00C2, 0x00C3, 0x00DE, 0x0178, 0x0100, 0x0102, 0x0000 };
const u16 expectedAsUTF16[] = { 0x00E0, 0x00E1, 0x00E2, 0x00E3, 0x00FE, 0x00FF, 0x0101, 0x0103, 0x0000 };
String8 toBeConverted(UTF16ToUTF8(toBeConvertedAsUTF16).get());
String8 expected(UTF16ToUTF8(expectedAsUTF16).get());
UTF8ToLower lowerString(toBeConverted.cStr());
String8 converted(lowerString.get());
EXPECT_EQ(expected, converted);
}
TEST(UnicodeTools, test_UTF8ToUpper_Ascii)
{
String8 toBeConverted = "azertyuiopqsdfghjklmwxcvbn1234567890";
String8 expected = "AZERTYUIOPQSDFGHJKLMWXCVBN1234567890";
UTF8ToUpper upperString(toBeConverted.cStr());
String8 converted(upperString.get());
EXPECT_EQ(expected, converted);
}
TEST(UnicodeTools, test_UTF8ToLower_Ascii)
{
String8 toBeConverted = "AZERTYUIOPQSDFGHJKLMWXCVBN1234567890";
String8 expected = "azertyuiopqsdfghjklmwxcvbn1234567890";
UTF8ToLower lowerString(toBeConverted.cStr());
String8 converted(lowerString.get());
EXPECT_EQ(expected, converted);
}
} // namespace ITF