From 65a2cb6964f3d1178718203f05e9143bd400c80e Mon Sep 17 00:00:00 2001 From: Ievgenii Meshcheriakov Date: Wed, 1 Sep 2021 12:46:57 +0200 Subject: [PATCH] corelib: Use char16_t and char32_t types for characters Use standard char16_t and char32_t types instead of ushort and uint. Remove members of QUtf8BaseTraits that use those integer types. Change-Id: I77b1a9106244835c813336a50417f6bbdfada288 Reviewed-by: Edward Welbourne --- src/corelib/io/qurlrecode.cpp | 81 +++++++++++------------ src/corelib/serialization/qjsonparser.cpp | 10 +-- src/corelib/serialization/qjsonwriter.cpp | 6 +- src/corelib/text/qstring.cpp | 6 +- src/corelib/text/qstringconverter.cpp | 74 ++++++++++----------- src/corelib/text/qstringconverter_p.h | 54 ++++----------- 6 files changed, 100 insertions(+), 131 deletions(-) diff --git a/src/corelib/io/qurlrecode.cpp b/src/corelib/io/qurlrecode.cpp index 44e0a8239d..fd5accb108 100644 --- a/src/corelib/io/qurlrecode.cpp +++ b/src/corelib/io/qurlrecode.cpp @@ -167,48 +167,45 @@ static const uchar reservedMask[96] = { 0xff // BSKP }; -static inline bool isHex(ushort c) +static inline bool isHex(char16_t c) { - return (c >= 'a' && c <= 'f') || - (c >= 'A' && c <= 'F') || - (c >= '0' && c <= '9'); + return (c >= u'a' && c <= u'f') || (c >= u'A' && c <= u'F') || (c >= u'0' && c <= u'9'); } -static inline bool isUpperHex(ushort c) +static inline bool isUpperHex(char16_t c) { // undefined behaviour if c isn't an hex char! return c < 0x60; } -static inline ushort toUpperHex(ushort c) +static inline char16_t toUpperHex(char16_t c) { return isUpperHex(c) ? c : c - 0x20; } -static inline ushort decodeNibble(ushort c) +static inline ushort decodeNibble(char16_t c) { - return c >= 'a' ? c - 'a' + 0xA : - c >= 'A' ? c - 'A' + 0xA : c - '0'; + return c >= u'a' ? c - u'a' + 0xA : c >= u'A' ? c - u'A' + 0xA : c - u'0'; } // if the sequence at input is 2*HEXDIG, returns its decoding // returns -1 if it isn't. // assumes that the range has been checked already -static inline ushort decodePercentEncoding(const ushort *input) +static inline char16_t decodePercentEncoding(const char16_t *input) { - ushort c1 = input[1]; - ushort c2 = input[2]; + char16_t c1 = input[1]; + char16_t c2 = input[2]; if (!isHex(c1) || !isHex(c2)) - return ushort(-1); + return char16_t(-1); return decodeNibble(c1) << 4 | decodeNibble(c2); } -static inline ushort encodeNibble(ushort c) +static inline char16_t encodeNibble(ushort c) { - return ushort(QtMiscUtils::toHexUpper(c)); + return QtMiscUtils::toHexUpper(c); } -static void ensureDetached(QString &result, ushort *&output, const ushort *begin, const ushort *input, const ushort *end, +static void ensureDetached(QString &result, char16_t *&output, const char16_t *begin, const char16_t *input, const char16_t *end, int add = 0) { if (!output) { @@ -221,7 +218,7 @@ static void ensureDetached(QString &result, ushort *&output, const ushort *begin result.resize(origSize + spaceNeeded); // we know that resize() above detached, so we bypass the reference count check - output = const_cast(reinterpret_cast(result.constData())) + output = const_cast(reinterpret_cast(result.constData())) + origSize; // copy the chars we've already processed @@ -260,7 +257,7 @@ struct QUrlUtf8Traits : public QUtf8BaseTraitsNoAscii static const bool allowNonCharacters = false; // override: our "bytes" are three percent-encoded UTF-16 characters - static void appendByte(ushort *&ptr, uchar b) + static void appendByte(char16_t *&ptr, uchar b) { // b >= 0x80, by construction, so percent-encode *ptr++ = '%'; @@ -268,9 +265,9 @@ struct QUrlUtf8Traits : public QUtf8BaseTraitsNoAscii *ptr++ = encodeNibble(b & 0xf); } - static uchar peekByte(const ushort *ptr, qsizetype n = 0) + static uchar peekByte(const char16_t *ptr, qsizetype n = 0) { - // decodePercentEncoding returns ushort(-1) if it can't decode, + // decodePercentEncoding returns char16_t(-1) if it can't decode, // which means we return 0xff, which is not a valid continuation byte. // If ptr[i * 3] is not '%', we'll multiply by zero and return 0, // also not a valid continuation byte (if it's '%', we multiply by 1). @@ -278,12 +275,12 @@ struct QUrlUtf8Traits : public QUtf8BaseTraitsNoAscii * uchar(ptr[n * 3] == '%'); } - static qptrdiff availableBytes(const ushort *ptr, const ushort *end) + static qptrdiff availableBytes(const char16_t *ptr, const char16_t *end) { return (end - ptr) / 3; } - static void advanceByte(const ushort *&ptr, int n = 1) + static void advanceByte(const char16_t *&ptr, int n = 1) { ptr += n * 3; } @@ -291,11 +288,11 @@ struct QUrlUtf8Traits : public QUtf8BaseTraitsNoAscii } // returns true if we performed an UTF-8 decoding -static bool encodedUtf8ToUtf16(QString &result, ushort *&output, const ushort *begin, const ushort *&input, - const ushort *end, ushort decoded) +static bool encodedUtf8ToUtf16(QString &result, char16_t *&output, const char16_t *begin, + const char16_t *&input, const char16_t *end, char16_t decoded) { - uint ucs4 = 0, *dst = &ucs4; - const ushort *src = input + 3;// skip the %XX that yielded \a decoded + char32_t ucs4 = 0, *dst = &ucs4; + const char16_t *src = input + 3;// skip the %XX that yielded \a decoded int charsNeeded = QUtf8Functions::fromUtf8(decoded, dst, src, end); if (charsNeeded < 0) return false; @@ -318,8 +315,8 @@ static bool encodedUtf8ToUtf16(QString &result, ushort *&output, const ushort *b return true; } -static void unicodeToEncodedUtf8(QString &result, ushort *&output, const ushort *begin, - const ushort *&input, const ushort *end, ushort decoded) +static void unicodeToEncodedUtf8(QString &result, char16_t *&output, const char16_t *begin, + const char16_t *&input, const char16_t *end, char16_t decoded) { // calculate the utf8 length and ensure enough space is available int utf8len = QChar::isHighSurrogate(decoded) ? 4 : decoded >= 0x800 ? 3 : 2; @@ -332,14 +329,14 @@ static void unicodeToEncodedUtf8(QString &result, ushort *&output, const ushort } else { // verify that there's enough space or expand int charsRemaining = end - input - 1; // not including this one - int pos = output - reinterpret_cast(result.constData()); + int pos = output - reinterpret_cast(result.constData()); int spaceRemaining = result.size() - pos; if (spaceRemaining < 3*charsRemaining + 3*utf8len) { // must resize result.resize(result.size() + 3*utf8len); // we know that resize() above detached, so we bypass the reference count check - output = const_cast(reinterpret_cast(result.constData())); + output = const_cast(reinterpret_cast(result.constData())); output += pos; } } @@ -372,16 +369,17 @@ static void unicodeToEncodedUtf8(QString &result, ushort *&output, const ushort } } -static int recode(QString &result, const ushort *begin, const ushort *end, QUrl::ComponentFormattingOptions encoding, - const uchar *actionTable, bool retryBadEncoding) +static int recode(QString &result, const char16_t *begin, const char16_t *end, + QUrl::ComponentFormattingOptions encoding, const uchar *actionTable, + bool retryBadEncoding) { const int origSize = result.size(); - const ushort *input = begin; - ushort *output = nullptr; + const char16_t *input = begin; + char16_t *output = nullptr; EncodingAction action = EncodeCharacter; for ( ; input != end; ++input) { - ushort c; + char16_t c; // try a run where no change is necessary for ( ; input != end; ++input) { c = *input; @@ -398,7 +396,7 @@ static int recode(QString &result, const ushort *begin, const ushort *end, QUrl: break; non_trivial: - uint decoded; + char16_t decoded; if (c == '%' && retryBadEncoding) { // always write "%25" ensureDetached(result, output, begin, input, end); @@ -408,7 +406,7 @@ non_trivial: continue; } else if (c == '%') { // check if the input is valid - if (input + 2 >= end || (decoded = decodePercentEncoding(input)) == ushort(-1)) { + if (input + 2 >= end || (decoded = decodePercentEncoding(input)) == char16_t(-1)) { // not valid, retry result.resize(origSize); return recode(result, begin, end, encoding, actionTable, true); @@ -468,7 +466,7 @@ non_trivial: } if (output) { - int len = output - reinterpret_cast(result.constData()); + int len = output - reinterpret_cast(result.constData()); result.truncate(len); return len - origSize; } @@ -603,7 +601,8 @@ static qsizetype decode(QString &appendTo, QStringView in) if (Q_UNLIKELY(end - input < 3 || !isHex(input[1]) || !isHex(input[2]))) { // badly-encoded data appendTo.resize(origSize + (end - begin)); - memcpy(static_cast(appendTo.begin() + origSize), static_cast(begin), (end - begin) * sizeof(ushort)); + memcpy(static_cast(appendTo.begin() + origSize), + static_cast(begin), (end - begin) * sizeof(*end)); return end - begin; } @@ -691,8 +690,8 @@ qt_urlRecode(QString &appendTo, QStringView in, actionTable[uchar(*p) - ' '] = *p >> 8; } - return recode(appendTo, reinterpret_cast(in.begin()), reinterpret_cast(in.end()), - encoding, actionTable, false); + return recode(appendTo, reinterpret_cast(in.begin()), + reinterpret_cast(in.end()), encoding, actionTable, false); } QT_END_NAMESPACE diff --git a/src/corelib/serialization/qjsonparser.cpp b/src/corelib/serialization/qjsonparser.cpp index 525cbfb3a0..9760fde2ed 100644 --- a/src/corelib/serialization/qjsonparser.cpp +++ b/src/corelib/serialization/qjsonparser.cpp @@ -766,7 +766,7 @@ bool Parser::parseNumber() unescaped = %x20-21 / %x23-5B / %x5D-10FFFF */ -static inline bool addHexDigit(char digit, uint *result) +static inline bool addHexDigit(char digit, char32_t *result) { *result <<= 4; if (digit >= '0' && digit <= '9') @@ -780,7 +780,7 @@ static inline bool addHexDigit(char digit, uint *result) return true; } -static inline bool scanEscapeSequence(const char *&json, const char *end, uint *ch) +static inline bool scanEscapeSequence(const char *&json, const char *end, char32_t *ch) { ++json; if (json >= end) @@ -825,7 +825,7 @@ static inline bool scanEscapeSequence(const char *&json, const char *end, uint * return true; } -static inline bool scanUtf8Char(const char *&json, const char *end, uint *result) +static inline bool scanUtf8Char(const char *&json, const char *end, char32_t *result) { const auto *usrc = reinterpret_cast(json); const auto *uend = reinterpret_cast(end); @@ -848,7 +848,7 @@ bool Parser::parseString() bool isUtf8 = true; bool isAscii = true; while (json < end) { - uint ch = 0; + char32_t ch = 0; if (*json == '"') break; if (*json == '\\') { @@ -890,7 +890,7 @@ bool Parser::parseString() QString ucs4; while (json < end) { - uint ch = 0; + char32_t ch = 0; if (*json == '"') break; else if (*json == '\\') { diff --git a/src/corelib/serialization/qjsonwriter.cpp b/src/corelib/serialization/qjsonwriter.cpp index 8610cdff7e..54aad864a5 100644 --- a/src/corelib/serialization/qjsonwriter.cpp +++ b/src/corelib/serialization/qjsonwriter.cpp @@ -65,8 +65,8 @@ static QByteArray escapedString(const QString &s) uchar *cursor = reinterpret_cast(const_cast(ba.constData())); const uchar *ba_end = cursor + ba.length(); - const ushort *src = reinterpret_cast(s.constBegin()); - const ushort *const end = reinterpret_cast(s.constEnd()); + const char16_t *src = reinterpret_cast(s.constBegin()); + const char16_t *const end = reinterpret_cast(s.constEnd()); while (src != end) { if (cursor >= ba_end - 6) { @@ -77,7 +77,7 @@ static QByteArray escapedString(const QString &s) ba_end = (const uchar *)ba.constData() + ba.length(); } - uint u = *src++; + char16_t u = *src++; if (u < 0x80) { if (u < 0x20 || u == 0x22 || u == 0x5c) { *cursor++ = '\\'; diff --git a/src/corelib/text/qstring.cpp b/src/corelib/text/qstring.cpp index 23da7e8b18..ec813ee750 100644 --- a/src/corelib/text/qstring.cpp +++ b/src/corelib/text/qstring.cpp @@ -885,8 +885,8 @@ static int ucstricmp8(const char *utf8, const char *utf8end, const QChar *utf16, QStringIterator src2(utf16, utf16end); while (src1 < end1 && src2.hasNext()) { - uint uc1 = 0; - uint *output = &uc1; + char32_t uc1 = 0; + char32_t *output = &uc1; uchar b = *src1++; int res = QUtf8Functions::fromUtf8(b, output, src1, end1); if (res < 0) { @@ -896,7 +896,7 @@ static int ucstricmp8(const char *utf8, const char *utf8end, const QChar *utf16, uc1 = QChar::toCaseFolded(uc1); } - uint uc2 = QChar::toCaseFolded(src2.next()); + char32_t uc2 = QChar::toCaseFolded(src2.next()); int diff = uc1 - uc2; // can't underflow if (diff) return diff; diff --git a/src/corelib/text/qstringconverter.cpp b/src/corelib/text/qstringconverter.cpp index df9efe7f67..09ac6512be 100644 --- a/src/corelib/text/qstringconverter.cpp +++ b/src/corelib/text/qstringconverter.cpp @@ -78,7 +78,7 @@ static Q_ALWAYS_INLINE uint qBitScanReverse(unsigned v) noexcept #endif #if defined(__SSE2__) && defined(QT_COMPILER_SUPPORTS_SSE2) -static inline bool simdEncodeAscii(uchar *&dst, const ushort *&nextAscii, const ushort *&src, const ushort *end) +static inline bool simdEncodeAscii(uchar *&dst, const char16_t *&nextAscii, const char16_t *&src, const char16_t *end) { // do sixteen characters at a time for ( ; end - src >= 16; src += 16, dst += 16) { @@ -142,7 +142,7 @@ static inline bool simdEncodeAscii(uchar *&dst, const ushort *&nextAscii, const return src == end; } -static inline bool simdDecodeAscii(ushort *&dst, const uchar *&nextAscii, const uchar *&src, const uchar *end) +static inline bool simdDecodeAscii(char16_t *&dst, const uchar *&nextAscii, const uchar *&src, const uchar *end) { // do sixteen characters at a time for ( ; end - src >= 16; src += 16, dst += 16) { @@ -361,7 +361,7 @@ static void simdCompareAscii(const char8_t *&src8, const char8_t *end8, const ch src16 += offset; } #elif defined(__ARM_NEON__) -static inline bool simdEncodeAscii(uchar *&dst, const ushort *&nextAscii, const ushort *&src, const ushort *end) +static inline bool simdEncodeAscii(uchar *&dst, const char16_t *&nextAscii, const char16_t *&src, const char16_t *end) { uint16x8_t maxAscii = vdupq_n_u16(0x7f); uint16x8_t mask1 = { 1, 1 << 2, 1 << 4, 1 << 6, 1 << 8, 1 << 10, 1 << 12, 1 << 14 }; @@ -370,7 +370,7 @@ static inline bool simdEncodeAscii(uchar *&dst, const ushort *&nextAscii, const // do sixteen characters at a time for ( ; end - src >= 16; src += 16, dst += 16) { // load 2 lanes (or: "load interleaved") - uint16x8x2_t in = vld2q_u16(src); + uint16x8x2_t in = vld2q_u16(reinterpret_cast(src)); // check if any of the elements > 0x7f, select 1 bit per element (element 0 -> bit 0, element 1 -> bit 1, etc), // add those together into a scalar, and merge the scalars. @@ -398,7 +398,7 @@ static inline bool simdEncodeAscii(uchar *&dst, const ushort *&nextAscii, const return src == end; } -static inline bool simdDecodeAscii(ushort *&dst, const uchar *&nextAscii, const uchar *&src, const uchar *end) +static inline bool simdDecodeAscii(char16_t *&dst, const uchar *&nextAscii, const uchar *&src, const uchar *end) { // do eight characters at a time uint8x8_t msb_mask = vdup_n_u8(0x80); @@ -408,7 +408,7 @@ static inline bool simdDecodeAscii(ushort *&dst, const uchar *&nextAscii, const uint8_t n = vaddv_u8(vand_u8(vcge_u8(c, msb_mask), add_mask)); if (!n) { // store - vst1q_u16(dst, vmovl_u8(c)); + vst1q_u16(reinterpret_cast(dst), vmovl_u8(c)); continue; } @@ -461,12 +461,12 @@ static void simdCompareAscii(const char8_t *&, const char8_t *, const char16_t * { } #else -static inline bool simdEncodeAscii(uchar *, const ushort *, const ushort *, const ushort *) +static inline bool simdEncodeAscii(uchar *, const char16_t *, const char16_t *, const char16_t *) { return false; } -static inline bool simdDecodeAscii(ushort *, const uchar *, const uchar *, const uchar *) +static inline bool simdDecodeAscii(char16_t *, const uchar *, const uchar *, const uchar *) { return false; } @@ -491,16 +491,16 @@ QByteArray QUtf8::convertFromUnicode(QStringView in) // create a QByteArray with the worst case scenario size QByteArray result(len * 3, Qt::Uninitialized); uchar *dst = reinterpret_cast(const_cast(result.constData())); - const ushort *src = reinterpret_cast(in.data()); - const ushort *const end = src + len; + const char16_t *src = reinterpret_cast(in.data()); + const char16_t *const end = src + len; while (src != end) { - const ushort *nextAscii = end; + const char16_t *nextAscii = end; if (simdEncodeAscii(dst, nextAscii, src, end)) break; do { - ushort u = *src++; + char16_t u = *src++; int res = QUtf8Functions::toUtf8(u, dst, src, end); if (res < 0) { // encoding error - append '?' @@ -542,8 +542,8 @@ char *QUtf8::convertFromUnicode(char *out, QStringView in, QStringConverter::Sta }; uchar *cursor = reinterpret_cast(out); - const ushort *src = reinterpret_cast(uc); - const ushort *const end = src + len; + const char16_t *src = reinterpret_cast(uc); + const char16_t *const end = src + len; if (!(state->flags & QStringDecoder::Flag::Stateless)) { if (state->remainingChars) { @@ -562,12 +562,12 @@ char *QUtf8::convertFromUnicode(char *out, QStringView in, QStringConverter::Sta } while (src != end) { - const ushort *nextAscii = end; + const char16_t *nextAscii = end; if (simdEncodeAscii(cursor, nextAscii, src, end)) break; do { - ushort uc = *src++; + char16_t uc = *src++; int res = QUtf8Functions::toUtf8(uc, cursor, src, end); if (Q_LIKELY(res >= 0)) continue; @@ -632,7 +632,7 @@ QString QUtf8::convertToUnicode(QByteArrayView in) QChar *QUtf8::convertToUnicode(QChar *buffer, QByteArrayView in) noexcept { - ushort *dst = reinterpret_cast(buffer); + char16_t *dst = reinterpret_cast(buffer); const uchar *const start = reinterpret_cast(in.data()); const uchar *src = start; const uchar *end = src + in.size(); @@ -694,14 +694,14 @@ QChar *QUtf8::convertToUnicode(QChar *out, QByteArrayView in, QStringConverter:: return out; - ushort replacement = QChar::ReplacementCharacter; + char16_t replacement = QChar::ReplacementCharacter; if (state->flags & QStringConverter::Flag::ConvertInvalidToNull) replacement = QChar::Null; int res; uchar ch = 0; - ushort *dst = reinterpret_cast(out); + char16_t *dst = reinterpret_cast(out); const uchar *src = reinterpret_cast(in.data()); const uchar *end = src + len; @@ -791,8 +791,8 @@ QChar *QUtf8::convertToUnicode(QChar *out, QByteArrayView in, QStringConverter:: struct QUtf8NoOutputTraits : public QUtf8BaseTraitsNoAscii { struct NoOutput {}; - static void appendUtf16(const NoOutput &, ushort) {} - static void appendUcs4(const NoOutput &, uint) {} + static void appendUtf16(const NoOutput &, char16_t) {} + static void appendUcs4(const NoOutput &, char32_t) {} }; QUtf8::ValidUtf8Result QUtf8::isValidUtf8(QByteArrayView in) @@ -865,7 +865,7 @@ int QUtf8::compareUtf8(QByteArrayView utf8, QStringView utf16) noexcept int QUtf8::compareUtf8(QByteArrayView utf8, QLatin1String s) { - uint uc1 = QChar::Null; + char32_t uc1 = QChar::Null; auto src1 = reinterpret_cast(utf8.data()); auto end1 = src1 + utf8.size(); auto src2 = reinterpret_cast(s.latin1()); @@ -873,14 +873,14 @@ int QUtf8::compareUtf8(QByteArrayView utf8, QLatin1String s) while (src1 < end1 && src2 < end2) { uchar b = *src1++; - uint *output = &uc1; + char32_t *output = &uc1; int res = QUtf8Functions::fromUtf8(b, output, src1, end1); if (res < 0) { // decoding error uc1 = QChar::ReplacementCharacter; } - uint uc2 = *src2++; + char32_t uc2 = *src2++; if (uc1 != uc2) return int(uc1) - int(uc2); } @@ -921,9 +921,9 @@ char *QUtf16::convertFromUnicode(char *out, QStringView in, QStringConverter::St out += 2; } if (endian == BigEndianness) - qToBigEndian(in.data(), in.length(), out); + qToBigEndian(in.data(), in.length(), out); else - qToLittleEndian(in.data(), in.length(), out); + qToLittleEndian(in.data(), in.length(), out); state->remainingChars = 0; state->internalState |= HeaderDone; @@ -998,9 +998,9 @@ QChar *QUtf16::convertToUnicode(QChar *out, QByteArrayView in, QStringConverter: int nPairs = (end - chars) >> 1; if (endian == BigEndianness) - qFromBigEndian(chars, nPairs, out); + qFromBigEndian(chars, nPairs, out); else - qFromLittleEndian(chars, nPairs, out); + qFromLittleEndian(chars, nPairs, out); out += nPairs; state->state_data[Endian] = endian; @@ -1064,7 +1064,7 @@ char *QUtf32::convertFromUnicode(char *out, QStringView in, QStringConverter::St const QChar *uc = in.data(); const QChar *end = in.data() + in.length(); QChar ch; - uint ucs4; + char32_t ucs4; if (state->remainingChars == 1) { auto character = state->state_data[Data]; Q_ASSERT(character <= 0xFFFF); @@ -1165,7 +1165,7 @@ QChar *QUtf32::convertToUnicode(QChar *out, QByteArrayView in, QStringConverter: endian = LittleEndianness; } } - uint code = (endian == BigEndianness) ? qFromBigEndian(tuple) : qFromLittleEndian(tuple); + char32_t code = (endian == BigEndianness) ? qFromBigEndian(tuple) : qFromLittleEndian(tuple); if (headerdone || code != QChar::ByteOrderMark) { if (QChar::requiresSurrogates(code)) { *out++ = QChar(QChar::highSurrogate(code)); @@ -1184,7 +1184,7 @@ QChar *QUtf32::convertToUnicode(QChar *out, QByteArrayView in, QStringConverter: while (chars < end) { tuple[num++] = *chars++; if (num == 4) { - uint code = (endian == BigEndianness) ? qFromBigEndian(tuple) : qFromLittleEndian(tuple); + char32_t code = (endian == BigEndianness) ? qFromBigEndian(tuple) : qFromLittleEndian(tuple); for (char16_t c : QChar::fromUcs4(code)) *out++ = c; num = 0; @@ -1764,10 +1764,10 @@ std::optional QStringConverter::encodingForData(QByt // someone set us up the BOM? qsizetype arraySize = data.size(); if (arraySize > 3) { - uint uc = qFromUnaligned(data.data()); - if (uc == qToBigEndian(uint(QChar::ByteOrderMark))) + char32_t uc = qFromUnaligned(data.data()); + if (uc == qToBigEndian(char32_t(QChar::ByteOrderMark))) return QStringConverter::Utf32BE; - if (uc == qToLittleEndian(uint(QChar::ByteOrderMark))) + if (uc == qToLittleEndian(char32_t(QChar::ByteOrderMark))) return QStringConverter::Utf32LE; if (expectedFirstCharacter) { // catch also anything starting with the expected character @@ -1784,10 +1784,10 @@ std::optional QStringConverter::encodingForData(QByt } if (arraySize > 1) { - ushort uc = qFromUnaligned(data.data()); - if (uc == qToBigEndian(ushort(QChar::ByteOrderMark))) + char16_t uc = qFromUnaligned(data.data()); + if (uc == qToBigEndian(char16_t(QChar::ByteOrderMark))) return QStringConverter::Utf16BE; - if (uc == qToLittleEndian(ushort(QChar::ByteOrderMark))) + if (uc == qToLittleEndian(char16_t(QChar::ByteOrderMark))) return QStringConverter::Utf16LE; if (expectedFirstCharacter) { // catch also anything starting with the expected character diff --git a/src/corelib/text/qstringconverter_p.h b/src/corelib/text/qstringconverter_p.h index 242f3f0303..2ad59af23c 100644 --- a/src/corelib/text/qstringconverter_p.h +++ b/src/corelib/text/qstringconverter_p.h @@ -70,9 +70,6 @@ struct QUtf8BaseTraits static const int Error = -1; static const int EndOfString = -2; - static bool isValidCharacter(uint u) - { return int(u) >= 0; } - static void appendByte(uchar *&ptr, uchar b) { *ptr++ = b; } @@ -97,54 +94,27 @@ struct QUtf8BaseTraits static void advanceByte(const char8_t *&ptr, int n = 1) { ptr += n; } - static void appendUtf16(ushort *&ptr, ushort uc) - { *ptr++ = uc; } - - static void appendUtf16(char16_t *&ptr, ushort uc) + static void appendUtf16(char16_t *&ptr, char16_t uc) { *ptr++ = char16_t(uc); } - static void appendUcs4(ushort *&ptr, uint uc) - { - appendUtf16(ptr, QChar::highSurrogate(uc)); - appendUtf16(ptr, QChar::lowSurrogate(uc)); - } - static void appendUcs4(char16_t *&ptr, char32_t uc) { appendUtf16(ptr, QChar::highSurrogate(uc)); appendUtf16(ptr, QChar::lowSurrogate(uc)); } - static ushort peekUtf16(const ushort *ptr, qsizetype n = 0) - { return ptr[n]; } - - static ushort peekUtf16(const char16_t *ptr, int n = 0) - { return ptr[n]; } - - static qptrdiff availableUtf16(const ushort *ptr, const ushort *end) - { return end - ptr; } + static char16_t peekUtf16(const char16_t *ptr, qsizetype n = 0) { return ptr[n]; } static qptrdiff availableUtf16(const char16_t *ptr, const char16_t *end) { return end - ptr; } - static void advanceUtf16(const ushort *&ptr, qsizetype n = 1) - { ptr += n; } + static void advanceUtf16(const char16_t *&ptr, qsizetype n = 1) { ptr += n; } - static void advanceUtf16(const char16_t *&ptr, int n = 1) - { ptr += n; } - - // it's possible to output to UCS-4 too - static void appendUtf16(uint *&ptr, ushort uc) - { *ptr++ = uc; } - - static void appendUtf16(char32_t *&ptr, ushort uc) + static void appendUtf16(char32_t *&ptr, char16_t uc) { *ptr++ = char32_t(uc); } - static void appendUcs4(uint *&ptr, uint uc) + static void appendUcs4(char32_t *&ptr, char32_t uc) { *ptr++ = uc; } - - static void appendUcs4(char32_t *&ptr, uint uc) - { *ptr++ = char32_t(uc); } }; struct QUtf8BaseTraitsNoAscii : public QUtf8BaseTraits @@ -159,7 +129,7 @@ namespace QUtf8Functions /// if \a u is a high surrogate, Error if the next isn't a low one, /// EndOfString if we run into the end of the string. template inline - int toUtf8(ushort u, OutputPtr &dst, InputPtr &src, InputPtr end) + int toUtf8(char16_t u, OutputPtr &dst, InputPtr &src, InputPtr end) { if (!Traits::skipAsciiHandling && u < 0x80) { // U+0000 to U+007F (US-ASCII) - one byte @@ -183,14 +153,14 @@ namespace QUtf8Functions if (Traits::availableUtf16(src, end) == 0) return Traits::EndOfString; - ushort low = Traits::peekUtf16(src); + char16_t low = Traits::peekUtf16(src); if (!QChar::isHighSurrogate(u)) return Traits::Error; if (!QChar::isLowSurrogate(low)) return Traits::Error; Traits::advanceUtf16(src); - uint ucs4 = QChar::surrogateToUcs4(u, low); + char32_t ucs4 = QChar::surrogateToUcs4(u, low); if (!Traits::allowNonCharacters && QChar::isNonCharacter(ucs4)) return Traits::Error; @@ -202,7 +172,7 @@ namespace QUtf8Functions Traits::appendByte(dst, 0x80 | (uchar(ucs4 >> 12) & 0x3f)); // for the rest of the bytes - u = ushort(ucs4); + u = char16_t(ucs4); } // second to last byte @@ -225,8 +195,8 @@ namespace QUtf8Functions qsizetype fromUtf8(uchar b, OutputPtr &dst, InputPtr &src, InputPtr end) { qsizetype charsNeeded; - uint min_uc; - uint uc; + char32_t min_uc; + char32_t uc; if (!Traits::skipAsciiHandling && b < 0x80) { // US-ASCII @@ -306,7 +276,7 @@ namespace QUtf8Functions if (!QChar::requiresSurrogates(uc)) { // UTF-8 decoded and no surrogates are required // detach if necessary - Traits::appendUtf16(dst, ushort(uc)); + Traits::appendUtf16(dst, char16_t(uc)); } else { // UTF-8 decoded to something that requires a surrogate pair Traits::appendUcs4(dst, uc);