Add an SSSE3 implementation of qt_memfill24

Brought to you by the PSHUFB instruction, introduced in SSSE3
(implementation also uses PALIGNR just because we can).

The tail functionality makes use of the fact that the low half of
"mval2" ends in the correct content, so we overwrite up to 8 bytes close
to the end.

Change-Id: Iba4b5c183776497d8ee1fffd15646a620829c335
Reviewed-by: Allan Sandfeld Jensen <allan.jensen@qt.io>
bb10
Thiago Macieira 2018-11-05 19:12:21 -08:00
parent 33177b0456
commit 3df79b2953
2 changed files with 76 additions and 1 deletions

View File

@ -6268,8 +6268,17 @@ void qt_memfill64(quint64 *dest, quint64 color, qsizetype count)
}
#endif
#if defined(QT_COMPILER_SUPPORTS_SSSE3) && defined(Q_CC_GNU) && !defined(Q_CC_INTEL) && !defined(Q_CC_CLANG)
__attribute__((optimize("no-tree-vectorize")))
#endif
void qt_memfill24(quint24 *dest, quint24 color, qsizetype count)
{
# ifdef QT_COMPILER_SUPPORTS_SSSE3
extern void qt_memfill24_ssse3(quint24 *, quint24, qsizetype);
if (qCpuHasFeature(SSSE3))
return qt_memfill24_ssse3(dest, color, count);
# endif
const quint32 v = color;
quint24 *end = dest + count;

View File

@ -1,6 +1,7 @@
/****************************************************************************
**
** Copyright (C) 2016 The Qt Company Ltd.
** Copyright (C) 2018 The Qt Company Ltd.
** Copyright (C) 2018 Intel Corporation.
** Contact: https://www.qt.io/licensing/
**
** This file is part of the QtGui module of the Qt Toolkit.
@ -188,6 +189,71 @@ const uint * QT_FASTCALL qt_fetchUntransformed_888_ssse3(uint *buffer, const Ope
return buffer;
}
void qt_memfill24_ssse3(quint24 *dest, quint24 color, qsizetype count)
{
// LCM of 12 and 16 bytes is 48 bytes (16 px)
quint32 v = color;
__m128i m = _mm_cvtsi32_si128(v);
quint24 *end = dest + count;
constexpr uchar x = 2, y = 1, z = 0;
Q_DECL_ALIGN(__m128i) static const uchar
shuffleMask[16 + 1] = { x, y, z, x, y, z, x, y, z, x, y, z, x, y, z, x, y };
__m128i mval1 = _mm_shuffle_epi8(m, _mm_load_si128(reinterpret_cast<const __m128i *>(shuffleMask)));
__m128i mval2 = _mm_shuffle_epi8(m, _mm_loadu_si128(reinterpret_cast<const __m128i *>(shuffleMask + 1)));
__m128i mval3 = _mm_alignr_epi8(mval2, mval1, 2);
for ( ; dest + 16 <= end; dest += 16) {
#ifdef __AVX__
// Store using 32-byte AVX instruction
__m256 mval12 = _mm256_castps128_ps256(_mm_castsi128_ps(mval1));
mval12 = _mm256_insertf128_ps(mval12, _mm_castsi128_ps(mval2), 1);
_mm256_storeu_ps(reinterpret_cast<float *>(dest), mval12);
#else
_mm_storeu_si128(reinterpret_cast<__m128i *>(dest) + 0, mval1);
_mm_storeu_si128(reinterpret_cast<__m128i *>(dest) + 1, mval2);
#endif
_mm_storeu_si128(reinterpret_cast<__m128i *>(dest) + 2, mval3);
}
if (count < 3) {
if (count > 1)
end[-2] = v;
if (count)
end[-1] = v;
return;
}
// less than 16px/48B left
uchar *ptr = reinterpret_cast<uchar *>(dest);
uchar *ptr_end = reinterpret_cast<uchar *>(end);
qptrdiff left = ptr_end - ptr;
if (left >= 24) {
// 8px/24B or more left
_mm_storeu_si128(reinterpret_cast<__m128i *>(ptr) + 0, mval1);
_mm_storel_epi64(reinterpret_cast<__m128i *>(ptr) + 1, mval2);
ptr += 24;
left -= 24;
}
// less than 8px/24B left
if (left >= 16) {
// but more than 5px/15B left
_mm_storeu_si128(reinterpret_cast<__m128i *>(ptr) , mval1);
} else if (left >= 8) {
// but more than 2px/6B left
_mm_storel_epi64(reinterpret_cast<__m128i *>(ptr), mval1);
}
if (left) {
// 1 or 2px left
// store 8 bytes ending with the right values (will overwrite a bit)
_mm_storel_epi64(reinterpret_cast<__m128i *>(ptr_end - 8), mval2);
}
}
QT_END_NAMESPACE
#endif // QT_COMPILER_SUPPORTS_SSSE3