From ea90e44eb10c2a2e4249eadfa548c57637481a34 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 6 Oct 2026 10:14:51 +0000 Subject: [PATCH] -remove SSE4.1 segmentation, Sobel, squared difference, statistic, stretch and texture alignment checks. Always use unaligned SSE4.1 loads and stores, and call LoadNose3, LoadBody3 and LoadTail3 without an align parameter. Co-authored-by: igor.ermolaev --- src/Simd/SimdSse41Segmentation.cpp | 89 +++----- src/Simd/SimdSse41Sobel.cpp | 245 +++++++++------------ src/Simd/SimdSse41SquaredDifferenceSum.cpp | 109 +++------ src/Simd/SimdSse41Statistic.cpp | 119 +++------- src/Simd/SimdSse41StatisticMoments.cpp | 33 +-- src/Simd/SimdSse41StretchGray2x2.cpp | 34 +-- src/Simd/SimdSse41Texture.cpp | 126 ++++------- 7 files changed, 243 insertions(+), 512 deletions(-) diff --git a/src/Simd/SimdSse41Segmentation.cpp b/src/Simd/SimdSse41Segmentation.cpp index 27e77eb282..9bd83f7263 100644 --- a/src/Simd/SimdSse41Segmentation.cpp +++ b/src/Simd/SimdSse41Segmentation.cpp @@ -30,16 +30,17 @@ namespace Simd #ifdef SIMD_SSE41_ENABLE namespace Sse41 { - template SIMD_INLINE void FillSingleHoles(uint8_t* mask, ptrdiff_t stride, __m128i index) + SIMD_INLINE void FillSingleHoles(uint8_t* mask, ptrdiff_t stride, __m128i index) { - const __m128i up = _mm_cmpeq_epi8(Load((__m128i*)(mask - stride)), index); - const __m128i left = _mm_cmpeq_epi8(Load((__m128i*)(mask - 1)), index); - const __m128i right = _mm_cmpeq_epi8(Load((__m128i*)(mask + 1)), index); - const __m128i down = _mm_cmpeq_epi8(Load((__m128i*)(mask + stride)), index); - StoreMasked((__m128i*)mask, index, _mm_and_si128(_mm_and_si128(up, left), _mm_and_si128(right, down))); + const __m128i up = _mm_cmpeq_epi8(_mm_loadu_si128((__m128i*)(mask - stride)), index); + const __m128i left = _mm_cmpeq_epi8(_mm_loadu_si128((__m128i*)(mask - 1)), index); + const __m128i right = _mm_cmpeq_epi8(_mm_loadu_si128((__m128i*)(mask + 1)), index); + const __m128i down = _mm_cmpeq_epi8(_mm_loadu_si128((__m128i*)(mask + stride)), index); + __m128i* pmask = (__m128i*)mask; + _mm_storeu_si128(pmask, _mm_blendv_epi8(_mm_loadu_si128(pmask), index, _mm_and_si128(_mm_and_si128(up, left), _mm_and_si128(right, down)))); } - template void SegmentationFillSingleHoles(uint8_t* mask, size_t stride, size_t width, size_t height, uint8_t index) + void SegmentationFillSingleHoles(uint8_t* mask, size_t stride, size_t width, size_t height, uint8_t index) { assert(width > A + 2 && height > 2); @@ -51,33 +52,25 @@ namespace Simd { mask += stride; - FillSingleHoles(mask + 1, stride, _index); + FillSingleHoles(mask + 1, stride, _index); for (size_t col = A; col < alignedWidth; col += A) - FillSingleHoles(mask + col, stride, _index); + FillSingleHoles(mask + col, stride, _index); if (alignedWidth != width) - FillSingleHoles(mask + width - A, stride, _index); + FillSingleHoles(mask + width - A, stride, _index); } } - void SegmentationFillSingleHoles(uint8_t* mask, size_t stride, size_t width, size_t height, uint8_t index) - { - if (Aligned(mask) && Aligned(stride)) - SegmentationFillSingleHoles(mask, stride, width, height, index); - else - SegmentationFillSingleHoles(mask, stride, width, height, index); - } - //----------------------------------------------------------------------------------------- - template SIMD_INLINE void ChangeIndex(uint8_t* mask, __m128i oldIndex, __m128i newIndex) + SIMD_INLINE void ChangeIndex(uint8_t* mask, __m128i oldIndex, __m128i newIndex) { - __m128i _mask = Load((__m128i*)mask); - Store((__m128i*)mask, Combine(_mm_cmpeq_epi8(_mask, oldIndex), newIndex, _mask)); + __m128i _mask = _mm_loadu_si128((__m128i*)mask); + _mm_storeu_si128((__m128i*)mask, Combine(_mm_cmpeq_epi8(_mask, oldIndex), newIndex, _mask)); } - template void SegmentationChangeIndex(uint8_t* mask, size_t stride, size_t width, size_t height, uint8_t oldIndex, uint8_t newIndex) + void SegmentationChangeIndex(uint8_t* mask, size_t stride, size_t width, size_t height, uint8_t oldIndex, uint8_t newIndex) { __m128i _oldIndex = _mm_set1_epi8((char)oldIndex); __m128i _newIndex = _mm_set1_epi8((char)newIndex); @@ -85,45 +78,37 @@ namespace Simd for (size_t row = 0; row < height; ++row) { for (size_t col = 0; col < alignedWidth; col += A) - ChangeIndex(mask + col, _oldIndex, _newIndex); + ChangeIndex(mask + col, _oldIndex, _newIndex); if (alignedWidth != width) - ChangeIndex(mask + width - A, _oldIndex, _newIndex); + ChangeIndex(mask + width - A, _oldIndex, _newIndex); mask += stride; } } - void SegmentationChangeIndex(uint8_t* mask, size_t stride, size_t width, size_t height, uint8_t oldIndex, uint8_t newIndex) - { - if (Aligned(mask) && Aligned(stride)) - SegmentationChangeIndex(mask, stride, width, height, oldIndex, newIndex); - else - SegmentationChangeIndex(mask, stride, width, height, oldIndex, newIndex); - } - //----------------------------------------------------------------------------------------- SIMD_INLINE void SegmentationPropagate2x2(const __m128i& parentOne, const __m128i& parentAll, const uint8_t* difference0, const uint8_t* difference1, uint8_t* child0, uint8_t* child1, size_t childCol, const __m128i& index, const __m128i& invalid, const __m128i& empty, const __m128i& threshold) { - const __m128i _difference0 = Load((__m128i*)(difference0 + childCol)); - const __m128i _difference1 = Load((__m128i*)(difference1 + childCol)); - const __m128i _child0 = Load((__m128i*)(child0 + childCol)); - const __m128i _child1 = Load((__m128i*)(child1 + childCol)); + const __m128i _difference0 = _mm_loadu_si128((__m128i*)(difference0 + childCol)); + const __m128i _difference1 = _mm_loadu_si128((__m128i*)(difference1 + childCol)); + const __m128i _child0 = _mm_loadu_si128((__m128i*)(child0 + childCol)); + const __m128i _child1 = _mm_loadu_si128((__m128i*)(child1 + childCol)); const __m128i condition0 = _mm_or_si128(parentAll, _mm_and_si128(parentOne, Greater8u(_difference0, threshold))); const __m128i condition1 = _mm_or_si128(parentAll, _mm_and_si128(parentOne, Greater8u(_difference1, threshold))); - Store((__m128i*)(child0 + childCol), Combine(Lesser8u(_child0, invalid), Combine(condition0, index, empty), _child0)); - Store((__m128i*)(child1 + childCol), Combine(Lesser8u(_child1, invalid), Combine(condition1, index, empty), _child1)); + _mm_storeu_si128((__m128i*)(child0 + childCol), Combine(Lesser8u(_child0, invalid), Combine(condition0, index, empty), _child0)); + _mm_storeu_si128((__m128i*)(child1 + childCol), Combine(Lesser8u(_child1, invalid), Combine(condition1, index, empty), _child1)); } - template SIMD_INLINE void SegmentationPropagate2x2(const uint8_t* parent0, const uint8_t* parent1, size_t parentCol, + SIMD_INLINE void SegmentationPropagate2x2(const uint8_t* parent0, const uint8_t* parent1, size_t parentCol, const uint8_t* difference0, const uint8_t* difference1, uint8_t* child0, uint8_t* child1, size_t childCol, const __m128i& index, const __m128i& invalid, const __m128i& empty, const __m128i& threshold) { - const __m128i parent00 = _mm_cmpeq_epi8(Load((__m128i*)(parent0 + parentCol)), index); - const __m128i parent01 = _mm_cmpeq_epi8(Load((__m128i*)(parent0 + parentCol + 1)), index); - const __m128i parent10 = _mm_cmpeq_epi8(Load((__m128i*)(parent1 + parentCol)), index); - const __m128i parent11 = _mm_cmpeq_epi8(Load((__m128i*)(parent1 + parentCol + 1)), index); + const __m128i parent00 = _mm_cmpeq_epi8(_mm_loadu_si128((__m128i*)(parent0 + parentCol)), index); + const __m128i parent01 = _mm_cmpeq_epi8(_mm_loadu_si128((__m128i*)(parent0 + parentCol + 1)), index); + const __m128i parent10 = _mm_cmpeq_epi8(_mm_loadu_si128((__m128i*)(parent1 + parentCol)), index); + const __m128i parent11 = _mm_cmpeq_epi8(_mm_loadu_si128((__m128i*)(parent1 + parentCol + 1)), index); const __m128i parentOne = _mm_or_si128(_mm_or_si128(parent00, parent01), _mm_or_si128(parent10, parent11)); const __m128i parentAll = _mm_and_si128(_mm_and_si128(parent00, parent01), _mm_and_si128(parent10, parent11)); @@ -134,7 +119,7 @@ namespace Simd difference0, difference1, child0, child1, childCol + A, index, invalid, empty, threshold); } - template void SegmentationPropagate2x2(const uint8_t* parent, size_t parentStride, size_t width, size_t height, + void SegmentationPropagate2x2(const uint8_t* parent, size_t parentStride, size_t width, size_t height, uint8_t* child, size_t childStride, const uint8_t* difference, size_t differenceStride, uint8_t currentIndex, uint8_t invalidIndex, uint8_t emptyIndex, uint8_t differenceThreshold) { @@ -158,26 +143,14 @@ namespace Simd uint8_t* child1 = child0 + childStride; for (size_t parentCol = 0, childCol = 1; parentCol < alignedWidth; parentCol += A, childCol += DA) - SegmentationPropagate2x2(parent0, parent1, parentCol, difference0, difference1, + SegmentationPropagate2x2(parent0, parent1, parentCol, difference0, difference1, child0, child1, childCol, index, invalid, empty, threshold); if (alignedWidth != width) - SegmentationPropagate2x2(parent0, parent1, width - A, difference0, difference1, + SegmentationPropagate2x2(parent0, parent1, width - A, difference0, difference1, child0, child1, (width - A) * 2 + 1, index, invalid, empty, threshold); } } - void SegmentationPropagate2x2(const uint8_t* parent, size_t parentStride, size_t width, size_t height, - uint8_t* child, size_t childStride, const uint8_t* difference, size_t differenceStride, - uint8_t currentIndex, uint8_t invalidIndex, uint8_t emptyIndex, uint8_t differenceThreshold) - { - if (Aligned(parent) && Aligned(parentStride)) - SegmentationPropagate2x2(parent, parentStride, width, height, child, childStride, - difference, differenceStride, currentIndex, invalidIndex, emptyIndex, differenceThreshold); - else - SegmentationPropagate2x2(parent, parentStride, width, height, child, childStride, - difference, differenceStride, currentIndex, invalidIndex, emptyIndex, differenceThreshold); - } - //----------------------------------------------------------------------------------------- SIMD_INLINE bool RowHasIndex(const uint8_t * mask, size_t alignedSize, size_t fullSize, __m128i index) diff --git a/src/Simd/SimdSse41Sobel.cpp b/src/Simd/SimdSse41Sobel.cpp index 1137b84018..c0b0bed35f 100644 --- a/src/Simd/SimdSse41Sobel.cpp +++ b/src/Simd/SimdSse41Sobel.cpp @@ -39,19 +39,17 @@ namespace Simd hi = ConditionalAbs(BinomialSum16(SubUnpackedU8<1>(a[0][2], a[0][0]), SubUnpackedU8<1>(a[1][2], a[1][0]), SubUnpackedU8<1>(a[2][2], a[2][0]))); } - template SIMD_INLINE void SobelDx(__m128i a[3][3], int16_t * dst) + template SIMD_INLINE void SobelDx(__m128i a[3][3], int16_t * dst) { __m128i lo, hi; SobelDx(a, lo, hi); - Store((__m128i*)dst + 0, lo); - Store((__m128i*)dst + 1, hi); + _mm_storeu_si128((__m128i*)dst + 0, lo); + _mm_storeu_si128((__m128i*)dst + 1, hi); } - template void SobelDx(const uint8_t * src, size_t srcStride, size_t width, size_t height, int16_t * dst, size_t dstStride) + template void SobelDx(const uint8_t * src, size_t srcStride, size_t width, size_t height, int16_t * dst, size_t dstStride) { assert(width > A); - if (align) - assert(Aligned(dst) && Aligned(dstStride, HA)); size_t bodyWidth = Simd::AlignHi(width, A) - A; const uint8_t *src0, *src1, *src2; @@ -70,18 +68,18 @@ namespace Simd LoadNoseDx(src0 + 0, a[0]); LoadNoseDx(src1 + 0, a[1]); LoadNoseDx(src2 + 0, a[2]); - SobelDx(a, dst + 0); + SobelDx(a, dst + 0); for (size_t col = A; col < bodyWidth; col += A) { LoadBodyDx(src0 + col, a[0]); LoadBodyDx(src1 + col, a[1]); LoadBodyDx(src2 + col, a[2]); - SobelDx(a, dst + col); + SobelDx(a, dst + col); } LoadTailDx(src0 + width - A, a[0]); LoadTailDx(src1 + width - A, a[1]); LoadTailDx(src2 + width - A, a[2]); - SobelDx(a, dst + width - A); + SobelDx(a, dst + width - A); dst += dstStride; } @@ -91,10 +89,7 @@ namespace Simd { assert(dstStride % sizeof(int16_t) == 0); - if (Aligned(dst) && Aligned(dstStride)) - SobelDx(src, srcStride, width, height, (int16_t *)dst, dstStride / sizeof(int16_t)); - else - SobelDx(src, srcStride, width, height, (int16_t *)dst, dstStride / sizeof(int16_t)); + SobelDx(src, srcStride, width, height, (int16_t *)dst, dstStride / sizeof(int16_t)); } //----------------------------------------------------------------------------------------- @@ -103,10 +98,7 @@ namespace Simd { assert(dstStride % sizeof(int16_t) == 0); - if (Aligned(dst) && Aligned(dstStride)) - SobelDx(src, srcStride, width, height, (int16_t *)dst, dstStride / sizeof(int16_t)); - else - SobelDx(src, srcStride, width, height, (int16_t *)dst, dstStride / sizeof(int16_t)); + SobelDx(src, srcStride, width, height, (int16_t *)dst, dstStride / sizeof(int16_t)); } //----------------------------------------------------------------------------------------- @@ -186,19 +178,17 @@ namespace Simd hi = ConditionalAbs(BinomialSum16(SubUnpackedU8<1>(a[2][0], a[0][0]), SubUnpackedU8<1>(a[2][1], a[0][1]), SubUnpackedU8<1>(a[2][2], a[0][2]))); } - template SIMD_INLINE void SobelDy(__m128i a[3][3], int16_t * dst) + template SIMD_INLINE void SobelDy(__m128i a[3][3], int16_t * dst) { __m128i lo, hi; SobelDy(a, lo, hi); - Store((__m128i*)dst + 0, lo); - Store((__m128i*)dst + 1, hi); + _mm_storeu_si128((__m128i*)dst + 0, lo); + _mm_storeu_si128((__m128i*)dst + 1, hi); } - template void SobelDy(const uint8_t * src, size_t srcStride, size_t width, size_t height, int16_t * dst, size_t dstStride) + template void SobelDy(const uint8_t * src, size_t srcStride, size_t width, size_t height, int16_t * dst, size_t dstStride) { assert(width > A); - if (align) - assert(Aligned(dst) && Aligned(dstStride, HA)); size_t bodyWidth = Simd::AlignHi(width, A) - A; const uint8_t *src0, *src1, *src2; @@ -214,18 +204,18 @@ namespace Simd if (row == height - 1) src2 = src1; - LoadNose3(src0 + 0, a[0]); - LoadNose3(src2 + 0, a[2]); - SobelDy(a, dst + 0); + LoadNose3<1>(src0 + 0, a[0]); + LoadNose3<1>(src2 + 0, a[2]); + SobelDy(a, dst + 0); for (size_t col = A; col < bodyWidth; col += A) { - LoadBody3(src0 + col, a[0]); - LoadBody3(src2 + col, a[2]); - SobelDy(a, dst + col); + LoadBody3<1>(src0 + col, a[0]); + LoadBody3<1>(src2 + col, a[2]); + SobelDy(a, dst + col); } - LoadTail3(src0 + width - A, a[0]); - LoadTail3(src2 + width - A, a[2]); - SobelDy(a, dst + width - A); + LoadTail3<1>(src0 + width - A, a[0]); + LoadTail3<1>(src2 + width - A, a[2]); + SobelDy(a, dst + width - A); dst += dstStride; } @@ -235,10 +225,7 @@ namespace Simd { assert(dstStride % sizeof(int16_t) == 0); - if (Aligned(src) && Aligned(srcStride) && Aligned(dst) && Aligned(dstStride)) - SobelDy(src, srcStride, width, height, (int16_t *)dst, dstStride / sizeof(int16_t)); - else - SobelDy(src, srcStride, width, height, (int16_t *)dst, dstStride / sizeof(int16_t)); + SobelDy(src, srcStride, width, height, (int16_t *)dst, dstStride / sizeof(int16_t)); } //----------------------------------------------------------------------------------------- @@ -247,10 +234,7 @@ namespace Simd { assert(dstStride % sizeof(int16_t) == 0); - if (Aligned(src) && Aligned(srcStride) && Aligned(dst) && Aligned(dstStride)) - SobelDy(src, srcStride, width, height, (int16_t *)dst, dstStride / sizeof(int16_t)); - else - SobelDy(src, srcStride, width, height, (int16_t *)dst, dstStride / sizeof(int16_t)); + SobelDy(src, srcStride, width, height, (int16_t *)dst, dstStride / sizeof(int16_t)); } //----------------------------------------------------------------------------------------- @@ -263,7 +247,7 @@ namespace Simd sum = _mm_add_epi32(sum, _mm_madd_epi16(hi, K16_0001)); } - template void SobelDyAbsSum(const uint8_t * src, size_t stride, size_t width, size_t height, uint64_t * sum) + void SobelDyAbsSum(const uint8_t * src, size_t stride, size_t width, size_t height, uint64_t * sum) { assert(width > A); @@ -286,17 +270,17 @@ namespace Simd __m128i rowSum = _mm_setzero_si128(); - LoadNose3(src0 + 0, a[0]); - LoadNose3(src2 + 0, a[2]); + LoadNose3<1>(src0 + 0, a[0]); + LoadNose3<1>(src2 + 0, a[2]); SobelDyAbsSum(a, rowSum); for (size_t col = A; col < bodyWidth; col += A) { - LoadBody3(src0 + col, a[0]); - LoadBody3(src2 + col, a[2]); + LoadBody3<1>(src0 + col, a[0]); + LoadBody3<1>(src2 + col, a[2]); SobelDyAbsSum(a, rowSum); } - LoadTail3(src0 + width - A, a[0]); - LoadTail3(src2 + width - A, a[2]); + LoadTail3<1>(src0 + width - A, a[0]); + LoadTail3<1>(src2 + width - A, a[2]); SetMask3x3(a, tailMask); SobelDyAbsSum(a, rowSum); @@ -305,78 +289,62 @@ namespace Simd *sum = ExtractInt64Sum(fullSum); } - void SobelDyAbsSum(const uint8_t * src, size_t stride, size_t width, size_t height, uint64_t * sum) - { - if (Aligned(src) && Aligned(stride)) - SobelDyAbsSum(src, stride, width, height, sum); - else - SobelDyAbsSum(src, stride, width, height, sum); - } - //----------------------------------------------------------------------------------------- - template SIMD_INLINE __m128i AnchorComponent(const int16_t* src, size_t step, const __m128i& current, const __m128i& threshold, const __m128i& mask) + SIMD_INLINE __m128i AnchorComponent(const int16_t* src, size_t step, const __m128i& current, const __m128i& threshold, const __m128i& mask) { - __m128i last = _mm_srli_epi16(Load((__m128i*)(src - step)), 1); - __m128i next = _mm_srli_epi16(Load((__m128i*)(src + step)), 1); + __m128i last = _mm_srli_epi16(_mm_loadu_si128((__m128i*)(src - step)), 1); + __m128i next = _mm_srli_epi16(_mm_loadu_si128((__m128i*)(src + step)), 1); return _mm_andnot_si128(_mm_or_si128(_mm_cmplt_epi16(_mm_sub_epi16(current, last), threshold), _mm_cmplt_epi16(_mm_sub_epi16(current, next), threshold)), mask); } - template SIMD_INLINE __m128i Anchor(const int16_t* src, size_t stride, const __m128i& threshold) + SIMD_INLINE __m128i Anchor(const int16_t* src, size_t stride, const __m128i& threshold) { - __m128i _src = Load((__m128i*)src); + __m128i _src = _mm_loadu_si128((__m128i*)src); __m128i direction = _mm_and_si128(_src, K16_0001); __m128i magnitude = _mm_srli_epi16(_src, 1); - __m128i vertical = AnchorComponent(src, 1, magnitude, threshold, _mm_cmpeq_epi16(direction, K16_0001)); - __m128i horizontal = AnchorComponent(src, stride, magnitude, threshold, _mm_cmpeq_epi16(direction, K_ZERO)); + __m128i vertical = AnchorComponent(src, 1, magnitude, threshold, _mm_cmpeq_epi16(direction, K16_0001)); + __m128i horizontal = AnchorComponent(src, stride, magnitude, threshold, _mm_cmpeq_epi16(direction, K_ZERO)); return _mm_andnot_si128(_mm_cmpeq_epi16(magnitude, K_ZERO), _mm_and_si128(_mm_or_si128(vertical, horizontal), K16_00FF)); } - template SIMD_INLINE void Anchor(const int16_t* src, size_t stride, const __m128i& threshold, uint8_t* dst) + SIMD_INLINE void Anchor(const int16_t* src, size_t stride, const __m128i& threshold, uint8_t* dst) { - __m128i lo = Anchor(src, stride, threshold); - __m128i hi = Anchor(src + HA, stride, threshold); - Store((__m128i*)dst, _mm_packus_epi16(lo, hi)); + __m128i lo = Anchor(src, stride, threshold); + __m128i hi = Anchor(src + HA, stride, threshold); + _mm_storeu_si128((__m128i*)dst, _mm_packus_epi16(lo, hi)); } - template void ContourAnchors(const int16_t* src, size_t srcStride, size_t width, size_t height, + void ContourAnchors(const uint8_t* src, size_t srcStride, size_t width, size_t height, size_t step, int16_t threshold, uint8_t* dst, size_t dstStride) { + assert(srcStride % sizeof(int16_t) == 0); + + const int16_t* src16 = (const int16_t*)src; + size_t srcStride16 = srcStride / sizeof(int16_t); + assert(width > A); - if (align) - assert(Aligned(src) && Aligned(srcStride, HA) && Aligned(dst) && Aligned(dstStride)); size_t bodyWidth = Simd::AlignHi(width, A) - A; __m128i _threshold = _mm_set1_epi16(threshold); memset(dst, 0, width); memset(dst + dstStride * (height - 1), 0, width); - src += srcStride; + src16 += srcStride16; dst += dstStride; for (size_t row = 1; row < height - 1; row += step) { dst[0] = 0; - Anchor(src + 1, srcStride, _threshold, dst + 1); + Anchor(src16 + 1, srcStride16, _threshold, dst + 1); for (size_t col = A; col < bodyWidth; col += A) - Anchor(src + col, srcStride, _threshold, dst + col); - Anchor(src + width - A - 1, srcStride, _threshold, dst + width - A - 1); + Anchor(src16 + col, srcStride16, _threshold, dst + col); + Anchor(src16 + width - A - 1, srcStride16, _threshold, dst + width - A - 1); dst[width - 1] = 0; - src += step * srcStride; + src16 += step * srcStride16; dst += step * dstStride; } } - void ContourAnchors(const uint8_t* src, size_t srcStride, size_t width, size_t height, - size_t step, int16_t threshold, uint8_t* dst, size_t dstStride) - { - assert(srcStride % sizeof(int16_t) == 0); - - if (Aligned(src) && Aligned(srcStride) && Aligned(dst) && Aligned(dstStride)) - ContourAnchors((const int16_t*)src, srcStride / sizeof(int16_t), width, height, step, threshold, dst, dstStride); - else - ContourAnchors((const int16_t*)src, srcStride / sizeof(int16_t), width, height, step, threshold, dst, dstStride); - } - //----------------------------------------------------------------------------------------- SIMD_INLINE __m128i ContourMetrics(__m128i dx, __m128i dy) @@ -393,19 +361,22 @@ namespace Simd hi = ContourMetrics(dxHi, dyHi); } - template SIMD_INLINE void ContourMetrics(__m128i a[3][3], int16_t * dst) + SIMD_INLINE void ContourMetrics(__m128i a[3][3], int16_t * dst) { __m128i lo, hi; ContourMetrics(a, lo, hi); - Store((__m128i*)dst + 0, lo); - Store((__m128i*)dst + 1, hi); + _mm_storeu_si128((__m128i*)dst + 0, lo); + _mm_storeu_si128((__m128i*)dst + 1, hi); } - template void ContourMetrics(const uint8_t * src, size_t srcStride, size_t width, size_t height, int16_t * dst, size_t dstStride) + void ContourMetrics(const uint8_t * src, size_t srcStride, size_t width, size_t height, uint8_t * dst, size_t dstStride) { + assert(dstStride % sizeof(int16_t) == 0); + + int16_t * dst16 = (int16_t *)dst; + size_t dstStride16 = dstStride / sizeof(int16_t); + assert(width > A); - if (align) - assert(Aligned(src) && Aligned(srcStride) && Aligned(dst) && Aligned(dstStride, HA)); size_t bodyWidth = Simd::AlignHi(width, A) - A; const uint8_t *src0, *src1, *src2; @@ -421,53 +392,46 @@ namespace Simd if (row == height - 1) src2 = src1; - LoadNose3(src0 + 0, a[0]); - LoadNose3(src1 + 0, a[1]); - LoadNose3(src2 + 0, a[2]); - ContourMetrics(a, dst + 0); + LoadNose3<1>(src0 + 0, a[0]); + LoadNose3<1>(src1 + 0, a[1]); + LoadNose3<1>(src2 + 0, a[2]); + ContourMetrics(a, dst16 + 0); for (size_t col = A; col < bodyWidth; col += A) { - LoadBody3(src0 + col, a[0]); - LoadBody3(src1 + col, a[1]); - LoadBody3(src2 + col, a[2]); - ContourMetrics(a, dst + col); + LoadBody3<1>(src0 + col, a[0]); + LoadBody3<1>(src1 + col, a[1]); + LoadBody3<1>(src2 + col, a[2]); + ContourMetrics(a, dst16 + col); } - LoadTail3(src0 + width - A, a[0]); - LoadTail3(src1 + width - A, a[1]); - LoadTail3(src2 + width - A, a[2]); - ContourMetrics(a, dst + width - A); + LoadTail3<1>(src0 + width - A, a[0]); + LoadTail3<1>(src1 + width - A, a[1]); + LoadTail3<1>(src2 + width - A, a[2]); + ContourMetrics(a, dst16 + width - A); - dst += dstStride; + dst16 += dstStride16; } } - void ContourMetrics(const uint8_t * src, size_t srcStride, size_t width, size_t height, uint8_t * dst, size_t dstStride) - { - assert(dstStride % sizeof(int16_t) == 0); - - if (Aligned(src) && Aligned(srcStride) && Aligned(dst) && Aligned(dstStride)) - ContourMetrics(src, srcStride, width, height, (int16_t *)dst, dstStride / sizeof(int16_t)); - else - ContourMetrics(src, srcStride, width, height, (int16_t *)dst, dstStride / sizeof(int16_t)); - } - //----------------------------------------------------------------------------------------- - template SIMD_INLINE void ContourMetricsMasked(__m128i a[3][3], const uint8_t * mask, const __m128i & indexMin, int16_t * dst) + SIMD_INLINE void ContourMetricsMasked(__m128i a[3][3], const uint8_t * mask, const __m128i & indexMin, int16_t * dst) { - __m128i m = GreaterOrEqual8u(Load((__m128i*)mask), indexMin); + __m128i m = GreaterOrEqual8u(_mm_loadu_si128((__m128i*)mask), indexMin); __m128i lo, hi; ContourMetrics(a, lo, hi); - Store((__m128i*)dst + 0, _mm_and_si128(lo, _mm_unpacklo_epi8(m, m))); - Store((__m128i*)dst + 1, _mm_and_si128(hi, _mm_unpackhi_epi8(m, m))); + _mm_storeu_si128((__m128i*)dst + 0, _mm_and_si128(lo, _mm_unpacklo_epi8(m, m))); + _mm_storeu_si128((__m128i*)dst + 1, _mm_and_si128(hi, _mm_unpackhi_epi8(m, m))); } - template void ContourMetricsMasked(const uint8_t * src, size_t srcStride, size_t width, size_t height, - const uint8_t * mask, size_t maskStride, uint8_t indexMin, int16_t * dst, size_t dstStride) + void ContourMetricsMasked(const uint8_t * src, size_t srcStride, size_t width, size_t height, + const uint8_t * mask, size_t maskStride, uint8_t indexMin, uint8_t * dst, size_t dstStride) { + assert(dstStride % sizeof(int16_t) == 0); + + int16_t * dst16 = (int16_t *)dst; + size_t dstStride16 = dstStride / sizeof(int16_t); + assert(width > A); - if (align) - assert(Aligned(src) && Aligned(srcStride) && Aligned(dst) && Aligned(dstStride, HA) && Aligned(mask) && Aligned(maskStride)); size_t bodyWidth = Simd::AlignHi(width, A) - A; const uint8_t *src0, *src1, *src2; @@ -484,37 +448,26 @@ namespace Simd if (row == height - 1) src2 = src1; - LoadNose3(src0 + 0, a[0]); - LoadNose3(src1 + 0, a[1]); - LoadNose3(src2 + 0, a[2]); - ContourMetricsMasked(a, mask + 0, _indexMin, dst + 0); + LoadNose3<1>(src0 + 0, a[0]); + LoadNose3<1>(src1 + 0, a[1]); + LoadNose3<1>(src2 + 0, a[2]); + ContourMetricsMasked(a, mask + 0, _indexMin, dst16 + 0); for (size_t col = A; col < bodyWidth; col += A) { - LoadBody3(src0 + col, a[0]); - LoadBody3(src1 + col, a[1]); - LoadBody3(src2 + col, a[2]); - ContourMetricsMasked(a, mask + col, _indexMin, dst + col); + LoadBody3<1>(src0 + col, a[0]); + LoadBody3<1>(src1 + col, a[1]); + LoadBody3<1>(src2 + col, a[2]); + ContourMetricsMasked(a, mask + col, _indexMin, dst16 + col); } - LoadTail3(src0 + width - A, a[0]); - LoadTail3(src1 + width - A, a[1]); - LoadTail3(src2 + width - A, a[2]); - ContourMetricsMasked(a, mask + width - A, _indexMin, dst + width - A); + LoadTail3<1>(src0 + width - A, a[0]); + LoadTail3<1>(src1 + width - A, a[1]); + LoadTail3<1>(src2 + width - A, a[2]); + ContourMetricsMasked(a, mask + width - A, _indexMin, dst16 + width - A); - dst += dstStride; + dst16 += dstStride16; mask += maskStride; } } - - void ContourMetricsMasked(const uint8_t * src, size_t srcStride, size_t width, size_t height, - const uint8_t * mask, size_t maskStride, uint8_t indexMin, uint8_t * dst, size_t dstStride) - { - assert(dstStride % sizeof(int16_t) == 0); - - if (Aligned(src) && Aligned(srcStride) && Aligned(dst) && Aligned(dstStride) && Aligned(mask) && Aligned(maskStride)) - ContourMetricsMasked(src, srcStride, width, height, mask, maskStride, indexMin, (int16_t *)dst, dstStride / sizeof(int16_t)); - else - ContourMetricsMasked(src, srcStride, width, height, mask, maskStride, indexMin, (int16_t *)dst, dstStride / sizeof(int16_t)); - } } #endif } diff --git a/src/Simd/SimdSse41SquaredDifferenceSum.cpp b/src/Simd/SimdSse41SquaredDifferenceSum.cpp index 63b144a556..6cd7080b87 100644 --- a/src/Simd/SimdSse41SquaredDifferenceSum.cpp +++ b/src/Simd/SimdSse41SquaredDifferenceSum.cpp @@ -38,15 +38,11 @@ namespace Simd return _mm_add_epi32(_mm_madd_epi16(lo, lo), _mm_madd_epi16(hi, hi)); } - template void SquaredDifferenceSum( + void SquaredDifferenceSum( const uint8_t *a, size_t aStride, const uint8_t *b, size_t bStride, size_t width, size_t height, uint64_t * sum) { assert(width < 0x10000); - if (align) - { - assert(Aligned(a) && Aligned(aStride) && Aligned(b) && Aligned(bStride)); - } size_t bodyWidth = AlignLo(width, A); __m128i tailMask = ShiftLeft(K_INV_ZERO, A - width + bodyWidth); @@ -56,14 +52,14 @@ namespace Simd __m128i rowSum = _mm_setzero_si128(); for (size_t col = 0; col < bodyWidth; col += A) { - const __m128i a_ = Load((__m128i*)(a + col)); - const __m128i b_ = Load((__m128i*)(b + col)); + const __m128i a_ = _mm_loadu_si128((__m128i*)(a + col)); + const __m128i b_ = _mm_loadu_si128((__m128i*)(b + col)); rowSum = _mm_add_epi32(rowSum, SquaredDifference(a_, b_)); } if (width - bodyWidth) { - const __m128i a_ = _mm_and_si128(tailMask, Load((__m128i*)(a + width - A))); - const __m128i b_ = _mm_and_si128(tailMask, Load((__m128i*)(b + width - A))); + const __m128i a_ = _mm_and_si128(tailMask, _mm_loadu_si128((__m128i*)(a + width - A))); + const __m128i b_ = _mm_and_si128(tailMask, _mm_loadu_si128((__m128i*)(b + width - A))); rowSum = _mm_add_epi32(rowSum, SquaredDifference(a_, b_)); } fullSum = _mm_add_epi64(fullSum, HorizontalSum32(rowSum)); @@ -73,27 +69,13 @@ namespace Simd *sum = ExtractInt64Sum(fullSum); } - void SquaredDifferenceSum(const uint8_t *a, size_t aStride, const uint8_t *b, size_t bStride, - size_t width, size_t height, uint64_t * sum) - { - if (Aligned(a) && Aligned(aStride) && Aligned(b) && Aligned(bStride)) - SquaredDifferenceSum(a, aStride, b, bStride, width, height, sum); - else - SquaredDifferenceSum(a, aStride, b, bStride, width, height, sum); - } - //----------------------------------------------------------------------------------------- - template void SquaredDifferenceSumMasked( + void SquaredDifferenceSumMasked( const uint8_t *a, size_t aStride, const uint8_t *b, size_t bStride, const uint8_t *mask, size_t maskStride, uint8_t index, size_t width, size_t height, uint64_t * sum) { assert(width < 0x10000); - if (align) - { - assert(Aligned(a) && Aligned(aStride) && Aligned(b) && Aligned(bStride)); - assert(Aligned(mask) && Aligned(maskStride)); - } size_t bodyWidth = AlignLo(width, A); __m128i tailMask = ShiftLeft(K_INV_ZERO, A - width + bodyWidth); @@ -104,16 +86,16 @@ namespace Simd __m128i rowSum = _mm_setzero_si128(); for (size_t col = 0; col < bodyWidth; col += A) { - const __m128i mask_ = LoadMaskI8((__m128i*)(mask + col), index_); - const __m128i a_ = _mm_and_si128(mask_, Load((__m128i*)(a + col))); - const __m128i b_ = _mm_and_si128(mask_, Load((__m128i*)(b + col))); + const __m128i mask_ = _mm_cmpeq_epi8(_mm_loadu_si128((__m128i*)(mask + col)), index_); + const __m128i a_ = _mm_and_si128(mask_, _mm_loadu_si128((__m128i*)(a + col))); + const __m128i b_ = _mm_and_si128(mask_, _mm_loadu_si128((__m128i*)(b + col))); rowSum = _mm_add_epi32(rowSum, SquaredDifference(a_, b_)); } if (width - bodyWidth) { - const __m128i mask_ = _mm_and_si128(tailMask, LoadMaskI8((__m128i*)(mask + width - A), index_)); - const __m128i a_ = _mm_and_si128(mask_, Load((__m128i*)(a + width - A))); - const __m128i b_ = _mm_and_si128(mask_, Load((__m128i*)(b + width - A))); + const __m128i mask_ = _mm_and_si128(tailMask, _mm_cmpeq_epi8(_mm_loadu_si128((__m128i*)(mask + width - A)), index_)); + const __m128i a_ = _mm_and_si128(mask_, _mm_loadu_si128((__m128i*)(a + width - A))); + const __m128i b_ = _mm_and_si128(mask_, _mm_loadu_si128((__m128i*)(b + width - A))); rowSum = _mm_add_epi32(rowSum, SquaredDifference(a_, b_)); } fullSum = _mm_add_epi64(fullSum, HorizontalSum32(rowSum)); @@ -122,32 +104,20 @@ namespace Simd mask += maskStride; } *sum = ExtractInt64Sum(fullSum); - } - - void SquaredDifferenceSumMasked(const uint8_t *a, size_t aStride, const uint8_t *b, size_t bStride, - const uint8_t *mask, size_t maskStride, uint8_t index, size_t width, size_t height, uint64_t * sum) - { - if (Aligned(a) && Aligned(aStride) && Aligned(b) && Aligned(bStride) && Aligned(mask) && Aligned(maskStride)) - SquaredDifferenceSumMasked(a, aStride, b, bStride, mask, maskStride, index, width, height, sum); - else - SquaredDifferenceSumMasked(a, aStride, b, bStride, mask, maskStride, index, width, height, sum); } //----------------------------------------------------------------------------------------- - template SIMD_INLINE void SquaredDifferenceSum32f(const float* a, const float* b, size_t offset, __m128& sum) + SIMD_INLINE void SquaredDifferenceSum32f(const float* a, const float* b, size_t offset, __m128& sum) { - __m128 _a = Load(a + offset); - __m128 _b = Load(b + offset); + __m128 _a = _mm_loadu_ps(a + offset); + __m128 _b = _mm_loadu_ps(b + offset); __m128 _d = _mm_sub_ps(_a, _b); sum = _mm_add_ps(sum, _mm_mul_ps(_d, _d)); } - template SIMD_INLINE void SquaredDifferenceSum32f(const float* a, const float* b, size_t size, float* sum) + void SquaredDifferenceSum32f(const float* a, const float* b, size_t size, float* sum) { - if (align) - assert(Aligned(a) && Aligned(b)); - *sum = 0; size_t partialAlignedSize = AlignLo(size, 4); size_t fullAlignedSize = AlignLo(size, 16); @@ -159,35 +129,27 @@ namespace Simd { for (; i < fullAlignedSize; i += 16) { - SquaredDifferenceSum32f(a, b, i, sums[0]); - SquaredDifferenceSum32f(a, b, i + 4, sums[1]); - SquaredDifferenceSum32f(a, b, i + 8, sums[2]); - SquaredDifferenceSum32f(a, b, i + 12, sums[3]); + SquaredDifferenceSum32f(a, b, i, sums[0]); + SquaredDifferenceSum32f(a, b, i + 4, sums[1]); + SquaredDifferenceSum32f(a, b, i + 8, sums[2]); + SquaredDifferenceSum32f(a, b, i + 12, sums[3]); } sums[0] = _mm_add_ps(_mm_add_ps(sums[0], sums[1]), _mm_add_ps(sums[2], sums[3])); } for (; i < partialAlignedSize; i += 4) - SquaredDifferenceSum32f(a, b, i, sums[0]); + SquaredDifferenceSum32f(a, b, i, sums[0]); *sum += ExtractSum(sums[0]); } for (; i < size; ++i) *sum += Simd::Square(a[i] - b[i]); } - void SquaredDifferenceSum32f(const float* a, const float* b, size_t size, float* sum) - { - if (Aligned(a) && Aligned(b)) - SquaredDifferenceSum32f(a, b, size, sum); - else - SquaredDifferenceSum32f(a, b, size, sum); - } - //----------------------------------------------------------------------------------------- - template SIMD_INLINE void SquaredDifferenceKahanSum32f(const float* a, const float* b, size_t offset, __m128& sum, __m128& correction) + SIMD_INLINE void SquaredDifferenceKahanSum32f(const float* a, const float* b, size_t offset, __m128& sum, __m128& correction) { - __m128 _a = Load(a + offset); - __m128 _b = Load(b + offset); + __m128 _a = _mm_loadu_ps(a + offset); + __m128 _b = _mm_loadu_ps(b + offset); __m128 _d = _mm_sub_ps(_a, _b); __m128 term = _mm_sub_ps(_mm_mul_ps(_d, _d), correction); __m128 temp = _mm_add_ps(sum, term); @@ -195,11 +157,8 @@ namespace Simd sum = temp; } - template SIMD_INLINE void SquaredDifferenceKahanSum32f(const float* a, const float* b, size_t size, float* sum) + void SquaredDifferenceKahanSum32f(const float* a, const float* b, size_t size, float* sum) { - if (align) - assert(Aligned(a) && Aligned(b)); - *sum = 0; size_t partialAlignedSize = AlignLo(size, 4); size_t fullAlignedSize = AlignLo(size, 16); @@ -212,27 +171,19 @@ namespace Simd { for (; i < fullAlignedSize; i += 16) { - SquaredDifferenceKahanSum32f(a, b, i, sums[0], corrections[0]); - SquaredDifferenceKahanSum32f(a, b, i + 4, sums[1], corrections[1]); - SquaredDifferenceKahanSum32f(a, b, i + 8, sums[2], corrections[2]); - SquaredDifferenceKahanSum32f(a, b, i + 12, sums[3], corrections[3]); + SquaredDifferenceKahanSum32f(a, b, i, sums[0], corrections[0]); + SquaredDifferenceKahanSum32f(a, b, i + 4, sums[1], corrections[1]); + SquaredDifferenceKahanSum32f(a, b, i + 8, sums[2], corrections[2]); + SquaredDifferenceKahanSum32f(a, b, i + 12, sums[3], corrections[3]); } } for (; i < partialAlignedSize; i += 4) - SquaredDifferenceKahanSum32f(a, b, i, sums[0], corrections[0]); + SquaredDifferenceKahanSum32f(a, b, i, sums[0], corrections[0]); *sum += ExtractSum(_mm_add_ps(_mm_add_ps(sums[0], sums[1]), _mm_add_ps(sums[2], sums[3]))); } for (; i < size; ++i) *sum += Simd::Square(a[i] - b[i]); } - - void SquaredDifferenceKahanSum32f(const float* a, const float* b, size_t size, float* sum) - { - if (Aligned(a) && Aligned(b)) - SquaredDifferenceKahanSum32f(a, b, size, sum); - else - SquaredDifferenceKahanSum32f(a, b, size, sum); - } } #endif } diff --git a/src/Simd/SimdSse41Statistic.cpp b/src/Simd/SimdSse41Statistic.cpp index a405cd0dfb..a7901c3bdf 100644 --- a/src/Simd/SimdSse41Statistic.cpp +++ b/src/Simd/SimdSse41Statistic.cpp @@ -36,12 +36,10 @@ namespace Simd #ifdef SIMD_SSE41_ENABLE namespace Sse41 { - template void GetStatistic(const uint8_t* src, size_t stride, size_t width, size_t height, + void GetStatistic(const uint8_t* src, size_t stride, size_t width, size_t height, uint8_t* min, uint8_t* max, uint8_t* average) { assert(width * height && width >= A); - if (align) - assert(Aligned(src) && Aligned(stride)); size_t bodyWidth = AlignLo(width, A); __m128i tailMask = ShiftLeft(K_INV_ZERO, A - width + bodyWidth); @@ -52,14 +50,14 @@ namespace Simd { for (size_t col = 0; col < bodyWidth; col += A) { - const __m128i value = Load((__m128i*)(src + col)); + const __m128i value = _mm_loadu_si128((__m128i*)(src + col)); min_ = _mm_min_epu8(min_, value); max_ = _mm_max_epu8(max_, value); sum = _mm_add_epi64(_mm_sad_epu8(value, K_ZERO), sum); } if (width - bodyWidth) { - const __m128i value = Load((__m128i*)(src + width - A)); + const __m128i value = _mm_loadu_si128((__m128i*)(src + width - A)); min_ = _mm_min_epu8(min_, value); max_ = _mm_max_epu8(max_, value); sum = _mm_add_epi64(_mm_sad_epu8(_mm_and_si128(tailMask, value), K_ZERO), sum); @@ -80,18 +78,9 @@ namespace Simd *average = (uint8_t)((ExtractInt64Sum(sum) + width * height / 2) / (width * height)); } - void GetStatistic(const uint8_t* src, size_t stride, size_t width, size_t height, - uint8_t* min, uint8_t* max, uint8_t* average) - { - if (Aligned(src) && Aligned(stride)) - GetStatistic(src, stride, width, height, min, max, average); - else - GetStatistic(src, stride, width, height, min, max, average); - } - //----------------------------------------------------------------------------------------- - template void GetRowSums(const uint8_t* src, size_t stride, size_t width, size_t height, uint32_t* sums) + void GetRowSums(const uint8_t* src, size_t stride, size_t width, size_t height, uint32_t* sums) { size_t alignedWidth = AlignLo(width, A); __m128i tailMask = ShiftLeft(K_INV_ZERO, A - width + alignedWidth); @@ -102,12 +91,12 @@ namespace Simd __m128i sum = _mm_setzero_si128(); for (size_t col = 0; col < alignedWidth; col += A) { - __m128i _src = Load((__m128i*)(src + col)); + __m128i _src = _mm_loadu_si128((__m128i*)(src + col)); sum = _mm_add_epi32(sum, _mm_sad_epu8(_src, K_ZERO)); } if (alignedWidth != width) { - __m128i _src = _mm_and_si128(Load((__m128i*)(src + width - A)), tailMask); + __m128i _src = _mm_and_si128(_mm_loadu_si128((__m128i*)(src + width - A)), tailMask); sum = _mm_add_epi32(sum, _mm_sad_epu8(_src, K_ZERO)); } sums[row] = ExtractInt32Sum(sum); @@ -115,17 +104,9 @@ namespace Simd } } - void GetRowSums(const uint8_t* src, size_t stride, size_t width, size_t height, uint32_t* sums) - { - if (Aligned(src) && Aligned(stride)) - GetRowSums(src, stride, width, height, sums); - else - GetRowSums(src, stride, width, height, sums); - } - //----------------------------------------------------------------------------------------- - template void GetAbsDyRowSums(const uint8_t* src, size_t stride, size_t width, size_t height, uint32_t* sums) + void GetAbsDyRowSums(const uint8_t* src, size_t stride, size_t width, size_t height, uint32_t* sums) { size_t alignedWidth = AlignLo(width, A); __m128i tailMask = ShiftLeft(K_INV_ZERO, A - width + alignedWidth); @@ -139,14 +120,14 @@ namespace Simd __m128i sum = _mm_setzero_si128(); for (size_t col = 0; col < alignedWidth; col += A) { - __m128i _src0 = Load((__m128i*)(src0 + col)); - __m128i _src1 = Load((__m128i*)(src1 + col)); + __m128i _src0 = _mm_loadu_si128((__m128i*)(src0 + col)); + __m128i _src1 = _mm_loadu_si128((__m128i*)(src1 + col)); sum = _mm_add_epi32(sum, _mm_sad_epu8(_src0, _src1)); } if (alignedWidth != width) { - __m128i _src0 = _mm_and_si128(Load((__m128i*)(src0 + width - A)), tailMask); - __m128i _src1 = _mm_and_si128(Load((__m128i*)(src1 + width - A)), tailMask); + __m128i _src0 = _mm_and_si128(_mm_loadu_si128((__m128i*)(src0 + width - A)), tailMask); + __m128i _src1 = _mm_and_si128(_mm_loadu_si128((__m128i*)(src1 + width - A)), tailMask); sum = _mm_add_epi32(sum, _mm_sad_epu8(_src0, _src1)); } sums[row] = ExtractInt32Sum(sum); @@ -155,21 +136,9 @@ namespace Simd } } - void GetAbsDyRowSums(const uint8_t* src, size_t stride, size_t width, size_t height, uint32_t* sums) - { - if (Aligned(src) && Aligned(stride)) - GetAbsDyRowSums(src, stride, width, height, sums); - else - GetAbsDyRowSums(src, stride, width, height, sums); - } - - //----------------------------------------------------------------------------------------- - - template void ValueSum(const uint8_t* src, size_t stride, size_t width, size_t height, uint64_t* sum) + void ValueSum(const uint8_t* src, size_t stride, size_t width, size_t height, uint64_t* sum) { assert(width >= A); - if (align) - assert(Aligned(src) && Aligned(stride)); size_t bodyWidth = AlignLo(width, A); __m128i tailMask = ShiftLeft(K_INV_ZERO, A - width + bodyWidth); @@ -178,12 +147,12 @@ namespace Simd { for (size_t col = 0; col < bodyWidth; col += A) { - const __m128i src_ = Load((__m128i*)(src + col)); + const __m128i src_ = _mm_loadu_si128((__m128i*)(src + col)); fullSum = _mm_add_epi64(_mm_sad_epu8(src_, K_ZERO), fullSum); } if (width - bodyWidth) { - const __m128i src_ = _mm_and_si128(tailMask, Load((__m128i*)(src + width - A))); + const __m128i src_ = _mm_and_si128(tailMask, _mm_loadu_si128((__m128i*)(src + width - A))); fullSum = _mm_add_epi64(_mm_sad_epu8(src_, K_ZERO), fullSum); } src += stride; @@ -191,14 +160,6 @@ namespace Simd *sum = ExtractInt64Sum(fullSum); } - void ValueSum(const uint8_t* src, size_t stride, size_t width, size_t height, uint64_t* sum) - { - if (Aligned(src) && Aligned(stride)) - ValueSum(src, stride, width, height, sum); - else - ValueSum(src, stride, width, height, sum); - } - //----------------------------------------------------------------------------------------- SIMD_INLINE __m128i Square(__m128i src) @@ -208,11 +169,9 @@ namespace Simd return _mm_add_epi32(_mm_madd_epi16(lo, lo), _mm_madd_epi16(hi, hi)); } - template void SquareSum(const uint8_t* src, size_t stride, size_t width, size_t height, uint64_t* sum) + void SquareSum(const uint8_t* src, size_t stride, size_t width, size_t height, uint64_t* sum) { assert(width >= A); - if (align) - assert(Aligned(src) && Aligned(stride)); size_t bodyWidth = AlignLo(width, A); __m128i tailMask = ShiftLeft(K_INV_ZERO, A - width + bodyWidth); @@ -222,12 +181,12 @@ namespace Simd __m128i rowSum = _mm_setzero_si128(); for (size_t col = 0; col < bodyWidth; col += A) { - const __m128i src_ = Load((__m128i*)(src + col)); + const __m128i src_ = _mm_loadu_si128((__m128i*)(src + col)); rowSum = _mm_add_epi32(rowSum, Square(src_)); } if (width - bodyWidth) { - const __m128i src_ = _mm_and_si128(tailMask, Load((__m128i*)(src + width - A))); + const __m128i src_ = _mm_and_si128(tailMask, _mm_loadu_si128((__m128i*)(src + width - A))); rowSum = _mm_add_epi32(rowSum, Square(src_)); } fullSum = _mm_add_epi64(fullSum, HorizontalSum32(rowSum)); @@ -236,21 +195,11 @@ namespace Simd *sum = ExtractInt64Sum(fullSum); } - void SquareSum(const uint8_t* src, size_t stride, size_t width, size_t height, uint64_t* sum) - { - if (Aligned(src) && Aligned(stride)) - SquareSum(src, stride, width, height, sum); - else - SquareSum(src, stride, width, height, sum); - } - //----------------------------------------------------------------------------------------- - template void ValueSquareSum(const uint8_t* src, size_t stride, size_t width, size_t height, uint64_t* valueSum, uint64_t* squareSum) + void ValueSquareSum(const uint8_t* src, size_t stride, size_t width, size_t height, uint64_t* valueSum, uint64_t* squareSum) { assert(width >= A); - if (align) - assert(Aligned(src) && Aligned(stride)); size_t bodyWidth = AlignLo(width, A); __m128i tailMask = ShiftLeft(K_INV_ZERO, A - width + bodyWidth); @@ -261,13 +210,13 @@ namespace Simd __m128i rowSquareSum = _mm_setzero_si128(); for (size_t col = 0; col < bodyWidth; col += A) { - const __m128i value = Load((__m128i*)(src + col)); + const __m128i value = _mm_loadu_si128((__m128i*)(src + col)); fullValueSum = _mm_add_epi64(_mm_sad_epu8(value, K_ZERO), fullValueSum); rowSquareSum = _mm_add_epi32(rowSquareSum, Square(value)); } if (width - bodyWidth) { - const __m128i value = _mm_and_si128(tailMask, Load((__m128i*)(src + width - A))); + const __m128i value = _mm_and_si128(tailMask, _mm_loadu_si128((__m128i*)(src + width - A))); fullValueSum = _mm_add_epi64(_mm_sad_epu8(value, K_ZERO), fullValueSum); rowSquareSum = _mm_add_epi32(rowSquareSum, Square(value)); } @@ -278,14 +227,6 @@ namespace Simd *squareSum = ExtractInt64Sum(fullSquareSum); } - void ValueSquareSum(const uint8_t* src, size_t stride, size_t width, size_t height, uint64_t* valueSum, uint64_t* squareSum) - { - if (Aligned(src) && Aligned(stride)) - ValueSquareSum(src, stride, width, height, valueSum, squareSum); - else - ValueSquareSum(src, stride, width, height, valueSum, squareSum); - } - //----------------------------------------------------------------------------------------- SIMD_INLINE __m128i Correlation(__m128i a, __m128i b) @@ -295,11 +236,9 @@ namespace Simd return _mm_add_epi32(lo, hi); } - template void CorrelationSum(const uint8_t* a, size_t aStride, const uint8_t* b, size_t bStride, size_t width, size_t height, uint64_t* sum) + void CorrelationSum(const uint8_t* a, size_t aStride, const uint8_t* b, size_t bStride, size_t width, size_t height, uint64_t* sum) { assert(width >= A); - if (align) - assert(Aligned(a) && Aligned(aStride) && Aligned(b) && Aligned(bStride)); size_t bodyWidth = AlignLo(width, A); __m128i tailMask = ShiftLeft(K_INV_ZERO, A - width + bodyWidth); @@ -309,14 +248,14 @@ namespace Simd __m128i rowSum = _mm_setzero_si128(); for (size_t col = 0; col < bodyWidth; col += A) { - const __m128i a_ = Load((__m128i*)(a + col)); - const __m128i b_ = Load((__m128i*)(b + col)); + const __m128i a_ = _mm_loadu_si128((__m128i*)(a + col)); + const __m128i b_ = _mm_loadu_si128((__m128i*)(b + col)); rowSum = _mm_add_epi32(rowSum, Correlation(a_, b_)); } if (width - bodyWidth) { - const __m128i a_ = _mm_and_si128(tailMask, Load((__m128i*)(a + width - A))); - const __m128i b_ = _mm_and_si128(tailMask, Load((__m128i*)(b + width - A))); + const __m128i a_ = _mm_and_si128(tailMask, _mm_loadu_si128((__m128i*)(a + width - A))); + const __m128i b_ = _mm_and_si128(tailMask, _mm_loadu_si128((__m128i*)(b + width - A))); rowSum = _mm_add_epi32(rowSum, Correlation(a_, b_)); } fullSum = _mm_add_epi64(fullSum, HorizontalSum32(rowSum)); @@ -326,14 +265,6 @@ namespace Simd *sum = ExtractInt64Sum(fullSum); } - void CorrelationSum(const uint8_t* a, size_t aStride, const uint8_t* b, size_t bStride, size_t width, size_t height, uint64_t* sum) - { - if (Aligned(a) && Aligned(aStride) && Aligned(b) && Aligned(bStride)) - CorrelationSum(a, aStride, b, bStride, width, height, sum); - else - CorrelationSum(a, aStride, b, bStride, width, height, sum); - } - //----------------------------------------------------------------------------------------- SIMD_INLINE __m128i Square8u(__m128i src) diff --git a/src/Simd/SimdSse41StatisticMoments.cpp b/src/Simd/SimdSse41StatisticMoments.cpp index 12180462fa..0f1e103c8c 100644 --- a/src/Simd/SimdSse41StatisticMoments.cpp +++ b/src/Simd/SimdSse41StatisticMoments.cpp @@ -48,7 +48,7 @@ namespace Simd col = _mm_add_epi16(col, K16_0008); } - template void GetObjectMoments(const uint8_t* src, size_t srcStride, size_t width, size_t height, const uint8_t * mask, size_t maskStride, uint8_t index, + void GetObjectMoments(const uint8_t* src, size_t srcStride, size_t width, size_t height, const uint8_t * mask, size_t maskStride, uint8_t index, __m128i & n, __m128i & s, __m128i & sx, __m128i & sy, __m128i & sxx, __m128i& sxy, __m128i& syy) { size_t widthA = AlignLo(width, A); @@ -74,12 +74,12 @@ namespace Simd { for (size_t col = colB; col < colE; col += A) { - __m128i _src = Load((__m128i*)(src + col)); + __m128i _src = _mm_loadu_si128((__m128i*)(src + col)); GetObjectMoments8(_src, K_INV_ZERO, _col, _n, _s, _sx, _sxx); } if (colB == widthB && widthA < width) { - __m128i _src = Load((__m128i*)(src + width - A)); + __m128i _src = _mm_loadu_si128((__m128i*)(src + width - A)); _col = tailCol; GetObjectMoments8(_src, tailMask, _col, _n, _s, _sx, _sxx); colE = width; @@ -89,12 +89,12 @@ namespace Simd { for (size_t col = colB; col < colE; col += A) { - __m128i _mask = _mm_cmpeq_epi8(Load((__m128i*)(mask + col)), _index); + __m128i _mask = _mm_cmpeq_epi8(_mm_loadu_si128((__m128i*)(mask + col)), _index); GetObjectMoments8(K8_01, _mask, _col, _n, _s, _sx, _sxx); } if (colB == widthB && widthA < width) { - __m128i _mask = _mm_and_si128(_mm_cmpeq_epi8(Load((__m128i*)(mask + width - A)), _index), tailMask); + __m128i _mask = _mm_and_si128(_mm_cmpeq_epi8(_mm_loadu_si128((__m128i*)(mask + width - A)), _index), tailMask); _col = tailCol; GetObjectMoments8(K8_01, _mask, _col, _n, _s, _sx, _sxx); colE = width; @@ -104,14 +104,14 @@ namespace Simd { for (size_t col = colB; col < colE; col += A) { - __m128i _src = Load((__m128i*)(src + col)); - __m128i _mask = _mm_cmpeq_epi8(Load((__m128i*)(mask + col)), _index); + __m128i _src = _mm_loadu_si128((__m128i*)(src + col)); + __m128i _mask = _mm_cmpeq_epi8(_mm_loadu_si128((__m128i*)(mask + col)), _index); GetObjectMoments8(_src, _mask, _col, _n, _s, _sx, _sxx); } if (colB == widthB && widthA < width) { - __m128i _mask = _mm_and_si128(_mm_cmpeq_epi8(Load((__m128i*)(mask + width - A)), _index), tailMask); - __m128i _src = Load((__m128i*)(src + width - A)); + __m128i _mask = _mm_and_si128(_mm_cmpeq_epi8(_mm_loadu_si128((__m128i*)(mask + width - A)), _index), tailMask); + __m128i _src = _mm_loadu_si128((__m128i*)(src + width - A)); _col = tailCol; GetObjectMoments8(_src, _mask, _col, _n, _s, _sx, _sxx); colE = width; @@ -152,12 +152,10 @@ namespace Simd } } - template void GetObjectMoments(const uint8_t* src, size_t srcStride, size_t width, size_t height, const uint8_t* mask, size_t maskStride, uint8_t index, + void GetObjectMoments(const uint8_t* src, size_t srcStride, size_t width, size_t height, const uint8_t* mask, size_t maskStride, uint8_t index, uint64_t* n, uint64_t* s, uint64_t* sx, uint64_t* sy, uint64_t* sxx, uint64_t* sxy, uint64_t* syy) { assert(width >= A && (src || mask)); - if (align) - assert((src == NULL || (Aligned(src) && Aligned(srcStride))) && (mask == NULL || (Aligned(mask) && Aligned(maskStride)))); __m128i _n = _mm_setzero_si128(); __m128i _s = _mm_setzero_si128(); @@ -167,7 +165,7 @@ namespace Simd __m128i _sxy = _mm_setzero_si128(); __m128i _syy = _mm_setzero_si128(); - GetObjectMoments(src, srcStride, width, height, mask, maskStride, index, _n, _s, _sx, _sy, _sxx, _sxy, _syy); + GetObjectMoments(src, srcStride, width, height, mask, maskStride, index, _n, _s, _sx, _sy, _sxx, _sxy, _syy); *n = ExtractInt64Sum(_n); *s = ExtractInt64Sum(_s); @@ -178,15 +176,6 @@ namespace Simd *syy = ExtractInt64Sum(_syy); } - void GetObjectMoments(const uint8_t* src, size_t srcStride, size_t width, size_t height, const uint8_t* mask, size_t maskStride, uint8_t index, - uint64_t* n, uint64_t* s, uint64_t* sx, uint64_t* sy, uint64_t* sxx, uint64_t* sxy, uint64_t* syy) - { - if ((src == NULL || (Aligned(src) && Aligned(srcStride))) && (mask == NULL || (Aligned(mask) && Aligned(maskStride)))) - GetObjectMoments(src, srcStride, width, height, mask, maskStride, index, n, s, sx, sy, sxx, sxy, syy); - else - GetObjectMoments(src, srcStride, width, height, mask, maskStride, index, n, s, sx, sy, sxx, sxy, syy); - } - //----------------------------------------------------------------------------------------- void GetMoments(const uint8_t* mask, size_t stride, size_t width, size_t height, uint8_t index, diff --git a/src/Simd/SimdSse41StretchGray2x2.cpp b/src/Simd/SimdSse41StretchGray2x2.cpp index cc21d55b11..d302552c71 100644 --- a/src/Simd/SimdSse41StretchGray2x2.cpp +++ b/src/Simd/SimdSse41StretchGray2x2.cpp @@ -29,22 +29,17 @@ namespace Simd #ifdef SIMD_SSE41_ENABLE namespace Sse41 { - template SIMD_INLINE void StoreUnpacked(__m128i value, uint8_t * dst) + SIMD_INLINE void StoreUnpacked(__m128i value, uint8_t * dst) { - Store((__m128i*)(dst + 0), _mm_unpacklo_epi8(value, value)); - Store((__m128i*)(dst + A), _mm_unpackhi_epi8(value, value)); + _mm_storeu_si128((__m128i*)(dst + 0), _mm_unpacklo_epi8(value, value)); + _mm_storeu_si128((__m128i*)(dst + A), _mm_unpackhi_epi8(value, value)); } - template void StretchGray2x2( + void StretchGray2x2( const uint8_t *src, size_t srcWidth, size_t srcHeight, size_t srcStride, uint8_t *dst, size_t dstWidth, size_t dstHeight, size_t dstStride) { assert(srcWidth * 2 == dstWidth && srcHeight * 2 == dstHeight && srcWidth >= A); - if (align) - { - assert(Aligned(src) && Aligned(srcStride)); - assert(Aligned(dst) && Aligned(dstStride)); - } size_t alignedWidth = AlignLo(srcWidth, A); for (size_t row = 0; row < srcHeight; ++row) @@ -53,29 +48,20 @@ namespace Simd uint8_t * dstOdd = dst + dstStride; for (size_t srcCol = 0, dstCol = 0; srcCol < alignedWidth; srcCol += A, dstCol += DA) { - __m128i value = Load((__m128i*)(src + srcCol)); - StoreUnpacked(value, dstEven + dstCol); - StoreUnpacked(value, dstOdd + dstCol); + __m128i value = _mm_loadu_si128((__m128i*)(src + srcCol)); + StoreUnpacked(value, dstEven + dstCol); + StoreUnpacked(value, dstOdd + dstCol); } if (alignedWidth != srcWidth) { - __m128i value = Load((__m128i*)(src + srcWidth - A)); - StoreUnpacked(value, dstEven + dstWidth - 2 * A); - StoreUnpacked(value, dstOdd + dstWidth - 2 * A); + __m128i value = _mm_loadu_si128((__m128i*)(src + srcWidth - A)); + StoreUnpacked(value, dstEven + dstWidth - 2 * A); + StoreUnpacked(value, dstOdd + dstWidth - 2 * A); } src += srcStride; dst += 2 * dstStride; } } - - void StretchGray2x2(const uint8_t *src, size_t srcWidth, size_t srcHeight, size_t srcStride, - uint8_t *dst, size_t dstWidth, size_t dstHeight, size_t dstStride) - { - if (Aligned(src) && Aligned(srcStride) && Aligned(dst) && Aligned(dstStride)) - StretchGray2x2(src, srcWidth, srcHeight, srcStride, dst, dstWidth, dstHeight, dstStride); - else - StretchGray2x2(src, srcWidth, srcHeight, srcStride, dst, dstWidth, dstHeight, dstStride); - } } #endif } diff --git a/src/Simd/SimdSse41Texture.cpp b/src/Simd/SimdSse41Texture.cpp index 58d530dc37..3d40aefd68 100644 --- a/src/Simd/SimdSse41Texture.cpp +++ b/src/Simd/SimdSse41Texture.cpp @@ -45,25 +45,21 @@ namespace Simd return _mm_packus_epi16(lo, hi); } - template SIMD_INLINE void TextureBoostedSaturatedGradient(const uint8_t * src, uint8_t * dx, uint8_t * dy, + SIMD_INLINE void TextureBoostedSaturatedGradient(const uint8_t * src, uint8_t * dx, uint8_t * dy, size_t stride, __m128i saturation, __m128i boost) { - const __m128i s10 = Load((__m128i*)(src - 1)); - const __m128i s12 = Load((__m128i*)(src + 1)); - const __m128i s01 = Load((__m128i*)(src - stride)); - const __m128i s21 = Load((__m128i*)(src + stride)); - Store((__m128i*)dx, TextureBoostedSaturatedGradient8(s10, s12, saturation, boost)); - Store((__m128i*)dy, TextureBoostedSaturatedGradient8(s01, s21, saturation, boost)); + const __m128i s10 = _mm_loadu_si128((__m128i*)(src - 1)); + const __m128i s12 = _mm_loadu_si128((__m128i*)(src + 1)); + const __m128i s01 = _mm_loadu_si128((__m128i*)(src - stride)); + const __m128i s21 = _mm_loadu_si128((__m128i*)(src + stride)); + _mm_storeu_si128((__m128i*)dx, TextureBoostedSaturatedGradient8(s10, s12, saturation, boost)); + _mm_storeu_si128((__m128i*)dy, TextureBoostedSaturatedGradient8(s01, s21, saturation, boost)); } - template void TextureBoostedSaturatedGradient(const uint8_t * src, size_t srcStride, size_t width, size_t height, + void TextureBoostedSaturatedGradient(const uint8_t * src, size_t srcStride, size_t width, size_t height, uint8_t saturation, uint8_t boost, uint8_t * dx, size_t dxStride, uint8_t * dy, size_t dyStride) { assert(width >= A && int(2)*saturation*boost <= 0xFF); - if (align) - { - assert(Aligned(src) && Aligned(srcStride) && Aligned(dx) && Aligned(dxStride) && Aligned(dy) && Aligned(dyStride)); - } size_t alignedWidth = AlignLo(width, A); __m128i _saturation = _mm_set1_epi16(saturation); @@ -77,9 +73,9 @@ namespace Simd for (size_t row = 2; row < height; ++row) { for (size_t col = 0; col < alignedWidth; col += A) - TextureBoostedSaturatedGradient(src + col, dx + col, dy + col, srcStride, _saturation, _boost); + TextureBoostedSaturatedGradient(src + col, dx + col, dy + col, srcStride, _saturation, _boost); if (width != alignedWidth) - TextureBoostedSaturatedGradient(src + width - A, dx + width - A, dy + width - A, srcStride, _saturation, _boost); + TextureBoostedSaturatedGradient(src + width - A, dx + width - A, dy + width - A, srcStride, _saturation, _boost); dx[0] = 0; dy[0] = 0; @@ -94,34 +90,21 @@ namespace Simd memset(dy, 0, width); } - void TextureBoostedSaturatedGradient(const uint8_t * src, size_t srcStride, size_t width, size_t height, - uint8_t saturation, uint8_t boost, uint8_t * dx, size_t dxStride, uint8_t * dy, size_t dyStride) - { - if (Aligned(src) && Aligned(srcStride) && Aligned(dx) && Aligned(dxStride) && Aligned(dy) && Aligned(dyStride)) - TextureBoostedSaturatedGradient(src, srcStride, width, height, saturation, boost, dx, dxStride, dy, dyStride); - else - TextureBoostedSaturatedGradient(src, srcStride, width, height, saturation, boost, dx, dxStride, dy, dyStride); - } - //----------------------------------------------------------------------------------------- - template SIMD_INLINE void TextureBoostedUv(const uint8_t* src, uint8_t* dst, __m128i min8, __m128i max8, __m128i boost16) + SIMD_INLINE void TextureBoostedUv(const uint8_t* src, uint8_t* dst, __m128i min8, __m128i max8, __m128i boost16) { - const __m128i _src = Load((__m128i*)src); + const __m128i _src = _mm_loadu_si128((__m128i*)src); const __m128i saturated = _mm_sub_epi8(_mm_max_epu8(min8, _mm_min_epu8(max8, _src)), min8); const __m128i lo = _mm_mullo_epi16(_mm_unpacklo_epi8(saturated, K_ZERO), boost16); const __m128i hi = _mm_mullo_epi16(_mm_unpackhi_epi8(saturated, K_ZERO), boost16); - Store((__m128i*)dst, _mm_packus_epi16(lo, hi)); + _mm_storeu_si128((__m128i*)dst, _mm_packus_epi16(lo, hi)); } - template void TextureBoostedUv(const uint8_t* src, size_t srcStride, size_t width, size_t height, + void TextureBoostedUv(const uint8_t* src, size_t srcStride, size_t width, size_t height, uint8_t boost, uint8_t* dst, size_t dstStride) { assert(width >= A && boost < 0x80); - if (align) - { - assert(Aligned(src) && Aligned(srcStride) && Aligned(dst) && Aligned(dstStride)); - } size_t alignedWidth = AlignLo(width, A); int min = 128 - (128 / boost); @@ -134,46 +117,33 @@ namespace Simd for (size_t row = 0; row < height; ++row) { for (size_t col = 0; col < alignedWidth; col += A) - TextureBoostedUv(src + col, dst + col, min8, max8, boost16); + TextureBoostedUv(src + col, dst + col, min8, max8, boost16); if (width != alignedWidth) - TextureBoostedUv(src + width - A, dst + width - A, min8, max8, boost16); + TextureBoostedUv(src + width - A, dst + width - A, min8, max8, boost16); src += srcStride; dst += dstStride; } } - void TextureBoostedUv(const uint8_t* src, size_t srcStride, size_t width, size_t height, - uint8_t boost, uint8_t* dst, size_t dstStride) - { - if (Aligned(src) && Aligned(srcStride) && Aligned(dst) && Aligned(dstStride)) - TextureBoostedUv(src, srcStride, width, height, boost, dst, dstStride); - else - TextureBoostedUv(src, srcStride, width, height, boost, dst, dstStride); - } - //----------------------------------------------------------------------------------------- - template SIMD_INLINE void TextureGetDifferenceSum(const uint8_t* src, const uint8_t* lo, const uint8_t* hi, + SIMD_INLINE void TextureGetDifferenceSum(const uint8_t* src, const uint8_t* lo, const uint8_t* hi, __m128i& positive, __m128i& negative, const __m128i& mask) { - const __m128i _src = Load((__m128i*)src); - const __m128i _lo = Load((__m128i*)lo); - const __m128i _hi = Load((__m128i*)hi); + const __m128i _src = _mm_loadu_si128((__m128i*)src); + const __m128i _lo = _mm_loadu_si128((__m128i*)lo); + const __m128i _hi = _mm_loadu_si128((__m128i*)hi); const __m128i average = _mm_and_si128(mask, _mm_avg_epu8(_lo, _hi)); const __m128i current = _mm_and_si128(mask, _src); positive = _mm_add_epi64(positive, _mm_sad_epu8(_mm_subs_epu8(current, average), K_ZERO)); negative = _mm_add_epi64(negative, _mm_sad_epu8(_mm_subs_epu8(average, current), K_ZERO)); } - template void TextureGetDifferenceSum(const uint8_t* src, size_t srcStride, size_t width, size_t height, + void TextureGetDifferenceSum(const uint8_t* src, size_t srcStride, size_t width, size_t height, const uint8_t* lo, size_t loStride, const uint8_t* hi, size_t hiStride, int64_t* sum) { assert(width >= A && sum != NULL); - if (align) - { - assert(Aligned(src) && Aligned(srcStride) && Aligned(lo) && Aligned(loStride) && Aligned(hi) && Aligned(hiStride)); - } size_t alignedWidth = AlignLo(width, A); __m128i tailMask = ShiftLeft(K_INV_ZERO, A - width + alignedWidth); @@ -182,9 +152,9 @@ namespace Simd for (size_t row = 0; row < height; ++row) { for (size_t col = 0; col < alignedWidth; col += A) - TextureGetDifferenceSum(src + col, lo + col, hi + col, positive, negative, K_INV_ZERO); + TextureGetDifferenceSum(src + col, lo + col, hi + col, positive, negative, K_INV_ZERO); if (width != alignedWidth) - TextureGetDifferenceSum(src + width - A, lo + width - A, hi + width - A, positive, negative, tailMask); + TextureGetDifferenceSum(src + width - A, lo + width - A, hi + width - A, positive, negative, tailMask); src += srcStride; lo += loStride; hi += hiStride; @@ -192,25 +162,18 @@ namespace Simd *sum = ExtractInt64Sum(positive) - ExtractInt64Sum(negative); } - void TextureGetDifferenceSum(const uint8_t* src, size_t srcStride, size_t width, size_t height, - const uint8_t* lo, size_t loStride, const uint8_t* hi, size_t hiStride, int64_t* sum) - { - if (Aligned(src) && Aligned(srcStride) && Aligned(lo) && Aligned(loStride) && Aligned(hi) && Aligned(hiStride)) - TextureGetDifferenceSum(src, srcStride, width, height, lo, loStride, hi, hiStride, sum); - else - TextureGetDifferenceSum(src, srcStride, width, height, lo, loStride, hi, hiStride, sum); - } - //----------------------------------------------------------------------------------------- - template void TexturePerformCompensation(const uint8_t* src, size_t srcStride, size_t width, size_t height, + void TexturePerformCompensation(const uint8_t* src, size_t srcStride, size_t width, size_t height, int shift, uint8_t* dst, size_t dstStride) { - assert(width >= A && shift > -0xFF && shift < 0xFF && shift != 0); - if (align) + if (shift == 0) { - assert(Aligned(src) && Aligned(srcStride) && Aligned(dst) && Aligned(dstStride)); + if (src != dst) + Base::Copy(src, srcStride, width, height, 1, dst, dstStride); + return; } + assert(width >= A && shift > -0xFF && shift < 0xFF && shift != 0); size_t alignedWidth = AlignLo(width, A); __m128i tailMask = src == dst ? ShiftLeft(K_INV_ZERO, A - width + alignedWidth) : K_INV_ZERO; @@ -221,13 +184,13 @@ namespace Simd { for (size_t col = 0; col < alignedWidth; col += A) { - const __m128i _src = Load((__m128i*) (src + col)); - Store((__m128i*) (dst + col), _mm_adds_epu8(_src, _shift)); + const __m128i _src = _mm_loadu_si128((__m128i*) (src + col)); + _mm_storeu_si128((__m128i*) (dst + col), _mm_adds_epu8(_src, _shift)); } if (width != alignedWidth) { - const __m128i _src = Load((__m128i*) (src + width - A)); - Store((__m128i*) (dst + width - A), _mm_adds_epu8(_src, _mm_and_si128(_shift, tailMask))); + const __m128i _src = _mm_loadu_si128((__m128i*) (src + width - A)); + _mm_storeu_si128((__m128i*) (dst + width - A), _mm_adds_epu8(_src, _mm_and_si128(_shift, tailMask))); } src += srcStride; dst += dstStride; @@ -240,34 +203,19 @@ namespace Simd { for (size_t col = 0; col < alignedWidth; col += A) { - const __m128i _src = Load((__m128i*) (src + col)); - Store((__m128i*) (dst + col), _mm_subs_epu8(_src, _shift)); + const __m128i _src = _mm_loadu_si128((__m128i*) (src + col)); + _mm_storeu_si128((__m128i*) (dst + col), _mm_subs_epu8(_src, _shift)); } if (width != alignedWidth) { - const __m128i _src = Load((__m128i*) (src + width - A)); - Store((__m128i*) (dst + width - A), _mm_subs_epu8(_src, _mm_and_si128(_shift, tailMask))); + const __m128i _src = _mm_loadu_si128((__m128i*) (src + width - A)); + _mm_storeu_si128((__m128i*) (dst + width - A), _mm_subs_epu8(_src, _mm_and_si128(_shift, tailMask))); } src += srcStride; dst += dstStride; } } } - - void TexturePerformCompensation(const uint8_t* src, size_t srcStride, size_t width, size_t height, - int shift, uint8_t* dst, size_t dstStride) - { - if (shift == 0) - { - if (src != dst) - Base::Copy(src, srcStride, width, height, 1, dst, dstStride); - return; - } - if (Aligned(src) && Aligned(srcStride) && Aligned(dst) && Aligned(dstStride)) - TexturePerformCompensation(src, srcStride, width, height, shift, dst, dstStride); - else - TexturePerformCompensation(src, srcStride, width, height, shift, dst, dstStride); - } } #endif }