From 68f98656ff06ac491e4f233f8a77b4a08fd4f87e Mon Sep 17 00:00:00 2001 From: Andrei Karas Date: Wed, 21 Dec 2016 20:09:30 +0300 Subject: Add simd function for dye replaceacolor (Software). --- src/CMakeLists.txt | 1 + src/Makefile.am | 1 + src/resources/dye/dyepalette.cpp | 92 ---------- src/resources/dye/dyepalette.h | 22 +++ src/resources/dye/dyepalette_replaceacolor.cpp | 233 +++++++++++++++++++++++++ 5 files changed, 257 insertions(+), 92 deletions(-) create mode 100644 src/resources/dye/dyepalette_replaceacolor.cpp (limited to 'src') diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt index 97802e33d..efc8b17f5 100644 --- a/src/CMakeLists.txt +++ b/src/CMakeLists.txt @@ -643,6 +643,7 @@ SET(SRCS resources/dye/dyecolor.h resources/dye/dyepalette.cpp resources/dye/dyepalette.h + resources/dye/dyepalette_replaceacolor.cpp resources/dye/dyepalette_replacescolor.cpp resources/dye/dyepalette_replacesoglcolor.cpp resources/effectdescription.h diff --git a/src/Makefile.am b/src/Makefile.am index 3ed36374c..bf5c33a5b 100644 --- a/src/Makefile.am +++ b/src/Makefile.am @@ -365,6 +365,7 @@ SRC += events/actionevent.h \ resources/dye/dyecolor.h \ resources/dye/dyepalette.cpp \ resources/dye/dyepalette.h \ + resources/dye/dyepalette_replaceacolor.cpp \ resources/dye/dyepalette_replacescolor.cpp \ resources/dye/dyepalette_replacesoglcolor.cpp \ resources/fboinfo.h \ diff --git a/src/resources/dye/dyepalette.cpp b/src/resources/dye/dyepalette.cpp index def804bf8..ff9ad9c32 100644 --- a/src/resources/dye/dyepalette.cpp +++ b/src/resources/dye/dyepalette.cpp @@ -224,98 +224,6 @@ void DyePalette::getColor(double intensity, intensity * colorJ.value[2]); } -void DyePalette::replaceAColor(uint32_t *restrict pixels, - const int bufSize) const restrict2 -{ - std::vector::const_iterator it_end = mColors.end(); - const size_t sz = mColors.size(); - if (!sz || !pixels) - return; - if (sz % 2) - -- it_end; - -#ifdef ENABLE_CILKPLUS - cilk_for (int ptr = 0; ptr < bufSize; ptr ++) - { - uint8_t *const p = reinterpret_cast(&pixels[ptr]); - const unsigned int data = pixels[ptr]; - - std::vector::const_iterator it = mColors.begin(); - while (it != it_end) - { - const DyeColor &col = *it; - ++ it; - const DyeColor &col2 = *it; - -#if SDL_BYTEORDER == SDL_BIG_ENDIAN - const unsigned int coldata = (col.value[3] << 24U) - | (col.value[2] << 16U) - | (col.value[1] << 8U) - | (col.value[0]); -#else // SDL_BYTEORDER == SDL_BIG_ENDIAN - - const unsigned int coldata = (col.value[3]) - | (col.value[2] << 8U) - | (col.value[1] << 16U) | - (col.value[0] << 24U); -#endif // SDL_BYTEORDER == SDL_BIG_ENDIAN - - if (data == coldata) - { - p[3] = col2.value[0]; - p[2] = col2.value[1]; - p[1] = col2.value[2]; - p[0] = col2.value[3]; - break; - } - - ++ it; - } - } - -#else // ENABLE_CILKPLUS - - for (const uint32_t *const p_end = pixels + CAST_SIZE(bufSize); - pixels != p_end; - ++pixels) - { - uint8_t *const p = reinterpret_cast(pixels); - const unsigned int data = *pixels; - - std::vector::const_iterator it = mColors.begin(); - while (it != it_end) - { - const DyeColor &col = *it; - ++ it; - const DyeColor &col2 = *it; - -#if SDL_BYTEORDER == SDL_BIG_ENDIAN - const unsigned int coldata = (col.value[3] << 24U) - | (col.value[2] << 16U) - | (col.value[1] << 8U) - | (col.value[0]); -#else // SDL_BYTEORDER == SDL_BIG_ENDIAN - const unsigned int coldata = (col.value[3]) - | (col.value[2] << 8U) - | (col.value[1] << 16U) | - (col.value[0] << 24U); -#endif // SDL_BYTEORDER == SDL_BIG_ENDIAN - - if (data == coldata) - { - p[3] = col2.value[0]; - p[2] = col2.value[1]; - p[1] = col2.value[2]; - p[0] = col2.value[3]; - break; - } - - ++ it; - } - } -#endif // ENABLE_CILKPLUS -} - void DyePalette::replaceAOGLColor(uint32_t *restrict pixels, const int bufSize) const restrict2 { diff --git a/src/resources/dye/dyepalette.h b/src/resources/dye/dyepalette.h index 4622527fd..edfe6b5ac 100644 --- a/src/resources/dye/dyepalette.h +++ b/src/resources/dye/dyepalette.h @@ -93,6 +93,28 @@ class DyePalette final void replaceAColor(uint32_t *restrict pixels, const int bufSize) const restrict2; + /** + * replace colors for SDL for A dye. + */ + void replaceAColorDefault(uint32_t *restrict pixels, + const int bufSize) const restrict2; + + /** + * replace colors for SDL for A dye. + */ + FUNCTION_SIMD_DEFAULT + void replaceAColorSimd(uint32_t *restrict pixels, + const int bufSize) const restrict2; + +#ifdef SIMD_SUPPORTED + /** + * replace colors for SDL for A dye. + */ + __attribute__ ((target ("avx2"))) + void replaceAColorSimd(uint32_t *restrict pixels, + const int bufSize) const restrict2; +#endif // SIMD_SUPPORTED + /** * replace colors for OpenGL for S dye. */ diff --git a/src/resources/dye/dyepalette_replaceacolor.cpp b/src/resources/dye/dyepalette_replaceacolor.cpp new file mode 100644 index 000000000..f98c52343 --- /dev/null +++ b/src/resources/dye/dyepalette_replaceacolor.cpp @@ -0,0 +1,233 @@ +/* + * The ManaPlus Client + * Copyright (C) 2007-2009 The Mana World Development Team + * Copyright (C) 2009-2010 The Mana Developers + * Copyright (C) 2011-2016 The ManaPlus Developers + * + * This file is part of The ManaPlus Client. + * + * This program is free software; you can redistribute it and/or modify + * it under the terms of the GNU General Public License as published by + * the Free Software Foundation; either version 2 of the License, or + * any later version. + * + * This program is distributed in the hope that it will be useful, + * but WITHOUT ANY WARRANTY; without even the implied warranty of + * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + * GNU General Public License for more details. + * + * You should have received a copy of the GNU General Public License + * along with this program. If not, see . + */ + +#include "resources/dye/dyepalette.h" + +#include "logger.h" + +#ifndef SDL_BIG_ENDIAN +#include +#endif // SDL_BYTEORDER + +#ifdef __x86_64__ +// avx2 +#include "immintrin.h" + +#if SDL_BYTEORDER == SDL_BIG_ENDIAN +#define buildHex(a, b, c, d) \ + (d) * 16777216U + (c) * 65536U + (b) * 256U + CAST_U32(a) +#else // SDL_BYTEORDER == SDL_BIG_ENDIAN +#define buildHex(a, b, c, d) \ + (a) * 16777216U + (b) * 65536U + (c) * 256U + CAST_U32(d) +#endif // SDL_BYTEORDER == SDL_BIG_ENDIAN +#endif // __x86_64__ + +#include "debug.h" + +void DyePalette::replaceAColor(uint32_t *restrict pixels, + const int bufSize) const restrict2 +{ +#ifdef SIMD_SUPPORTED + replaceAColorSimd(pixels, bufSize); +#else // SIMD_SUPPORTED + replaceAColorDefault(pixels, bufSize); +#endif // SIMD_SUPPORTED +} + +void DyePalette::replaceAColorDefault(uint32_t *restrict pixels, + const int bufSize) const restrict2 +{ + std::vector::const_iterator it_end = mColors.end(); + const size_t sz = mColors.size(); + if (!sz || !pixels) + return; + if (sz % 2) + -- it_end; + +#ifdef ENABLE_CILKPLUS + cilk_for (int ptr = 0; ptr < bufSize; ptr ++) + { + uint8_t *const p = reinterpret_cast(&pixels[ptr]); + const unsigned int data = pixels[ptr]; + + std::vector::const_iterator it = mColors.begin(); + while (it != it_end) + { + const DyeColor &col = *it; + ++ it; + const DyeColor &col2 = *it; + +#if SDL_BYTEORDER == SDL_BIG_ENDIAN + const unsigned int coldata = (col.value[3] << 24U) + | (col.value[2] << 16U) + | (col.value[1] << 8U) + | (col.value[0]); +#else // SDL_BYTEORDER == SDL_BIG_ENDIAN + + const unsigned int coldata = (col.value[3]) + | (col.value[2] << 8U) + | (col.value[1] << 16U) | + (col.value[0] << 24U); +#endif // SDL_BYTEORDER == SDL_BIG_ENDIAN + + if (data == coldata) + { + p[3] = col2.value[0]; + p[2] = col2.value[1]; + p[1] = col2.value[2]; + p[0] = col2.value[3]; + break; + } + + ++ it; + } + } + +#else // ENABLE_CILKPLUS + + for (const uint32_t *const p_end = pixels + CAST_SIZE(bufSize); + pixels != p_end; + ++pixels) + { + uint8_t *const p = reinterpret_cast(pixels); + const unsigned int data = *pixels; + + std::vector::const_iterator it = mColors.begin(); + while (it != it_end) + { + const DyeColor &col = *it; + ++ it; + const DyeColor &col2 = *it; + +#if SDL_BYTEORDER == SDL_BIG_ENDIAN + const unsigned int coldata = (col.value[3] << 24U) + | (col.value[2] << 16U) + | (col.value[1] << 8U) + | (col.value[0]); +#else // SDL_BYTEORDER == SDL_BIG_ENDIAN + const unsigned int coldata = (col.value[3]) + | (col.value[2] << 8U) + | (col.value[1] << 16U) | + (col.value[0] << 24U); +#endif // SDL_BYTEORDER == SDL_BIG_ENDIAN + + if (data == coldata) + { + p[3] = col2.value[0]; + p[2] = col2.value[1]; + p[1] = col2.value[2]; + p[0] = col2.value[3]; + break; + } + + ++ it; + } + } +#endif // ENABLE_CILKPLUS +} + +#ifdef SIMD_SUPPORTED +static void print256(const char *const text, const __m256i &val); +static void print256(const char *const text, const __m256i &val) +{ + printf("%s 0x%016llx%016llx%016llx%016llx\n", text, val[0], val[1], val[2], val[3]); +} + +__attribute__ ((target ("avx2"))) +void DyePalette::replaceAColorSimd(uint32_t *restrict pixels, + const int bufSize) const restrict2 +{ + std::vector::const_iterator it_end = mColors.end(); + const size_t sz = mColors.size(); + if (!sz || !pixels) + return; + if (sz % 2) + -- it_end; + const int mod = bufSize % 8; + const int bufEnd = bufSize - mod; + + for (int ptr = 0; ptr < bufEnd; ptr += 8) + { + //__m256i base = _mm256_load_si256(reinterpret_cast<__m256i*>(pixels)); + __m256i base = _mm256_loadu_si256(reinterpret_cast<__m256i*>(&pixels[ptr])); + + std::vector::const_iterator it = mColors.begin(); + while (it != it_end) + { + const DyeColor &col = *it; + ++ it; + const DyeColor &col2 = *it; + + __m256i newMask = _mm256_set1_epi32(buildHex(col2.value[0], col2.value[1], col2.value[2], col2.value[3])); + __m256i cmpMask = _mm256_set1_epi32(buildHex(col.value[0], col.value[1], col.value[2], col.value[3])); + __m256i cmpRes = _mm256_cmpeq_epi32(base, cmpMask); + __m256i srcAnd = _mm256_andnot_si256(cmpRes, base); + __m256i dstAnd = _mm256_and_si256(cmpRes, newMask); + base = _mm256_or_si256(srcAnd, dstAnd); + + ++ it; + } + //print256("res ", base); + //_mm256_store_si256(reinterpret_cast<__m256i*>(pixels), base); + _mm256_storeu_si256(reinterpret_cast<__m256i*>(&pixels[ptr]), base); + } + + // complete end without simd + for (int ptr = bufSize - mod; ptr < bufSize; ptr ++) + { + uint8_t *const p = reinterpret_cast(&pixels[ptr]); + const unsigned int data = pixels[ptr]; + + std::vector::const_iterator it = mColors.begin(); + while (it != it_end) + { + const DyeColor &col = *it; + ++ it; + const DyeColor &col2 = *it; + + const unsigned int coldata = (col.value[3]) | + (col.value[2] << 8U) | + (col.value[1] << 16U) | + (col.value[0] << 24U); + + if (data == coldata) + { + p[3] = col2.value[0]; + p[2] = col2.value[1]; + p[1] = col2.value[2]; + p[0] = col2.value[3]; + break; + } + + ++ it; + } + } +} + +#endif // SIMD_SUPPORTED + +FUNCTION_SIMD_DEFAULT +void DyePalette::replaceAColorSimd(uint32_t *restrict pixels, + const int bufSize) const restrict2 +{ + replaceAColorDefault(pixels, bufSize); +} -- cgit v1.2.3-60-g2f50