From 1dcb529da1483ef67426236480abe79e5e952ef0 Mon Sep 17 00:00:00 2001 From: Andrei Karas Date: Tue, 20 Dec 2016 23:51:08 +0300 Subject: Add simd function for dye replace color (software). --- src/CMakeLists.txt | 1 + src/Makefile.am | 1 + src/localconsts.h | 10 ++ src/resources/dye/dyecolor.h | 13 +- src/resources/dye/dyepalette.cpp | 115 ------------ src/resources/dye/dyepalette.h | 22 +++ src/resources/dye/dyepalette_replacescolor.cpp | 237 +++++++++++++++++++++++++ 7 files changed, 283 insertions(+), 116 deletions(-) create mode 100644 src/resources/dye/dyepalette_replacescolor.cpp diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt index fbbde02c1..c72e42bad 100644 --- a/src/CMakeLists.txt +++ b/src/CMakeLists.txt @@ -643,6 +643,7 @@ SET(SRCS resources/dye/dyecolor.h resources/dye/dyepalette.cpp resources/dye/dyepalette.h + resources/dye/dyepalette_replacescolor.cpp resources/effectdescription.h resources/emoteinfo.h resources/emotesprite.h diff --git a/src/Makefile.am b/src/Makefile.am index ebf972ca2..1466f573c 100644 --- a/src/Makefile.am +++ b/src/Makefile.am @@ -365,6 +365,7 @@ SRC += events/actionevent.h \ resources/dye/dyecolor.h \ resources/dye/dyepalette.cpp \ resources/dye/dyepalette.h \ + resources/dye/dyepalette_replacescolor.cpp \ resources/fboinfo.h \ resources/frame.h \ resources/image/image.cpp \ diff --git a/src/localconsts.h b/src/localconsts.h index 00108b5eb..2d2b2d548 100644 --- a/src/localconsts.h +++ b/src/localconsts.h @@ -138,6 +138,16 @@ #define A_INLINE __attribute__ ((always_inline)) #endif // ENABLE_CILKPLUS +#ifdef __x86_64__ +#define SIMD_SUPPORTED +#endif // __x86_64__ + +#ifdef SIMD_SUPPORTED +#define FUNCTION_SIMD_DEFAULT __attribute__ ((target ("default"))) +#else // SIMD_SUPPORTED +#define FUNCTION_SIMD_DEFAULT +#endif // SIMD_SUPPORTED + #ifdef __INTEL_COMPILER #define RETURNS_NONNULL #else // __INTEL_COMPILER diff --git a/src/resources/dye/dyecolor.h b/src/resources/dye/dyecolor.h index c28428a0d..11e12a038 100644 --- a/src/resources/dye/dyecolor.h +++ b/src/resources/dye/dyecolor.h @@ -21,6 +21,10 @@ #ifndef RESOURCES_DYE_DYECOLOR_H #define RESOURCES_DYE_DYECOLOR_H +#ifndef SDL_BIG_ENDIAN +#include +#endif // SDL_BYTEORDER + #include "localconsts.h" struct DyeColor final @@ -38,6 +42,7 @@ struct DyeColor final value[1] = g; value[2] = b; value[3] = 255; +// value2 = buildHex(r, g, b, 255); } DyeColor(const uint8_t r, @@ -49,9 +54,15 @@ struct DyeColor final value[1] = g; value[2] = b; value[3] = a; +// value2 = buildHex(r, g, b, a); } - uint8_t value[4]; + union + { + uint8_t value[4]; + uint32_t value1; + }; +// uint32_t value2; }; #endif // RESOURCES_DYE_DYECOLOR_H diff --git a/src/resources/dye/dyepalette.cpp b/src/resources/dye/dyepalette.cpp index dbb8e4973..8ca07ecc7 100644 --- a/src/resources/dye/dyepalette.cpp +++ b/src/resources/dye/dyepalette.cpp @@ -224,121 +224,6 @@ void DyePalette::getColor(double intensity, intensity * colorJ.value[2]); } -void DyePalette::replaceSColor(uint32_t *restrict pixels, - const int bufSize) const restrict2 -{ - std::vector::const_iterator it_end = mColors.end(); - const size_t sz = mColors.size(); - if (!sz || !pixels) - return; - if (sz % 2) - -- it_end; - -#ifdef ENABLE_CILKPLUS - cilk_for (int ptr = 0; ptr < bufSize; ptr ++) - { - uint8_t *const p = reinterpret_cast(&pixels[ptr]); -#if SDL_BYTEORDER == SDL_BIG_ENDIAN - const int alpha = pixels[ptr] & 0xff000000; - const unsigned int data = pixels[ptr] & 0x00ffffff; -#else // SDL_BYTEORDER == SDL_BIG_ENDIAN - - const int alpha = *p & 0xff; - const unsigned int data = pixels[ptr] & 0xffffff00; -#endif // SDL_BYTEORDER == SDL_BIG_ENDIAN - -// logger->log("c:%04d %08x", c, *pixels); -// logger->log("data: %08x", data); - if (!alpha) - { -// logger->log("skip: %08x", *pixels); - } - else - { - std::vector::const_iterator it = mColors.begin(); - while (it != it_end) - { - const DyeColor &col = *it; - ++ it; - const DyeColor &col2 = *it; - -#if SDL_BYTEORDER == SDL_BIG_ENDIAN - const unsigned int coldata = (col.value[2] << 16U) - | (col.value[1] << 8U) | (col.value[0]); -#else // SDL_BYTEORDER == SDL_BIG_ENDIAN - const unsigned int coldata = (col.value[2] << 8U) - | (col.value[1] << 16U) | (col.value[0] << 24U); -#endif // SDL_BYTEORDER == SDL_BIG_ENDIAN - -// logger->log("coldata: %08x", coldata); - if (data == coldata) - { -// logger->log("correct"); - p[3] = col2.value[0]; - p[2] = col2.value[1]; - p[1] = col2.value[2]; - break; - } - ++ it; - } - } - } -#else // ENABLE_CILKPLUS - - for (const uint32_t *const p_end = pixels + CAST_SIZE(bufSize); - pixels != p_end; - ++ pixels) - { - uint8_t *const p = reinterpret_cast(pixels); -#if SDL_BYTEORDER == SDL_BIG_ENDIAN - const int alpha = *pixels & 0xff000000; - const unsigned int data = (*pixels) & 0x00ffffff; -#else // SDL_BYTEORDER == SDL_BIG_ENDIAN - - const int alpha = *p & 0xff; - const unsigned int data = (*pixels) & 0xffffff00; -#endif // SDL_BYTEORDER == SDL_BIG_ENDIAN - -// logger->log("c:%04d %08x", c, *pixels); -// logger->log("data: %08x", data); - if (!alpha) - { -// logger->log("skip: %08x", *pixels); - continue; - } - - std::vector::const_iterator it = mColors.begin(); - while (it != it_end) - { - const DyeColor &col = *it; - ++ it; - const DyeColor &col2 = *it; - -#if SDL_BYTEORDER == SDL_BIG_ENDIAN - const unsigned int coldata = (col.value[2] << 16U) - | (col.value[1] << 8U) | (col.value[0]); -#else // SDL_BYTEORDER == SDL_BIG_ENDIAN - - const unsigned int coldata = (col.value[2] << 8U) - | (col.value[1] << 16U) | (col.value[0] << 24U); -#endif // SDL_BYTEORDER == SDL_BIG_ENDIAN - -// logger->log("coldata: %08x", coldata); - if (data == coldata) - { -// logger->log("correct"); - p[3] = col2.value[0]; - p[2] = col2.value[1]; - p[1] = col2.value[2]; - break; - } - - ++ it; - } - } -#endif // ENABLE_CILKPLUS -} - void DyePalette::replaceAColor(uint32_t *restrict pixels, const int bufSize) const restrict2 { diff --git a/src/resources/dye/dyepalette.h b/src/resources/dye/dyepalette.h index 5f71180de..0e7567c56 100644 --- a/src/resources/dye/dyepalette.h +++ b/src/resources/dye/dyepalette.h @@ -65,6 +65,28 @@ class DyePalette final void replaceSColor(uint32_t *restrict pixels, const int bufSize) const restrict2; + /** + * replace colors for SDL for S dye. + */ + void replaceSColorDefault(uint32_t *restrict pixels, + const int bufSize) const restrict2; + + /** + * replace colors for SDL for S dye. + */ + FUNCTION_SIMD_DEFAULT + void replaceSColor256(uint32_t *restrict pixels, + const int bufSize) const restrict2; + +#ifdef SIMD_SUPPORTED + /** + * replace colors for SDL for S dye. + */ + __attribute__ ((target ("avx2"))) + void replaceSColor256(uint32_t *restrict pixels, + const int bufSize) const restrict2; +#endif // SIMD_SUPPORTED + /** * replace colors for SDL for A dye. */ diff --git a/src/resources/dye/dyepalette_replacescolor.cpp b/src/resources/dye/dyepalette_replacescolor.cpp new file mode 100644 index 000000000..98525108e --- /dev/null +++ b/src/resources/dye/dyepalette_replacescolor.cpp @@ -0,0 +1,237 @@ +/* + * The ManaPlus Client + * Copyright (C) 2011-2016 The ManaPlus Developers + * + * This file is part of The ManaPlus Client. + * + * This program is free software; you can redistribute it and/or modify + * it under the terms of the GNU General Public License as published by + * the Free Software Foundation; either version 2 of the License, or + * any later version. + * + * This program is distributed in the hope that it will be useful, + * but WITHOUT ANY WARRANTY; without even the implied warranty of + * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + * GNU General Public License for more details. + * + * You should have received a copy of the GNU General Public License + * along with this program. If not, see . + */ + +#include "resources/dye/dyepalette.h" + +#include "logger.h" + +#ifndef SDL_BIG_ENDIAN +#include +#endif // SDL_BYTEORDER + +#ifdef __x86_64__ +// avx2 +#include "immintrin.h" + +#if SDL_BYTEORDER == SDL_BIG_ENDIAN +#define buildHex(a, b, c, d) \ + (d) * 16777216U + (c) * 65536U + (b) * 256U + CAST_U32(a) +#else // SDL_BYTEORDER == SDL_BIG_ENDIAN +#define buildHex(a, b, c, d) \ + (a) * 16777216U + (b) * 65536U + (c) * 256U + CAST_U32(d) +#endif // SDL_BYTEORDER == SDL_BIG_ENDIAN +#endif // __x86_64__ + +#include "debug.h" + +void DyePalette::replaceSColor(uint32_t *restrict pixels, + const int bufSize) const restrict2 +{ +#ifdef __x86_64__ + if (bufSize % 8 == 0) + replaceSColor256(pixels, bufSize); + else + replaceSColorDefault(pixels, bufSize); +#else // __x86_64__ + replaceSColorDefault(pixels, bufSize); +#endif // __x86_64__ +} + +void DyePalette::replaceSColorDefault(uint32_t *restrict pixels, + const int bufSize) const restrict2 +{ + std::vector::const_iterator it_end = mColors.end(); + const size_t sz = mColors.size(); + if (!sz || !pixels) + return; + if (sz % 2) + -- it_end; + +#ifdef ENABLE_CILKPLUS + cilk_for (int ptr = 0; ptr < bufSize; ptr ++) + { + uint8_t *const p = reinterpret_cast(&pixels[ptr]); +#if SDL_BYTEORDER == SDL_BIG_ENDIAN + const int alpha = pixels[ptr] & 0xff000000; + const unsigned int data = pixels[ptr] & 0x00ffffff; +#else // SDL_BYTEORDER == SDL_BIG_ENDIAN + + const int alpha = *p & 0xff; + const unsigned int data = pixels[ptr] & 0xffffff00; +#endif // SDL_BYTEORDER == SDL_BIG_ENDIAN + +// logger->log("c:%04d %08x", c, *pixels); +// logger->log("data: %08x", data); + if (!alpha) + { +// logger->log("skip: %08x", *pixels); + } + else + { + std::vector::const_iterator it = mColors.begin(); + while (it != it_end) + { + const DyeColor &col = *it; + ++ it; + const DyeColor &col2 = *it; + +#if SDL_BYTEORDER == SDL_BIG_ENDIAN + const unsigned int coldata = (col.value[2] << 16U) + | (col.value[1] << 8U) | (col.value[0]); +#else // SDL_BYTEORDER == SDL_BIG_ENDIAN + const unsigned int coldata = (col.value[2] << 8U) + | (col.value[1] << 16U) | (col.value[0] << 24U); +#endif // SDL_BYTEORDER == SDL_BIG_ENDIAN + +// logger->log("coldata: %08x", coldata); + if (data == coldata) + { +// logger->log("correct"); + p[3] = col2.value[0]; + p[2] = col2.value[1]; + p[1] = col2.value[2]; + break; + } + ++ it; + } + } + } +#else // ENABLE_CILKPLUS + + for (const uint32_t *const p_end = pixels + CAST_SIZE(bufSize); + pixels != p_end; + ++ pixels) + { + uint8_t *const p = reinterpret_cast(pixels); +#if SDL_BYTEORDER == SDL_BIG_ENDIAN + const int alpha = *pixels & 0xff000000; + const unsigned int data = (*pixels) & 0x00ffffff; +#else // SDL_BYTEORDER == SDL_BIG_ENDIAN + + const int alpha = *p & 0xff; + const unsigned int data = (*pixels) & 0xffffff00; +#endif // SDL_BYTEORDER == SDL_BIG_ENDIAN + +// logger->log("c:%04d %08x", c, *pixels); +// logger->log("data: %08x", data); + if (!alpha) + { +// logger->log("skip: %08x", *pixels); + continue; + } + + std::vector::const_iterator it = mColors.begin(); + while (it != it_end) + { + const DyeColor &col = *it; + ++ it; + const DyeColor &col2 = *it; + +#if SDL_BYTEORDER == SDL_BIG_ENDIAN + const unsigned int coldata = (col.value[2] << 16U) + | (col.value[1] << 8U) | (col.value[0]); +#else // SDL_BYTEORDER == SDL_BIG_ENDIAN + + const unsigned int coldata = (col.value[2] << 8U) + | (col.value[1] << 16U) | (col.value[0] << 24U); +#endif // SDL_BYTEORDER == SDL_BIG_ENDIAN + +// logger->log("coldata: %08x", coldata); + if (data == coldata) + { +// logger->log("correct"); + p[3] = col2.value[0]; + p[2] = col2.value[1]; + p[1] = col2.value[2]; + break; + } + + ++ it; + } + } +#endif // ENABLE_CILKPLUS +} + +#ifdef __x86_64__ +void print256(const char *const text, const __m256i &val); +void print256(const char *const text, const __m256i &val) +{ + printf("%s 0x%016llx%016llx%016llx%016llx\n", text, val[0], val[1], val[2], val[3]); +} + +__attribute__ ((target ("avx2"))) +void DyePalette::replaceSColor256(uint32_t *restrict pixels, + const int bufSize) const restrict2 +{ + std::vector::const_iterator it_end = mColors.end(); + const size_t sz = mColors.size(); + if (!sz || !pixels) + return; + if (sz % 2) + -- it_end; + + for (const uint32_t *const p_end = pixels + CAST_SIZE(bufSize); + pixels != p_end; + pixels += 8) + { + __m256i mask = _mm256_set1_epi32(0xffffff00); + //__m256i base = _mm256_load_si256(reinterpret_cast<__m256i*>(pixels)); + __m256i base = _mm256_loadu_si256(reinterpret_cast<__m256i*>(pixels)); + //print256("mask ", mask); + + std::vector::const_iterator it = mColors.begin(); + while (it != it_end) + { + //print256("base ", base); + const DyeColor &col = *it; + ++ it; + const DyeColor &col2 = *it; + + __m256i base2 = _mm256_and_si256(mask, base); + //print256("base2 ", base2); + __m256i newMask = _mm256_set1_epi32(buildHex(col2.value[0], col2.value[1], col2.value[2], 0)); + //print256("newMask ", newMask); + __m256i cmpMask = _mm256_set1_epi32(buildHex(col.value[0], col.value[1], col.value[2], 0)); + //print256("cmpMask ", cmpMask); + __m256i cmpRes = _mm256_cmpeq_epi32(base2, cmpMask); + //print256("cmpRes ", cmpRes); + cmpRes = _mm256_and_si256(mask, cmpRes); + //print256("cmpRes ", cmpRes); + __m256i srcAnd = _mm256_andnot_si256(cmpRes, base); + //print256("srcAnd ", srcAnd); + __m256i dstAnd = _mm256_and_si256(cmpRes, newMask); + //print256("dstAnd ", dstAnd); + base = _mm256_or_si256(srcAnd, dstAnd); + ++ it; + } + //print256("res ", base); + //_mm256_store_si256(reinterpret_cast<__m256i*>(pixels), base); + _mm256_storeu_si256(reinterpret_cast<__m256i*>(pixels), base); + } +} + +#endif // __x86_64__ + +FUNCTION_SIMD_DEFAULT +void DyePalette::replaceSColor256(uint32_t *restrict pixels, + const int bufSize) const restrict2 +{ + replaceSColorDefault(pixels, bufSize); +} -- cgit v1.2.3-70-g09d2