mirror of
https://github.com/ultravideo/uvg266.git
synced 2024-11-27 11:24:05 +00:00
[ibc] Add CRC32C functions, with SSE 4.2 optimized CRC calculations
This commit is contained in:
parent
7252befc17
commit
a32a318d18
|
@ -145,6 +145,7 @@ target_include_directories(uvg266 PUBLIC src/strategies)
|
|||
|
||||
file(GLOB LIB_SOURCES_STRATEGIES_AVX2 RELATIVE ${PROJECT_SOURCE_DIR} "src/strategies/avx2/*.c")
|
||||
file(GLOB LIB_SOURCES_STRATEGIES_SSE41 RELATIVE ${PROJECT_SOURCE_DIR} "src/strategies/sse41/*.c")
|
||||
file(GLOB LIB_SOURCES_STRATEGIES_SSE42 RELATIVE ${PROJECT_SOURCE_DIR} "src/strategies/sse42/*.c")
|
||||
|
||||
set(CLI_SOURCES "src/encmain.c" "src/cli.c" "src/cli.h" "src/yuv_io.c" "src/yuv_io.h")
|
||||
|
||||
|
@ -175,7 +176,8 @@ else()
|
|||
list(APPEND ALLOW_AVX2 "x86_64" "AMD64")
|
||||
if(${CMAKE_SYSTEM_PROCESSOR} IN_LIST ALLOW_AVX2)
|
||||
set_property( SOURCE ${LIB_SOURCES_STRATEGIES_AVX2} APPEND PROPERTY COMPILE_FLAGS "-mavx2 -mbmi -mpopcnt -mlzcnt -mbmi2" )
|
||||
set_property( SOURCE ${LIB_SOURCES_STRATEGIES_SSE41} APPEND PROPERTY COMPILE_FLAGS "-msse4.1" )
|
||||
set_property( SOURCE ${LIB_SOURCES_STRATEGIES_SSE41} APPEND PROPERTY COMPILE_FLAGS "-msse4.1" )
|
||||
set_property( SOURCE ${LIB_SOURCES_STRATEGIES_SSE42} APPEND PROPERTY COMPILE_FLAGS "-msse4.2" )
|
||||
endif()
|
||||
set(THREADS_PREFER_PTHREAD_FLAG ON)
|
||||
find_package(Threads REQUIRED)
|
||||
|
|
|
@ -34,6 +34,11 @@
|
|||
#include <stdlib.h>
|
||||
#include <stdint.h>
|
||||
|
||||
// The ratio of the hashmap bucket size to the maximum number of elements
|
||||
#define UVG_HASHMAP_RATIO 0.35
|
||||
// Use Hashmap for 4x4 blocks
|
||||
#define UVG_HASHMAP_BLOCKSIZE 4
|
||||
|
||||
typedef struct uvg_hashmap_node {
|
||||
uint32_t key;
|
||||
uint32_t value;
|
||||
|
|
17
src/inter.c
17
src/inter.c
|
@ -1666,6 +1666,13 @@ void uvg_inter_get_mv_cand_cua(const encoder_state_t * const state,
|
|||
uvg_round_precision(INTERNAL_MV_PREC, 2, &mv_cand[1][0], &mv_cand[1][1]);
|
||||
}
|
||||
|
||||
/**
|
||||
• \brief Checks if two CUs have similar motion vectors. The function takes two CUs and compares their motion vectors.
|
||||
• \param cu1 first CU
|
||||
• \param cu2 second CU
|
||||
• \return returns 0 if the two CUs have dissimilar motion vectors, and 1 if the motions are similar.
|
||||
*/
|
||||
|
||||
static bool is_duplicate_candidate(const cu_info_t* cu1, const cu_info_t* cu2)
|
||||
{
|
||||
if (!cu2) return false;
|
||||
|
@ -1684,6 +1691,16 @@ static bool is_duplicate_candidate(const cu_info_t* cu1, const cu_info_t* cu2)
|
|||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Adds a merge candidate to the list of possible candidates, if it is not a duplicate.
|
||||
*
|
||||
* \param cand The candidate to be added.
|
||||
* \param possible_duplicate1 The first possible duplicate candidate to check for duplication.
|
||||
* \param possible_duplicate2 The second possible duplicate candidate to check for duplication.
|
||||
* \param merge_cand_out The output parameter to store the merge candidate information.
|
||||
*
|
||||
* @return Returns true if the merge candidate was added successfully, false otherwise.
|
||||
*/
|
||||
static bool add_merge_candidate(const cu_info_t *cand,
|
||||
const cu_info_t *possible_duplicate1,
|
||||
const cu_info_t *possible_duplicate2,
|
||||
|
|
|
@ -793,9 +793,51 @@ static void generate_residual_generic(const uvg_pixel* ref_in, const uvg_pixel*
|
|||
}
|
||||
}
|
||||
|
||||
INLINE static uint32_t uvg_crc32c_4_generic(uint32_t crc, const uvg_pixel *buf)
|
||||
{
|
||||
crc = (crc >> 8) ^ uvg_crc_table[(crc ^ buf[0]) & 0xFF];
|
||||
crc = (crc >> 8) ^ uvg_crc_table[(crc ^ buf[1]) & 0xFF];
|
||||
crc = (crc >> 8) ^ uvg_crc_table[(crc ^ buf[2]) & 0xFF];
|
||||
crc = (crc >> 8) ^ uvg_crc_table[(crc ^ buf[3]) & 0xFF];
|
||||
return crc;
|
||||
}
|
||||
|
||||
static uint32_t uvg_crc32c_4x4_8bit_generic(const uvg_pixel *buf, uint32_t pic_stride)
|
||||
{
|
||||
uint32_t crc = 0xFFFFFFFF;
|
||||
crc = uvg_crc32c_4_generic(crc, &buf[0 * pic_stride]);
|
||||
crc = uvg_crc32c_4_generic(crc, &buf[1 * pic_stride]);
|
||||
crc = uvg_crc32c_4_generic(crc, &buf[2 * pic_stride]);
|
||||
crc = uvg_crc32c_4_generic(crc, &buf[3 * pic_stride]);
|
||||
return crc ^ 0xFFFFFFFF;
|
||||
}
|
||||
|
||||
static uint32_t uvg_crc32c_4x4_16bit_generic(const uvg_pixel *buf, uint32_t pic_stride)
|
||||
{
|
||||
uint32_t crc = 0xFFFFFFFF;
|
||||
crc = uvg_crc32c_4_generic(crc, &buf[0 * pic_stride]);
|
||||
crc = uvg_crc32c_4_generic(crc, &buf[0 * pic_stride] + 4);
|
||||
|
||||
crc = uvg_crc32c_4_generic(crc, &buf[1 * pic_stride]);
|
||||
crc = uvg_crc32c_4_generic(crc, &buf[1 * pic_stride] + 4);
|
||||
|
||||
crc = uvg_crc32c_4_generic(crc, &buf[2 * pic_stride]);
|
||||
crc = uvg_crc32c_4_generic(crc, &buf[2 * pic_stride] + 4);
|
||||
|
||||
crc = uvg_crc32c_4_generic(crc, &buf[3 * pic_stride]);
|
||||
crc = uvg_crc32c_4_generic(crc, &buf[3 * pic_stride] + 4);
|
||||
return crc ^ 0xFFFFFFFF;
|
||||
}
|
||||
|
||||
int uvg_strategy_register_picture_generic(void* opaque, uint8_t bitdepth)
|
||||
{
|
||||
bool success = true;
|
||||
if (bitdepth == 8) {
|
||||
success &= uvg_strategyselector_register(opaque, "uvg_crc32c_4x4", "generic", 0, &uvg_crc32c_4x4_8bit_generic);
|
||||
} else {
|
||||
success &= uvg_strategyselector_register(opaque, "uvg_crc32c_4x4", "generic", 0, &uvg_crc32c_4x4_16bit_generic);
|
||||
}
|
||||
|
||||
|
||||
success &= uvg_strategyselector_register(opaque, "reg_sad", "generic", 0, ®_sad_generic);
|
||||
|
||||
|
|
80
src/strategies/sse42/picture-sse42.c
Normal file
80
src/strategies/sse42/picture-sse42.c
Normal file
|
@ -0,0 +1,80 @@
|
|||
/*****************************************************************************
|
||||
* This file is part of uvg266 VVC encoder.
|
||||
*
|
||||
* Copyright (c) 2023, Tampere University, ITU/ISO/IEC, project contributors
|
||||
* All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without modification,
|
||||
* are permitted provided that the following conditions are met:
|
||||
*
|
||||
* * Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* * Redistributions in binary form must reproduce the above copyright notice, this
|
||||
* list of conditions and the following disclaimer in the documentation and/or
|
||||
* other materials provided with the distribution.
|
||||
*
|
||||
* * Neither the name of the Tampere University or ITU/ISO/IEC nor the names of its
|
||||
* contributors may be used to endorse or promote products derived from
|
||||
* this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
|
||||
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
|
||||
* WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE FOR
|
||||
* ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
|
||||
* INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
|
||||
* LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION HOWEVER CAUSED AND ON
|
||||
* ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
* SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
* INCLUDING NEGLIGENCE OR OTHERWISE ARISING IN ANY WAY OUT OF THE USE OF THIS
|
||||
****************************************************************************/
|
||||
|
||||
#include "global.h"
|
||||
|
||||
#if COMPILE_INTEL_SSE42
|
||||
#include "uvg266.h"
|
||||
|
||||
#include "strategies/sse42/picture-sse42.h"
|
||||
|
||||
#include <immintrin.h>
|
||||
#include <stdlib.h>
|
||||
|
||||
#include "strategyselector.h"
|
||||
|
||||
|
||||
|
||||
static uint32_t uvg_crc32c_4x4_8bit_sse42(const uvg_pixel *buf, uint32_t pic_stride)
|
||||
{
|
||||
uint32_t crc = 0xFFFFFFFF;
|
||||
crc = _mm_crc32_u32(crc, *((uint32_t *)&buf[0 * pic_stride]));
|
||||
crc = _mm_crc32_u32(crc, *((uint32_t *)&buf[1 * pic_stride]));
|
||||
crc = _mm_crc32_u32(crc, *((uint32_t *)&buf[2 * pic_stride]));
|
||||
crc = _mm_crc32_u32(crc, *((uint32_t *)&buf[3 * pic_stride]));
|
||||
return crc ^ 0xFFFFFFFF;
|
||||
}
|
||||
|
||||
static uint32_t uvg_crc32c_4x4_16bit_sse42(const uvg_pixel *buf, uint32_t pic_stride)
|
||||
{
|
||||
uint32_t crc = 0xFFFFFFFF;
|
||||
crc = _mm_crc32_u64(crc, *((uint32_t *)&buf[0 * pic_stride]));
|
||||
crc = _mm_crc32_u64(crc, *((uint32_t *)&buf[1 * pic_stride]));
|
||||
crc = _mm_crc32_u64(crc, *((uint32_t *)&buf[2 * pic_stride]));
|
||||
crc = _mm_crc32_u64(crc, *((uint32_t *)&buf[3 * pic_stride]));
|
||||
return crc ^ 0xFFFFFFFF;
|
||||
}
|
||||
|
||||
|
||||
#endif //COMPILE_INTEL_SSE42
|
||||
|
||||
int uvg_strategy_register_picture_sse41(void* opaque, uint8_t bitdepth) {
|
||||
bool success = true;
|
||||
#if COMPILE_INTEL_SSE42
|
||||
if (bitdepth == 8){
|
||||
success &= uvg_strategyselector_register(opaque, "uvg_crc32c_4x4", "sse42", 0, &uvg_crc32c_4x4_8bit_sse42);
|
||||
} else {
|
||||
success &= uvg_strategyselector_register(opaque, "uvg_crc32c_4x4", "sse42", 0, &uvg_crc32c_4x4_16bit_sse42);
|
||||
}
|
||||
#endif
|
||||
return success;
|
||||
}
|
45
src/strategies/sse42/picture-sse42.h
Normal file
45
src/strategies/sse42/picture-sse42.h
Normal file
|
@ -0,0 +1,45 @@
|
|||
#pragma once
|
||||
|
||||
/*****************************************************************************
|
||||
* This file is part of uvg266 VVC encoder.
|
||||
*
|
||||
* Copyright (c) 2022, Tampere University, ITU/ISO/IEC, project contributors
|
||||
* All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without modification,
|
||||
* are permitted provided that the following conditions are met:
|
||||
*
|
||||
* * Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* * Redistributions in binary form must reproduce the above copyright notice, this
|
||||
* list of conditions and the following disclaimer in the documentation and/or
|
||||
* other materials provided with the distribution.
|
||||
*
|
||||
* * Neither the name of the Tampere University or ITU/ISO/IEC nor the names of its
|
||||
* contributors may be used to endorse or promote products derived from
|
||||
* this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
|
||||
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
|
||||
* WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE FOR
|
||||
* ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
|
||||
* INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
|
||||
* LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION HOWEVER CAUSED AND ON
|
||||
* ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
* SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
* INCLUDING NEGLIGENCE OR OTHERWISE ARISING IN ANY WAY OUT OF THE USE OF THIS
|
||||
****************************************************************************/
|
||||
|
||||
/**
|
||||
* \ingroup Optimization
|
||||
* \file
|
||||
* Optimizations for SSE4.2.
|
||||
*/
|
||||
|
||||
#include "global.h" // IWYU pragma: keep
|
||||
#include "uvg266.h"
|
||||
|
||||
|
||||
int uvg_strategy_register_picture_sse42(void* opaque, uint8_t bitdepth);
|
|
@ -41,6 +41,7 @@
|
|||
|
||||
|
||||
// Define function pointers.
|
||||
crc32c_4x4_func * uvg_crc32c_4x4;
|
||||
reg_sad_func * uvg_reg_sad = 0;
|
||||
|
||||
cost_pixel_nxn_func * uvg_sad_4x4 = 0;
|
||||
|
@ -83,6 +84,8 @@ pixel_var_func *uvg_pixel_var = 0;
|
|||
generate_residual_func *uvg_generate_residual = 0;
|
||||
|
||||
|
||||
|
||||
|
||||
int uvg_strategy_register_picture(void* opaque, uint8_t bitdepth) {
|
||||
bool success = true;
|
||||
|
||||
|
@ -206,3 +209,50 @@ cost_pixel_nxn_multi_func * uvg_pixels_get_sad_dual_func(unsigned n)
|
|||
return NULL;
|
||||
}
|
||||
}
|
||||
|
||||
// Precomputed CRC32C lookup table for polynomial 0x04C11DB7
|
||||
const uint32_t uvg_crc_table[256] = {
|
||||
0x00000000, 0xf26b8303, 0xe13b70f7, 0x1350f3f4, 0xc79a971f, 0x35f1141c,
|
||||
0x26a1e7e8, 0xd4ca64eb, 0x8ad958cf, 0x78b2dbcc, 0x6be22838, 0x9989ab3b,
|
||||
0x4d43cfd0, 0xbf284cd3, 0xac78bf27, 0x5e133c24, 0x105ec76f, 0xe235446c,
|
||||
0xf165b798, 0x030e349b, 0xd7c45070, 0x25afd373, 0x36ff2087, 0xc494a384,
|
||||
0x9a879fa0, 0x68ec1ca3, 0x7bbcef57, 0x89d76c54, 0x5d1d08bf, 0xaf768bbc,
|
||||
0xbc267848, 0x4e4dfb4b, 0x20bd8ede, 0xd2d60ddd, 0xc186fe29, 0x33ed7d2a,
|
||||
0xe72719c1, 0x154c9ac2, 0x061c6936, 0xf477ea35, 0xaa64d611, 0x580f5512,
|
||||
0x4b5fa6e6, 0xb93425e5, 0x6dfe410e, 0x9f95c20d, 0x8cc531f9, 0x7eaeb2fa,
|
||||
0x30e349b1, 0xc288cab2, 0xd1d83946, 0x23b3ba45, 0xf779deae, 0x05125dad,
|
||||
0x1642ae59, 0xe4292d5a, 0xba3a117e, 0x4851927d, 0x5b016189, 0xa96ae28a,
|
||||
0x7da08661, 0x8fcb0562, 0x9c9bf696, 0x6ef07595, 0x417b1dbc, 0xb3109ebf,
|
||||
0xa0406d4b, 0x522bee48, 0x86e18aa3, 0x748a09a0, 0x67dafa54, 0x95b17957,
|
||||
0xcba24573, 0x39c9c670, 0x2a993584, 0xd8f2b687, 0x0c38d26c, 0xfe53516f,
|
||||
0xed03a29b, 0x1f682198, 0x5125dad3, 0xa34e59d0, 0xb01eaa24, 0x42752927,
|
||||
0x96bf4dcc, 0x64d4cecf, 0x77843d3b, 0x85efbe38, 0xdbfc821c, 0x2997011f,
|
||||
0x3ac7f2eb, 0xc8ac71e8, 0x1c661503, 0xee0d9600, 0xfd5d65f4, 0x0f36e6f7,
|
||||
0x61c69362, 0x93ad1061, 0x80fde395, 0x72966096, 0xa65c047d, 0x5437877e,
|
||||
0x4767748a, 0xb50cf789, 0xeb1fcbad, 0x197448ae, 0x0a24bb5a, 0xf84f3859,
|
||||
0x2c855cb2, 0xdeeedfb1, 0xcdbe2c45, 0x3fd5af46, 0x7198540d, 0x83f3d70e,
|
||||
0x90a324fa, 0x62c8a7f9, 0xb602c312, 0x44694011, 0x5739b3e5, 0xa55230e6,
|
||||
0xfb410cc2, 0x092a8fc1, 0x1a7a7c35, 0xe811ff36, 0x3cdb9bdd, 0xceb018de,
|
||||
0xdde0eb2a, 0x2f8b6829, 0x82f63b78, 0x709db87b, 0x63cd4b8f, 0x91a6c88c,
|
||||
0x456cac67, 0xb7072f64, 0xa457dc90, 0x563c5f93, 0x082f63b7, 0xfa44e0b4,
|
||||
0xe9141340, 0x1b7f9043, 0xcfb5f4a8, 0x3dde77ab, 0x2e8e845f, 0xdce5075c,
|
||||
0x92a8fc17, 0x60c37f14, 0x73938ce0, 0x81f80fe3, 0x55326b08, 0xa759e80b,
|
||||
0xb4091bff, 0x466298fc, 0x1871a4d8, 0xea1a27db, 0xf94ad42f, 0x0b21572c,
|
||||
0xdfeb33c7, 0x2d80b0c4, 0x3ed04330, 0xccbbc033, 0xa24bb5a6, 0x502036a5,
|
||||
0x4370c551, 0xb11b4652, 0x65d122b9, 0x97baa1ba, 0x84ea524e, 0x7681d14d,
|
||||
0x2892ed69, 0xdaf96e6a, 0xc9a99d9e, 0x3bc21e9d, 0xef087a76, 0x1d63f975,
|
||||
0x0e330a81, 0xfc588982, 0xb21572c9, 0x407ef1ca, 0x532e023e, 0xa145813d,
|
||||
0x758fe5d6, 0x87e466d5, 0x94b49521, 0x66df1622, 0x38cc2a06, 0xcaa7a905,
|
||||
0xd9f75af1, 0x2b9cd9f2, 0xff56bd19, 0x0d3d3e1a, 0x1e6dcdee, 0xec064eed,
|
||||
0xc38d26c4, 0x31e6a5c7, 0x22b65633, 0xd0ddd530, 0x0417b1db, 0xf67c32d8,
|
||||
0xe52cc12c, 0x1747422f, 0x49547e0b, 0xbb3ffd08, 0xa86f0efc, 0x5a048dff,
|
||||
0x8ecee914, 0x7ca56a17, 0x6ff599e3, 0x9d9e1ae0, 0xd3d3e1ab, 0x21b862a8,
|
||||
0x32e8915c, 0xc083125f, 0x144976b4, 0xe622f5b7, 0xf5720643, 0x07198540,
|
||||
0x590ab964, 0xab613a67, 0xb831c993, 0x4a5a4a90, 0x9e902e7b, 0x6cfbad78,
|
||||
0x7fab5e8c, 0x8dc0dd8f, 0xe330a81a, 0x115b2b19, 0x020bd8ed, 0xf0605bee,
|
||||
0x24aa3f05, 0xd6c1bc06, 0xc5914ff2, 0x37faccf1, 0x69e9f0d5, 0x9b8273d6,
|
||||
0x88d28022, 0x7ab90321, 0xae7367ca, 0x5c18e4c9, 0x4f48173d, 0xbd23943e,
|
||||
0xf36e6f75, 0x0105ec76, 0x12551f82, 0xe03e9c81, 0x34f4f86a, 0xc69f7b69,
|
||||
0xd5cf889d, 0x27a40b9e, 0x79b737ba, 0x8bdcb4b9, 0x988c474d, 0x6ae7c44e,
|
||||
0xbe2da0a5, 0x4c4623a6, 0x5f16d052, 0xad7d5351,
|
||||
};
|
|
@ -151,7 +151,14 @@ typedef double (pixel_var_func)(const uvg_pixel *buf, const uint32_t len);
|
|||
|
||||
typedef void (generate_residual_func)(const uvg_pixel* ref_in, const uvg_pixel* pred_in, int16_t* residual, int width, int ref_stride, int pred_stride);
|
||||
|
||||
|
||||
extern const uint32_t uvg_crc_table[256];
|
||||
|
||||
typedef uint32_t(crc32c_4x4_func)(const uvg_pixel *buf, uint32_t pic_stride);
|
||||
|
||||
// Declare function pointers.
|
||||
extern crc32c_4x4_func * uvg_crc32c_4x4;
|
||||
|
||||
extern reg_sad_func * uvg_reg_sad;
|
||||
|
||||
extern cost_pixel_nxn_func * uvg_sad_4x4;
|
||||
|
|
Loading…
Reference in a new issue