mirror of
https://github.com/openharmony/third_party_openh264.git
synced 2026-07-19 19:13:35 -04:00
[Encoder] Add single-block SSE2 4x4 DCT/IDCT routines
We do four blocks at a time when possible, but need to handle single blocks at a time for intra prediction. ~2.31x speedup over MMX for the DCT on Haswell. ~1.92x speedup over MMX for the IDCT on Haswell.
This commit is contained in:
@@ -147,8 +147,10 @@ static void Sub8x8DctAnchor (int16_t iDct[4][4][4], uint8_t* pPix1, uint8_t* pPi
|
||||
}
|
||||
static void TestDctT4 (PDctFunc func) {
|
||||
int16_t iDctRef[4][4];
|
||||
uint8_t uiPix1[16 * FENC_STRIDE], uiPix2[16 * FDEC_STRIDE];
|
||||
int16_t iDct[16];
|
||||
CMemoryAlign cMemoryAlign (0);
|
||||
ALLOC_MEMORY (uint8_t, uiPix1, 16 * FENC_STRIDE);
|
||||
ALLOC_MEMORY (uint8_t, uiPix2, 16 * FDEC_STRIDE);
|
||||
ALLOC_MEMORY (int16_t, iDct, 16);
|
||||
for (int i = 0; i < 4; i++) {
|
||||
for (int j = 0; j < 4; j++) {
|
||||
uiPix1[i * FENC_STRIDE + j] = rand() & 255;
|
||||
@@ -160,6 +162,9 @@ static void TestDctT4 (PDctFunc func) {
|
||||
for (int i = 0; i < 4; i++)
|
||||
for (int j = 0; j < 4; j++)
|
||||
EXPECT_EQ (iDctRef[j][i], iDct[i * 4 + j]);
|
||||
FREE_MEMORY (uiPix1);
|
||||
FREE_MEMORY (uiPix2);
|
||||
FREE_MEMORY (iDct);
|
||||
}
|
||||
static void TestDctFourT4 (PDctFunc func) {
|
||||
int16_t iDctRef[4][4][4];
|
||||
@@ -195,6 +200,10 @@ TEST (EncodeMbAuxTest, WelsDctT4_mmx) {
|
||||
TestDctT4 (WelsDctT4_mmx);
|
||||
}
|
||||
|
||||
TEST (EncodeMbAuxTest, WelsDctT4_sse2) {
|
||||
TestDctT4 (WelsDctT4_sse2);
|
||||
}
|
||||
|
||||
TEST (EncodeMbAuxTest, WelsDctFourT4_sse2) {
|
||||
TestDctFourT4 (WelsDctFourT4_sse2);
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user