From d0dd01b8ce8bc5f477d70f1c127d795418c5efb5 Mon Sep 17 00:00:00 2001 From: Yaowu Xu Date: Wed, 16 Jun 2010 12:52:18 -0700 Subject: Redo the forward 4x4 dct The new fdct lowers the round trip sum squared error for a 4x4 block ~0.12. or ~0.008/pixel. For reference, the old matrix multiply version has average round trip error 1.46 for a 4x4 block. Thanks to "derf" for his suggestions and references. Change-Id: I5559d1e81d333b319404ab16b336b739f87afc79 --- vp8/encoder/x86/csystemdependent.c | 26 ++- vp8/encoder/x86/dct_mmx.asm | 392 +++------------------------------ vp8/encoder/x86/dct_x86.h | 13 +- vp8/encoder/x86/x86_csystemdependent.c | 34 ++- 4 files changed, 71 insertions(+), 394 deletions(-) (limited to 'vp8/encoder/x86') diff --git a/vp8/encoder/x86/csystemdependent.c b/vp8/encoder/x86/csystemdependent.c index 6aeac508f..bf12fee54 100644 --- a/vp8/encoder/x86/csystemdependent.c +++ b/vp8/encoder/x86/csystemdependent.c @@ -181,10 +181,17 @@ void vp8_cmachine_specific_config(void) // Willamette instruction set available: vp8_mbuverror = vp8_mbuverror_xmm; vp8_fast_quantize_b = vp8_fast_quantize_b_sse; +#if 0 //new fdct vp8_short_fdct4x4 = vp8_short_fdct4x4_mmx; vp8_short_fdct8x4 = vp8_short_fdct8x4_mmx; - vp8_fast_fdct4x4 = vp8_fast_fdct4x4_mmx; - vp8_fast_fdct8x4 = vp8_fast_fdct8x4_wmt; + vp8_fast_fdct4x4 = vp8_short_fdct4x4_mmx; + vp8_fast_fdct8x4 = vp8_short_fdct8x4_wmt; +#else + vp8_short_fdct4x4 = vp8_short_fdct4x4_c; + vp8_short_fdct8x4 = vp8_short_fdct8x4_c; + vp8_fast_fdct4x4 = vp8_short_fdct4x4_c; + vp8_fast_fdct8x4 = vp8_fast_fdct8x4_c; +#endif vp8_subtract_b = vp8_subtract_b_mmx; vp8_subtract_mbuv = vp8_subtract_mbuv_mmx; vp8_variance4x4 = vp8_variance4x4_mmx; @@ -218,10 +225,17 @@ void vp8_cmachine_specific_config(void) // MMX instruction set available: vp8_mbuverror = vp8_mbuverror_mmx; vp8_fast_quantize_b = vp8_fast_quantize_b_mmx; +#if 0 // new fdct vp8_short_fdct4x4 = vp8_short_fdct4x4_mmx; vp8_short_fdct8x4 = vp8_short_fdct8x4_mmx; - vp8_fast_fdct4x4 = vp8_fast_fdct4x4_mmx; - vp8_fast_fdct8x4 = vp8_fast_fdct8x4_mmx; + vp8_fast_fdct4x4 = vp8_short_fdct4x4_mmx; + vp8_fast_fdct8x4 = vp8_short_fdct8x4_mmx; +#else + vp8_short_fdct4x4 = vp8_short_fdct4x4_c; + vp8_short_fdct8x4 = vp8_short_fdct8x4_c; + vp8_fast_fdct4x4 = vp8_short_fdct4x4_c; + vp8_fast_fdct8x4 = vp8_fast_fdct8x4_c; +#endif vp8_subtract_b = vp8_subtract_b_mmx; vp8_subtract_mbuv = vp8_subtract_mbuv_mmx; vp8_variance4x4 = vp8_variance4x4_mmx; @@ -254,10 +268,10 @@ void vp8_cmachine_specific_config(void) { // Pure C: vp8_mbuverror = vp8_mbuverror_c; - vp8_fast_quantize_b = vp8_fast_quantize_b_c; + vp8_fast_quantize_b = vp8_fast_quantize_b_c; vp8_short_fdct4x4 = vp8_short_fdct4x4_c; vp8_short_fdct8x4 = vp8_short_fdct8x4_c; - vp8_fast_fdct4x4 = vp8_fast_fdct4x4_c; + vp8_fast_fdct4x4 = vp8_short_fdct4x4_c; vp8_fast_fdct8x4 = vp8_fast_fdct8x4_c; vp8_subtract_b = vp8_subtract_b_c; vp8_subtract_mbuv = vp8_subtract_mbuv_c; diff --git a/vp8/encoder/x86/dct_mmx.asm b/vp8/encoder/x86/dct_mmx.asm index 32d6610aa..ff96c49f3 100644 --- a/vp8/encoder/x86/dct_mmx.asm +++ b/vp8/encoder/x86/dct_mmx.asm @@ -13,8 +13,7 @@ section .text global sym(vp8_short_fdct4x4_mmx) - global sym(vp8_fast_fdct4x4_mmx) - global sym(vp8_fast_fdct8x4_wmt) + global sym(vp8_short_fdct8x4_wmt) %define DCTCONSTANTSBITS (16) @@ -24,339 +23,8 @@ section .text %define x_c3 (25080) ; cos(pi*3/8) * (1<<15) -%define _1STSTAGESHIFT 14 -%define _2NDSTAGESHIFT 16 - -; using matrix multiply with source and destbuffer has a pitch ;void vp8_short_fdct4x4_mmx(short *input, short *output, int pitch) sym(vp8_short_fdct4x4_mmx): - push rbp - mov rbp, rsp - SHADOW_ARGS_TO_STACK 3 - GET_GOT rbx - push rsi - push rdi - ; end prolog - - mov rsi, arg(0) ;input - mov rdi, arg(1) ;output - - movsxd rax, dword ptr arg(2) ;pitch - lea rdx, [dct_matrix GLOBAL] - - movq mm0, [rsi ] - movq mm1, [rsi + rax] - - movq mm2, [rsi + rax*2] - lea rsi, [rsi + rax*2] - - movq mm3, [rsi + rax] - - ; first column - movq mm4, mm0 - movq mm7, [rdx] - - pmaddwd mm4, mm7 - movq mm5, mm1 - - pmaddwd mm5, mm7 - movq mm6, mm4 - - punpckldq mm4, mm5 - punpckhdq mm6, mm5 - - paddd mm4, mm6 - movq mm5, mm2 - - - pmaddwd mm5, mm7 - movq mm6, mm3 - - pmaddwd mm6, mm7 - movq mm7, mm5 - - punpckldq mm5, mm6 - punpckhdq mm7, mm6 - - paddd mm5, mm7 - movq mm6, [dct1st_stage_rounding_mmx GLOBAL] - - paddd mm4, mm6 - paddd mm5, mm6 - - psrad mm4, _1STSTAGESHIFT - psrad mm5, _1STSTAGESHIFT - - packssdw mm4, mm5 - movq [rdi], mm4 - - ;second column - movq mm4, mm0 - - pmaddwd mm4, [rdx+8] - movq mm5, mm1 - - pmaddwd mm5, [rdx+8] - movq mm6, mm4 - - punpckldq mm4, mm5 - punpckhdq mm6, mm5 - - paddd mm4, mm6 - movq mm5, mm2 - - pmaddwd mm5, [rdx+8] - movq mm6, mm3 - - pmaddwd mm6, [rdx+8] - movq mm7, mm5 - - punpckldq mm5, mm6 - punpckhdq mm7, mm6 - - paddd mm5, mm7 - movq mm6, [dct1st_stage_rounding_mmx GLOBAL] - - paddd mm4, mm6 - paddd mm5, mm6 - - psrad mm4, _1STSTAGESHIFT - psrad mm5, _1STSTAGESHIFT - - packssdw mm4, mm5 - movq [rdi+8], mm4 - - - ;third column - movq mm4, mm0 - - pmaddwd mm4, [rdx+16] - movq mm5, mm1 - - pmaddwd mm5, [rdx+16] - movq mm6, mm4 - - punpckldq mm4, mm5 - punpckhdq mm6, mm5 - - paddd mm4, mm6 - movq mm5, mm2 - - pmaddwd mm5, [rdx+16] - movq mm6, mm3 - - pmaddwd mm6, [rdx+16] - movq mm7, mm5 - - punpckldq mm5, mm6 - punpckhdq mm7, mm6 - - paddd mm5, mm7 - movq mm6, [dct1st_stage_rounding_mmx GLOBAL] - - paddd mm4, mm6 - paddd mm5, mm6 - - psrad mm4, _1STSTAGESHIFT - psrad mm5, _1STSTAGESHIFT - - packssdw mm4, mm5 - movq [rdi+16], mm4 - - ;fourth column (this is the last column, so we do not have save the source any more) - - pmaddwd mm0, [rdx+24] - - pmaddwd mm1, [rdx+24] - movq mm6, mm0 - - punpckldq mm0, mm1 - punpckhdq mm6, mm1 - - paddd mm0, mm6 - - pmaddwd mm2, [rdx+24] - - pmaddwd mm3, [rdx+24] - movq mm7, mm2 - - punpckldq mm2, mm3 - punpckhdq mm7, mm3 - - paddd mm2, mm7 - movq mm6, [dct1st_stage_rounding_mmx GLOBAL] - - paddd mm0, mm6 - paddd mm2, mm6 - - psrad mm0, _1STSTAGESHIFT - psrad mm2, _1STSTAGESHIFT - - packssdw mm0, mm2 - - movq mm3, mm0 - - ; done with one pass - ; now start second pass - movq mm0, [rdi ] - movq mm1, [rdi+ 8] - movq mm2, [rdi+ 16] - - movq mm4, mm0 - - pmaddwd mm4, [rdx] - movq mm5, mm1 - - pmaddwd mm5, [rdx] - movq mm6, mm4 - - punpckldq mm4, mm5 - punpckhdq mm6, mm5 - - paddd mm4, mm6 - movq mm5, mm2 - - pmaddwd mm5, [rdx] - movq mm6, mm3 - - pmaddwd mm6, [rdx] - movq mm7, mm5 - - punpckldq mm5, mm6 - punpckhdq mm7, mm6 - - paddd mm5, mm7 - movq mm6, [dct2nd_stage_rounding_mmx GLOBAL] - - paddd mm4, mm6 - paddd mm5, mm6 - - psrad mm4, _2NDSTAGESHIFT - psrad mm5, _2NDSTAGESHIFT - - packssdw mm4, mm5 - movq [rdi], mm4 - - ;second column - movq mm4, mm0 - - pmaddwd mm4, [rdx+8] - movq mm5, mm1 - - pmaddwd mm5, [rdx+8] - movq mm6, mm4 - - punpckldq mm4, mm5 - punpckhdq mm6, mm5 - - paddd mm4, mm6 - movq mm5, mm2 - - pmaddwd mm5, [rdx+8] - movq mm6, mm3 - - pmaddwd mm6, [rdx+8] - movq mm7, mm5 - - punpckldq mm5, mm6 - punpckhdq mm7, mm6 - - paddd mm5, mm7 - movq mm6, [dct2nd_stage_rounding_mmx GLOBAL] - - paddd mm4, mm6 - paddd mm5, mm6 - - psrad mm4, _2NDSTAGESHIFT - psrad mm5, _2NDSTAGESHIFT - - packssdw mm4, mm5 - movq [rdi+8], mm4 - - - ;third column - movq mm4, mm0 - - pmaddwd mm4, [rdx+16] - movq mm5, mm1 - - pmaddwd mm5, [rdx+16] - movq mm6, mm4 - - punpckldq mm4, mm5 - punpckhdq mm6, mm5 - - paddd mm4, mm6 - movq mm5, mm2 - - pmaddwd mm5, [rdx+16] - movq mm6, mm3 - - pmaddwd mm6, [rdx+16] - movq mm7, mm5 - - punpckldq mm5, mm6 - punpckhdq mm7, mm6 - - paddd mm5, mm7 - movq mm6, [dct2nd_stage_rounding_mmx GLOBAL] - - paddd mm4, mm6 - paddd mm5, mm6 - - psrad mm4, _2NDSTAGESHIFT - psrad mm5, _2NDSTAGESHIFT - - packssdw mm4, mm5 - movq [rdi+16], mm4 - - ;fourth column - movq mm4, mm0 - - pmaddwd mm4, [rdx+24] - movq mm5, mm1 - - pmaddwd mm5, [rdx+24] - movq mm6, mm4 - - punpckldq mm4, mm5 - punpckhdq mm6, mm5 - - paddd mm4, mm6 - movq mm5, mm2 - - pmaddwd mm5, [rdx+24] - movq mm6, mm3 - - pmaddwd mm6, [rdx+24] - movq mm7, mm5 - - punpckldq mm5, mm6 - punpckhdq mm7, mm6 - - paddd mm5, mm7 - movq mm6, [dct2nd_stage_rounding_mmx GLOBAL] - - paddd mm4, mm6 - paddd mm5, mm6 - - psrad mm4, _2NDSTAGESHIFT - psrad mm5, _2NDSTAGESHIFT - - packssdw mm4, mm5 - movq [rdi+24], mm4 - - ; begin epilog - pop rdi - pop rsi - RESTORE_GOT - UNSHADOW_ARGS - pop rbp - ret - - -;void vp8_fast_fdct4x4_mmx(short *input, short *output, int pitch) -sym(vp8_fast_fdct4x4_mmx): push rbp mov rbp, rsp SHADOW_ARGS_TO_STACK 3 @@ -379,11 +47,11 @@ sym(vp8_fast_fdct4x4_mmx): movq mm3, [rcx + rax] ; get the constants ;shift to left by 1 for prescision - paddw mm0, mm0 - paddw mm1, mm1 + psllw mm0, 3 + psllw mm1, 3 - psllw mm2, 1 - psllw mm3, 1 + psllw mm2, 3 + psllw mm3, 3 ; transpose for the second stage movq mm4, mm0 ; 00 01 02 03 @@ -531,20 +199,23 @@ sym(vp8_fast_fdct4x4_mmx): movq mm3, mm5 ; done with vertical - pcmpeqw mm4, mm4 - pcmpeqw mm5, mm5 - psrlw mm4, 15 - psrlw mm5, 15 + pcmpeqw mm4, mm4 + pcmpeqw mm5, mm5 + psrlw mm4, 15 + psrlw mm5, 15 + + psllw mm4, 2 + psllw mm5, 2 paddw mm0, mm4 paddw mm1, mm5 paddw mm2, mm4 paddw mm3, mm5 - psraw mm0, 1 - psraw mm1, 1 - psraw mm2, 1 - psraw mm3, 1 + psraw mm0, 3 + psraw mm1, 3 + psraw mm2, 3 + psraw mm3, 3 movq [rdi ], mm0 movq [rdi+ 8], mm1 @@ -560,8 +231,8 @@ sym(vp8_fast_fdct4x4_mmx): ret -;void vp8_fast_fdct8x4_wmt(short *input, short *output, int pitch) -sym(vp8_fast_fdct8x4_wmt): +;void vp8_short_fdct8x4_wmt(short *input, short *output, int pitch) +sym(vp8_short_fdct8x4_wmt): push rbp mov rbp, rsp SHADOW_ARGS_TO_STACK 3 @@ -584,11 +255,11 @@ sym(vp8_fast_fdct8x4_wmt): movdqa xmm3, [rcx + rax] ; get the constants ;shift to left by 1 for prescision - psllw xmm0, 1 - psllw xmm2, 1 + psllw xmm0, 3 + psllw xmm2, 3 - psllw xmm4, 1 - psllw xmm3, 1 + psllw xmm4, 3 + psllw xmm3, 3 ; transpose for the second stage movdqa xmm1, xmm0 ; 00 01 02 03 04 05 06 07 @@ -758,20 +429,23 @@ sym(vp8_fast_fdct8x4_wmt): ; done with vertical - pcmpeqw xmm4, xmm4 - pcmpeqw xmm5, xmm5; - psrlw xmm4, 15 - psrlw xmm5, 15 + pcmpeqw xmm4, xmm4 + pcmpeqw xmm5, xmm5; + psrlw xmm4, 15 + psrlw xmm5, 15 + + psllw xmm4, 2 + psllw xmm5, 2 paddw xmm0, xmm4 paddw xmm1, xmm5 paddw xmm2, xmm4 paddw xmm3, xmm5 - psraw xmm0, 1 - psraw xmm1, 1 - psraw xmm2, 1 - psraw xmm3, 1 + psraw xmm0, 3 + psraw xmm1, 3 + psraw xmm2, 3 + psraw xmm3, 3 movq QWORD PTR[rdi ], xmm0 movq QWORD PTR[rdi+ 8], xmm1 diff --git a/vp8/encoder/x86/dct_x86.h b/vp8/encoder/x86/dct_x86.h index 05d018043..ada16d34f 100644 --- a/vp8/encoder/x86/dct_x86.h +++ b/vp8/encoder/x86/dct_x86.h @@ -22,31 +22,22 @@ #if HAVE_MMX extern prototype_fdct(vp8_short_fdct4x4_mmx); extern prototype_fdct(vp8_short_fdct8x4_mmx); -extern prototype_fdct(vp8_fast_fdct4x4_mmx); -extern prototype_fdct(vp8_fast_fdct8x4_mmx); #if !CONFIG_RUNTIME_CPU_DETECT +#if 0 new c version, #undef vp8_fdct_short4x4 #define vp8_fdct_short4x4 vp8_short_fdct4x4_mmx #undef vp8_fdct_short8x4 #define vp8_fdct_short8x4 vp8_short_fdct8x4_mmx - -#undef vp8_fdct_fast4x4 -#define vp8_fdct_fast4x4 vp8_fast_fdct4x4_mmx - -#undef vp8_fdct_fast8x4 -#define vp8_fdct_fast8x4 vp8_fast_fdct8x4_mmx +#endif #endif #endif #if HAVE_SSE2 -extern prototype_fdct(vp8_short_fdct4x4_wmt); extern prototype_fdct(vp8_short_fdct8x4_wmt); -extern prototype_fdct(vp8_fast_fdct8x4_wmt); - extern prototype_fdct(vp8_short_walsh4x4_sse2); #if !CONFIG_RUNTIME_CPU_DETECT diff --git a/vp8/encoder/x86/x86_csystemdependent.c b/vp8/encoder/x86/x86_csystemdependent.c index f3750455b..0fb82e60e 100644 --- a/vp8/encoder/x86/x86_csystemdependent.c +++ b/vp8/encoder/x86/x86_csystemdependent.c @@ -18,15 +18,10 @@ #if HAVE_MMX void vp8_short_fdct8x4_mmx(short *input, short *output, int pitch) { - vp8_short_fdct4x4_mmx(input, output, pitch); - vp8_short_fdct4x4_mmx(input + 4, output + 16, pitch); + vp8_short_fdct4x4_c(input, output, pitch); + vp8_short_fdct4x4_c(input + 4, output + 16, pitch); } -void vp8_fast_fdct8x4_mmx(short *input, short *output, int pitch) -{ - vp8_fast_fdct4x4_mmx(input, output , pitch); - vp8_fast_fdct4x4_mmx(input + 4, output + 16, pitch); -} int vp8_fast_quantize_b_impl_mmx(short *coeff_ptr, short *zbin_ptr, short *qcoeff_ptr, short *dequant_ptr, @@ -87,11 +82,6 @@ void vp8_subtract_b_mmx(BLOCK *be, BLOCKD *bd, int pitch) #endif #if HAVE_SSE2 -void vp8_short_fdct8x4_wmt(short *input, short *output, int pitch) -{ - vp8_short_fdct4x4_wmt(input, output, pitch); - vp8_short_fdct4x4_wmt(input + 4, output + 16, pitch); -} int vp8_fast_quantize_b_impl_sse(short *coeff_ptr, short *zbin_ptr, short *qcoeff_ptr, short *dequant_ptr, @@ -221,11 +211,19 @@ void vp8_arch_x86_encoder_init(VP8_COMP *cpi) cpi->rtcd.variance.get8x8var = vp8_get8x8var_mmx; cpi->rtcd.variance.get16x16var = vp8_get16x16var_mmx; cpi->rtcd.variance.get4x4sse_cs = vp8_get4x4sse_cs_mmx; - +#if 0 // new fdct cpi->rtcd.fdct.short4x4 = vp8_short_fdct4x4_mmx; cpi->rtcd.fdct.short8x4 = vp8_short_fdct8x4_mmx; - cpi->rtcd.fdct.fast4x4 = vp8_fast_fdct4x4_mmx; - cpi->rtcd.fdct.fast8x4 = vp8_fast_fdct8x4_mmx; + cpi->rtcd.fdct.fast4x4 = vp8_short_fdct4x4_mmx; + cpi->rtcd.fdct.fast8x4 = vp8_short_fdct8x4_mmx; +#else + cpi->rtcd.fdct.short4x4 = vp8_short_fdct4x4_c; + cpi->rtcd.fdct.short8x4 = vp8_short_fdct8x4_c; + cpi->rtcd.fdct.fast4x4 = vp8_short_fdct4x4_c; + cpi->rtcd.fdct.fast8x4 = vp8_short_fdct8x4_c; + +#endif + cpi->rtcd.fdct.walsh_short4x4 = vp8_short_walsh4x4_c; cpi->rtcd.encodemb.berr = vp8_block_error_mmx; @@ -270,13 +268,13 @@ void vp8_arch_x86_encoder_init(VP8_COMP *cpi) cpi->rtcd.variance.get16x16var = vp8_get16x16var_sse2; /* cpi->rtcd.variance.get4x4sse_cs not implemented for wmt */; -#if 0 +#if 0 //new fdct /* short SSE2 DCT currently disabled, does not match the MMX version */ cpi->rtcd.fdct.short4x4 = vp8_short_fdct4x4_wmt; cpi->rtcd.fdct.short8x4 = vp8_short_fdct8x4_wmt; -#endif /* cpi->rtcd.fdct.fast4x4 not implemented for wmt */; - cpi->rtcd.fdct.fast8x4 = vp8_fast_fdct8x4_wmt; + cpi->rtcd.fdct.fast8x4 = vp8_short_fdct8x4_wmt; +#endif cpi->rtcd.fdct.walsh_short4x4 = vp8_short_walsh4x4_sse2; cpi->rtcd.encodemb.berr = vp8_block_error_xmm; -- cgit v1.2.3