John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 1 | /* |
John Koleszar | c2140b8 | 2010-09-09 08:16:39 -0400 | [diff] [blame] | 2 | * Copyright (c) 2010 The WebM project authors. All Rights Reserved. |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 3 | * |
John Koleszar | 94c52e4 | 2010-06-18 12:39:21 -0400 | [diff] [blame] | 4 | * Use of this source code is governed by a BSD-style license |
John Koleszar | 09202d8 | 2010-06-04 16:19:40 -0400 | [diff] [blame] | 5 | * that can be found in the LICENSE file in the root of the source |
| 6 | * tree. An additional intellectual property rights grant can be found |
John Koleszar | 94c52e4 | 2010-06-18 12:39:21 -0400 | [diff] [blame] | 7 | * in the file PATENTS. All contributing project authors may |
John Koleszar | 09202d8 | 2010-06-04 16:19:40 -0400 | [diff] [blame] | 8 | * be found in the AUTHORS file in the root of the source tree. |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 9 | */ |
| 10 | |
| 11 | |
| 12 | #include "vpx_ports/config.h" |
| 13 | #include "vpx_ports/x86.h" |
John Koleszar | 02321de | 2011-02-10 14:41:38 -0500 | [diff] [blame] | 14 | #include "vp8/encoder/variance.h" |
| 15 | #include "vp8/encoder/onyx_int.h" |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 16 | |
| 17 | |
| 18 | #if HAVE_MMX |
Johann | 92b0e54 | 2011-06-14 11:31:50 -0400 | [diff] [blame] | 19 | void vp8_short_fdct8x4_mmx(short *input, short *output, int pitch) |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 20 | { |
Fritz Koenig | 5f0e061 | 2010-10-21 10:53:15 -0700 | [diff] [blame] | 21 | vp8_short_fdct4x4_mmx(input, output, pitch); |
| 22 | vp8_short_fdct4x4_mmx(input + 4, output + 16, pitch); |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 23 | } |
| 24 | |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 25 | int vp8_fast_quantize_b_impl_mmx(short *coeff_ptr, short *zbin_ptr, |
| 26 | short *qcoeff_ptr, short *dequant_ptr, |
| 27 | short *scan_mask, short *round_ptr, |
| 28 | short *quant_ptr, short *dqcoeff_ptr); |
Johann | 92b0e54 | 2011-06-14 11:31:50 -0400 | [diff] [blame] | 29 | void vp8_fast_quantize_b_mmx(BLOCK *b, BLOCKD *d) |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 30 | { |
Timothy B. Terriberry | 8f75ea6 | 2010-10-21 17:04:30 -0700 | [diff] [blame] | 31 | short *scan_mask = vp8_default_zig_zag_mask;//d->scan_order_mask_ptr; |
| 32 | short *coeff_ptr = b->coeff; |
| 33 | short *zbin_ptr = b->zbin; |
| 34 | short *round_ptr = b->round; |
Scott LaVarnway | 516ea84 | 2010-12-28 14:51:46 -0500 | [diff] [blame] | 35 | short *quant_ptr = b->quant_fast; |
Timothy B. Terriberry | 8f75ea6 | 2010-10-21 17:04:30 -0700 | [diff] [blame] | 36 | short *qcoeff_ptr = d->qcoeff; |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 37 | short *dqcoeff_ptr = d->dqcoeff; |
Timothy B. Terriberry | 8f75ea6 | 2010-10-21 17:04:30 -0700 | [diff] [blame] | 38 | short *dequant_ptr = d->dequant; |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 39 | |
| 40 | d->eob = vp8_fast_quantize_b_impl_mmx( |
| 41 | coeff_ptr, |
| 42 | zbin_ptr, |
| 43 | qcoeff_ptr, |
| 44 | dequant_ptr, |
| 45 | scan_mask, |
| 46 | |
| 47 | round_ptr, |
| 48 | quant_ptr, |
| 49 | dqcoeff_ptr |
| 50 | ); |
| 51 | } |
| 52 | |
| 53 | int vp8_mbblock_error_mmx_impl(short *coeff_ptr, short *dcoef_ptr, int dc); |
Johann | 92b0e54 | 2011-06-14 11:31:50 -0400 | [diff] [blame] | 54 | int vp8_mbblock_error_mmx(MACROBLOCK *mb, int dc) |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 55 | { |
| 56 | short *coeff_ptr = mb->block[0].coeff; |
| 57 | short *dcoef_ptr = mb->e_mbd.block[0].dqcoeff; |
| 58 | return vp8_mbblock_error_mmx_impl(coeff_ptr, dcoef_ptr, dc); |
| 59 | } |
| 60 | |
| 61 | int vp8_mbuverror_mmx_impl(short *s_ptr, short *d_ptr); |
Johann | 92b0e54 | 2011-06-14 11:31:50 -0400 | [diff] [blame] | 62 | int vp8_mbuverror_mmx(MACROBLOCK *mb) |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 63 | { |
| 64 | short *s_ptr = &mb->coeff[256]; |
| 65 | short *d_ptr = &mb->e_mbd.dqcoeff[256]; |
| 66 | return vp8_mbuverror_mmx_impl(s_ptr, d_ptr); |
| 67 | } |
| 68 | |
| 69 | void vp8_subtract_b_mmx_impl(unsigned char *z, int src_stride, |
| 70 | short *diff, unsigned char *predictor, |
| 71 | int pitch); |
Johann | 92b0e54 | 2011-06-14 11:31:50 -0400 | [diff] [blame] | 72 | void vp8_subtract_b_mmx(BLOCK *be, BLOCKD *bd, int pitch) |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 73 | { |
| 74 | unsigned char *z = *(be->base_src) + be->src; |
| 75 | unsigned int src_stride = be->src_stride; |
| 76 | short *diff = &be->src_diff[0]; |
| 77 | unsigned char *predictor = &bd->predictor[0]; |
| 78 | vp8_subtract_b_mmx_impl(z, src_stride, diff, predictor, pitch); |
| 79 | } |
| 80 | |
| 81 | #endif |
| 82 | |
| 83 | #if HAVE_SSE2 |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 84 | int vp8_mbblock_error_xmm_impl(short *coeff_ptr, short *dcoef_ptr, int dc); |
Johann | 92b0e54 | 2011-06-14 11:31:50 -0400 | [diff] [blame] | 85 | int vp8_mbblock_error_xmm(MACROBLOCK *mb, int dc) |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 86 | { |
| 87 | short *coeff_ptr = mb->block[0].coeff; |
| 88 | short *dcoef_ptr = mb->e_mbd.block[0].dqcoeff; |
| 89 | return vp8_mbblock_error_xmm_impl(coeff_ptr, dcoef_ptr, dc); |
| 90 | } |
| 91 | |
| 92 | int vp8_mbuverror_xmm_impl(short *s_ptr, short *d_ptr); |
Johann | 92b0e54 | 2011-06-14 11:31:50 -0400 | [diff] [blame] | 93 | int vp8_mbuverror_xmm(MACROBLOCK *mb) |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 94 | { |
| 95 | short *s_ptr = &mb->coeff[256]; |
| 96 | short *d_ptr = &mb->e_mbd.dqcoeff[256]; |
| 97 | return vp8_mbuverror_xmm_impl(s_ptr, d_ptr); |
| 98 | } |
| 99 | |
Yunqing Wang | 4db2076 | 2010-10-18 14:15:15 -0400 | [diff] [blame] | 100 | void vp8_subtract_b_sse2_impl(unsigned char *z, int src_stride, |
| 101 | short *diff, unsigned char *predictor, |
| 102 | int pitch); |
Johann | 92b0e54 | 2011-06-14 11:31:50 -0400 | [diff] [blame] | 103 | void vp8_subtract_b_sse2(BLOCK *be, BLOCKD *bd, int pitch) |
Yunqing Wang | 4db2076 | 2010-10-18 14:15:15 -0400 | [diff] [blame] | 104 | { |
| 105 | unsigned char *z = *(be->base_src) + be->src; |
| 106 | unsigned int src_stride = be->src_stride; |
| 107 | short *diff = &be->src_diff[0]; |
| 108 | unsigned char *predictor = &bd->predictor[0]; |
| 109 | vp8_subtract_b_sse2_impl(z, src_stride, diff, predictor, pitch); |
| 110 | } |
| 111 | |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 112 | #endif |
| 113 | |
Scott LaVarnway | ff4a71f | 2010-11-01 16:24:15 -0400 | [diff] [blame] | 114 | #if HAVE_SSSE3 |
Yaowu Xu | 57ad189 | 2011-04-29 09:37:59 -0700 | [diff] [blame] | 115 | #if CONFIG_INTERNAL_STATS |
Jim Bankoski | 3f6f728 | 2011-03-08 09:05:18 -0500 | [diff] [blame] | 116 | #if ARCH_X86_64 |
| 117 | typedef void ssimpf |
| 118 | ( |
| 119 | unsigned char *s, |
| 120 | int sp, |
| 121 | unsigned char *r, |
| 122 | int rp, |
| 123 | unsigned long *sum_s, |
| 124 | unsigned long *sum_r, |
| 125 | unsigned long *sum_sq_s, |
| 126 | unsigned long *sum_sq_r, |
| 127 | unsigned long *sum_sxr |
| 128 | ); |
| 129 | |
| 130 | extern ssimpf vp8_ssim_parms_16x16_sse3; |
| 131 | extern ssimpf vp8_ssim_parms_8x8_sse3; |
| 132 | #endif |
| 133 | #endif |
Scott LaVarnway | ff4a71f | 2010-11-01 16:24:15 -0400 | [diff] [blame] | 134 | #endif |
| 135 | |
| 136 | |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 137 | void vp8_arch_x86_encoder_init(VP8_COMP *cpi) |
| 138 | { |
| 139 | #if CONFIG_RUNTIME_CPU_DETECT |
| 140 | int flags = x86_simd_caps(); |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 141 | |
| 142 | /* Note: |
| 143 | * |
| 144 | * This platform can be built without runtime CPU detection as well. If |
| 145 | * you modify any of the function mappings present in this file, be sure |
| 146 | * to also update them in static mapings (<arch>/filename_<arch>.h) |
| 147 | */ |
| 148 | |
| 149 | /* Override default functions with fastest ones for this CPU. */ |
| 150 | #if HAVE_MMX |
Johann | a7d4d3c | 2011-05-09 11:16:31 -0400 | [diff] [blame] | 151 | if (flags & HAS_MMX) |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 152 | { |
| 153 | cpi->rtcd.variance.sad16x16 = vp8_sad16x16_mmx; |
| 154 | cpi->rtcd.variance.sad16x8 = vp8_sad16x8_mmx; |
| 155 | cpi->rtcd.variance.sad8x16 = vp8_sad8x16_mmx; |
| 156 | cpi->rtcd.variance.sad8x8 = vp8_sad8x8_mmx; |
| 157 | cpi->rtcd.variance.sad4x4 = vp8_sad4x4_mmx; |
| 158 | |
| 159 | cpi->rtcd.variance.var4x4 = vp8_variance4x4_mmx; |
| 160 | cpi->rtcd.variance.var8x8 = vp8_variance8x8_mmx; |
| 161 | cpi->rtcd.variance.var8x16 = vp8_variance8x16_mmx; |
| 162 | cpi->rtcd.variance.var16x8 = vp8_variance16x8_mmx; |
| 163 | cpi->rtcd.variance.var16x16 = vp8_variance16x16_mmx; |
| 164 | |
| 165 | cpi->rtcd.variance.subpixvar4x4 = vp8_sub_pixel_variance4x4_mmx; |
| 166 | cpi->rtcd.variance.subpixvar8x8 = vp8_sub_pixel_variance8x8_mmx; |
| 167 | cpi->rtcd.variance.subpixvar8x16 = vp8_sub_pixel_variance8x16_mmx; |
| 168 | cpi->rtcd.variance.subpixvar16x8 = vp8_sub_pixel_variance16x8_mmx; |
| 169 | cpi->rtcd.variance.subpixvar16x16 = vp8_sub_pixel_variance16x16_mmx; |
John Koleszar | a0ae368 | 2010-10-27 11:28:43 -0400 | [diff] [blame] | 170 | cpi->rtcd.variance.halfpixvar16x16_h = vp8_variance_halfpixvar16x16_h_mmx; |
| 171 | cpi->rtcd.variance.halfpixvar16x16_v = vp8_variance_halfpixvar16x16_v_mmx; |
| 172 | cpi->rtcd.variance.halfpixvar16x16_hv = vp8_variance_halfpixvar16x16_hv_mmx; |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 173 | cpi->rtcd.variance.subpixmse16x16 = vp8_sub_pixel_mse16x16_mmx; |
| 174 | |
| 175 | cpi->rtcd.variance.mse16x16 = vp8_mse16x16_mmx; |
| 176 | cpi->rtcd.variance.getmbss = vp8_get_mb_ss_mmx; |
| 177 | |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 178 | cpi->rtcd.variance.get4x4sse_cs = vp8_get4x4sse_cs_mmx; |
Fritz Koenig | 5f0e061 | 2010-10-21 10:53:15 -0700 | [diff] [blame] | 179 | |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 180 | cpi->rtcd.fdct.short4x4 = vp8_short_fdct4x4_mmx; |
Johann | 92b0e54 | 2011-06-14 11:31:50 -0400 | [diff] [blame] | 181 | cpi->rtcd.fdct.short8x4 = vp8_short_fdct8x4_mmx; |
Yaowu Xu | d0dd01b | 2010-06-16 12:52:18 -0700 | [diff] [blame] | 182 | cpi->rtcd.fdct.fast4x4 = vp8_short_fdct4x4_mmx; |
Johann | 92b0e54 | 2011-06-14 11:31:50 -0400 | [diff] [blame] | 183 | cpi->rtcd.fdct.fast8x4 = vp8_short_fdct8x4_mmx; |
Yaowu Xu | d0dd01b | 2010-06-16 12:52:18 -0700 | [diff] [blame] | 184 | |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 185 | cpi->rtcd.fdct.walsh_short4x4 = vp8_short_walsh4x4_c; |
| 186 | |
| 187 | cpi->rtcd.encodemb.berr = vp8_block_error_mmx; |
Johann | 92b0e54 | 2011-06-14 11:31:50 -0400 | [diff] [blame] | 188 | cpi->rtcd.encodemb.mberr = vp8_mbblock_error_mmx; |
| 189 | cpi->rtcd.encodemb.mbuverr = vp8_mbuverror_mmx; |
| 190 | cpi->rtcd.encodemb.subb = vp8_subtract_b_mmx; |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 191 | cpi->rtcd.encodemb.submby = vp8_subtract_mby_mmx; |
| 192 | cpi->rtcd.encodemb.submbuv = vp8_subtract_mbuv_mmx; |
| 193 | |
Johann | 92b0e54 | 2011-06-14 11:31:50 -0400 | [diff] [blame] | 194 | /*cpi->rtcd.quantize.fastquantb = vp8_fast_quantize_b_mmx;*/ |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 195 | } |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 196 | #endif |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 197 | |
Yunqing Wang | 71ecb5d | 2010-10-27 08:45:24 -0400 | [diff] [blame] | 198 | #if HAVE_SSE2 |
Johann | a7d4d3c | 2011-05-09 11:16:31 -0400 | [diff] [blame] | 199 | if (flags & HAS_SSE2) |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 200 | { |
| 201 | cpi->rtcd.variance.sad16x16 = vp8_sad16x16_wmt; |
| 202 | cpi->rtcd.variance.sad16x8 = vp8_sad16x8_wmt; |
| 203 | cpi->rtcd.variance.sad8x16 = vp8_sad8x16_wmt; |
| 204 | cpi->rtcd.variance.sad8x8 = vp8_sad8x8_wmt; |
| 205 | cpi->rtcd.variance.sad4x4 = vp8_sad4x4_wmt; |
Yunqing Wang | 20bd144 | 2011-06-28 09:14:13 -0400 | [diff] [blame] | 206 | cpi->rtcd.variance.copy32xn = vp8_copy32xn_sse2; |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 207 | |
| 208 | cpi->rtcd.variance.var4x4 = vp8_variance4x4_wmt; |
| 209 | cpi->rtcd.variance.var8x8 = vp8_variance8x8_wmt; |
| 210 | cpi->rtcd.variance.var8x16 = vp8_variance8x16_wmt; |
| 211 | cpi->rtcd.variance.var16x8 = vp8_variance16x8_wmt; |
| 212 | cpi->rtcd.variance.var16x16 = vp8_variance16x16_wmt; |
| 213 | |
| 214 | cpi->rtcd.variance.subpixvar4x4 = vp8_sub_pixel_variance4x4_wmt; |
| 215 | cpi->rtcd.variance.subpixvar8x8 = vp8_sub_pixel_variance8x8_wmt; |
| 216 | cpi->rtcd.variance.subpixvar8x16 = vp8_sub_pixel_variance8x16_wmt; |
| 217 | cpi->rtcd.variance.subpixvar16x8 = vp8_sub_pixel_variance16x8_wmt; |
| 218 | cpi->rtcd.variance.subpixvar16x16 = vp8_sub_pixel_variance16x16_wmt; |
John Koleszar | a0ae368 | 2010-10-27 11:28:43 -0400 | [diff] [blame] | 219 | cpi->rtcd.variance.halfpixvar16x16_h = vp8_variance_halfpixvar16x16_h_wmt; |
| 220 | cpi->rtcd.variance.halfpixvar16x16_v = vp8_variance_halfpixvar16x16_v_wmt; |
| 221 | cpi->rtcd.variance.halfpixvar16x16_hv = vp8_variance_halfpixvar16x16_hv_wmt; |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 222 | cpi->rtcd.variance.subpixmse16x16 = vp8_sub_pixel_mse16x16_wmt; |
| 223 | |
| 224 | cpi->rtcd.variance.mse16x16 = vp8_mse16x16_wmt; |
| 225 | cpi->rtcd.variance.getmbss = vp8_get_mb_ss_sse2; |
| 226 | |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 227 | /* cpi->rtcd.variance.get4x4sse_cs not implemented for wmt */; |
| 228 | |
Scott LaVarnway | f1a3b1e | 2010-06-24 13:11:30 -0400 | [diff] [blame] | 229 | cpi->rtcd.fdct.short4x4 = vp8_short_fdct4x4_sse2; |
| 230 | cpi->rtcd.fdct.short8x4 = vp8_short_fdct8x4_sse2; |
| 231 | cpi->rtcd.fdct.fast4x4 = vp8_short_fdct4x4_sse2; |
| 232 | cpi->rtcd.fdct.fast8x4 = vp8_short_fdct8x4_sse2; |
| 233 | |
Yunqing Wang | fc94ffc | 2010-10-21 10:26:50 -0400 | [diff] [blame] | 234 | cpi->rtcd.fdct.walsh_short4x4 = vp8_short_walsh4x4_sse2 ; |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 235 | |
| 236 | cpi->rtcd.encodemb.berr = vp8_block_error_xmm; |
Johann | 92b0e54 | 2011-06-14 11:31:50 -0400 | [diff] [blame] | 237 | cpi->rtcd.encodemb.mberr = vp8_mbblock_error_xmm; |
| 238 | cpi->rtcd.encodemb.mbuverr = vp8_mbuverror_xmm; |
| 239 | cpi->rtcd.encodemb.subb = vp8_subtract_b_sse2; |
Yunqing Wang | 4db2076 | 2010-10-18 14:15:15 -0400 | [diff] [blame] | 240 | cpi->rtcd.encodemb.submby = vp8_subtract_mby_sse2; |
| 241 | cpi->rtcd.encodemb.submbuv = vp8_subtract_mbuv_sse2; |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 242 | |
Johann | 8edaf6e | 2011-02-10 14:57:43 -0500 | [diff] [blame] | 243 | cpi->rtcd.quantize.quantb = vp8_regular_quantize_b_sse2; |
Johann | c32e0ec | 2011-03-24 13:31:10 -0400 | [diff] [blame] | 244 | cpi->rtcd.quantize.fastquantb = vp8_fast_quantize_b_sse2; |
Johann | 8b0cf5f | 2010-12-22 11:23:51 -0500 | [diff] [blame] | 245 | |
Attila Nagy | 7af0d90 | 2011-02-22 10:29:23 +0200 | [diff] [blame] | 246 | #if !(CONFIG_REALTIME_ONLY) |
Johann | 8b0cf5f | 2010-12-22 11:23:51 -0500 | [diff] [blame] | 247 | cpi->rtcd.temporal.apply = vp8_temporal_filter_apply_sse2; |
Attila Nagy | 7af0d90 | 2011-02-22 10:29:23 +0200 | [diff] [blame] | 248 | #endif |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 249 | } |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 250 | #endif |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 251 | |
Yunqing Wang | 71ecb5d | 2010-10-27 08:45:24 -0400 | [diff] [blame] | 252 | #if HAVE_SSE3 |
Johann | a7d4d3c | 2011-05-09 11:16:31 -0400 | [diff] [blame] | 253 | if (flags & HAS_SSE3) |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 254 | { |
| 255 | cpi->rtcd.variance.sad16x16 = vp8_sad16x16_sse3; |
| 256 | cpi->rtcd.variance.sad16x16x3 = vp8_sad16x16x3_sse3; |
| 257 | cpi->rtcd.variance.sad16x8x3 = vp8_sad16x8x3_sse3; |
| 258 | cpi->rtcd.variance.sad8x16x3 = vp8_sad8x16x3_sse3; |
| 259 | cpi->rtcd.variance.sad8x8x3 = vp8_sad8x8x3_sse3; |
| 260 | cpi->rtcd.variance.sad4x4x3 = vp8_sad4x4x3_sse3; |
| 261 | cpi->rtcd.search.full_search = vp8_full_search_sadx3; |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 262 | cpi->rtcd.variance.sad16x16x4d = vp8_sad16x16x4d_sse3; |
| 263 | cpi->rtcd.variance.sad16x8x4d = vp8_sad16x8x4d_sse3; |
| 264 | cpi->rtcd.variance.sad8x16x4d = vp8_sad8x16x4d_sse3; |
| 265 | cpi->rtcd.variance.sad8x8x4d = vp8_sad8x8x4d_sse3; |
| 266 | cpi->rtcd.variance.sad4x4x4d = vp8_sad4x4x4d_sse3; |
Yunqing Wang | 20bd144 | 2011-06-28 09:14:13 -0400 | [diff] [blame] | 267 | cpi->rtcd.variance.copy32xn = vp8_copy32xn_sse3; |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 268 | cpi->rtcd.search.diamond_search = vp8_diamond_search_sadx4; |
Yunqing Wang | cb7b1fb | 2011-05-06 12:51:31 -0400 | [diff] [blame] | 269 | cpi->rtcd.search.refining_search = vp8_refining_search_sadx4; |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 270 | } |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 271 | #endif |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 272 | |
Yunqing Wang | 71ecb5d | 2010-10-27 08:45:24 -0400 | [diff] [blame] | 273 | #if HAVE_SSSE3 |
Johann | a7d4d3c | 2011-05-09 11:16:31 -0400 | [diff] [blame] | 274 | if (flags & HAS_SSSE3) |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 275 | { |
| 276 | cpi->rtcd.variance.sad16x16x3 = vp8_sad16x16x3_ssse3; |
| 277 | cpi->rtcd.variance.sad16x8x3 = vp8_sad16x8x3_ssse3; |
Scott LaVarnway | ff4a71f | 2010-11-01 16:24:15 -0400 | [diff] [blame] | 278 | |
Yunqing Wang | 7b8e7f0 | 2011-03-09 11:16:30 -0500 | [diff] [blame] | 279 | cpi->rtcd.variance.subpixvar16x8 = vp8_sub_pixel_variance16x8_ssse3; |
Yunqing Wang | 244e2e1 | 2011-03-03 19:02:45 -0500 | [diff] [blame] | 280 | cpi->rtcd.variance.subpixvar16x16 = vp8_sub_pixel_variance16x16_ssse3; |
| 281 | |
Johann Koenig | 0870200 | 2011-04-07 16:40:05 -0400 | [diff] [blame] | 282 | cpi->rtcd.quantize.fastquantb = vp8_fast_quantize_b_ssse3; |
Scott LaVarnway | ff4a71f | 2010-11-01 16:24:15 -0400 | [diff] [blame] | 283 | |
Yaowu Xu | 57ad189 | 2011-04-29 09:37:59 -0700 | [diff] [blame] | 284 | #if CONFIG_INTERNAL_STATS |
Jim Bankoski | 3f6f728 | 2011-03-08 09:05:18 -0500 | [diff] [blame] | 285 | #if ARCH_X86_64 |
| 286 | cpi->rtcd.variance.ssimpf_8x8 = vp8_ssim_parms_8x8_sse3; |
| 287 | cpi->rtcd.variance.ssimpf = vp8_ssim_parms_16x16_sse3; |
| 288 | #endif |
| 289 | #endif |
| 290 | |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 291 | } |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 292 | #endif |
Yunqing Wang | 71ecb5d | 2010-10-27 08:45:24 -0400 | [diff] [blame] | 293 | |
Jim Bankoski | 3f6f728 | 2011-03-08 09:05:18 -0500 | [diff] [blame] | 294 | |
| 295 | |
Yunqing Wang | 71ecb5d | 2010-10-27 08:45:24 -0400 | [diff] [blame] | 296 | #if HAVE_SSE4_1 |
Johann | a7d4d3c | 2011-05-09 11:16:31 -0400 | [diff] [blame] | 297 | if (flags & HAS_SSE4_1) |
Yunqing Wang | 71ecb5d | 2010-10-27 08:45:24 -0400 | [diff] [blame] | 298 | { |
| 299 | cpi->rtcd.variance.sad16x16x8 = vp8_sad16x16x8_sse4; |
| 300 | cpi->rtcd.variance.sad16x8x8 = vp8_sad16x8x8_sse4; |
| 301 | cpi->rtcd.variance.sad8x16x8 = vp8_sad8x16x8_sse4; |
| 302 | cpi->rtcd.variance.sad8x8x8 = vp8_sad8x8x8_sse4; |
| 303 | cpi->rtcd.variance.sad4x4x8 = vp8_sad4x4x8_sse4; |
| 304 | cpi->rtcd.search.full_search = vp8_full_search_sadx8; |
Johann | 508ae1b | 2011-04-13 16:38:02 -0400 | [diff] [blame] | 305 | |
| 306 | cpi->rtcd.quantize.quantb = vp8_regular_quantize_b_sse4; |
Yunqing Wang | 71ecb5d | 2010-10-27 08:45:24 -0400 | [diff] [blame] | 307 | } |
| 308 | #endif |
| 309 | |
John Koleszar | 0ea50ce | 2010-05-18 11:58:33 -0400 | [diff] [blame] | 310 | #endif |
| 311 | } |