diff libavcodec/x86/dsputil_mmx.h @ 2:897f711a7157

rearrange to work with autoconf
author Nina Engelhardt <nengel@mailbox.tu-berlin.de>
date Tue, 25 Sep 2012 15:55:33 +0200
parents
children
line diff
     1.1 --- /dev/null	Thu Jan 01 00:00:00 1970 +0000
     1.2 +++ b/libavcodec/x86/dsputil_mmx.h	Tue Sep 25 15:55:33 2012 +0200
     1.3 @@ -0,0 +1,170 @@
     1.4 +/*
     1.5 + * MMX optimized DSP utils
     1.6 + * Copyright (c) 2007  Aurelien Jacobs <aurel@gnuage.org>
     1.7 + *
     1.8 + * This file is part of FFmpeg.
     1.9 + *
    1.10 + * FFmpeg is free software; you can redistribute it and/or
    1.11 + * modify it under the terms of the GNU Lesser General Public
    1.12 + * License as published by the Free Software Foundation; either
    1.13 + * version 2.1 of the License, or (at your option) any later version.
    1.14 + *
    1.15 + * FFmpeg is distributed in the hope that it will be useful,
    1.16 + * but WITHOUT ANY WARRANTY; without even the implied warranty of
    1.17 + * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
    1.18 + * Lesser General Public License for more details.
    1.19 + *
    1.20 + * You should have received a copy of the GNU Lesser General Public
    1.21 + * License along with FFmpeg; if not, write to the Free Software
    1.22 + * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
    1.23 + */
    1.24 +
    1.25 +#ifndef AVCODEC_X86_DSPUTIL_MMX_H
    1.26 +#define AVCODEC_X86_DSPUTIL_MMX_H
    1.27 +
    1.28 +#include <stdint.h>
    1.29 +#include "libavcodec/dsputil.h"
    1.30 +
    1.31 +typedef struct { uint64_t a, b; } xmm_reg;
    1.32 +
    1.33 +extern const uint64_t ff_bone;
    1.34 +extern const uint64_t ff_wtwo;
    1.35 +
    1.36 +extern const uint64_t ff_pdw_80000000[2];
    1.37 +
    1.38 +extern const uint64_t ff_pw_3;
    1.39 +extern const uint64_t ff_pw_4;
    1.40 +extern const xmm_reg  ff_pw_5;
    1.41 +extern const xmm_reg  ff_pw_8;
    1.42 +extern const uint64_t ff_pw_15;
    1.43 +extern const xmm_reg  ff_pw_16;
    1.44 +extern const uint64_t ff_pw_20;
    1.45 +extern const xmm_reg  ff_pw_28;
    1.46 +extern const xmm_reg  ff_pw_32;
    1.47 +extern const uint64_t ff_pw_42;
    1.48 +extern const xmm_reg  ff_pw_64;
    1.49 +extern const uint64_t ff_pw_96;
    1.50 +extern const uint64_t ff_pw_128;
    1.51 +extern const uint64_t ff_pw_255;
    1.52 +
    1.53 +extern const uint64_t ff_pb_1;
    1.54 +extern const uint64_t ff_pb_3;
    1.55 +extern const uint64_t ff_pb_7;
    1.56 +extern const uint64_t ff_pb_1F;
    1.57 +extern const uint64_t ff_pb_3F;
    1.58 +extern const uint64_t ff_pb_81;
    1.59 +extern const uint64_t ff_pb_A1;
    1.60 +extern const uint64_t ff_pb_FC;
    1.61 +
    1.62 +extern const double ff_pd_1[2];
    1.63 +extern const double ff_pd_2[2];
    1.64 +
    1.65 +#define LOAD4(stride,in,a,b,c,d)\
    1.66 +    "movq 0*"#stride"+"#in", "#a"\n\t"\
    1.67 +    "movq 1*"#stride"+"#in", "#b"\n\t"\
    1.68 +    "movq 2*"#stride"+"#in", "#c"\n\t"\
    1.69 +    "movq 3*"#stride"+"#in", "#d"\n\t"
    1.70 +
    1.71 +#define STORE4(stride,out,a,b,c,d)\
    1.72 +    "movq "#a", 0*"#stride"+"#out"\n\t"\
    1.73 +    "movq "#b", 1*"#stride"+"#out"\n\t"\
    1.74 +    "movq "#c", 2*"#stride"+"#out"\n\t"\
    1.75 +    "movq "#d", 3*"#stride"+"#out"\n\t"
    1.76 +
    1.77 +/* in/out: mma=mma+mmb, mmb=mmb-mma */
    1.78 +#define SUMSUB_BA( a, b ) \
    1.79 +    "paddw "#b", "#a" \n\t"\
    1.80 +    "paddw "#b", "#b" \n\t"\
    1.81 +    "psubw "#a", "#b" \n\t"
    1.82 +
    1.83 +#define SBUTTERFLY(a,b,t,n,m)\
    1.84 +    "mov" #m " " #a ", " #t "         \n\t" /* abcd */\
    1.85 +    "punpckl" #n " " #b ", " #a "     \n\t" /* aebf */\
    1.86 +    "punpckh" #n " " #b ", " #t "     \n\t" /* cgdh */\
    1.87 +
    1.88 +#define TRANSPOSE4(a,b,c,d,t)\
    1.89 +    SBUTTERFLY(a,b,t,wd,q) /* a=aebf t=cgdh */\
    1.90 +    SBUTTERFLY(c,d,b,wd,q) /* c=imjn b=kolp */\
    1.91 +    SBUTTERFLY(a,c,d,dq,q) /* a=aeim d=bfjn */\
    1.92 +    SBUTTERFLY(t,b,c,dq,q) /* t=cgko c=dhlp */
    1.93 +
    1.94 +// e,f,g,h can be memory
    1.95 +// out: a,d,t,c
    1.96 +#define TRANSPOSE8x4(a,b,c,d,e,f,g,h,t)\
    1.97 +    "punpcklbw " #e ", " #a " \n\t" /* a0 e0 a1 e1 a2 e2 a3 e3 */\
    1.98 +    "punpcklbw " #f ", " #b " \n\t" /* b0 f0 b1 f1 b2 f2 b3 f3 */\
    1.99 +    "punpcklbw " #g ", " #c " \n\t" /* c0 g0 c1 g1 c2 g2 d3 g3 */\
   1.100 +    "punpcklbw " #h ", " #d " \n\t" /* d0 h0 d1 h1 d2 h2 d3 h3 */\
   1.101 +    SBUTTERFLY(a, b, t, bw, q)   /* a= a0 b0 e0 f0 a1 b1 e1 f1 */\
   1.102 +                                 /* t= a2 b2 e2 f2 a3 b3 e3 f3 */\
   1.103 +    SBUTTERFLY(c, d, b, bw, q)   /* c= c0 d0 g0 h0 c1 d1 g1 h1 */\
   1.104 +                                 /* b= c2 d2 g2 h2 c3 d3 g3 h3 */\
   1.105 +    SBUTTERFLY(a, c, d, wd, q)   /* a= a0 b0 c0 d0 e0 f0 g0 h0 */\
   1.106 +                                 /* d= a1 b1 c1 d1 e1 f1 g1 h1 */\
   1.107 +    SBUTTERFLY(t, b, c, wd, q)   /* t= a2 b2 c2 d2 e2 f2 g2 h2 */\
   1.108 +                                 /* c= a3 b3 c3 d3 e3 f3 g3 h3 */
   1.109 +
   1.110 +#if ARCH_X86_64
   1.111 +// permutes 01234567 -> 05736421
   1.112 +#define TRANSPOSE8(a,b,c,d,e,f,g,h,t)\
   1.113 +    SBUTTERFLY(a,b,%%xmm8,wd,dqa)\
   1.114 +    SBUTTERFLY(c,d,b,wd,dqa)\
   1.115 +    SBUTTERFLY(e,f,d,wd,dqa)\
   1.116 +    SBUTTERFLY(g,h,f,wd,dqa)\
   1.117 +    SBUTTERFLY(a,c,h,dq,dqa)\
   1.118 +    SBUTTERFLY(%%xmm8,b,c,dq,dqa)\
   1.119 +    SBUTTERFLY(e,g,b,dq,dqa)\
   1.120 +    SBUTTERFLY(d,f,g,dq,dqa)\
   1.121 +    SBUTTERFLY(a,e,f,qdq,dqa)\
   1.122 +    SBUTTERFLY(%%xmm8,d,e,qdq,dqa)\
   1.123 +    SBUTTERFLY(h,b,d,qdq,dqa)\
   1.124 +    SBUTTERFLY(c,g,b,qdq,dqa)\
   1.125 +    "movdqa %%xmm8, "#g"              \n\t"
   1.126 +#else
   1.127 +#define TRANSPOSE8(a,b,c,d,e,f,g,h,t)\
   1.128 +    "movdqa "#h", "#t"                \n\t"\
   1.129 +    SBUTTERFLY(a,b,h,wd,dqa)\
   1.130 +    "movdqa "#h", 16"#t"              \n\t"\
   1.131 +    "movdqa "#t", "#h"                \n\t"\
   1.132 +    SBUTTERFLY(c,d,b,wd,dqa)\
   1.133 +    SBUTTERFLY(e,f,d,wd,dqa)\
   1.134 +    SBUTTERFLY(g,h,f,wd,dqa)\
   1.135 +    SBUTTERFLY(a,c,h,dq,dqa)\
   1.136 +    "movdqa "#h", "#t"                \n\t"\
   1.137 +    "movdqa 16"#t", "#h"              \n\t"\
   1.138 +    SBUTTERFLY(h,b,c,dq,dqa)\
   1.139 +    SBUTTERFLY(e,g,b,dq,dqa)\
   1.140 +    SBUTTERFLY(d,f,g,dq,dqa)\
   1.141 +    SBUTTERFLY(a,e,f,qdq,dqa)\
   1.142 +    SBUTTERFLY(h,d,e,qdq,dqa)\
   1.143 +    "movdqa "#h", 16"#t"              \n\t"\
   1.144 +    "movdqa "#t", "#h"                \n\t"\
   1.145 +    SBUTTERFLY(h,b,d,qdq,dqa)\
   1.146 +    SBUTTERFLY(c,g,b,qdq,dqa)\
   1.147 +    "movdqa 16"#t", "#g"              \n\t"
   1.148 +#endif
   1.149 +
   1.150 +#define MOVQ_WONE(regd) \
   1.151 +    __asm__ volatile ( \
   1.152 +    "pcmpeqd %%" #regd ", %%" #regd " \n\t" \
   1.153 +    "psrlw $15, %%" #regd ::)
   1.154 +
   1.155 +void add_pixels_clamped_mmx(const DCTELEM *block, uint8_t *pixels, int line_size);
   1.156 +void put_pixels_clamped_mmx(const DCTELEM *block, uint8_t *pixels, int line_size);
   1.157 +void put_signed_pixels_clamped_mmx(const DCTELEM *block, uint8_t *pixels, int line_size);
   1.158 +
   1.159 +void ff_put_cavs_qpel8_mc00_mmx2(uint8_t *dst, uint8_t *src, int stride);
   1.160 +void ff_avg_cavs_qpel8_mc00_mmx2(uint8_t *dst, uint8_t *src, int stride);
   1.161 +void ff_put_cavs_qpel16_mc00_mmx2(uint8_t *dst, uint8_t *src, int stride);
   1.162 +void ff_avg_cavs_qpel16_mc00_mmx2(uint8_t *dst, uint8_t *src, int stride);
   1.163 +
   1.164 +void ff_put_vc1_mspel_mc00_mmx(uint8_t *dst, const uint8_t *src, int stride, int rnd);
   1.165 +void ff_avg_vc1_mspel_mc00_mmx2(uint8_t *dst, const uint8_t *src, int stride, int rnd);
   1.166 +
   1.167 +void ff_lpc_compute_autocorr_sse2(const int32_t *data, int len, int lag,
   1.168 +                                   double *autoc);
   1.169 +
   1.170 +void ff_mmx_idct(DCTELEM *block);
   1.171 +void ff_mmxext_idct(DCTELEM *block);
   1.172 +
   1.173 +#endif /* AVCODEC_X86_DSPUTIL_MMX_H */