1 ;******************************************************************************
2 ;* VP9 IDCT SIMD optimizations
4 ;* Copyright (C) 2013 Clément Bœsch <u pkh me>
5 ;* Copyright (C) 2013 Ronald S. Bultje <rsbultje gmail com>
7 ;* This file is part of FFmpeg.
9 ;* FFmpeg is free software; you can redistribute it and/or
10 ;* modify it under the terms of the GNU Lesser General Public
11 ;* License as published by the Free Software Foundation; either
12 ;* version 2.1 of the License, or (at your option) any later version.
14 ;* FFmpeg is distributed in the hope that it will be useful,
15 ;* but WITHOUT ANY WARRANTY; without even the implied warranty of
16 ;* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
17 ;* Lesser General Public License for more details.
19 ;* You should have received a copy of the GNU Lesser General Public
20 ;* License along with FFmpeg; if not, write to the Free Software
21 ;* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
22 ;******************************************************************************
39 ; (a*x + b*y + round) >> shift
40 %macro VP9_MULSUB_2W_2X 5 ; dst1, dst2/src, round, coefs1, coefs2
49 %macro VP9_MULSUB_2W_4X 7 ; dst1, dst2, coef1, coef2, rnd, tmp1/src, tmp2
50 VP9_MULSUB_2W_2X %7, %6, %5, [pw_m%3_%4], [pw_%4_%3]
51 VP9_MULSUB_2W_2X %1, %2, %5, [pw_m%3_%4], [pw_%4_%3]
56 %macro VP9_UNPACK_MULSUB_2W_4X 7-9 ; dst1, dst2, (src1, src2,) coef1, coef2, rnd, tmp1, tmp2
58 punpckhwd m%6, m%2, m%1
60 VP9_MULSUB_2W_4X %1, %2, %3, %4, %5, %6, %7
62 punpckhwd m%8, m%4, m%3
63 punpcklwd m%2, m%4, m%3
64 VP9_MULSUB_2W_4X %1, %2, %5, %6, %7, %8, %9
68 %macro VP9_IDCT4_1D_FINALIZE 0
69 SUMSUB_BA w, 3, 2, 4 ; m3=t3+t0, m2=-t3+t0
70 SUMSUB_BA w, 1, 0, 4 ; m1=t2+t1, m0=-t2+t1
71 SWAP 0, 3, 2 ; 3102 -> 0123
76 SUMSUB_BA w, 2, 0, 4 ; m2=IN(0)+IN(2) m0=IN(0)-IN(2)
77 pmulhrsw m2, m6 ; m2=t0
78 pmulhrsw m0, m6 ; m0=t1
80 VP9_UNPACK_MULSUB_2W_4X 0, 2, 11585, 11585, m7, 4, 5 ; m0=t1, m1=t0
82 VP9_UNPACK_MULSUB_2W_4X 1, 3, 15137, 6270, m7, 4, 5 ; m1=t2, m3=t3
86 %macro VP9_IADST4_1D 0
96 pmaddwd xmm1, xmm0, [pw_5283_13377]
97 pmaddwd xmm4, xmm0, [pw_9929_13377]
99 pmaddwd xmm6, xmm0, [pw_13377_0]
101 pmaddwd xmm0, [pw_15212_m13377]
102 pmaddwd xmm3, xmm2, [pw_15212_9929]
103 %if notcpuflag(ssse3)
104 pmaddwd xmm7, xmm2, [pw_m13377_13377]
106 pmaddwd xmm2, [pw_m5283_m15212]
115 %if notcpuflag(ssse3)
125 pmulhrsw m3, [pw_13377x2] ; out2
132 %if notcpuflag(ssse3)
135 movdq2q m0, xmm0 ; out3
136 movdq2q m1, xmm1 ; out0
137 movdq2q m2, xmm4 ; out1
138 %if notcpuflag(ssse3)
139 movdq2q m3, xmm6 ; out2