1cabdff1aSopenharmony_ci; Chinese AVS video (AVS1-P2, JiZhun profile) decoder
2cabdff1aSopenharmony_ci; Copyright (c) 2006  Stefan Gehrer <stefan.gehrer@gmx.de>
3cabdff1aSopenharmony_ci;
4cabdff1aSopenharmony_ci; MMX-optimized DSP functions, based on H.264 optimizations by
5cabdff1aSopenharmony_ci; Michael Niedermayer and Loren Merritt
6cabdff1aSopenharmony_ci; Conversion from gcc syntax to x264asm syntax with modifications
7cabdff1aSopenharmony_ci; by Ronald S. Bultje <rsbultje@gmail.com>
8cabdff1aSopenharmony_ci;
9cabdff1aSopenharmony_ci; This file is part of FFmpeg.
10cabdff1aSopenharmony_ci;
11cabdff1aSopenharmony_ci; FFmpeg is free software; you can redistribute it and/or
12cabdff1aSopenharmony_ci; modify it under the terms of the GNU Lesser General Public
13cabdff1aSopenharmony_ci; License as published by the Free Software Foundation; either
14cabdff1aSopenharmony_ci; version 2.1 of the License, or (at your option) any later version.
15cabdff1aSopenharmony_ci;
16cabdff1aSopenharmony_ci; FFmpeg is distributed in the hope that it will be useful,
17cabdff1aSopenharmony_ci; but WITHOUT ANY WARRANTY; without even the implied warranty of
18cabdff1aSopenharmony_ci; MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
19cabdff1aSopenharmony_ci; Lesser General Public License for more details.
20cabdff1aSopenharmony_ci;
21cabdff1aSopenharmony_ci; You should have received a copy of the GNU Lesser General Public License
22cabdff1aSopenharmony_ci; along with FFmpeg; if not, write to the Free Software Foundation,
23cabdff1aSopenharmony_ci; Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
24cabdff1aSopenharmony_ci
25cabdff1aSopenharmony_ci%include "libavutil/x86/x86util.asm"
26cabdff1aSopenharmony_ci
27cabdff1aSopenharmony_cicextern pw_4
28cabdff1aSopenharmony_cicextern pw_64
29cabdff1aSopenharmony_ci
30cabdff1aSopenharmony_ciSECTION .text
31cabdff1aSopenharmony_ci
32cabdff1aSopenharmony_ci%macro CAVS_IDCT8_1D 2-3 1 ; source, round, init_load
33cabdff1aSopenharmony_ci%if %3 == 1
34cabdff1aSopenharmony_ci    mova            m4, [%1+7*16]       ; m4 = src7
35cabdff1aSopenharmony_ci    mova            m5, [%1+1*16]       ; m5 = src1
36cabdff1aSopenharmony_ci    mova            m2, [%1+5*16]       ; m2 = src5
37cabdff1aSopenharmony_ci    mova            m7, [%1+3*16]       ; m7 = src3
38cabdff1aSopenharmony_ci%else
39cabdff1aSopenharmony_ci    SWAP             1, 7
40cabdff1aSopenharmony_ci    SWAP             4, 6
41cabdff1aSopenharmony_ci%endif
42cabdff1aSopenharmony_ci    mova            m0, m4
43cabdff1aSopenharmony_ci    mova            m3, m5
44cabdff1aSopenharmony_ci    mova            m6, m2
45cabdff1aSopenharmony_ci    mova            m1, m7
46cabdff1aSopenharmony_ci
47cabdff1aSopenharmony_ci    paddw           m4, m4              ; m4 = 2*src7
48cabdff1aSopenharmony_ci    paddw           m3, m3              ; m3 = 2*src1
49cabdff1aSopenharmony_ci    paddw           m6, m6              ; m6 = 2*src5
50cabdff1aSopenharmony_ci    paddw           m1, m1              ; m1 = 2*src3
51cabdff1aSopenharmony_ci    paddw           m0, m4              ; m0 = 3*src7
52cabdff1aSopenharmony_ci    paddw           m5, m3              ; m5 = 3*src1
53cabdff1aSopenharmony_ci    paddw           m2, m6              ; m2 = 3*src5
54cabdff1aSopenharmony_ci    paddw           m7, m1              ; m7 = 3*src3
55cabdff1aSopenharmony_ci    psubw           m5, m4              ; m5 = 3*src1 - 2*src7 = a0
56cabdff1aSopenharmony_ci    paddw           m7, m6              ; m7 = 3*src3 - 2*src5 = a1
57cabdff1aSopenharmony_ci    psubw           m1, m2              ; m1 = 2*src3 - 3*src5 = a2
58cabdff1aSopenharmony_ci    paddw           m3, m0              ; m3 = 2*src1 - 3*src7 = a3
59cabdff1aSopenharmony_ci
60cabdff1aSopenharmony_ci    mova            m4, m5
61cabdff1aSopenharmony_ci    mova            m6, m7
62cabdff1aSopenharmony_ci    mova            m0, m3
63cabdff1aSopenharmony_ci    mova            m2, m1
64cabdff1aSopenharmony_ci    SUMSUB_BA     w, 7, 5               ; m7 = a0 + a1, m5 = a0 - a1
65cabdff1aSopenharmony_ci    paddw           m7, m3              ; m7 = a0 + a1 + a3
66cabdff1aSopenharmony_ci    paddw           m5, m1              ; m5 = a0 - a1 + a2
67cabdff1aSopenharmony_ci    paddw           m7, m7
68cabdff1aSopenharmony_ci    paddw           m5, m5
69cabdff1aSopenharmony_ci    paddw           m7, m6              ; m7 = b4
70cabdff1aSopenharmony_ci    paddw           m5, m4              ; m5 = b5
71cabdff1aSopenharmony_ci
72cabdff1aSopenharmony_ci    SUMSUB_BA     w, 1, 3               ; m1 = a3 + a2, m3 = a3 - a2
73cabdff1aSopenharmony_ci    psubw           m4, m1              ; m4 = a0 - a2 - a3
74cabdff1aSopenharmony_ci    mova            m1, m4              ; m1 = a0 - a2 - a3
75cabdff1aSopenharmony_ci    psubw           m3, m6              ; m3 = a3 - a2 - a1
76cabdff1aSopenharmony_ci    paddw           m1, m1
77cabdff1aSopenharmony_ci    paddw           m3, m3
78cabdff1aSopenharmony_ci    psubw           m1, m2              ; m1 = b7
79cabdff1aSopenharmony_ci    paddw           m3, m0              ; m3 = b6
80cabdff1aSopenharmony_ci
81cabdff1aSopenharmony_ci    mova            m2, [%1+2*16]       ; m2 = src2
82cabdff1aSopenharmony_ci    mova            m6, [%1+6*16]       ; m6 = src6
83cabdff1aSopenharmony_ci    mova            m4, m2
84cabdff1aSopenharmony_ci    mova            m0, m6
85cabdff1aSopenharmony_ci    psllw           m4, 2               ; m4 = 4*src2
86cabdff1aSopenharmony_ci    psllw           m6, 2               ; m6 = 4*src6
87cabdff1aSopenharmony_ci    paddw           m2, m4              ; m2 = 5*src2
88cabdff1aSopenharmony_ci    paddw           m0, m6              ; m0 = 5*src6
89cabdff1aSopenharmony_ci    paddw           m2, m2
90cabdff1aSopenharmony_ci    paddw           m0, m0
91cabdff1aSopenharmony_ci    psubw           m4, m0              ; m4 = 4*src2 - 10*src6 = a7
92cabdff1aSopenharmony_ci    paddw           m6, m2              ; m6 = 4*src6 + 10*src2 = a6
93cabdff1aSopenharmony_ci
94cabdff1aSopenharmony_ci    mova            m2, [%1+0*16]       ; m2 = src0
95cabdff1aSopenharmony_ci    mova            m0, [%1+4*16]       ; m0 = src4
96cabdff1aSopenharmony_ci    SUMSUB_BA     w, 0, 2               ; m0 = src0 + src4, m2 = src0 - src4
97cabdff1aSopenharmony_ci    psllw           m0, 3
98cabdff1aSopenharmony_ci    psllw           m2, 3
99cabdff1aSopenharmony_ci    paddw           m0, %2              ; add rounding bias
100cabdff1aSopenharmony_ci    paddw           m2, %2              ; add rounding bias
101cabdff1aSopenharmony_ci
102cabdff1aSopenharmony_ci    SUMSUB_BA     w, 6, 0               ; m6 = a4 + a6, m0 = a4 - a6
103cabdff1aSopenharmony_ci    SUMSUB_BA     w, 4, 2               ; m4 = a5 + a7, m2 = a5 - a7
104cabdff1aSopenharmony_ci    SUMSUB_BA     w, 7, 6               ; m7 = dst0, m6 = dst7
105cabdff1aSopenharmony_ci    SUMSUB_BA     w, 5, 4               ; m5 = dst1, m4 = dst6
106cabdff1aSopenharmony_ci    SUMSUB_BA     w, 3, 2               ; m3 = dst2, m2 = dst5
107cabdff1aSopenharmony_ci    SUMSUB_BA     w, 1, 0               ; m1 = dst3, m0 = dst4
108cabdff1aSopenharmony_ci%endmacro
109cabdff1aSopenharmony_ci
110cabdff1aSopenharmony_ciINIT_XMM sse2
111cabdff1aSopenharmony_cicglobal cavs_idct8, 2, 2, 8 + ARCH_X86_64, 0 - 8 * 16, out, in
112cabdff1aSopenharmony_ci    CAVS_IDCT8_1D  inq, [pw_4]
113cabdff1aSopenharmony_ci    psraw           m7, 3
114cabdff1aSopenharmony_ci    psraw           m6, 3
115cabdff1aSopenharmony_ci    psraw           m5, 3
116cabdff1aSopenharmony_ci    psraw           m4, 3
117cabdff1aSopenharmony_ci    psraw           m3, 3
118cabdff1aSopenharmony_ci    psraw           m2, 3
119cabdff1aSopenharmony_ci    psraw           m1, 3
120cabdff1aSopenharmony_ci    psraw           m0, 3
121cabdff1aSopenharmony_ci%if ARCH_X86_64
122cabdff1aSopenharmony_ci    TRANSPOSE8x8W    7, 5, 3, 1, 0, 2, 4, 6, 8
123cabdff1aSopenharmony_ci    mova    [rsp+4*16], m0
124cabdff1aSopenharmony_ci%else
125cabdff1aSopenharmony_ci    mova    [rsp+0*16], m4
126cabdff1aSopenharmony_ci    TRANSPOSE8x8W    7, 5, 3, 1, 0, 2, 4, 6, [rsp+0*16], [rsp+4*16], 1
127cabdff1aSopenharmony_ci%endif
128cabdff1aSopenharmony_ci    mova    [rsp+0*16], m7
129cabdff1aSopenharmony_ci    mova    [rsp+2*16], m3
130cabdff1aSopenharmony_ci    mova    [rsp+6*16], m4
131cabdff1aSopenharmony_ci    CAVS_IDCT8_1D  rsp, [pw_64], 0
132cabdff1aSopenharmony_ci    psraw           m7, 7
133cabdff1aSopenharmony_ci    psraw           m6, 7
134cabdff1aSopenharmony_ci    psraw           m5, 7
135cabdff1aSopenharmony_ci    psraw           m4, 7
136cabdff1aSopenharmony_ci    psraw           m3, 7
137cabdff1aSopenharmony_ci    psraw           m2, 7
138cabdff1aSopenharmony_ci    psraw           m1, 7
139cabdff1aSopenharmony_ci    psraw           m0, 7
140cabdff1aSopenharmony_ci
141cabdff1aSopenharmony_ci    mova   [outq+0*16], m7
142cabdff1aSopenharmony_ci    mova   [outq+1*16], m5
143cabdff1aSopenharmony_ci    mova   [outq+2*16], m3
144cabdff1aSopenharmony_ci    mova   [outq+3*16], m1
145cabdff1aSopenharmony_ci    mova   [outq+4*16], m0
146cabdff1aSopenharmony_ci    mova   [outq+5*16], m2
147cabdff1aSopenharmony_ci    mova   [outq+6*16], m4
148cabdff1aSopenharmony_ci    mova   [outq+7*16], m6
149cabdff1aSopenharmony_ci    RET
150