1cabdff1aSopenharmony_ci; Chinese AVS video (AVS1-P2, JiZhun profile) decoder 2cabdff1aSopenharmony_ci; Copyright (c) 2006 Stefan Gehrer <stefan.gehrer@gmx.de> 3cabdff1aSopenharmony_ci; 4cabdff1aSopenharmony_ci; MMX-optimized DSP functions, based on H.264 optimizations by 5cabdff1aSopenharmony_ci; Michael Niedermayer and Loren Merritt 6cabdff1aSopenharmony_ci; Conversion from gcc syntax to x264asm syntax with modifications 7cabdff1aSopenharmony_ci; by Ronald S. Bultje <rsbultje@gmail.com> 8cabdff1aSopenharmony_ci; 9cabdff1aSopenharmony_ci; This file is part of FFmpeg. 10cabdff1aSopenharmony_ci; 11cabdff1aSopenharmony_ci; FFmpeg is free software; you can redistribute it and/or 12cabdff1aSopenharmony_ci; modify it under the terms of the GNU Lesser General Public 13cabdff1aSopenharmony_ci; License as published by the Free Software Foundation; either 14cabdff1aSopenharmony_ci; version 2.1 of the License, or (at your option) any later version. 15cabdff1aSopenharmony_ci; 16cabdff1aSopenharmony_ci; FFmpeg is distributed in the hope that it will be useful, 17cabdff1aSopenharmony_ci; but WITHOUT ANY WARRANTY; without even the implied warranty of 18cabdff1aSopenharmony_ci; MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU 19cabdff1aSopenharmony_ci; Lesser General Public License for more details. 20cabdff1aSopenharmony_ci; 21cabdff1aSopenharmony_ci; You should have received a copy of the GNU Lesser General Public License 22cabdff1aSopenharmony_ci; along with FFmpeg; if not, write to the Free Software Foundation, 23cabdff1aSopenharmony_ci; Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA 24cabdff1aSopenharmony_ci 25cabdff1aSopenharmony_ci%include "libavutil/x86/x86util.asm" 26cabdff1aSopenharmony_ci 27cabdff1aSopenharmony_cicextern pw_4 28cabdff1aSopenharmony_cicextern pw_64 29cabdff1aSopenharmony_ci 30cabdff1aSopenharmony_ciSECTION .text 31cabdff1aSopenharmony_ci 32cabdff1aSopenharmony_ci%macro CAVS_IDCT8_1D 2-3 1 ; source, round, init_load 33cabdff1aSopenharmony_ci%if %3 == 1 34cabdff1aSopenharmony_ci mova m4, [%1+7*16] ; m4 = src7 35cabdff1aSopenharmony_ci mova m5, [%1+1*16] ; m5 = src1 36cabdff1aSopenharmony_ci mova m2, [%1+5*16] ; m2 = src5 37cabdff1aSopenharmony_ci mova m7, [%1+3*16] ; m7 = src3 38cabdff1aSopenharmony_ci%else 39cabdff1aSopenharmony_ci SWAP 1, 7 40cabdff1aSopenharmony_ci SWAP 4, 6 41cabdff1aSopenharmony_ci%endif 42cabdff1aSopenharmony_ci mova m0, m4 43cabdff1aSopenharmony_ci mova m3, m5 44cabdff1aSopenharmony_ci mova m6, m2 45cabdff1aSopenharmony_ci mova m1, m7 46cabdff1aSopenharmony_ci 47cabdff1aSopenharmony_ci paddw m4, m4 ; m4 = 2*src7 48cabdff1aSopenharmony_ci paddw m3, m3 ; m3 = 2*src1 49cabdff1aSopenharmony_ci paddw m6, m6 ; m6 = 2*src5 50cabdff1aSopenharmony_ci paddw m1, m1 ; m1 = 2*src3 51cabdff1aSopenharmony_ci paddw m0, m4 ; m0 = 3*src7 52cabdff1aSopenharmony_ci paddw m5, m3 ; m5 = 3*src1 53cabdff1aSopenharmony_ci paddw m2, m6 ; m2 = 3*src5 54cabdff1aSopenharmony_ci paddw m7, m1 ; m7 = 3*src3 55cabdff1aSopenharmony_ci psubw m5, m4 ; m5 = 3*src1 - 2*src7 = a0 56cabdff1aSopenharmony_ci paddw m7, m6 ; m7 = 3*src3 - 2*src5 = a1 57cabdff1aSopenharmony_ci psubw m1, m2 ; m1 = 2*src3 - 3*src5 = a2 58cabdff1aSopenharmony_ci paddw m3, m0 ; m3 = 2*src1 - 3*src7 = a3 59cabdff1aSopenharmony_ci 60cabdff1aSopenharmony_ci mova m4, m5 61cabdff1aSopenharmony_ci mova m6, m7 62cabdff1aSopenharmony_ci mova m0, m3 63cabdff1aSopenharmony_ci mova m2, m1 64cabdff1aSopenharmony_ci SUMSUB_BA w, 7, 5 ; m7 = a0 + a1, m5 = a0 - a1 65cabdff1aSopenharmony_ci paddw m7, m3 ; m7 = a0 + a1 + a3 66cabdff1aSopenharmony_ci paddw m5, m1 ; m5 = a0 - a1 + a2 67cabdff1aSopenharmony_ci paddw m7, m7 68cabdff1aSopenharmony_ci paddw m5, m5 69cabdff1aSopenharmony_ci paddw m7, m6 ; m7 = b4 70cabdff1aSopenharmony_ci paddw m5, m4 ; m5 = b5 71cabdff1aSopenharmony_ci 72cabdff1aSopenharmony_ci SUMSUB_BA w, 1, 3 ; m1 = a3 + a2, m3 = a3 - a2 73cabdff1aSopenharmony_ci psubw m4, m1 ; m4 = a0 - a2 - a3 74cabdff1aSopenharmony_ci mova m1, m4 ; m1 = a0 - a2 - a3 75cabdff1aSopenharmony_ci psubw m3, m6 ; m3 = a3 - a2 - a1 76cabdff1aSopenharmony_ci paddw m1, m1 77cabdff1aSopenharmony_ci paddw m3, m3 78cabdff1aSopenharmony_ci psubw m1, m2 ; m1 = b7 79cabdff1aSopenharmony_ci paddw m3, m0 ; m3 = b6 80cabdff1aSopenharmony_ci 81cabdff1aSopenharmony_ci mova m2, [%1+2*16] ; m2 = src2 82cabdff1aSopenharmony_ci mova m6, [%1+6*16] ; m6 = src6 83cabdff1aSopenharmony_ci mova m4, m2 84cabdff1aSopenharmony_ci mova m0, m6 85cabdff1aSopenharmony_ci psllw m4, 2 ; m4 = 4*src2 86cabdff1aSopenharmony_ci psllw m6, 2 ; m6 = 4*src6 87cabdff1aSopenharmony_ci paddw m2, m4 ; m2 = 5*src2 88cabdff1aSopenharmony_ci paddw m0, m6 ; m0 = 5*src6 89cabdff1aSopenharmony_ci paddw m2, m2 90cabdff1aSopenharmony_ci paddw m0, m0 91cabdff1aSopenharmony_ci psubw m4, m0 ; m4 = 4*src2 - 10*src6 = a7 92cabdff1aSopenharmony_ci paddw m6, m2 ; m6 = 4*src6 + 10*src2 = a6 93cabdff1aSopenharmony_ci 94cabdff1aSopenharmony_ci mova m2, [%1+0*16] ; m2 = src0 95cabdff1aSopenharmony_ci mova m0, [%1+4*16] ; m0 = src4 96cabdff1aSopenharmony_ci SUMSUB_BA w, 0, 2 ; m0 = src0 + src4, m2 = src0 - src4 97cabdff1aSopenharmony_ci psllw m0, 3 98cabdff1aSopenharmony_ci psllw m2, 3 99cabdff1aSopenharmony_ci paddw m0, %2 ; add rounding bias 100cabdff1aSopenharmony_ci paddw m2, %2 ; add rounding bias 101cabdff1aSopenharmony_ci 102cabdff1aSopenharmony_ci SUMSUB_BA w, 6, 0 ; m6 = a4 + a6, m0 = a4 - a6 103cabdff1aSopenharmony_ci SUMSUB_BA w, 4, 2 ; m4 = a5 + a7, m2 = a5 - a7 104cabdff1aSopenharmony_ci SUMSUB_BA w, 7, 6 ; m7 = dst0, m6 = dst7 105cabdff1aSopenharmony_ci SUMSUB_BA w, 5, 4 ; m5 = dst1, m4 = dst6 106cabdff1aSopenharmony_ci SUMSUB_BA w, 3, 2 ; m3 = dst2, m2 = dst5 107cabdff1aSopenharmony_ci SUMSUB_BA w, 1, 0 ; m1 = dst3, m0 = dst4 108cabdff1aSopenharmony_ci%endmacro 109cabdff1aSopenharmony_ci 110cabdff1aSopenharmony_ciINIT_XMM sse2 111cabdff1aSopenharmony_cicglobal cavs_idct8, 2, 2, 8 + ARCH_X86_64, 0 - 8 * 16, out, in 112cabdff1aSopenharmony_ci CAVS_IDCT8_1D inq, [pw_4] 113cabdff1aSopenharmony_ci psraw m7, 3 114cabdff1aSopenharmony_ci psraw m6, 3 115cabdff1aSopenharmony_ci psraw m5, 3 116cabdff1aSopenharmony_ci psraw m4, 3 117cabdff1aSopenharmony_ci psraw m3, 3 118cabdff1aSopenharmony_ci psraw m2, 3 119cabdff1aSopenharmony_ci psraw m1, 3 120cabdff1aSopenharmony_ci psraw m0, 3 121cabdff1aSopenharmony_ci%if ARCH_X86_64 122cabdff1aSopenharmony_ci TRANSPOSE8x8W 7, 5, 3, 1, 0, 2, 4, 6, 8 123cabdff1aSopenharmony_ci mova [rsp+4*16], m0 124cabdff1aSopenharmony_ci%else 125cabdff1aSopenharmony_ci mova [rsp+0*16], m4 126cabdff1aSopenharmony_ci TRANSPOSE8x8W 7, 5, 3, 1, 0, 2, 4, 6, [rsp+0*16], [rsp+4*16], 1 127cabdff1aSopenharmony_ci%endif 128cabdff1aSopenharmony_ci mova [rsp+0*16], m7 129cabdff1aSopenharmony_ci mova [rsp+2*16], m3 130cabdff1aSopenharmony_ci mova [rsp+6*16], m4 131cabdff1aSopenharmony_ci CAVS_IDCT8_1D rsp, [pw_64], 0 132cabdff1aSopenharmony_ci psraw m7, 7 133cabdff1aSopenharmony_ci psraw m6, 7 134cabdff1aSopenharmony_ci psraw m5, 7 135cabdff1aSopenharmony_ci psraw m4, 7 136cabdff1aSopenharmony_ci psraw m3, 7 137cabdff1aSopenharmony_ci psraw m2, 7 138cabdff1aSopenharmony_ci psraw m1, 7 139cabdff1aSopenharmony_ci psraw m0, 7 140cabdff1aSopenharmony_ci 141cabdff1aSopenharmony_ci mova [outq+0*16], m7 142cabdff1aSopenharmony_ci mova [outq+1*16], m5 143cabdff1aSopenharmony_ci mova [outq+2*16], m3 144cabdff1aSopenharmony_ci mova [outq+3*16], m1 145cabdff1aSopenharmony_ci mova [outq+4*16], m0 146cabdff1aSopenharmony_ci mova [outq+5*16], m2 147cabdff1aSopenharmony_ci mova [outq+6*16], m4 148cabdff1aSopenharmony_ci mova [outq+7*16], m6 149cabdff1aSopenharmony_ci RET 150