diff options
author | Wu Jianhua <jianhua.wu@intel.com> | 2022-03-11 15:52:10 +0800 |
---|---|---|
committer | Haihao Xiang <haihao.xiang@intel.com> | 2022-04-24 14:46:41 +0800 |
commit | c1790b60d643100266192c2bbaefb2c76eba6e5a (patch) | |
tree | 8fce0a70d09538379db19eb9fe80a97edec2e99e /libavcodec/x86/hevc_mc.asm | |
parent | d4cd8830bdac3e26c8e75cd92e574c159fecc4f7 (diff) | |
download | ffmpeg-c1790b60d643100266192c2bbaefb2c76eba6e5a.tar.gz |
avcodec/x86/hevc_mc: add qpel_h16_8_avx512icl
ff_hevc_put_hevc_qpel_h16_8_sse4 3290870
ff_hevc_put_hevc_qpel_h16_8_avx512icl 1730033
Reviewed-by: Henrik Gramner <henrik@gramner.com>
Signed-off-by: Wu Jianhua <jianhua.wu@intel.com>
Diffstat (limited to 'libavcodec/x86/hevc_mc.asm')
-rw-r--r-- | libavcodec/x86/hevc_mc.asm | 26 |
1 files changed, 25 insertions, 1 deletions
diff --git a/libavcodec/x86/hevc_mc.asm b/libavcodec/x86/hevc_mc.asm index 52fa3ec948..4e39cdd7fe 100644 --- a/libavcodec/x86/hevc_mc.asm +++ b/libavcodec/x86/hevc_mc.asm @@ -89,6 +89,7 @@ QPEL_TABLE 10, 8, w, avx2 QPEL_TABLE 8, 1, b, avx512icl_h QPEL_TABLE 8, 1, d, avx512icl_v +QPEL_TABLE 16, 1, b, avx512icl_h pb_qpel_shuffle_index: db 0, 1, 2, 3 db 1, 2, 3, 4 @@ -98,6 +99,14 @@ pb_qpel_shuffle_index: db 0, 1, 2, 3 db 5, 6, 7, 8 db 6, 7, 8, 9 db 7, 8, 9, 10 + db 8, 9, 10, 11 + db 9, 10, 11, 12 + db 10, 11, 12, 13 + db 11, 12, 13, 14 + db 12, 13, 14, 15 + db 13, 14, 15, 16 + db 14, 15, 16, 17 + db 15, 16, 17, 18 db 4, 5, 6, 7 db 5, 6, 7, 8 db 6, 7, 8, 9 @@ -106,6 +115,14 @@ pb_qpel_shuffle_index: db 0, 1, 2, 3 db 9, 10, 11, 12 db 10, 11, 12, 13 db 11, 12, 13, 14 + db 12, 13, 14, 15 + db 13, 14, 15, 16 + db 14, 15, 16, 17 + db 15, 16, 17, 18 + db 16, 17, 18, 19 + db 17, 18, 19, 20 + db 18, 19, 20, 21 + db 19, 20, 21, 22 SECTION .text @@ -1712,7 +1729,7 @@ HEVC_PUT_HEVC_QPEL_HV 16, 10 %macro QPEL_LOAD_SHUF 2 movu m%1, [pb_qpel_shuffle_index + 0] - movu m%2, [pb_qpel_shuffle_index + 32] + movu m%2, [pb_qpel_shuffle_index + 64] %endmacro ; required: m0-m5 @@ -1720,7 +1737,11 @@ HEVC_PUT_HEVC_QPEL_HV 16, 10 ; %2: name for src %macro QPEL_H_LOAD_COMPUTE 2 pxor m%1, m%1 +%if mmsize == 64 + movu ym4, [%2q - 3] +%else movu xm4, [%2q - 3] +%endif vpermb m5, m2, m4 vpermb m4, m3, m4 vpdpbusd m%1, m5, m0 @@ -1805,5 +1826,8 @@ INIT_YMM avx512icl HEVC_PUT_HEVC_QPEL_AVX512ICL 8, 8 HEVC_PUT_HEVC_QPEL_HV_AVX512ICL 8, 8 +INIT_ZMM avx512icl +HEVC_PUT_HEVC_QPEL_AVX512ICL 16, 8 + %endif %endif |