aboutsummaryrefslogtreecommitdiffstats
path: root/libavutil/riscv/asm.S
blob: 1e6358dcb517b5ee4555887b4925af50ecf9d59f (plain) (blame)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
/*
 * Copyright © 2022 Rémi Denis-Courmont.
 * Loosely based on earlier work copyrighted by Måns Rullgård, 2008.
 *
 * This file is part of FFmpeg.
 *
 * FFmpeg is free software; you can redistribute it and/or
 * modify it under the terms of the GNU Lesser General Public
 * License as published by the Free Software Foundation; either
 * version 2.1 of the License, or (at your option) any later version.
 *
 * FFmpeg is distributed in the hope that it will be useful,
 * but WITHOUT ANY WARRANTY; without even the implied warranty of
 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
 * Lesser General Public License for more details.
 *
 * You should have received a copy of the GNU Lesser General Public
 * License along with FFmpeg; if not, write to the Free Software
 * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
 */

#if defined (__riscv_float_abi_soft)
#define NOHWF
#define NOHWD
#define HWF   #
#define HWD   #
#elif defined (__riscv_float_abi_single)
#define NOHWF #
#define NOHWD
#define HWF
#define HWD   #
#else
#define NOHWF #
#define NOHWD #
#define HWF
#define HWD
#endif

        .macro func sym, ext1=, ext2=
            .text
            .align 2

            .option push
            .ifnb \ext1
            .option arch, +\ext1
            .ifnb \ext2
            .option arch, +\ext2
            .endif
            .endif

            .global \sym
            .hidden \sym
            .type   \sym, %function
            \sym:

            .macro endfunc
                .size   \sym, . - \sym
                .option pop
                .previous
                .purgem endfunc
            .endm
        .endm

        .macro const sym, align=3, relocate=0
            .if \relocate
                .pushsection .data.rel.ro
            .else
                .pushsection .rodata
            .endif
            .align \align
            \sym:

            .macro endconst
                .size  \sym, . - \sym
                .popsection
                .purgem endconst
            .endm
        .endm

#if !defined (__riscv_zba)
        /* SH{1,2,3}ADD definitions for pre-Zba assemblers */
        .macro  shnadd n, rd, rs1, rs2
        .insn r OP, 2 * \n, 16, \rd, \rs1, \rs2
        .endm

        .macro  sh1add rd, rs1, rs2
        shnadd  1, \rd, \rs1, \rs2
        .endm

        .macro  sh2add rd, rs1, rs2
        shnadd  2, \rd, \rs1, \rs2
        .endm

        .macro  sh3add rd, rs1, rs2
        shnadd  3, \rd, \rs1, \rs2
        .endm
#endif

#if defined (__riscv_v_elen)
# define RV_V_ELEN __riscv_v_elen
#else
/* Run-time detection of the V extension implies ELEN >= 64. */
# define RV_V_ELEN 64
#endif
#if RV_V_ELEN == 32
# define VSEW_MAX 2
#else
# define VSEW_MAX 3
#endif

        .macro  parse_vtype ew, tp, mp
        .ifc    \ew,e8
        .equ    vsew, 0
        .else
        .ifc    \ew,e16
        .equ    vsew, 1
        .else
        .ifc    \ew,e32
        .equ    vsew, 2
        .else
        .ifc    \ew,e64
        .equ    vsew, 3
        .else
        .error  "Unknown element width \ew"
        .endif
        .endif
        .endif
        .endif

        .ifc    \tp,tu
        .equ    tp, 0
        .else
        .ifc    \tp,ta
        .equ    tp, 1
        .else
        .error  "Unknown tail policy \tp"
        .endif
        .endif

        .ifc    \mp,mu
        .equ    mp, 0
        .else
        .ifc    \mp,ma
        .equ    mp, 1
        .else
        .error  "Unknown mask policy \mp"
        .endif
        .endif
        .endm

        /**
         * Gets the vector type with the smallest suitable LMUL value.
         * @param[out] rd vector type destination register
         * @param vl vector length constant
         * @param ew element width: e8, e16, e32 or e64
         * @param tp tail policy: tu or ta
         * @param mp mask policty: mu or ma
         */
        .macro  vtype_ivli rd, avl, ew, tp=tu, mp=mu
        .if     \avl <= 1
        .equ    log2vl, 0
        .elseif \avl <= 2
        .equ    log2vl, 1
        .elseif \avl <= 4
        .equ    log2vl, 2
        .elseif \avl <= 8
        .equ    log2vl, 3
        .elseif \avl <= 16
        .equ    log2vl, 4
        .elseif \avl <= 32
        .equ    log2vl, 5
        .elseif \avl <= 64
        .equ    log2vl, 6
        .elseif \avl <= 128
        .equ    log2vl, 7
        .else
        .error  "Vector length \avl out of range"
        .endif
        parse_vtype \ew, \tp, \mp
        csrr    \rd, vlenb
        clz     \rd, \rd
        addi    \rd, \rd, log2vl + 1 + VSEW_MAX - __riscv_xlen
        max     \rd, \rd, zero // VLMUL must be >= VSEW - VSEW_MAX
        .if     vsew < VSEW_MAX
        addi    \rd, \rd, vsew - VSEW_MAX
        andi    \rd, \rd, 7
        .endif
        ori     \rd, \rd, (vsew << 3) | (tp << 6) | (mp << 7)
        .endm

        /**
         * Gets the vector type with the smallest suitable LMUL value.
         * @param[out] rd vector type destination register
         * @param rs vector length source register
         * @param[out] tmp temporary register to be clobbered
         * @param ew element width: e8, e16, e32 or e64
         * @param tp tail policy: tu or ta
         * @param mp mask policty: mu or ma
         */
        .macro  vtype_vli rd, rs, tmp, ew, tp=tu, mp=mu
        parse_vtype \ew, \tp, \mp
        /*
         * The difference between the CLZ's notionally equals the VLMUL value
         * for 4-bit elements. But we want the value for SEW_MAX-bit elements.
         */
        slli    \tmp, \rs, 1 + VSEW_MAX
        csrr    \rd, vlenb
        addi    \tmp, \tmp, -1
        clz     \rd, \rd
        clz     \tmp, \tmp
        sub     \rd, \rd, \tmp
        max     \rd, \rd, zero // VLMUL must be >= VSEW - VSEW_MAX
        .if     vsew < VSEW_MAX
        addi    \rd, \rd, vsew - VSEW_MAX
        andi    \rd, \rd, 7
        .endif
        ori     \rd, \rd, (vsew << 3) | (tp << 6) | (mp << 7)
        .endm

        /**
         * Widens a vector type.
         * @param[out] rd widened vector type destination register
         * @param rs vector type source register
         * @param n number of times to widen (once by default)
         */
        .macro  vwtypei rd, rs, n=1
        xori    \rd, \rs, 4
        addi    \rd, \rd, (\n) * 011
        xori    \rd, \rd, 4
        .endm

        /**
         * Narrows a vector type.
         * @param[out] rd narrowed vector type destination register
         * @param rs vector type source register
         * @param n number of times to narrow (once by default)
         */
        .macro  vntypei rd, rs, n=1
        vwtypei \rd, \rs, -(\n)
        .endm