-
Notifications
You must be signed in to change notification settings - Fork 260
Expand file tree
/
Copy pathbitsetops_avx512_amd64.s
More file actions
363 lines (330 loc) · 6.86 KB
/
Copy pathbitsetops_avx512_amd64.s
File metadata and controls
363 lines (330 loc) · 6.86 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
//go:build amd64 && !appengine
// +build amd64,!appengine
#include "textflag.h"
// AVX-512 word operations on bitmap containers.
//
// Each routine computes dst[i] = a[i] op b[i] over a whole container, eight
// words at a time. The Card variants additionally return the population count
// of the result using VPOPCNTQ, so a caller that needs both the result and its
// cardinality makes a single pass over the container instead of two.
//
// dst may alias a or b: within an iteration the sources are loaded before the
// destination is stored, and iterations never touch each other's words.
//
// Each iteration handles four ZMM registers, i.e. 32 words; a scalar tail
// handles the trailing len%32 words, so any slice length works. The Card
// variants use four accumulators to keep the adds off one dependency chain.
//
// Go assembler conventions are as in popcnt_avx2_amd64.s: operands are written
// source(s) first and destination last, and a []uint64 argument is a
// {ptr,len,cap} header, so the second and third slices start at +24(FP) and
// +48(FP), and any result follows the arguments.
// HSUM512 horizontally sums the eight 64-bit lanes of Z4 into out.
#define HSUM512(out) \
VEXTRACTI64X4 $1, Z4, Y1 \
VPADDQ Y1, Y4, Y1 \
VEXTRACTI128 $1, Y1, X2 \
VPADDQ X2, X1, X1 \
VPSHUFD $0x4e, X1, X2 \
VPADDQ X2, X1, X1 \
VMOVQ X1, out
// func orSliceAVX512(dst, a, b []uint64)
TEXT ·orSliceAVX512(SB), NOSPLIT, $0-72
MOVQ dst_base+0(FP), DX
MOVQ a_base+24(FP), SI
MOVQ a_len+32(FP), BX
MOVQ b_base+48(FP), DI
MOVQ BX, CX
SHRQ $5, CX
TESTQ CX, CX
JZ or_tail
or_loop:
VMOVDQU64 (SI), Z0
VMOVDQU64 64(SI), Z1
VMOVDQU64 128(SI), Z2
VMOVDQU64 192(SI), Z3
VPORQ (DI), Z0, Z0
VPORQ 64(DI), Z1, Z1
VPORQ 128(DI), Z2, Z2
VPORQ 192(DI), Z3, Z3
VMOVDQU64 Z0, (DX)
VMOVDQU64 Z1, 64(DX)
VMOVDQU64 Z2, 128(DX)
VMOVDQU64 Z3, 192(DX)
ADDQ $256, SI
ADDQ $256, DI
ADDQ $256, DX
DECQ CX
JNZ or_loop
or_tail:
ANDQ $31, BX
JZ or_done
or_scalar:
MOVQ (SI), R9
ORQ (DI), R9
MOVQ R9, (DX)
ADDQ $8, SI
ADDQ $8, DI
ADDQ $8, DX
DECQ BX
JNZ or_scalar
or_done:
VZEROUPPER
RET
// func andSliceAVX512(dst, a, b []uint64)
TEXT ·andSliceAVX512(SB), NOSPLIT, $0-72
MOVQ dst_base+0(FP), DX
MOVQ a_base+24(FP), SI
MOVQ a_len+32(FP), BX
MOVQ b_base+48(FP), DI
MOVQ BX, CX
SHRQ $5, CX
TESTQ CX, CX
JZ and_tail
and_loop:
VMOVDQU64 (SI), Z0
VMOVDQU64 64(SI), Z1
VMOVDQU64 128(SI), Z2
VMOVDQU64 192(SI), Z3
VPANDQ (DI), Z0, Z0
VPANDQ 64(DI), Z1, Z1
VPANDQ 128(DI), Z2, Z2
VPANDQ 192(DI), Z3, Z3
VMOVDQU64 Z0, (DX)
VMOVDQU64 Z1, 64(DX)
VMOVDQU64 Z2, 128(DX)
VMOVDQU64 Z3, 192(DX)
ADDQ $256, SI
ADDQ $256, DI
ADDQ $256, DX
DECQ CX
JNZ and_loop
and_tail:
ANDQ $31, BX
JZ and_done
and_scalar:
MOVQ (SI), R9
ANDQ (DI), R9
MOVQ R9, (DX)
ADDQ $8, SI
ADDQ $8, DI
ADDQ $8, DX
DECQ BX
JNZ and_scalar
and_done:
VZEROUPPER
RET
// func xorSliceAVX512(dst, a, b []uint64)
TEXT ·xorSliceAVX512(SB), NOSPLIT, $0-72
MOVQ dst_base+0(FP), DX
MOVQ a_base+24(FP), SI
MOVQ a_len+32(FP), BX
MOVQ b_base+48(FP), DI
MOVQ BX, CX
SHRQ $5, CX
TESTQ CX, CX
JZ xor_tail
xor_loop:
VMOVDQU64 (SI), Z0
VMOVDQU64 64(SI), Z1
VMOVDQU64 128(SI), Z2
VMOVDQU64 192(SI), Z3
VPXORQ (DI), Z0, Z0
VPXORQ 64(DI), Z1, Z1
VPXORQ 128(DI), Z2, Z2
VPXORQ 192(DI), Z3, Z3
VMOVDQU64 Z0, (DX)
VMOVDQU64 Z1, 64(DX)
VMOVDQU64 Z2, 128(DX)
VMOVDQU64 Z3, 192(DX)
ADDQ $256, SI
ADDQ $256, DI
ADDQ $256, DX
DECQ CX
JNZ xor_loop
xor_tail:
ANDQ $31, BX
JZ xor_done
xor_scalar:
MOVQ (SI), R9
XORQ (DI), R9
MOVQ R9, (DX)
ADDQ $8, SI
ADDQ $8, DI
ADDQ $8, DX
DECQ BX
JNZ xor_scalar
xor_done:
VZEROUPPER
RET
// func andNotSliceAVX512(dst, a, b []uint64)
// VPANDN negates its first source and only the second may come from
// memory, so b goes into the register and a is read from memory:
// "VPANDNQ (SI), Zb, Zb" gives (NOT b) AND a = a &^ b.
TEXT ·andNotSliceAVX512(SB), NOSPLIT, $0-72
MOVQ dst_base+0(FP), DX
MOVQ a_base+24(FP), SI
MOVQ a_len+32(FP), BX
MOVQ b_base+48(FP), DI
MOVQ BX, CX
SHRQ $5, CX
TESTQ CX, CX
JZ andnot_tail
andnot_loop:
VMOVDQU64 (DI), Z0
VMOVDQU64 64(DI), Z1
VMOVDQU64 128(DI), Z2
VMOVDQU64 192(DI), Z3
VPANDNQ (SI), Z0, Z0
VPANDNQ 64(SI), Z1, Z1
VPANDNQ 128(SI), Z2, Z2
VPANDNQ 192(SI), Z3, Z3
VMOVDQU64 Z0, (DX)
VMOVDQU64 Z1, 64(DX)
VMOVDQU64 Z2, 128(DX)
VMOVDQU64 Z3, 192(DX)
ADDQ $256, SI
ADDQ $256, DI
ADDQ $256, DX
DECQ CX
JNZ andnot_loop
andnot_tail:
ANDQ $31, BX
JZ andnot_done
andnot_scalar:
MOVQ (DI), R9
NOTQ R9
ANDQ (SI), R9
MOVQ R9, (DX)
ADDQ $8, SI
ADDQ $8, DI
ADDQ $8, DX
DECQ BX
JNZ andnot_scalar
andnot_done:
VZEROUPPER
RET
// func orCardSliceAVX512(dst, a, b []uint64) uint64
TEXT ·orCardSliceAVX512(SB), NOSPLIT, $0-80
MOVQ dst_base+0(FP), DX
MOVQ a_base+24(FP), SI
MOVQ a_len+32(FP), BX
MOVQ b_base+48(FP), DI
VPXORQ Z4, Z4, Z4
VPXORQ Z5, Z5, Z5
VPXORQ Z6, Z6, Z6
VPXORQ Z7, Z7, Z7
MOVQ BX, CX
SHRQ $5, CX
TESTQ CX, CX
JZ orc_tail
orc_loop:
VMOVDQU64 (SI), Z0
VMOVDQU64 64(SI), Z1
VMOVDQU64 128(SI), Z2
VMOVDQU64 192(SI), Z3
VPORQ (DI), Z0, Z0
VMOVDQU64 Z0, (DX)
VPOPCNTQ Z0, Z0
VPADDQ Z0, Z4, Z4
VPORQ 64(DI), Z1, Z1
VMOVDQU64 Z1, 64(DX)
VPOPCNTQ Z1, Z1
VPADDQ Z1, Z5, Z5
VPORQ 128(DI), Z2, Z2
VMOVDQU64 Z2, 128(DX)
VPOPCNTQ Z2, Z2
VPADDQ Z2, Z6, Z6
VPORQ 192(DI), Z3, Z3
VMOVDQU64 Z3, 192(DX)
VPOPCNTQ Z3, Z3
VPADDQ Z3, Z7, Z7
ADDQ $256, SI
ADDQ $256, DI
ADDQ $256, DX
DECQ CX
JNZ orc_loop
orc_tail:
VPADDQ Z5, Z4, Z4
VPADDQ Z7, Z6, Z6
VPADDQ Z6, Z4, Z4
HSUM512(AX)
ANDQ $31, BX
JZ orc_done
orc_scalar:
MOVQ (SI), R9
ORQ (DI), R9
MOVQ R9, (DX)
POPCNTQ R9, R9
ADDQ R9, AX
ADDQ $8, SI
ADDQ $8, DI
ADDQ $8, DX
DECQ BX
JNZ orc_scalar
orc_done:
VZEROUPPER
MOVQ AX, ret+72(FP)
RET
// func andCardSliceAVX512(dst, a, b []uint64) uint64
TEXT ·andCardSliceAVX512(SB), NOSPLIT, $0-80
MOVQ dst_base+0(FP), DX
MOVQ a_base+24(FP), SI
MOVQ a_len+32(FP), BX
MOVQ b_base+48(FP), DI
VPXORQ Z4, Z4, Z4
VPXORQ Z5, Z5, Z5
VPXORQ Z6, Z6, Z6
VPXORQ Z7, Z7, Z7
MOVQ BX, CX
SHRQ $5, CX
TESTQ CX, CX
JZ andc_tail
andc_loop:
VMOVDQU64 (SI), Z0
VMOVDQU64 64(SI), Z1
VMOVDQU64 128(SI), Z2
VMOVDQU64 192(SI), Z3
VPANDQ (DI), Z0, Z0
VMOVDQU64 Z0, (DX)
VPOPCNTQ Z0, Z0
VPADDQ Z0, Z4, Z4
VPANDQ 64(DI), Z1, Z1
VMOVDQU64 Z1, 64(DX)
VPOPCNTQ Z1, Z1
VPADDQ Z1, Z5, Z5
VPANDQ 128(DI), Z2, Z2
VMOVDQU64 Z2, 128(DX)
VPOPCNTQ Z2, Z2
VPADDQ Z2, Z6, Z6
VPANDQ 192(DI), Z3, Z3
VMOVDQU64 Z3, 192(DX)
VPOPCNTQ Z3, Z3
VPADDQ Z3, Z7, Z7
ADDQ $256, SI
ADDQ $256, DI
ADDQ $256, DX
DECQ CX
JNZ andc_loop
andc_tail:
VPADDQ Z5, Z4, Z4
VPADDQ Z7, Z6, Z6
VPADDQ Z6, Z4, Z4
HSUM512(AX)
ANDQ $31, BX
JZ andc_done
andc_scalar:
MOVQ (SI), R9
ANDQ (DI), R9
MOVQ R9, (DX)
POPCNTQ R9, R9
ADDQ R9, AX
ADDQ $8, SI
ADDQ $8, DI
ADDQ $8, DX
DECQ BX
JNZ andc_scalar
andc_done:
VZEROUPPER
MOVQ AX, ret+72(FP)
RET