diff --git a/bitmapcontainer.go b/bitmapcontainer.go index 416da732..e97da14f 100644 --- a/bitmapcontainer.go +++ b/bitmapcontainer.go @@ -562,10 +562,7 @@ func (bc *bitmapContainer) orArrayCardinality(value2 *arrayContainer) int { func (bc *bitmapContainer) orBitmap(value2 *bitmapContainer) container { answer := newBitmapContainer() - for k := 0; k < len(answer.bitmap); k++ { - answer.bitmap[k] = bc.bitmap[k] | value2.bitmap[k] - } - answer.computeCardinality() + answer.cardinality = int(orCardSlice(answer.bitmap, bc.bitmap, value2.bitmap)) if answer.isFull() { return newRunContainer16Range(0, MaxUint16) } @@ -601,11 +598,7 @@ func (bc *bitmapContainer) iorArray(ac *arrayContainer) container { func (bc *bitmapContainer) iorBitmap(value2 *bitmapContainer) container { answer := bc - answer.cardinality = 0 - for k := 0; k < len(answer.bitmap); k++ { - answer.bitmap[k] = bc.bitmap[k] | value2.bitmap[k] - } - answer.computeCardinality() + answer.cardinality = int(orCardSlice(answer.bitmap, bc.bitmap, value2.bitmap)) if bc.isFull() { return newRunContainer16Range(0, MaxUint16) } @@ -654,16 +647,7 @@ func (bc *bitmapContainer) lazyIORBitmap(value2 *bitmapContainer) container { bitmap := answer.bitmap other := value2.bitmap - // Bitmap containers always span bitmapContainerSize words. Prove the - // bounds once so the compiler can eliminate the checks in the unrolled loop. - _ = bitmap[bitmapContainerSize-1] - _ = other[bitmapContainerSize-1] - for k := 0; k < bitmapContainerSize; k += 4 { - bitmap[k] |= other[k] - bitmap[k+1] |= other[k+1] - bitmap[k+2] |= other[k+2] - bitmap[k+3] |= other[k+3] - } + orSlice(bitmap, bitmap, other) answer.cardinality = invalidCardinality return answer } @@ -674,17 +658,7 @@ func (bc *bitmapContainer) lazyORBitmap(value2 *bitmapContainer) container { left := bc.bitmap right := value2.bitmap - // Bitmap containers always span bitmapContainerSize words. Prove the - // bounds once so the compiler can eliminate the checks in the unrolled loop. - _ = bitmap[bitmapContainerSize-1] - _ = left[bitmapContainerSize-1] - _ = right[bitmapContainerSize-1] - for k := 0; k < bitmapContainerSize; k += 4 { - bitmap[k] = left[k] | right[k] - bitmap[k+1] = left[k+1] | right[k+1] - bitmap[k+2] = left[k+2] | right[k+2] - bitmap[k+3] = left[k+3] | right[k+3] - } + orSlice(bitmap, left, right) answer.cardinality = invalidCardinality return answer } @@ -744,9 +718,7 @@ func (bc *bitmapContainer) xorBitmap(value2 *bitmapContainer) container { if newCardinality > arrayDefaultMaxSize { answer := newBitmapContainer() - for k := 0; k < len(answer.bitmap); k++ { - answer.bitmap[k] = bc.bitmap[k] ^ value2.bitmap[k] - } + xorSlice(answer.bitmap, bc.bitmap, value2.bitmap) answer.cardinality = newCardinality if answer.isFull() { return newRunContainer16Range(0, MaxUint16) @@ -868,9 +840,7 @@ func (bc *bitmapContainer) andBitmap(value2 *bitmapContainer) container { newcardinality := int(popcntAndSlice(bc.bitmap, value2.bitmap)) if newcardinality > arrayDefaultMaxSize { answer := newBitmapContainer() - for k := 0; k < len(answer.bitmap); k++ { - answer.bitmap[k] = bc.bitmap[k] & value2.bitmap[k] - } + andSlice(answer.bitmap, bc.bitmap, value2.bitmap) answer.cardinality = newcardinality return answer } @@ -901,10 +871,7 @@ func (bc *bitmapContainer) intersectsBitmap(value2 *bitmapContainer) bool { } func (bc *bitmapContainer) iandBitmap(value2 *bitmapContainer) container { - newcardinality := int(popcntAndSlice(bc.bitmap, value2.bitmap)) - for k := 0; k < len(bc.bitmap); k++ { - bc.bitmap[k] = bc.bitmap[k] & value2.bitmap[k] - } + newcardinality := int(andCardSlice(bc.bitmap, bc.bitmap, value2.bitmap)) bc.cardinality = newcardinality if newcardinality <= arrayDefaultMaxSize { @@ -938,9 +905,7 @@ func (bc *bitmapContainer) ixorRun16(value2 *runContainer16) container { func (bc *bitmapContainer) ixorBitmap(value2 *bitmapContainer) container { newCardinality := int(popcntXorSlice(bc.bitmap, value2.bitmap)) if newCardinality > arrayDefaultMaxSize { - for k := 0; k < len(bc.bitmap); k++ { - bc.bitmap[k] = bc.bitmap[k] ^ value2.bitmap[k] - } + xorSlice(bc.bitmap, bc.bitmap, value2.bitmap) bc.cardinality = newCardinality return bc } @@ -1064,9 +1029,7 @@ func (bc *bitmapContainer) andNotBitmap(value2 *bitmapContainer) container { newCardinality := int(popcntMaskSlice(bc.bitmap, value2.bitmap)) if newCardinality > arrayDefaultMaxSize { answer := newBitmapContainer() - for k := 0; k < len(answer.bitmap); k++ { - answer.bitmap[k] = bc.bitmap[k] &^ value2.bitmap[k] - } + andNotSlice(answer.bitmap, bc.bitmap, value2.bitmap) answer.cardinality = newCardinality return answer } @@ -1077,9 +1040,7 @@ func (bc *bitmapContainer) andNotBitmap(value2 *bitmapContainer) container { func (bc *bitmapContainer) iandNotBitmapSurely(value2 *bitmapContainer) container { newCardinality := int(popcntMaskSlice(bc.bitmap, value2.bitmap)) - for k := 0; k < len(bc.bitmap); k++ { - bc.bitmap[k] = bc.bitmap[k] &^ value2.bitmap[k] - } + andNotSlice(bc.bitmap, bc.bitmap, value2.bitmap) bc.cardinality = newCardinality if bc.getCardinality() <= arrayDefaultMaxSize { return bc.toArrayContainer() diff --git a/bitsetops.go b/bitsetops.go new file mode 100644 index 00000000..ffb579ef --- /dev/null +++ b/bitsetops.go @@ -0,0 +1,53 @@ +package roaring + +import "math/bits" + +// Portable implementations of the bitmap-container word operations. Each writes +// dst[i] = a[i] op b[i]; the Card variants also return the population count of +// the result, which callers would otherwise obtain with a second pass. +// +// dst may alias a or b: every element is read before it is written. + +func orSliceGo(dst, a, b []uint64) { + for i := range dst { + dst[i] = a[i] | b[i] + } +} + +func andSliceGo(dst, a, b []uint64) { + for i := range dst { + dst[i] = a[i] & b[i] + } +} + +func xorSliceGo(dst, a, b []uint64) { + for i := range dst { + dst[i] = a[i] ^ b[i] + } +} + +func andNotSliceGo(dst, a, b []uint64) { + for i := range dst { + dst[i] = a[i] &^ b[i] + } +} + +func orCardSliceGo(dst, a, b []uint64) uint64 { + card := 0 + for i := range dst { + v := a[i] | b[i] + dst[i] = v + card += bits.OnesCount64(v) + } + return uint64(card) +} + +func andCardSliceGo(dst, a, b []uint64) uint64 { + card := 0 + for i := range dst { + v := a[i] & b[i] + dst[i] = v + card += bits.OnesCount64(v) + } + return uint64(card) +} diff --git a/bitsetops_avx512_amd64.go b/bitsetops_avx512_amd64.go new file mode 100644 index 00000000..b25909cc --- /dev/null +++ b/bitsetops_avx512_amd64.go @@ -0,0 +1,85 @@ +//go:build amd64 && !appengine +// +build amd64,!appengine + +package roaring + +import "golang.org/x/sys/cpu" + +// The functions below are implemented in bitsetops_avx512_amd64.s. The Card +// variants fuse the write with the population count of the result, so a +// bitmap-container operation that needs both makes one pass over the container +// instead of two. + +//go:noescape +func orSliceAVX512(dst, a, b []uint64) + +//go:noescape +func andSliceAVX512(dst, a, b []uint64) + +//go:noescape +func xorSliceAVX512(dst, a, b []uint64) + +//go:noescape +func andNotSliceAVX512(dst, a, b []uint64) + +//go:noescape +func orCardSliceAVX512(dst, a, b []uint64) uint64 + +//go:noescape +func andCardSliceAVX512(dst, a, b []uint64) uint64 + +// useAVX512BitsetOps requires AVX512_VPOPCNTDQ because the fused kernels use +// VPOPCNTQ; the plain writes only need AVX512F, but they are gated together so +// a single flag governs the whole file. x/sys/cpu verifies operating-system +// support for the ZMM state and honors GODEBUG=cpu.avx512vpopcntdq=off. +var useAVX512BitsetOps = cpu.X86.HasAVX512VPOPCNTDQ + +func orSlice(dst, a, b []uint64) { + if useAVX512BitsetOps { + orSliceAVX512(dst, a, b) + return + } + orSliceGo(dst, a, b) +} + +func andSlice(dst, a, b []uint64) { + if useAVX512BitsetOps { + andSliceAVX512(dst, a, b) + return + } + andSliceGo(dst, a, b) +} + +func xorSlice(dst, a, b []uint64) { + if useAVX512BitsetOps { + xorSliceAVX512(dst, a, b) + return + } + xorSliceGo(dst, a, b) +} + +func andNotSlice(dst, a, b []uint64) { + if useAVX512BitsetOps { + andNotSliceAVX512(dst, a, b) + return + } + andNotSliceGo(dst, a, b) +} + +func orCardSlice(dst, a, b []uint64) uint64 { + if useAVX512BitsetOps { + return orCardSliceAVX512(dst, a, b) + } + // Without a vector population count the fused loop is no faster than the + // two passes the callers used before, and popcntSlice may itself be AVX2. + orSliceGo(dst, a, b) + return popcntSlice(dst) +} + +func andCardSlice(dst, a, b []uint64) uint64 { + if useAVX512BitsetOps { + return andCardSliceAVX512(dst, a, b) + } + andSliceGo(dst, a, b) + return popcntSlice(dst) +} diff --git a/bitsetops_avx512_amd64.s b/bitsetops_avx512_amd64.s new file mode 100644 index 00000000..5cd23435 --- /dev/null +++ b/bitsetops_avx512_amd64.s @@ -0,0 +1,363 @@ +//go:build amd64 && !appengine +// +build amd64,!appengine + +#include "textflag.h" + +// AVX-512 word operations on bitmap containers. +// +// Each routine computes dst[i] = a[i] op b[i] over a whole container, eight +// words at a time. The Card variants additionally return the population count +// of the result using VPOPCNTQ, so a caller that needs both the result and its +// cardinality makes a single pass over the container instead of two. +// +// dst may alias a or b: within an iteration the sources are loaded before the +// destination is stored, and iterations never touch each other's words. +// +// Each iteration handles four ZMM registers, i.e. 32 words; a scalar tail +// handles the trailing len%32 words, so any slice length works. The Card +// variants use four accumulators to keep the adds off one dependency chain. +// +// Go assembler conventions are as in popcnt_avx2_amd64.s: operands are written +// source(s) first and destination last, and a []uint64 argument is a +// {ptr,len,cap} header, so the second and third slices start at +24(FP) and +// +48(FP), and any result follows the arguments. + +// HSUM512 horizontally sums the eight 64-bit lanes of Z4 into out. +#define HSUM512(out) \ + VEXTRACTI64X4 $1, Z4, Y1 \ + VPADDQ Y1, Y4, Y1 \ + VEXTRACTI128 $1, Y1, X2 \ + VPADDQ X2, X1, X1 \ + VPSHUFD $0x4e, X1, X2 \ + VPADDQ X2, X1, X1 \ + VMOVQ X1, out + +// func orSliceAVX512(dst, a, b []uint64) +TEXT ·orSliceAVX512(SB), NOSPLIT, $0-72 + MOVQ dst_base+0(FP), DX + MOVQ a_base+24(FP), SI + MOVQ a_len+32(FP), BX + MOVQ b_base+48(FP), DI + MOVQ BX, CX + SHRQ $5, CX + TESTQ CX, CX + JZ or_tail + +or_loop: + VMOVDQU64 (SI), Z0 + VMOVDQU64 64(SI), Z1 + VMOVDQU64 128(SI), Z2 + VMOVDQU64 192(SI), Z3 + VPORQ (DI), Z0, Z0 + VPORQ 64(DI), Z1, Z1 + VPORQ 128(DI), Z2, Z2 + VPORQ 192(DI), Z3, Z3 + VMOVDQU64 Z0, (DX) + VMOVDQU64 Z1, 64(DX) + VMOVDQU64 Z2, 128(DX) + VMOVDQU64 Z3, 192(DX) + ADDQ $256, SI + ADDQ $256, DI + ADDQ $256, DX + DECQ CX + JNZ or_loop + +or_tail: + ANDQ $31, BX + JZ or_done + +or_scalar: + MOVQ (SI), R9 + ORQ (DI), R9 + MOVQ R9, (DX) + ADDQ $8, SI + ADDQ $8, DI + ADDQ $8, DX + DECQ BX + JNZ or_scalar + +or_done: + VZEROUPPER + RET + +// func andSliceAVX512(dst, a, b []uint64) +TEXT ·andSliceAVX512(SB), NOSPLIT, $0-72 + MOVQ dst_base+0(FP), DX + MOVQ a_base+24(FP), SI + MOVQ a_len+32(FP), BX + MOVQ b_base+48(FP), DI + MOVQ BX, CX + SHRQ $5, CX + TESTQ CX, CX + JZ and_tail + +and_loop: + VMOVDQU64 (SI), Z0 + VMOVDQU64 64(SI), Z1 + VMOVDQU64 128(SI), Z2 + VMOVDQU64 192(SI), Z3 + VPANDQ (DI), Z0, Z0 + VPANDQ 64(DI), Z1, Z1 + VPANDQ 128(DI), Z2, Z2 + VPANDQ 192(DI), Z3, Z3 + VMOVDQU64 Z0, (DX) + VMOVDQU64 Z1, 64(DX) + VMOVDQU64 Z2, 128(DX) + VMOVDQU64 Z3, 192(DX) + ADDQ $256, SI + ADDQ $256, DI + ADDQ $256, DX + DECQ CX + JNZ and_loop + +and_tail: + ANDQ $31, BX + JZ and_done + +and_scalar: + MOVQ (SI), R9 + ANDQ (DI), R9 + MOVQ R9, (DX) + ADDQ $8, SI + ADDQ $8, DI + ADDQ $8, DX + DECQ BX + JNZ and_scalar + +and_done: + VZEROUPPER + RET + +// func xorSliceAVX512(dst, a, b []uint64) +TEXT ·xorSliceAVX512(SB), NOSPLIT, $0-72 + MOVQ dst_base+0(FP), DX + MOVQ a_base+24(FP), SI + MOVQ a_len+32(FP), BX + MOVQ b_base+48(FP), DI + MOVQ BX, CX + SHRQ $5, CX + TESTQ CX, CX + JZ xor_tail + +xor_loop: + VMOVDQU64 (SI), Z0 + VMOVDQU64 64(SI), Z1 + VMOVDQU64 128(SI), Z2 + VMOVDQU64 192(SI), Z3 + VPXORQ (DI), Z0, Z0 + VPXORQ 64(DI), Z1, Z1 + VPXORQ 128(DI), Z2, Z2 + VPXORQ 192(DI), Z3, Z3 + VMOVDQU64 Z0, (DX) + VMOVDQU64 Z1, 64(DX) + VMOVDQU64 Z2, 128(DX) + VMOVDQU64 Z3, 192(DX) + ADDQ $256, SI + ADDQ $256, DI + ADDQ $256, DX + DECQ CX + JNZ xor_loop + +xor_tail: + ANDQ $31, BX + JZ xor_done + +xor_scalar: + MOVQ (SI), R9 + XORQ (DI), R9 + MOVQ R9, (DX) + ADDQ $8, SI + ADDQ $8, DI + ADDQ $8, DX + DECQ BX + JNZ xor_scalar + +xor_done: + VZEROUPPER + RET + +// func andNotSliceAVX512(dst, a, b []uint64) +// VPANDN negates its first source and only the second may come from +// memory, so b goes into the register and a is read from memory: +// "VPANDNQ (SI), Zb, Zb" gives (NOT b) AND a = a &^ b. +TEXT ·andNotSliceAVX512(SB), NOSPLIT, $0-72 + MOVQ dst_base+0(FP), DX + MOVQ a_base+24(FP), SI + MOVQ a_len+32(FP), BX + MOVQ b_base+48(FP), DI + MOVQ BX, CX + SHRQ $5, CX + TESTQ CX, CX + JZ andnot_tail + +andnot_loop: + VMOVDQU64 (DI), Z0 + VMOVDQU64 64(DI), Z1 + VMOVDQU64 128(DI), Z2 + VMOVDQU64 192(DI), Z3 + VPANDNQ (SI), Z0, Z0 + VPANDNQ 64(SI), Z1, Z1 + VPANDNQ 128(SI), Z2, Z2 + VPANDNQ 192(SI), Z3, Z3 + VMOVDQU64 Z0, (DX) + VMOVDQU64 Z1, 64(DX) + VMOVDQU64 Z2, 128(DX) + VMOVDQU64 Z3, 192(DX) + ADDQ $256, SI + ADDQ $256, DI + ADDQ $256, DX + DECQ CX + JNZ andnot_loop + +andnot_tail: + ANDQ $31, BX + JZ andnot_done + +andnot_scalar: + MOVQ (DI), R9 + NOTQ R9 + ANDQ (SI), R9 + MOVQ R9, (DX) + ADDQ $8, SI + ADDQ $8, DI + ADDQ $8, DX + DECQ BX + JNZ andnot_scalar + +andnot_done: + VZEROUPPER + RET + +// func orCardSliceAVX512(dst, a, b []uint64) uint64 +TEXT ·orCardSliceAVX512(SB), NOSPLIT, $0-80 + MOVQ dst_base+0(FP), DX + MOVQ a_base+24(FP), SI + MOVQ a_len+32(FP), BX + MOVQ b_base+48(FP), DI + VPXORQ Z4, Z4, Z4 + VPXORQ Z5, Z5, Z5 + VPXORQ Z6, Z6, Z6 + VPXORQ Z7, Z7, Z7 + MOVQ BX, CX + SHRQ $5, CX + TESTQ CX, CX + JZ orc_tail + +orc_loop: + VMOVDQU64 (SI), Z0 + VMOVDQU64 64(SI), Z1 + VMOVDQU64 128(SI), Z2 + VMOVDQU64 192(SI), Z3 + VPORQ (DI), Z0, Z0 + VMOVDQU64 Z0, (DX) + VPOPCNTQ Z0, Z0 + VPADDQ Z0, Z4, Z4 + VPORQ 64(DI), Z1, Z1 + VMOVDQU64 Z1, 64(DX) + VPOPCNTQ Z1, Z1 + VPADDQ Z1, Z5, Z5 + VPORQ 128(DI), Z2, Z2 + VMOVDQU64 Z2, 128(DX) + VPOPCNTQ Z2, Z2 + VPADDQ Z2, Z6, Z6 + VPORQ 192(DI), Z3, Z3 + VMOVDQU64 Z3, 192(DX) + VPOPCNTQ Z3, Z3 + VPADDQ Z3, Z7, Z7 + ADDQ $256, SI + ADDQ $256, DI + ADDQ $256, DX + DECQ CX + JNZ orc_loop + +orc_tail: + VPADDQ Z5, Z4, Z4 + VPADDQ Z7, Z6, Z6 + VPADDQ Z6, Z4, Z4 + HSUM512(AX) + ANDQ $31, BX + JZ orc_done + +orc_scalar: + MOVQ (SI), R9 + ORQ (DI), R9 + MOVQ R9, (DX) + POPCNTQ R9, R9 + ADDQ R9, AX + ADDQ $8, SI + ADDQ $8, DI + ADDQ $8, DX + DECQ BX + JNZ orc_scalar + +orc_done: + VZEROUPPER + MOVQ AX, ret+72(FP) + RET + +// func andCardSliceAVX512(dst, a, b []uint64) uint64 +TEXT ·andCardSliceAVX512(SB), NOSPLIT, $0-80 + MOVQ dst_base+0(FP), DX + MOVQ a_base+24(FP), SI + MOVQ a_len+32(FP), BX + MOVQ b_base+48(FP), DI + VPXORQ Z4, Z4, Z4 + VPXORQ Z5, Z5, Z5 + VPXORQ Z6, Z6, Z6 + VPXORQ Z7, Z7, Z7 + MOVQ BX, CX + SHRQ $5, CX + TESTQ CX, CX + JZ andc_tail + +andc_loop: + VMOVDQU64 (SI), Z0 + VMOVDQU64 64(SI), Z1 + VMOVDQU64 128(SI), Z2 + VMOVDQU64 192(SI), Z3 + VPANDQ (DI), Z0, Z0 + VMOVDQU64 Z0, (DX) + VPOPCNTQ Z0, Z0 + VPADDQ Z0, Z4, Z4 + VPANDQ 64(DI), Z1, Z1 + VMOVDQU64 Z1, 64(DX) + VPOPCNTQ Z1, Z1 + VPADDQ Z1, Z5, Z5 + VPANDQ 128(DI), Z2, Z2 + VMOVDQU64 Z2, 128(DX) + VPOPCNTQ Z2, Z2 + VPADDQ Z2, Z6, Z6 + VPANDQ 192(DI), Z3, Z3 + VMOVDQU64 Z3, 192(DX) + VPOPCNTQ Z3, Z3 + VPADDQ Z3, Z7, Z7 + ADDQ $256, SI + ADDQ $256, DI + ADDQ $256, DX + DECQ CX + JNZ andc_loop + +andc_tail: + VPADDQ Z5, Z4, Z4 + VPADDQ Z7, Z6, Z6 + VPADDQ Z6, Z4, Z4 + HSUM512(AX) + ANDQ $31, BX + JZ andc_done + +andc_scalar: + MOVQ (SI), R9 + ANDQ (DI), R9 + MOVQ R9, (DX) + POPCNTQ R9, R9 + ADDQ R9, AX + ADDQ $8, SI + ADDQ $8, DI + ADDQ $8, DX + DECQ BX + JNZ andc_scalar + +andc_done: + VZEROUPPER + MOVQ AX, ret+72(FP) + RET diff --git a/bitsetops_generic.go b/bitsetops_generic.go new file mode 100644 index 00000000..938e6fb4 --- /dev/null +++ b/bitsetops_generic.go @@ -0,0 +1,12 @@ +//go:build !amd64 || appengine +// +build !amd64 appengine + +package roaring + +func orSlice(dst, a, b []uint64) { orSliceGo(dst, a, b) } +func andSlice(dst, a, b []uint64) { andSliceGo(dst, a, b) } +func xorSlice(dst, a, b []uint64) { xorSliceGo(dst, a, b) } +func andNotSlice(dst, a, b []uint64) { andNotSliceGo(dst, a, b) } + +func orCardSlice(dst, a, b []uint64) uint64 { return orCardSliceGo(dst, a, b) } +func andCardSlice(dst, a, b []uint64) uint64 { return andCardSliceGo(dst, a, b) } diff --git a/bitsetops_test.go b/bitsetops_test.go new file mode 100644 index 00000000..8d277e22 --- /dev/null +++ b/bitsetops_test.go @@ -0,0 +1,195 @@ +package roaring + +import ( + "math/rand" + "testing" + + "github.com/stretchr/testify/assert" +) + +// lengths exercise the vector main loop (multiples of 32 words), the scalar +// tail (len % 32 != 0), and the empty case. +var bitsetOpsTestLengths = []int{0, 1, 3, 8, 31, 32, 33, 64, 65, 1023, 1024, 1025} + +func randomWords(r *rand.Rand, n int) []uint64 { + s := make([]uint64, n) + for i := range s { + switch i % 4 { + case 0: + s[i] = r.Uint64() + case 1: + s[i] = 0 + case 2: + s[i] = ^uint64(0) + default: + s[i] = r.Uint64() & r.Uint64() + } + } + return s +} + +func TestBitsetOpsMatchGo(t *testing.T) { + r := rand.New(rand.NewSource(3)) + for _, n := range bitsetOpsTestLengths { + a := randomWords(r, n) + b := randomWords(r, n) + want := make([]uint64, n) + got := make([]uint64, n) + + orSliceGo(want, a, b) + orSlice(got, a, b) + assert.Equalf(t, want, got, "orSlice len=%d", n) + + andSliceGo(want, a, b) + andSlice(got, a, b) + assert.Equalf(t, want, got, "andSlice len=%d", n) + + xorSliceGo(want, a, b) + xorSlice(got, a, b) + assert.Equalf(t, want, got, "xorSlice len=%d", n) + + andNotSliceGo(want, a, b) + andNotSlice(got, a, b) + assert.Equalf(t, want, got, "andNotSlice len=%d", n) + + wantCard := orCardSliceGo(want, a, b) + gotCard := orCardSlice(got, a, b) + assert.Equalf(t, want, got, "orCardSlice result len=%d", n) + assert.Equalf(t, wantCard, gotCard, "orCardSlice cardinality len=%d", n) + + wantCard = andCardSliceGo(want, a, b) + gotCard = andCardSlice(got, a, b) + assert.Equalf(t, want, got, "andCardSlice result len=%d", n) + assert.Equalf(t, wantCard, gotCard, "andCardSlice cardinality len=%d", n) + } +} + +// The container operations pass the destination as one of the sources; make +// sure aliasing is handled. +func TestBitsetOpsAliasing(t *testing.T) { + r := rand.New(rand.NewSource(4)) + for _, n := range bitsetOpsTestLengths { + a := randomWords(r, n) + b := randomWords(r, n) + + // copyOf keeps the empty case an empty slice rather than nil, so the + // comparisons below stay about the contents. + copyOf := func(src []uint64) []uint64 { + out := make([]uint64, len(src)) + copy(out, src) + return out + } + want := make([]uint64, n) + orSliceGo(want, a, b) + dst := copyOf(a) + orSlice(dst, dst, b) + assert.Equalf(t, want, dst, "orSlice dst==a len=%d", n) + + andNotSliceGo(want, a, b) + dst = copyOf(a) + andNotSlice(dst, dst, b) + assert.Equalf(t, want, dst, "andNotSlice dst==a len=%d", n) + + wantCard := andCardSliceGo(want, a, b) + dst = copyOf(a) + gotCard := andCardSlice(dst, dst, b) + assert.Equalf(t, want, dst, "andCardSlice dst==a len=%d", n) + assert.Equalf(t, wantCard, gotCard, "andCardSlice dst==a cardinality len=%d", n) + + // destination aliasing the second source + orSliceGo(want, a, b) + dst = copyOf(b) + orSlice(dst, a, dst) + assert.Equalf(t, want, dst, "orSlice dst==b len=%d", n) + } +} + +func benchBitsetOp(b *testing.B, fn func(dst, x, y []uint64)) { + r := rand.New(rand.NewSource(1)) + x := randomWords(r, bitmapContainerSize) + y := randomWords(r, bitmapContainerSize) + dst := make([]uint64, bitmapContainerSize) + b.SetBytes(int64(bitmapContainerSize * 8)) + for b.Loop() { + fn(dst, x, y) + } +} + +func BenchmarkOrSlice(b *testing.B) { benchBitsetOp(b, orSlice) } +func BenchmarkAndSlice(b *testing.B) { benchBitsetOp(b, andSlice) } + +func BenchmarkOrCardSlice(b *testing.B) { + r := rand.New(rand.NewSource(1)) + x := randomWords(r, bitmapContainerSize) + y := randomWords(r, bitmapContainerSize) + dst := make([]uint64, bitmapContainerSize) + var sink uint64 + b.SetBytes(int64(bitmapContainerSize * 8)) + for b.Loop() { + sink = orCardSlice(dst, x, y) + } + _ = sink +} + +// The shape the container code used before: write with a scalar loop, then +// make a second pass to count. +func BenchmarkOrThenCountTwoPass(b *testing.B) { + r := rand.New(rand.NewSource(1)) + x := randomWords(r, bitmapContainerSize) + y := randomWords(r, bitmapContainerSize) + dst := make([]uint64, bitmapContainerSize) + var sink uint64 + b.SetBytes(int64(bitmapContainerSize * 8)) + for b.Loop() { + orSliceGo(dst, x, y) + sink = popcntSlice(dst) + } + _ = sink +} + +// two bitmaps of dense bitmap containers: the case the bitmap-container word +// operations actually govern +func denseBitmapPair(containers int) (*Bitmap, *Bitmap) { + r := rand.New(rand.NewSource(9)) + mk := func() *Bitmap { + words := make([]uint64, bitmapContainerSize*containers) + for i := range words { + words[i] = r.Uint64() + } + return FromDense(words, false) + } + return mk(), mk() +} + +func BenchmarkDenseBitmapOps(b *testing.B) { + x, y := denseBitmapPair(64) + b.Run("Or", func(b *testing.B) { + for b.Loop() { + sinkU += Or(x, y).GetCardinality() + } + }) + b.Run("And", func(b *testing.B) { + for b.Loop() { + sinkU += And(x, y).GetCardinality() + } + }) + b.Run("Xor", func(b *testing.B) { + for b.Loop() { + sinkU += Xor(x, y).GetCardinality() + } + }) + b.Run("AndNot", func(b *testing.B) { + for b.Loop() { + sinkU += AndNot(x, y).GetCardinality() + } + }) + b.Run("IOr", func(b *testing.B) { + for b.Loop() { + z := x.Clone() + z.Or(y) + sinkU += z.GetCardinality() + } + }) +} + +var sinkU uint64 diff --git a/go.mod b/go.mod index 34ab80ef..89089272 100644 --- a/go.mod +++ b/go.mod @@ -9,6 +9,7 @@ require ( github.com/google/uuid v1.6.0 github.com/mschoch/smat v0.2.0 github.com/stretchr/testify v1.11.1 + golang.org/x/sys v0.30.0 ) require ( diff --git a/go.sum b/go.sum index 4666bde7..a5482567 100644 --- a/go.sum +++ b/go.sum @@ -18,6 +18,8 @@ golang.org/x/mod v0.23.0 h1:Zb7khfcRGKk+kqfxFaP5tZqCnDZMjC5VtUBs87Hr6QM= golang.org/x/mod v0.23.0/go.mod h1:6SkKJ3Xj0I0BrPOZoBy3bdMptDDU9oJrpohJ3eWZ1fY= golang.org/x/sync v0.11.0 h1:GGz8+XQP4FvTTrjZPzNKTMFtSXH80RAzG+5ghFPgK9w= golang.org/x/sync v0.11.0/go.mod h1:Czt+wKu1gCyEFDUtn0jG5QVvpJ6rzVqr5aXyt9drQfk= +golang.org/x/sys v0.30.0 h1:QjkSwP/36a20jFYWkSue1YwXzLmsV5Gfq7Eiy72C1uc= +golang.org/x/sys v0.30.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA= golang.org/x/text v0.22.0 h1:bofq7m3/HAFvbF51jz3Q9wLg3jkvSPuiZu/pD1XwgtM= golang.org/x/text v0.22.0/go.mod h1:YRoo4H8PVmsu+E3Ou7cqLVH8oXWIHVoX0jqUWALQhfY= golang.org/x/tools v0.30.0 h1:BgcpHewrV5AUp2G9MebG4XPFI1E2W41zU1SaqVA9vJY=