CI / pre-commit (pull_request) Successful in 1m59s
CI / release (arm64, , ubuntu-latest-arm64) (pull_request) Successful in 2m14s
CI / test (-DCMAKE_BUILD_TYPE=Debug -DMSAN_TOOLCHAIN_PATH=/opt/msan, debug) (pull_request) Successful in 3m50s
CI / test (-DCMAKE_CXX_FLAGS=-DUSE_64_BIT=1, 64-bit-versions) (pull_request) Successful in 3m17s
CI / test (-DCMAKE_C_COMPILER=gcc -DCMAKE_CXX_COMPILER=g++, gcc) (pull_request) Successful in 3m14s
CI / test (-DUSE_SIMD_FALLBACK=ON, simd-fallback) (pull_request) Successful in 3m16s
CI / release (amd64, -DMSAN_TOOLCHAIN_PATH=/opt/msan, ubuntu-latest-amd64) (pull_request) Successful in 5m33s
CI / coverage (pull_request) Successful in 3m44s
85 lines
3.7 KiB
ArmAsm
85 lines
3.7 KiB
ArmAsm
// SIMD operations on potentially-indeterminate Node16::index[16] bytes.
|
|
// Written in assembly because loading and operating on indeterminate values
|
|
// is undefined behavior in C++ ([basic.indet]) but well-defined in assembly.
|
|
// The caller is responsible for masking the returned bitfield to
|
|
// [0, numChildren) before using it.
|
|
//
|
|
// Unlike x86-64 (which has pmovmskb), AArch64 has no single instruction that
|
|
// produces a 1-bit-per-byte mask. Each function therefore returns a 64-bit
|
|
// "nibble mask" in x0: nibble i (bits [4i, 4i+4)) is 0xf iff the condition
|
|
// holds at index i. Bit (4i + 3) is the high bit of byte i's result. Callers
|
|
// locate a set lane with countr_zero(bitfield) / 4 and mask the valid lanes
|
|
// with (uint64_t(1) << (numChildren * 4)) - 1.
|
|
//
|
|
// AArch64 AAPCS:
|
|
// x0 = const uint8_t *idx (16 bytes, may contain indeterminate data)
|
|
// w1 = uint8_t key (find_eq_16, find_ge_16)
|
|
// w1 = uint8_t begin (mask_in_range_16)
|
|
// w2 = uint8_t end (mask_in_range_16)
|
|
//
|
|
// These functions are only ever called directly (never indirectly), so they
|
|
// do not need BTI landing pads; the object is still marked BTI/PAC/GCS-aware
|
|
// below so a -z force-bti link keeps BTI enabled for the whole binary.
|
|
|
|
.text
|
|
|
|
// uint64_t find_eq_16(const uint8_t idx[16], uint8_t key)
|
|
// nibble i = 0xf iff idx[i] == key
|
|
.globl find_eq_16
|
|
.type find_eq_16, %function
|
|
find_eq_16:
|
|
dup v1.16b, w1 // broadcast key
|
|
ldr q0, [x0] // load 16 bytes (may be indeterminate)
|
|
cmeq v0.16b, v0.16b, v1.16b // 0xff for each match
|
|
shrn v0.8b, v0.8h, 4 // pack 16 byte-flags into 8 nibble-pairs
|
|
umov x0, v0.d[0]
|
|
ret
|
|
.size find_eq_16, .-find_eq_16
|
|
|
|
// uint64_t find_ge_16(const uint8_t idx[16], uint8_t child)
|
|
// nibble i = 0xf iff idx[i] >= child (unsigned)
|
|
// cmhs gives unsigned ">=" (higher-or-same): Vd = Vn >= Vm.
|
|
.globl find_ge_16
|
|
.type find_ge_16, %function
|
|
find_ge_16:
|
|
dup v1.16b, w1 // broadcast child
|
|
ldr q0, [x0] // load 16 bytes
|
|
cmhs v0.16b, v0.16b, v1.16b // 0xff where idx[i] >= child (unsigned)
|
|
shrn v0.8b, v0.8h, 4
|
|
umov x0, v0.d[0]
|
|
ret
|
|
.size find_ge_16, .-find_ge_16
|
|
|
|
// uint64_t mask_in_range_16(const uint8_t idx[16], uint8_t begin, uint8_t end)
|
|
// nibble i = 0xf iff begin <= idx[i] < end (unsigned, wrapping arithmetic)
|
|
// Logic: (idx[i] - begin) < (end - begin), valid when end - begin < 256.
|
|
// cmhi gives unsigned ">" (higher): Vd = Vn > Vm. We want
|
|
// (end - begin) > (idx - begin), so Vn = (end - begin).
|
|
.globl mask_in_range_16
|
|
.type mask_in_range_16, %function
|
|
mask_in_range_16:
|
|
dup v1.16b, w1 // broadcast begin
|
|
dup v2.16b, w2 // broadcast end
|
|
ldr q0, [x0] // load 16 bytes
|
|
sub v0.16b, v0.16b, v1.16b // idx - begin (wrapping)
|
|
sub v2.16b, v2.16b, v1.16b // end - begin (range size)
|
|
cmhi v0.16b, v2.16b, v0.16b // 0xff where (end-begin) > (idx-begin)
|
|
shrn v0.8b, v0.8h, 4
|
|
umov x0, v0.d[0]
|
|
ret
|
|
.size mask_in_range_16, .-mask_in_range_16
|
|
|
|
// Declare AArch64 branch-protection compatibility, matching what the
|
|
// compiler emits for -mbranch-protection=standard (BTI + PAC + GCS). This
|
|
// keeps the object indistinguishable from C/C++ translation units for
|
|
// linkers enforcing BTI (-z force-bti). The functions above are only ever
|
|
// called directly, so they need no BTI landing pads; PAC/GCS compatibility
|
|
// holds trivially since they use no stack.
|
|
|
|
.aeabi_subsection aeabi_feature_and_bits, optional, ULEB128
|
|
.aeabi_attribute Tag_Feature_BTI, 1
|
|
.aeabi_attribute Tag_Feature_PAC, 1
|
|
.aeabi_attribute Tag_Feature_GCS, 1
|
|
|
|
.section .note.GNU-stack,"",@progbits
|