// SIMD operations on potentially-indeterminate Node16::index[16] bytes. // Written in assembly because loading and operating on indeterminate values // is undefined behavior in C++ ([basic.indet]) but well-defined in assembly. // The caller is responsible for masking the returned bitfield to // [0, numChildren) before using it. // // Unlike x86-64 (which has pmovmskb), AArch64 has no single instruction that // produces a 1-bit-per-byte mask. Each function therefore returns a 64-bit // "nibble mask" in x0: nibble i (bits [4i, 4i+4)) is 0xf iff the condition // holds at index i. Bit (4i + 3) is the high bit of byte i's result. Callers // locate a set lane with countr_zero(bitfield) / 4 and mask the valid lanes // with (uint64_t(1) << (numChildren * 4)) - 1. // // AArch64 AAPCS: // x0 = const uint8_t *idx (16 bytes, may contain indeterminate data) // w1 = uint8_t key (find_eq_16, find_ge_16) // w1 = uint8_t begin (mask_in_range_16) // w2 = uint8_t end (mask_in_range_16) // // These functions are only ever called directly (never indirectly), so they // do not need BTI landing pads; the object is still marked BTI/PAC/GCS-aware // below so a -z force-bti link keeps BTI enabled for the whole binary. .text // uint64_t find_eq_16(const uint8_t idx[16], uint8_t key) // nibble i = 0xf iff idx[i] == key .globl find_eq_16 .type find_eq_16, %function find_eq_16: dup v1.16b, w1 // broadcast key ldr q0, [x0] // load 16 bytes (may be indeterminate) cmeq v0.16b, v0.16b, v1.16b // 0xff for each match shrn v0.8b, v0.8h, 4 // pack 16 byte-flags into 8 nibble-pairs umov x0, v0.d[0] ret .size find_eq_16, .-find_eq_16 // uint64_t find_ge_16(const uint8_t idx[16], uint8_t child) // nibble i = 0xf iff idx[i] >= child (unsigned) // cmhs gives unsigned ">=" (higher-or-same): Vd = Vn >= Vm. .globl find_ge_16 .type find_ge_16, %function find_ge_16: dup v1.16b, w1 // broadcast child ldr q0, [x0] // load 16 bytes cmhs v0.16b, v0.16b, v1.16b // 0xff where idx[i] >= child (unsigned) shrn v0.8b, v0.8h, 4 umov x0, v0.d[0] ret .size find_ge_16, .-find_ge_16 // uint64_t mask_in_range_16(const uint8_t idx[16], uint8_t begin, uint8_t end) // nibble i = 0xf iff begin <= idx[i] < end (unsigned, wrapping arithmetic) // Logic: (idx[i] - begin) < (end - begin), valid when end - begin < 256. // cmhi gives unsigned ">" (higher): Vd = Vn > Vm. We want // (end - begin) > (idx - begin), so Vn = (end - begin). .globl mask_in_range_16 .type mask_in_range_16, %function mask_in_range_16: dup v1.16b, w1 // broadcast begin dup v2.16b, w2 // broadcast end ldr q0, [x0] // load 16 bytes sub v0.16b, v0.16b, v1.16b // idx - begin (wrapping) sub v2.16b, v2.16b, v1.16b // end - begin (range size) cmhi v0.16b, v2.16b, v0.16b // 0xff where (end-begin) > (idx-begin) shrn v0.8b, v0.8h, 4 umov x0, v0.d[0] ret .size mask_in_range_16, .-mask_in_range_16 // Declare AArch64 branch-protection compatibility, matching what the // compiler emits for -mbranch-protection=standard (BTI + PAC + GCS). This // keeps the object indistinguishable from C/C++ translation units for // linkers enforcing BTI (-z force-bti). The functions above are only ever // called directly, so they need no BTI landing pads; PAC/GCS compatibility // holds trivially since they use no stack. .aeabi_subsection aeabi_feature_and_bits, optional, ULEB128 .aeabi_attribute Tag_Feature_BTI, 1 .aeabi_attribute Tag_Feature_PAC, 1 .aeabi_attribute Tag_Feature_GCS, 1 .section .note.GNU-stack,"",@progbits