7#include <native/attributes.h>
8#include <native/targets.h>
24export namespace native {
26struct simd_traits<jam::ptr<T>> {
using storage_type = std::uint32_t; };
29namespace jam::detail {
32template<
class T,
class Raw,
class Self>
33struct alignas(Raw) ptr_storage {
34 using value_type = ptr<T>;
35 static constexpr std::size_t lanes = Raw::lanes;
37 std::array<value_type, lanes> values_{};
39 [[nodiscard]] native_inline
constexpr value_type
const & operator[](std::size_t i)
const noexcept {
return values_[i]; }
40 [[nodiscard]] native_inline
constexpr value_type & operator[](std::size_t i)
noexcept {
return values_[i]; }
42 constexpr ptr_storage() noexcept = default;
43 native_inline constexpr ptr_storage(ptr_storage const & other) noexcept {
assign(other); }
44 native_inline
constexpr ptr_storage & operator=(ptr_storage
const & other)
noexcept {
assign(other);
return *
this; }
45 native_inline
constexpr void unsafe_assign(ptr_storage
const & other)
noexcept {
46 for (std::size_t i = 0; i != lanes; ++i) values_[i].
unsafe_assign(other.values_[i]);
48 native_inline
constexpr void remember() noexcept {
49 if !
consteval {
if (
auto * h = heap::current()) h->remember(std::span<value_type>{values_}); }
51 native_inline
constexpr void assign(ptr_storage
const & other)
noexcept {
unsafe_assign(other); remember(); }
53 native_inline
constexpr ptr_storage(value_type
const & value)
noexcept {
54 for (
auto & lane : values_) lane.unsafe_assign(value);
58 native_inline
constexpr explicit ptr_storage(std::array<value_type, lanes>
const & values)
noexcept
60 for (std::size_t i = 0; i != lanes; ++i) values_[i].
unsafe_assign(values[i]);
64 native_inline
constexpr explicit ptr_storage(Raw
const & raw)
noexcept {
65 std::array<std::uint32_t, lanes> offsets;
66 std::memcpy(offsets.data(), &raw,
sizeof(offsets));
67 for (std::size_t i = 0; i != lanes; ++i) values_[i].
unsafe_assign(offsets[i]);
71 template<std::
size_t Alignment = 1>
72 [[nodiscard]]
static native_inline
constexpr Self load_memory(value_type
const * p)
noexcept {
73 std::array<value_type, lanes> values;
74 for (std::size_t i = 0; i != lanes; ++i) values[i].
unsafe_assign(p[i]);
78 template<std::
size_t Alignment = 1>
79 native_inline
constexpr void store_memory(value_type * p)
const noexcept {
80 for (std::size_t i = 0; i != lanes; ++i) p[i].
unsafe_assign(values_[i]);
81 if !
consteval {
if (
auto * h = heap::current()) h->remember(std::span<value_type>{p, lanes}); }
84 [[nodiscard]]
static native_inline
constexpr Self load(value_type
const * p)
noexcept {
return load_memory(p); }
86 native_inline
constexpr void store(value_type * p)
const noexcept { store_memory(p); }
89#if defined(__x86_64__) || defined(_M_X64)
90#define JAM_PTR_TARGETS avx512, avx2, scalar
92#define JAM_PTR_TARGETS neon, scalar
95#define JAM_PTR_SIMD(name, ISA, ...) \
96export namespace native { \
101template<class T, class Raw, class Self> \
102 requires (native::target<Raw::architecture, __VA_ARGS__> == native::target<ISA, __VA_ARGS__>) \
103struct simd_customization<jam::ptr<T>, Raw, Self> : jam::detail::ptr_storage<T, Raw, Self> { \
104 using base = jam::detail::ptr_storage<T, Raw, Self>; \
107 using raw_type = Raw; \
108 using mask_type = typename Raw::mask_type; \
110 using base::values_; \
113 [[nodiscard]] native_inline constexpr Raw to_native() const noexcept { \
114 std::array<std::uint32_t, lanes> offsets; \
115 for (std::size_t i = 0; i != lanes; ++i) offsets[i] = values_[i].get(); \
116 return Raw::load(offsets.data()); \
119 [[nodiscard]] friend native_inline constexpr mask_type operator==(Self const & a, Self const & b) noexcept { \
120 if constexpr (requires(Raw x) { x == x; }) return a.to_native() == b.to_native(); \
122 std::uint64_t bits = 0; \
123 for (std::size_t i = 0; i != lanes; ++i) bits |= std::uint64_t(a[i] == b[i]) << i; \
124 return mask_type::from_bitset(bits); \
128 [[nodiscard]] friend native_inline constexpr mask_type operator!=(Self const & a, Self const & b) noexcept { \
129 if constexpr (requires(Raw x) { x != x; }) return a.to_native() != b.to_native(); \
131 std::uint64_t bits = 0; \
132 for (std::size_t i = 0; i != lanes; ++i) bits |= std::uint64_t(a[i] != b[i]) << i; \
133 return mask_type::from_bitset(bits); \
137 [[nodiscard]] friend native_inline constexpr mask_type operator!=(Self const & a, std::nullptr_t) noexcept { \
138 if constexpr (requires(Raw x) { x != x; }) return a.to_native() != Raw{std::uint32_t{0}}; \
140 std::uint64_t bits = 0; \
141 for (std::size_t i = 0; i != lanes; ++i) bits |= std::uint64_t(bool(a[i])) << i; \
142 return mask_type::from_bitset(bits); \
146 [[nodiscard]] friend native_inline constexpr mask_type operator==(Self const & a, std::nullptr_t) noexcept { \
147 if constexpr (requires(Raw x) { x == x; }) return a.to_native() == Raw{std::uint32_t{0}}; \
149 std::uint64_t bits = 0; \
150 for (std::size_t i = 0; i != lanes; ++i) bits |= std::uint64_t(!a[i]) << i; \
151 return mask_type::from_bitset(bits); \
156NATIVE_TARGET_VARIANTS(pointer, JAM_PTR_SIMD, JAM_PTR_TARGETS)
159export namespace jam {
161template<
class T, std::
size_t N, native::isa<> A>
163 native::simd<
ptr<T>, N, A>
const & source)
noexcept { target.unsafe_assign(source); }
165template<
class T, std::
size_t N, native::isa<> A>
166native_inline
void assign(native::simd<ptr<T>, N, A> & target,
167 native::simd<
ptr<T>, N, A>
const & source)
noexcept { target.assign(source); }
169template<
class T, std::
size_t N, native::isa<> A, std::
size_t K>
170native_inline
void unsafe_assign(native::wide<native::simd<ptr<T>, N, A>, K> & target,
171 native::wide<native::simd<ptr<T>, N, A>, K>
const & source)
noexcept {
172 for (std::size_t i = 0; i != K; ++i) target.registers[i].unsafe_assign(source.registers[i]);
175template<
class T, std::
size_t N, native::isa<> A, std::
size_t K>
176native_inline
void assign(native::wide<native::simd<ptr<T>, N, A>, K> & target,
177 native::wide<native::simd<ptr<T>, N, A>, K>
const & source)
noexcept {
178 for (std::size_t i = 0; i != K; ++i) target.registers[i].assign(source.registers[i]);
182template<
class T, std::
size_t N, native::isa<> A>
183struct tracer<native::simd<ptr<T>, N, A>> : detail::structural_tracer {
184 template<
class Visitor>
185 static native_inline
constexpr void trace(Visitor & visit, native::simd<
ptr<T>, N, A>
const & value)
186 noexcept requires detail::visits<ptr<T>, Visitor> {
187 for (std::size_t i = 0; i != N; ++i) detail::trace_parts(visit, value[i]);
191template<
class V, std::
size_t K>
192struct tracer<native::wide<V, K>> : detail::structural_tracer {
193 template<
class Visitor>
194 static native_inline
constexpr void trace(Visitor & visit, native::wide<V, K>
const & value)
195 noexcept requires detail::visits<V, Visitor> {
196 for (
auto const & part : value.registers) detail::trace_parts(visit, part);
201namespace jam::detail {
203using gather_bits = std::conditional_t<
sizeof(U) == 4, std::uint32_t, std::uint64_t>;
204template<
class T,
class U, std::
size_t N, native::isa<> A>
205concept gather_shape = !std::is_const_v<U> && !std::is_volatile_v<U> &&
206 (std::same_as<U, std::uint32_t> || std::same_as<U, std::int32_t> ||
207 std::same_as<U, std::uint64_t> || std::same_as<U, std::int64_t> ||
208 std::same_as<U, float> || std::same_as<U, double> || is_ptr<U>) &&
209 requires {
sizeof(native::simd<U, N, A>);
sizeof(native::simd<gather_bits<U>, N, A>); };
216template<
class T,
class U>
217native_inline native_const std::ptrdiff_t member_displacement(U T::* member)
noexcept {
218 assert(member !=
nullptr);
220 static_assert(
sizeof(member) ==
sizeof(std::int32_t),
"gather requires the flat MS data-member-pointer ABI");
221 return __builtin_bit_cast(std::int32_t, member);
223 static_assert(
sizeof(member) ==
sizeof(std::ptrdiff_t),
"gather requires the Itanium data-member-pointer ABI");
224 return __builtin_bit_cast(std::ptrdiff_t, member);
229#if defined(__x86_64__) || defined(_M_X64)
232#define JAM_GATHER_X86 \
233 if constexpr (A.has(native::x86_feature::avx2) && (N == 4 || N == 8 || N == 16)) { \
234 using V = native::simd<std::uint32_t, N, A>; \
235 using I = native::simd<std::int32_t, N, A>; \
236 auto const displacement = detail::member_displacement(member); \
237 auto const encoded = nodes.to_native(); \
238 auto const indices = __builtin_bit_cast(I, encoded & V{std::uint32_t{0x7fffffffu}}); \
239 for (unsigned generation = 0; generation != 2; ++generation) { \
240 auto const selected = active & ((encoded >> native::imm<31>) == V{generation}); \
241 auto const & g = generation ? arena->young() : arena->old(); \
242 auto const address = reinterpret_cast<std::uintptr_t>(g.data()) + displacement; \
243 auto const * base = reinterpret_cast<B const *>(address); \
244 if constexpr (sizeof(B) == 4) { \
245 if constexpr (A.has(native::x86_feature::avx512f)) \
246 result = native::mask_vpgatherdd<8>(result, selected, base, indices); \
247 else result = native::mask_vpgatherdd<8>(result, \
248 I::from_native(selected.to_native()), base, indices); \
250 if constexpr (A.has(native::x86_feature::avx512f)) \
251 result = native::mask_vpgatherdq<8>(result, selected, base, indices); \
253 using M64 = native::simd<std::int64_t, N, A>; \
254 using D = std::int32_t __attribute__((ext_vector_type(N))); \
255 using Q = std::int64_t __attribute__((ext_vector_type(N))); \
256 auto const mask = __builtin_bit_cast(M64, __builtin_convertvector(__builtin_bit_cast(D, selected), Q)); \
257 result = native::mask_vpgatherdq<8>(result, mask, base, indices); \
263#define JAM_GATHER_X86
266#define JAM_GATHERS(name, ISA, ...) \
267export namespace jam { \
268template<class T, class U, class Owner, std::size_t N, native::isa<> A> \
269 requires detail::gather_shape<T, U, N, A> && std::convertible_to<U Owner::*, U T::*> && \
270 (native::target<A, __VA_ARGS__> == native::target<ISA, __VA_ARGS__>) \
274[[nodiscard]] native_inline auto gather(native::simd<ptr<T>, N, A> const & nodes, \
275 U Owner::* selected, typename native::simd<ptr<T>, N, A>::mask_type active) noexcept { \
276 U T::* member = selected; \
277 using B = detail::gather_bits<U>; \
278 using R = native::simd<B, N, A>; \
279 using Out = native::simd<U, N, A>; \
280 auto * arena = heap::current(); \
281 assert(member != nullptr); \
285 auto const bits = active.to_bitset(); \
286 std::array<B, N> values{}; \
287 for (std::size_t i = 0; i != N; ++i) if ((bits >> i) & 1) { \
288 auto const * p = arena->address(nodes[i]); \
289 std::memcpy(&values[i], &(p->*member), sizeof(B)); \
291 result = R::load(values.data()); \
293 if constexpr (detail::is_ptr<U>) return Out{result}; \
294 else return __builtin_bit_cast(Out, result); \
297template<class T, class U, class Owner, std::size_t N, native::isa<> A> \
298 requires detail::gather_shape<T, U, N, A> && std::convertible_to<U Owner::*, U T::*> && \
299 (native::target<A, __VA_ARGS__> == native::target<ISA, __VA_ARGS__>) \
301[[nodiscard]] native_inline auto gather(native::simd<ptr<T>, N, A> const & nodes, \
302 U Owner::* member) noexcept { \
303 return gather(nodes, member, nodes != nullptr); \
306template<class T, class U, class Owner, std::size_t N, native::isa<> A, std::size_t K> \
307 requires detail::gather_shape<T, U, N, A> && std::convertible_to<U Owner::*, U T::*> && \
308 (native::target<A, __VA_ARGS__> == native::target<ISA, __VA_ARGS__>) \
310[[nodiscard]] native_inline auto gather(native::wide<native::simd<ptr<T>, N, A>, K> const & nodes, \
311 U Owner::* member, native::wide<typename native::simd<ptr<T>, N, A>::mask_type, K> const & active) noexcept { \
312 auto const & [...p] = nodes.registers; \
313 auto const & [...m] = active.registers; \
314 return native::wide<native::simd<U, N, A>, K>{std::array<native::simd<U, N, A>, K>{gather(p, member, m)...}}; \
317template<class T, class U, class Owner, std::size_t N, native::isa<> A, std::size_t K> \
318 requires detail::gather_shape<T, U, N, A> && std::convertible_to<U Owner::*, U T::*> && \
319 (native::target<A, __VA_ARGS__> == native::target<ISA, __VA_ARGS__>) \
321[[nodiscard]] native_inline auto gather(native::wide<native::simd<ptr<T>, N, A>, K> const & nodes, \
322 U Owner::* member) noexcept { \
323 using M = typename native::simd<ptr<T>, N, A>::mask_type; \
324 auto const & [...p] = nodes.registers; \
325 native::wide<M, K> active{std::array<M, K>{(p != nullptr)...}}; \
326 return gather(nodes, member, active); \
331export namespace jam {
339template<
class T,
class U,
class Owner, std::
size_t N, native::isa<> A>
340native::simd<U, N, A>
gather(native::simd<
ptr<T>, N, A>
const & nodes,
341 U Owner::* member,
typename native::simd<
ptr<T>, N, A>::mask_type active)
noexcept;
344template<
class T,
class U,
class Owner, std::
size_t N, native::isa<> A>
345native::simd<U, N, A>
gather(native::simd<
ptr<T>, N, A>
const & nodes, U Owner::* member)
noexcept;
348template<
class T,
class U,
class Owner, std::
size_t N, native::isa<> A, std::
size_t K>
349native::wide<native::simd<U, N, A>, K>
gather(native::wide<native::simd<
ptr<T>, N, A>, K>
const & nodes,
350 U Owner::* member, native::wide<
typename native::simd<
ptr<T>, N, A>::mask_type, K>
const & active)
noexcept;
353template<
class T,
class U,
class Owner, std::
size_t N, native::isa<> A, std::
size_t K>
354native::wide<native::simd<U, N, A>, K>
gather(native::wide<native::simd<
ptr<T>, N, A>, K>
const & nodes,
355 U Owner::* member)
noexcept;
358NATIVE_TARGET_VARIANTS(pointer_gather, JAM_GATHERS, JAM_PTR_TARGETS)
362#undef JAM_PTR_TARGETS
A typed heap offset, with no root registration or ownership.
constexpr void unsafe_assign(std::array< ptr< T >, N > &target, std::array< ptr< T >, N > const &source) noexcept
Bulk-copy pointer bits without registration; the caller owns the barrier.
constexpr void assign(std::array< ptr< T >, N > &target, std::array< ptr< T >, N > const &source) noexcept
Copy an array and register young lanes with one old-range check.
native::simd< U, N, A > gather(native::simd< ptr< T >, N, A > const &nodes, U Owner::*member, typename native::simd< ptr< T >, N, A >::mask_type active) noexcept
Gather a member from selected lanes; inactive lanes return zero or null.
Per-value tracing customization; values without a hook or manifest are leaves.An allocation-only clai...