jam 0.0.1
A compacting generational garbage collector for C++26
Loading...
Searching...
No Matches
simd.ccm
Go to the documentation of this file.
1// SPDX-FileCopyrightText: 2026 Edward Kmett
2// SPDX-License-Identifier: BSD-2-Clause OR Apache-2.0
3
6module;
7#include <native/attributes.h>
8#include <native/targets.h>
9#include <array>
10#include <cassert>
11#include <bit>
12#include <concepts>
13#include <cstddef>
14#include <cstdint>
15#include <cstring>
16#include <span>
17#include <type_traits>
18#include <utility>
19
20export module jam:simd;
21export import :heap;
22export import native;
23
24export namespace native {
25template<class T>
26struct simd_traits<jam::ptr<T>> { using storage_type = std::uint32_t; };
27}
28
29namespace jam::detail {
30// Typed storage and lane access stay available to baseline GC workers. Inherited
31// constructor bridges must not require the target ISA of the vector operations.
32template<class T, class Raw, class Self>
33struct alignas(Raw) ptr_storage {
34 using value_type = ptr<T>;
35 static constexpr std::size_t lanes = Raw::lanes;
36protected:
37 std::array<value_type, lanes> values_{};
38public:
39 [[nodiscard]] native_inline constexpr value_type const & operator[](std::size_t i) const noexcept { return values_[i]; }
40 [[nodiscard]] native_inline constexpr value_type & operator[](std::size_t i) noexcept { return values_[i]; }
42 constexpr ptr_storage() noexcept = default;
43 native_inline constexpr ptr_storage(ptr_storage const & other) noexcept { assign(other); }
44 native_inline constexpr ptr_storage & operator=(ptr_storage const & other) noexcept { assign(other); return *this; }
45 native_inline constexpr void unsafe_assign(ptr_storage const & other) noexcept {
46 for (std::size_t i = 0; i != lanes; ++i) values_[i].unsafe_assign(other.values_[i]);
47 }
48 native_inline constexpr void remember() noexcept {
49 if !consteval { if (auto * h = heap::current()) h->remember(std::span<value_type>{values_}); }
50 }
51 native_inline constexpr void assign(ptr_storage const & other) noexcept { unsafe_assign(other); remember(); }
53 native_inline constexpr ptr_storage(value_type const & value) noexcept {
54 for (auto & lane : values_) lane.unsafe_assign(value);
55 remember();
56 }
58 native_inline constexpr explicit ptr_storage(std::array<value_type, lanes> const & values) noexcept
59 {
60 for (std::size_t i = 0; i != lanes; ++i) values_[i].unsafe_assign(values[i]);
61 remember();
62 }
64 native_inline constexpr explicit ptr_storage(Raw const & raw) noexcept {
65 std::array<std::uint32_t, lanes> offsets;
66 std::memcpy(offsets.data(), &raw, sizeof(offsets));
67 for (std::size_t i = 0; i != lanes; ++i) values_[i].unsafe_assign(offsets[i]);
68 remember();
69 }
71 template<std::size_t Alignment = 1>
72 [[nodiscard]] static native_inline constexpr Self load_memory(value_type const * p) noexcept {
73 std::array<value_type, lanes> values;
74 for (std::size_t i = 0; i != lanes; ++i) values[i].unsafe_assign(p[i]);
75 return Self{values};
76 }
78 template<std::size_t Alignment = 1>
79 native_inline constexpr void store_memory(value_type * p) const noexcept {
80 for (std::size_t i = 0; i != lanes; ++i) p[i].unsafe_assign(values_[i]);
81 if !consteval { if (auto * h = heap::current()) h->remember(std::span<value_type>{p, lanes}); }
82 }
84 [[nodiscard]] static native_inline constexpr Self load(value_type const * p) noexcept { return load_memory(p); }
86 native_inline constexpr void store(value_type * p) const noexcept { store_memory(p); }
87};
88}
89#if defined(__x86_64__) || defined(_M_X64)
90#define JAM_PTR_TARGETS avx512, avx2, scalar
91#else
92#define JAM_PTR_TARGETS neon, scalar
93#endif
94
95#define JAM_PTR_SIMD(name, ISA, ...) \
96export namespace native { \
97 \
101template<class T, class Raw, class Self> \
102 requires (native::target<Raw::architecture, __VA_ARGS__> == native::target<ISA, __VA_ARGS__>) \
103struct simd_customization<jam::ptr<T>, Raw, Self> : jam::detail::ptr_storage<T, Raw, Self> { \
104 using base = jam::detail::ptr_storage<T, Raw, Self>; \
105 using base::base; \
106 using base::lanes; \
107 using raw_type = Raw; \
108 using mask_type = typename Raw::mask_type; \
109private: \
110 using base::values_; \
111public: \
112 \
113 [[nodiscard]] native_inline constexpr Raw to_native() const noexcept { \
114 std::array<std::uint32_t, lanes> offsets; \
115 for (std::size_t i = 0; i != lanes; ++i) offsets[i] = values_[i].get(); \
116 return Raw::load(offsets.data()); \
117 } \
118 \
119 [[nodiscard]] friend native_inline constexpr mask_type operator==(Self const & a, Self const & b) noexcept { \
120 if constexpr (requires(Raw x) { x == x; }) return a.to_native() == b.to_native(); \
121 else { \
122 std::uint64_t bits = 0; \
123 for (std::size_t i = 0; i != lanes; ++i) bits |= std::uint64_t(a[i] == b[i]) << i; \
124 return mask_type::from_bitset(bits); \
125 } \
126 } \
127 \
128 [[nodiscard]] friend native_inline constexpr mask_type operator!=(Self const & a, Self const & b) noexcept { \
129 if constexpr (requires(Raw x) { x != x; }) return a.to_native() != b.to_native(); \
130 else { \
131 std::uint64_t bits = 0; \
132 for (std::size_t i = 0; i != lanes; ++i) bits |= std::uint64_t(a[i] != b[i]) << i; \
133 return mask_type::from_bitset(bits); \
134 } \
135 } \
136 \
137 [[nodiscard]] friend native_inline constexpr mask_type operator!=(Self const & a, std::nullptr_t) noexcept { \
138 if constexpr (requires(Raw x) { x != x; }) return a.to_native() != Raw{std::uint32_t{0}}; \
139 else { \
140 std::uint64_t bits = 0; \
141 for (std::size_t i = 0; i != lanes; ++i) bits |= std::uint64_t(bool(a[i])) << i; \
142 return mask_type::from_bitset(bits); \
143 } \
144 } \
145 \
146 [[nodiscard]] friend native_inline constexpr mask_type operator==(Self const & a, std::nullptr_t) noexcept { \
147 if constexpr (requires(Raw x) { x == x; }) return a.to_native() == Raw{std::uint32_t{0}}; \
148 else { \
149 std::uint64_t bits = 0; \
150 for (std::size_t i = 0; i != lanes; ++i) bits |= std::uint64_t(!a[i]) << i; \
151 return mask_type::from_bitset(bits); \
152 } \
153 } \
154}; \
155}
156NATIVE_TARGET_VARIANTS(pointer, JAM_PTR_SIMD, JAM_PTR_TARGETS)
157#undef JAM_PTR_SIMD
158
159export namespace jam {
161template<class T, std::size_t N, native::isa<> A>
162native_inline void unsafe_assign(native::simd<ptr<T>, N, A> & target,
163 native::simd<ptr<T>, N, A> const & source) noexcept { target.unsafe_assign(source); }
165template<class T, std::size_t N, native::isa<> A>
166native_inline void assign(native::simd<ptr<T>, N, A> & target,
167 native::simd<ptr<T>, N, A> const & source) noexcept { target.assign(source); }
169template<class T, std::size_t N, native::isa<> A, std::size_t K>
170native_inline void unsafe_assign(native::wide<native::simd<ptr<T>, N, A>, K> & target,
171 native::wide<native::simd<ptr<T>, N, A>, K> const & source) noexcept {
172 for (std::size_t i = 0; i != K; ++i) target.registers[i].unsafe_assign(source.registers[i]);
175template<class T, std::size_t N, native::isa<> A, std::size_t K>
176native_inline void assign(native::wide<native::simd<ptr<T>, N, A>, K> & target,
177 native::wide<native::simd<ptr<T>, N, A>, K> const & source) noexcept {
178 for (std::size_t i = 0; i != K; ++i) target.registers[i].assign(source.registers[i]);
179}
182template<class T, std::size_t N, native::isa<> A>
183struct tracer<native::simd<ptr<T>, N, A>> : detail::structural_tracer {
184 template<class Visitor>
185 static native_inline constexpr void trace(Visitor & visit, native::simd<ptr<T>, N, A> const & value)
186 noexcept requires detail::visits<ptr<T>, Visitor> {
187 for (std::size_t i = 0; i != N; ++i) detail::trace_parts(visit, value[i]);
188 }
191template<class V, std::size_t K>
192struct tracer<native::wide<V, K>> : detail::structural_tracer {
193 template<class Visitor>
194 static native_inline constexpr void trace(Visitor & visit, native::wide<V, K> const & value)
195 noexcept requires detail::visits<V, Visitor> {
196 for (auto const & part : value.registers) detail::trace_parts(visit, part);
197 }
198};
199}
200
201namespace jam::detail {
202template<class U>
203using gather_bits = std::conditional_t<sizeof(U) == 4, std::uint32_t, std::uint64_t>;
204template<class T, class U, std::size_t N, native::isa<> A>
205concept gather_shape = !std::is_const_v<U> && !std::is_volatile_v<U> &&
206 (std::same_as<U, std::uint32_t> || std::same_as<U, std::int32_t> ||
207 std::same_as<U, std::uint64_t> || std::same_as<U, std::int64_t> ||
208 std::same_as<U, float> || std::same_as<U, double> || is_ptr<U>) &&
209 requires { sizeof(native::simd<U, N, A>); sizeof(native::simd<gather_bits<U>, N, A>); };
210
211// For supported nonvirtual inheritance, these ABIs encode data-member pointers
212// as flat byte displacements. Decode that documented
213// ABI representation rather than applying a member pointer to a fabricated object
214// or extracting a vector mask just to find a sample object. Extended MS member
215// pointer representations are deliberately rejected instead of guessed.
216template<class T, class U>
217native_inline native_const std::ptrdiff_t member_displacement(U T::* member) noexcept {
218 assert(member != nullptr);
219#if defined(_WIN32)
220 static_assert(sizeof(member) == sizeof(std::int32_t), "gather requires the flat MS data-member-pointer ABI");
221 return __builtin_bit_cast(std::int32_t, member);
222#else
223 static_assert(sizeof(member) == sizeof(std::ptrdiff_t), "gather requires the Itanium data-member-pointer ABI");
224 return __builtin_bit_cast(std::ptrdiff_t, member);
225#endif
226}
227}
228
229#if defined(__x86_64__) || defined(_M_X64)
230// Each generation has 31-bit nonnegative cell indices. Gather from its own
231// base under a disjoint mask, preserving the other generation's result lanes.
232#define JAM_GATHER_X86 \
233 if constexpr (A.has(native::x86_feature::avx2) && (N == 4 || N == 8 || N == 16)) { \
234 using V = native::simd<std::uint32_t, N, A>; \
235 using I = native::simd<std::int32_t, N, A>; \
236 auto const displacement = detail::member_displacement(member); \
237 auto const encoded = nodes.to_native(); \
238 auto const indices = __builtin_bit_cast(I, encoded & V{std::uint32_t{0x7fffffffu}}); \
239 for (unsigned generation = 0; generation != 2; ++generation) { \
240 auto const selected = active & ((encoded >> native::imm<31>) == V{generation}); \
241 auto const & g = generation ? arena->young() : arena->old(); \
242 auto const address = reinterpret_cast<std::uintptr_t>(g.data()) + displacement; \
243 auto const * base = reinterpret_cast<B const *>(address); \
244 if constexpr (sizeof(B) == 4) { \
245 if constexpr (A.has(native::x86_feature::avx512f)) \
246 result = native::mask_vpgatherdd<8>(result, selected, base, indices); \
247 else result = native::mask_vpgatherdd<8>(result, \
248 I::from_native(selected.to_native()), base, indices); \
249 } else { \
250 if constexpr (A.has(native::x86_feature::avx512f)) \
251 result = native::mask_vpgatherdq<8>(result, selected, base, indices); \
252 else { \
253 using M64 = native::simd<std::int64_t, N, A>; \
254 using D = std::int32_t __attribute__((ext_vector_type(N))); \
255 using Q = std::int64_t __attribute__((ext_vector_type(N))); \
256 auto const mask = __builtin_bit_cast(M64, __builtin_convertvector(__builtin_bit_cast(D, selected), Q)); \
257 result = native::mask_vpgatherdq<8>(result, mask, base, indices); \
258 } \
259 } \
260 } \
261 } else
262#else
263#define JAM_GATHER_X86
264#endif
265
266#define JAM_GATHERS(name, ISA, ...) \
267export namespace jam { \
268template<class T, class U, class Owner, std::size_t N, native::isa<> A> \
269 requires detail::gather_shape<T, U, N, A> && std::convertible_to<U Owner::*, U T::*> && \
270 (native::target<A, __VA_ARGS__> == native::target<ISA, __VA_ARGS__>) \
271 \
272 \
273 \
274[[nodiscard]] native_inline auto gather(native::simd<ptr<T>, N, A> const & nodes, \
275 U Owner::* selected, typename native::simd<ptr<T>, N, A>::mask_type active) noexcept { \
276 U T::* member = selected; \
277 using B = detail::gather_bits<U>; \
278 using R = native::simd<B, N, A>; \
279 using Out = native::simd<U, N, A>; \
280 auto * arena = heap::current(); \
281 assert(member != nullptr); \
282 R result{}; \
283 JAM_GATHER_X86 \
284 { \
285 auto const bits = active.to_bitset(); \
286 std::array<B, N> values{}; \
287 for (std::size_t i = 0; i != N; ++i) if ((bits >> i) & 1) { \
288 auto const * p = arena->address(nodes[i]); \
289 std::memcpy(&values[i], &(p->*member), sizeof(B)); \
290 } \
291 result = R::load(values.data()); \
292 } \
293 if constexpr (detail::is_ptr<U>) return Out{result}; \
294 else return __builtin_bit_cast(Out, result); \
295} \
296 \
297template<class T, class U, class Owner, std::size_t N, native::isa<> A> \
298 requires detail::gather_shape<T, U, N, A> && std::convertible_to<U Owner::*, U T::*> && \
299 (native::target<A, __VA_ARGS__> == native::target<ISA, __VA_ARGS__>) \
300 \
301[[nodiscard]] native_inline auto gather(native::simd<ptr<T>, N, A> const & nodes, \
302 U Owner::* member) noexcept { \
303 return gather(nodes, member, nodes != nullptr); \
304} \
305 \
306template<class T, class U, class Owner, std::size_t N, native::isa<> A, std::size_t K> \
307 requires detail::gather_shape<T, U, N, A> && std::convertible_to<U Owner::*, U T::*> && \
308 (native::target<A, __VA_ARGS__> == native::target<ISA, __VA_ARGS__>) \
309 \
310[[nodiscard]] native_inline auto gather(native::wide<native::simd<ptr<T>, N, A>, K> const & nodes, \
311 U Owner::* member, native::wide<typename native::simd<ptr<T>, N, A>::mask_type, K> const & active) noexcept { \
312 auto const & [...p] = nodes.registers; \
313 auto const & [...m] = active.registers; \
314 return native::wide<native::simd<U, N, A>, K>{std::array<native::simd<U, N, A>, K>{gather(p, member, m)...}}; \
315} \
316 \
317template<class T, class U, class Owner, std::size_t N, native::isa<> A, std::size_t K> \
318 requires detail::gather_shape<T, U, N, A> && std::convertible_to<U Owner::*, U T::*> && \
319 (native::target<A, __VA_ARGS__> == native::target<ISA, __VA_ARGS__>) \
320 \
321[[nodiscard]] native_inline auto gather(native::wide<native::simd<ptr<T>, N, A>, K> const & nodes, \
322 U Owner::* member) noexcept { \
323 using M = typename native::simd<ptr<T>, N, A>::mask_type; \
324 auto const & [...p] = nodes.registers; \
325 native::wide<M, K> active{std::array<M, K>{(p != nullptr)...}}; \
326 return gather(nodes, member, active); \
327} \
328}
329
330#ifdef JAM_DOXYGEN
331export namespace jam {
339template<class T, class U, class Owner, std::size_t N, native::isa<> A>
340native::simd<U, N, A> gather(native::simd<ptr<T>, N, A> const & nodes,
341 U Owner::* member, typename native::simd<ptr<T>, N, A>::mask_type active) noexcept;
344template<class T, class U, class Owner, std::size_t N, native::isa<> A>
345native::simd<U, N, A> gather(native::simd<ptr<T>, N, A> const & nodes, U Owner::* member) noexcept;
348template<class T, class U, class Owner, std::size_t N, native::isa<> A, std::size_t K>
349native::wide<native::simd<U, N, A>, K> gather(native::wide<native::simd<ptr<T>, N, A>, K> const & nodes,
350 U Owner::* member, native::wide<typename native::simd<ptr<T>, N, A>::mask_type, K> const & active) noexcept;
353template<class T, class U, class Owner, std::size_t N, native::isa<> A, std::size_t K>
354native::wide<native::simd<U, N, A>, K> gather(native::wide<native::simd<ptr<T>, N, A>, K> const & nodes,
355 U Owner::* member) noexcept;
356}
357#else
358NATIVE_TARGET_VARIANTS(pointer_gather, JAM_GATHERS, JAM_PTR_TARGETS)
359#endif
360#undef JAM_GATHERS
361#undef JAM_GATHER_X86
362#undef JAM_PTR_TARGETS
A typed heap offset, with no root registration or ownership.
Definition heap.ccm:570
constexpr void unsafe_assign(std::array< ptr< T >, N > &target, std::array< ptr< T >, N > const &source) noexcept
Bulk-copy pointer bits without registration; the caller owns the barrier.
Definition heap.ccm:2456
constexpr void assign(std::array< ptr< T >, N > &target, std::array< ptr< T >, N > const &source) noexcept
Copy an array and register young lanes with one old-range check.
Definition heap.ccm:2461
native::simd< U, N, A > gather(native::simd< ptr< T >, N, A > const &nodes, U Owner::*member, typename native::simd< ptr< T >, N, A >::mask_type active) noexcept
Gather a member from selected lanes; inactive lanes return zero or null.
Per-value tracing customization; values without a hook or manifest are leaves.An allocation-only clai...
Definition heap.ccm:515