Line data Source code
1 : #ifndef HEADER_fd_src_util_scratch_fd_scratch_h
2 : #define HEADER_fd_src_util_scratch_fd_scratch_h
3 :
4 : /* APIs for high performance scratch pad memory allocation. There
5 : are two allocators provided. One is fd_alloca, which is an alignment
6 : aware equivalent of alloca. It is meant for use anywhere alloca
7 : would normally be used. This is only available if the built target
8 : has the FD_HAS_ALLOCA capability. The second as fd_scratch_alloc.
9 : It is meant for use in situations that have very complex and large
10 : temporary memory usage. */
11 :
12 : #include "../tile/fd_tile.h"
13 :
14 : /* FD_SCRATCH_ALLOC_ALIGN_DEFAULT is the default alignment to use for
15 : allocations.
16 :
17 : Default should be at least 16 for consistent cross platform behavior
18 : that is language conformant across a wide range of targets (i.e. the
19 : largest primitive type across all possible build ... practically
20 : sizeof(int128)). This also naturally covers SSE natural alignment on
21 : x86. 8 could be used if features like int128 and so forth and still
22 : be linguistically conformant (sizeof(ulong) here is the limit).
23 : Likewise, 32, 64, 128 could be used to guarantee all allocations will
24 : have natural AVX/AVX2, natural AVX-512 / cache-line,
25 : adjacent-cache-line-prefetch false sharing avoidance / natural GPU
26 : alignment properties.
27 :
28 : 128 for default was picked as double x86 cache line for ACLPF false
29 : sharing avoidance and for consistency with GPU warp sizes ... i.e.
30 : the default allocation behaviors are naturally interthread
31 : communication false sharing resistant and GPU friendly. This also
32 : naturally covers cases like SSE, AVX, AVX2 and AVX-512. */
33 :
34 4298435 : #define FD_SCRATCH_ALIGN_DEFAULT (128UL) /* integer power-of-2 >=16 */
35 :
36 : /* FD_SCRATCH_{SMEM,FMEM}_ALIGN give the alignment requirements for
37 : the memory regions used to a scratch pad memory. There are not many
38 : restrictions on the SMEM alignment practically other than it be a
39 : reasonable integer power of two. 128 was picked to harmonize with
40 : FD_SCRATCH_ALIGN_DEFAULT (which does have more technical motivations
41 : behind its choice) but this is not strictly required.
42 : FD_SCRATCH_FMEM_ALIGN is required to be sizeof(ulong). */
43 :
44 98313 : #define FD_SCRATCH_SMEM_ALIGN (128UL) /* integer power-of-2, harmonized with ALIGN_DEFAULT */
45 : #define FD_SCRATCH_FMEM_ALIGN (8UL) /* ==sizeof(ulong) but avoids bugs with some compilers */
46 :
47 : FD_PROTOTYPES_BEGIN
48 :
49 : /* Private APIs *******************************************************/
50 :
51 : #if FD_DCHECK_STYLE>0
52 : extern FD_TL int fd_scratch_in_prepare;
53 : #endif
54 :
55 : extern FD_TL ulong fd_scratch_private_start;
56 : extern FD_TL ulong fd_scratch_private_free;
57 : extern FD_TL ulong fd_scratch_private_stop;
58 :
59 : extern FD_TL ulong * fd_scratch_private_frame;
60 : extern FD_TL ulong fd_scratch_private_frame_cnt;
61 : extern FD_TL ulong fd_scratch_private_frame_max;
62 :
63 : FD_FN_CONST static inline int
64 3544594 : fd_scratch_private_align_is_valid( ulong align ) {
65 3544594 : return !(align & (align-1UL)); /* returns true if power or 2 or zero, compile time typically */
66 3544594 : }
67 :
68 : FD_FN_CONST static inline ulong
69 3544594 : fd_scratch_private_true_align( ulong align ) {
70 3544594 : return fd_ulong_if( !align, FD_SCRATCH_ALIGN_DEFAULT, align ); /* compile time typically */
71 3544594 : }
72 :
73 : /* Public APIs ********************************************************/
74 :
75 : /* Constructor APIs */
76 :
77 : /* fd_scratch_smem_{align,footprint} return the alignment and footprint
78 : of a memory region suitable for use as a scratch pad memory that can
79 : hold up to smax bytes. There are very few restrictions on the nature
80 : of this memory. It could even be just a flat address space that is
81 : not backed by an actual physical memory as far as scratch is
82 : concerned. In typical use cases though, the scratch pad memory
83 : should point to a region of huge or gigantic page backed memory on
84 : the caller's numa node.
85 :
86 : A shared memory region for smem is fine for smem. This could be used
87 : for example to allow other threads / processes to access a scratch
88 : allocation from this thread for the lifetime of a scratch allocation.
89 :
90 : Even more generally, a shared memory region for both smem and fmem
91 : could make it is theoretically possible to have a scratch pad memory
92 : that is shared across multiple threads / processes. The API is not
93 : well designed for such though (the main reason to use fmem in shared
94 : memory would be convenience and/or adding hot swapping
95 : functionality). In the common scratch scenario, every thread would
96 : attach to their local join of the shared smem and shared fmem. But
97 : since the operations below are not designed to be thread safe, the
98 : threads would have to protect against concurrent use of push and pop
99 : (and attach would probably need to be tweaked to make it easier to
100 : attach to an already in use scratch pad).
101 :
102 : Compile time allocation is possible via the FD_SCRATCH_SMEM_ALIGN
103 : define. E.g.:
104 :
105 : uchar my_smem[ MY_SMAX ] __attribute__((aligned(FD_SCRATCH_SMEM_ALIGN)));
106 :
107 : will be valid to use as a scratch smem with space for up to MY_SMAX
108 : bytes. */
109 :
110 49158 : FD_FN_CONST static inline ulong fd_scratch_smem_align( void ) { return FD_SCRATCH_SMEM_ALIGN; }
111 :
112 : FD_FN_CONST static inline ulong
113 49155 : fd_scratch_smem_footprint( ulong smax ) {
114 49155 : return fd_ulong_align_up( smax, FD_SCRATCH_SMEM_ALIGN );
115 49155 : }
116 :
117 : /* fd_scratch_fmem_{align,footprint} return the alignment and footprint
118 : of a memory region suitable for holding the scratch pad memory
119 : metadata (typically very small). The scratch pad memory will be
120 : capable of holding up to depth scratch frames.
121 :
122 : Compile time allocation is possible via the FD_SCRATCH_FMEM_ALIGN
123 : define. E.g.
124 :
125 : ulong my_fmem[ MY_DEPTH ] __attribute((aligned(FD_SCRATCH_FMEM_ALIGN)));
126 :
127 : or, even simpler:
128 :
129 : ulong my_fmem[ MY_DEPTH ];
130 :
131 : will be valid to use as a scratch fmem with space for up to depth
132 : frames. The attribute variant is not strictly necessary, just for
133 : consistency with the smem above (where it is required). */
134 :
135 9 : FD_FN_CONST static inline ulong fd_scratch_fmem_align ( void ) { return sizeof(ulong); }
136 51 : FD_FN_CONST static inline ulong fd_scratch_fmem_footprint( ulong depth ) { return sizeof(ulong)*depth; }
137 :
138 : /* fd_scratch_attach attaches the calling thread to memory regions
139 : sufficient to hold up to smax (positive) bytes and with up to depth
140 : (positive) frames. smem/fmem should have the required alignment and
141 : footprint specified for smax/depth from the above and be non-NULL).
142 : The caller has a read/write interest in these regions while attached
143 : (and thus the local lifetime of these regions must cover the lifetime
144 : of the attachment). Only one scratch pad memory may be attached to a
145 : caller at a time. This cannot fail from the caller's point of view
146 : (if handholding is enabled, it will abort the caller with a
147 : descriptive error message if used obviously in error). */
148 :
149 : static inline void
150 : fd_scratch_attach( void * smem,
151 : void * fmem,
152 : ulong smax,
153 15 : ulong depth ) {
154 :
155 15 : FD_DCHECK_CRIT( !fd_scratch_private_frame_max, "already attached" );
156 15 : FD_DCHECK_CRIT( !!smem, "bad smem" );
157 15 : FD_DCHECK_CRIT( !!fmem, "bad fmem" );
158 15 : FD_DCHECK_CRIT( !!smax, "bad smax" );
159 15 : FD_DCHECK_CRIT( !!depth, "bad depth" );
160 : # if FD_DCHECK_STYLE>0
161 : fd_scratch_in_prepare = 0;
162 : # endif
163 :
164 15 : fd_scratch_private_start = (ulong)smem;
165 15 : fd_scratch_private_free = fd_scratch_private_start;
166 15 : fd_scratch_private_stop = fd_scratch_private_start + smax;
167 :
168 15 : fd_scratch_private_frame = (ulong *)fmem;
169 15 : fd_scratch_private_frame_cnt = 0UL;
170 15 : fd_scratch_private_frame_max = depth;
171 :
172 : # if FD_HAS_DEEPASAN
173 : /* Poison the entire smem region. Underpoison the boundaries to respect
174 : alignment requirements. */
175 : ulong aligned_start = fd_ulong_align_up( fd_scratch_private_start, FD_ASAN_ALIGN );
176 : ulong aligned_end = fd_ulong_align_dn( fd_scratch_private_stop, FD_ASAN_ALIGN );
177 : fd_asan_poison( (void*)aligned_start, aligned_end - aligned_start );
178 : # endif
179 : #if FD_HAS_MSAN
180 : /* Mark the entire smem region as uninitialized. */
181 : ulong aligned_start = fd_ulong_align_up( fd_scratch_private_start, FD_MSAN_ALIGN );
182 : ulong aligned_end = fd_ulong_align_dn( fd_scratch_private_stop, FD_MSAN_ALIGN );
183 : fd_msan_poison( (void*)aligned_start, aligned_end - aligned_start );
184 : #endif
185 15 : }
186 :
187 : /* fd_scratch_detach detaches the calling thread from its current
188 : attachment. Returns smem used on attach and, if opt_fmem is
189 : non-NULL, opt_fmem[0] will contain the fmem used on attach on return.
190 :
191 : This relinquishes the calling threads read/write interest on these
192 : memory regions. All the caller's scratch frames are popped, any
193 : prepare in progress is canceled and all the caller's scratch
194 : allocations are freed implicitly by this.
195 :
196 : This cannot fail from the caller's point of view (if handholding is
197 : enabled, it will abort the caller with a descriptive error message if
198 : used obviously in error). */
199 :
200 : static inline void *
201 15 : fd_scratch_detach( void ** _opt_fmem ) {
202 :
203 15 : FD_DCHECK_CRIT( !!fd_scratch_private_frame_max, "not attached" );
204 : # if FD_DCHECK_STYLE>0
205 : fd_scratch_in_prepare = 0;
206 : # endif
207 :
208 : # if FD_HAS_DEEPASAN
209 : /* Unpoison the entire scratch space. There should now be an underlying
210 : allocation which has not been poisoned. */
211 : ulong aligned_start = fd_ulong_align_up( fd_scratch_private_start, FD_ASAN_ALIGN );
212 : ulong aligned_stop = fd_ulong_align_dn( fd_scratch_private_stop, FD_ASAN_ALIGN );
213 : fd_asan_unpoison( (void*)aligned_start, aligned_stop - aligned_start );
214 : # endif
215 :
216 15 : void * smem = (void *)fd_scratch_private_start;
217 15 : void * fmem = (void *)fd_scratch_private_frame;
218 :
219 15 : fd_scratch_private_start = 0UL;
220 15 : fd_scratch_private_free = 0UL;
221 15 : fd_scratch_private_stop = 0UL;
222 :
223 15 : fd_scratch_private_frame = NULL;
224 15 : fd_scratch_private_frame_cnt = 0UL;
225 15 : fd_scratch_private_frame_max = 0UL;
226 :
227 15 : if( _opt_fmem ) _opt_fmem[0] = fmem;
228 15 : return smem;
229 15 : }
230 :
231 : /* User APIs */
232 :
233 : /* fd_scratch_{used,free} returns the number of bytes used/free in the
234 : caller's scratch. Returns 0 if not attached. Because of alignment
235 : overheads, an allocation is guaranteed to succeed if free>=sz+align-1
236 : where align is the actual alignment required for the allocation (e.g.
237 : align==0 -> default, align<min -> min). It is guaranteed to fail if
238 : free<sz. It might succeed or fail in between depending on the
239 : alignments of previously allocations. These are freaky fast (O(3)
240 : fast asm operations under the hood). */
241 :
242 9 : static inline ulong fd_scratch_used( void ) { return fd_scratch_private_free - fd_scratch_private_start; }
243 9 : static inline ulong fd_scratch_free( void ) { return fd_scratch_private_stop - fd_scratch_private_free; }
244 :
245 : /* fd_scratch_frame_{used,free} returns the number of scratch frames
246 : used/free in the caller's scratch. Returns 0 if not attached. push
247 : is guaranteed to succeed if free is non-zero and guaranteed to fail
248 : otherwise. pop is guaranteed to succeed if used is non-zero and
249 : guaranteed to fail otherwise. These are freaky fast (O(1-3) fast asm
250 : operations under the hood). */
251 :
252 2954058 : static inline ulong fd_scratch_frame_used( void ) { return fd_scratch_private_frame_cnt; }
253 2999381 : static inline ulong fd_scratch_frame_free( void ) { return fd_scratch_private_frame_max - fd_scratch_private_frame_cnt; }
254 :
255 : /* fd_scratch_reset frees all allocations (if any) and pops all scratch
256 : frames (if any) such that the caller's scratch will be in the same
257 : state it was immediately after attach. The caller must be attached
258 : to a scratch memory to use. This cannot fail from the caller's point
259 : of view (if handholding is enabled, it will abort the caller with a
260 : descriptive error message if used obviously in error). This is
261 : freaky fast (O(3) fast asm operations under the hood). */
262 :
263 : static inline void
264 730 : fd_scratch_reset( void ) {
265 730 : FD_DCHECK_CRIT( !!fd_scratch_private_frame_max, "not attached" );
266 : # if FD_DCHECK_STYLE>0
267 : fd_scratch_in_prepare = 0;
268 : # endif
269 730 : fd_scratch_private_free = fd_scratch_private_start;
270 730 : fd_scratch_private_frame_cnt = 0UL;
271 :
272 : /* Poison entire scratch space again. */
273 : # if FD_HAS_DEEPASAN
274 : ulong aligned_start = fd_ulong_align_up( fd_scratch_private_start, FD_ASAN_ALIGN );
275 : ulong aligned_stop = fd_ulong_align_dn( fd_scratch_private_stop, FD_ASAN_ALIGN );
276 : fd_asan_poison( (void*)aligned_start, aligned_stop - aligned_start );
277 : # endif
278 : # if FD_HAS_MSAN
279 : ulong aligned_start = fd_ulong_align_up( fd_scratch_private_start, FD_MSAN_ALIGN );
280 : ulong aligned_stop = fd_ulong_align_dn( fd_scratch_private_stop, FD_MSAN_ALIGN );
281 : fd_msan_poison( (void*)aligned_start, aligned_stop - aligned_start );
282 : # endif
283 730 : }
284 :
285 : /* fd_scratch_push creates a new scratch frame and makes it the current
286 : frame. Assumes caller is attached to a scratch with space for a new
287 : frame. This cannot fail from the caller's point of view (if
288 : handholding is enabled, it will abort the caller with a descriptive
289 : error message if used obviously in error). This is freaky fast (O(5)
290 : fast asm operations under the hood). */
291 :
292 : FD_FN_UNUSED static void /* Work around -Winline */
293 45531 : fd_scratch_push( void ) {
294 45531 : FD_DCHECK_CRIT( !!fd_scratch_private_frame_max, "not attached" );
295 45531 : FD_DCHECK_CRIT( fd_scratch_private_frame_cnt < fd_scratch_private_frame_max, "too many frames" );
296 : # if FD_DCHECK_STYLE>0
297 : fd_scratch_in_prepare = 0;
298 : # endif
299 45531 : fd_scratch_private_frame[ fd_scratch_private_frame_cnt++ ] = fd_scratch_private_free;
300 :
301 : /* Poison to end of scratch region to account for case of in-prep allocation
302 : getting implictly cancelled. */
303 : # if FD_HAS_DEEPASAN
304 : ulong aligned_start = fd_ulong_align_up( fd_scratch_private_free, FD_ASAN_ALIGN );
305 : ulong aligned_stop = fd_ulong_align_dn( fd_scratch_private_stop, FD_ASAN_ALIGN );
306 : fd_asan_poison( (void*)aligned_start, aligned_stop - aligned_start );
307 : # endif
308 : #if FD_HAS_MSAN
309 : ulong aligned_start = fd_ulong_align_up( fd_scratch_private_free, FD_MSAN_ALIGN );
310 : ulong aligned_stop = fd_ulong_align_dn( fd_scratch_private_stop, FD_MSAN_ALIGN );
311 : fd_msan_poison( (void*)aligned_start, aligned_stop - aligned_start );
312 : #endif
313 45531 : }
314 :
315 : /* fd_scratch_pop frees all allocations in the current scratch frame,
316 : destroys the current scratch frame and makes the previous frame (if
317 : there is one) the current stack frame (and leaves the caller without
318 : a current frame if there is not one). Assumes the caller is attached
319 : to a scratch memory with at least one frame in use. This cannot fail
320 : from the caller's point of view (if handholding is enabled, it will
321 : abort the caller with a descriptive error message if used obviously
322 : in error). This is freaky fast (O(5) fast asm operations under the
323 : hood). */
324 :
325 : FD_FN_UNUSED static void /* Work around -Winline */
326 40751 : fd_scratch_pop( void ) {
327 40751 : FD_DCHECK_CRIT( !!fd_scratch_private_frame_max, "not attached" );
328 40751 : FD_DCHECK_CRIT( !!fd_scratch_private_frame_cnt, "unmatched pop" );
329 : # if FD_DCHECK_STYLE>0
330 : fd_scratch_in_prepare = 0;
331 : # endif
332 40751 : fd_scratch_private_free = fd_scratch_private_frame[ --fd_scratch_private_frame_cnt ];
333 :
334 : # if FD_HAS_DEEPASAN
335 : /* On a pop() operation, the entire range from fd_scratch_private_free to the
336 : end of the scratch space can be safely poisoned. The region must be aligned
337 : to accomodate asan manual poisoning requirements. */
338 : ulong aligned_start = fd_ulong_align_up( fd_scratch_private_free, FD_ASAN_ALIGN );
339 : ulong aligned_stop = fd_ulong_align_dn( fd_scratch_private_stop, FD_ASAN_ALIGN );
340 : fd_asan_poison( (void*)aligned_start, aligned_stop - aligned_start );
341 : # endif
342 : #if FD_HAS_MSAN
343 : ulong aligned_start = fd_ulong_align_up( fd_scratch_private_free, FD_MSAN_ALIGN );
344 : ulong aligned_stop = fd_ulong_align_dn( fd_scratch_private_stop, FD_MSAN_ALIGN );
345 : fd_msan_poison( (void*)aligned_start, aligned_stop - aligned_start );
346 : #endif
347 40751 : }
348 :
349 : /* fd_scratch_prepare starts an allocation of unknown size and known
350 : alignment align (0 means use default alignment) in the caller's
351 : current scratch frame. Returns a pointer in the caller's address
352 : space with alignment align to the first byte of a region with
353 : fd_scratch_free() (as observed after this function returns) bytes
354 : available. The caller is free to clobber any bytes in this region.
355 :
356 : fd_scratch_publish finishes an in-progress allocation. end points at
357 : the first byte after the final allocation. Assumes there is a
358 : matching prepare. A published allocation can be subsequently
359 : trimmed.
360 :
361 : fd_scratch_cancel cancels an in-progress allocation. This is a no-op
362 : if there is no matching prepare. If the prepare had alignment other
363 : than 1, it is possible that some alignment padding needed for the
364 : allocation will still be used in the caller's current scratch frame.
365 : If this is not acceptable, the prepare should use an alignment of 1
366 : and manually align the return.
367 :
368 : This allows idioms like:
369 :
370 : uchar * p = (uchar *)fd_scratch_prepare( align );
371 :
372 : if( FD_UNLIKELY( fd_scratch_free() < app_max_sz ) ) {
373 :
374 : fd_scratch_cancel();
375 :
376 : ... handle too little scratch space to handle application
377 : ... worst case needs here
378 :
379 : } else {
380 :
381 : ... populate sz bytes to p where sz is in [0,app_max_sz]
382 : p += sz;
383 :
384 : fd_scratch_publish( p );
385 :
386 : ... at this point, scratch is as though
387 : ... fd_scratch_alloc( align, sz ) was called above
388 :
389 : }
390 :
391 : Ideally every prepare should be matched with a publish or a cancel,
392 : only one prepare can be in-progress at a time on a thread and prepares
393 : cannot be nested. As such virtually all other scratch operations
394 : will implicitly cancel any in-progress prepare, including attach /
395 : detach / push / pop / prepare / alloc / trim. */
396 :
397 : FD_FN_UNUSED static void * /* Work around -Winline */
398 942795 : fd_scratch_prepare( ulong align ) {
399 :
400 942795 : FD_DCHECK_CRIT( !!fd_scratch_private_frame_cnt, "unmatched push" );
401 942795 : FD_DCHECK_CRIT( fd_scratch_private_align_is_valid( align ), "bad align" );
402 :
403 : # if FD_HAS_DEEPASAN
404 : /* Need 8 byte alignment. */
405 : align = fd_ulong_align_up( align, FD_ASAN_ALIGN );
406 : # endif
407 942795 : ulong true_align = fd_scratch_private_true_align( align );
408 942795 : ulong smem = fd_ulong_align_up( fd_scratch_private_free, true_align );
409 :
410 942795 : FD_DCHECK_CRIT( smem >= fd_scratch_private_free, "prepare align overflow" );
411 942795 : FD_DCHECK_CRIT( smem <= fd_scratch_private_stop, "prepare overflow" );
412 : # if FD_DCHECK_STYLE>0
413 : fd_scratch_in_prepare = 1;
414 : # endif
415 :
416 : # if FD_HAS_DEEPASAN
417 : /* The user can clobber any byte in [smem,stop) and unpoisoning is
418 : exact at the end of a region, so unpoison exactly that. */
419 : fd_asan_unpoison( (void*)smem, fd_scratch_private_stop - smem );
420 : # endif
421 :
422 942795 : fd_scratch_private_free = smem;
423 942795 : return (void *)smem;
424 942795 : }
425 :
426 : static inline void
427 754072 : fd_scratch_publish( void * _end ) {
428 754072 : ulong end = (ulong)_end;
429 :
430 : # if FD_DCHECK_STYLE>0
431 : FD_DCHECK_CRIT( !!fd_scratch_in_prepare, "unmatched prepare" );
432 : # endif
433 754072 : FD_DCHECK_CRIT( end >= fd_scratch_private_free, "publish underflow" );
434 754072 : FD_DCHECK_CRIT( end <= fd_scratch_private_stop, "publish overflow" );
435 : # if FD_DCHECK_STYLE>0
436 : fd_scratch_in_prepare = 0;
437 : # endif
438 :
439 : /* Poison everything that is trimmed off. Conservatively poison potentially
440 : less than the region that is trimmed to respect alignment requirements. */
441 : # if FD_HAS_DEEPASAN
442 : ulong aligned_free = fd_ulong_align_dn( fd_scratch_private_free, FD_ASAN_ALIGN );
443 : ulong aligned_end = fd_ulong_align_up( end, FD_ASAN_ALIGN );
444 : ulong aligned_stop = fd_ulong_align_dn( fd_scratch_private_stop, FD_ASAN_ALIGN );
445 : fd_asan_poison( (void*)aligned_end, aligned_stop - aligned_end );
446 : fd_asan_unpoison( (void*)aligned_free, aligned_end - aligned_free );
447 : # endif
448 : # if FD_HAS_MSAN
449 : ulong aligned_free = fd_ulong_align_dn( fd_scratch_private_free, FD_ASAN_ALIGN );
450 : ulong aligned_end = fd_ulong_align_up( end, FD_ASAN_ALIGN );
451 : ulong aligned_stop = fd_ulong_align_dn( fd_scratch_private_stop, FD_MSAN_ALIGN );
452 : fd_msan_poison( (void*)aligned_end, aligned_stop - aligned_end );
453 : fd_msan_unpoison( (void*)aligned_free, aligned_end - aligned_free );
454 : # endif
455 :
456 754072 : fd_scratch_private_free = end;
457 754072 : }
458 :
459 : static inline void
460 188723 : fd_scratch_cancel( void ) {
461 :
462 : # if FD_DCHECK_STYLE>0
463 : FD_DCHECK_CRIT( !!fd_scratch_in_prepare, "unmatched prepare" );
464 : fd_scratch_in_prepare = 0;
465 : # endif
466 :
467 188723 : }
468 :
469 : /* fd_scratch_alloc allocates sz bytes with alignment align in the
470 : caller's current scratch frame. There should be no prepare in
471 : progress. Note that this has same function signature as
472 : aligned_alloc (and not by accident). It does have some less
473 : restrictive behaviors though.
474 :
475 : align must be 0 or an integer power of 2. 0 will be treated as
476 : FD_SCRATCH_ALIGN_DEFAULT.
477 :
478 : sz need not be a multiple of align. Further, the underlying
479 : allocator does not implicitly round up sz to an align multiple (as
480 : such, scratch can allocate additional items in any tail padding that
481 : might have been implicitly reserved had it rounded up). That is, if
482 : you really want to round up allocations to a multiple of align, then
483 : manually align up sz ... e.g. pass fd_ulong_align_up(sz,align) when
484 : align is non-zero to this call (this could be implemented as a
485 : compile time mode with some small extra overhead if desirable).
486 :
487 : sz 0 is fine. This will currently return a properly aligned non-NULL
488 : pointer (the allocator might do some allocation under the hood to get
489 : the desired alignment and it is possible this might fail ... there is
490 : a case for returning NULL or an arbitrary but appropriately aligned
491 : non-NULL and this could be implemented as a compile time mode with
492 : some small extra overhead if desirable).
493 :
494 : This cannot fail from the caller's point of view (if handholding is
495 : enabled, it will abort the caller with a descriptive error message if
496 : used obviously in error).
497 :
498 : This is freaky fast (O(5) fast asm operations under the hood). */
499 :
500 : FD_FN_UNUSED static void * /* Work around -Winline */
501 : fd_scratch_alloc( ulong align,
502 376912 : ulong sz ) {
503 376912 : ulong smem = (ulong)fd_scratch_prepare( align );
504 376912 : ulong end = smem + sz;
505 :
506 376912 : FD_DCHECK_CRIT( end >= smem, "sz overflow" );
507 376912 : FD_DCHECK_CRIT( end <= fd_scratch_private_stop, "sz overflow" );
508 :
509 376912 : fd_scratch_publish( (void *)end );
510 376912 : return (void *)smem;
511 376912 : }
512 :
513 : /* fd_scratch_trim trims the size of the most recent scratch allocation
514 : in the current scratch frame (technically it can be used to trim the
515 : size of the entire current scratch frame but doing more than the most
516 : recent scratch allocation is strongly discouraged). Assumes there is
517 : a current scratch frame and the caller is not in a prepare. end
518 : points at the first byte to free in the most recent scratch
519 : allocation (or the first byte after the most recent scratch
520 : allocation). This allows idioms like:
521 :
522 : uchar * p = (uchar *)fd_scratch_alloc( align, max_sz );
523 :
524 : ... populate sz bytes of p where sz is in [0,max_sz]
525 : p += sz;
526 :
527 : fd_scratch_trim( p );
528 :
529 : ... now the thread's scratch is as though original call was
530 : ... p = fd_scratch_alloc( align, sz );
531 :
532 : This cannot fail from the caller's point of view (if handholding is
533 : enabled, this will abort the caller with a descriptive error message
534 : if used obviously in error).
535 :
536 : Note that an allocation be repeatedly trimmed.
537 :
538 : Note also that trim can nest. E.g. a thread can call a function that
539 : uses scratch with its own properly matched scratch pushes and pops.
540 : On function return, trim will still work on the most recent scratch
541 : alloc in that frame by the caller.
542 :
543 : This is freaky fast (O(1) fast asm operations under the hood). */
544 :
545 : static inline void
546 753841 : fd_scratch_trim( void * _end ) {
547 753841 : ulong end = (ulong)_end;
548 :
549 753841 : FD_DCHECK_CRIT( !!fd_scratch_private_frame_cnt, "unmatched push" );
550 753841 : FD_DCHECK_CRIT( end >= fd_scratch_private_frame[ fd_scratch_private_frame_cnt-1UL ], "trim underflow" );
551 753841 : FD_DCHECK_CRIT( end <= fd_scratch_private_free, "trim overflow" );
552 : # if FD_DCHECK_STYLE>0
553 : fd_scratch_in_prepare = 0;
554 : # endif
555 :
556 : # if FD_HAS_DEEPASAN
557 : /* The region to poison should be from _end to the end of the scratch's region.
558 : The same alignment considerations need to be taken into account. */
559 : ulong aligned_end = fd_ulong_align_up( end, FD_ASAN_ALIGN );
560 : ulong aligned_stop = fd_ulong_align_dn( fd_scratch_private_stop, FD_ASAN_ALIGN );
561 : fd_asan_poison( (void*)aligned_end, aligned_stop - aligned_end );
562 : # endif
563 : # if FD_HAS_MSAN
564 : ulong aligned_end = fd_ulong_align_up( end, FD_MSAN_ALIGN );
565 : ulong aligned_stop = fd_ulong_align_dn( fd_scratch_private_stop, FD_MSAN_ALIGN );
566 : fd_msan_poison( (void*)aligned_end, aligned_stop - aligned_end );
567 : # endif
568 :
569 753841 : fd_scratch_private_free = end;
570 753841 : }
571 :
572 : /* fd_scratch_*_is_safe returns false (0) if the operation is obviously
573 : unsafe to do at the time of the call or true otherwise.
574 : Specifically:
575 :
576 : fd_scratch_attach_is_safe() returns 1 if the calling thread is not
577 : already attached to scratch.
578 :
579 : fd_scratch_detach_is_safe() returns 1 if the calling thread is
580 : already attached to scratch.
581 :
582 : fd_scratch_reset_is_safe() returns 1 if the calling thread is already
583 : attached to scratch.
584 :
585 : fd_scratch_push_is_safe() returns 1 if there is at least one frame
586 : available and 0 otherwise.
587 :
588 : fd_scratch_pop_is_safe() returns 1 if there is at least one frame
589 : in use and 0 otherwise.
590 :
591 : fd_scratch_prepare_is_safe( align ) returns 1 if there is a current
592 : frame for the allocation and enough scratch pad memory to start
593 : preparing an allocation with alignment align.
594 :
595 : fd_scratch_publish_is_safe( end ) returns 1 if end is a valid
596 : location to complete an allocation in preparation. If handholding is
597 : enabled, will additionally check that there is a prepare already in
598 : progress.
599 :
600 : fd_scratch_cancel_is_safe() returns 1.
601 :
602 : fd_scratch_alloc_is_safe( align, sz ) returns 1 if there is a current
603 : frame for the allocation and enough scratch pad memory for an
604 : allocation with alignment align and size sz.
605 :
606 : fd_scratch_trim_is_safe( end ) returns 1 if there is a current frame
607 : and that current frame can be trimmed to end safely.
608 :
609 : These are safe to call at any time and also freak fast handful of
610 : assembly operations. */
611 :
612 0 : FD_FN_PURE static inline int fd_scratch_attach_is_safe( void ) { return !fd_scratch_private_frame_max; }
613 0 : FD_FN_PURE static inline int fd_scratch_detach_is_safe( void ) { return !!fd_scratch_private_frame_max; }
614 0 : FD_FN_PURE static inline int fd_scratch_reset_is_safe ( void ) { return !!fd_scratch_private_frame_max; }
615 5998546 : FD_FN_PURE static inline int fd_scratch_push_is_safe ( void ) { return fd_scratch_private_frame_cnt<fd_scratch_private_frame_max; }
616 5907812 : FD_FN_PURE static inline int fd_scratch_pop_is_safe ( void ) { return !!fd_scratch_private_frame_cnt; }
617 :
618 : FD_FN_PURE static inline int
619 0 : fd_scratch_prepare_is_safe( ulong align ) {
620 0 : if( FD_UNLIKELY( !fd_scratch_private_frame_cnt ) ) return 0; /* No current frame */
621 0 : if( FD_UNLIKELY( !fd_scratch_private_align_is_valid( align ) ) ) return 0; /* Bad alignment, compile time typically */
622 0 : ulong true_align = fd_scratch_private_true_align( align ); /* compile time typically */
623 0 : ulong smem = fd_ulong_align_up( fd_scratch_private_free, true_align );
624 0 : if( FD_UNLIKELY( smem < fd_scratch_private_free ) ) return 0; /* alignment overflow */
625 0 : if( FD_UNLIKELY( smem > fd_scratch_private_stop ) ) return 0; /* insufficient scratch */
626 0 : return 1;
627 0 : }
628 :
629 : FD_FN_PURE static inline int
630 0 : fd_scratch_publish_is_safe( void * _end ) {
631 0 : ulong end = (ulong)_end;
632 0 : # if FD_DCHECK_STYLE>0
633 0 : if( FD_UNLIKELY( !fd_scratch_in_prepare ) ) return 0; /* Not in prepare */
634 0 : # endif
635 0 : if( FD_UNLIKELY( end < fd_scratch_private_free ) ) return 0; /* Backward */
636 0 : if( FD_UNLIKELY( end > fd_scratch_private_stop ) ) return 0; /* Out of bounds */
637 0 : return 1;
638 0 : }
639 :
640 : FD_FN_CONST static inline int
641 0 : fd_scratch_cancel_is_safe( void ) {
642 0 : return 1;
643 0 : }
644 :
645 : FD_FN_PURE static inline int
646 : fd_scratch_alloc_is_safe( ulong align,
647 2913379 : ulong sz ) {
648 2913379 : if( FD_UNLIKELY( !fd_scratch_private_frame_cnt ) ) return 0; /* No current frame */
649 2601799 : if( FD_UNLIKELY( !fd_scratch_private_align_is_valid( align ) ) ) return 0; /* Bad align, compile time typically */
650 2601799 : ulong true_align = fd_scratch_private_true_align( align ); /* compile time typically */
651 2601799 : ulong smem = fd_ulong_align_up( fd_scratch_private_free, true_align );
652 2601799 : if( FD_UNLIKELY( smem < fd_scratch_private_free ) ) return 0; /* align overflow */
653 2601799 : ulong free = smem + sz;
654 2601799 : if( FD_UNLIKELY( free < smem ) ) return 0; /* sz overflow */
655 2601799 : if( FD_UNLIKELY( free > fd_scratch_private_stop ) ) return 0; /* too little space */
656 753904 : return 1;
657 2601799 : }
658 :
659 : FD_FN_PURE static inline int
660 0 : fd_scratch_trim_is_safe( void * _end ) {
661 0 : ulong end = (ulong)_end;
662 0 : if( FD_UNLIKELY( !fd_scratch_private_frame_cnt ) ) return 0; /* No current frame */
663 0 : if( FD_UNLIKELY( end < fd_scratch_private_frame[ fd_scratch_private_frame_cnt-1UL ] ) ) return 0; /* Trim underflow */
664 0 : if( FD_UNLIKELY( end > fd_scratch_private_free ) ) return 0; /* Trim overflow */
665 0 : return 1;
666 0 : }
667 :
668 : /* FD_SCRATCH_SCOPE_{BEGIN,END} create a `do { ... } while(0);` scope in
669 : which a temporary scratch frame is available. Nested scopes are
670 : permitted. This scratch frame is automatically destroyed when
671 : exiting the scope normally (e.g. by 'break', 'return', or reaching
672 : the end). Uses a dummy variable with a cleanup attribute under the
673 : hood. U.B. if scope is left abnormally (e.g. longjmp(), exception,
674 : abort(), etc.). Use as follows:
675 :
676 : FD_SCRATCH_SCOPE_BEGIN {
677 : ...
678 : fd_scratch_alloc( ... );
679 : ...
680 : }
681 : FD_SCRATCH_SCOPE_END; */
682 :
683 : FD_FN_UNUSED static inline void
684 68 : fd_scratch_scoped_pop_private( void * _unused ) {
685 68 : (void)_unused;
686 68 : fd_scratch_pop();
687 68 : }
688 :
689 68 : #define FD_SCRATCH_SCOPE_BEGIN do { \
690 68 : fd_scratch_push(); \
691 68 : int __fd_scratch_guard_ ## __LINE__ \
692 68 : __attribute__((cleanup(fd_scratch_scoped_pop_private))) \
693 68 : __attribute__((unused)) = 0; \
694 68 : do
695 :
696 68 : #define FD_SCRATCH_SCOPE_END while(0); } while(0)
697 :
698 : /* fd_alloca is variant of alloca that works like aligned_alloc. That
699 : is, it returns an allocation of sz bytes with an alignment of at
700 : least align. Like alloca, this allocation will be in the stack frame
701 : of the calling function with a lifetime of until the calling function
702 : returns. Stack overflow handling is likewise identical to alloca
703 : (stack overflows will overlap the top stack guard, typically
704 : triggering a seg fault when the overflow region is touched that will
705 : be caught and handled by the logger to terminate the calling thread
706 : group). As such, like alloca, these really should only be used for
707 : smallish (<< few KiB) quick allocations in bounded recursion depth
708 : circumstances.
709 :
710 : Like fd_scratch_alloc, align must be an 0 or a non-negative integer
711 : power of 2. 0 will be treated as align_default. align smaller than
712 : align_min will be bumped up to align_min.
713 :
714 : The caller promises request will not overflow the stack. This has to
715 : be implemented as a macro for linguistic reasons and align should be
716 : safe against multiple evaluation and, due to compiler limitations,
717 : must be a compile time constant. Returns non-NULL on success and
718 : NULL on failure (in most situations, can never fail from the caller's
719 : POV). sz==0 is okay (and will return non-NULL). */
720 :
721 : #if FD_HAS_ALLOCA
722 :
723 : /* Work around compiler limitations */
724 9 : #define FD_SCRATCH_PRIVATE_TRUE_ALIGN( align ) ((align) ? (align) : FD_SCRATCH_ALIGN_DEFAULT)
725 :
726 6 : #define fd_alloca(align,sz) __builtin_alloca_with_align( fd_ulong_max( (sz), 1UL ), \
727 6 : 8UL*FD_SCRATCH_PRIVATE_TRUE_ALIGN( (align) ) /*bits*/ )
728 :
729 : /* fd_alloca_check does fd_alloca but it will FD_LOG_CRIT with a
730 : detailed message if the request would cause a stack overflow or leave
731 : so little available free stack that subsequent normal thread
732 : operations would be at risk.
733 :
734 : Note that returning NULL on failure is not an option as this would no
735 : longer be a drop-in instrumented replacement for fd_alloca (this
736 : would also require even more linguistic hacks to keep the fd_alloca
737 : at the appropriate scope). Likewise, testing the allocated region is
738 : within the stack post allocation is not an option as the FD_LOG_CRIT
739 : invocation would then try to use stack with the already overflowed
740 : allocation in it (there is no easy portable way to guarantee an
741 : alloca has been freed short of returning from the function in which
742 : the alloca was performed). Using FD_LOG_ERR instead of FD_LOG_CRIT
743 : is a potentially viable alternative error handling behavior though.
744 :
745 : This has to be implemented as a macro for linguistic reasons. It is
746 : recommended this only be used for development / debugging / testing
747 : purposes (e.g. if you are doing alloca in production that are large
748 : enough you are worried about stack overflow, you probably should be
749 : using fd_scratch, fd_alloc or fd_wksp depending on performance and
750 : persistence needs or, better still, architecting to not need any
751 : temporary memory allocations at all). If the caller's stack
752 : diagnostics could not be successfully initialized (this is logged),
753 : this will always FD_LOG_CRIT. */
754 :
755 : #if !FD_HAS_ASAN
756 :
757 : extern FD_TL ulong fd_alloca_check_private_sz;
758 :
759 : #define fd_alloca_check( align, sz ) \
760 3 : ( fd_alloca_check_private_sz = (sz), \
761 3 : (__extension__({ \
762 3 : ulong _fd_alloca_check_private_pad_max = FD_SCRATCH_PRIVATE_TRUE_ALIGN( (align) ) - 1UL; \
763 3 : ulong _fd_alloca_check_private_footprint = fd_alloca_check_private_sz + _fd_alloca_check_private_pad_max; \
764 3 : if( FD_UNLIKELY( (_fd_alloca_check_private_footprint < _fd_alloca_check_private_pad_max ) | \
765 3 : (_fd_alloca_check_private_footprint > (31UL*(fd_tile_stack_est_free() >> 5))) ) ) \
766 3 : FD_LOG_CRIT(( "fd_alloca_check( " #align ", " #sz " ) stack overflow" )); \
767 3 : })), \
768 3 : fd_alloca( (align), fd_alloca_check_private_sz ) )
769 :
770 : #else /* FD_HAS_ASAN */
771 :
772 : /* AddressSanitizer provides its own alloca safety instrumentation
773 : which are more powerful than the above fd_alloca_check heuristics. */
774 :
775 : #define fd_alloca_check fd_alloca
776 :
777 : #endif /* FD_HAS_ASAN */
778 : #endif /* FD_HAS_ALLOCA */
779 :
780 : FD_PROTOTYPES_END
781 :
782 : #endif /* HEADER_fd_src_util_scratch_fd_scratch_h */
|