Line data Source code
1 : #ifndef HEADER_fd_src_util_fd_util_base_h
2 : #define HEADER_fd_src_util_fd_util_base_h
3 :
4 : /* Base development environment */
5 :
6 : /* Compiler checks ****************************************************/
7 :
8 : #ifdef __cplusplus
9 :
10 : #if __cplusplus<201703L
11 : #error "Firedancer requires C++17 or later"
12 : #endif
13 :
14 : #else
15 :
16 : #if __STDC_VERSION__<201710L
17 : #error "Firedancer requires C Standard version C17 or later"
18 : #endif
19 :
20 : #endif //__cplusplus
21 :
22 : /* Build target capabilities ******************************************/
23 :
24 : /* Different build targets often have different levels of support for
25 : various language and hardware features. The presence of various
26 : features can be tested at preprocessor, compile, or run time via the
27 : below capability macros.
28 :
29 : Code that does not exploit any of these capabilities written within
30 : the base development environment should be broadly portable across a
31 : range of build targets ranging from on-chain virtual machines to
32 : commodity hosts to custom hardware.
33 :
34 : As such, highly portable yet high performance code is possible by
35 : writing generic implementations that do not exploit any of the below
36 : capabilities as a portable fallback along with build target specific
37 : optimized implementations that are invoked when the build target
38 : supports the appropriate capabilities.
39 :
40 : The base development itself provides lots of functionality to help
41 : with implementing portable fallbacks while making very minimal
42 : assumptions about the build targets and zero use of 3rd party
43 : libraries (these might make unknown additional assumptions about the
44 : build target, including availability of a quality implementation of
45 : the library on the build target). */
46 :
47 : /* FD_HAS_HOSTED: If the build target is hosted (e.g. resides on a host
48 : with a POSIX-ish environment ... practically speaking, stdio.h,
49 : stdlib.h, unistd.h, et al more or less behave normally ...
50 : pedantically XOPEN_SOURCE=700), FD_HAS_HOSTED will be 1. It will be
51 : zero otherwise. */
52 :
53 : #ifndef FD_HAS_HOSTED
54 : #define FD_HAS_HOSTED 0
55 : #endif
56 :
57 : /* FD_HAS_ATOMIC: If the build target supports atomic operations
58 : between threads accessing a common memory region (include threads
59 : that reside in different processes on a host communicating via a
60 : shared memory region with potentially different local virtual
61 : mappings). Practically speaking, does atomic compare-and-swap et al
62 : work? */
63 :
64 : #ifndef FD_HAS_ATOMIC
65 : #define FD_HAS_ATOMIC 0
66 : #endif
67 :
68 : /* FD_HAS_THREADS: If the build target supports a POSIX-ish notion of
69 : threads (e.g. practically speaking, global variables declared within
70 : a compile unit are visible to more than one thread of execution,
71 : pthread.h / threading parts of C standard, the atomics parts of the
72 : C standard, ... more or less work normally), FD_HAS_THREADS will be
73 : 1. It will be zero otherwise. FD_HAS_THREADS implies FD_HAS_HOSTED
74 : and FD_HAS_ATOMIC. */
75 :
76 : #ifndef FD_HAS_THREADS
77 : #define FD_HAS_THREADS 0
78 : #endif
79 :
80 : /* FD_HAS_INT128: If the build target supports reasonably efficient
81 : 128-bit wide integer operations, define FD_HAS_INT128 to 1 to enable
82 : use of them in implementations. */
83 :
84 : #ifndef FD_HAS_INT128
85 : #define FD_HAS_INT128 0
86 : #endif
87 :
88 : /* FD_HAS_DOUBLE: If the build target supports reasonably efficient
89 : IEEE 754 64-bit wide double precision floating point operations,
90 : define FD_HAS_DOUBLE to 1 to enable use of them in implementations.
91 : Note that even if the build target does not, va_args handling in the
92 : C / C++ language requires promotion of a float in a va_arg list to a
93 : double. Thus, C / C++ language that support IEEE 754 float also
94 : implies a minimum level of support for double (though not necessarily
95 : efficient or IEEE 754). That is, even if a target does not have
96 : FD_HAS_DOUBLE, there might still be limited use of double in va_arg
97 : list handling. */
98 :
99 : #ifndef FD_HAS_DOUBLE
100 : #define FD_HAS_DOUBLE 0
101 : #endif
102 :
103 : /* FD_HAS_ALLOCA: If the build target supports fast alloca-style
104 : dynamic stack memory allocation (e.g. alloca.h / __builtin_alloca
105 : more or less work normally), define FD_HAS_ALLOCA to 1 to enable use
106 : of it in implementations. */
107 :
108 : #ifndef FD_HAS_ALLOCA
109 : #define FD_HAS_ALLOCA 0
110 : #endif
111 :
112 : /* FD_HAS_X86: If the build target supports x86 specific features and
113 : can benefit from x86 specific optimizations, define FD_HAS_X86. Code
114 : needing more specific target features (Intel / AMD / SSE / AVX2 /
115 : AVX512 / etc) can specialize further as necessary with even more
116 : precise capabilities (that in turn imply FD_HAS_X86). */
117 :
118 : #ifndef FD_HAS_X86
119 : #define FD_HAS_X86 0
120 : #endif
121 :
122 : /* These allow even more precise targeting for X86. */
123 :
124 : /* FD_HAS_SSE indicates the target supports Intel SSE4 style SIMD
125 : (basically do the 128-bit wide parts of "x86intrin.h" work).
126 : Recommend using the simd/fd_sse.h APIs instead of raw Intel
127 : intrinsics for readability and to facilitate portability to non-x86
128 : platforms. Implies FD_HAS_X86. */
129 :
130 : #ifndef FD_HAS_SSE
131 : #define FD_HAS_SSE 0
132 : #endif
133 :
134 : /* FD_HAS_AVX indicates the target supports Intel AVX2 style SIMD
135 : (basically do the 256-bit wide parts of "x86intrin.h" work).
136 : Recommend using the simd/fd_avx.h APIs instead of raw Intel
137 : intrinsics for readability and to facilitate portability to non-x86
138 : platforms. Implies FD_HAS_SSE. */
139 :
140 : #ifndef FD_HAS_AVX
141 : #define FD_HAS_AVX 0
142 : #endif
143 :
144 : /* FD_HAS_AVX512 indicates the target supports Intel AVX-512 style SIMD
145 : (basically do the 512-bit wide parts of "x86intrin.h" work).
146 : Recommend using the simd/fd_avx512.h APIs instead of raw Intel
147 : intrinsics for readability and to facilitate portability to non-x86
148 : platforms. Implies FD_HAS_AVX. */
149 :
150 : #ifndef FD_HAS_AVX512
151 2 : #define FD_HAS_AVX512 0
152 : #endif
153 :
154 : /* FD_HAS_SHANI indicates that the target supports Intel SHA extensions
155 : which accelerate SHA-1 and SHA-256 computation. This extension is
156 : also called SHA-NI or SHA_NI (Secure Hash Algorithm New
157 : Instructions). Although proposed in 2013, they're only supported on
158 : Intel Ice Lake and AMD Zen CPUs and newer. Implies FD_HAS_AVX. */
159 :
160 : #ifndef FD_HAS_SHANI
161 : #define FD_HAS_SHANI 0
162 : #endif
163 :
164 : /* FD_HAS_GFNI indicates that the target supports Intel Galois Field
165 : extensions, which accelerate operations over binary extension fields,
166 : especially GF(2^8). These instructions are supported on Intel Ice
167 : Lake and newer and AMD Zen4 and newer CPUs. Implies FD_HAS_AVX. */
168 :
169 : #ifndef FD_HAS_GFNI
170 : #define FD_HAS_GFNI 0
171 : #endif
172 :
173 : /* FD_HAS_AESNI indicates that the target supports AES-NI extensions,
174 : which accelerate AES encryption and decryption. While AVX predates
175 : the original AES-NI extension, the combination of AES-NI+AVX adds
176 : additional opcodes (such as vaesenc, a more flexible variant of
177 : aesenc). Thus, implies FD_HAS_AVX. A conservative estimate for
178 : minimum platform support is Intel Haswell or AMD Zen. */
179 :
180 : #ifndef FD_HAS_AESNI
181 : #define FD_HAS_AESNI 0
182 : #endif
183 :
184 : /* FD_HAS_ARM: If the build target supports armv8-a specific features
185 : and can benefit from aarch64 specific optimizations, define
186 : FD_HAS_ARM. */
187 :
188 : #ifndef FD_HAS_ARM
189 : #define FD_HAS_ARM 0
190 : #endif
191 :
192 : /* FD_HAS_LZ4 indicates that the target supports LZ4 compression.
193 : Roughly, does "#include <lz4.h>" and the APIs therein work? */
194 :
195 : #ifndef FD_HAS_LZ4
196 57 : #define FD_HAS_LZ4 0
197 : #endif
198 :
199 : /* FD_HAS_COVERAGE indicates that the build target is built with coverage instrumentation. */
200 :
201 : #ifndef FD_HAS_COVERAGE
202 : #define FD_HAS_COVERAGE 0
203 : #endif
204 :
205 : /* FD_HAS_ASAN indicates that the build target is using ASAN. */
206 :
207 : #ifndef FD_HAS_ASAN
208 : #define FD_HAS_ASAN 0
209 : #endif
210 :
211 : /* FD_HAS_UBSAN indicates that the build target is using UBSAN. */
212 :
213 : #ifndef FD_HAS_UBSAN
214 : #define FD_HAS_UBSAN 0
215 : #endif
216 :
217 : /* FD_HAS_DEEPASAN indicates that the build target is using ASAN with manual
218 : memory poisoning for fd_alloc, fd_wksp, and fd_scratch. */
219 :
220 : #ifndef FD_HAS_DEEPASAN
221 : #define FD_HAS_DEEPASAN 0
222 : #endif
223 :
224 : /* Base development environment ***************************************/
225 :
226 : /* The functionality provided by these vanilla headers are always
227 : available within the base development environment. Notably, stdio.h
228 : / stdlib.h / et al are not included here as these make lots of
229 : assumptions about the build target that may not be true (especially
230 : for on-chain and custom hardware use). Code should prefer the fd
231 : util equivalents for such functionality when possible. */
232 :
233 : #include <stdalign.h>
234 : #include <string.h>
235 : #include <limits.h>
236 : #include <float.h>
237 :
238 : /* Work around some library naming irregularities */
239 : /* FIXME: Consider this for FLOAT/FLT, DOUBLE/DBL too? */
240 :
241 3 : #define SHORT_MIN SHRT_MIN
242 3 : #define SHORT_MAX SHRT_MAX
243 476532 : #define USHORT_MAX USHRT_MAX
244 :
245 : /* Primitive types ****************************************************/
246 :
247 : /* These typedefs provide single token regularized names for all the
248 : primitive types in the base development environment:
249 :
250 : char !
251 : schar ! short int long int128 !!
252 : uchar ushort uint ulong uint128 !!
253 : float
254 : double !!!
255 :
256 : ! Does not assume the sign of char. A naked char should be treated
257 : as cstr character and mathematical operations should be avoided on
258 : them. This is less than ideal as the patterns for integer types in
259 : the C/C++ language spec itself are far more consistent with a naked
260 : char naturally being treated as signed (see above). But there are
261 : lots of conflicts between architectures, languages and standard
262 : libraries about this so any use of a naked char shouldn't assume
263 : the sign ... sigh.
264 :
265 : !! Only available if FD_HAS_INT128 is defined
266 :
267 : !!! Should only be used if FD_HAS_DOUBLE is defined but see note in
268 : FD_HAS_DOUBLE about C/C++ silent promotions of float to double in
269 : va_arg lists.
270 :
271 : Note also that these token names more naturally interoperate with
272 : integer constant declarations, type generic code generation
273 : techniques, with printf-style format strings than the stdint.h /
274 : inttypes.h handling.
275 :
276 : To minimize portability issues, unexpected silent type conversion
277 : issues, align with typical developer implicit usage, align with
278 : typical build target usage, ..., assumes char / short / int / long
279 : are 8 / 16 / 32 / 64 twos complement integers and float is IEEE-754
280 : single precision. Further assumes little endian, truncating signed
281 : integer division, sign extending (arithmetic) signed right shift and
282 : signed left shift behaves the same as an unsigned left shift from bit
283 : operations point of view (technically the standard says signed left
284 : shift is undefined if the result would overflow). Also, except for
285 : int128/uint128, assumes that aligned access to these will be
286 : naturally atomic. Lastly assumes that unaligned access to these is
287 : functionally valid but does not assume that unaligned access to these
288 : is efficient or atomic.
289 :
290 : For values meant to be held in registers, code should prefer long /
291 : ulong types (improves asm generation given the prevalence of 64-bit
292 : targets and also to avoid lots of tricky bugs with silent promotions
293 : in the language ... e.g. ushort should ideally only be used for
294 : in-memory representations).
295 :
296 : These are currently not prefixed given how often they are used. If
297 : this becomes problematic prefixes can be added as necessary.
298 : Specifically, C++ allows typedefs to be defined multiple times so
299 : long as they are equivalent. Inequivalent collisions are not
300 : supported but should be rare (e.g. if a 3rd party header thinks
301 : "ulong" should be something other than "unsigned long", the 3rd party
302 : header probably should be nuked from orbit). C11 and forward also
303 : allow multiple equivalent typedefs. C99 and earlier don't but this
304 : is typically only a warning and then only if pedantic warnings are
305 : enabled. Thus, if we want to support users using C99 and earlier who
306 : want to do a strict compile and have a superfluous collision with
307 : these types in other libraries, uncomment the below (or do something
308 : equivalent for the compiler). */
309 :
310 : //#pragma GCC diagnostic push
311 : //#pragma GCC diagnostic ignored "-Wpedantic"
312 :
313 : typedef signed char schar; /* See above note of sadness */
314 :
315 : typedef unsigned char uchar;
316 : typedef unsigned short ushort;
317 : typedef unsigned int uint;
318 : typedef unsigned long ulong;
319 :
320 : #ifdef __SIZEOF_INT128__
321 :
322 : __extension__ typedef __int128 int128;
323 : __extension__ typedef unsigned __int128 uint128;
324 :
325 1229993316 : #define UINT128_MAX (~(uint128)0)
326 6 : #define INT128_MAX ((int128)(UINT128_MAX>>1))
327 3 : #define INT128_MIN (-INT128_MAX-(int128)1)
328 :
329 : #endif
330 :
331 : //#pragma GCC diagnostic pop
332 :
333 : /* Compiler tricks ****************************************************/
334 :
335 : /* FD_STRINGIFY,FD_CONCAT{2,3,4}: Various macros for token
336 : stringification and pasting. FD_STRINGIFY returns the argument as a
337 : cstr (e.g. FD_STRINGIFY(foo) -> "foo"). FD_CONCAT* pastes the tokens
338 : together into a single token (e.g. FD_CONCAT3(a,b,c) -> abc). The
339 : EXPAND variants first expand their arguments and then do the token
340 : operation (e.g. FD_EXPAND_THEN_STRINGIFY(__LINE__) -> "104" if done
341 : on line 104 of the source code file). */
342 :
343 0 : #define FD_STRINGIFY(x)#x
344 7394889 : #define FD_CONCAT2(a,b)a##b
345 39662589 : #define FD_CONCAT3(a,b,c)a##b##c
346 54 : #define FD_CONCAT4(a,b,c,d)a##b##c##d
347 :
348 : #define FD_EXPAND_THEN_STRINGIFY(x)FD_STRINGIFY(x)
349 4219467 : #define FD_EXPAND_THEN_CONCAT2(a,b)FD_CONCAT2(a,b)
350 >16813*10^7 : #define FD_EXPAND_THEN_CONCAT3(a,b,c)FD_CONCAT3(a,b,c)
351 6686721 : #define FD_EXPAND_THEN_CONCAT4(a,b,c,d)FD_CONCAT4(a,b,c,d)
352 :
353 : /* FD_VA_ARGS_SELECT(__VA_ARGS__,e32,e31,...e1): Macro that expands to
354 : en at compile time where n is number of items in the __VA_ARGS__
355 : list. If __VA_ARGS__ is empty, returns e1. Assumes __VA_ARGS__ has
356 : at most 32 arguments. Useful for making a variadic macro whose
357 : behavior depends on the number of arguments in __VA_ARGS__. */
358 :
359 : #define FD_VA_ARGS_SELECT(A,B,C,D,E,F,G,H,I,J,K,L,M,N,O,P,Q,R,S,T,U,V,W,X,Y,Z,a,b,c,d,e,f,_,...)_
360 :
361 : /* FD_SRC_LOCATION returns a const cstr holding the line of code where
362 : FD_SRC_LOCATION was used. */
363 :
364 : #define FD_SRC_LOCATION __FILE__ "(" FD_EXPAND_THEN_STRINGIFY(__LINE__) ")"
365 :
366 : /* FD_STATIC_ASSERT tests at compile time if c is non-zero. If not,
367 : it aborts the compile with an error. err itself should be a token
368 : (e.g. not a string, no whitespace, etc). */
369 :
370 : #ifdef __cplusplus
371 : #define FD_STATIC_ASSERT(c,err) static_assert(c, #err)
372 : #else
373 160542 : #define FD_STATIC_ASSERT(c,err) _Static_assert(c, #err)
374 : #endif
375 :
376 : /* FD_ADDRESS_OF_PACKED_MEMBER(x): Linguistically does &(x) but without
377 : recent compiler complaints that &x might be unaligned if x is a
378 : member of a packed datastructure. (Often needed for interfacing with
379 : hardware / packets / etc.) */
380 :
381 3 : #define FD_ADDRESS_OF_PACKED_MEMBER( x ) (__extension__({ \
382 3 : char * _fd_aopm = (char *)&(x); \
383 3 : __asm__( "# FD_ADDRESS_OF_PACKED_MEMBER(" #x ") @" FD_SRC_LOCATION : "+r" (_fd_aopm) :: ); \
384 3 : (__typeof__(&(x)))_fd_aopm; \
385 3 : }))
386 :
387 : /* FD_PROTOTYPES_{BEGIN,END}: Headers that might be included in C++
388 : source should encapsulate the prototypes of code and globals
389 : contained in compilation units compiled as C with a
390 : FD_PROTOTYPES_{BEGIN,END} pair. */
391 :
392 : #ifdef __cplusplus
393 : #define FD_PROTOTYPES_BEGIN extern "C" {
394 : #else
395 : #define FD_PROTOTYPES_BEGIN
396 : #endif
397 :
398 : #ifdef __cplusplus
399 : #define FD_PROTOTYPES_END }
400 : #else
401 : #define FD_PROTOTYPES_END
402 : #endif
403 :
404 : /* FD_ASM_LG_ALIGN(lg_n) expands to an alignment assembler directive
405 : appropriate for the current architecture/ABI. The resulting align
406 : is 2^(lg_n) bytes, i.e. FD_ASM_LG_ALIGN(3) aligns by 8 bytes. */
407 :
408 : #if defined(__aarch64__)
409 : #define FD_ASM_LG_ALIGN(lg_n) ".align " #lg_n "\n"
410 : #elif defined(__x86_64__) || defined(__powerpc64__) || defined(__riscv)
411 : #define FD_ASM_LG_ALIGN(lg_n) ".p2align " #lg_n "\n"
412 : #endif
413 :
414 : /* FD_IMPORT declares a variable name and initializes with the contents
415 : of the file at path (with potentially some assembly directives for
416 : additional footer info). It is equivalent to:
417 :
418 : type const name[] __attribute__((aligned(align))) = {
419 :
420 : ... code that would initialize the contents of name to the
421 : ... raw binary data found in the file at path at compile time
422 : ... (with any appended information as specified by footer)
423 :
424 : };
425 :
426 : ulong const name_sz = ... number of bytes pointed to by name;
427 :
428 : More precisely, this creates a symbol "name" in the object file that
429 : points to a read-only copy of the raw data in the file at "path" as
430 : it was at compile time. 2^lg_align specifies the minimum alignment
431 : required for the copy's first byte as an unsuffixed decimal integer.
432 : footer are assembly commands to permit additional data to be appended
433 : to the copy (use "" for footer if no footer is necessary).
434 :
435 : Then it exposes a pointer to this copy in the current compilation
436 : unit as name and the byte size as name_sz. name_sz covers the first
437 : byte of the included data to the last byte of the footer inclusive.
438 :
439 : The dummy linker symbol _fd_import_name_sz will also be created in
440 : the object file as some under the hood magic to make this work. This
441 : should not be used in any compile unit as some compilers (I'm looking
442 : at you clang-15, but apparently not clang-10) will sometimes mangle
443 : its value from what it was set to in the object file even when marked
444 : as absolute in the object file.
445 :
446 : This should only be used at global scope and should be done at most
447 : once over all object files / libraries used to make a program. If
448 : other compilation units want to make use of an import in a different
449 : compilation unit, they should declare:
450 :
451 : extern type const name[] __attribute__((aligned(align)));
452 :
453 : and/or:
454 :
455 : extern ulong const name_sz;
456 :
457 : as necessary (that is, do the usual to use name and name_sz as shown
458 : for the pseudo code above).
459 :
460 : Important safety tip! gcc -M will generally not detect the
461 : dependency this creates between the importing file and the imported
462 : file. This can cause incremental builds to miss changes to the
463 : imported file. Ideally, we would have FD_IMPORT automatically do
464 : something like:
465 :
466 : _Pragma( "GCC dependency \"" path "\" )
467 :
468 : This doesn't work as is because _Pragma needs some macro expansion
469 : hacks to accept this (this is doable). After that workaround, this
470 : still doesn't work because, due to tooling limitations, the pragma
471 : path is relative to the source file directory and the FD_IMPORT path
472 : is relative to the make directory (working around this would
473 : require a __FILE__-like directive for the source code directory base
474 : path). Even if that did exist, it might still not work because
475 : out-of-tree builds often require some substitutions to the gcc -M
476 : generated dependencies that this might not pick up (at least not
477 : without some build system surgery). And then it still wouldn't work
478 : because gcc -M seems to ignore all of this anyways (which is the
479 : actual show stopper as this pragma does something subtly different
480 : than what the name suggests and there isn't any obvious support for a
481 : "pseudo-include".) Another reminder that make clean and fast builds
482 : are our friend. */
483 :
484 : #if defined(__ELF__)
485 :
486 : #define FD_IMPORT( name, path, type, lg_align, footer ) \
487 : __asm__( ".section .rodata,\"a\",@progbits\n" \
488 : ".type " #name ",@object\n" \
489 : ".globl " #name "\n" \
490 : FD_ASM_LG_ALIGN(lg_align) \
491 : #name ":\n" \
492 : ".incbin \"" path "\"\n" \
493 : footer "\n" \
494 : ".size " #name ",. - " #name "\n" \
495 : "_fd_import_" #name "_sz = . - " #name "\n" \
496 : ".type " #name "_sz,@object\n" \
497 : ".globl " #name "_sz\n" \
498 : FD_ASM_LG_ALIGN(3) \
499 : #name "_sz:\n" \
500 : ".quad _fd_import_" #name "_sz\n" \
501 : ".size " #name "_sz,8\n" \
502 : ".previous\n" ); \
503 : extern type const name[] __attribute__((aligned(1<<(lg_align)))); \
504 : extern ulong const name##_sz
505 :
506 : #elif defined(__MACH__)
507 :
508 : #define FD_IMPORT( name, path, type, lg_align, footer ) \
509 : __asm__( ".section __DATA,__const\n" \
510 : ".globl _" #name "\n" \
511 : FD_ASM_LG_ALIGN(lg_align) \
512 : "_" #name ":\n" \
513 : ".incbin \"" path "\"\n" \
514 : footer "\n" \
515 : "_fd_import_" #name "_sz = . - _" #name "\n" \
516 : ".globl _" #name "_sz\n" \
517 : FD_ASM_LG_ALIGN(3) \
518 : "_" #name "_sz:\n" \
519 : ".quad _fd_import_" #name "_sz\n" \
520 : ".previous\n" ); \
521 : extern type const name[] __attribute__((aligned(1<<(lg_align)))); \
522 : extern ulong const name##_sz
523 :
524 : #endif
525 :
526 : /* FD_IMPORT_{BINARY,CSTR} are common cases for FD_IMPORT.
527 :
528 : In BINARY, the file is imported into the object file and exposed to
529 : the caller as a uchar binary data. name_sz will be the number of
530 : bytes in the file at time of import. name will have 128 byte
531 : alignment.
532 :
533 : In CSTR, the file is imported into the object file with a '\0'
534 : termination appended and exposed to the caller as a cstr. Assuming
535 : the file is text (i.e. has no internal '\0's), strlen(name) will be
536 : the number of bytes in the file and name_sz will be strlen(name)+1.
537 : name can have arbitrary alignment. */
538 :
539 : #ifdef FD_IMPORT
540 : #define FD_IMPORT_BINARY(name, path) FD_IMPORT( name, path, uchar, 7, "" )
541 : #define FD_IMPORT_CSTR( name, path) FD_IMPORT( name, path, char, 1, ".byte 0" )
542 : #endif
543 :
544 : /* Optimizer tricks ***************************************************/
545 :
546 : /* FD_RESTRICT is a pointer modifier to designate a pointer as
547 : restricted. Hoops jumped because C++-17 still doesn't understand
548 : restrict ... sigh */
549 :
550 : #ifndef FD_RESTRICT
551 : #ifdef __cplusplus
552 : #define FD_RESTRICT __restrict
553 : #else
554 : #define FD_RESTRICT restrict
555 : #endif
556 : #endif
557 :
558 : /* fd_type_pun(p), fd_type_pun_const(p): These allow use of type
559 : punning while keeping strict aliasing optimizations enabled (e.g.
560 : some UNIX APIs, like sockaddr related APIs are dependent on type
561 : punning). These allow these API's to be used cleanly while keeping
562 : strict aliasing optimizations enabled and strict alias checking done. */
563 :
564 : static inline void *
565 675093687 : fd_type_pun( void * p ) {
566 675093687 : __asm__( "# fd_type_pun @" FD_SRC_LOCATION : "+r" (p) :: "memory" );
567 675093687 : return p;
568 675093687 : }
569 :
570 : static inline void const *
571 219411113 : fd_type_pun_const( void const * p ) {
572 219411113 : __asm__( "# fd_type_pun_const @" FD_SRC_LOCATION : "+r" (p) :: "memory" );
573 219411113 : return p;
574 219411113 : }
575 :
576 : /* FD_{LIKELY,UNLIKELY}(c): Evaluates c and returns whether it is
577 : logical true/false as long (1L/0L). It also hints to the optimizer
578 : whether it should optimize for the case of c evaluating as
579 : true/false. */
580 :
581 98137962161 : #define FD_LIKELY(c) __builtin_expect( !!(c), 1L )
582 >28735*10^7 : #define FD_UNLIKELY(c) __builtin_expect( !!(c), 0L )
583 :
584 : /* FD_FN_PURE hints to the optimizer that the function, roughly
585 : speaking, does not have side effects. As such, the compiler can
586 : replace a call to the function with the result of an earlier call to
587 : that function provided the inputs and memory used haven't changed.
588 :
589 : IMPORTANT SAFETY TIP! Recent compilers seem to take an undocumented
590 : and debatable stance that pure functions do no writes to memory.
591 : This is a sufficient condition for the above but not a necessary one.
592 :
593 : Consider, for example, the real world case of an otherwise pure
594 : function that uses pass-by-reference to return more than one value
595 : (an unpleasant practice that is sadly often necessary because C/C++,
596 : compilers and underlying platform ABIs are very bad at helping
597 : developers simply and clearly express their intent to return multiple
598 : values and then generate good assembly for such).
599 :
600 : If called multiple times sequentially, all but the first call to such
601 : a "pure" function could be optimized away because the non-volatile
602 : memory writes done in the all but the 1st call for the
603 : pass-by-reference-returns write the same value to normal memory that
604 : was written on the 1st call. That is, these calls return the same
605 : value for their direct return and do writes that do not have any
606 : visible effect.
607 :
608 : Thus, while it is safe for the compiler to eliminate all but the
609 : first call via techniques like common subexpression elimination, it
610 : is not safe for the compiler to infer that the first call did no
611 : writes.
612 :
613 : But recent compilers seem to do exactly that.
614 :
615 : Sigh ... we can't use FD_FN_PURE on such functions because of all the
616 : above linguistic, compiler, documentation and ABI infinite sadness.
617 :
618 : TL;DR To be safe against the above vagaries, recommend using
619 : FD_FN_PURE to annotate functions that do no memory writes (including
620 : trivial memory writes) and try to design HPC APIs to avoid returning
621 : multiple values as much as possible.
622 :
623 : Followup: FD_FN_PURE expands to nothing by default given additional
624 : confusion between how current languages, compilers, CI, fuzzing, and
625 : developers interpret this function attribute. We keep it around
626 : given it documents the intent of various APIs and so it can be
627 : manually enabled to find implementation surprises during bullet
628 : proofing (e.g. under compiler options like "extra-brutality").
629 : Hopefully someday, pure function attributes will someday be handled
630 : more consistently across the board. */
631 :
632 : #ifndef FD_FN_PURE
633 : #define FD_FN_PURE
634 : #endif
635 :
636 : /* FD_FN_CONST is like pure but also, even stronger, indicates that the
637 : function does not depend on the state of memory. See note above
638 : about why this expands to nothing by default. */
639 :
640 : #ifndef FD_FN_CONST
641 : #define FD_FN_CONST
642 : #endif
643 :
644 : /* FD_FN_UNUSED indicates that it is okay if the function with static
645 : linkage is not used. Allows working around -Winline in header only
646 : APIs where the compiler decides not to actually inline the function.
647 : (This belief, frequently promulgated by anti-macro cults, that "An
648 : Inline Function is As Fast As a Macro" ... an entire section in gcc's
649 : documentation devoted to it in fact ... remains among the biggest
650 : lies in computer science. Yes, an inline function is as fast as a
651 : macro ... when the compiler actually decides to treat the inline
652 : keyword more than just for entertainment purposes only. Which, as
653 : -Winline proves, it frequently doesn't. Sigh ... force_inline like
654 : compiler extensions might be an alternative here but they have their
655 : own portability issues.) */
656 :
657 48 : #define FD_FN_UNUSED __attribute__((unused))
658 :
659 : /* FD_FN_UNSANITIZED tells the compiler to disable AddressSanitizer and
660 : UndefinedBehaviorSanitizer instrumentation. For some functions, this
661 : can improve instrumented compile time by ~30x. */
662 :
663 : #if FD_HAS_MSAN
664 : #define FD_FN_UNSANITIZED __attribute__((no_sanitize("memory")))
665 : #else
666 : #define FD_FN_UNSANITIZED __attribute__((no_sanitize("address", "undefined")))
667 : #endif
668 :
669 : /* FD_FN_SENSITIVE instruments the compiler to sanitize sensitive functions.
670 : https://eprint.iacr.org/2023/1713 (Sec 3.2)
671 : - Clear all registers with __attribute__((zero_call_used_regs("all")))
672 : - Clear stack with __attribute__((strub)), available in gcc 14+ */
673 :
674 : #if __has_attribute(strub)
675 : #define FD_FN_SENSITIVE __attribute__((strub)) __attribute__((zero_call_used_regs("all")))
676 : #elif __has_attribute(zero_call_used_regs)
677 : #define FD_FN_SENSITIVE __attribute__((zero_call_used_regs("all")))
678 : #else
679 : #define FD_FN_SENSITIVE
680 : #endif
681 :
682 : /* FD_PARAM_UNUSED indicates that it is okay if the function parameter is not
683 : used. */
684 :
685 : #define FD_PARAM_UNUSED __attribute__((unused))
686 :
687 : /* FD_TYPE_PACKED indicates that a type is to be packed, resetting its
688 : alignment to 1. */
689 :
690 : #define FD_TYPE_PACKED __attribute__((packed))
691 :
692 : /* FD_WARN_UNUSED tells the compiler the result (from a function) should
693 : be checked. This is useful to force callers to either check the result
694 : or deliberately and explicitly ignore it. Good for result codes and
695 : errors */
696 :
697 : #define FD_WARN_UNUSED __attribute__ ((warn_unused_result))
698 :
699 : /* FD_FALLTHRU tells the compiler that a case in a switch falls through
700 : to the next case. This avoids the compiler complaining, in cases where
701 : it is an intentional fall through.
702 : The "while(0)" avoids a compiler complaint in the event the case
703 : has no statement, example:
704 : switch( return_code ) {
705 : case RETURN_CASE_1: FD_FALLTHRU;
706 : case RETURN_CASE_2: FD_FALLTHRU;
707 : case RETURN_CASE_3:
708 : case_123();
709 : default:
710 : case_other();
711 : }
712 :
713 : See C++17 [[fallthrough]] and gcc __attribute__((fallthrough)) */
714 :
715 : #define FD_FALLTHRU while(0) __attribute__((fallthrough))
716 :
717 : /* FD_COMPILER_FORGET(var): Tells the compiler that it shouldn't use
718 : any knowledge it has about the provided register-compatible variable
719 : var for optimizations going forward (i.e. the variable has changed in
720 : a deterministic but unknown-to-the-compiler way where the actual
721 : change is the identity operation). Useful for inhibiting various
722 : branch nest misoptimizations (compilers unfortunately tend to
723 : radically underestimate the impact in raw average performance and
724 : jitter and the probability of branch mispredicts or the cost to the
725 : CPU of having lots of branches). This is not asm volatile (use
726 : UNPREDICTABLE below for that) and has no clobbers. So if var is not
727 : used after the forget, the compiler can optimize the FORGET away
728 : (along with operations preceding it used to produce var). */
729 :
730 25948934601 : #define FD_COMPILER_FORGET(var) __asm__( "# FD_COMPILER_FORGET(" #var ")@" FD_SRC_LOCATION : "+r" (var) )
731 :
732 : /* FD_COMPILER_UNPREDICTABLE(var): Same as FD_COMPILER_FORGET(var) but
733 : the provided variable has changed in a non-deterministic way from the
734 : compiler's POV (e.g. the value in the variable on output should not
735 : be treated as a compile time constant even if it is one
736 : linguistically). Useful for suppressing unwanted
737 : compile-time-const-based optimizations like hoisting operations with
738 : useful CPU side effects out of a critical loop. */
739 :
740 40845311 : #define FD_COMPILER_UNPREDICTABLE(var) __asm__ __volatile__( "# FD_COMPILER_UNPREDICTABLE(" #var ")@" FD_SRC_LOCATION : "+m,r" (var) )
741 :
742 : /* Atomic tricks ******************************************************/
743 :
744 : /* FD_COMPILER_MFENCE(): Tells the compiler that it can't move any
745 : memory operations (load or store) from before the MFENCE to after the
746 : MFENCE (and vice versa). The processor itself might still reorder
747 : around the fence though (that requires platform specific fences). */
748 :
749 33165621426 : #define FD_COMPILER_MFENCE() __asm__ __volatile__( "# FD_COMPILER_MFENCE()@" FD_SRC_LOCATION ::: "memory" )
750 :
751 : /* FD_HW_MFENCE(): A full hardware memory fence. All prior stores
752 : are globally visible before any subsequent loads execute. Use
753 : when a compiler fence is insufficient, e.g. epoch-based safe
754 : reclamation where a store must be visible to another core before
755 : a dependent load on this core (StoreLoad barrier). */
756 :
757 : #if FD_HAS_X86
758 10437747 : #define FD_HW_MFENCE() __asm__ __volatile__( "lock addl $0, (%%rsp)" ::: "memory", "cc" )
759 24615969696 : #define FD_HW_MFENCE_LD() FD_COMPILER_MFENCE()
760 : #define FD_HW_MFENCE_ST() FD_COMPILER_MFENCE()
761 : #elif FD_HAS_ARM
762 : #define FD_HW_MFENCE() __asm__ __volatile__( "dmb ish" ::: "memory" )
763 : #define FD_HW_MFENCE_LD() __asm__ __volatile__( "dmb ishld" ::: "memory" )
764 : #define FD_HW_MFENCE_ST() __asm__ __volatile__( "dmb ishst" ::: "memory" )
765 : #else
766 : #define FD_HW_MFENCE() __sync_synchronize()
767 : #define FD_HW_MFENCE_LD() __sync_synchronize()
768 : #define FD_HW_MFENCE_ST() __sync_synchronize()
769 : #endif
770 :
771 : /* FD_SPIN_PAUSE(): Yields the logical core of the calling thread to
772 : the other logical cores sharing the same underlying physical core for
773 : a few clocks without yielding it to the operating system scheduler.
774 : Typically useful for shared memory spin polling loops, especially if
775 : hyperthreading is in use. IMPORTANT SAFETY TIP! This might act as a
776 : FD_COMPILER_MFENCE on some combinations of toolchains and targets
777 : (e.g. gcc documents that __builtin_ia32_pause also does a compiler
778 : memory fence) but this should not be relied upon for portable code
779 : (consider making this a compiler memory fence on all platforms?) */
780 :
781 : #if FD_HAS_X86
782 19516110082 : #define FD_SPIN_PAUSE() __builtin_ia32_pause()
783 : #elif FD_HAS_ARM
784 : #define FD_SPIN_PAUSE() __asm__ __volatile__( "yield" ::: "memory" )
785 : #else
786 : #define FD_SPIN_PAUSE() ((void)0)
787 : #endif
788 :
789 : /* FD_YIELD(): Yields the logical core of the calling thread to the
790 : operating system scheduler if a hosted target and does a spin pause
791 : otherwise. */
792 :
793 : #if FD_HAS_HOSTED
794 2857740 : #define FD_YIELD() fd_yield()
795 : #else
796 : #define FD_YIELD() FD_SPIN_PAUSE()
797 : #endif
798 :
799 : /* FD_VOLATILE_CONST(x): Tells the compiler that it is not able to
800 : predict the value obtained by dereferencing x and that dereferencing
801 : x might have other side effects (e.g. maybe another thread could
802 : change the value and the compiler has no way of knowing this).
803 : Generally speaking, the volatile keyword is broken linguistically.
804 : Volatility is not a property of the variable but of the
805 : dereferencing of a variable (e.g. what is volatile from the POV of a
806 : reader of a shared variable is not necessarily volatile from the POV
807 : of a writer of that shared variable in a different thread). */
808 :
809 1556816673 : #define FD_VOLATILE_CONST(x) (*((volatile const __typeof__((x)) *)&(x)))
810 :
811 : /* FD_VOLATILE(x): tells the compiler that it is not able to predict the
812 : effect of modifying x and that dereferencing x might have other side
813 : effects (e.g. maybe another thread is spinning on x waiting for its
814 : value to change and the compiler has no way of knowing this). */
815 :
816 659572089 : #define FD_VOLATILE(x) (*((volatile __typeof__((x)) *)&(x)))
817 :
818 : /* FD_ATOMIC_FETCH_AND_{ADD,SUB,OR,AND,XOR}(p,v):
819 :
820 : FD_ATOMIC_FETCH_AND_ADD(p,v) does
821 : f = *p;
822 : *p = f + v
823 : return f;
824 : as a single atomic operation. Similarly for the other variants. */
825 :
826 3204424 : #define FD_ATOMIC_FETCH_AND_ADD(p,v) __sync_fetch_and_add( (p), (v) )
827 3120807 : #define FD_ATOMIC_FETCH_AND_SUB(p,v) __sync_fetch_and_sub( (p), (v) )
828 80274883 : #define FD_ATOMIC_FETCH_AND_OR( p,v) __sync_fetch_and_or( (p), (v) )
829 3606 : #define FD_ATOMIC_FETCH_AND_AND(p,v) __sync_fetch_and_and( (p), (v) )
830 : #define FD_ATOMIC_FETCH_AND_XOR(p,v) __sync_fetch_and_xor( (p), (v) )
831 :
832 : /* FD_ATOMIC_{ADD,SUB,OR,AND,XOR}_AND_FETCH(p,v):
833 :
834 : FD_ATOMIC_{ADD,SUB,OR,AND,XOR}_AND_FETCH(p,v) does
835 : r = *p + v;
836 : *p = r;
837 : return r;
838 : as a single atomic operation. Similarly for the other variants. */
839 :
840 390648 : #define FD_ATOMIC_ADD_AND_FETCH(p,v) __sync_add_and_fetch( (p), (v) )
841 393210 : #define FD_ATOMIC_SUB_AND_FETCH(p,v) __sync_sub_and_fetch( (p), (v) )
842 : #define FD_ATOMIC_OR_AND_FETCH( p,v) __sync_or_and_fetch( (p), (v) )
843 : #define FD_ATOMIC_AND_AND_FETCH(p,v) __sync_and_and_fetch( (p), (v) )
844 : #define FD_ATOMIC_XOR_AND_FETCH(p,v) __sync_xor_and_fetch( (p), (v) )
845 :
846 : /* FD_ATOMIC_CAS(p,c,s):
847 :
848 : o = FD_ATOMIC_CAS(p,c,s) conceptually does:
849 : o = *p;
850 : if( o==c ) *p = s;
851 : return o
852 : as a single atomic operation. */
853 :
854 70882901 : #define FD_ATOMIC_CAS(p,c,s) __sync_val_compare_and_swap( (p), (c), (s) )
855 :
856 : /* FD_ATOMIC_XCHG(p,v):
857 :
858 : o = FD_ATOMIC_XCHG( p, v ) conceptually does:
859 : o = *p
860 : *p = v
861 : return o
862 : as a single atomic operation. */
863 :
864 4080387 : #define FD_ATOMIC_XCHG(p,v) __atomic_exchange_n( (p), (v), __ATOMIC_SEQ_CST )
865 :
866 : /* FD_TL: This indicates that the variable should be thread local.
867 :
868 : FD_ONCE_{BEGIN,END}: The block:
869 :
870 : FD_ONCE_BEGIN {
871 : ... code ...
872 : } FD_ONCE_END
873 :
874 : linguistically behaves like:
875 :
876 : do {
877 : ... code ...
878 : } while(0)
879 :
880 : But provides a low overhead guarantee that:
881 : - The block will be executed at most once over all threads
882 : in a process (i.e. the set of threads which share global
883 : variables).
884 : - No thread in a process that encounters the block will continue
885 : past it until it has executed once.
886 :
887 : This implies that caller promises a ONCE block will execute in a
888 : finite time. (Meant for doing simple lightweight initializations.)
889 :
890 : It is okay to nest ONCE blocks. The thread that executes the
891 : outermost will execute all the nested once as part of executing the
892 : outermost.
893 :
894 : A ONCE implicitly provides a compiler memory fence to reduce the risk
895 : that the compiler will assume that operations done in the once block
896 : on another thread have not been done (e.g. propagating pre-once block
897 : variable values into post-once block code). It is up to the user to
898 : provide any necessary hardware fencing (usually not necessary).
899 :
900 : FD_THREAD_ONCE_{BEGIN,END}: The block:
901 :
902 : FD_THREAD_ONCE_BEGIN {
903 : ... code ...
904 : } FD_THREAD_ONCE_END;
905 :
906 : is similar except the guarantee is that the block only covers the
907 : invoking thread and it does not provide any fencing. If a thread
908 : once begin is nested inside a once begin, that thread once begin will
909 : only be executed on the thread that executes the thread once begin.
910 : It is similarly okay to nest ONCE block inside a THREAD_ONCE block.
911 :
912 : FD_TURNSTILE_{BEGIN,BLOCKED,END} implement a turnstile for all
913 : threads in a process. Only one thread can be in the turnstile at a
914 : time. Usage:
915 :
916 : FD_TURNSTILE_BEGIN(blocking) {
917 :
918 : ... At this point, we are the only thread executing this block of
919 : ... code.
920 : ...
921 : ... Do operations that must be done by threads one-at-a-time
922 : ... here.
923 : ...
924 : ... Because compiler memory fences are done just before entering
925 : ... and after exiting this block, there is typically no need to
926 : ... use any atomics / volatile / fencing here. That is, we can
927 : ... just write "normal" code on platforms where writes to memory
928 : ... become visible to other threads in the order in which they
929 : ... were issued in the machine code (e.g. x86) as others will not
930 : ... proceed with this block until they exit it. YMMV for non-x86
931 : ... platforms (probably need additional hardware store fences in
932 : ... these macros).
933 : ...
934 : ... It is safe to use "break" and/or "continue" within this
935 : ... block. The block will exit with the appropriate compiler
936 : ... fencing and unlocking. Execution will resume immediately
937 : ... after FD_TURNSTILE_END.
938 :
939 : ... IMPORTANT SAFETY TIP! DO NOT RETURN FROM THIS BLOCK.
940 :
941 : } FD_TURNSTILE_BLOCKED {
942 :
943 : ... At this point, there was another thread in the turnstile when
944 : ... we tried to enter the turnstile.
945 : ...
946 : ... Handle blocked here.
947 : ...
948 : ... On exiting this block, if blocking was zero, we will resume
949 : ... execution immediately after FD_TURNSTILE_END. If blocking
950 : ... was non-zero, we will resume execution immediately before
951 : ... FD_TURNSTILE_BEGIN (e.g. we will retry again after a short
952 : ... spin pause).
953 : ...
954 : ... It is safe to use "break" and/or "continue" within this
955 : ... block. Both will exit this block and resume execution
956 : ... at the location indicated as per what blocking specified
957 : ... when the turnstile was entered.
958 : ...
959 : ... It is technically safe to return from this block but
960 : ... also extremely gross.
961 :
962 : } FD_TURNSTILE_END; */
963 :
964 : #if FD_HAS_THREADS /* Potentially more than one thread in the process */
965 :
966 : #ifndef FD_TL
967 : #define FD_TL __thread
968 : #endif
969 :
970 4757 : #define FD_ONCE_BEGIN do { \
971 4757 : FD_COMPILER_MFENCE(); \
972 4757 : static volatile int _fd_once_block_state = 0; \
973 4757 : for(;;) { \
974 4757 : int _fd_once_block_tmp = _fd_once_block_state; \
975 4757 : if( FD_LIKELY( _fd_once_block_tmp>0 ) ) break; \
976 4757 : if( FD_LIKELY( !_fd_once_block_tmp ) && \
977 2716 : FD_LIKELY( !FD_ATOMIC_CAS( &_fd_once_block_state, 0, -1 ) ) ) { \
978 2716 : do
979 :
980 : #define FD_ONCE_END \
981 2716 : while(0); \
982 2716 : FD_COMPILER_MFENCE(); \
983 2716 : _fd_once_block_state = 1; \
984 2716 : break; \
985 2716 : } \
986 2716 : FD_YIELD(); \
987 0 : } \
988 4757 : } while(0)
989 :
990 36 : #define FD_THREAD_ONCE_BEGIN do { \
991 36 : static FD_TL int _fd_thread_once_block_state = 0; \
992 36 : if( FD_UNLIKELY( !_fd_thread_once_block_state ) ) { \
993 9 : do
994 :
995 : #define FD_THREAD_ONCE_END \
996 9 : while(0); \
997 9 : _fd_thread_once_block_state = 1; \
998 9 : } \
999 36 : } while(0)
1000 :
1001 9 : #define FD_TURNSTILE_BEGIN(blocking) do { \
1002 9 : static volatile int _fd_turnstile_state = 0; \
1003 9 : int _fd_turnstile_blocking = (blocking); \
1004 9 : for(;;) { \
1005 9 : int _fd_turnstile_tmp = _fd_turnstile_state; \
1006 9 : if( FD_LIKELY( !_fd_turnstile_tmp ) && \
1007 9 : FD_LIKELY( !FD_ATOMIC_CAS( &_fd_turnstile_state, 0, 1 ) ) ) { \
1008 9 : FD_COMPILER_MFENCE(); \
1009 9 : do
1010 :
1011 : #define FD_TURNSTILE_BLOCKED \
1012 9 : while(0); \
1013 9 : FD_COMPILER_MFENCE(); \
1014 9 : _fd_turnstile_state = 0; \
1015 9 : FD_COMPILER_MFENCE(); \
1016 9 : break; \
1017 9 : } \
1018 9 : FD_COMPILER_MFENCE(); \
1019 0 : do
1020 :
1021 : #define FD_TURNSTILE_END \
1022 0 : while(0); \
1023 0 : FD_COMPILER_MFENCE(); \
1024 0 : if( !_fd_turnstile_blocking ) break; /* likely compile time */ \
1025 0 : FD_SPIN_PAUSE(); \
1026 0 : } \
1027 9 : } while(0)
1028 :
1029 : #else /* Only one thread in the process */
1030 :
1031 : #ifndef FD_TL
1032 : #define FD_TL /**/
1033 : #endif
1034 :
1035 : #define FD_ONCE_BEGIN do { \
1036 : static int _fd_once_block_state = 0; \
1037 : if( FD_UNLIKELY( !_fd_once_block_state ) ) { \
1038 : do
1039 :
1040 : #define FD_ONCE_END \
1041 : while(0); \
1042 : _fd_once_block_state = 1; \
1043 : } \
1044 : } while(0)
1045 :
1046 : #define FD_THREAD_ONCE_BEGIN FD_ONCE_BEGIN
1047 : #define FD_THREAD_ONCE_END FD_ONCE_END
1048 :
1049 : #define FD_TURNSTILE_BEGIN(blocking) do { \
1050 : (void)(blocking); \
1051 : FD_COMPILER_MFENCE(); \
1052 : if( 1 ) { \
1053 : do
1054 :
1055 : #define FD_TURNSTILE_BLOCKED \
1056 : while(0); \
1057 : } else { \
1058 : do
1059 :
1060 : #define FD_TURNSTILE_END \
1061 : while(0); \
1062 : } \
1063 : FD_COMPILER_MFENCE(); \
1064 : } while(0)
1065 :
1066 : #endif
1067 :
1068 : /* An ideal fd_clock_func_t is a function such that:
1069 :
1070 : long dx = clock( args );
1071 : ... stuff ...
1072 : dx = clock( args ) - dx;
1073 :
1074 : yields a strictly positive dx where dx approximates the amount of
1075 : wallclock time elapsed on the caller in some clock specific unit
1076 : (e.g. nanoseconds, CPU ticks, etc) for a reasonable amount of "stuff"
1077 : (including no "stuff"). args allows arbitrary clock specific context
1078 : to be passed to the clock implementation. (clocks that need a
1079 : non-const args can cast away the const in the implementation or cast
1080 : the function pointer as necessary.) */
1081 :
1082 : typedef long (*fd_clock_func_t)( void const * args );
1083 :
1084 : FD_PROTOTYPES_BEGIN
1085 :
1086 : /* fd_memcpy(d,s,sz): On modern x86 in some circumstances, rep mov will
1087 : be faster than memcpy under the hood (basically due to RFO /
1088 : read-for-ownership optimizations in the cache protocol under the hood
1089 : that aren't easily done from the ISA ... see Intel docs on enhanced
1090 : rep mov). Compile time configurable though as this is not always
1091 : true. So application can tune to taste. Hard to beat rep mov for
1092 : code density though (2 bytes) and pretty hard to beat in situations
1093 : needing a completely generic memcpy. But it can be beaten in
1094 : specialized situations for the usual reasons. */
1095 :
1096 : /* FIXME: CONSIDER MEMCMP TOO! */
1097 : /* FIXME: CONSIDER MEMCPY RELATED FUNC ATTRS */
1098 :
1099 : #ifndef FD_USE_ARCH_MEMCPY
1100 : #define FD_USE_ARCH_MEMCPY 0
1101 : #endif
1102 :
1103 : #if FD_HAS_X86 && FD_USE_ARCH_MEMCPY && !defined(CBMC) && !FD_HAS_DEEPASAN && !FD_HAS_MSAN
1104 :
1105 : static inline void *
1106 : fd_memcpy( void * FD_RESTRICT d,
1107 : void const * FD_RESTRICT s,
1108 170793978 : ulong sz ) {
1109 170793978 : void * p = d;
1110 170793978 : __asm__ __volatile__( "rep movsb" : "+D" (p), "+S" (s), "+c" (sz) :: "memory" );
1111 170793978 : return d;
1112 170793978 : }
1113 :
1114 : #elif FD_HAS_MSAN
1115 :
1116 : void * __msan_memcpy( void * dest, void const * src, ulong n );
1117 :
1118 : static inline void *
1119 : fd_memcpy( void * FD_RESTRICT d,
1120 : void const * FD_RESTRICT s,
1121 : ulong sz ) {
1122 : return __msan_memcpy( d, s, sz );
1123 : }
1124 :
1125 : #else
1126 :
1127 : static inline void *
1128 : fd_memcpy( void * FD_RESTRICT d,
1129 : void const * FD_RESTRICT s,
1130 343876275 : ulong sz ) {
1131 : #if defined(CBMC) || FD_HAS_ASAN
1132 : if( FD_UNLIKELY( !sz ) ) return d; /* Standard says sz 0 is UB, uncomment if target is insane and doesn't treat sz 0 as a nop */
1133 : #endif
1134 343876275 : return memcpy( d, s, sz );
1135 343876275 : }
1136 :
1137 : #endif
1138 :
1139 : /* fd_memset(d,c,sz): architecturally optimized memset. See fd_memcpy
1140 : for considerations. */
1141 :
1142 : /* FIXME: CONSIDER MEMSET RELATED FUNC ATTRS */
1143 :
1144 : #ifndef FD_USE_ARCH_MEMSET
1145 : #define FD_USE_ARCH_MEMSET 0
1146 : #endif
1147 :
1148 : #if FD_HAS_X86 && FD_USE_ARCH_MEMSET && !defined(CBMC) && !FD_HAS_DEEPASAN && !FD_HAS_MSAN
1149 :
1150 : static inline void *
1151 : fd_memset( void * d,
1152 : int c,
1153 31730744 : ulong sz ) {
1154 31730744 : void * p = d;
1155 31730744 : __asm__ __volatile__( "rep stosb" : "+D" (p), "+c" (sz) : "a" (c) : "memory" );
1156 31730744 : return d;
1157 31730744 : }
1158 :
1159 : #else
1160 :
1161 : static inline void *
1162 : fd_memset( void * d,
1163 : int c,
1164 62042624 : ulong sz ) {
1165 : # ifdef CBMC
1166 : if( FD_UNLIKELY( !sz ) ) return d; /* See fd_memcpy note */
1167 : # endif
1168 62042624 : return memset( d, c, sz );
1169 62042624 : }
1170 :
1171 : #endif
1172 :
1173 : /* Calling fd_memzero_explicit will fill the provided region with zeroes.
1174 : It is guaranteed to not be optimized away. */
1175 : FD_FN_UNUSED static inline void
1176 : fd_memzero_explicit( void * d,
1177 1046526 : ulong sz ) {
1178 : /* We don't want to depend on explicit_bzero or memset_s, so the simplest
1179 : way to ensure the memset is not optimized away is to use a compiler fence,
1180 : identical to how explicit_bzero is implemented.
1181 : https://elixir.bootlin.com/glibc/glibc-2.40/source/string/explicit_bzero.c#L33 */
1182 1046526 : memset( d, 0, sz );
1183 1046526 : __asm__ __volatile__( "" ::: "memory" );
1184 1046526 : }
1185 :
1186 : /* fd_memeq(s0,s1,sz): Compares two blocks of memory. Returns 1 if
1187 : equal or sz is zero and 0 otherwise. No memory accesses made if sz
1188 : is zero (pointers may be invalid). On x86, uses repe cmpsb which is
1189 : preferable to __builtin_memcmp in some cases. */
1190 :
1191 : #ifndef FD_USE_ARCH_MEMEQ
1192 : #define FD_USE_ARCH_MEMEQ 0
1193 : #endif
1194 :
1195 : #if FD_HAS_X86 && FD_USE_ARCH_MEMEQ && defined(__GCC_ASM_FLAG_OUTPUTS__) && __STDC_VERSION__>=199901L
1196 :
1197 : FD_FN_PURE static inline int
1198 : fd_memeq( void const * s0,
1199 : void const * s1,
1200 : ulong sz ) {
1201 : /* ZF flag is set and exported in two cases:
1202 : a) size is zero (via test)
1203 : b) buffer is equal (via repe cmpsb) */
1204 : int r;
1205 : __asm__( "test %3, %3;"
1206 : "repe cmpsb"
1207 : : "=@cce" (r), "+S" (s0), "+D" (s1), "+c" (sz)
1208 : : "m" (*(char const (*)[sz]) s0), "m" (*(char const (*)[sz]) s1)
1209 : : "cc" );
1210 : return r;
1211 : }
1212 :
1213 : #else
1214 :
1215 : FD_FN_PURE static inline int
1216 : fd_memeq( void const * s1,
1217 : void const * s2,
1218 46540403 : ulong sz ) {
1219 46540403 : return 0==memcmp( s1, s2, sz );
1220 46540403 : }
1221 :
1222 : #endif
1223 :
1224 : /* Returns 1 if all sz bytes starting at s are zero, 0 otherwise. */
1225 : FD_FN_PURE static inline int
1226 : fd_mem_iszero( uchar const * s,
1227 15 : ulong sz ) {
1228 15 : for( ulong i=0UL; i<sz; i++ ) {
1229 15 : if( s[i]!=0 ) return 0;
1230 15 : }
1231 0 : return 1;
1232 15 : }
1233 :
1234 : /* fd_hash(seed,buf,sz), fd_hash_memcpy(seed,d,s,sz): High quality
1235 : (full avalanche) high speed variable length buffer -> 64-bit hash
1236 : function (memcpy_hash is often as fast as plain memcpy). Based on
1237 : the xxhash-r39 (open source BSD licensed) implementation. In-place
1238 : and out-of-place variants provided (out-of-place variant assumes dst
1239 : and src do not overlap). Caller promises valid input arguments,
1240 : cannot fail given valid input arguments. sz==0 is fine. */
1241 :
1242 : FD_FN_PURE ulong
1243 : fd_hash( ulong seed,
1244 : void const * buf,
1245 : ulong sz );
1246 :
1247 : ulong
1248 : fd_hash_memcpy( ulong seed,
1249 : void * FD_RESTRICT d,
1250 : void const * FD_RESTRICT s,
1251 : ulong sz );
1252 :
1253 : #ifndef FD_TICKCOUNT_STYLE
1254 : #if FD_HAS_X86 /* Use RDTSC */
1255 : #define FD_TICKCOUNT_STYLE 1
1256 : #elif FD_HAS_ARM /* Use CNTVCT_EL0 */
1257 : #define FD_TICKCOUNT_STYLE 2
1258 : #else /* Use portable fallback */
1259 : #define FD_TICKCOUNT_STYLE 0
1260 : #endif
1261 : #endif
1262 :
1263 : #if FD_TICKCOUNT_STYLE==0 /* Portable fallback (slow). Ticks at 1 ns / tick */
1264 :
1265 : #define fd_tickcount() fd_log_wallclock() /* TODO: fix ugly pre-log usage */
1266 :
1267 : #elif FD_TICKCOUNT_STYLE==1 /* RDTSC (fast) */
1268 :
1269 : /* fd_tickcount: Reads the hardware invariant tickcounter ("RDTSC").
1270 : This monotonically increases at an approximately constant rate
1271 : relative to the system wallclock and is synchronous across all CPUs
1272 : on a host.
1273 :
1274 : The rate this ticks at is not precisely defined (see Intel docs for
1275 : more details) but it is typically in the ballpark of the CPU base
1276 : clock frequency. The relationship to the wallclock is very well
1277 : approximated as linear over short periods of time (i.e. less than a
1278 : fraction of a second) and this should not exhibit any sudden changes
1279 : in its rate relative to the wallclock. Notably, its rate is not
1280 : directly impacted by CPU clock frequency adaptation / Turbo mode (see
1281 : other Intel performance monitoring counters for various CPU cycle
1282 : counters). It can drift over longer periods of time for the usual
1283 : clock synchronization reasons.
1284 :
1285 : This is a reasonably fast O(1) cost (~6-8 ns on recent Intel).
1286 : Because of all compiler options and parallel execution going on in
1287 : modern CPUs cores, other instructions might be reordered around this
1288 : by the compiler and/or CPU. It is up to the user to do lower level
1289 : tricks as necessary when the precise location of this in the
1290 : execution stream and/or when executed by the CPU is needed. (This is
1291 : often unnecessary as such levels of precision are not frequently
1292 : required and often have self-defeating overheads.)
1293 :
1294 : It is worth noting that RDTSC and/or (even more frequently) lower
1295 : level performance counters are often restricted from use in user
1296 : space applications. It is recommended that applications use this
1297 : primarily for debugging / performance tuning on unrestricted hosts
1298 : and/or when the developer is confident that applications using this
1299 : will have appropriate permissions when deployed. */
1300 :
1301 2884449420 : #define fd_tickcount() ((long)__builtin_ia32_rdtsc())
1302 :
1303 : #elif FD_TICKCOUNT_STYLE==2 /* armv8 (fast) */
1304 :
1305 : /* fd_tickcount (ARM): https://developer.arm.com/documentation/ddi0601/2021-12/AArch64-Registers/CNTVCT-EL0--Counter-timer-Virtual-Count-register
1306 : Approx 24 MHz on Apple M1. */
1307 :
1308 : static inline long
1309 : fd_tickcount( void ) {
1310 : /* consider using 'isb' */
1311 : ulong value;
1312 : __asm__ __volatile__ (
1313 : "mrs %0, cntvct_el0\n"
1314 : "nop"
1315 : : "=r" (value) );
1316 : return (long)value;
1317 : }
1318 :
1319 : #else
1320 : #error "Unknown FD_TICKCOUNT_STYLE"
1321 : #endif
1322 :
1323 : long _fd_tickcount( void const * _ ); /* fd_clock_func_t compat */
1324 :
1325 : #if FD_HAS_HOSTED
1326 :
1327 : /* fd_yield yields the calling thread to the operating system scheduler. */
1328 :
1329 : void
1330 : fd_yield( void );
1331 :
1332 : #endif
1333 :
1334 : #if FD_HAS_ARM
1335 :
1336 : /* fd_arm_stp16 stores two ulongs to a 16-byte memory location.
1337 : If LSE2 and p is aligned, is single-copy atomic. */
1338 :
1339 : static inline void
1340 : fd_arm_stp16( ulong * p,
1341 : ulong a,
1342 : ulong b ) {
1343 : __asm__(
1344 : "stp %x[a], %x[b], [%[p]]"
1345 : :
1346 : : [a] "r"(a), [b] "r"(b), [p] "r"(p)
1347 : : "memory"
1348 : );
1349 : }
1350 :
1351 : /* fd_arm_ldp16 loads two ulongs from a 16-byte memory location.
1352 : If LSE2 and p is aligned, is single-copy atomic. */
1353 :
1354 : #define fd_arm_ldp16(p_,a_,b_) \
1355 : __asm__( \
1356 : "ldp %x[a], %x[b], [%[p]]" \
1357 : : [a] "=r"(a_), [b] "=r"(b_) \
1358 : : [p] "r"(p_) \
1359 : : "memory" \
1360 : )
1361 :
1362 : #endif /* FD_HAS_ARM */
1363 :
1364 : FD_PROTOTYPES_END
1365 :
1366 : #endif /* HEADER_fd_src_util_fd_util_base_h */
|