Line data Source code
1 : #ifndef HEADER_fd_src_flamenco_accdb_fd_accdb_private_h 2 : #define HEADER_fd_src_flamenco_accdb_fd_accdb_private_h 3 : 4 : #include "fd_accdb_base.h" 5 : #include "fd_accdb_shmem.h" 6 : #include "fd_accdb_cache.h" 7 : 8 : static inline void 9 90 : spin_lock_acquire( int * lock ) { 10 90 : # if FD_HAS_THREADS 11 42028825 : for(;;) { 12 42028825 : if( FD_LIKELY( !FD_ATOMIC_CAS( lock, 0, 1 ) ) ) break; 13 42028735 : FD_SPIN_PAUSE(); 14 42028735 : } 15 : # else 16 : *lock = 1; 17 : # endif 18 90 : FD_COMPILER_MFENCE(); 19 90 : } 20 : 21 : static inline void 22 90 : spin_lock_release( int * lock ) { 23 90 : FD_COMPILER_MFENCE(); 24 90 : # if FD_HAS_THREADS 25 90 : FD_VOLATILE( *lock ) = 0; 26 : # else 27 : *lock = 0; 28 : # endif 29 90 : } 30 : 31 : struct fd_accdb_txn { 32 : union { 33 : struct { uint next; } pool; 34 : struct { uint next; } fork; 35 : }; 36 : 37 : uint acc_pool_idx; 38 : }; 39 : 40 : typedef struct fd_accdb_txn fd_accdb_txn_t; 41 : 42 : #define POOL_NAME txn_pool 43 457728 : #define POOL_ELE_T fd_accdb_txn_t 44 1899 : #define POOL_NEXT pool.next 45 : #define POOL_IDX_T uint 46 1054083 : #define POOL_IDX_WIDTH 32 47 : #define POOL_IMPL_STYLE 0 48 : #define POOL_LAZY 1 49 : 50 : #include "../../util/tmpl/fd_pool_para.c" 51 : 52 : #define SET_NAME descends_set 53 : #define SET_IMPL_STYLE 1 54 : #include "../../util/tmpl/fd_set_dynamic.c" 55 : 56 : struct fd_accdb_fork_shmem { 57 : uint generation; 58 : 59 : fd_accdb_fork_id_t parent_id; 60 : fd_accdb_fork_id_t child_id; 61 : fd_accdb_fork_id_t sibling_id; 62 : 63 : struct { 64 : ulong next; 65 : } pool; 66 : 67 : uint txn_head; 68 : }; 69 : 70 : typedef struct fd_accdb_fork_shmem fd_accdb_fork_shmem_t; 71 : 72 : #define POOL_NAME fork_pool 73 101076 : #define POOL_ELE_T fd_accdb_fork_shmem_t 74 79617 : #define POOL_NEXT pool.next 75 : #define POOL_IDX_T ulong 76 : #define POOL_IMPL_STYLE 0 77 : 78 : #include "../../util/tmpl/fd_pool_para.c" 79 : 80 : struct fd_accdb_partition { 81 : ulong marked_compaction; 82 : ulong write_offset; 83 : ulong compaction_offset; 84 : 85 : ulong bytes_freed; 86 : 87 : uchar layer; /* compaction tier this partition belongs to. */ 88 : 89 : ulong read_ops; 90 : ulong bytes_read; 91 : ulong write_ops; 92 : ulong bytes_written; 93 : 94 : /* Tickcount (fd_tickcount) of the partition's lifecycle events. Set 95 : only at partition creation and again when the partition closes 96 : (i.e. the layer's write head rotates off it). */ 97 : long created_ticks; 98 : long filled_ticks; 99 : 100 : /* Compaction lifecycle flags. queued is set when the partition is 101 : pushed onto the compaction_dlist and cleared when it is popped. 102 : compacting_now is set by the compaction tile around the actual 103 : compaction work for this partition. */ 104 : uchar queued; 105 : uchar compacting_now; 106 : 107 : /* Per-compaction-pass telemetry accumulators for the 108 : accdb_compaction_completed event. Reset when a partition's 109 : compaction begins (compacting_now set in background_compact) and 110 : accumulated as records are scanned/relocated. */ 111 : long compaction_start_wallclock; /* timestamp when this pass began */ 112 : ulong compaction_accounts_relocated; /* live records moved this pass */ 113 : ulong compaction_bytes_relocated; /* bytes moved this pass */ 114 : ulong compaction_dead_records; /* records skipped (no live index entry) this pass */ 115 : 116 : /* Epoch at which this partition was enqueued for compaction. Set by 117 : fd_accdb_shmem_bytes_freed when the partition crosses the 118 : freed-bytes threshold. The compaction tile will not begin reading 119 : from this partition until all joiners that were in an 120 : epoch-protected critical section at enqueue time have exited, 121 : ensuring any in-flight pwritev2 to this partition has completed. */ 122 : ulong compaction_ready_epoch; 123 : 124 : /* Epoch at which this partition was enqueued for deferred freeing. 125 : Set by compaction when the partition finishes compaction, and 126 : checked by the reclamation scan to determine when it is safe to 127 : release the partition back to the pool. */ 128 : ulong epoch_tag; 129 : 130 : ulong pool_next; 131 : 132 : ulong dlist_prev; 133 : ulong dlist_next; 134 : }; 135 : 136 : typedef struct fd_accdb_partition fd_accdb_partition_t; 137 : 138 : #define POOL_NAME partition_pool 139 : #define POOL_T fd_accdb_partition_t 140 78 : #define POOL_NEXT pool_next 141 : #define POOL_IDX_T ulong 142 : #define POOL_IMPL_STYLE 1 143 : 144 : #include "../../util/tmpl/fd_pool.c" 145 : 146 : #define DLIST_NAME compaction_dlist 147 : #define DLIST_ELE_T fd_accdb_partition_t 148 33 : #define DLIST_PREV dlist_prev 149 36 : #define DLIST_NEXT dlist_next 150 : #define DLIST_IMPL_STYLE 1 151 : 152 : #include "../../util/tmpl/fd_dlist.c" 153 : 154 : /* deferred_free_dlist reuses the same prev/next fields as 155 : compaction_dlist. A partition is in at most one of the two lists at 156 : any time: it is popped from compaction_dlist before being pushed onto 157 : deferred_free_dlist. */ 158 : 159 : #define DLIST_NAME deferred_free_dlist 160 : #define DLIST_ELE_T fd_accdb_partition_t 161 3 : #define DLIST_PREV dlist_prev 162 6 : #define DLIST_NEXT dlist_next 163 : #define DLIST_IMPL_STYLE 1 164 : 165 : #include "../../util/tmpl/fd_dlist.c" 166 : 167 : struct fd_accdb_cache_key { 168 : uchar pubkey[ 32UL ]; 169 : uint generation; 170 : }; 171 : 172 : typedef struct fd_accdb_cache_key fd_accdb_cache_key_t; 173 : 174 : struct __attribute__((aligned(64))) fd_accdb_accmeta { 175 : fd_accdb_cache_key_t key; 176 : 177 : struct { 178 : uint next; 179 : } map; 180 : 181 : union { 182 : struct { 183 : uint next; 184 : } pool; 185 : uint cache_idx; 186 : }; 187 : 188 : uint executable_size; 189 : 190 : ulong lamports; 191 : 192 : /* Pack offset and fork_id together into a single ulong to pack the 193 : struct into a single 64 byte cache line. This is a performance win 194 : of 2-3%. */ 195 : ulong offset_fork; 196 : }; 197 : 198 : typedef struct fd_accdb_accmeta fd_accdb_accmeta_t; 199 : 200 : FD_STATIC_ASSERT( alignof(fd_accdb_accmeta_t)==64, layout ); 201 : FD_STATIC_ASSERT( sizeof (fd_accdb_accmeta_t)==64, layout ); 202 : 203 496290 : #define FD_ACCDB_OFF_BITS 48UL 204 363411 : #define FD_ACCDB_OFF_MASK ((1UL<<FD_ACCDB_OFF_BITS)-1UL) /* 0x0000_FFFF_FFFF_FFFF */ 205 232713 : #define FD_ACCDB_OFF_INVAL FD_ACCDB_OFF_MASK /* sentinel: offset bits all-ones */ 206 : 207 : /* The `size` field in fd_accdb_disk_meta_t (named executable_size in 208 : fd_accdb_accmeta_t) packs five things into 32 bits: 209 : 210 : bit 31 executable flag (FD_ACCDB_SIZE_EXEC_BIT) 211 : bit 30 cache_valid flag, in-memory only (FD_ACCDB_SIZE_CACHE_VALID_BIT) 212 : bit 29 cache_claim flag, in-memory only (FD_ACCDB_SIZE_CACHE_CLAIM_BIT) 213 : bit 28 pd_write flag, in-memory only (FD_ACCDB_SIZE_PD_WRITE_BIT) 214 : bits 27..0 data length in bytes (FD_ACCDB_SIZE_MASK) 215 : 216 : The data length is therefore 28 bits, max 256 MiB, still well above 217 : FD_RUNTIME_ACC_SZ_MAX of 10 MiB (enforced by the static assert below). 218 : 219 : The three upper flag bits exist only in the in-memory index, never on 220 : disk: 221 : - cache_valid (bit 30): when set, cache_idx holds a valid 222 : (class, idx) pair; when clear, cache_idx must not be dereferenced 223 : (it may hold a snapshot slot number or garbage). 224 : - cache_claim (bit 29): a short-lived eviction/install lock taken 225 : via CAS while a writer mutates the cache_idx <-> cache-line 226 : binding, so concurrent readers and the evictor do not race. 227 : - pd_write (bit 28): set on a committed version whose write changed 228 : BPF upgradeable-loader deploy status this slot (Deploy/Upgrade/ 229 : Extend/Close). Meaningful only while the accmeta's key.generation 230 : equals a live fork's generation (i.e. the version was committed on 231 : that fork this slot); across snapshot/root boundaries the bit is 232 : dead by construction because the generation no longer matches any 233 : live fork. Carried explicitly by the two commit sites in 234 : fd_accdb_release and nowhere else. 235 : 236 : The on-disk representation (written via SIZE_PACK / SIZE_DATA) carries 237 : no in-memory flag: persisted bytes are unchanged, and compaction's 238 : copy_file_range preserves the record headers verbatim without 239 : rewriting them. */ 240 : 241 148512 : #define FD_ACCDB_SIZE_EXEC_BIT (1U<<31) 242 1515 : #define FD_ACCDB_SIZE_CACHE_VALID_BIT (1U<<30) 243 5418 : #define FD_ACCDB_SIZE_CACHE_CLAIM_BIT (1U<<29) 244 3684 : #define FD_ACCDB_SIZE_PD_WRITE_BIT (1U<<28) 245 309333 : #define FD_ACCDB_SIZE_MASK ((1U<<28)-1U) 246 115725 : #define FD_ACCDB_SIZE_PACK(sz,exec) ((uint)(sz) | ((exec) ? FD_ACCDB_SIZE_EXEC_BIT : 0U)) 247 309333 : #define FD_ACCDB_SIZE_DATA(packed) ((packed) & FD_ACCDB_SIZE_MASK) 248 97542 : #define FD_ACCDB_SIZE_EXEC(packed) (!!((packed) & FD_ACCDB_SIZE_EXEC_BIT)) 249 1485 : #define FD_ACCDB_SIZE_CACHE_VALID(p) (!!((p) & FD_ACCDB_SIZE_CACHE_VALID_BIT)) 250 30 : #define FD_ACCDB_SIZE_CACHE_CLAIM(p) (!!((p) & FD_ACCDB_SIZE_CACHE_CLAIM_BIT)) 251 210 : #define FD_ACCDB_SIZE_PD_WRITE(p) (!!((p) & FD_ACCDB_SIZE_PD_WRITE_BIT)) 252 : 253 : FD_STATIC_ASSERT( (10UL<<20) < (1UL<<28), pd_write_bit_collides_with_len ); 254 : 255 : static inline ulong 256 1119 : fd_accdb_acc_offset( fd_accdb_accmeta_t const * acc ) { 257 1119 : return acc->offset_fork & FD_ACCDB_OFF_MASK; 258 1119 : } 259 : 260 : static inline ushort 261 20295 : fd_accdb_acc_fork_id( fd_accdb_accmeta_t const * acc ) { 262 20295 : return (ushort)( acc->offset_fork >> FD_ACCDB_OFF_BITS ); 263 20295 : } 264 : 265 : static inline ulong 266 : fd_accdb_acc_pack_offset_fork( ulong offset, 267 112584 : ushort fork_id ) { 268 112584 : return ( (ulong)fork_id << FD_ACCDB_OFF_BITS ) | ( offset & FD_ACCDB_OFF_MASK ); 269 112584 : } 270 : 271 : /* fd_accdb_acc_xchg_offset atomically replaces the 48-bit offset 272 : portion of acc->offset_fork with new_offset while preserving the 273 : 16-bit fork_id, and returns the previous 48-bit offset. Uses a 274 : CAS loop so that concurrent compaction CAS and release-overwrite 275 : exchanges serialize correctly. */ 276 : 277 : static inline ulong 278 : fd_accdb_acc_xchg_offset( fd_accdb_accmeta_t * acc, 279 4992 : ulong new_offset ) { 280 4992 : for(;;) { 281 4992 : ulong old_packed = FD_VOLATILE_CONST( acc->offset_fork ); 282 4992 : ulong new_packed = ( old_packed & ~FD_ACCDB_OFF_MASK ) | ( new_offset & FD_ACCDB_OFF_MASK ); 283 4992 : if( FD_LIKELY( FD_ATOMIC_CAS( &acc->offset_fork, old_packed, new_packed )==old_packed ) ) 284 4992 : return old_packed & FD_ACCDB_OFF_MASK; 285 0 : FD_SPIN_PAUSE(); 286 0 : } 287 4992 : } 288 : 289 : /* Packing helpers for the embedded acc cache index. 3 bits class 290 : in bits 31-29, 29 bits line index. INVAL is the sentinel for 291 : "no cached location known". FD_ACCDB_CACHE_LINE_MAX (defined in 292 : fd_accdb_cache.h) is the exclusive upper bound on representable 293 : line indices; per-class slot counts must not exceed it, or cidx 294 : values would alias. */ 295 : 296 214986 : #define FD_ACCDB_ACC_CIDX_IDX_MASK ((uint)(FD_ACCDB_CACHE_LINE_MAX-1UL)) 297 3240 : #define FD_ACCDB_ACC_CIDX_INVAL UINT_MAX 298 116025 : #define FD_ACCDB_ACC_CIDX_PACK(c,i) ((uint)( ((uint)(c)<<FD_ACCDB_CACHE_LINE_BITS) | ((uint)(i) & FD_ACCDB_ACC_CIDX_IDX_MASK) )) 299 100380 : #define FD_ACCDB_ACC_CIDX_CLASS(ci) ((ulong)((uint)(ci) >> FD_ACCDB_CACHE_LINE_BITS)) 300 98961 : #define FD_ACCDB_ACC_CIDX_IDX(ci) ((ulong)((uint)(ci) & FD_ACCDB_ACC_CIDX_IDX_MASK)) 301 : 302 : #define POOL_NAME acc_pool 303 455274 : #define POOL_ELE_T fd_accdb_accmeta_t 304 1329 : #define POOL_NEXT pool.next 305 : #define POOL_IDX_T uint 306 1054962 : #define POOL_IDX_WIDTH 32 307 : #define POOL_IMPL_STYLE 0 308 : #define POOL_LAZY 1 309 : 310 : #include "../../util/tmpl/fd_pool_para.c" 311 : 312 : struct fd_accdb_cache_line { 313 : fd_accdb_cache_key_t key; 314 : 315 : uint acc_idx; 316 : uint cache_idx; 317 : 318 : uint refcnt; 319 : uchar persisted; 320 : uchar referenced; 321 : 322 : uint next; 323 : 324 : uchar owner[ 32UL ]; 325 : }; 326 : 327 : typedef struct fd_accdb_cache_line fd_accdb_cache_line_t; 328 : 329 : typedef struct __attribute__((aligned(64))) { ulong val; } accdb_offset_t; 330 : 331 : /* Partition offsets are packed into accdb_offset_t as: 332 : bits 63..51: partition pool index 333 : bits 50..0 : byte offset within the partition */ 334 : 335 13335 : #define FD_ACCDB_PARTITION_OFF_BITS 51UL 336 : 337 : static FD_FN_CONST inline accdb_offset_t 338 : accdb_offset( ulong partition_idx, 339 12570 : ulong partition_offset ) { 340 12570 : return (accdb_offset_t){ .val = (partition_idx<<FD_ACCDB_PARTITION_OFF_BITS) | partition_offset }; 341 12570 : } 342 : 343 : static FD_FN_PURE inline ulong 344 387 : packed_partition_idx( accdb_offset_t const * offset ) { 345 387 : return offset->val>>FD_ACCDB_PARTITION_OFF_BITS; 346 387 : } 347 : 348 : static FD_FN_PURE inline ulong 349 378 : packed_partition_offset( accdb_offset_t const * offset ) { 350 378 : return offset->val & ((1UL<<FD_ACCDB_PARTITION_OFF_BITS)-1UL); 351 378 : } 352 : 353 : static FD_FN_PURE inline ulong 354 : packed_partition_file_offset( accdb_offset_t const * offset, 355 102 : ulong partition_sz ) { 356 102 : return (packed_partition_idx( offset )*partition_sz + packed_partition_offset( offset )); 357 102 : } 358 : 359 : /* Maximum number of concurrent joiners (tiles) that can publish an 360 : epoch in the accdb. Each joiner claims a slot in the shared epoch 361 : array during fd_accdb_new. Must be less than or equal to 256 so 362 : refcnt in cache lines can safely track the number of threads 363 : referencing each cache line without overflow. With a uint refcnt 364 : field, 256 joiners is well within range. */ 365 1070919 : #define FD_ACCDB_MAX_JOINERS (256UL) 366 : 367 : /* EVICT_SENTINEL: stored in refcnt to indicate a cache line is being 368 : claimed by an eviction scan. Any thread seeing this value must treat 369 : the line as unavailable. */ 370 : #define FD_ACCDB_EVICT_SENTINEL UINT_MAX 371 : 372 : struct fd_accdb_shmem_private { 373 : int partition_lock __attribute__((aligned(64))); 374 : 375 : /* Set non-zero by the snapin tile while a snapshot is being loaded. 376 : Suppresses compaction enqueue so the compaction tile does not race 377 : with bulk snapshot writes; fd_accdb_snapshot_load_end performs a 378 : one-shot sweep to enqueue any partitions that crossed the 379 : fragmentation threshold during the load. */ 380 : int snapshot_loading; 381 : 382 : /* Set at construction (fd_accdb_shmem_new) when this validator 383 : supports bundles. A bundle coalesces up to 384 : FD_ACCDB_MAX_TXN_PER_ACQUIRE transactions into one acquire, so when 385 : set, fd_accdb_acquire_inner permits pubkeys_cnt up to 386 : FD_ACCDB_MAX_ACQUIRE_CNT instead of the single-transaction limit 387 : FD_ACCDB_MAX_TX_ACCOUNT_LOCKS. */ 388 : int bundle_enabled; 389 : 390 : /* Per-class CLOCK sweep position. Atomically incremented by 391 : eviction scans (modulo cache_class_max[c]). Each element is on 392 : its own cacheline to avoid false sharing between classes. */ 393 : struct __attribute__((aligned(64))) { ulong val; } clock_hand[ FD_ACCDB_CACHE_CLASS_CNT ]; 394 : 395 : /* Per-class CAS free list (Treiber stack) for fully-freed cache 396 : lines. ver_top packs a 32-bit ABA version counter in bits 397 : 63..32 and a uint pool index in bits 31..0. UINT_MAX in the 398 : low 32 bits means empty. */ 399 : struct __attribute__((aligned(64))) { ulong ver_top; } cache_free[ FD_ACCDB_CACHE_CLASS_CNT ]; 400 : 401 : /* Per-class approximate depth of the CAS free list. Atomically 402 : incremented on push, decremented on pop. Used by the 403 : background pre-eviction loop to decide when to refill. */ 404 : struct __attribute__((aligned(64))) { ulong val; } cache_free_cnt[ FD_ACCDB_CACHE_CLASS_CNT ]; 405 : 406 : fd_accdb_fork_id_t root_fork_id; 407 : 408 : ulong seed; 409 : 410 : /* generation is a monotonically increasing counter assigned to each 411 : fork on creation. When a fork is rooted, its pool slot (fork_id) 412 : is freed and may be recycled by a new fork, making fork_id in 413 : on-disk metadata useless for identifying entries from that freed 414 : fork. But generation persists in disk metadata and is never 415 : recycled. 416 : 417 : Any rooted fork is by definition an ancestor of all live forks, so 418 : entries with generation <= root_fork->generation are 419 : unconditionally visible without consulting descends_set. For 420 : entries with generation > root_fork->generation, the fork_id is 421 : still valid and descends_set is used to check ancestry. 422 : 423 : KEY INVARIANT: descends_set is ONLY consulted when generation > 424 : root_generation, which means the fork_id has NOT been rooted yet 425 : and its pool slot is still live. This is what makes it safe for 426 : fork_slot_defer to eagerly clear descends_set bits for retired 427 : forks: rooted fork bits are dead (bypassed by the generation fast 428 : path), and purged fork bits were already 0 in all live forks' 429 : descends_sets (a purged fork is never an ancestor of a live fork). 430 : */ 431 : uint generation; 432 : 433 : /* Lazy initial-allocation counter per size class. Atomically 434 : incremented by acquire_cache_line (with undo on overflow). Each 435 : element is on its own cacheline to avoid false sharing between 436 : classes. */ 437 : struct __attribute__((aligned(64))) { ulong val; } cache_class_init[ FD_ACCDB_CACHE_CLASS_CNT ]; 438 : 439 : ulong cache_class_max[ FD_ACCDB_CACHE_CLASS_CNT ]; 440 : 441 : /* Byte offsets from shmem base to the per-class cache regions. Each 442 : region is cache_class_max[c] * fd_accdb_cache_slot_sz[c] bytes, 443 : with each slot holding an fd_accdb_cache_line_t header followed by 444 : up to (fd_accdb_cache_slot_sz[c] - META_SZ) bytes of account data. 445 : */ 446 : ulong cache_region_off[ FD_ACCDB_CACHE_CLASS_CNT ]; 447 : 448 : /* Background pre-eviction watermarks (computed once in shmem_new). 449 : cache_free_target[c]: desired free-list depth for class c. 450 : cache_free_low_water[c]: trigger threshold ((target*3)/4). */ 451 : ulong cache_free_target [ FD_ACCDB_CACHE_CLASS_CNT ]; 452 : ulong cache_free_low_water[ FD_ACCDB_CACHE_CLASS_CNT ]; 453 : 454 : /* cache_class_used[i].val holds the number of reserved cache 455 : slots in size class i. Acquire atomically increments; if the 456 : result exceeds cache_class_max[i] the reservation overflowed 457 : and the thread subtracts back and retries. Release atomically 458 : decrements. Each element is on its own cacheline to avoid 459 : false sharing between classes. Invariant: 460 : used[i].val + available[i] == cache_class_max[i] 461 : at all times. */ 462 : struct __attribute__((aligned(64))) { ulong val; } cache_class_used[ FD_ACCDB_CACHE_CLASS_CNT ]; 463 : 464 : /* Per-layer write heads. whead[0] is the hot (execution) write 465 : head, updated with atomic fetch-and-add by acquire/release 466 : threads. whead[1..N-1] are compaction write heads, each 467 : single-writer (compaction tile only). */ 468 : accdb_offset_t whead[ FD_ACCDB_COMPACTION_LAYER_CNT ]; 469 : int has_partition[ FD_ACCDB_COMPACTION_LAYER_CNT ]; 470 : 471 : ulong partition_cnt; 472 : ulong partition_sz; 473 : ulong partition_max; 474 : 475 : ulong chain_cnt; 476 : ulong max_live_slots; 477 : ulong max_accounts; 478 : ulong max_account_writes_per_slot; 479 : 480 : /* Hard upper bound on concurrent joiners, set at construction. 481 : Used to determine whether cache_class_used tracking can be 482 : skipped for a given class (when max[c] >= MIN_RESERVED * 483 : joiner_cnt, every reservation succeeds trivially). */ 484 : ulong joiner_cnt_max; 485 : 486 : /* Per-joiner worst-case cache reservation, set at construction. 487 : Used by fd_accdb_reset to replicate the cache_class_used sentinel 488 : logic from fd_accdb_shmem_new. */ 489 : ulong cache_min_reserved; 490 : 491 : ulong partition_pool_off; 492 : 493 : /* compaction_dlist_off[k] is the byte offset (from shmem base) of 494 : the dlist sentinel for layer k. Partitions at layer k that 495 : reach the freed-bytes threshold are enqueued here for compaction 496 : into layer k+1, or into layer k itself for the deepest layer. */ 497 : ulong compaction_dlist_off[ FD_ACCDB_COMPACTION_LAYER_CNT ]; 498 : 499 : /* Epoch-based safe reclamation for compacted partitions. 500 : 501 : epoch is a monotonically increasing counter incremented by the 502 : compaction tile each time a partition finishes compaction. The 503 : completed partition is tagged with the current epoch and pushed 504 : onto a deferred-free list instead of being released immediately. 505 : 506 : joiner_epochs[i] holds the epoch observed by joiner i at the start 507 : of its epoch-protected critical section, or ULONG_MAX when idle. 508 : The compaction tile scans this array to find the minimum observed 509 : epoch; any deferred partition tagged with an epoch strictly less 510 : than that minimum is safe to release, because every epoch-protected 511 : operation that could have snapshotted an offset into that partition 512 : has since exited its critical section. 513 : 514 : joiner_cnt is claimed via atomic fetch-and-add in fd_accdb_new 515 : and never decremented. */ 516 : ulong epoch __attribute__((aligned(64))); 517 : 518 : /* Synchronization with snapshot producer to inhibit compaction 519 : Holds one of FD_ACCDB_SNAPSHOT_SYNC_* */ 520 : ulong snapshot_sync __attribute__((aligned(64))); 521 : 522 : /* Each joiner epoch is padded to a full cache line to prevent 523 : false sharing between joiners writing to adjacent slots. */ 524 : struct __attribute__((aligned(64))) { ulong val; } joiner_epochs[ FD_ACCDB_MAX_JOINERS ]; 525 : ulong joiner_cnt __attribute__((aligned(64))); 526 : ulong deferred_free_dlist_off; 527 : 528 : fd_accdb_shmem_metrics_t shmetrics[1]; 529 : 530 : /* Command slot for T1 -> T2 offloading of advance_root / purge. 531 : Padded to its own cache line to avoid false sharing with the 532 : hot epoch / joiner_epochs fields above. */ 533 : 534 56340101 : #define FD_ACCDB_CMD_IDLE (0U) 535 1002 : #define FD_ACCDB_CMD_ADVANCE_ROOT (1U) 536 42 : #define FD_ACCDB_CMD_PURGE (2U) 537 6 : #define FD_ACCDB_CMD_CLEAR_DEFERRED (3U) 538 6 : #define FD_ACCDB_CMD_DRAIN_DEFERRED (4U) 539 : 540 : uint cmd_op __attribute__((aligned(64))); /* FD_ACCDB_CMD_* */ 541 : ushort cmd_fork_id; /* argument */ 542 : 543 : /* T2-only side buffer for deferred acc unlinks. Holds the indices of 544 : accs that have been CAS-unlinked from their map chains in the 545 : current advance_root or purge call but cannot yet have pool.next 546 : written: a concurrent cold_load_acc may stomp it via the cache_idx 547 : union alias. After wait_for_epoch_drain the list is materialized 548 : into pool.next links and released via acc_pool_release_chain. 549 : 550 : T2 is the sole writer (advance_root and purge both run on T2 via 551 : the cmd offload above), so cnt is plain. Capacity is 552 : txn_max = max_live_slots * max_account_writes_per_slot, the same 553 : bound used to size txn_pool. Stored as a byte offset from shmem 554 : base. */ 555 : 556 : ulong deferred_acc_buf_off; 557 : ulong deferred_acc_buf_cnt; 558 : ulong deferred_acc_buf_max; 559 : ulong deferred_acc_epoch; 560 : 561 : acc_pool_shmem_t acc_pool [1]; 562 : fork_pool_shmem_t fork_pool[1]; 563 : txn_pool_shmem_t txn_pool [1]; 564 : 565 : /* Track accounts modified since full snapshot. 566 : 567 : ------------++++++++....... - pruned blocks 568 : ^ ^ ^ ^ + active blocks 569 : full snap root head incremental (future) . future blocks 570 : 571 : The validator thus needs to track the addresses of all accounts 572 : that have changed since the last full snapshot. accdb forks cannot 573 : be used for this as fork information at the full snapshot slot is 574 : discarded. 575 : 576 : accdb deltas track this missing information. */ 577 : 578 : struct { 579 : ulong seed; 580 : ulong chain_off; 581 : uint chain_cnt; /* power of 2 */ 582 : uint chain_mask; /* chain_cnt-1, contiguous runs of one bits */ 583 : ulong ele_off; 584 : ulong ele_max; 585 : ulong head; /* bump alloc head */ 586 : } delta; 587 : 588 : ulong magic; /* ==FD_ACCDB_SHMEM_MAGIC */ 589 : }; 590 : 591 : #endif /* HEADER_fd_src_flamenco_accdb_fd_accdb_private_h */