LCOV - code coverage report
Current view: top level - flamenco/accdb - fd_accdb_private.h (source / functions) Hit Total Coverage
Test: cov.lcov Lines: 80 82 97.6 %
Date: 2026-09-17 04:28:31 Functions: 19 80 23.8 %

          Line data    Source code
       1             : #ifndef HEADER_fd_src_flamenco_accdb_fd_accdb_private_h
       2             : #define HEADER_fd_src_flamenco_accdb_fd_accdb_private_h
       3             : 
       4             : #include "fd_accdb_base.h"
       5             : #include "fd_accdb_shmem.h"
       6             : #include "fd_accdb_cache.h"
       7             : 
       8             : static inline void
       9          90 : spin_lock_acquire( int * lock ) {
      10          90 : # if FD_HAS_THREADS
      11    42028825 :   for(;;) {
      12    42028825 :     if( FD_LIKELY( !FD_ATOMIC_CAS( lock, 0, 1 ) ) ) break;
      13    42028735 :     FD_SPIN_PAUSE();
      14    42028735 :   }
      15             : # else
      16             :   *lock = 1;
      17             : # endif
      18          90 :   FD_COMPILER_MFENCE();
      19          90 : }
      20             : 
      21             : static inline void
      22          90 : spin_lock_release( int * lock ) {
      23          90 :   FD_COMPILER_MFENCE();
      24          90 : # if FD_HAS_THREADS
      25          90 :   FD_VOLATILE( *lock ) = 0;
      26             : # else
      27             :   *lock = 0;
      28             : # endif
      29          90 : }
      30             : 
      31             : struct fd_accdb_txn {
      32             :   union {
      33             :     struct { uint next; } pool;
      34             :     struct { uint next; } fork;
      35             :   };
      36             : 
      37             :   uint acc_pool_idx;
      38             : };
      39             : 
      40             : typedef struct fd_accdb_txn fd_accdb_txn_t;
      41             : 
      42             : #define POOL_NAME       txn_pool
      43      457728 : #define POOL_ELE_T      fd_accdb_txn_t
      44        1899 : #define POOL_NEXT       pool.next
      45             : #define POOL_IDX_T      uint
      46     1054083 : #define POOL_IDX_WIDTH  32
      47             : #define POOL_IMPL_STYLE 0
      48             : #define POOL_LAZY       1
      49             : 
      50             : #include "../../util/tmpl/fd_pool_para.c"
      51             : 
      52             : #define SET_NAME       descends_set
      53             : #define SET_IMPL_STYLE 1
      54             : #include "../../util/tmpl/fd_set_dynamic.c"
      55             : 
      56             : struct fd_accdb_fork_shmem {
      57             :   uint generation;
      58             : 
      59             :   fd_accdb_fork_id_t parent_id;
      60             :   fd_accdb_fork_id_t child_id;
      61             :   fd_accdb_fork_id_t sibling_id;
      62             : 
      63             :   struct {
      64             :     ulong next;
      65             :   } pool;
      66             : 
      67             :   uint txn_head;
      68             : };
      69             : 
      70             : typedef struct fd_accdb_fork_shmem fd_accdb_fork_shmem_t;
      71             : 
      72             : #define POOL_NAME       fork_pool
      73      101076 : #define POOL_ELE_T      fd_accdb_fork_shmem_t
      74       79617 : #define POOL_NEXT       pool.next
      75             : #define POOL_IDX_T      ulong
      76             : #define POOL_IMPL_STYLE 0
      77             : 
      78             : #include "../../util/tmpl/fd_pool_para.c"
      79             : 
      80             : struct fd_accdb_partition {
      81             :   ulong marked_compaction;
      82             :   ulong write_offset;
      83             :   ulong compaction_offset;
      84             : 
      85             :   ulong bytes_freed;
      86             : 
      87             :   uchar layer; /* compaction tier this partition belongs to. */
      88             : 
      89             :   ulong read_ops;
      90             :   ulong bytes_read;
      91             :   ulong write_ops;
      92             :   ulong bytes_written;
      93             : 
      94             :   /* Tickcount (fd_tickcount) of the partition's lifecycle events.  Set
      95             :      only at partition creation and again when the partition closes
      96             :      (i.e. the layer's write head rotates off it). */
      97             :   long created_ticks;
      98             :   long filled_ticks;
      99             : 
     100             :   /* Compaction lifecycle flags.  queued is set when the partition is
     101             :      pushed onto the compaction_dlist and cleared when it is popped.
     102             :      compacting_now is set by the compaction tile around the actual
     103             :      compaction work for this partition. */
     104             :   uchar queued;
     105             :   uchar compacting_now;
     106             : 
     107             :   /* Per-compaction-pass telemetry accumulators for the
     108             :      accdb_compaction_completed event.  Reset when a partition's
     109             :      compaction begins (compacting_now set in background_compact) and
     110             :      accumulated as records are scanned/relocated. */
     111             :   long  compaction_start_wallclock;    /* timestamp when this pass began */
     112             :   ulong compaction_accounts_relocated; /* live records moved this pass */
     113             :   ulong compaction_bytes_relocated;    /* bytes moved this pass */
     114             :   ulong compaction_dead_records;       /* records skipped (no live index entry) this pass */
     115             : 
     116             :   /* Epoch at which this partition was enqueued for compaction.  Set by
     117             :      fd_accdb_shmem_bytes_freed when the partition crosses the
     118             :      freed-bytes threshold.  The compaction tile will not begin reading
     119             :      from this partition until all joiners that were in an
     120             :      epoch-protected critical section at enqueue time have exited,
     121             :      ensuring any in-flight pwritev2 to this partition has completed. */
     122             :   ulong compaction_ready_epoch;
     123             : 
     124             :   /* Epoch at which this partition was enqueued for deferred freeing.
     125             :      Set by compaction when the partition finishes compaction, and
     126             :      checked by the reclamation scan to determine when it is safe to
     127             :      release the partition back to the pool. */
     128             :   ulong epoch_tag;
     129             : 
     130             :   ulong pool_next;
     131             : 
     132             :   ulong dlist_prev;
     133             :   ulong dlist_next;
     134             : };
     135             : 
     136             : typedef struct fd_accdb_partition fd_accdb_partition_t;
     137             : 
     138             : #define POOL_NAME       partition_pool
     139             : #define POOL_T          fd_accdb_partition_t
     140          78 : #define POOL_NEXT       pool_next
     141             : #define POOL_IDX_T      ulong
     142             : #define POOL_IMPL_STYLE 1
     143             : 
     144             : #include "../../util/tmpl/fd_pool.c"
     145             : 
     146             : #define DLIST_NAME       compaction_dlist
     147             : #define DLIST_ELE_T      fd_accdb_partition_t
     148          33 : #define DLIST_PREV       dlist_prev
     149          36 : #define DLIST_NEXT       dlist_next
     150             : #define DLIST_IMPL_STYLE 1
     151             : 
     152             : #include "../../util/tmpl/fd_dlist.c"
     153             : 
     154             : /* deferred_free_dlist reuses the same prev/next fields as
     155             :    compaction_dlist.  A partition is in at most one of the two lists at
     156             :    any time: it is popped from compaction_dlist before being pushed onto
     157             :    deferred_free_dlist. */
     158             : 
     159             : #define DLIST_NAME       deferred_free_dlist
     160             : #define DLIST_ELE_T      fd_accdb_partition_t
     161           3 : #define DLIST_PREV       dlist_prev
     162           6 : #define DLIST_NEXT       dlist_next
     163             : #define DLIST_IMPL_STYLE 1
     164             : 
     165             : #include "../../util/tmpl/fd_dlist.c"
     166             : 
     167             : struct fd_accdb_cache_key {
     168             :   uchar pubkey[ 32UL ];
     169             :   uint generation;
     170             : };
     171             : 
     172             : typedef struct fd_accdb_cache_key fd_accdb_cache_key_t;
     173             : 
     174             : struct __attribute__((aligned(64))) fd_accdb_accmeta {
     175             :   fd_accdb_cache_key_t key;
     176             : 
     177             :   struct {
     178             :     uint next;
     179             :   } map;
     180             : 
     181             :   union {
     182             :     struct {
     183             :       uint next;
     184             :     } pool;
     185             :     uint cache_idx;
     186             :   };
     187             : 
     188             :   uint   executable_size;
     189             : 
     190             :   ulong  lamports;
     191             : 
     192             :   /* Pack offset and fork_id together into a single ulong to pack the
     193             :      struct into a single 64 byte cache line.  This is a performance win
     194             :      of 2-3%. */
     195             :   ulong  offset_fork;
     196             : };
     197             : 
     198             : typedef struct fd_accdb_accmeta fd_accdb_accmeta_t;
     199             : 
     200             : FD_STATIC_ASSERT( alignof(fd_accdb_accmeta_t)==64, layout );
     201             : FD_STATIC_ASSERT( sizeof (fd_accdb_accmeta_t)==64, layout );
     202             : 
     203      496290 : #define FD_ACCDB_OFF_BITS  48UL
     204      363411 : #define FD_ACCDB_OFF_MASK  ((1UL<<FD_ACCDB_OFF_BITS)-1UL)       /* 0x0000_FFFF_FFFF_FFFF */
     205      232713 : #define FD_ACCDB_OFF_INVAL FD_ACCDB_OFF_MASK                    /* sentinel: offset bits all-ones */
     206             : 
     207             : /* The `size` field in fd_accdb_disk_meta_t (named executable_size in
     208             :    fd_accdb_accmeta_t) packs five things into 32 bits:
     209             : 
     210             :      bit  31     executable flag                       (FD_ACCDB_SIZE_EXEC_BIT)
     211             :      bit  30     cache_valid flag, in-memory only      (FD_ACCDB_SIZE_CACHE_VALID_BIT)
     212             :      bit  29     cache_claim flag, in-memory only      (FD_ACCDB_SIZE_CACHE_CLAIM_BIT)
     213             :      bit  28     pd_write flag,    in-memory only      (FD_ACCDB_SIZE_PD_WRITE_BIT)
     214             :      bits 27..0  data length in bytes                  (FD_ACCDB_SIZE_MASK)
     215             : 
     216             :    The data length is therefore 28 bits, max 256 MiB, still well above
     217             :    FD_RUNTIME_ACC_SZ_MAX of 10 MiB (enforced by the static assert below).
     218             : 
     219             :    The three upper flag bits exist only in the in-memory index, never on
     220             :    disk:
     221             :      - cache_valid (bit 30): when set, cache_idx holds a valid
     222             :        (class, idx) pair; when clear, cache_idx must not be dereferenced
     223             :        (it may hold a snapshot slot number or garbage).
     224             :      - cache_claim (bit 29): a short-lived eviction/install lock taken
     225             :        via CAS while a writer mutates the cache_idx <-> cache-line
     226             :        binding, so concurrent readers and the evictor do not race.
     227             :      - pd_write (bit 28): set on a committed version whose write changed
     228             :        BPF upgradeable-loader deploy status this slot (Deploy/Upgrade/
     229             :        Extend/Close).  Meaningful only while the accmeta's key.generation
     230             :        equals a live fork's generation (i.e. the version was committed on
     231             :        that fork this slot); across snapshot/root boundaries the bit is
     232             :        dead by construction because the generation no longer matches any
     233             :        live fork.  Carried explicitly by the two commit sites in
     234             :        fd_accdb_release and nowhere else.
     235             : 
     236             :    The on-disk representation (written via SIZE_PACK / SIZE_DATA) carries
     237             :    no in-memory flag: persisted bytes are unchanged, and compaction's
     238             :    copy_file_range preserves the record headers verbatim without
     239             :    rewriting them. */
     240             : 
     241      148512 : #define FD_ACCDB_SIZE_EXEC_BIT        (1U<<31)
     242        1515 : #define FD_ACCDB_SIZE_CACHE_VALID_BIT (1U<<30)
     243        5418 : #define FD_ACCDB_SIZE_CACHE_CLAIM_BIT (1U<<29)
     244        3684 : #define FD_ACCDB_SIZE_PD_WRITE_BIT    (1U<<28)
     245      309333 : #define FD_ACCDB_SIZE_MASK            ((1U<<28)-1U)
     246      115725 : #define FD_ACCDB_SIZE_PACK(sz,exec)   ((uint)(sz) | ((exec) ? FD_ACCDB_SIZE_EXEC_BIT : 0U))
     247      309333 : #define FD_ACCDB_SIZE_DATA(packed)    ((packed) & FD_ACCDB_SIZE_MASK)
     248       97542 : #define FD_ACCDB_SIZE_EXEC(packed)    (!!((packed) & FD_ACCDB_SIZE_EXEC_BIT))
     249        1485 : #define FD_ACCDB_SIZE_CACHE_VALID(p)  (!!((p) & FD_ACCDB_SIZE_CACHE_VALID_BIT))
     250          30 : #define FD_ACCDB_SIZE_CACHE_CLAIM(p)  (!!((p) & FD_ACCDB_SIZE_CACHE_CLAIM_BIT))
     251         210 : #define FD_ACCDB_SIZE_PD_WRITE(p)     (!!((p) & FD_ACCDB_SIZE_PD_WRITE_BIT))
     252             : 
     253             : FD_STATIC_ASSERT( (10UL<<20) < (1UL<<28), pd_write_bit_collides_with_len );
     254             : 
     255             : static inline ulong
     256        1119 : fd_accdb_acc_offset( fd_accdb_accmeta_t const * acc ) {
     257        1119 :   return acc->offset_fork & FD_ACCDB_OFF_MASK;
     258        1119 : }
     259             : 
     260             : static inline ushort
     261       20295 : fd_accdb_acc_fork_id( fd_accdb_accmeta_t const * acc ) {
     262       20295 :   return (ushort)( acc->offset_fork >> FD_ACCDB_OFF_BITS );
     263       20295 : }
     264             : 
     265             : static inline ulong
     266             : fd_accdb_acc_pack_offset_fork( ulong  offset,
     267      112584 :                                ushort fork_id ) {
     268      112584 :   return ( (ulong)fork_id << FD_ACCDB_OFF_BITS ) | ( offset & FD_ACCDB_OFF_MASK );
     269      112584 : }
     270             : 
     271             : /* fd_accdb_acc_xchg_offset atomically replaces the 48-bit offset
     272             :    portion of acc->offset_fork with new_offset while preserving the
     273             :    16-bit fork_id, and returns the previous 48-bit offset.  Uses a
     274             :    CAS loop so that concurrent compaction CAS and release-overwrite
     275             :    exchanges serialize correctly. */
     276             : 
     277             : static inline ulong
     278             : fd_accdb_acc_xchg_offset( fd_accdb_accmeta_t * acc,
     279        4992 :                            ulong               new_offset ) {
     280        4992 :   for(;;) {
     281        4992 :     ulong old_packed = FD_VOLATILE_CONST( acc->offset_fork );
     282        4992 :     ulong new_packed = ( old_packed & ~FD_ACCDB_OFF_MASK ) | ( new_offset & FD_ACCDB_OFF_MASK );
     283        4992 :     if( FD_LIKELY( FD_ATOMIC_CAS( &acc->offset_fork, old_packed, new_packed )==old_packed ) )
     284        4992 :       return old_packed & FD_ACCDB_OFF_MASK;
     285           0 :     FD_SPIN_PAUSE();
     286           0 :   }
     287        4992 : }
     288             : 
     289             : /* Packing helpers for the embedded acc cache index.  3 bits class
     290             :    in bits 31-29, 29 bits line index.  INVAL is the sentinel for
     291             :    "no cached location known".  FD_ACCDB_CACHE_LINE_MAX (defined in
     292             :    fd_accdb_cache.h) is the exclusive upper bound on representable
     293             :    line indices; per-class slot counts must not exceed it, or cidx
     294             :    values would alias. */
     295             : 
     296      214986 : #define FD_ACCDB_ACC_CIDX_IDX_MASK  ((uint)(FD_ACCDB_CACHE_LINE_MAX-1UL))
     297        3240 : #define FD_ACCDB_ACC_CIDX_INVAL     UINT_MAX
     298      116025 : #define FD_ACCDB_ACC_CIDX_PACK(c,i) ((uint)( ((uint)(c)<<FD_ACCDB_CACHE_LINE_BITS) | ((uint)(i) & FD_ACCDB_ACC_CIDX_IDX_MASK) ))
     299      100380 : #define FD_ACCDB_ACC_CIDX_CLASS(ci) ((ulong)((uint)(ci) >> FD_ACCDB_CACHE_LINE_BITS))
     300       98961 : #define FD_ACCDB_ACC_CIDX_IDX(ci)   ((ulong)((uint)(ci) & FD_ACCDB_ACC_CIDX_IDX_MASK))
     301             : 
     302             : #define POOL_NAME       acc_pool
     303      455274 : #define POOL_ELE_T      fd_accdb_accmeta_t
     304        1329 : #define POOL_NEXT       pool.next
     305             : #define POOL_IDX_T      uint
     306     1054962 : #define POOL_IDX_WIDTH  32
     307             : #define POOL_IMPL_STYLE 0
     308             : #define POOL_LAZY       1
     309             : 
     310             : #include "../../util/tmpl/fd_pool_para.c"
     311             : 
     312             : struct fd_accdb_cache_line {
     313             :   fd_accdb_cache_key_t key;
     314             : 
     315             :   uint acc_idx;
     316             :   uint cache_idx;
     317             : 
     318             :   uint  refcnt;
     319             :   uchar persisted;
     320             :   uchar referenced;
     321             : 
     322             :   uint next;
     323             : 
     324             :   uchar owner[ 32UL ];
     325             : };
     326             : 
     327             : typedef struct fd_accdb_cache_line fd_accdb_cache_line_t;
     328             : 
     329             : typedef struct __attribute__((aligned(64))) { ulong val; } accdb_offset_t;
     330             : 
     331             : /* Partition offsets are packed into accdb_offset_t as:
     332             :      bits 63..51: partition pool index
     333             :      bits 50..0 : byte offset within the partition */
     334             : 
     335       13335 : #define FD_ACCDB_PARTITION_OFF_BITS 51UL
     336             : 
     337             : static FD_FN_CONST inline accdb_offset_t
     338             : accdb_offset( ulong partition_idx,
     339       12570 :               ulong partition_offset ) {
     340       12570 :   return (accdb_offset_t){ .val = (partition_idx<<FD_ACCDB_PARTITION_OFF_BITS) | partition_offset };
     341       12570 : }
     342             : 
     343             : static FD_FN_PURE inline ulong
     344         387 : packed_partition_idx( accdb_offset_t const * offset ) {
     345         387 :   return offset->val>>FD_ACCDB_PARTITION_OFF_BITS;
     346         387 : }
     347             : 
     348             : static FD_FN_PURE inline ulong
     349         378 : packed_partition_offset( accdb_offset_t const * offset ) {
     350         378 :   return offset->val & ((1UL<<FD_ACCDB_PARTITION_OFF_BITS)-1UL);
     351         378 : }
     352             : 
     353             : static FD_FN_PURE inline ulong
     354             : packed_partition_file_offset( accdb_offset_t const * offset,
     355         102 :                               ulong                  partition_sz ) {
     356         102 :    return (packed_partition_idx( offset )*partition_sz + packed_partition_offset( offset ));
     357         102 : }
     358             : 
     359             : /* Maximum number of concurrent joiners (tiles) that can publish an
     360             :    epoch in the accdb.  Each joiner claims a slot in the shared epoch
     361             :    array during fd_accdb_new.  Must be less than or equal to 256 so
     362             :    refcnt in cache lines can safely track the number of threads
     363             :    referencing each cache line without overflow.  With a uint refcnt
     364             :    field, 256 joiners is well within range. */
     365     1070919 : #define FD_ACCDB_MAX_JOINERS (256UL)
     366             : 
     367             : /* EVICT_SENTINEL: stored in refcnt to indicate a cache line is being
     368             :    claimed by an eviction scan.  Any thread seeing this value must treat
     369             :    the line as unavailable. */
     370             : #define FD_ACCDB_EVICT_SENTINEL UINT_MAX
     371             : 
     372             : struct fd_accdb_shmem_private {
     373             :   int partition_lock  __attribute__((aligned(64)));
     374             : 
     375             :   /* Set non-zero by the snapin tile while a snapshot is being loaded.
     376             :      Suppresses compaction enqueue so the compaction tile does not race
     377             :      with bulk snapshot writes; fd_accdb_snapshot_load_end performs a
     378             :      one-shot sweep to enqueue any partitions that crossed the
     379             :      fragmentation threshold during the load. */
     380             :   int snapshot_loading;
     381             : 
     382             :   /* Set at construction (fd_accdb_shmem_new) when this validator
     383             :      supports bundles.  A bundle coalesces up to
     384             :      FD_ACCDB_MAX_TXN_PER_ACQUIRE transactions into one acquire, so when
     385             :      set, fd_accdb_acquire_inner permits pubkeys_cnt up to
     386             :      FD_ACCDB_MAX_ACQUIRE_CNT instead of the single-transaction limit
     387             :      FD_ACCDB_MAX_TX_ACCOUNT_LOCKS. */
     388             :   int bundle_enabled;
     389             : 
     390             :   /* Per-class CLOCK sweep position.  Atomically incremented by
     391             :      eviction scans (modulo cache_class_max[c]).  Each element is on
     392             :      its own cacheline to avoid false sharing between classes. */
     393             :   struct __attribute__((aligned(64))) { ulong val; } clock_hand[ FD_ACCDB_CACHE_CLASS_CNT ];
     394             : 
     395             :   /* Per-class CAS free list (Treiber stack) for fully-freed cache
     396             :      lines.  ver_top packs a 32-bit ABA version counter in bits
     397             :      63..32 and a uint pool index in bits 31..0.  UINT_MAX in the
     398             :      low 32 bits means empty. */
     399             :   struct __attribute__((aligned(64))) { ulong ver_top; } cache_free[ FD_ACCDB_CACHE_CLASS_CNT ];
     400             : 
     401             :   /* Per-class approximate depth of the CAS free list.  Atomically
     402             :      incremented on push, decremented on pop.  Used by the
     403             :      background pre-eviction loop to decide when to refill. */
     404             :   struct __attribute__((aligned(64))) { ulong val; } cache_free_cnt[ FD_ACCDB_CACHE_CLASS_CNT ];
     405             : 
     406             :   fd_accdb_fork_id_t root_fork_id;
     407             : 
     408             :   ulong seed;
     409             : 
     410             :   /* generation is a monotonically increasing counter assigned to each
     411             :      fork on creation.  When a fork is rooted, its pool slot (fork_id)
     412             :      is freed and may be recycled by a new fork, making fork_id in
     413             :      on-disk metadata useless for identifying entries from that freed
     414             :      fork.  But generation persists in disk metadata and is never
     415             :      recycled.
     416             : 
     417             :      Any rooted fork is by definition an ancestor of all live forks, so
     418             :      entries with generation <= root_fork->generation are
     419             :      unconditionally visible without consulting descends_set.  For
     420             :      entries with generation > root_fork->generation, the fork_id is
     421             :      still valid and descends_set is used to check ancestry.
     422             : 
     423             :      KEY INVARIANT: descends_set is ONLY consulted when generation >
     424             :      root_generation, which means the fork_id has NOT been rooted yet
     425             :      and its pool slot is still live.  This is what makes it safe for
     426             :      fork_slot_defer to eagerly clear descends_set bits for retired
     427             :      forks: rooted fork bits are dead (bypassed by the generation fast
     428             :      path), and purged fork bits were already 0 in all live forks'
     429             :      descends_sets (a purged fork is never an ancestor of a live fork).
     430             :      */
     431             :   uint generation;
     432             : 
     433             :   /* Lazy initial-allocation counter per size class.  Atomically
     434             :      incremented by acquire_cache_line (with undo on overflow). Each
     435             :      element is on its own cacheline to avoid false sharing between
     436             :      classes. */
     437             :   struct __attribute__((aligned(64))) { ulong val; } cache_class_init[ FD_ACCDB_CACHE_CLASS_CNT ];
     438             : 
     439             :   ulong cache_class_max[ FD_ACCDB_CACHE_CLASS_CNT ];
     440             : 
     441             :   /* Byte offsets from shmem base to the per-class cache regions. Each
     442             :      region is cache_class_max[c] * fd_accdb_cache_slot_sz[c] bytes,
     443             :      with each slot holding an fd_accdb_cache_line_t header followed by
     444             :      up to (fd_accdb_cache_slot_sz[c] - META_SZ) bytes of account data.
     445             :      */
     446             :   ulong cache_region_off[ FD_ACCDB_CACHE_CLASS_CNT ];
     447             : 
     448             :   /* Background pre-eviction watermarks (computed once in shmem_new).
     449             :      cache_free_target[c]: desired free-list depth for class c.
     450             :      cache_free_low_water[c]: trigger threshold ((target*3)/4). */
     451             :   ulong cache_free_target   [ FD_ACCDB_CACHE_CLASS_CNT ];
     452             :   ulong cache_free_low_water[ FD_ACCDB_CACHE_CLASS_CNT ];
     453             : 
     454             :   /* cache_class_used[i].val holds the number of reserved cache
     455             :      slots in size class i.  Acquire atomically increments; if the
     456             :      result exceeds cache_class_max[i] the reservation overflowed
     457             :      and the thread subtracts back and retries.  Release atomically
     458             :      decrements.  Each element is on its own cacheline to avoid
     459             :      false sharing between classes.  Invariant:
     460             :        used[i].val + available[i] == cache_class_max[i]
     461             :      at all times. */
     462             :   struct __attribute__((aligned(64))) { ulong val; } cache_class_used[ FD_ACCDB_CACHE_CLASS_CNT ];
     463             : 
     464             :   /* Per-layer write heads.  whead[0] is the hot (execution) write
     465             :      head, updated with atomic fetch-and-add by acquire/release
     466             :      threads.  whead[1..N-1] are compaction write heads, each
     467             :      single-writer (compaction tile only). */
     468             :   accdb_offset_t whead[ FD_ACCDB_COMPACTION_LAYER_CNT ];
     469             :   int            has_partition[ FD_ACCDB_COMPACTION_LAYER_CNT ];
     470             : 
     471             :   ulong partition_cnt;
     472             :   ulong partition_sz;
     473             :   ulong partition_max;
     474             : 
     475             :   ulong chain_cnt;
     476             :   ulong max_live_slots;
     477             :   ulong max_accounts;
     478             :   ulong max_account_writes_per_slot;
     479             : 
     480             :   /* Hard upper bound on concurrent joiners, set at construction.
     481             :      Used to determine whether cache_class_used tracking can be
     482             :      skipped for a given class (when max[c] >= MIN_RESERVED *
     483             :      joiner_cnt, every reservation succeeds trivially). */
     484             :   ulong joiner_cnt_max;
     485             : 
     486             :   /* Per-joiner worst-case cache reservation, set at construction.
     487             :      Used by fd_accdb_reset to replicate the cache_class_used sentinel
     488             :      logic from fd_accdb_shmem_new. */
     489             :   ulong cache_min_reserved;
     490             : 
     491             :   ulong partition_pool_off;
     492             : 
     493             :   /* compaction_dlist_off[k] is the byte offset (from shmem base) of
     494             :      the dlist sentinel for layer k.  Partitions at layer k that
     495             :      reach the freed-bytes threshold are enqueued here for compaction
     496             :      into layer k+1, or into layer k itself for the deepest layer. */
     497             :   ulong compaction_dlist_off[ FD_ACCDB_COMPACTION_LAYER_CNT ];
     498             : 
     499             :   /* Epoch-based safe reclamation for compacted partitions.
     500             : 
     501             :      epoch is a monotonically increasing counter incremented by the
     502             :      compaction tile each time a partition finishes compaction.  The
     503             :      completed partition is tagged with the current epoch and pushed
     504             :      onto a deferred-free list instead of being released immediately.
     505             : 
     506             :      joiner_epochs[i] holds the epoch observed by joiner i at the start
     507             :      of its epoch-protected critical section, or ULONG_MAX when idle.
     508             :      The compaction tile scans this array to find the minimum observed
     509             :      epoch; any deferred partition tagged with an epoch strictly less
     510             :      than that minimum is safe to release, because every epoch-protected
     511             :      operation that could have snapshotted an offset into that partition
     512             :      has since exited its critical section.
     513             : 
     514             :      joiner_cnt is claimed via atomic fetch-and-add in fd_accdb_new
     515             :      and never decremented. */
     516             :   ulong epoch __attribute__((aligned(64)));
     517             : 
     518             :   /* Synchronization with snapshot producer to inhibit compaction
     519             :      Holds one of FD_ACCDB_SNAPSHOT_SYNC_* */
     520             :   ulong snapshot_sync __attribute__((aligned(64)));
     521             : 
     522             :   /* Each joiner epoch is padded to a full cache line to prevent
     523             :      false sharing between joiners writing to adjacent slots. */
     524             :   struct __attribute__((aligned(64))) { ulong val; } joiner_epochs[ FD_ACCDB_MAX_JOINERS ];
     525             :   ulong joiner_cnt __attribute__((aligned(64)));
     526             :   ulong deferred_free_dlist_off;
     527             : 
     528             :   fd_accdb_shmem_metrics_t shmetrics[1];
     529             : 
     530             :   /* Command slot for T1 -> T2 offloading of advance_root / purge.
     531             :      Padded to its own cache line to avoid false sharing with the
     532             :      hot epoch / joiner_epochs fields above. */
     533             : 
     534    56340101 : #define FD_ACCDB_CMD_IDLE            (0U)
     535        1002 : #define FD_ACCDB_CMD_ADVANCE_ROOT    (1U)
     536          42 : #define FD_ACCDB_CMD_PURGE           (2U)
     537           6 : #define FD_ACCDB_CMD_CLEAR_DEFERRED  (3U)
     538           6 : #define FD_ACCDB_CMD_DRAIN_DEFERRED  (4U)
     539             : 
     540             :   uint   cmd_op       __attribute__((aligned(64))); /* FD_ACCDB_CMD_* */
     541             :   ushort cmd_fork_id;                               /* argument       */
     542             : 
     543             :   /* T2-only side buffer for deferred acc unlinks.  Holds the indices of
     544             :      accs that have been CAS-unlinked from their map chains in the
     545             :      current advance_root or purge call but cannot yet have pool.next
     546             :      written: a concurrent cold_load_acc may stomp it via the cache_idx
     547             :      union alias.  After wait_for_epoch_drain the list is materialized
     548             :      into pool.next links and released via acc_pool_release_chain.
     549             : 
     550             :      T2 is the sole writer (advance_root and purge both run on T2 via
     551             :      the cmd offload above), so cnt is plain.  Capacity is
     552             :      txn_max = max_live_slots * max_account_writes_per_slot, the same
     553             :      bound used to size txn_pool.  Stored as a byte offset from shmem
     554             :      base. */
     555             : 
     556             :   ulong deferred_acc_buf_off;
     557             :   ulong deferred_acc_buf_cnt;
     558             :   ulong deferred_acc_buf_max;
     559             :   ulong deferred_acc_epoch;
     560             : 
     561             :   acc_pool_shmem_t  acc_pool [1];
     562             :   fork_pool_shmem_t fork_pool[1];
     563             :   txn_pool_shmem_t  txn_pool [1];
     564             : 
     565             :   /* Track accounts modified since full snapshot.
     566             : 
     567             :        ------------++++++++.......                       - pruned blocks
     568             :        ^           ^      ^      ^                       + active blocks
     569             :        full snap   root   head   incremental (future)    . future blocks
     570             : 
     571             :      The validator thus needs to track the addresses of all accounts
     572             :      that have changed since the last full snapshot.  accdb forks cannot
     573             :      be used for this as fork information at the full snapshot slot is
     574             :      discarded.
     575             : 
     576             :      accdb deltas track this missing information. */
     577             : 
     578             :   struct {
     579             :     ulong seed;
     580             :     ulong chain_off;
     581             :     uint  chain_cnt;  /* power of 2 */
     582             :     uint  chain_mask; /* chain_cnt-1, contiguous runs of one bits */
     583             :     ulong ele_off;
     584             :     ulong ele_max;
     585             :     ulong head; /* bump alloc head */
     586             :   } delta;
     587             : 
     588             :   ulong magic; /* ==FD_ACCDB_SHMEM_MAGIC */
     589             : };
     590             : 
     591             : #endif /* HEADER_fd_src_flamenco_accdb_fd_accdb_private_h */

Generated by: LCOV version 1.14