LCOV - code coverage report
Current view: top level - flamenco/accdb - fd_accdb.c (source / functions) Hit Total Coverage
Test: cov.lcov Lines: 1780 2480 71.8 %
Date: 2026-09-17 04:28:31 Functions: 61 68 89.7 %

          Line data    Source code
       1             : #define _GNU_SOURCE
       2             : #include "fd_accdb.h"
       3             : #include "fd_accdb_shmem.h"
       4             : #include "fd_accdb_private.h"
       5             : #include "../../util/fd_hash32.h"
       6             : 
       7             : #if FD_TMPL_USE_HANDHOLDING
       8             : #include "../../ballet/txn/fd_txn.h"
       9             : #include "../../ballet/base58/fd_base58.h"
      10             : #endif
      11             : #include "../../util/racesan/fd_racesan_target.h"
      12             : 
      13             : #include "../../disco/events/generated/fd_event_gen.h"
      14             : 
      15             : FD_STATIC_ASSERT( sizeof(fd_accdb_cache_line_t)==FD_ACCDB_CACHE_META_SZ, cache_meta_sz );
      16             : 
      17             : #if FD_HAS_RACESAN
      18             : /* Test-only telemetry: background_compact publishes the pubkey + dest
      19             :    offset of the record it is about to relocation-CAS at the
      20             :    accdb_compact:pre_offset_cas hook, so test_accdb_racesan can PROVE the
      21             :    parked relocation is the account it set up (avoiding a vacuous test).
      22             :    Zero-cost / absent in production (racesan off). */
      23             : uchar fd_accdb_dbg_reloc_pubkey[ 32UL ];
      24             : ulong fd_accdb_dbg_reloc_dest;
      25             : ulong fd_accdb_dbg_reloc_cnt;
      26             : #endif
      27             : 
      28             : #include <stddef.h>
      29             : #include <unistd.h>
      30             : #include <fcntl.h>
      31             : #include <errno.h>
      32             : #include <sys/uio.h>
      33             : 
      34             : struct fd_accdb_fork {
      35             :   fd_accdb_fork_shmem_t * shmem;
      36             :   descends_set_t * descends;
      37             : };
      38             : 
      39             : typedef struct fd_accdb_fork fd_accdb_fork_t;
      40             : 
      41      315459 : #define FD_ACCDB_ACQUIRE_STATE_IDLE    (0)
      42         351 : #define FD_ACCDB_ACQUIRE_STATE_PHASE_A (1)
      43      311403 : #define FD_ACCDB_ACQUIRE_STATE_OPEN    (2)
      44             : 
      45             : struct __attribute__((aligned(FD_ACCDB_ALIGN))) fd_accdb_private {
      46             :   int fd;
      47             : 
      48             :   int acquire_state;
      49             : 
      50             :   fd_accdb_shmem_t * shmem;
      51             : 
      52             :   fd_accdb_fork_t * fork_pool;
      53             :   fork_pool_t fork_shmem_pool[1];
      54             : 
      55             :   fd_accdb_accmeta_t * acc_pool;
      56             :   acc_pool_t acc_pool_join[1];
      57             :   uint * acc_map;
      58             : 
      59             :   uchar * cache [ FD_ACCDB_CACHE_CLASS_CNT ];
      60             : 
      61             :   fd_accdb_partition_t * partition_pool;
      62             :   compaction_dlist_t * compaction_dlist[ FD_ACCDB_COMPACTION_LAYER_CNT ];
      63             :   deferred_free_dlist_t * deferred_free_dlist;
      64             : 
      65             :   txn_pool_t txn_pool[1];
      66             : 
      67             :   /* Pointer into shmem->joiner_epochs[ my_slot ].val for writer
      68             :      joiners, or into a private per-tile fseq for read-only joiners.
      69             :      Set to the current global epoch on entry to an epoch-protected
      70             :      operation, and ULONG_MAX on exit.  Used to determine when
      71             :      deferred frees are safe. */
      72             :   ulong * my_epoch_slot;
      73             : 
      74             :   /* Read-only pointers to external epoch slots (e.g. fseqs owned by
      75             :      RO consumer tiles like the rpc tile).  Scanned in addition to
      76             :      shmem->joiner_epochs[] by compaction's deferred-free
      77             :      reclamation.  Borrowed; the caller of fd_accdb_new owns the
      78             :      storage. */
      79             :   ulong const * const * external_epoch_slots;
      80             :   ulong                 external_epoch_cnt;
      81             : 
      82             :   /* Side buffer of acc pool indices that have been CAS-unlinked from
      83             :      their hash chains but cannot be released back to acc_pool yet,
      84             :      because concurrent readers (acquire / compact) may still be
      85             :      traversing the removed nodes via map.next.  The batch is released
      86             :      once all joiner_epochs exceed shmem->deferred_acc_epoch.  Indices
      87             :      are written here (not into pool.next) until after the epoch drain
      88             :      because pool.next is union-aliased to cache_idx, which a concurrent
      89             :      cold_load_acc may still write through a captured pointer.  Backed
      90             :      by shmem->deferred_acc_buf_off; cnt and epoch live in shmem too. */
      91             :   uint * deferred_acc_buf;
      92             : 
      93             :   /* Chain of fork pool slots whose IDs are still potentially
      94             :      referenced by concurrent readers (via descends_set_test or
      95             :      root_fork_id snapshot).  The chain is released back to fork_pool
      96             :      once all joiner_epochs exceed deferred_fork_epoch.  NULL head
      97             :      means no deferred forks. */
      98             :   fd_accdb_fork_shmem_t * deferred_fork_head;
      99             :   fd_accdb_fork_shmem_t * deferred_fork_tail;
     100             :   ulong                   deferred_fork_epoch;
     101             : 
     102             :   fd_accdb_metrics_t metrics[1];
     103             : 
     104             :   /* Set by fd_accdb_snapshot_load_begin/end.  When non-zero, layer-0
     105             :      partition handoffs (in change_partition) re-tier the partitions
     106             :      that fell out of the snapshot-load working set: P-2 to Warm and
     107             :      P-3 to Cold.  This backfills tiering for snapshot-loaded data
     108             :      that never gets a second write (and therefore would otherwise
     109             :      never be promoted by compaction). */
     110             :   int snapshot_loading;
     111             : 
     112             :   /* Track account addresses changed since a full snap.
     113             :      Used to determine which accounts should be packed into a full
     114             :      snapshot (including tombstones for accounts no longer present in
     115             :      accdb, but present in the full snapshot). */
     116             :   struct {
     117             :     uint *             chains;
     118             :     fd_accdb_delta_t * pool;
     119             :     struct {
     120             :       uchar const * pubkey;
     121             :       uint          chain;
     122             :     } scratch[ FD_ACCDB_MAX_ACQUIRE_CNT ];
     123             :   } delta;
     124             : 
     125             :   /* Write counters that are not published yet.  Metrics are aggregated
     126             :      in batches to avoid expensive atomic operations on each write.
     127             :      64-byte aligned to fit in a single cache line. */
     128             :   struct {
     129             :     ulong bytes;         /* bytes reserved on partition_idx */
     130             :     ulong num_ops;       /* reservations behind those bytes */
     131             :     ulong partition_idx; /* set while num_ops>0 */
     132             :   } write_stats __attribute__((aligned(64)));
     133             : };
     134             : 
     135             : static inline fd_accdb_cache_line_t *
     136             : cache_line( fd_accdb_t * accdb,
     137             :             ulong        cls,
     138     2289933 :             ulong        idx ) {
     139     2289933 :   return (fd_accdb_cache_line_t *)( accdb->cache[ cls ] + idx * fd_accdb_cache_slot_sz[ cls ] );
     140     2289933 : }
     141             : 
     142             : /* Bump the per-partition read counters for the partition that contains
     143             :    file_offset.  Called at preadv2 sites.  Writes are counted at
     144             :    allocate time (see fd_accdb_partition_write_bump) so that they reflect
     145             :    bytes committed to a partition rather than syscalls — the snapshot
     146             :    loader bypasses pwritev2 entirely, but every write still goes through
     147             :    reserve_next_write. */
     148             : static inline void
     149             : fd_accdb_partition_read_bump( fd_accdb_t * accdb,
     150             :                               ulong        file_offset,
     151          42 :                               ulong        bytes ) {
     152          42 :   if( FD_UNLIKELY( !bytes ) ) return;
     153             :   /* Readonly joiners have no partition_pool join (see
     154             :      fd_accdb_join_readonly) and do not contribute to per-partition
     155             :      read telemetry today; their disk reads still show up in the
     156             :      joiner-local fd_accdb_metrics_t bytes_read/read_ops. */
     157          42 :   if( FD_UNLIKELY( !accdb->partition_pool ) ) return;
     158          42 :   ulong partition_idx = file_offset / accdb->shmem->partition_sz;
     159          42 :   ulong partition_cnt = partition_pool_max( accdb->partition_pool );
     160          42 :   FD_CHECK_CRIT( partition_idx<partition_cnt, "read partition index out of range" );
     161          42 :   fd_accdb_partition_t * p = partition_pool_ele( accdb->partition_pool, partition_idx );
     162          42 :   FD_ATOMIC_FETCH_AND_ADD( &p->bytes_read, bytes );
     163          42 :   FD_ATOMIC_FETCH_AND_ADD( &p->read_ops,   1UL   );
     164          42 : }
     165             : 
     166             : /* Bump the per-partition write counters.  bytes is how much landed on
     167             :    this partition, over num_ops reservations. */
     168             : static inline void
     169             : fd_accdb_partition_write_bump( fd_accdb_t * accdb,
     170             :                                ulong        partition_idx,
     171             :                                ulong        bytes,
     172          57 :                                ulong        num_ops ) {
     173          57 :   if( FD_UNLIKELY( !bytes ) ) return;
     174          57 :   ulong partition_cnt = partition_pool_max( accdb->partition_pool );
     175          57 :   FD_CHECK_CRIT( partition_idx<partition_cnt, "write partition index out of range" );
     176          57 :   fd_accdb_partition_t * p = partition_pool_ele( accdb->partition_pool, partition_idx );
     177          57 :   FD_ATOMIC_FETCH_AND_ADD( &p->bytes_written, bytes   );
     178          57 :   FD_ATOMIC_FETCH_AND_ADD( &p->write_ops,     num_ops );
     179          57 : }
     180             : 
     181             : void
     182          69 : fd_accdb_flush_metrics( fd_accdb_t * accdb ) {
     183          69 :   ulong bytes    = accdb->write_stats.bytes;
     184          69 :   ulong num_ops  = accdb->write_stats.num_ops;
     185          69 :   ulong part_idx = accdb->write_stats.partition_idx;
     186             : 
     187          69 :   if( !num_ops ) return;
     188             : 
     189          54 :   memset( &accdb->write_stats, 0, sizeof(accdb->write_stats) );
     190             : 
     191          54 :   FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->shmetrics->disk_current_bytes, bytes );
     192          54 :   fd_accdb_partition_write_bump( accdb, part_idx, bytes, num_ops );
     193          54 : }
     194             : 
     195             : static inline ulong
     196             : cache_line_idx( fd_accdb_t *                  accdb,
     197             :                 ulong                         cls,
     198     1965678 :                 fd_accdb_cache_line_t const * line ) {
     199     1965678 :   return (ulong)( (uchar const *)line - accdb->cache[ cls ] ) / fd_accdb_cache_slot_sz[ cls ];
     200     1965678 : }
     201             : 
     202             : #if FD_TMPL_USE_HANDHOLDING
     203             : static inline int
     204             : fd_accdb_ptr_in_region( fd_accdb_t const * accdb,
     205             :                         ulong              cls,
     206             :                         void const *       ptr ) {
     207             :   if( FD_UNLIKELY( cls>=FD_ACCDB_CACHE_CLASS_CNT ) ) return 0;
     208             : 
     209             :   uchar const * base = accdb->cache[ cls ];
     210             :   if( FD_UNLIKELY( !base ) ) return 0;
     211             : 
     212             :   ulong slot_sz   = fd_accdb_cache_slot_sz[ cls ];
     213             :   ulong region_sz = accdb->shmem->cache_class_max[ cls ] * slot_sz;
     214             :   uchar const * p = (uchar const *)ptr;
     215             : 
     216             :   if( FD_UNLIKELY( p<base || p>=base+region_sz ) ) return 0;
     217             :   return ( (ulong)( p - base ) % slot_sz )==FD_ACCDB_CACHE_META_SZ;
     218             : }
     219             : #endif
     220             : 
     221             : FD_FN_CONST ulong
     222       12792 : fd_accdb_align( void ) {
     223       12792 :   return FD_ACCDB_ALIGN;
     224       12792 : }
     225             : 
     226             : FD_FN_CONST ulong
     227         279 : fd_accdb_footprint( ulong max_live_slots ) {
     228         279 :   ulong l;
     229         279 :   l = FD_LAYOUT_INIT;
     230         279 :   l = FD_LAYOUT_APPEND( l, FD_ACCDB_ALIGN,           sizeof(fd_accdb_t)                     );
     231         279 :   l = FD_LAYOUT_APPEND( l, alignof(fd_accdb_fork_t), max_live_slots*sizeof(fd_accdb_fork_t) );
     232         279 :   return FD_LAYOUT_FINI( l, FD_ACCDB_ALIGN );
     233         279 : }
     234             : 
     235             : void *
     236             : fd_accdb_new( void *              ljoin,
     237             :               fd_accdb_shmem_t *  shmem,
     238             :               int                 fd,
     239             :               ulong               external_epoch_cnt,
     240        4170 :               ulong const **      external_epoch_slots ) {
     241        4170 :   if( FD_UNLIKELY( !ljoin ) ) {
     242           0 :     FD_LOG_WARNING(( "NULL ljoin" ));
     243           0 :     return NULL;
     244           0 :   }
     245             : 
     246        4170 :   if( FD_UNLIKELY( !fd_ulong_is_aligned( (ulong)ljoin, fd_accdb_align() ) ) ) {
     247           0 :     FD_LOG_WARNING(( "misaligned ljoin" ));
     248           0 :     return NULL;
     249           0 :   }
     250             : 
     251        4170 :   if( FD_UNLIKELY( fd<0 ) ) {
     252           0 :     FD_LOG_WARNING(( "fd must be a valid file descriptor" ));
     253           0 :     return NULL;
     254           0 :   }
     255             : 
     256        4170 :   ulong max_live_slots = shmem->max_live_slots;
     257        4170 :   ulong max_accounts = shmem->max_accounts;
     258        4170 :   ulong max_account_writes_per_slot = shmem->max_account_writes_per_slot;
     259        4170 :   ulong partition_cnt = shmem->partition_cnt;
     260             : 
     261        4170 :   ulong chain_cnt = fd_ulong_pow2_up( (max_accounts>>1) + (max_accounts&1UL) );
     262        4170 :   ulong txn_max = max_live_slots * max_account_writes_per_slot;
     263             : 
     264        4170 :   FD_SCRATCH_ALLOC_INIT( l, shmem );
     265        4170 :                              FD_SCRATCH_ALLOC_APPEND( l, FD_ACCDB_SHMEM_ALIGN,           sizeof(fd_accdb_shmem_t)                                );
     266        4170 :   void * _fork_pool_ele    = FD_SCRATCH_ALLOC_APPEND( l, alignof(fd_accdb_fork_shmem_t), max_live_slots*sizeof(fd_accdb_fork_shmem_t)            );
     267        4170 :   void * _descends_sets    = FD_SCRATCH_ALLOC_APPEND( l, descends_set_align(),           max_live_slots*descends_set_footprint( max_live_slots ) );
     268        4170 :   void * _acc_map          = FD_SCRATCH_ALLOC_APPEND( l, alignof(uint),                  chain_cnt*sizeof(uint)                                  );
     269        4170 :   void * _acc_pool_ele     = FD_SCRATCH_ALLOC_APPEND( l, alignof(fd_accdb_accmeta_t),        max_accounts*sizeof(fd_accdb_accmeta_t)             );
     270        4170 :   void * _txn_pool_ele     = FD_SCRATCH_ALLOC_APPEND( l, alignof(fd_accdb_txn_t),        txn_max*sizeof(fd_accdb_txn_t)                          );
     271        4170 :   void * _partition_pool   = FD_SCRATCH_ALLOC_APPEND( l, partition_pool_align(),         partition_pool_footprint( partition_cnt )               );
     272        4170 :   void * _compaction_dlists[ FD_ACCDB_COMPACTION_LAYER_CNT ];
     273       16680 :   for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
     274       12510 :     _compaction_dlists[ k ] = FD_SCRATCH_ALLOC_APPEND( l, compaction_dlist_align(), compaction_dlist_footprint()                                 );
     275       12510 :   }
     276        4170 :   void * _deferred_free_dlist = FD_SCRATCH_ALLOC_APPEND( l, deferred_free_dlist_align(), deferred_free_dlist_footprint()                         );
     277             : 
     278        4170 :   FD_SCRATCH_ALLOC_INIT( l2, ljoin );
     279        4170 :   fd_accdb_t * accdb      = FD_SCRATCH_ALLOC_APPEND( l2, fd_accdb_align(),         sizeof(fd_accdb_t)                     );
     280        4170 :   void * _local_fork_pool = FD_SCRATCH_ALLOC_APPEND( l2, alignof(fd_accdb_fork_t), max_live_slots*sizeof(fd_accdb_fork_t) );
     281             : 
     282        4170 :   accdb->fd = fd;
     283        4170 :   accdb->acquire_state = FD_ACCDB_ACQUIRE_STATE_IDLE;
     284        4170 :   accdb->snapshot_loading = 0;
     285             : 
     286        4170 :   accdb->shmem = (fd_accdb_shmem_t *)shmem;
     287        4170 :   FD_TEST( acc_pool_join( accdb->acc_pool_join, shmem->acc_pool, _acc_pool_ele, max_accounts ) );
     288        4170 :   accdb->acc_pool = accdb->acc_pool_join->ele;
     289        4170 :   accdb->acc_map = _acc_map;
     290        4170 :   FD_TEST( txn_pool_join( accdb->txn_pool, shmem->txn_pool, _txn_pool_ele, txn_max ) );
     291       37530 :   for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) accdb->cache[ c ] = (uchar *)shmem + shmem->cache_region_off[ c ];
     292        4170 :   accdb->partition_pool = partition_pool_join( _partition_pool );
     293        4170 :   FD_TEST( accdb->partition_pool );
     294       16680 :   for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
     295       12510 :     accdb->compaction_dlist[ k ] = compaction_dlist_join( _compaction_dlists[ k ] );
     296       12510 :     FD_TEST( accdb->compaction_dlist[ k ] );
     297       12510 :   }
     298        4170 :   accdb->deferred_free_dlist = deferred_free_dlist_join( _deferred_free_dlist );
     299        4170 :   FD_TEST( accdb->deferred_free_dlist );
     300             : 
     301        4170 :   FD_TEST( fork_pool_join( accdb->fork_shmem_pool, shmem->fork_pool, _fork_pool_ele, max_live_slots ) );
     302        4170 :   accdb->fork_pool = _local_fork_pool;
     303       74547 :   for( ulong i=0UL; i<max_live_slots; i++ ) {
     304       70377 :     fd_accdb_fork_t * fork = &accdb->fork_pool[ i ];
     305       70377 :     fork->shmem = fork_pool_ele( accdb->fork_shmem_pool, i );
     306       70377 :     fork->descends = descends_set_join( (uchar *)_descends_sets + i*descends_set_footprint( max_live_slots ) );
     307       70377 :     FD_TEST( fork->shmem );
     308       70377 :     FD_TEST( fork->descends );
     309       70377 :   }
     310             : 
     311        4170 :   ulong epoch_idx = FD_ATOMIC_FETCH_AND_ADD( &shmem->joiner_cnt, 1UL );
     312        4170 :   FD_TEST( epoch_idx<shmem->joiner_cnt_max );
     313        4170 :   accdb->my_epoch_slot = &shmem->joiner_epochs[ epoch_idx ].val;
     314             : 
     315        4170 :   accdb->external_epoch_slots = external_epoch_slots;
     316        4170 :   accdb->external_epoch_cnt   = external_epoch_cnt;
     317             : 
     318        4170 :   accdb->deferred_acc_buf = (uint *)( (uchar *)shmem + shmem->deferred_acc_buf_off );
     319             : 
     320        4170 :   accdb->delta.chains = (uint *)             ( (uchar *)shmem + shmem->delta.chain_off );
     321        4170 :   accdb->delta.pool   = (fd_accdb_delta_t *) ( (uchar *)shmem + shmem->delta.ele_off   );
     322             : 
     323        4170 :   accdb->deferred_fork_head  = NULL;
     324        4170 :   accdb->deferred_fork_tail  = NULL;
     325        4170 :   accdb->deferred_fork_epoch = 0UL;
     326             : 
     327        4170 :   memset( accdb->metrics,      0, sizeof(fd_accdb_metrics_t) );
     328        4170 :   memset( &accdb->write_stats, 0, sizeof(accdb->write_stats) );
     329             : 
     330        4170 :   return accdb;
     331        4170 : }
     332             : 
     333             : static inline void wait_cmd( fd_accdb_t * accdb );
     334             : static inline void submit_cmd( fd_accdb_t * accdb, uint op, ushort fork_id );
     335             : 
     336             : void
     337           3 : fd_accdb_reset( fd_accdb_t * accdb ) {
     338           3 :   fd_accdb_shmem_t * shmem = accdb->shmem;
     339             : 
     340             :   /* Wait for any pending background command (advance_root / purge) on
     341             :      T2 to finish before clobbering shared state. */
     342           3 :   wait_cmd( accdb );
     343             : 
     344             :   /* Reset pools through the joiner's existing pointers.  acc_pool and
     345             :      txn_pool use POOL_LAZY=1 so reset is O(1).  fork_pool and
     346             :      partition_pool rebuild their free lists in O(max_live_slots) and
     347             :      O(partition_cnt), both small. */
     348           3 :   acc_pool_reset( accdb->acc_pool_join );
     349           3 :   txn_pool_reset( accdb->txn_pool );
     350           3 :   fork_pool_reset( accdb->fork_shmem_pool );
     351           3 :   partition_pool_reset( accdb->partition_pool );
     352             : 
     353             :   /* Clear hash chains */
     354           3 :   fd_memset( accdb->acc_map, 0xFF, shmem->chain_cnt*sizeof(uint) );
     355             : 
     356             :   /* Empty dlists */
     357          12 :   for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
     358           9 :     compaction_dlist_remove_all( accdb->compaction_dlist[ k ], accdb->partition_pool );
     359           9 :   }
     360           3 :   deferred_free_dlist_remove_all( accdb->deferred_free_dlist, accdb->partition_pool );
     361             : 
     362             :   /* Null descends_sets. */
     363         195 :   for( ulong i=0UL; i<shmem->max_live_slots; i++ ) {
     364         192 :     descends_set_null( accdb->fork_pool[ i ].descends );
     365         192 :   }
     366             : 
     367             :   /* Reset shmem scalar fields. */
     368           3 :   shmem->root_fork_id   = (fd_accdb_fork_id_t){ .val = USHORT_MAX };
     369           3 :   shmem->generation     = 0U;
     370           3 :   shmem->partition_lock = 0;
     371           3 :   shmem->partition_max  = 0UL;
     372             : 
     373             :   /* Write heads: sentinel values that force partition-switch on first
     374             :      write. */
     375          12 :   for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
     376           9 :     shmem->whead[ k ] = accdb_offset( shmem->partition_cnt, shmem->partition_sz );
     377           9 :     shmem->has_partition[ k ] = 0;
     378           9 :   }
     379             : 
     380             :   /* Cache state */
     381          27 :   for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) {
     382          24 :     shmem->clock_hand[ c ].val       = 0UL;
     383          24 :     shmem->cache_free[ c ].ver_top   = (ulong)UINT_MAX;
     384          24 :     shmem->cache_free_cnt[ c ].val   = 0UL;
     385          24 :     shmem->cache_class_init[ c ].val = 0UL;
     386          24 :     if( shmem->cache_class_max[ c ]>=shmem->cache_min_reserved*shmem->joiner_cnt_max )
     387          24 :       shmem->cache_class_used[ c ].val = ULONG_MAX;
     388           0 :     else
     389           0 :       shmem->cache_class_used[ c ].val = 0UL;
     390          24 :   }
     391             : 
     392             :   /* Reset every cache slot's metadata to empty sentinels. */
     393          27 :   for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) {
     394          24 :     ulong slot_sz = fd_accdb_cache_slot_sz[ c ];
     395       82200 :     for( ulong i=0UL; i<shmem->cache_class_max[ c ]; i++ ) {
     396       82176 :       fd_accdb_cache_line_t * line = (fd_accdb_cache_line_t *)( accdb->cache[ c ] + i*slot_sz );
     397       82176 :       line->key.generation = UINT_MAX;
     398       82176 :       line->acc_idx        = UINT_MAX;
     399       82176 :       line->refcnt         = 0U;
     400       82176 :       line->referenced     = 0;
     401       82176 :       line->persisted      = 1;
     402       82176 :     }
     403          24 :   }
     404             : 
     405             :   /* Epoch system: reset epoch and all slot values to idle, but
     406             :      preserve joiner_cnt and each tile's my_epoch_slot pointer so that
     407             :      tiles which joined during init keep their original slot indices. */
     408           3 :   shmem->epoch = 1UL;
     409         771 :   for( ulong i=0UL; i<FD_ACCDB_MAX_JOINERS; i++ ) shmem->joiner_epochs[ i ].val = ULONG_MAX;
     410             : 
     411             :   /* Deferred acc buffer. */
     412           3 :   shmem->deferred_acc_buf_cnt = 0UL;
     413           3 :   shmem->deferred_acc_epoch   = 0UL;
     414             : 
     415             :   /* Shared metrics: zero gauges that reflect current state (now empty)
     416             :      but preserve counters and accounts_capacity. */
     417           3 :   shmem->shmetrics->accounts_total       = 0UL;
     418           3 :   shmem->shmetrics->disk_allocated_bytes = 0UL;
     419           3 :   shmem->shmetrics->disk_current_bytes   = 0UL;
     420           3 :   shmem->shmetrics->disk_used_bytes      = 0UL;
     421           3 :   shmem->shmetrics->in_compaction        = 0;
     422             : 
     423             :   /* Command slot */
     424           3 :   shmem->cmd_op      = FD_ACCDB_CMD_IDLE;
     425           3 :   shmem->cmd_fork_id = USHORT_MAX;
     426             : 
     427           3 :   shmem->snapshot_loading = 0;
     428             : 
     429           3 :   FD_COMPILER_MFENCE();
     430             : 
     431             :   /* Tell the accdb tile to clear its stale deferred fork chain.
     432             :      Its deferred_fork_head/tail now reference recycled pool elements;
     433             :      it must discard them before processing any future advance_root or
     434             :      purge command.  The command is asynchronous; the next advance_root
     435             :      or purge call will wait for it to complete via wait_cmd. */
     436           3 :   submit_cmd( accdb, FD_ACCDB_CMD_CLEAR_DEFERRED, 0 );
     437             : 
     438             :   /* Reset local state */
     439           3 :   accdb->deferred_fork_head  = NULL;
     440           3 :   accdb->deferred_fork_tail  = NULL;
     441           3 :   accdb->deferred_fork_epoch = 0UL;
     442           3 :   accdb->snapshot_loading    = 0;
     443           3 :   accdb->acquire_state       = FD_ACCDB_ACQUIRE_STATE_IDLE;
     444           3 :   memset( &accdb->write_stats, 0, sizeof(accdb->write_stats) );
     445           3 : }
     446             : 
     447             : void
     448          21 : fd_accdb_snapshot_load_begin( fd_accdb_t * accdb ) {
     449          21 :   FD_CHECK_CRIT( fd_accdb_snapshot_sync_state( &accdb->shmem->snapshot_sync )==FD_ACCDB_SNAPSHOT_SYNC_IDLE,
     450          21 :                  "snapshot load started while snapshot production active" );
     451          21 :   accdb->snapshot_loading = 1;
     452          21 :   FD_VOLATILE( accdb->shmem->snapshot_loading ) = 1;
     453          21 : }
     454             : 
     455             : static inline void
     456             : change_partition( fd_accdb_t *           accdb,
     457             :                   accdb_offset_t const * offset_before,
     458             :                   accdb_offset_t *       out_offset,
     459             :                   int *                  has_partition,
     460             :                   uchar                  layer );
     461             : 
     462             : void
     463          21 : fd_accdb_snapshot_load_end( fd_accdb_t * accdb ) {
     464          21 :   fd_accdb_flush_metrics( accdb );
     465          21 :   spin_lock_acquire( &accdb->shmem->partition_lock );
     466             : 
     467             :   /* Force the next layer-0 write onto a fresh Hot partition so we do
     468             :      not keep appending live execution writes to the tail of a partition
     469             :      that was tagged Cold during snapshot load.  Must run while
     470             :      snapshot_loading is still set so the partition we just closed
     471             :      (the snapshot-tagged Cold one) is not enqueued for compaction by
     472             :      change_partition's tail-credit try_enqueue.  change_partition will
     473             :      retag the newly-allocated partition as Cold (because the flag is
     474             :      still set), so we fix it back to Hot below. */
     475          21 :   if( FD_LIKELY( accdb->shmem->has_partition[ 0 ] ) ) {
     476          21 :     change_partition( accdb, &accdb->shmem->whead[ 0 ], &accdb->shmem->whead[ 0 ], &accdb->shmem->has_partition[ 0 ], 0 );
     477          21 :     ulong new_idx = packed_partition_idx( &accdb->shmem->whead[ 0 ] );
     478          21 :     fd_accdb_partition_t * newp = partition_pool_ele( accdb->partition_pool, new_idx );
     479          21 :     FD_VOLATILE( newp->layer ) = 0;
     480          21 :   }
     481             : 
     482          21 :   accdb->snapshot_loading = 0;
     483          21 :   FD_VOLATILE( accdb->shmem->snapshot_loading ) = 0;
     484             : 
     485             :   /* Sweep all partitions written during the load — any that crossed
     486             :      the fragmentation threshold while enqueue was suppressed are
     487             :      re-checked now and pushed onto the compaction queue. */
     488          21 :   ulong partition_max = accdb->shmem->partition_max;
     489          90 :   for( ulong p=0UL; p<partition_max; p++ ) {
     490          69 :     fd_accdb_shmem_try_enqueue_compaction( accdb->shmem, p );
     491          69 :   }
     492             : 
     493          21 :   spin_lock_release( &accdb->shmem->partition_lock );
     494          21 : }
     495             : 
     496             : static inline uint
     497             : delta_chain( fd_accdb_shmem_t const * accdb,
     498           0 :              uchar const              pubkey[ 32 ] ) {
     499           0 :   uint hash = (uint)fd_hash32( pubkey, accdb->delta.seed );
     500           0 :   return hash & accdb->delta.chain_mask;
     501           0 : }
     502             : 
     503             : static int
     504             : delta_insert( fd_accdb_t * accdb,
     505        1458 :               uchar const  pubkey[ 32 ] ) {
     506             :   /* FIXME consider batch inserting */
     507        1458 :   fd_accdb_shmem_t * shmem = accdb->shmem;
     508        1458 :   if( FD_UNLIKELY( shmem->delta.head >= shmem->delta.ele_max ) ) return 0;
     509             : 
     510           0 :   uint *             chains = accdb->delta.chains;
     511           0 :   uint *             chain  = &chains[ delta_chain( shmem, pubkey ) ];
     512           0 :   fd_accdb_delta_t * pool   = accdb->delta.pool;
     513             : 
     514           0 :   uint head = *chain;
     515           0 :   for( uint cur=head; cur!=UINT_MAX; cur=pool[ cur ].next ) {
     516           0 :     if( FD_UNLIKELY( !memcmp( pool[ cur ].pubkey, pubkey, 32UL ) ) ) return 1;
     517           0 :   }
     518             : 
     519           0 :   ulong idx = shmem->delta.head++;
     520           0 :   fd_accdb_delta_t * delta = &pool[ idx ];
     521           0 :   delta->next = head;
     522           0 :   memcpy( delta->pubkey, pubkey, 32UL );
     523           0 :   *chain = (uint)idx;
     524           0 :   return 1;
     525           0 : }
     526             : 
     527             : int
     528             : fd_accdb_snapshot_recover_delta( fd_accdb_t *       accdb,
     529           0 :                                  fd_accdb_fork_id_t fork_id ) {
     530           0 :   if( FD_UNLIKELY( fork_id.val>=fork_pool_ele_max( accdb->fork_shmem_pool ) ) ) {
     531           0 :     FD_LOG_CRIT(( "fd_accdb_snapshot_populate_delta: invalid fork id %u (capacity %lu)",
     532           0 :                   (uint)fork_id.val, fork_pool_ele_max( accdb->fork_shmem_pool ) ));
     533           0 :   }
     534             : 
     535           0 :   uint txn_idx = accdb->fork_pool[ fork_id.val ].shmem->txn_head;
     536           0 :   while( txn_idx!=UINT_MAX ) {
     537           0 :     fd_accdb_txn_t const * txn = txn_pool_ele( accdb->txn_pool, (ulong)txn_idx );
     538           0 :     fd_accdb_accmeta_t const * acc = &accdb->acc_pool[ txn->acc_pool_idx ];
     539           0 :     if( FD_UNLIKELY( !delta_insert( accdb, acc->key.pubkey ) ) ) return -1;
     540           0 :     txn_idx = txn->fork.next;
     541           0 :   }
     542           0 :   return 0;
     543           0 : }
     544             : 
     545             : void
     546             : fd_accdb_snapshot_save_whead( fd_accdb_t *                   accdb,
     547          12 :                               fd_accdb_snapshot_recovery_t * out ) {
     548             :   /* Flush metrics to update disk_current_bytes. */
     549          12 :   fd_accdb_flush_metrics( accdb );
     550             : 
     551          12 :   out->whead_val          = FD_VOLATILE_CONST( accdb->shmem->whead[ 0 ].val );
     552          12 :   out->has_partition      = FD_VOLATILE_CONST( accdb->shmem->has_partition[ 0 ] );
     553          12 :   out->partition_max      = FD_VOLATILE_CONST( accdb->shmem->partition_max );
     554          12 :   out->disk_current_bytes = FD_VOLATILE_CONST( accdb->shmem->shmetrics->disk_current_bytes );
     555             : 
     556          12 :   if( out->has_partition ) {
     557          12 :     accdb_offset_t whead = { .val = out->whead_val };
     558          12 :     ulong idx = packed_partition_idx( &whead );
     559          12 :     fd_accdb_partition_t * part = partition_pool_ele( accdb->partition_pool, idx );
     560          12 :     out->savepoint_bytes_freed = FD_VOLATILE_CONST( part->bytes_freed );
     561          12 :   } else {
     562           0 :     out->savepoint_bytes_freed = 0UL;
     563           0 :   }
     564          12 : }
     565             : 
     566             : void
     567             : fd_accdb_snapshot_revert_whead( fd_accdb_t *                         accdb,
     568          12 :                                 fd_accdb_snapshot_recovery_t const * recover ) {
     569          12 :   fd_accdb_shmem_t * shmem = accdb->shmem;
     570             : 
     571             :   /* Partitions are about to be released, so flush metrics first. */
     572          12 :   fd_accdb_flush_metrics( accdb );
     573             : 
     574             :   /* Wait for any pending background command (purge) on T2 to finish
     575             :      before releasing partitions. */
     576          12 :   wait_cmd( accdb );
     577             : 
     578          12 :   ulong cur_partition_max = shmem->partition_max;
     579             : 
     580             :   /* Materialize the active partition's write_offset from the whead
     581             :      before releasing.  Closed partitions have write_offset set by
     582             :      change_partition, but the last active partition still has
     583             :      write_offset == 0 from its initialization.  The real byte offset
     584             :      is encoded in whead[0]. */
     585          12 :   if( shmem->has_partition[ 0 ] && cur_partition_max>recover->partition_max ) {
     586           6 :     ulong active_idx = packed_partition_idx( &shmem->whead[ 0 ] );
     587           6 :     if( active_idx>=recover->partition_max && active_idx<cur_partition_max ) {
     588           6 :       fd_accdb_partition_t * active = partition_pool_ele( accdb->partition_pool, active_idx );
     589           6 :       active->write_offset = packed_partition_offset( &shmem->whead[ 0 ] );
     590           6 :     }
     591           6 :   }
     592             : 
     593             :   /* Release partitions that have been previously allocated.  Must hold
     594             :      partition_lock because partition_pool_ele_release mutates the
     595             :      pool free list.  Before releasing, unlink any partition that sits
     596             :      on a compaction dlist (queued flag).
     597             : 
     598             :      Release in descending index order so that the LIFO free list
     599             :      re-acquires them in ascending order (P, P+1, P+2, ...).  This
     600             :      keeps reserve_next_write in sync with snapwr, which advances
     601             :      its flat file offset sequentially. */
     602          12 :   spin_lock_acquire( &shmem->partition_lock );
     603          24 :   for( ulong p=cur_partition_max; p>recover->partition_max; p-- ) {
     604          12 :     fd_accdb_partition_t * part = partition_pool_ele( accdb->partition_pool, p-1UL );
     605          12 :     if( FD_UNLIKELY( part->queued ) ) {
     606           6 :       compaction_dlist_ele_remove( accdb->compaction_dlist[ part->layer ], part, accdb->partition_pool );
     607           6 :     }
     608          12 :     FD_VOLATILE( part->write_offset ) = 0UL;
     609          12 :     partition_pool_ele_release( accdb->partition_pool, part );
     610          12 :   }
     611             : 
     612          12 :   shmem->whead[ 0 ].val     = recover->whead_val;
     613          12 :   shmem->has_partition[ 0 ] = recover->has_partition;
     614          12 :   shmem->partition_max      = recover->partition_max;
     615             : 
     616             :   /* disk_used_bytes is NOT saved/restored here.  It is implicitly
     617             :      reverted by purge_inner -> acc_unlink, which decrements
     618             :      disk_used_bytes for each unlinked entry.  The caller must
     619             :      complete the purge before calling revert_whead. */
     620             : 
     621          12 :   shmem->shmetrics->disk_current_bytes = recover->disk_current_bytes;
     622          12 :   shmem->shmetrics->disk_allocated_bytes = recover->partition_max * shmem->partition_sz;
     623             : 
     624          12 :   if( recover->has_partition ) {
     625          12 :     accdb_offset_t sp_off = (accdb_offset_t){ .val = recover->whead_val };
     626          12 :     ulong sp_idx = packed_partition_idx( &sp_off );
     627          12 :     fd_accdb_partition_t * sp = partition_pool_ele( accdb->partition_pool, sp_idx );
     628          12 :     sp->bytes_freed   = recover->savepoint_bytes_freed;
     629          12 :     sp->write_offset  = 0UL;
     630          12 :   }
     631             : 
     632          12 :   spin_lock_release( &shmem->partition_lock );
     633          12 : }
     634             : 
     635             : fd_accdb_t *
     636        4170 : fd_accdb_join( void * shaccdb ) {
     637        4170 :   if( FD_UNLIKELY( !shaccdb ) ) {
     638           0 :     FD_LOG_WARNING(( "NULL shaccdb" ));
     639           0 :     return NULL;
     640           0 :   }
     641             : 
     642        4170 :   if( FD_UNLIKELY( !fd_ulong_is_aligned( (ulong)shaccdb, fd_accdb_align() ) ) ) {
     643           0 :     FD_LOG_WARNING(( "misaligned shaccdb" ));
     644           0 :     return NULL;
     645           0 :   }
     646             : 
     647        4170 :   return (fd_accdb_t*)shaccdb;
     648        4170 : }
     649             : 
     650             : fd_accdb_t *
     651             : fd_accdb_join_readonly( void *             ljoin,
     652             :                         fd_accdb_shmem_t * shmem,
     653             :                         ulong *            my_epoch_slot_rw,
     654           0 :                         int                fd_ro ) {
     655           0 :   if( FD_UNLIKELY( !ljoin ) ) {
     656           0 :     FD_LOG_WARNING(( "NULL ljoin" ));
     657           0 :     return NULL;
     658           0 :   }
     659             : 
     660           0 :   if( FD_UNLIKELY( !fd_ulong_is_aligned( (ulong)ljoin, fd_accdb_align() ) ) ) {
     661           0 :     FD_LOG_WARNING(( "misaligned ljoin" ));
     662           0 :     return NULL;
     663           0 :   }
     664             : 
     665           0 :   if( FD_UNLIKELY( !my_epoch_slot_rw ) ) {
     666           0 :     FD_LOG_WARNING(( "NULL my_epoch_slot_rw" ));
     667           0 :     return NULL;
     668           0 :   }
     669             : 
     670           0 :   ulong max_live_slots               = shmem->max_live_slots;
     671           0 :   ulong max_accounts                 = shmem->max_accounts;
     672           0 :   ulong max_account_writes_per_slot  = shmem->max_account_writes_per_slot;
     673           0 :   ulong partition_cnt                = shmem->partition_cnt;
     674             : 
     675           0 :   ulong chain_cnt = fd_ulong_pow2_up( (max_accounts>>1) + (max_accounts&1UL) );
     676           0 :   ulong txn_max   = max_live_slots * max_account_writes_per_slot;
     677             : 
     678             :   /* Recompute the same shmem scratch layout that fd_accdb_shmem_new
     679             :      used.  All FD_SCRATCH_ALLOC_APPEND calls here only compute pointer
     680             :      offsets — they do not write to shmem. */
     681           0 :   FD_SCRATCH_ALLOC_INIT( l, shmem );
     682           0 :                              FD_SCRATCH_ALLOC_APPEND( l, FD_ACCDB_SHMEM_ALIGN,           sizeof(fd_accdb_shmem_t)                                );
     683           0 :   void * _fork_pool_ele    = FD_SCRATCH_ALLOC_APPEND( l, alignof(fd_accdb_fork_shmem_t), max_live_slots*sizeof(fd_accdb_fork_shmem_t)            );
     684           0 :   void * _descends_sets    = FD_SCRATCH_ALLOC_APPEND( l, descends_set_align(),           max_live_slots*descends_set_footprint( max_live_slots ) );
     685           0 :   void * _acc_map          = FD_SCRATCH_ALLOC_APPEND( l, alignof(uint),                  chain_cnt*sizeof(uint)                                  );
     686           0 :   void * _acc_pool_ele     = FD_SCRATCH_ALLOC_APPEND( l, alignof(fd_accdb_accmeta_t),    max_accounts*sizeof(fd_accdb_accmeta_t)                 );
     687           0 :                              FD_SCRATCH_ALLOC_APPEND( l, alignof(fd_accdb_txn_t),        txn_max*sizeof(fd_accdb_txn_t)                          );
     688           0 :                              FD_SCRATCH_ALLOC_APPEND( l, partition_pool_align(),         partition_pool_footprint( partition_cnt )               );
     689           0 :   for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
     690           0 :                              FD_SCRATCH_ALLOC_APPEND( l, compaction_dlist_align(),       compaction_dlist_footprint()                            );
     691           0 :   }
     692           0 :                              FD_SCRATCH_ALLOC_APPEND( l, deferred_free_dlist_align(),    deferred_free_dlist_footprint()                         );
     693             : 
     694           0 :   FD_SCRATCH_ALLOC_INIT( l2, ljoin );
     695           0 :   fd_accdb_t * accdb      = FD_SCRATCH_ALLOC_APPEND( l2, fd_accdb_align(),         sizeof(fd_accdb_t)                     );
     696           0 :   void * _local_fork_pool = FD_SCRATCH_ALLOC_APPEND( l2, alignof(fd_accdb_fork_t), max_live_slots*sizeof(fd_accdb_fork_t) );
     697             : 
     698           0 :   accdb->fd    = fd_ro;
     699           0 :   accdb->acquire_state = FD_ACCDB_ACQUIRE_STATE_IDLE;
     700           0 :   accdb->shmem = shmem;
     701           0 :   FD_TEST( acc_pool_join( accdb->acc_pool_join, shmem->acc_pool, _acc_pool_ele, max_accounts ) );
     702           0 :   accdb->acc_pool = accdb->acc_pool_join->ele;
     703           0 :   accdb->acc_map  = _acc_map;
     704           0 :   for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) accdb->cache[ c ] = (uchar *)shmem + shmem->cache_region_off[ c ];
     705             : 
     706             :   /* Writer-only structures: leave NULL so any accidental writer-path
     707             :      call from a readonly joiner crashes loudly rather than corrupting
     708             :      state. */
     709           0 :   accdb->partition_pool      = NULL;
     710           0 :   for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) accdb->compaction_dlist[ k ] = NULL;
     711           0 :   accdb->deferred_free_dlist = NULL;
     712           0 :   accdb->delta.chains        = NULL;
     713           0 :   accdb->delta.pool          = NULL;
     714             : 
     715           0 :   FD_TEST( fork_pool_join( accdb->fork_shmem_pool, shmem->fork_pool, _fork_pool_ele, max_live_slots ) );
     716           0 :   accdb->fork_pool = _local_fork_pool;
     717           0 :   for( ulong i=0UL; i<max_live_slots; i++ ) {
     718           0 :     fd_accdb_fork_t * fork = &accdb->fork_pool[ i ];
     719           0 :     fork->shmem    = fork_pool_ele( accdb->fork_shmem_pool, i );
     720           0 :     fork->descends = descends_set_join( (uchar *)_descends_sets + i*descends_set_footprint( max_live_slots ) );
     721           0 :     FD_TEST( fork->shmem );
     722           0 :     FD_TEST( fork->descends );
     723           0 :   }
     724             : 
     725             :   /* my_epoch_slot_rw points at memory owned by this joiner (e.g. a
     726             :      private per-tile fseq) that the joiner can write to.  The
     727             :      accdb tile sees it via its external_epoch_slots[] array (mapped
     728             :      read-only) and includes it in its compaction epoch scan.
     729             :      Storing through this pointer is the only side effect a readonly
     730             :      joiner has on shared state. */
     731           0 :   accdb->my_epoch_slot = my_epoch_slot_rw;
     732             : 
     733             :   /* Readonly joiners do not own external slots themselves; only the
     734             :      compaction tile / writer joiners do. */
     735           0 :   accdb->external_epoch_slots = NULL;
     736           0 :   accdb->external_epoch_cnt   = 0UL;
     737             : 
     738           0 :   accdb->deferred_acc_buf    = NULL;
     739           0 :   accdb->deferred_fork_head  = NULL;
     740           0 :   accdb->deferred_fork_tail  = NULL;
     741           0 :   accdb->deferred_fork_epoch = 0UL;
     742             : 
     743           0 :   memset( accdb->metrics,      0, sizeof(fd_accdb_metrics_t) );
     744           0 :   memset( &accdb->write_stats, 0, sizeof(accdb->write_stats) );
     745             : 
     746           0 :   return accdb;
     747           0 : }
     748             : 
     749             : /* T1 -> T2 cmd channel.  Two states on cmd_op:
     750             : 
     751             :      IDLE     - no cmd in flight
     752             :      non-IDLE - cmd pending; T2 will process it then flip back to IDLE
     753             : 
     754             :    T1 submits by writing fork_id then cmd_op (non-IDLE).  T2 processes
     755             :    by reading fork_id then writing cmd_op = IDLE.  T1 waits for IDLE
     756             :    before submitting again, so T2 never sees a half-written cmd and
     757             :    never re-processes the same cmd. */
     758             : 
     759             : static inline void
     760        9369 : wait_cmd( fd_accdb_t * accdb ) {
     761        9369 :   fd_accdb_shmem_t * shmem = accdb->shmem;
     762    11559410 :   while( FD_VOLATILE_CONST( shmem->cmd_op )!=FD_ACCDB_CMD_IDLE ) FD_SPIN_PAUSE();
     763        9369 :   FD_COMPILER_MFENCE();
     764        9369 : }
     765             : 
     766             : static inline void
     767             : submit_cmd( fd_accdb_t * accdb,
     768             :             uint         op,
     769         528 :             ushort       fork_id ) {
     770         528 :   fd_accdb_shmem_t * shmem = accdb->shmem;
     771         528 :   FD_VOLATILE( shmem->cmd_fork_id ) = fork_id;
     772         528 :   FD_COMPILER_MFENCE();
     773         528 :   FD_VOLATILE( shmem->cmd_op ) = op;
     774         528 : }
     775             : 
     776             : fd_accdb_fork_id_t
     777             : fd_accdb_attach_child( fd_accdb_t *       accdb,
     778        8829 :                        fd_accdb_fork_id_t parent_fork_id ) {
     779             :   /* fork_pool_acquire is not NULL-checked: replay gates attaches on
     780             :      fd_banks_can_start_bank, and wait_cmd ensures the prior
     781             :      advance_root has fully run on T2, so
     782             :      live + deferred forks <= max_live_slots. */
     783        8829 :   wait_cmd( accdb );
     784             : 
     785        8829 :   if( FD_UNLIKELY( accdb->shmem->generation==UINT_MAX ) ) FD_LOG_ERR(( "accdb ran out of generation sequence numbers, restart required" ));
     786             : 
     787        8829 :   fd_accdb_fork_shmem_t * acquired = fork_pool_acquire( accdb->fork_shmem_pool );
     788        8829 :   if( FD_UNLIKELY( !acquired ) ) {
     789           3 :     submit_cmd( accdb, FD_ACCDB_CMD_DRAIN_DEFERRED, USHORT_MAX );
     790           3 :     wait_cmd( accdb );
     791           3 :     acquired = fork_pool_acquire( accdb->fork_shmem_pool );
     792           3 :     FD_CHECK_CRIT( acquired, "accdb fork pool exhausted after deferred drain" );
     793           3 :   }
     794        8829 :   ulong idx = fork_pool_idx( accdb->fork_shmem_pool, acquired );
     795             : 
     796        8829 :   fd_accdb_fork_t * fork = &accdb->fork_pool[ idx ];
     797        8829 :   fd_accdb_fork_id_t fork_id = { .val = (ushort)idx };
     798             : 
     799        8829 :   fork->shmem->child_id = (fd_accdb_fork_id_t){ .val = USHORT_MAX };
     800             : 
     801        8829 :   if( FD_LIKELY( parent_fork_id.val==USHORT_MAX ) ) {
     802        4110 :     fork->shmem->parent_id  = (fd_accdb_fork_id_t){ .val = USHORT_MAX };
     803        4110 :     fork->shmem->sibling_id = (fd_accdb_fork_id_t){ .val = USHORT_MAX };
     804             : 
     805        4110 :     descends_set_null( fork->descends );
     806        4110 :     accdb->shmem->root_fork_id = fork_id;
     807        4719 :   } else {
     808        4719 :     fd_accdb_fork_t * parent = &accdb->fork_pool[ parent_fork_id.val ];
     809        4719 :     fork->shmem->parent_id  = parent_fork_id;
     810             : 
     811        4719 :     descends_set_copy( fork->descends, parent->descends );
     812        4719 :     descends_set_insert( fork->descends, parent_fork_id.val );
     813             : 
     814             :     /* Atomically prepend to parent's child list.  T2 (background_purge)
     815             :        may concurrently unlink a different child from the same list, so
     816             :        we must CAS here. */
     817        4719 :     FD_COMPILER_MFENCE();
     818        4719 :     for(;;) {
     819        4719 :       ushort old_head = FD_VOLATILE_CONST( parent->shmem->child_id.val );
     820        4719 :       fork->shmem->sibling_id = (fd_accdb_fork_id_t){ .val = old_head };
     821        4719 :       FD_COMPILER_MFENCE();
     822        4719 :       if( FD_LIKELY( FD_ATOMIC_CAS( &parent->shmem->child_id.val, old_head, fork_id.val )==old_head ) ) break;
     823           0 :       FD_SPIN_PAUSE();
     824           0 :     }
     825        4719 :   }
     826             : 
     827        8829 :   fork->shmem->generation = accdb->shmem->generation++;
     828        8829 :   fork->shmem->txn_head = UINT_MAX;
     829             : 
     830        8829 :   FD_TEST( !descends_set_test( fork->descends, fork_id.val ) );
     831             : 
     832        8829 :   return fork_id;
     833        8829 : }
     834             : 
     835             : /* evict_clear_acc_cache_ref atomically tears down acc->cache_idx and
     836             :    acc->executable_size.CACHE_VALID for an acc that is being evicted
     837             :    from cache line (size_class, line_idx).  The caller must already
     838             :    hold an exclusive claim on the line (line->refcnt ==
     839             :    FD_ACCDB_EVICT_SENTINEL) so that no concurrent thread can pin the
     840             :    line.
     841             : 
     842             :    The naive sequence (clear cache_idx, clear VALID) lets a reader in
     843             :    cold_load_acc see VALID=1 and read a stale INVAL cache_idx, which
     844             :    decodes to an OOB cache_line pointer.  The reverse sequence (clear
     845             :    VALID, clear cache_idx) lets a concurrent cold_load_acc observe
     846             :    VALID=0/CLAIM=0 and start publishing a *new* cache_idx + VALID=1
     847             :    between our two stores; our later cache_idx=INVAL would then
     848             :    stomp on the cold-loader's published idx.
     849             : 
     850             :    We close both races by acquiring CACHE_CLAIM_BIT before mutating
     851             :    acc->cache_idx.  cold_load_acc spins while CLAIM is held, so it
     852             :    cannot enter the publish path concurrently.  If CLAIM is already
     853             :    held, a cold-loader is already mid-publish; in that case
     854             :    acc->cache_idx is being repointed away from our line, and we must
     855             :    not touch it.  After mutation we release CLAIM.
     856             : 
     857             :    Verifies acc->cache_idx still encodes (size_class, line_idx) before
     858             :    clobbering, in case the acc was concurrently re-published into a
     859             :    different line (e.g. by a previous cold_load_acc completing before
     860             :    we arrived). */
     861             : 
     862             : static inline void
     863             : evict_clear_acc_cache_ref( fd_accdb_accmeta_t * accmeta,
     864             :                            ulong                size_class,
     865         366 :                            ulong                line_idx ) {
     866         366 :   uint expected_cidx = FD_ACCDB_ACC_CIDX_PACK( (uint)size_class, (uint)line_idx );
     867             : 
     868             :   /* CAS-acquire CLAIM.  If a cold-loader already holds CLAIM, they
     869             :      own the publish path; bail without touching accmeta fields (their
     870             :      republish is repointing accmeta->cache_idx away from our line). */
     871         366 :   for(;;) {
     872         366 :     uint cur = FD_VOLATILE_CONST( accmeta->executable_size );
     873         366 :     if( FD_UNLIKELY( cur & FD_ACCDB_SIZE_CACHE_CLAIM_BIT ) ) return;
     874         366 :     uint nxt = cur | FD_ACCDB_SIZE_CACHE_CLAIM_BIT;
     875         366 :     if( FD_LIKELY( FD_ATOMIC_CAS( &accmeta->executable_size, cur, nxt )==cur ) ) break;
     876           0 :     fd_racesan_hook( "accdb_evict_clear:claim_wait" );
     877           0 :     FD_SPIN_PAUSE();
     878           0 :   }
     879             : 
     880         366 :   fd_racesan_hook( "accdb_evict_clear:post_claim" );
     881             : 
     882             :   /* CLAIM held.  If accmeta->cache_idx still points at our line, clear
     883             :      VALID and INVAL the cache_idx.  Otherwise the accmeta was already
     884             :      re-published into a different line; leave it alone. */
     885         366 :   if( FD_LIKELY( FD_VOLATILE_CONST( accmeta->cache_idx )==expected_cidx ) ) {
     886         366 :     FD_ATOMIC_FETCH_AND_AND( &accmeta->executable_size, ~FD_ACCDB_SIZE_CACHE_VALID_BIT );
     887         366 :     FD_VOLATILE( accmeta->cache_idx ) = FD_ACCDB_ACC_CIDX_INVAL;
     888         366 :   }
     889             : 
     890             :   /* Release CLAIM. */
     891         366 :   FD_ATOMIC_FETCH_AND_AND( &accmeta->executable_size, ~FD_ACCDB_SIZE_CACHE_CLAIM_BIT );
     892         366 : }
     893             : 
     894             : /* cache_free_push pushes a fully-freed cache line onto the per-class
     895             :    CAS free list (Treiber stack).  The caller must have already
     896             :    invalidated the line (key.generation==UINT_MAX, acc_idx==UINT_MAX)
     897             :    and set persisted=1 before pushing.  Those stores must also be
     898             :    ordered before the store that releases refcnt to 0. */
     899             : 
     900             : static inline void
     901             : cache_free_push( fd_accdb_t * accdb,
     902             :                  ulong        size_class,
     903      819951 :                  fd_accdb_cache_line_t * line ) {
     904      819951 :   ulong line_idx = cache_line_idx( accdb, size_class, line );
     905      819951 :   for(;;) {
     906      819951 :     ulong old_vt  = FD_VOLATILE_CONST( accdb->shmem->cache_free[ size_class ].ver_top );
     907      819951 :     uint  old_top = (uint)( old_vt & (ulong)UINT_MAX );
     908      819951 :     uint  old_ver = (uint)( old_vt >> 32 );
     909      819951 :     line->next = old_top;
     910      819951 :     FD_COMPILER_MFENCE();
     911      819951 :     ulong new_vt = ((ulong)(uint)( old_ver+1U ) << 32) | (ulong)(uint)line_idx;
     912      819951 :     if( FD_LIKELY( FD_ATOMIC_CAS( &accdb->shmem->cache_free[ size_class ].ver_top, old_vt, new_vt )==old_vt ) ) {
     913      819951 :       FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->cache_free_cnt[ size_class ].val, 1UL );
     914      819951 :       return;
     915      819951 :     }
     916           0 :     FD_SPIN_PAUSE();
     917           0 :   }
     918      819951 : }
     919             : 
     920             : /* cache_free_pop pops a line from the per-class CAS free list.  Returns
     921             :    NULL if the list is empty. */
     922             : 
     923             : static inline fd_accdb_cache_line_t *
     924             : cache_free_pop( fd_accdb_t * accdb,
     925      932526 :                 ulong        size_class ) {
     926      932526 :   for(;;) {
     927      932526 :     ulong old_vt  = FD_VOLATILE_CONST( accdb->shmem->cache_free[ size_class ].ver_top );
     928      932526 :     uint  old_top = (uint)( old_vt & (ulong)UINT_MAX );
     929      932526 :     if( FD_UNLIKELY( old_top==UINT_MAX ) ) return NULL;
     930      788325 :     uint  old_ver = (uint)( old_vt >> 32 );
     931      788325 :     fd_accdb_cache_line_t * top = cache_line( accdb, size_class, (ulong)old_top );
     932      788325 :     uint next = FD_VOLATILE_CONST( top->next );
     933      788325 :     ulong new_vt = ((ulong)(uint)( old_ver+1U ) << 32) | (ulong)next;
     934      788325 :     if( FD_LIKELY( FD_ATOMIC_CAS( &accdb->shmem->cache_free[ size_class ].ver_top, old_vt, new_vt )==old_vt ) ) {
     935      788325 :       FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->cache_free_cnt[ size_class ].val, 1UL );
     936      788325 :       return top;
     937      788325 :     }
     938           0 :     FD_SPIN_PAUSE();
     939           0 :   }
     940      932526 : }
     941             : 
     942             : /* cache_try_pin attempts a lock-free pin of a cache-hit line.  Returns
     943             :    the line if successfully pinned, or NULL if the line is being evicted
     944             :    or was recycled (ABA). */
     945             : 
     946             : static inline fd_accdb_cache_line_t *
     947             : cache_try_pin( fd_accdb_cache_line_t * line,
     948             :                uchar const             pubkey[ 32 ],
     949       97542 :                uint                    generation ) {
     950       97542 :   for(;;) {
     951       97542 :     uint old_rc = FD_VOLATILE_CONST( line->refcnt );
     952       97542 :     if( FD_UNLIKELY( old_rc==FD_ACCDB_EVICT_SENTINEL ) ) return NULL;
     953             :     /* No saturation guard needed: refcnt is a uint and at most
     954             :        FD_ACCDB_MAX_JOINERS (256) threads can pin concurrently,
     955             :        so old_rc+1 can never reach FD_ACCDB_EVICT_SENTINEL
     956             :        (UINT_MAX) or wrap. */
     957       97542 :     if( FD_LIKELY( FD_ATOMIC_CAS( &line->refcnt, old_rc, old_rc+1U )==old_rc ) ) {
     958             :       /* Pinned.  ABA check: verify the key hasn't changed under us. */
     959       97542 :       fd_racesan_hook( "accdb_try_pin:post_cas" );
     960       97542 :       FD_COMPILER_MFENCE();
     961       97542 :       if( FD_UNLIKELY( line->key.generation!=generation ||
     962       97542 :                        memcmp( line->key.pubkey, pubkey, 32UL ) ) ) {
     963           0 :         FD_ATOMIC_FETCH_AND_SUB( &line->refcnt, 1U );
     964           0 :         return NULL;
     965           0 :       }
     966       97542 :       line->referenced = 1;
     967       97542 :       fd_racesan_hook( "cache_try_pin:pinned" );
     968       97542 :       return line;
     969       97542 :     }
     970           0 :     FD_SPIN_PAUSE();
     971           0 :   }
     972       97542 : }
     973             : 
     974             : /* wait_for_epoch_drain spins until every joiner's published epoch
     975             :    exceeds tag, meaning all readers that were active at epoch=tag have
     976             :    since exited their critical sections. */
     977             : 
     978             : static void
     979             : wait_for_epoch_drain( fd_accdb_t * accdb,
     980         864 :                       ulong        tag ) {
     981         864 :   for(;;) {
     982         864 :     ulong min_epoch = ULONG_MAX;
     983         864 :     ulong joiner_cnt = FD_VOLATILE_CONST( accdb->shmem->joiner_cnt );
     984        1746 :     for( ulong t=0UL; t<joiner_cnt; t++ ) {
     985         882 :       ulong e = FD_VOLATILE_CONST( accdb->shmem->joiner_epochs[ t ].val );
     986         882 :       if( FD_LIKELY( e<min_epoch ) ) min_epoch = e;
     987         882 :     }
     988         864 :     for( ulong t=0UL; t<accdb->external_epoch_cnt; t++ ) {
     989           0 :       ulong e = FD_VOLATILE_CONST( *accdb->external_epoch_slots[ t ] );
     990           0 :       if( FD_LIKELY( e<min_epoch ) ) min_epoch = e;
     991           0 :     }
     992         864 :     if( FD_LIKELY( tag<min_epoch ) ) break;
     993           0 :     fd_racesan_hook( "accdb_epoch_drain:wait" );
     994           0 :     FD_SPIN_PAUSE();
     995           0 :   }
     996         864 : }
     997             : 
     998             : /* drain_deferred_frees releases back to their respective pools any acc
     999             :    batch and/or fork slots that were unlinked in a prior advance_root /
    1000             :    purge call.  The resources cannot be released immediately because
    1001             :    concurrent readers may still reference them. We wait until every
    1002             :    joiner's published epoch exceeds the tag stamped when each resource
    1003             :    was unlinked.
    1004             : 
    1005             :    Must be called before creating new deferred batches (there is at most
    1006             :    one of each outstanding at a time). */
    1007             : 
    1008             : static void
    1009         525 : drain_deferred_frees( fd_accdb_t * accdb ) {
    1010         525 :   if( FD_UNLIKELY( accdb->deferred_fork_head ) ) {
    1011         435 :     wait_for_epoch_drain( accdb, accdb->deferred_fork_epoch );
    1012         435 :     fork_pool_release_chain( accdb->fork_shmem_pool, accdb->deferred_fork_head, accdb->deferred_fork_tail );
    1013         435 :     accdb->deferred_fork_head = NULL;
    1014         435 :     accdb->deferred_fork_tail = NULL;
    1015         435 :   }
    1016             : 
    1017         525 :   ulong n = accdb->shmem->deferred_acc_buf_cnt;
    1018         525 :   if( FD_LIKELY( !n ) ) return;
    1019         429 :   wait_for_epoch_drain( accdb, accdb->shmem->deferred_acc_epoch );
    1020             : 
    1021             :   /* All readers that could have been holding a captured pointer to any
    1022             :      of these accs at unlink time have now exited their epoch sections.
    1023             :      It is safe to materialize pool.next links and hand the chain to
    1024             :      acc_pool_release_chain. */
    1025         429 :   uint *               buf      = accdb->deferred_acc_buf;
    1026         429 :   fd_accdb_accmeta_t * acc_pool = accdb->acc_pool;
    1027             : 
    1028             :   /* Late-publish sweep: a concurrent acquire evictor may have published
    1029             :      a new offset into one of these accmetas after acc_unlink's
    1030             :      xchg-to-INVAL but before exiting its epoch.  Now that the epoch has
    1031             :      drained, any such publish is complete and visible.  Free the
    1032             :      orphaned disk bytes here, before the accmeta is released to the
    1033             :      pool and its fields recycled. */
    1034         429 :   ulong acc_pool_cap = acc_pool_ele_max( accdb->acc_pool_join );
    1035        1515 :   for( ulong i=0UL; i<n; i++ ) {
    1036        1086 :     FD_TEST( (ulong)buf[ i ]<acc_pool_cap );
    1037        1086 :     fd_accdb_accmeta_t * accmeta = &acc_pool[ buf[ i ] ];
    1038        1086 :     ulong off = fd_accdb_acc_offset( accmeta );
    1039        1086 :     if( FD_UNLIKELY( off!=FD_ACCDB_OFF_INVAL ) ) {
    1040           0 :       ulong entry_sz = (ulong)FD_ACCDB_SIZE_DATA(accmeta->executable_size)+sizeof(fd_accdb_disk_meta_t);
    1041           0 :       fd_accdb_shmem_bytes_freed( accdb->shmem, off, entry_sz );
    1042           0 :       FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->shmetrics->disk_used_bytes, entry_sz );
    1043           0 :     }
    1044        1086 :   }
    1045             : 
    1046        1086 :   for( ulong i=0UL; i+1UL<n; i++ ) {
    1047         657 :     acc_pool[ buf[ i ] ].pool.next = acc_pool_private_cidx( (ulong)buf[ i+1UL ] );
    1048         657 :   }
    1049         429 :   fd_accdb_accmeta_t * head = &acc_pool[ buf[ 0UL ] ];
    1050         429 :   fd_accdb_accmeta_t * tail = &acc_pool[ buf[ n-1UL ] ];
    1051         429 :   acc_pool_release_chain( accdb->acc_pool_join, head, tail );
    1052         429 :   accdb->shmem->deferred_acc_buf_cnt = 0UL;
    1053         429 : }
    1054             : 
    1055             : /* deferred_acc_append records an unlinked acc index in the side buffer
    1056             :    for later release after wait_for_epoch_drain.  T2 is the sole writer.
    1057             :    The chain link from acc->pool.next is NOT laid down here: pool.next
    1058             :    is union-aliased to cache_idx, and a concurrent cold_load_acc may
    1059             :    still publish through a captured pointer until the epoch drains.
    1060             :    Materialization of the chain happens in drain_deferred_frees. */
    1061             : 
    1062             : static inline void
    1063             : deferred_acc_append( fd_accdb_t * accdb,
    1064        1455 :                      uint         acc_idx ) {
    1065        1455 :   fd_accdb_shmem_t * shmem = accdb->shmem;
    1066        1455 :   FD_TEST( shmem->deferred_acc_buf_cnt<shmem->deferred_acc_buf_max );
    1067        1455 :   accdb->deferred_acc_buf[ shmem->deferred_acc_buf_cnt++ ] = acc_idx;
    1068        1455 : }
    1069             : 
    1070             : /* acc_unlink unlinks an account from its hash map chain, frees any
    1071             :    associated disk bytes, and invalidates a stale cache reference.  Does
    1072             :    NOT release the acc pool slot — the caller is responsible for that
    1073             :    (or for batching releases).
    1074             : 
    1075             :    prev is the previous element in the map chain (UINT_MAX if acc_idx is
    1076             :    the head).
    1077             : 
    1078             :    CONCURRENCY: The chain link being removed is swapped out with a CAS
    1079             :    so that a concurrent fd_accdb_release prepending to the same chain
    1080             :    cannot lose its update.  If a head-removal CAS fails (a new node was
    1081             :    prepended since we loaded the head), we re-walk from the new head to
    1082             :    find the target as an interior node.  Interior CAS cannot fail from
    1083             :    inserts (inserts only touch the head) and only one remover exists at
    1084             :    a time (advance_root / purge are serialized). */
    1085             : 
    1086             : static inline void
    1087             : acc_unlink( fd_accdb_t * accdb,
    1088             :             uint         map_idx,
    1089             :             uint         prev,
    1090        1455 :             uint         acc_idx ) {
    1091        1455 :   fd_accdb_accmeta_t * accmeta = &accdb->acc_pool[ acc_idx ];
    1092             : 
    1093             :   /* Atomically capture and clear the offset.  Two races to defuse:
    1094             : 
    1095             :      (1) A concurrent fd_accdb_acquire_inner that is CLOCK-evicting the
    1096             :          cache line currently holding this acc's data may have already
    1097             :          xchg'd the offset to INVAL in step 5-6 and freed the old disk
    1098             :          bytes.  Without atomicity we would re-read the old offset and
    1099             :          free those same bytes a second time.  The xchg here serializes:
    1100             :          whoever wins sees the real offset and frees; the loser sees
    1101             :          INVAL and skips.
    1102             : 
    1103             :      (2) That same evictor may also be mid-flight to publish a NEW
    1104             :          offset in step 9 (after step 5-6's free but before step 9's
    1105             :          store).  That late publish lands on an accmeta that is about
    1106             :          to be chain-unlinked and deferred-released.  drain_deferred_
    1107             :          frees sweeps the deferred buffer after epoch drain to catch
    1108             :          the late publish and free the orphaned bytes. */
    1109        1455 :   ulong entry_sz = (ulong)FD_ACCDB_SIZE_DATA(accmeta->executable_size)+sizeof(fd_accdb_disk_meta_t);
    1110        1455 :   ulong old_offset = fd_accdb_acc_xchg_offset( accmeta, FD_ACCDB_OFF_INVAL );
    1111        1455 :   if( FD_LIKELY( old_offset!=FD_ACCDB_OFF_INVAL ) ) {
    1112          36 :     fd_accdb_shmem_bytes_freed( accdb->shmem, old_offset, entry_sz );
    1113          36 :     FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->shmetrics->disk_used_bytes, entry_sz );
    1114          36 :   }
    1115        1455 :   FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->shmetrics->accounts_total, 1UL );
    1116        1455 :   accdb->metrics->accounts_deleted++;
    1117             : 
    1118        1455 :   if( FD_LIKELY( prev==UINT_MAX ) ) {
    1119             :     /* Head removal — CAS may fail if a concurrent insert prepended a
    1120             :        new node.  On failure the target is now interior. */
    1121          48 :     for(;;) {
    1122          48 :       uint old_head = FD_VOLATILE_CONST( accdb->acc_map[ map_idx ] );
    1123          48 :       if( FD_LIKELY( old_head==acc_idx ) ) {
    1124          48 :         if( FD_LIKELY( FD_ATOMIC_CAS( &accdb->acc_map[ map_idx ], acc_idx, accmeta->map.next )==acc_idx ) ) break;
    1125           0 :         FD_SPIN_PAUSE();
    1126           0 :         continue;
    1127          48 :       }
    1128             :       /* Head changed — walk from new head to find prev for interior
    1129             :          removal.  The target must still be in the chain because only
    1130             :          this thread removes elements. */
    1131           0 :       prev = old_head;
    1132           0 :       while( FD_VOLATILE_CONST( accdb->acc_pool[ prev ].map.next )!=acc_idx ) prev = FD_VOLATILE_CONST( accdb->acc_pool[ prev ].map.next );
    1133           0 :       FD_ATOMIC_CAS( &accdb->acc_pool[ prev ].map.next, acc_idx, accmeta->map.next );
    1134           0 :       break;
    1135          48 :     }
    1136        1407 :   } else {
    1137        1407 :     FD_ATOMIC_CAS( &accdb->acc_pool[ prev ].map.next, acc_idx, accmeta->map.next );
    1138        1407 :   }
    1139             : 
    1140        1455 :   fd_racesan_hook( "accdb_acc_unlink:post_splice" );
    1141             : 
    1142             :   /* If the freed acc still has a cached location, invalidate it and
    1143             :      try to reclaim the cache line so the eviction path does not try
    1144             :      to write back stale data from a recycled pool slot.  Lock-free:
    1145             :      CAS the refcnt 0 -> EVICT_SENTINEL to claim it exclusively, then
    1146             :      push to the CAS free list.  If the line is pinned (refcnt>0),
    1147             :      skip, the pinner's release will handle it.
    1148             : 
    1149             :      Acquire CACHE_CLAIM_BIT before touching acc->cache_idx /
    1150             :      CACHE_VALID — see evict_clear_acc_cache_ref for the protocol.
    1151             :      Without CLAIM, a concurrent cold_load_acc can publish a fresh
    1152             :      (cache_idx, VALID=1) pair into this acc between our two stores,
    1153             :      and our subsequent cache_idx=INVAL stomps onto the freelist
    1154             :      pool.next field (the union sibling of cache_idx), corrupting the
    1155             :      pool.  Unlike evict_clear_acc_cache_ref, we cannot bail when CLAIM
    1156             :      is held: this acc is being permanently unlinked, so we must
    1157             :      spin-wait for the cold-loader to release CLAIM and then invalidate
    1158             :      whatever cache_idx is current. */
    1159        1455 :   uint cur_es;
    1160        1455 :   for(;;) {
    1161        1455 :     cur_es = FD_VOLATILE_CONST( accmeta->executable_size );
    1162        1455 :     if( FD_UNLIKELY( cur_es & FD_ACCDB_SIZE_CACHE_CLAIM_BIT ) ) { FD_SPIN_PAUSE(); continue; }
    1163        1455 :     uint nxt_es = cur_es | FD_ACCDB_SIZE_CACHE_CLAIM_BIT;
    1164        1455 :     if( FD_LIKELY( FD_ATOMIC_CAS( &accmeta->executable_size, cur_es, nxt_es )==cur_es ) ) break;
    1165           0 :     FD_SPIN_PAUSE();
    1166           0 :   }
    1167             : 
    1168        1455 :   uint cidx = FD_ACCDB_ACC_CIDX_INVAL;
    1169        1455 :   int  had_valid = FD_ACCDB_SIZE_CACHE_VALID( cur_es );
    1170        1455 :   if( FD_UNLIKELY( had_valid ) ) {
    1171        1419 :     cidx = FD_VOLATILE_CONST( accmeta->cache_idx );
    1172             :     /* Clear VALID before INVAL'ing cache_idx — matches the order in
    1173             :        evict_clear_acc_cache_ref so cold_load_acc's "VALID=1 +
    1174             :        cidx=INVAL" spin path resolves on the next iteration when it
    1175             :        observes VALID=0. */
    1176        1419 :     FD_ATOMIC_FETCH_AND_AND( &accmeta->executable_size, ~FD_ACCDB_SIZE_CACHE_VALID_BIT );
    1177        1419 :     FD_VOLATILE( accmeta->cache_idx ) = FD_ACCDB_ACC_CIDX_INVAL;
    1178        1419 :   }
    1179             : 
    1180             :   /* Release CLAIM. */
    1181        1455 :   FD_ATOMIC_FETCH_AND_AND( &accmeta->executable_size, ~FD_ACCDB_SIZE_CACHE_CLAIM_BIT );
    1182             : 
    1183        1455 :   if( FD_UNLIKELY( had_valid ) ) {
    1184        1419 :     fd_accdb_cache_line_t * stale = cache_line( accdb, FD_ACCDB_ACC_CIDX_CLASS( cidx ), FD_ACCDB_ACC_CIDX_IDX( cidx ) );
    1185        1419 :     fd_racesan_hook( "acc_unlink:pre_reclaim_cas" );
    1186        1419 :     uint old_rc = FD_ATOMIC_CAS( &stale->refcnt, 0U, FD_ACCDB_EVICT_SENTINEL );
    1187        1419 :     fd_racesan_hook( "acc_unlink:post_reclaim_cas" );
    1188        1419 :     if( FD_LIKELY( !old_rc ) ) {
    1189             :       /* Claimed.  Validate key (ABA, slot could have been recycled
    1190             :          between our read of cache_idx and the CAS). */
    1191        1419 :       if( FD_LIKELY( stale->key.generation==accmeta->key.generation &&
    1192        1419 :                      !memcmp( stale->key.pubkey, accmeta->key.pubkey, 32UL ) ) ) {
    1193        1419 :         ulong sc = FD_ACCDB_ACC_CIDX_CLASS( cidx );
    1194        1419 :         stale->key.generation = UINT_MAX;
    1195        1419 :         stale->persisted = 1;
    1196        1419 :         stale->acc_idx   = UINT_MAX;
    1197        1419 :         FD_COMPILER_MFENCE();
    1198        1419 :         FD_VOLATILE( stale->refcnt ) = 0;
    1199        1419 :         cache_free_push( accdb, sc, stale );
    1200        1419 :       } else {
    1201             :         /* Wrong line (ABA).  Release claim. */
    1202           0 :         FD_VOLATILE( stale->refcnt ) = 0;
    1203           0 :       }
    1204        1419 :     }
    1205           0 :     else if( FD_LIKELY( old_rc!=FD_ACCDB_EVICT_SENTINEL ) ) {
    1206             :       /* The CAS lost to a non-sentinel refcnt, but that does not prove
    1207             :          `stale` is still our line.  Between capturing cidx and here we
    1208             :          released the claim, so we could have evicted `stale` and
    1209             :          recycled it to an unrelated account. */
    1210           0 :       fd_accdb_cache_line_t * mine = cache_try_pin( stale, accmeta->key.pubkey, accmeta->key.generation );
    1211           0 :       if( FD_LIKELY( mine ) ) {
    1212             :         /* Genuinely our line, still pinned by a reader.  The accmeta
    1213             :            slot is about to be deferred-released and recycled; if a
    1214             :            later writeback of this dirty line fires, it would pair the
    1215             :            recycled accmeta's pubkey with the old owner/data.  Set
    1216             :            persisted so the writeback gate never fires. */
    1217           0 :         FD_VOLATILE( mine->persisted ) = 1;
    1218             : 
    1219             :         /* Only the tombstone self-unlink may be pinned here old-version
    1220             :            and purge unlinks are never pinned, because a reader on a
    1221             :            live fork resolves to the newest version, not the one these
    1222             :            unlink. */
    1223           0 :         FD_TEST( accmeta->lamports==0UL );
    1224             : 
    1225           0 :         FD_ATOMIC_FETCH_AND_SUB( &mine->refcnt, 1U );
    1226           0 :       }
    1227             :       /* Else was recycled to a foreign account.  Nothing to neutralize,
    1228             :          leave the line alone. */
    1229           0 :     } else {
    1230             :       /* A foreground evictor already claimed this line.  It holds its
    1231             :          epoch acquire and writeback, so drain_deferred_frees cannot
    1232             :          recycle the slot before it finishes. Its writeback names the
    1233             :          old account correctly, no poison. */
    1234           0 :     }
    1235        1419 :   }
    1236        1455 : }
    1237             : 
    1238             : /* fork_slot_defer removes fork_id from every descends_set and chains
    1239             :    the fork pool slot onto the deferred fork chain for later release.
    1240             :    The slot must not be released immediately because concurrent readers
    1241             :    may still reference the fork ID via descends_set or stale chain
    1242             :    walks.
    1243             : 
    1244             :    The eager descends_set_remove here is safe despite being a
    1245             :    non-atomic RMW that races with concurrent descends_set_test in
    1246             :    fd_accdb_acquire, for two reasons:
    1247             : 
    1248             :    (a) Rooted parent forks: after advance_root publishes the new
    1249             :        root_fork_id, any acquire loads root_generation >=
    1250             :        parent->generation.  Every account from the old parent has
    1251             :        generation <= parent->generation, so the
    1252             :        "generation > root_generation" gate in the chain walk is
    1253             :        never satisfied and the parent's bit is never tested.
    1254             : 
    1255             :    (b) Purged / pruned sibling forks: a purged fork is by
    1256             :        definition not an ancestor of any live fork, so its bit
    1257             :        was never set in any live fork's descends_set.  Clearing
    1258             :        it is a literal no-op.
    1259             : 
    1260             :    Fork-id ABA after slot reuse is also safe: the fork pool slot
    1261             :    is not released until drain_deferred_frees, which waits until
    1262             :    all epoch-protected readers have exited.  On x86 (TSO), the
    1263             :    synchronization chain (T2: bit clear -> epoch FAA; reader:
    1264             :    epoch load -> epoch_slot store -> mfence -> bit read) guarantees
    1265             :    that any reader entering a new epoch section after the drain
    1266             :    will observe the cleared bit before the slot is recycled by
    1267             :    attach_child. */
    1268             : 
    1269             : static inline void
    1270             : fork_slot_defer( fd_accdb_t *              accdb,
    1271             :                  fd_accdb_fork_id_t         fork_id,
    1272             :                  fd_accdb_fork_shmem_t **   fork_head,
    1273         534 :                  fd_accdb_fork_shmem_t **   fork_tail ) {
    1274       11391 :   for( ulong i=0UL; i<accdb->shmem->max_live_slots; i++ ) descends_set_remove( accdb->fork_pool[ i ].descends, fork_id.val );
    1275         534 :   fd_accdb_fork_shmem_t * shmem = fork_pool_ele( accdb->fork_shmem_pool, (ulong)fork_id.val );
    1276         534 :   if( *fork_tail ) (*fork_tail)->pool.next = fork_pool_private_cidx( (ulong)fork_id.val );
    1277         522 :   else             *fork_head = shmem;
    1278         534 :   *fork_tail = shmem;
    1279         534 : }
    1280             : 
    1281             : static void
    1282             : purge_inner( fd_accdb_t *              accdb,
    1283             :              fd_accdb_fork_id_t         fork_id,
    1284             :              fd_accdb_fork_shmem_t **   fork_head,
    1285          33 :              fd_accdb_fork_shmem_t **   fork_tail ) {
    1286          33 :   fd_accdb_fork_t * fork = &accdb->fork_pool[ fork_id.val ];
    1287             : 
    1288          33 :   fd_accdb_fork_id_t child = fork->shmem->child_id;
    1289          42 :   while( child.val!=USHORT_MAX ) {
    1290           9 :     fd_accdb_fork_id_t next = accdb->fork_pool[ child.val ].shmem->sibling_id;
    1291           9 :     purge_inner( accdb, child, fork_head, fork_tail );
    1292           9 :     child = next;
    1293           9 :   }
    1294             : 
    1295          33 :   uint txn = fork->shmem->txn_head;
    1296          33 :   if( txn!=UINT_MAX ) {
    1297          30 :     fd_accdb_txn_t * txn_head = txn_pool_ele( accdb->txn_pool, (ulong)txn );
    1298          30 :     fd_accdb_txn_t * txn_tail = NULL;
    1299          81 :     while( txn!=UINT_MAX ) {
    1300          51 :       fd_accdb_txn_t * txne = txn_pool_ele( accdb->txn_pool, (ulong)txn );
    1301             : 
    1302          51 :       uint acc_idx     = txne->acc_pool_idx;
    1303          51 :       uint acc_map_idx = (uint)(fd_hash32( accdb->acc_pool[ acc_idx ].key.pubkey, accdb->shmem->seed ) &
    1304          51 :                                 (accdb->shmem->chain_cnt-1UL));
    1305             : 
    1306          51 :       uint prev = UINT_MAX;
    1307          51 :       uint cur = FD_VOLATILE_CONST( accdb->acc_map[ acc_map_idx ] );
    1308          54 :       while( cur!=acc_idx ) {
    1309           3 :         prev = cur;
    1310           3 :         cur = FD_VOLATILE_CONST( accdb->acc_pool[ cur ].map.next );
    1311           3 :       }
    1312             : 
    1313          51 :       fd_racesan_hook( "accdb_purge:pre_unlink" );
    1314          51 :       acc_unlink( accdb, acc_map_idx, prev, acc_idx );
    1315          51 :       deferred_acc_append( accdb, acc_idx );
    1316             : 
    1317          51 :       txn_tail = txne;
    1318          51 :       txn = txne->fork.next;
    1319          51 :     }
    1320          30 :     txn_pool_release_chain( accdb->txn_pool, txn_head, txn_tail );
    1321          30 :   }
    1322             : 
    1323          33 :   fork_slot_defer( accdb, fork_id, fork_head, fork_tail );
    1324          33 : }
    1325             : 
    1326             : static inline void
    1327             : remove_children( fd_accdb_t *              accdb,
    1328             :                  fd_accdb_fork_t *          fork,
    1329             :                  fd_accdb_fork_t *          except,
    1330             :                  fd_accdb_fork_shmem_t **   fork_head,
    1331         501 :                  fd_accdb_fork_shmem_t **   fork_tail ) {
    1332         501 :   fd_accdb_fork_id_t sibling_idx = fork->shmem->child_id;
    1333        1005 :   while( sibling_idx.val!=USHORT_MAX ) {
    1334         504 :     fd_accdb_fork_t * sibling = &accdb->fork_pool[ sibling_idx.val ];
    1335         504 :     fd_accdb_fork_id_t cur_idx = sibling_idx;
    1336             : 
    1337         504 :     sibling_idx = sibling->shmem->sibling_id;
    1338         504 :     if( FD_UNLIKELY( sibling==except ) ) continue;
    1339             : 
    1340           3 :     purge_inner( accdb, cur_idx, fork_head, fork_tail );
    1341           3 :   }
    1342         501 : }
    1343             : 
    1344             : static void
    1345             : background_advance_root( fd_accdb_t *       accdb,
    1346         501 :                          fd_accdb_fork_id_t fork_id ) {
    1347         501 :   drain_deferred_frees( accdb );
    1348             : 
    1349             :   /* The caller guarantees that rooting is sequential: each call
    1350             :      advances the root by exactly one slot (the immediate child of the
    1351             :      current root).  Skipping levels is not supported. */
    1352         501 :   fd_accdb_fork_t * fork = &accdb->fork_pool[ fork_id.val ];
    1353         501 :   FD_TEST( fork->shmem->parent_id.val==accdb->shmem->root_fork_id.val );
    1354         501 :   FD_TEST( fork->shmem->parent_id.val!=USHORT_MAX );
    1355             : 
    1356         501 :   fd_accdb_fork_t * parent_fork = &accdb->fork_pool[ fork->shmem->parent_id.val ];
    1357             : 
    1358             :   /* Accumulate freed fork pool slots across remove_children and the
    1359             :      old-version cleanup below into a chain that will be deferred-
    1360             :      released after the epoch bump.  Freed acc pool slots are recorded
    1361             :      in the shmem side buffer via deferred_acc_append (they cannot be
    1362             :      chained via pool.next yet — see comment on the side buffer). */
    1363         501 :   fd_accdb_fork_shmem_t * fork_head = NULL;
    1364         501 :   fd_accdb_fork_shmem_t * fork_tail = NULL;
    1365             : 
    1366             :   /* When a fork is rooted, any competing forks can be immediately
    1367             :      removed as they will not be needed again.  This includes child
    1368             :      forks of the pruned siblings as well. */
    1369         501 :   remove_children( accdb, parent_fork, fork, &fork_head, &fork_tail );
    1370             : 
    1371             :   /* And for any accounts which were updated in the newly rooted slot,
    1372             :      we will now never need to access any older version, so we can
    1373             :      discard any slots earlier than the one we are rooting. */
    1374         501 :   uint txn = fork->shmem->txn_head;
    1375         501 :   if( txn!=UINT_MAX ) {
    1376         498 :     fd_accdb_txn_t * txn_head = txn_pool_ele( accdb->txn_pool, (ulong)txn );
    1377         498 :     fd_accdb_txn_t * txn_tail = NULL;
    1378        1956 :     while( txn!=UINT_MAX ) {
    1379        1458 :       fd_accdb_txn_t * txne = txn_pool_ele( accdb->txn_pool, (ulong)txn );
    1380             : 
    1381        1458 :       fd_accdb_accmeta_t const * new_acc = &accdb->acc_pool[ txne->acc_pool_idx ];
    1382        1458 :       uint acc_map_idx = (uint)(fd_hash32( new_acc->key.pubkey, accdb->shmem->seed ) &
    1383        1458 :                                 (accdb->shmem->chain_cnt-1UL));
    1384             : 
    1385        1458 :       delta_insert( accdb, new_acc->key.pubkey );
    1386             : 
    1387        1458 :       uint prev          = UINT_MAX;
    1388        1458 :       uint new_acc_prev  = UINT_MAX; /* prev of new_acc on the chain when we encounter it (UINT_MAX if head or never seen) */
    1389        1458 :       int  new_acc_seen  = 0;
    1390        1458 :       uint acc = FD_VOLATILE_CONST( accdb->acc_map[ acc_map_idx ] );
    1391        1458 :       FD_TEST( acc!=UINT_MAX );
    1392        4644 :       while( acc!=UINT_MAX ) {
    1393        3186 :         fd_accdb_accmeta_t const * cur_acc = &accdb->acc_pool[ acc ];
    1394        3186 :         uint cur_next = FD_VOLATILE_CONST( cur_acc->map.next );
    1395             : 
    1396        3186 :         if( FD_LIKELY( acc==txne->acc_pool_idx ) ) {
    1397        1458 :           new_acc_prev = prev;
    1398        1458 :           new_acc_seen = 1;
    1399        1458 :           prev = acc;
    1400        1458 :           acc = cur_next;
    1401        1458 :           continue;
    1402        1458 :         }
    1403             : 
    1404        1728 :         if( FD_LIKELY( (cur_acc->key.generation<=parent_fork->shmem->generation || descends_set_test( fork->descends, fd_accdb_acc_fork_id(cur_acc) ) ) && !memcmp( new_acc->key.pubkey, cur_acc->key.pubkey, 32UL ) ) ) {
    1405        1404 :           uint next = cur_next;
    1406        1404 :           fd_racesan_hook( "accdb_advance:pre_unlink" );
    1407        1404 :           acc_unlink( accdb, acc_map_idx, prev, acc );
    1408        1404 :           deferred_acc_append( accdb, acc );
    1409        1404 :           acc = next;
    1410        1404 :         } else {
    1411         324 :           prev = acc;
    1412         324 :           acc = cur_next;
    1413         324 :         }
    1414        1728 :       }
    1415             : 
    1416             :       /* If the newly rooted version is a tombstone (lamports==0, e.g.
    1417             :          account was closed), drop it from the index too: no fork can
    1418             :          reach it anymore, and keeping it around just wastes a hash
    1419             :          slot and the disk bytes it occupies.
    1420             : 
    1421             :          If a later txn on this same fork wrote the same pubkey, that
    1422             :          txn's inner walk above would have already unlinked this txn's
    1423             :          new_acc as an "older version" - in that case new_acc_seen=0
    1424             :          and we skip, since the freelist cleanup is already done. */
    1425        1458 :       if( FD_UNLIKELY( new_acc_seen && new_acc->lamports==0UL ) ) {
    1426           0 :         uint new_acc_idx = (uint)txne->acc_pool_idx;
    1427           0 :         acc_unlink( accdb, acc_map_idx, new_acc_prev, new_acc_idx );
    1428           0 :         deferred_acc_append( accdb, new_acc_idx );
    1429           0 :       }
    1430             : 
    1431        1458 :       txn_tail = txne;
    1432        1458 :       txn = txne->fork.next;
    1433        1458 :     }
    1434         498 :     txn_pool_release_chain( accdb->txn_pool, txn_head, txn_tail );
    1435         498 :   }
    1436             : 
    1437         501 :   uint parent_txn = parent_fork->shmem->txn_head;
    1438         501 :   if( parent_txn!=UINT_MAX ) {
    1439          69 :     fd_accdb_txn_t * parent_head = txn_pool_ele( accdb->txn_pool, (ulong)parent_txn );
    1440          69 :     fd_accdb_txn_t * parent_tail = NULL;
    1441        1437 :     while( parent_txn!=UINT_MAX ) {
    1442        1368 :       fd_accdb_txn_t * t = txn_pool_ele( accdb->txn_pool, (ulong)parent_txn );
    1443        1368 :       parent_tail = t;
    1444        1368 :       parent_txn = t->fork.next;
    1445        1368 :     }
    1446          69 :     txn_pool_release_chain( accdb->txn_pool, parent_head, parent_tail );
    1447          69 :   }
    1448             : 
    1449             :   /* Remove the parent from all descends_sets and chain it for deferred
    1450             :      release, so that when the slot is eventually recycled to a new
    1451             :      fork, no concurrent reader can mistake the new fork for the old
    1452             :      ancestor.  Entries from the freed parent are still visible via the
    1453             :      generation <= root_generation fast path in reads. */
    1454         501 :   fd_accdb_fork_id_t old_parent_id = fork->shmem->parent_id;
    1455         501 :   fork_slot_defer( accdb, old_parent_id, &fork_head, &fork_tail );
    1456             : 
    1457         501 :   fork->shmem->parent_id  = (fd_accdb_fork_id_t){ .val = USHORT_MAX };
    1458         501 :   fork->shmem->sibling_id = (fd_accdb_fork_id_t){ .val = USHORT_MAX };
    1459         501 :   fork->shmem->txn_head   = UINT_MAX;
    1460         501 :   descends_set_null( fork->descends );
    1461             : 
    1462             :   /* Publish the new root_fork_id BEFORE bumping the epoch and deferring
    1463             :      the parent slot.  On x86-64 (TSO) a concurrent reader that still
    1464             :      loads the old root_fork_id is guaranteed to see the parent shmem in
    1465             :      its original (not-yet-recycled) state because the slot has not been
    1466             :      released yet.  A reader that loads the new root_fork_id uses the
    1467             :      new fork. */
    1468         501 :   fd_racesan_hook( "accdb_advance:pre_publish_root" );
    1469         501 :   accdb->shmem->root_fork_id = fork_id;
    1470         501 :   FD_COMPILER_MFENCE();
    1471         501 :   fd_racesan_hook( "accdb_advance:post_publish_root" );
    1472             : 
    1473             :   /* Bump epoch and defer both the acc batch and parent fork slot. They
    1474             :      will be released at the next drain_deferred_frees call once all
    1475             :      concurrent readers have exited.  The acc batch lives in the shmem
    1476             :      side buffer; only its epoch tag needs setting here. */
    1477         501 :   ulong tag = FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->epoch, 1UL );
    1478         501 :   if( FD_LIKELY( accdb->shmem->deferred_acc_buf_cnt ) ) {
    1479         495 :     accdb->shmem->deferred_acc_epoch = tag;
    1480         495 :   }
    1481         501 :   if( FD_LIKELY( fork_head ) ) {
    1482         501 :     accdb->deferred_fork_head  = fork_head;
    1483         501 :     accdb->deferred_fork_tail  = fork_tail;
    1484         501 :     accdb->deferred_fork_epoch = tag;
    1485         501 :   }
    1486         501 : }
    1487             : 
    1488             : void
    1489             : fd_accdb_advance_root( fd_accdb_t *       accdb,
    1490         501 :                        fd_accdb_fork_id_t fork_id ) {
    1491         501 :   FD_CHECK_CRIT( fd_accdb_snapshot_sync_state( &accdb->shmem->snapshot_sync )!=FD_ACCDB_SNAPSHOT_SYNC_RUNNING,
    1492         501 :                  "fd_accdb_advance_root called during snapshot production" );
    1493         501 :   wait_cmd( accdb );
    1494         501 :   submit_cmd( accdb, FD_ACCDB_CMD_ADVANCE_ROOT, fork_id.val );
    1495         501 : }
    1496             : 
    1497             : /* background_purge does the heavy lifting of purge on T2: unlink the
    1498             :    fork from the parent's child list, drain deferred frees, recursively
    1499             :    purge the fork subtree, and defer-release the freed acc pool
    1500             :    elements.  The sibling-list unlink is done here (not on T1) because
    1501             :    advance_root / remove_children also mutate sibling lists on T2, and
    1502             :    T2 is single-threaded so plain stores are safe. */
    1503             : 
    1504             : static void
    1505             : background_purge( fd_accdb_t *       accdb,
    1506          21 :                   fd_accdb_fork_id_t fork_id ) {
    1507             :   /* Unlink fork_id from its parent's child list.  This runs on T2
    1508             :      which is the sole mutator of sibling lists (advance_root and
    1509             :      remove_children also run on T2), so plain stores are safe. */
    1510          21 :   fd_accdb_fork_t * fork = &accdb->fork_pool[ fork_id.val ];
    1511          21 :   fd_accdb_fork_id_t parent_id = fork->shmem->parent_id;
    1512          21 :   if( FD_LIKELY( parent_id.val!=USHORT_MAX ) ) {
    1513          21 :     fd_accdb_fork_t * parent = &accdb->fork_pool[ parent_id.val ];
    1514          21 :     if( FD_UNLIKELY( parent->shmem->child_id.val==fork_id.val ) ) {
    1515          21 :       parent->shmem->child_id = fork->shmem->sibling_id;
    1516          21 :     } else {
    1517           0 :       fd_accdb_fork_id_t prev_id = parent->shmem->child_id;
    1518           0 :       while( prev_id.val!=USHORT_MAX ) {
    1519           0 :         fd_accdb_fork_t * prev = &accdb->fork_pool[ prev_id.val ];
    1520           0 :         if( prev->shmem->sibling_id.val==fork_id.val ) {
    1521           0 :           prev->shmem->sibling_id = fork->shmem->sibling_id;
    1522           0 :           break;
    1523           0 :         }
    1524           0 :         prev_id = prev->shmem->sibling_id;
    1525           0 :       }
    1526           0 :     }
    1527          21 :   }
    1528             : 
    1529          21 :   drain_deferred_frees( accdb );
    1530             : 
    1531          21 :   fd_accdb_fork_shmem_t * fork_head = NULL;
    1532          21 :   fd_accdb_fork_shmem_t * fork_tail = NULL;
    1533          21 :   purge_inner( accdb, fork_id, &fork_head, &fork_tail );
    1534             : 
    1535          21 :   ulong tag = FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->epoch, 1UL );
    1536          21 :   if( FD_LIKELY( accdb->shmem->deferred_acc_buf_cnt ) ) {
    1537          18 :     accdb->shmem->deferred_acc_epoch = tag;
    1538          18 :   }
    1539          21 :   if( FD_LIKELY( fork_head ) ) {
    1540          21 :     accdb->deferred_fork_head  = fork_head;
    1541          21 :     accdb->deferred_fork_tail  = fork_tail;
    1542          21 :     accdb->deferred_fork_epoch = tag;
    1543          21 :   }
    1544          21 : }
    1545             : 
    1546             : void
    1547             : fd_accdb_purge( fd_accdb_t *       accdb,
    1548          21 :                 fd_accdb_fork_id_t fork_id ) {
    1549          21 :   FD_TEST( fork_id.val!=accdb->shmem->root_fork_id.val );
    1550             : 
    1551          21 :   wait_cmd( accdb );
    1552          21 :   submit_cmd( accdb, FD_ACCDB_CMD_PURGE, fork_id.val );
    1553          21 : }
    1554             : 
    1555             : static inline fd_accdb_cache_line_t *
    1556             : acquire_cache_line( fd_accdb_t * accdb,
    1557             :                     ulong        size_class,
    1558      932526 :                     uint *       out_evicted_acc_idx ) {
    1559             :   /* Priority 1: CAS free list — already invalidated,
    1560             :      persisted==1, generation==UINT_MAX.  Cheapest path. */
    1561      932526 :   fd_accdb_cache_line_t * result = cache_free_pop( accdb, size_class );
    1562      932526 :   if( FD_LIKELY( result ) ) {
    1563      788325 :     while( FD_UNLIKELY( FD_ATOMIC_CAS( &result->refcnt, 0U, 1U )!=0U ) ) {
    1564           0 :       fd_racesan_hook( "accdb_freepop:refcnt_wait" );
    1565           0 :       FD_SPIN_PAUSE();
    1566           0 :     }
    1567      788325 :     result->referenced = 0;
    1568      788325 :     *out_evicted_acc_idx = UINT_MAX;
    1569      788325 :     return result;
    1570      788325 :   }
    1571             : 
    1572             :   /* Priority 2: Lazy initial allocation — atomic FAA with undo on
    1573             :      overflow.  Safe for concurrent callers. */
    1574      144201 :   ulong old_init = FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->cache_class_init[ size_class ].val, 1UL );
    1575      144201 :   if( FD_LIKELY( old_init<accdb->shmem->cache_class_max[ size_class ] ) ) {
    1576      144189 :     result = cache_line( accdb, size_class, old_init );
    1577      144189 :     result->refcnt         = 1;
    1578      144189 :     result->persisted      = 1;
    1579      144189 :     result->referenced     = 0;
    1580      144189 :     result->acc_idx        = UINT_MAX;
    1581      144189 :     result->key.generation = UINT_MAX;
    1582      144189 :     *out_evicted_acc_idx   = UINT_MAX;
    1583      144189 :     return result;
    1584      144189 :   }
    1585          12 :   FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->cache_class_init[ size_class ].val, 1UL );
    1586             : 
    1587             :   /* Priority 3: CLOCK sweep ... scan forward giving second chances. */
    1588          30 :   for(;;) {
    1589          30 :     ulong hand = FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->clock_hand[ size_class ].val, 1UL ) % accdb->shmem->cache_class_max[ size_class ];
    1590          30 :     fd_accdb_cache_line_t * line = cache_line( accdb, size_class, hand );
    1591             : 
    1592          30 :     if( FD_UNLIKELY( line->key.generation==UINT_MAX && line->acc_idx==UINT_MAX ) ) continue;
    1593             : 
    1594          30 :     fd_racesan_hook( "accdb_clock:pre_refcnt" );
    1595          30 :     uint rc = FD_VOLATILE_CONST( line->refcnt );
    1596          30 :     if( FD_UNLIKELY( rc!=0U ) ) continue; /* Pinned or being evicted */
    1597             : 
    1598          30 :     if( FD_UNLIKELY( line->referenced ) ) {
    1599          18 :       line->referenced = 0;
    1600          18 :       continue; /* Second chance */
    1601          18 :     }
    1602             : 
    1603          12 :     if( FD_UNLIKELY( FD_ATOMIC_CAS( &line->refcnt, 0U, FD_ACCDB_EVICT_SENTINEL )!=0U ) ) continue;
    1604             : 
    1605          12 :     if( FD_UNLIKELY( line->acc_idx==UINT_MAX && line->key.generation==UINT_MAX ) ) {
    1606           0 :       FD_VOLATILE( line->refcnt ) = 0;
    1607           0 :       continue;
    1608           0 :     }
    1609             : 
    1610             :     /* The line is now claimed for eviction (refcnt==EVICT_SENTINEL).  A
    1611             :        concurrent acc_unlink that targets this same line's accmeta will
    1612             :        observe the sentinel here and take its do-nothing branch — see the
    1613             :        test_accdb_racesan SENTINEL case. */
    1614          12 :     fd_racesan_hook( "clock_evict:post_sentinel" );
    1615             : 
    1616          12 :     if( FD_LIKELY( line->acc_idx!=UINT_MAX ) ) {
    1617          12 :       evict_clear_acc_cache_ref( &accdb->acc_pool[ line->acc_idx ], size_class, hand );
    1618          12 :     }
    1619          12 :     *out_evicted_acc_idx    = line->persisted ? UINT_MAX : line->acc_idx;
    1620          12 :     line->key.generation    = UINT_MAX;
    1621          12 :     line->refcnt            = 1;
    1622          12 :     line->referenced        = 0;
    1623          12 :     return line;
    1624          12 :   }
    1625             : 
    1626           0 :   FD_TEST( 0 );
    1627           0 :   return NULL;
    1628           0 : }
    1629             : 
    1630             : static inline void
    1631             : change_partition( fd_accdb_t *           accdb,
    1632             :                   accdb_offset_t const * offset_before,
    1633             :                   accdb_offset_t *       out_offset,
    1634             :                   int *                  has_partition,
    1635          63 :                   uchar                  layer ) {
    1636             :   /* New data will not fit in the current partition, so we need to
    1637             :      move to the next one.  */
    1638          63 :   ulong partition_idx_before = packed_partition_idx( offset_before );
    1639          63 :   ulong partition_offset_before = packed_partition_offset( offset_before );
    1640          63 :   if( FD_LIKELY( *has_partition ) ) {
    1641          39 :     fd_accdb_partition_t * before = partition_pool_ele( accdb->partition_pool, partition_idx_before );
    1642          39 :     before->write_offset = partition_offset_before;
    1643          39 :   }
    1644             : 
    1645             :   /* Single rdtsc per partition lifecycle event: stamp the closing
    1646             :      partition's filled time and the new partition's created time off
    1647             :      the same sample. */
    1648          63 :   long now_ticks = (long)fd_tickcount();
    1649             : 
    1650          63 :   ulong free_size = accdb->shmem->partition_sz - partition_offset_before;
    1651          63 :   if( FD_LIKELY( *has_partition ) ) {
    1652          39 :     fd_accdb_partition_t * old = partition_pool_ele( accdb->partition_pool, partition_idx_before );
    1653          39 :     FD_ATOMIC_FETCH_AND_ADD( &old->bytes_freed, free_size );
    1654          39 :     FD_VOLATILE( old->filled_ticks ) = now_ticks;
    1655             :     /* The tail slack is now committed dead — count it as current
    1656             :        (written-through) so fragmentation reflects it. */
    1657          39 :     FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->shmetrics->disk_current_bytes, free_size );
    1658          39 :   }
    1659             : 
    1660          63 :   if( FD_UNLIKELY( !partition_pool_free( accdb->partition_pool ) ) ) FD_LOG_ERR(( "accounts database file is at capacity" ));
    1661          63 :   fd_accdb_partition_t * partition = partition_pool_ele_acquire( accdb->partition_pool );
    1662          63 :   partition->bytes_freed       = 0UL;
    1663          63 :   partition->marked_compaction = 0;
    1664          63 :   partition->layer             = layer;
    1665          63 :   partition->read_ops          = 0UL;
    1666          63 :   partition->bytes_read        = 0UL;
    1667          63 :   partition->write_ops         = 0UL;
    1668          63 :   partition->bytes_written     = 0UL;
    1669          63 :   partition->write_offset      = 0UL;
    1670          63 :   partition->compaction_offset = 0UL;
    1671          63 :   partition->created_ticks     = now_ticks;
    1672          63 :   partition->filled_ticks      = 0L;
    1673          63 :   partition->queued            = 0;
    1674          63 :   partition->compacting_now    = 0;
    1675             : 
    1676          63 :   ulong new_partition_idx = partition_pool_idx( accdb->partition_pool, partition );
    1677          63 :   int had_partition = *has_partition;
    1678          63 :   *out_offset   = accdb_offset( new_partition_idx, 0UL );
    1679          63 :   FD_COMPILER_MFENCE();
    1680          63 :   *has_partition = 1;
    1681             : 
    1682             :   /* Now that the write head has been rotated away from the old
    1683             :      partition, check if it should be enqueued for compaction.  We call
    1684             :      try_enqueue directly because the caller already holds
    1685             :      partition_lock (calling fd_accdb_shmem_bytes_freed here would
    1686             :      deadlock on the non-reentrant lock).  Skip when
    1687             :      has_partition was 0, because the sentinel partition_idx is
    1688             :      not a valid pool element. */
    1689          63 :   if( FD_LIKELY( had_partition && partition_idx_before!=new_partition_idx ) ) {
    1690          39 :     fd_accdb_shmem_try_enqueue_compaction( accdb->shmem, partition_idx_before );
    1691          39 :   }
    1692             : 
    1693             :   /* Snapshot-load tiering: accounts loaded from a snapshot never get
    1694             :      a second write, so compaction-driven promotion never fires and
    1695             :      they would otherwise live in Hot forever.  When snapshot_loading
    1696             :      is set, tag the new partition as Cold up front.  We do not set
    1697             :      has_partition[Cold] / whead[Cold] — those are owned by the
    1698             :      compaction tile and represent the live Cold write head, which is
    1699             :      independent of snapshot-loaded partitions that happen to be
    1700             :      labeled Cold. */
    1701          63 :   if( FD_UNLIKELY( accdb->snapshot_loading && layer==0 ) ) {
    1702          51 :     FD_VOLATILE( partition->layer ) = FD_ACCDB_COMPACTION_LAYER_CNT-1UL;
    1703          51 :   }
    1704             : 
    1705          63 :   if( FD_UNLIKELY( new_partition_idx>=accdb->shmem->partition_max ) ) {
    1706          63 :     FD_LOG_INFO(( "growing accounts database from %lu GiB to %lu GiB", accdb->shmem->partition_max*accdb->shmem->partition_sz/(1UL<<30UL), (new_partition_idx+1UL)*accdb->shmem->partition_sz/(1UL<<30UL) ));
    1707             : 
    1708          63 :     int result = fallocate( accdb->fd, 0, (long)(new_partition_idx*accdb->shmem->partition_sz), (long)accdb->shmem->partition_sz );
    1709          63 :     if( FD_UNLIKELY( -1==result ) ) {
    1710           0 :       if( FD_LIKELY( errno==ENOSPC ) ) FD_LOG_ERR(( "fallocate() failed (%d-%s). The accounts database filled "
    1711           0 :                                                     "the disk it is on, trying to grow from %lu GiB to %lu GiB. Please "
    1712           0 :                                                     "free up disk space and restart the validator.",
    1713           0 :                                                     errno, fd_io_strerror( errno ), accdb->shmem->partition_max*accdb->shmem->partition_sz/(1UL<<30UL), (new_partition_idx+1UL)*accdb->shmem->partition_sz/(1UL<<30UL) ));
    1714           0 :       else FD_LOG_ERR(( "fallocate() failed (%d-%s)", errno, fd_io_strerror( errno ) ));
    1715           0 :     }
    1716             : 
    1717             :     /* CAS loop: the compaction tile may also be growing the file
    1718             :        concurrently, so neither path may clobber the other. */
    1719          63 :     for(;;) {
    1720          63 :       ulong cur = accdb->shmem->partition_max;
    1721          63 :       if( FD_LIKELY( new_partition_idx+1UL<=cur ) ) break;
    1722          63 :       if( FD_LIKELY( FD_ATOMIC_CAS( &accdb->shmem->partition_max, cur, new_partition_idx+1UL )==cur ) ) {
    1723          63 :         fd_event_accdb_partition_added_t ev = {
    1724          63 :           .partition_idx        = new_partition_idx,
    1725          63 :           .prior_partition_idx  = had_partition ? partition_idx_before : ULONG_MAX,
    1726          63 :           .layer                = layer,
    1727          63 :           .old_partition_max    = cur,
    1728          63 :           .new_partition_max    = new_partition_idx+1UL,
    1729          63 :           .partition_sz         = accdb->shmem->partition_sz,
    1730          63 :           .disk_allocated_bytes = (new_partition_idx+1UL)*accdb->shmem->partition_sz,
    1731          63 :         };
    1732          63 :         fd_event_report_accdb_partition_added( &ev );
    1733          63 :         break;
    1734          63 :       }
    1735          63 :     }
    1736          63 :     accdb->shmem->shmetrics->disk_allocated_bytes = accdb->shmem->partition_max*accdb->shmem->partition_sz;
    1737          63 :   }
    1738          63 : }
    1739             : 
    1740             : /* Reserve sz bytes in the layer-0 write head and set
    1741             :    out_partition_idx to where they landed.  Bumps no counters. */
    1742             : 
    1743             : static inline ulong
    1744             : reserve_next_write( fd_accdb_t * accdb,
    1745             :                     ulong        sz,
    1746          99 :                     ulong *      out_partition_idx ) {
    1747         138 :   for(;;) {
    1748         138 :     accdb_offset_t offset = { .val = FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->whead[ 0 ].val, sz ) };
    1749         138 :     if( FD_LIKELY( packed_partition_offset( &offset )+sz<=accdb->shmem->partition_sz ) ) {
    1750          99 :       *out_partition_idx = packed_partition_idx( &offset );
    1751          99 :       return packed_partition_file_offset( &offset, accdb->shmem->partition_sz );
    1752          99 :     }
    1753             : 
    1754          39 :     if( FD_UNLIKELY( packed_partition_offset( &offset )>accdb->shmem->partition_sz ) ) {
    1755             :       /* This can happen if another thread also raced to allocate the
    1756             :          next write and won.  Wait for the partition switch to finish
    1757             :          before retrying, so we do not keep doing fetch-and-adds that
    1758             :          advance the offset further past the boundary.
    1759             : 
    1760             :          A switch is detected by the head moving to a different
    1761             :          partition index OR its offset dropping back to a valid position
    1762             :          (a switch resets the offset to 0).  We must not key the wait
    1763             :          solely on the index changing: the initial write head is a
    1764             :          sentinel whose packed index can coincide with a real pool
    1765             :          index. */
    1766           0 :       ulong stale_partition = packed_partition_idx( &offset );
    1767           0 :       for(;;) {
    1768           0 :         accdb_offset_t cur = { .val = FD_VOLATILE_CONST( accdb->shmem->whead[ 0 ].val ) };
    1769           0 :         if( packed_partition_idx( &cur )!=stale_partition ) break;
    1770           0 :         if( packed_partition_offset( &cur )<=accdb->shmem->partition_sz ) break;
    1771           0 :         FD_SPIN_PAUSE();
    1772           0 :       }
    1773           0 :       continue;
    1774           0 :     }
    1775             : 
    1776          39 :     spin_lock_acquire( &accdb->shmem->partition_lock );
    1777          39 :     change_partition( accdb, &offset, &accdb->shmem->whead[ 0 ], &accdb->shmem->has_partition[ 0 ], 0 );
    1778          39 :     spin_lock_release( &accdb->shmem->partition_lock );
    1779          39 :   }
    1780          99 : }
    1781             : 
    1782             : /* Reserve sz bytes.  Layer-0 write metrics are deferred until
    1783             :    explicitly flushed. */
    1784             : 
    1785             : static inline ulong
    1786             : allocate_next_write( fd_accdb_t * accdb,
    1787          99 :                      ulong        sz ) {
    1788          99 :   ulong partition_idx;
    1789          99 :   ulong file_offset = reserve_next_write( accdb, sz, &partition_idx );
    1790             : 
    1791             :   /* Very rarely the reservation crosses into a new partition.  Since
    1792             :      stats are aggregated per-partition, any accumulated stats are
    1793             :      flushed and reset to accommodate the new partition. */
    1794          99 :   if( FD_LIKELY( accdb->write_stats.num_ops ) &&
    1795          99 :       FD_UNLIKELY( accdb->write_stats.partition_idx!=partition_idx ) ) {
    1796          15 :     fd_accdb_flush_metrics( accdb );
    1797          15 :   }
    1798             : 
    1799          99 :   accdb->write_stats.partition_idx  = partition_idx;
    1800          99 :   accdb->write_stats.bytes         += sz;
    1801          99 :   accdb->write_stats.num_ops++;
    1802          99 :   return file_offset;
    1803          99 : }
    1804             : 
    1805             : /* Compaction write allocation.  Single-threaded: only the compaction
    1806             :    tile calls these, so the compaction write heads do not need atomic
    1807             :    fetch-and-add.  dest_layer is the target layer (1..N-1). */
    1808             : 
    1809             : static inline ulong
    1810             : allocate_next_compaction_write( fd_accdb_t * accdb,
    1811             :                                 ulong        sz,
    1812           3 :                                 ulong        dest_layer ) {
    1813           3 :   accdb_offset_t offset = accdb->shmem->whead[ dest_layer ];
    1814           3 :   if( FD_UNLIKELY( !accdb->shmem->has_partition[ dest_layer ] ||
    1815           3 :                     packed_partition_offset( &offset )+sz>accdb->shmem->partition_sz ) ) {
    1816           3 :     spin_lock_acquire( &accdb->shmem->partition_lock );
    1817           3 :     change_partition( accdb, &offset, &accdb->shmem->whead[ dest_layer ], &accdb->shmem->has_partition[ dest_layer ], (uchar)dest_layer );
    1818           3 :     spin_lock_release( &accdb->shmem->partition_lock );
    1819           3 :     offset = accdb->shmem->whead[ dest_layer ];
    1820           3 :   }
    1821           3 :   accdb->shmem->whead[ dest_layer ].val += sz;
    1822           3 :   FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->shmetrics->disk_current_bytes, sz );
    1823           3 :   ulong file_offset = packed_partition_file_offset( &offset, accdb->shmem->partition_sz );
    1824           3 :   fd_accdb_partition_write_bump( accdb, packed_partition_idx( &offset ), sz, 1UL );
    1825           3 :   return file_offset;
    1826           3 : }
    1827             : 
    1828             : /* fd_accdb_compact relocates one record from the oldest partition
    1829             :    queued for compaction at src_layer into the write head for the
    1830             :    next colder tier, or the same tier for the deepest layer.  It is
    1831             :    designed to be called repeatedly from a dedicated compaction tile.
    1832             :    If there is work to do, *charge_busy is set to 1; otherwise 0 is
    1833             :    left unchanged and the call returns immediately.
    1834             : 
    1835             :    src_layer must be in 0..FD_ACCDB_COMPACTION_LAYER_CNT-1. */
    1836             : 
    1837             : static void
    1838             : background_compact( fd_accdb_t * accdb,
    1839             :                     ulong        src_layer,
    1840     9805245 :                     int *        charge_busy ) {
    1841     9805245 :   FD_COMPILER_MFENCE();
    1842     9805245 :   FD_VOLATILE( *accdb->my_epoch_slot ) = FD_VOLATILE_CONST( accdb->shmem->epoch );
    1843     9805245 :   FD_HW_MFENCE(); /* StoreLoad: epoch store must be globally visible
    1844             :                      before any subsequent loads so the deferred
    1845             :                      reclamation scan does not miss us. */
    1846             : 
    1847             :   /* Reclaim any deferred-free partitions whose epoch has been observed
    1848             :      by all joiners (i.e. no epoch-publishing joiner could still be
    1849             :      referencing data in them).  Scan writer slots [0, joiner_cnt)
    1850             :      plus each external (read-only) joiner's private epoch fseq. */
    1851     9805245 :   ulong min_epoch = ULONG_MAX;
    1852     9805245 :   ulong joiner_cnt = FD_VOLATILE_CONST( accdb->shmem->joiner_cnt );
    1853    24134193 :   for( ulong t=0UL; t<joiner_cnt; t++ ) {
    1854    14328948 :     ulong e = FD_VOLATILE_CONST( accdb->shmem->joiner_epochs[ t ].val );
    1855    14328948 :     if( FD_LIKELY( e<min_epoch ) ) min_epoch = e;
    1856    14328948 :   }
    1857     9805245 :   for( ulong t=0UL; t<accdb->external_epoch_cnt; t++ ) {
    1858           0 :     ulong e = FD_VOLATILE_CONST( *accdb->external_epoch_slots[ t ] );
    1859           0 :     if( FD_LIKELY( e<min_epoch ) ) min_epoch = e;
    1860           0 :   }
    1861     9805248 :   for(;;) {
    1862     9805248 :     if( FD_LIKELY( deferred_free_dlist_is_empty( accdb->deferred_free_dlist, accdb->partition_pool ) ) ) break;
    1863           3 :     fd_accdb_partition_t * p = deferred_free_dlist_ele_peek_head( accdb->deferred_free_dlist, accdb->partition_pool );
    1864           3 :     if( FD_LIKELY( p->epoch_tag>=min_epoch ) ) break;
    1865             : 
    1866           3 :     fd_racesan_hook( "accdb_reclaim:pre_free_partition" );
    1867             : 
    1868           3 :     spin_lock_acquire( &accdb->shmem->partition_lock );
    1869           3 :     deferred_free_dlist_ele_pop_head( accdb->deferred_free_dlist, accdb->partition_pool );
    1870           3 :     FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->shmetrics->disk_current_bytes, accdb->shmem->partition_sz );
    1871           3 :     FD_VOLATILE( p->write_offset ) = 0UL;
    1872           3 :     partition_pool_ele_release( accdb->partition_pool, p );
    1873           3 :     spin_lock_release( &accdb->shmem->partition_lock );
    1874           3 :   }
    1875             : 
    1876     9805245 :   if( FD_LIKELY( compaction_dlist_is_empty( accdb->compaction_dlist[ src_layer ], accdb->partition_pool ) ) ) {
    1877     9805236 :     FD_COMPILER_MFENCE();
    1878     9805236 :     FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
    1879     9805236 :     return;
    1880     9805236 :   }
    1881           9 :   fd_accdb_partition_t * compact = compaction_dlist_ele_peek_head( accdb->compaction_dlist[ src_layer ], accdb->partition_pool );
    1882           9 :   if( FD_UNLIKELY( !compact ) ) {
    1883           0 :     FD_COMPILER_MFENCE();
    1884           0 :     FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
    1885           0 :     return;
    1886           0 :   }
    1887             : 
    1888             :   /* Wait until all epoch-publishing joiners that were active when this
    1889             :      partition was enqueued for compaction have exited, ensuring any
    1890             :      in-flight pwritev2 to this partition has completed before we start
    1891             :      reading from it. */
    1892           9 :   if( FD_UNLIKELY( compact->compaction_ready_epoch>=min_epoch ) ) {
    1893           0 :     FD_COMPILER_MFENCE();
    1894           0 :     FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
    1895           0 :     return;
    1896           0 :   }
    1897             : 
    1898           9 :   *charge_busy = 1;
    1899             : 
    1900           9 :   if( FD_UNLIKELY( !compact->compacting_now ) ) {
    1901           3 :     compact->compaction_start_wallclock    = fd_log_wallclock();
    1902           3 :     compact->compaction_accounts_relocated = 0UL;
    1903           3 :     compact->compaction_bytes_relocated    = 0UL;
    1904           3 :     compact->compaction_dead_records       = 0UL;
    1905           3 :   }
    1906           9 :   FD_VOLATILE( compact->queued )         = 0;
    1907           9 :   FD_VOLATILE( compact->compacting_now ) = 1;
    1908             : 
    1909           9 :   fd_accdb_disk_meta_t meta[1];
    1910             : 
    1911           9 :   ulong compact_base = partition_pool_idx( accdb->partition_pool, compact )*accdb->shmem->partition_sz;
    1912             : 
    1913             :   /* Read the on-disk metadata header at the current compaction
    1914             :      cursor within the partition being compacted. */
    1915           9 :   ulong bytes_read = 0UL;
    1916          18 :   while( FD_UNLIKELY( bytes_read<sizeof(fd_accdb_disk_meta_t) ) ) {
    1917           9 :     long result = pread( accdb->fd, ((uchar *)meta)+bytes_read, sizeof(fd_accdb_disk_meta_t)-bytes_read, (long)(compact_base+compact->compaction_offset+bytes_read) );
    1918           9 :     if( FD_UNLIKELY( -1==result && (errno==EINTR || errno==EAGAIN || errno==EWOULDBLOCK ) ) ) continue;
    1919           9 :     else if( FD_UNLIKELY( -1==result ) ) FD_LOG_ERR(( "pread() failed (%d-%s)", errno, fd_io_strerror( errno ) ));
    1920           9 :     else if( FD_UNLIKELY( !result ) ) FD_LOG_ERR(( "accounts database is corrupt, data expected at offset %lu with size %lu exceeded file extents",
    1921           9 :                                                    compact_base+compact->compaction_offset+bytes_read, sizeof(fd_accdb_disk_meta_t) ));
    1922           9 :     fd_accdb_partition_read_bump( accdb, compact_base+compact->compaction_offset, (ulong)result );
    1923           9 :     bytes_read += (ulong)result;
    1924           9 :   }
    1925             : 
    1926             :   /* Walk the hash chain to find a live index entry whose on-disk
    1927             :      offset matches the record we are compacting. */
    1928           9 :   fd_accdb_accmeta_t * accmeta = NULL;
    1929           9 :   ulong source_packed = 0UL;
    1930           9 :   uint acc_idx = FD_VOLATILE_CONST( accdb->acc_map[ fd_hash32( meta->pubkey, accdb->shmem->seed )&(accdb->shmem->chain_cnt-1UL) ] );
    1931          15 :   while( acc_idx!=UINT_MAX ) {
    1932           9 :     fd_accdb_accmeta_t * candidate = &accdb->acc_pool[ acc_idx ];
    1933           9 :     uint next_idx = FD_VOLATILE_CONST( candidate->map.next );
    1934           9 :     ulong candidate_packed = FD_VOLATILE_CONST( candidate->offset_fork );
    1935           9 :     if( FD_LIKELY( (candidate_packed & FD_ACCDB_OFF_MASK)==compact_base+compact->compaction_offset ) ) {
    1936           3 :       accmeta       = candidate;
    1937           3 :       source_packed = candidate_packed;
    1938           3 :       break;
    1939           3 :     }
    1940           6 :     acc_idx = next_idx;
    1941           6 :   }
    1942             : 
    1943           9 :   ulong record_sz  = sizeof(fd_accdb_disk_meta_t) + (ulong)meta->size;
    1944           9 :   ulong bytes_copied = 0UL;
    1945           9 :   if( FD_UNLIKELY( !accmeta ) ) {
    1946             :     /* Dead record — the index entry was already removed, so this
    1947             :        on-disk extent is garbage.  Nothing to relocate. */
    1948           6 :     compact->compaction_dead_records++;
    1949           6 :   } else {
    1950           3 :     ulong dest_layer  = fd_ulong_min( src_layer+1UL, FD_ACCDB_COMPACTION_LAYER_CNT-1UL );
    1951           3 :     ulong dest_offset = allocate_next_compaction_write( accdb, record_sz, dest_layer );
    1952             : 
    1953           6 :     while( FD_UNLIKELY( bytes_copied<record_sz ) ) {
    1954           3 :       long in_off  = (long)(compact_base + compact->compaction_offset + bytes_copied);
    1955           3 :       long out_off = (long)(dest_offset + bytes_copied);
    1956             : 
    1957           3 :       long result = copy_file_range( accdb->fd, &in_off, accdb->fd, &out_off, record_sz-bytes_copied, 0 );
    1958           3 :       if( FD_UNLIKELY( -1==result && (errno==EINTR || errno==EAGAIN || errno==EWOULDBLOCK ) ) ) continue;
    1959           3 :       else if( FD_UNLIKELY( -1==result ) ) FD_LOG_ERR(( "copy_file_range() failed (%d-%s)", errno, fd_io_strerror( errno ) ));
    1960           3 :       else if( FD_UNLIKELY( !result ) ) FD_LOG_ERR(( "accounts database is corrupt, data expected at offset %lu with size %lu exceeded file extents",
    1961           3 :                                                       compact_base+compact->compaction_offset+bytes_copied, record_sz ));
    1962           3 :       fd_accdb_partition_read_bump( accdb, compact_base+compact->compaction_offset+bytes_copied, (ulong)result );
    1963           3 :       bytes_copied += (ulong)result;
    1964           3 :       accdb->metrics->copy_ops++;
    1965           3 :     }
    1966             : 
    1967           3 :     accdb->shmem->shmetrics->accounts_relocated++;
    1968           3 :     accdb->shmem->shmetrics->accounts_relocated_bytes += bytes_copied;
    1969           3 :     compact->compaction_accounts_relocated++;
    1970           3 :     compact->compaction_bytes_relocated += bytes_copied;
    1971             : 
    1972             :     /* Ensure the data is on disk before publishing the new offset,
    1973             :        so concurrent acquire threads do not preadv2 from a location
    1974             :        that hasn't been written yet. */
    1975           3 :     FD_COMPILER_MFENCE();
    1976             : 
    1977             :      /* CAS the offset from the exact source record we copied to the new
    1978             :        destination.  If a concurrent release overwrote the offset to
    1979             :        FD_ACCDB_OFF_INVAL (dirty sentinel for a new commit), or later
    1980             :        published a newer on-disk location, the CAS fails and we treat
    1981             :        the relocated copy as stale.  We CAS the full packed
    1982             :        offset_fork so the fork_id is preserved and so we only publish
    1983             :        the relocation if the copied source record is still current. */
    1984           3 :      ulong new_packed = ( source_packed & ~FD_ACCDB_OFF_MASK ) | ( dest_offset & FD_ACCDB_OFF_MASK );
    1985             : 
    1986             : #if FD_HAS_RACESAN
    1987             :      fd_memcpy( fd_accdb_dbg_reloc_pubkey, accmeta->key.pubkey, 32UL );
    1988             :      fd_accdb_dbg_reloc_dest = dest_offset;
    1989             :      fd_accdb_dbg_reloc_cnt++;
    1990             : #endif
    1991             : 
    1992           3 :      fd_racesan_hook( "accdb_compact:pre_offset_cas" );
    1993           3 :      if( FD_UNLIKELY( FD_ATOMIC_CAS( &accmeta->offset_fork, source_packed, new_packed )!=source_packed ) ) {
    1994             :       /* Record was superseded by a concurrent overwrite commit.
    1995             :          The disk space we just wrote is dead on arrival — account
    1996             :          it as freed so compaction can reclaim it later. */
    1997           0 :       fd_accdb_shmem_bytes_freed( accdb->shmem, dest_offset, record_sz );
    1998           0 :       bytes_copied = 0UL;
    1999           0 :     }
    2000           3 :   }
    2001             : 
    2002           9 :   fd_racesan_hook( "accdb_compact:post_relocate" );
    2003             : 
    2004           9 :   compact->compaction_offset += record_sz;
    2005             : 
    2006           9 :   if( FD_UNLIKELY( compact->compaction_offset>=compact->write_offset ) ) {
    2007           3 :     FD_LOG_INFO(( "compaction of partition %lu completed", partition_pool_idx( accdb->partition_pool, compact ) ));
    2008             : 
    2009           3 :     fd_event_accdb_compaction_completed_t ev = {
    2010           3 :       .partition_idx      = partition_pool_idx( accdb->partition_pool, compact ),
    2011           3 :       .src_layer          = (uchar)src_layer,
    2012           3 :       .dest_layer         = (uchar)fd_ulong_min( src_layer+1UL, FD_ACCDB_COMPACTION_LAYER_CNT-1UL ),
    2013           3 :       .bytes_scanned      = compact->write_offset,
    2014           3 :       .bytes_freed        = compact->bytes_freed,
    2015           3 :       .accounts_relocated = compact->compaction_accounts_relocated,
    2016           3 :       .bytes_relocated    = compact->compaction_bytes_relocated,
    2017           3 :       .dead_records       = compact->compaction_dead_records,
    2018           3 :       .start_time         = (ulong)compact->compaction_start_wallclock,
    2019           3 :       .end_time           = (ulong)fd_log_wallclock(),
    2020           3 :     };
    2021           3 :     fd_event_report_accdb_compaction_completed( &ev );
    2022             : 
    2023             :     /* Ensure the new acc->offset_fork stores above are visible to other
    2024             :        cores before the source partition is moved to the deferred-free
    2025             :        list.  On x86 (TSO) hardware store ordering already guarantees
    2026             :        this, but the compiler fence prevents the compiler from sinking
    2027             :        the offset store past the inlined pool/dlist mutations below. */
    2028           3 :     FD_COMPILER_MFENCE();
    2029             : 
    2030             :     /* Bump the global epoch and tag this partition so the reclamation
    2031             :        scan knows when all epoch-publishing joiners that could reference
    2032             :        data in this partition have exited. */
    2033           3 :     ulong tag = FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->epoch, 1UL );
    2034           3 :     compact->epoch_tag = tag;
    2035             : 
    2036             :     /* partition_lock serializes these dlist/pool mutations with
    2037             :        concurrent push_tail in fd_accdb_shmem_bytes_freed and
    2038             :        partition_pool_ele_acquire in change_partition.  Neither fd_dlist
    2039             :        nor fd_pool are thread-safe, so all mutations must be under the
    2040             :        same lock. */
    2041           3 :     spin_lock_acquire( &accdb->shmem->partition_lock );
    2042             : 
    2043           3 :     accdb->shmem->shmetrics->partitions_freed++;
    2044           3 :     compaction_dlist_ele_pop_head( accdb->compaction_dlist[ src_layer ], accdb->partition_pool );
    2045           3 :     FD_VOLATILE( compact->compacting_now ) = 0;
    2046           3 :     FD_VOLATILE( compact->queued )         = 0;
    2047           3 :     deferred_free_dlist_ele_push_tail( accdb->deferred_free_dlist, compact, accdb->partition_pool );
    2048             : 
    2049           3 :     accdb->shmem->shmetrics->compactions_completed++;
    2050           3 :     if( FD_LIKELY( compaction_dlist_is_empty( accdb->compaction_dlist[ src_layer ], accdb->partition_pool ) ) ) {
    2051           3 :       accdb->shmem->shmetrics->in_compaction = 0;
    2052           3 :     } else {
    2053           0 :       fd_accdb_partition_t * next = compaction_dlist_ele_peek_head( accdb->compaction_dlist[ src_layer ], accdb->partition_pool );
    2054           0 :       FD_LOG_INFO(( "compaction of layer %lu partition %lu started", src_layer, partition_pool_idx( accdb->partition_pool, next ) ));
    2055           0 :     }
    2056             : 
    2057           3 :     spin_lock_release( &accdb->shmem->partition_lock );
    2058           3 :   }
    2059             : 
    2060           9 :   accdb->metrics->bytes_read += bytes_read + bytes_copied;
    2061           9 :   accdb->metrics->bytes_written += bytes_copied;
    2062             : 
    2063           9 :   FD_COMPILER_MFENCE();
    2064           9 :   FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
    2065           9 : }
    2066             : 
    2067             : /* cold_load_acc resolves the cache slot for `acc` when STEP 1's
    2068             :    cache_try_pin failed.  It uses bit 29 of executable_size as a
    2069             :    single-claimer lock so that two concurrent acquirers cannot each
    2070             :    install their own cache slot for the same acc (which would orphan
    2071             :    one slot with a dangling line->acc_idx and eventually corrupt
    2072             :    acc->cache_valid via CLOCK).
    2073             : 
    2074             :    Protocol per acc:
    2075             :      - If cache_valid is set, retry cache_try_pin (another thread
    2076             :        finished the cold-load while we were here).  On success, mark
    2077             :        exists_in_cache so STEP 4 will not write back the slot.
    2078             :      - If claim is set, spin (another thread is mid-cold-load).
    2079             :      - Otherwise CAS-set the claim bit.  Winner allocates a cache
    2080             :        line, populates the placeholder (acc_idx=UINT_MAX), publishes
    2081             :        cache_idx, then atomically (CAS-loop) sets cache_valid and
    2082             :        clears claim.
    2083             : 
    2084             :    The eviction sites that clear cache_valid must use FETCH_AND with
    2085             :    ~CACHE_VALID_BIT (preserving the claim bit) to interact correctly
    2086             :    with this protocol. */
    2087             : 
    2088             : static fd_accdb_cache_line_t *
    2089             : cold_load_acc( fd_accdb_t *     accdb,
    2090             :                fd_accdb_accmeta_t * accmeta,
    2091             :                uchar const *    pubkey,
    2092             :                int *            out_exists_in_cache,
    2093          30 :                uint *           out_evicted_acc_idx ) {
    2094          30 :   for(;;) {
    2095          30 :     uint old_es  = FD_VOLATILE_CONST( accmeta->executable_size );
    2096          30 :     int  valid   = FD_ACCDB_SIZE_CACHE_VALID( old_es );
    2097          30 :     int  claimed = FD_ACCDB_SIZE_CACHE_CLAIM( old_es );
    2098             : 
    2099          30 :     if( FD_UNLIKELY( valid ) ) {
    2100             :       /* old_es snapshot saw VALID=1 but a concurrent
    2101             :          evict_clear_acc_cache_ref may have cleared VALID and stored
    2102             :          cache_idx=INVAL between our snapshot and this load.  Decoding
    2103             :          INVAL would yield a wild cache_line pointer; retry the loop
    2104             :          instead (next iteration will see VALID=0). */
    2105           0 :       uint cidx = FD_VOLATILE_CONST( accmeta->cache_idx );
    2106           0 :       if( FD_UNLIKELY( cidx==FD_ACCDB_ACC_CIDX_INVAL ) ) { FD_SPIN_PAUSE(); continue; }
    2107           0 :       fd_accdb_cache_line_t * hit = cache_line( accdb, FD_ACCDB_ACC_CIDX_CLASS( cidx ), FD_ACCDB_ACC_CIDX_IDX( cidx ) );
    2108           0 :       fd_racesan_hook( "accdb_cold_load:pre_try_pin" );
    2109           0 :       fd_accdb_cache_line_t * pinned = cache_try_pin( hit, pubkey, accmeta->key.generation );
    2110           0 :       if( FD_LIKELY( pinned ) ) {
    2111           0 :         *out_exists_in_cache  = 1;
    2112           0 :         *out_evicted_acc_idx  = UINT_MAX;
    2113           0 :         return pinned;
    2114           0 :       }
    2115           0 :       FD_SPIN_PAUSE();
    2116           0 :       continue;
    2117           0 :     }
    2118             : 
    2119          30 :     if( FD_UNLIKELY( claimed ) ) {
    2120           0 :       fd_racesan_hook( "accdb_cold_load:claim_wait" );
    2121           0 :       FD_SPIN_PAUSE();
    2122           0 :       continue;
    2123           0 :     }
    2124             : 
    2125          30 :     if( FD_UNLIKELY( FD_ATOMIC_CAS( &accmeta->executable_size, old_es, old_es | FD_ACCDB_SIZE_CACHE_CLAIM_BIT )!=old_es ) ) {
    2126           0 :       FD_SPIN_PAUSE();
    2127           0 :       continue;
    2128           0 :     }
    2129             : 
    2130             :     /* We hold the claim.  Allocate a cache line and publish. */
    2131          30 :     ulong size_class = fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( old_es ) );
    2132          30 :     fd_accdb_cache_line_t * line = acquire_cache_line( accdb, size_class, out_evicted_acc_idx );
    2133          30 :     fd_memcpy( line->key.pubkey, accmeta->key.pubkey, 32UL );
    2134          30 :     line->key.generation = accmeta->key.generation;
    2135             :     /* Leave acc_idx at UINT_MAX (the "loading" sentinel) until step 12
    2136             :        publishes it after the preadv2 fence.  Concurrent threads that
    2137             :        pin via cache_idx will spin on this in step 13. */
    2138          30 :     line->acc_idx = UINT_MAX;
    2139          30 :     FD_COMPILER_MFENCE();
    2140          30 :     FD_VOLATILE( accmeta->cache_idx ) = FD_ACCDB_ACC_CIDX_PACK( (uint)size_class, (uint)cache_line_idx( accdb, size_class, line ) );
    2141          30 :     FD_COMPILER_MFENCE();
    2142             : 
    2143          30 :     fd_racesan_hook( "accdb_cold_load:pre_valid" );
    2144             : 
    2145             :     /* Atomically set CACHE_VALID_BIT and clear CACHE_CLAIM_BIT.
    2146             :        Eviction may have flipped CACHE_VALID_BIT on us between our
    2147             :        claim and now (it preserves CLAIM but can clear VALID); the
    2148             :        CAS loop tolerates that.  The data length and exec bits stay
    2149             :        unchanged. */
    2150          30 :     for(;;) {
    2151          30 :       uint cur = FD_VOLATILE_CONST( accmeta->executable_size );
    2152          30 :       uint nxt = (cur & ~FD_ACCDB_SIZE_CACHE_CLAIM_BIT) | FD_ACCDB_SIZE_CACHE_VALID_BIT;
    2153          30 :       if( FD_LIKELY( FD_ATOMIC_CAS( &accmeta->executable_size, cur, nxt )==cur ) ) break;
    2154           0 :       FD_SPIN_PAUSE();
    2155           0 :     }
    2156             : 
    2157          30 :     *out_exists_in_cache = 0;
    2158          30 :     return line;
    2159          30 :   }
    2160          30 : }
    2161             : 
    2162      311052 : #define RESERVATION_TYPE_SIMPLE            (0)
    2163         351 : #define RESERVATION_TYPE_MAYBE_PROGRAMDATA (1)
    2164         351 : #define RESERVATION_TYPE_ALREADY_RESERVED  (2)
    2165             : 
    2166             : static void
    2167             : fd_accdb_acquire_inner( fd_accdb_t *          accdb,
    2168             :                         fd_accdb_fork_id_t    fork_id,
    2169             :                         int                   reservation_type,
    2170             :                         ulong                 reserved_cnt,
    2171             :                         ulong                 pubkeys_cnt,
    2172             :                         uchar const * const * pubkeys,
    2173             :                         int *                 writable,
    2174      311754 :                         fd_acc_t *            out_accs ) {
    2175      311754 :   accdb->metrics->acquire_calls++;
    2176             : 
    2177      311754 :   ulong max_acquire_cnt = accdb->shmem->bundle_enabled ? FD_ACCDB_MAX_ACQUIRE_CNT : FD_ACCDB_MAX_TX_ACCOUNT_LOCKS;
    2178      311754 :   FD_TEST( pubkeys_cnt<=max_acquire_cnt );
    2179             : 
    2180      311754 :   FD_TEST( FD_VOLATILE_CONST( *accdb->my_epoch_slot )==ULONG_MAX );
    2181             : 
    2182      311754 :   FD_COMPILER_MFENCE();
    2183      311754 :   FD_VOLATILE( *accdb->my_epoch_slot ) = FD_VOLATILE_CONST( accdb->shmem->epoch );
    2184      311754 :   FD_HW_MFENCE(); /* StoreLoad: epoch store must be globally visible
    2185             :                      before any subsequent loads so the deferred
    2186             :                      reclamation scan does not miss us */
    2187             : 
    2188             :   // STEP 1.
    2189             :   //   Locate each account in the fork and index structure, to determine
    2190             :   //   if it already exists, its size and other metadata, and which
    2191             :   //   specific slot (generation) it was last written in.
    2192             : 
    2193      311754 :   fd_accdb_fork_t * fork = &accdb->fork_pool[ fork_id.val ];
    2194      311754 :   uint root_generation = accdb->fork_pool[ accdb->shmem->root_fork_id.val ].shmem->generation;
    2195             : 
    2196      311754 :   fd_racesan_hook( "accdb_acquire:post_root_gen" );
    2197             : 
    2198      311754 :   fd_accdb_accmeta_t * accmetas[ FD_ACCDB_MAX_ACQUIRE_CNT ];
    2199      311754 :   ulong acc_map_idxs[ FD_ACCDB_MAX_ACQUIRE_CNT ];
    2200             : 
    2201             :   /* Walk the hash chain for each pubkey and take the first visible
    2202             :      match.  Correctness relies on newer entries always being prepended
    2203             :      to the chain head, which is guaranteed because replay processes
    2204             :      writes in slot order and release always inserts at the head.
    2205             : 
    2206             :      CONCURRENCY: This chain walk runs epoch-protected.  A concurrent
    2207             :      fd_accdb_release may prepend a new node to the same chain while
    2208             :      we walk it.  This is safe on x86-64 (TSO): the releasing thread
    2209             :      stores all acc fields (pubkey, generation, map.next, ...) before
    2210             :      publishing the new head via a CAS on acc_map[idx], and TSO
    2211             :      guarantees a reading core that observes the new head also observes
    2212             :      all prior stores to the node.  A reader that does not yet see the
    2213             :      new head simply sees an older (still valid) version of the chain.
    2214             :      On weakly-ordered architectures an explicit acquire fence would be
    2215             :      needed before the chain walk and a release fence in
    2216             :      fd_accdb_release before the head-pointer store.  Multiple
    2217             :      concurrent releases serialize on the CAS of the chain head. */
    2218      624351 :   for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
    2219      312597 :     acc_map_idxs[ i ] = fd_hash32( pubkeys[ i ], accdb->shmem->seed )&(accdb->shmem->chain_cnt-1UL);
    2220      312597 :     uint acc = FD_VOLATILE_CONST( accdb->acc_map[ acc_map_idxs[ i ] ] );
    2221      389556 :     while( acc!=UINT_MAX ) {
    2222      174531 :       fd_accdb_accmeta_t const * candidate_acc = &accdb->acc_pool[ acc ];
    2223      174531 :       uint next_acc = FD_VOLATILE_CONST( candidate_acc->map.next );
    2224             : 
    2225      174531 :       fd_racesan_hook( "accdb_acquire:post_next" );
    2226             : 
    2227      174531 :       if( FD_UNLIKELY( (candidate_acc->key.generation>root_generation &&
    2228      174531 :                         fd_accdb_acc_fork_id(candidate_acc)!=fork_id.val &&
    2229      174531 :                         !descends_set_test( fork->descends, fd_accdb_acc_fork_id(candidate_acc) )) ) ||
    2230      174531 :                         memcmp( pubkeys[ i ], candidate_acc->key.pubkey, 32UL ) ) {
    2231       76959 :         acc = next_acc;
    2232       76959 :         continue;
    2233       76959 :       }
    2234             : 
    2235       97572 :       break;
    2236      174531 :     }
    2237      312597 :     if( FD_UNLIKELY( acc==UINT_MAX ) )                                       accmetas[ i ] = NULL;
    2238       97572 :     else                                                                     accmetas[ i ] = &accdb->acc_pool[ acc ];
    2239             : 
    2240             : #if FD_TMPL_USE_HANDHOLDING
    2241             :     if( FD_UNLIKELY( accmetas[ i ] ) ) {
    2242             :       fd_accdb_accmeta_t const * sel = accmetas[ i ];
    2243             :       FD_TEST( !memcmp( sel->key.pubkey, pubkeys[ i ], 32UL ) );
    2244             :       FD_TEST( sel->key.generation<=root_generation ||
    2245             :                fd_accdb_acc_fork_id( sel )==fork_id.val ||
    2246             :                descends_set_test( fork->descends, fd_accdb_acc_fork_id( sel ) ) );
    2247             :       FD_TEST( sel->key.generation<=FD_VOLATILE_CONST( accdb->shmem->generation ) );
    2248             :     }
    2249             : #endif
    2250             : 
    2251      312597 :     if( FD_UNLIKELY( accmetas[ i ] && !writable[ i ] && !accmetas[ i ]->lamports ) ) accmetas[ i ] = NULL;
    2252             : 
    2253             :     /* Attribute this acquired account to a size class for per-class
    2254             :        rate metrics.  Use the account's current size class when known;
    2255             :        otherwise (new account) bucket as class 0. */
    2256      312597 :     ulong acq_class = 0UL;
    2257      312597 :     if( FD_LIKELY( accmetas[ i ] ) ) acq_class = fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size ) );
    2258      312597 :     if( FD_LIKELY( writable[ i ] ) ) accdb->metrics->writable_accounts_acquired_per_class[ acq_class ]++;
    2259      196035 :     else                             accdb->metrics->accounts_acquired_per_class[ acq_class ]++;
    2260      312597 :   }
    2261             : 
    2262             :   // STEP 2.
    2263             :   //   The two-phase programdata acquire (acquire_a then acquire_b)
    2264             :   //   works as follows: acquire_a (RESERVATION_TYPE_MAYBE_PROGRAMDATA)
    2265             :   //   over-reserves one slot in every live size class per candidate
    2266             :   //   account (reserved_cnt total per class), because it does not yet
    2267             :   //   know which accounts have programdata or what size class it lands
    2268             :   //   in.  acquire_b then resolves the actual programdata pubkeys and
    2269             :   //   re-enters here with RESERVATION_TYPE_ALREADY_RESERVED to refund
    2270             :   //   the surplus.  Keep one reservation per found programdata account
    2271             :   //   in its own size class (consumed later by release) and give the
    2272             :   //   rest back.
    2273      311754 :   if( FD_UNLIKELY( reservation_type==RESERVATION_TYPE_ALREADY_RESERVED ) ) {
    2274         351 :     ulong refund[ FD_ACCDB_CACHE_CLASS_CNT ] = {0};
    2275        3159 :     for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) {
    2276        2808 :       if( FD_LIKELY( accdb->shmem->cache_class_used[ j ].val!=ULONG_MAX ) ) refund[ j ] = reserved_cnt;
    2277        2808 :     }
    2278         390 :     for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
    2279          39 :       if( FD_LIKELY( accmetas[ i ] ) ) {
    2280          36 :         ulong cls = fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size ) );
    2281          36 :         if( FD_LIKELY( accdb->shmem->cache_class_used[ cls ].val!=ULONG_MAX ) ) {
    2282           3 :           FD_TEST( refund[ cls ]>0UL );
    2283           3 :           refund[ cls ]--;
    2284           3 :         }
    2285          36 :       }
    2286          39 :     }
    2287        3159 :     for( ulong k=0UL; k<FD_ACCDB_CACHE_CLASS_CNT; k++ ) {
    2288        2808 :       if( FD_UNLIKELY( refund[ k ] ) ) FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->cache_class_used[ k ].val, refund[ k ] );
    2289        2808 :     }
    2290         351 :   }
    2291             : 
    2292             :   // STEP 3.
    2293             :   //   We are potentially going to need to read the account data off of
    2294             :   //   disk into the cache, if the account(s) are not in the cache so
    2295             :   //   reserve the necessary cache space.  This is done with an "atomic
    2296             :   //   subtract" spin loop on the cache class counters, which is
    2297             :   //   actually faster than doing a real CAS on a packed ulong.
    2298             :   //
    2299             :   //   For reads, we only need space to copy the account data into a
    2300             :   //   single right-sized cache line, but for writes ... we need to
    2301             :   //   reserve one of every size class.  The reason is we are going to
    2302             :   //   need a 10MiB staging buffer for the executor to write to (it may
    2303             :   //   grow the account, so needs the max size class).  Even if the
    2304             :   //   account is already in the 10MiB cache class, we need another one
    2305             :   //   because a transaction can fail half way, so we need scratch space
    2306             :   //   to be able to unwind.
    2307             :   //
    2308             :   //   So we acquire one of each size class.  Then when the transaction
    2309             :   //   finishes, if it succeeded, we will copy the data back to the
    2310             :   //   whichever size-class is now right-sized post execution.
    2311      311754 :   if( FD_LIKELY( reservation_type==RESERVATION_TYPE_SIMPLE || reservation_type==RESERVATION_TYPE_MAYBE_PROGRAMDATA ) ) {
    2312      311403 :     ulong requested_buckets[ FD_ACCDB_CACHE_CLASS_CNT ] = {0};
    2313      623961 :     for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
    2314      312558 :       if( FD_LIKELY( accmetas[ i ] || writable[ i ] ) ) {
    2315      199203 :         if( FD_LIKELY( accmetas[ i ] ) ) {
    2316       97536 :           if( FD_UNLIKELY( accdb->shmem->cache_class_used[ fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size ) ) ].val!=ULONG_MAX ) ) {
    2317           0 :             requested_buckets[ fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size ) ) ]++;
    2318           0 :           }
    2319       97536 :         }
    2320      199203 :         if( FD_UNLIKELY( writable[ i ] ) ) {
    2321     1049058 :           for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) {
    2322      932496 :             if( FD_UNLIKELY( accdb->shmem->cache_class_used[ j ].val!=ULONG_MAX ) ) {
    2323          54 :               requested_buckets[ j ]++;
    2324          54 :             }
    2325      932496 :           }
    2326      116562 :         }
    2327      199203 :       }
    2328             : 
    2329      312558 :       if( FD_LIKELY( reservation_type==RESERVATION_TYPE_MAYBE_PROGRAMDATA ) ) {
    2330             :         /* Any account could also have an implied reference to a
    2331             :           programdata account, which we don't know yet ... so we need to
    2332             :           reserve worst case space if they all went to the same size
    2333             :           class.  This reservation runs unconditionally per pubkey (not
    2334             :           gated on accmetas/writable) so that acquire_b can refund based on
    2335             :           pubkeys_cnt without needing to re-derive the live-account set. */
    2336       13284 :         for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) {
    2337       11808 :           if( FD_UNLIKELY( accdb->shmem->cache_class_used[ j ].val!=ULONG_MAX ) ) {
    2338          36 :             requested_buckets[ j ]++;
    2339          36 :           }
    2340       11808 :         }
    2341        1476 :       }
    2342      312558 :     }
    2343             : 
    2344             :     /* TODO: This over-reserves cache slots for writable accounts that
    2345             :        already exist.  For each such account we reserve one line in the
    2346             :        account's size class (for the read into cache) AND one line in
    2347             :        every size class (for the write destination buffers). But if the
    2348             :        account is already resident in cache (which is the common case
    2349             :        for hot accounts), the read-into-cache line is unnecessary — we
    2350             :        will get a cache hit in step 4 and never use it.  The fix is to
    2351             :        probe acc->cache_idx here and skip the per-account size class
    2352             :        reservation per-account size class reservation when a hit is
    2353             :        found. This would reduce peak reservation by up to one line per
    2354             :        writable account per acquire batch, lowering contention on the
    2355             :        cache class counters and allowing smaller cache provisioning. */
    2356             : 
    2357             :     /* Reserve cache slots by atomically incrementing the shared used
    2358             :        counters.  If any class exceeds its max, the reservation
    2359             :        overflowed — subtract back partial grabs and retry. */
    2360      311403 :     for(;;) {
    2361      311403 :       int acquire_failed = 0;
    2362      311403 :       ulong grabbed[ FD_ACCDB_CACHE_CLASS_CNT ] = {0};
    2363     2802627 :       for( ulong i=0UL; i<FD_ACCDB_CACHE_CLASS_CNT; i++ ) {
    2364     2491224 :         if( FD_LIKELY( !requested_buckets[ i ] ) ) continue;
    2365          72 :         ulong new_used = FD_ATOMIC_ADD_AND_FETCH( &accdb->shmem->cache_class_used[ i ].val, requested_buckets[ i ] );
    2366          72 :         if( FD_UNLIKELY( new_used>accdb->shmem->cache_class_max[ i ] ) ) {
    2367           0 :           FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->cache_class_used[ i ].val, requested_buckets[ i ] );
    2368           0 :           acquire_failed = 1;
    2369          72 :         } else {
    2370          72 :           grabbed[ i ] = requested_buckets[ i ];
    2371          72 :         }
    2372          72 :         if( FD_UNLIKELY( acquire_failed ) ) {
    2373           0 :           accdb->metrics->acquire_failed++;
    2374           0 :           for( ulong j=0UL; j<i; j++ ) {
    2375           0 :             if( grabbed[ j ] ) FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->cache_class_used[ j ].val, grabbed[ j ] );
    2376           0 :           }
    2377           0 :           FD_SPIN_PAUSE();
    2378           0 :           break;
    2379           0 :         }
    2380          72 :       }
    2381      311403 :       if( FD_LIKELY( !acquire_failed ) ) break;
    2382      311403 :     }
    2383      311403 :   }
    2384             : 
    2385             :   // STEP 4.
    2386             :   //   For any accounts that are not in cache, we now need to actually
    2387             :   //   retrieve the cache pointers from our structures.  Space has been
    2388             :   //   reserved already, so this step is guaranteed to succeed, and is
    2389             :   //   just pulling the cache lines out of the free lists and marking
    2390             :   //   them as in-use.
    2391             :   //
    2392             :   //   This step is fully lock-free.  Cache hits are pinned with an
    2393             :   //   atomic CAS on refcnt (cache_try_pin).  Eviction uses the CLOCK
    2394             :   //   algorithm.  The CAS free list provides immediate recycling of
    2395             :   //   fully-freed lines.
    2396             : 
    2397      311754 :   int exists_in_cache[ FD_ACCDB_MAX_ACQUIRE_CNT ];
    2398      311754 :   fd_accdb_cache_line_t * original_cache_line[ FD_ACCDB_MAX_ACQUIRE_CNT ];
    2399      311754 :   fd_accdb_cache_line_t * destination_cache_lines[ FD_ACCDB_MAX_ACQUIRE_CNT ][ FD_ACCDB_CACHE_CLASS_CNT ];
    2400             : 
    2401             :   /* Saved acc_pool indices of evicted dirty cache lines.  These are
    2402             :      captured before clearing acc_idx to UINT_MAX on the line struct, so
    2403             :      that the sentinel protocol (step 14) works correctly while the
    2404             :      evicted account metadata is still available for writeback in steps
    2405             :      4 and 6. */
    2406      311754 :   uint evicted_dest_acc[ FD_ACCDB_MAX_ACQUIRE_CNT ][ FD_ACCDB_CACHE_CLASS_CNT ];
    2407      311754 :   uint evicted_orig_acc[ FD_ACCDB_MAX_ACQUIRE_CNT ];
    2408             : 
    2409      624351 :   for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
    2410      312597 :     if( FD_UNLIKELY( !accmetas[ i ] && !writable[ i ] ) ) continue;
    2411             : 
    2412      199239 :     original_cache_line[ i ] = NULL;
    2413      199239 :     if( FD_LIKELY( accmetas[ i ] ) ) {
    2414       97572 :       if( FD_LIKELY( FD_ACCDB_SIZE_CACHE_VALID( FD_VOLATILE_CONST( accmetas[ i ]->executable_size ) ) ) ) {
    2415             :         /* Concurrent evict_clear_acc_cache_ref clears VALID then stores
    2416             :            cache_idx=INVAL.  We may have observed VALID=1 just before the
    2417             :            writer cleared it, so cidx can read as INVAL here; decoding it
    2418             :            would yield a wild cache_line pointer.  Skip on INVAL.  Any
    2419             :            other stale cidx is harmless: cache_try_pin's ABA generation
    2420             :            check rejects a recycled line. */
    2421       97542 :         uint cidx = FD_VOLATILE_CONST( accmetas[ i ]->cache_idx );
    2422       97542 :         if( FD_LIKELY( cidx!=FD_ACCDB_ACC_CIDX_INVAL ) ) {
    2423       97542 :           fd_accdb_cache_line_t * hit = cache_line( accdb, FD_ACCDB_ACC_CIDX_CLASS( cidx ), FD_ACCDB_ACC_CIDX_IDX( cidx ) );
    2424       97542 :           fd_racesan_hook( "accdb_acquire:pre_try_pin" );
    2425       97542 :           original_cache_line[ i ] = cache_try_pin( hit, pubkeys[ i ], accmetas[ i ]->key.generation );
    2426             : #if FD_TMPL_USE_HANDHOLDING
    2427             :           if( FD_LIKELY( original_cache_line[ i ] ) ) {
    2428             :             FD_TEST( original_cache_line[ i ]->key.generation==accmetas[ i ]->key.generation &&
    2429             :                      !memcmp( original_cache_line[ i ]->key.pubkey, pubkeys[ i ], 32UL ) );
    2430             :             uint rc = FD_VOLATILE_CONST( original_cache_line[ i ]->refcnt );
    2431             :             FD_TEST( rc>0U && rc!=FD_ACCDB_EVICT_SENTINEL );
    2432             :           }
    2433             : #endif
    2434       97542 :         }
    2435       97542 :       }
    2436       97572 :     }
    2437      199239 :     exists_in_cache[ i ] = original_cache_line[ i ]!=NULL;
    2438             : 
    2439      199239 :     if( FD_UNLIKELY( writable[ i ] ) ) {
    2440     1049058 :       for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) destination_cache_lines[ i ][ j ] = acquire_cache_line( accdb, j, &evicted_dest_acc[ i ][ j ] );
    2441      116562 :       if( FD_UNLIKELY( accmetas[ i ] && !original_cache_line[ i ] ) ) {
    2442           0 :         original_cache_line[ i ] = cold_load_acc( accdb, accmetas[ i ], pubkeys[ i ], &exists_in_cache[ i ], &evicted_orig_acc[ i ] );
    2443           0 :       }
    2444      116562 :     } else {
    2445       82677 :       if( FD_UNLIKELY( !original_cache_line[ i ] ) ) {
    2446          30 :         original_cache_line[ i ] = cold_load_acc( accdb, accmetas[ i ], pubkeys[ i ], &exists_in_cache[ i ], &evicted_orig_acc[ i ] );
    2447          30 :       }
    2448       82677 :     }
    2449      199239 :   }
    2450             : 
    2451             :   // STEP 5.
    2452             :   //   For any cache lines we have retrieved, which we might potentially
    2453             :   //   be about to trash (by writing stuff in there), we need to write
    2454             :   //   them back to disk first if they are dirty.  This is the process of
    2455             :   //   "persisting" (a/k/a evicting) whatever was previously in the
    2456             :   //   cache line we are about to use.
    2457             :   //
    2458             :   //   This step does not actually persist the data to disk, it just
    2459             :   //   constructs a series of iovecs (write instructions) which will be
    2460             :   //   used later to do the actual write.  The reason is that we want to
    2461             :   //   batch all the writes together into a single writev call, to
    2462             :   //   minimize overhead, and also keep the actual writes at the end of
    2463             :   //   the function and independent of the specific control flow, so
    2464             :   //   that they could be offloaded to another thread of made
    2465             :   //   asynchronous (e.g. with io_uring) in the future without needing
    2466             :   //   to change the rest of the logic.
    2467             : 
    2468      311754 :   int write_ops_cnt = 0;
    2469      311754 :   int write_meta_cnt = 0;
    2470      311754 :   ulong total_write_sz = 0UL;
    2471      311754 :   fd_accdb_disk_meta_t write_metas[ (FD_ACCDB_CACHE_CLASS_CNT+1UL)*FD_ACCDB_MAX_ACQUIRE_CNT ];
    2472      311754 :   struct iovec write_ops[ 2UL*(FD_ACCDB_CACHE_CLASS_CNT+1UL)*FD_ACCDB_MAX_ACQUIRE_CNT ];
    2473             : 
    2474      624351 :   for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
    2475      312597 :     if( FD_UNLIKELY( !accmetas[ i ] && !writable[ i ] ) ) continue;
    2476             : 
    2477      199239 :     if( FD_UNLIKELY( writable[ i ] ) ) {
    2478     1049058 :       for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) {
    2479      932496 :         if( FD_LIKELY( evicted_dest_acc[ i ][ j ]==UINT_MAX ) ) continue;
    2480           0 :         accdb->metrics->accounts_evicted++;
    2481           0 :         accdb->metrics->accounts_evicted_per_class[ j ]++;
    2482             : 
    2483           0 :         fd_accdb_accmeta_t const * evicted = &accdb->acc_pool[ evicted_dest_acc[ i ][ j ] ];
    2484           0 :         fd_racesan_hook( "writeback:pre_synth" );
    2485           0 :         total_write_sz += sizeof(fd_accdb_disk_meta_t) + FD_ACCDB_SIZE_DATA( evicted->executable_size );
    2486           0 :         FD_TEST( write_meta_cnt<(int)(sizeof(write_metas)/sizeof(write_metas[0])) );
    2487           0 :         fd_memcpy( write_metas[ write_meta_cnt ].pubkey, evicted->key.pubkey, 32UL );
    2488           0 :         write_metas[ write_meta_cnt ].size       = FD_ACCDB_SIZE_DATA( evicted->executable_size );
    2489           0 :         write_metas[ write_meta_cnt ].generation = evicted->key.generation;
    2490           0 :         fd_memcpy( write_metas[ write_meta_cnt ].owner, destination_cache_lines[ i ][ j ]->owner, 32UL );
    2491           0 :         write_ops[ write_ops_cnt++ ] = (struct iovec){ .iov_base = &write_metas[ write_meta_cnt ], .iov_len = sizeof(fd_accdb_disk_meta_t) };
    2492           0 :         write_meta_cnt++;
    2493           0 :         write_ops[ write_ops_cnt++ ] = (struct iovec){ .iov_base = destination_cache_lines[ i ][ j ]+1UL, .iov_len = FD_ACCDB_SIZE_DATA( evicted->executable_size ) };
    2494           0 :       }
    2495      116562 :       if( FD_UNLIKELY( accmetas[ i ] && !exists_in_cache[ i ] && evicted_orig_acc[ i ]!=UINT_MAX ) ) {
    2496           0 :         fd_accdb_accmeta_t const * evicted = &accdb->acc_pool[ evicted_orig_acc[ i ] ];
    2497           0 :         accdb->metrics->accounts_evicted++;
    2498           0 :         accdb->metrics->accounts_evicted_per_class[ fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( evicted->executable_size ) ) ]++;
    2499             : 
    2500           0 :         total_write_sz += sizeof(fd_accdb_disk_meta_t) + FD_ACCDB_SIZE_DATA( evicted->executable_size );
    2501           0 :         FD_TEST( write_meta_cnt<(int)(sizeof(write_metas)/sizeof(write_metas[0])) );
    2502           0 :         fd_memcpy( write_metas[ write_meta_cnt ].pubkey, evicted->key.pubkey, 32UL );
    2503           0 :         write_metas[ write_meta_cnt ].size       = FD_ACCDB_SIZE_DATA( evicted->executable_size );
    2504           0 :         write_metas[ write_meta_cnt ].generation = evicted->key.generation;
    2505           0 :         fd_memcpy( write_metas[ write_meta_cnt ].owner, original_cache_line[ i ]->owner, 32UL );
    2506           0 :         write_ops[ write_ops_cnt++ ] = (struct iovec){ .iov_base = &write_metas[ write_meta_cnt ], .iov_len = sizeof(fd_accdb_disk_meta_t) };
    2507           0 :         write_meta_cnt++;
    2508           0 :         write_ops[ write_ops_cnt++ ] = (struct iovec){ .iov_base = original_cache_line[ i ]+1UL, .iov_len = FD_ACCDB_SIZE_DATA( evicted->executable_size ) };
    2509           0 :       }
    2510      116562 :     } else {
    2511       82677 :       if( FD_LIKELY( exists_in_cache[ i ] || evicted_orig_acc[ i ]==UINT_MAX ) ) continue;
    2512           0 :       fd_accdb_accmeta_t const * evicted = &accdb->acc_pool[ evicted_orig_acc[ i ] ];
    2513           0 :       accdb->metrics->accounts_evicted++;
    2514           0 :       accdb->metrics->accounts_evicted_per_class[ fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( evicted->executable_size ) ) ]++;
    2515           0 :       total_write_sz += sizeof(fd_accdb_disk_meta_t) + FD_ACCDB_SIZE_DATA( evicted->executable_size );
    2516           0 :       FD_TEST( write_meta_cnt<(int)(sizeof(write_metas)/sizeof(write_metas[0])) );
    2517           0 :       fd_memcpy( write_metas[ write_meta_cnt ].pubkey, evicted->key.pubkey, 32UL );
    2518           0 :       write_metas[ write_meta_cnt ].size       = FD_ACCDB_SIZE_DATA( evicted->executable_size );
    2519           0 :       write_metas[ write_meta_cnt ].generation = evicted->key.generation;
    2520           0 :       fd_memcpy( write_metas[ write_meta_cnt ].owner, original_cache_line[ i ]->owner, 32UL );
    2521           0 :       write_ops[ write_ops_cnt++ ] = (struct iovec){ .iov_base = &write_metas[ write_meta_cnt ], .iov_len = sizeof(fd_accdb_disk_meta_t) };
    2522           0 :       write_meta_cnt++;
    2523           0 :       write_ops[ write_ops_cnt++ ] = (struct iovec){ .iov_base = original_cache_line[ i ]+1UL, .iov_len = FD_ACCDB_SIZE_DATA( evicted->executable_size ) };
    2524           0 :     }
    2525      199239 :   }
    2526             : 
    2527             :   // STEP 6-7.
    2528             :   //   Compute the file offset for the writes we are about to do and
    2529             :   //   build the pending offset table.  The common case is a single
    2530             :   //   atomic fetch-add on the write head, reserving a contiguous
    2531             :   //   region.  If the total eviction batch is too large to fit in one
    2532             :   //   partition (extremely unlikely — requires many dirty 10MiB
    2533             :   //   evictions), fall back to per-entry allocation so that each
    2534             :   //   individual write fits in a single partition.
    2535             :   //
    2536             :   //   The actual stores to evicted->offset_fork and line->persisted
    2537             :   //   are deferred until after pwritev2 completes (Step 9-10), so
    2538             :   //   a concurrent acquire spinning on offset==FD_ACCDB_OFF_INVAL
    2539             :   //   does not proceed to preadv2 from a location that hasn't been
    2540             :   //   written.
    2541      311754 :   int                     pending_cnt = 0;
    2542      311754 :   fd_accdb_accmeta_t *    pending_accs [ (FD_ACCDB_CACHE_CLASS_CNT+1UL)*FD_ACCDB_MAX_ACQUIRE_CNT ];
    2543      311754 :   ulong                   pending_offs [ (FD_ACCDB_CACHE_CLASS_CNT+1UL)*FD_ACCDB_MAX_ACQUIRE_CNT ];
    2544      311754 :   fd_accdb_cache_line_t * pending_lines[ (FD_ACCDB_CACHE_CLASS_CNT+1UL)*FD_ACCDB_MAX_ACQUIRE_CNT ];
    2545             : 
    2546      311754 :   ulong file_offset;
    2547      311754 :   int   batch_contiguous;
    2548      311754 :   if( FD_LIKELY( total_write_sz && total_write_sz<=accdb->shmem->partition_sz ) ) {
    2549           0 :     file_offset      = allocate_next_write( accdb, total_write_sz );
    2550           0 :     batch_contiguous = 1;
    2551      311754 :   } else {
    2552      311754 :     file_offset      = 0UL;
    2553      311754 :     batch_contiguous = 0;
    2554      311754 :   }
    2555             : 
    2556      311754 :   ulong cumulative_offset = 0UL;
    2557      624351 :   for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
    2558      312597 :     if( FD_UNLIKELY( !accmetas[ i ] && !writable[ i ] ) ) continue;
    2559             : 
    2560      199239 :     if( FD_UNLIKELY( writable[ i ] ) ) {
    2561     1049058 :       for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) {
    2562      932496 :         if( FD_LIKELY( evicted_dest_acc[ i ][ j ]==UINT_MAX ) ) continue;
    2563             : 
    2564           0 :         fd_accdb_accmeta_t * evicted = &accdb->acc_pool[ evicted_dest_acc[ i ][ j ] ];
    2565           0 :         ulong entry_sz = sizeof(fd_accdb_disk_meta_t) + (ulong)FD_ACCDB_SIZE_DATA( evicted->executable_size );
    2566             :         /* xchg-to-INVAL atomically captures the old offset and prevents
    2567             :            a concurrent acc_unlink from also reading and freeing it (the
    2568             :            xchg there will see INVAL and skip).  Step 10 republishes the
    2569             :            new offset; the spinner at line ~2082 tolerates the transient
    2570             :            INVAL.  Same pattern as the overwrite path at line ~2388. */
    2571           0 :         ulong old_off = fd_accdb_acc_xchg_offset( evicted, FD_ACCDB_OFF_INVAL );
    2572           0 :         if( FD_LIKELY( old_off!=FD_ACCDB_OFF_INVAL ) ) {
    2573           0 :           fd_accdb_shmem_bytes_freed( accdb->shmem, old_off, entry_sz );
    2574           0 :           FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->shmetrics->disk_used_bytes, entry_sz );
    2575           0 :         }
    2576           0 :         FD_TEST( pending_cnt<(int)(sizeof(pending_accs)/sizeof(pending_accs[0])) );
    2577           0 :         pending_accs [ pending_cnt ] = evicted;
    2578           0 :         if( FD_LIKELY( batch_contiguous ) ) pending_offs[ pending_cnt ] = file_offset + cumulative_offset;
    2579           0 :         else                                pending_offs[ pending_cnt ] = allocate_next_write( accdb, entry_sz );
    2580           0 :         pending_lines[ pending_cnt ] = destination_cache_lines[ i ][ j ];
    2581           0 :         pending_cnt++;
    2582           0 :         cumulative_offset += entry_sz;
    2583           0 :         FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->shmetrics->disk_used_bytes, entry_sz );
    2584           0 :       }
    2585      116562 :       if( FD_UNLIKELY( accmetas[ i ] && !exists_in_cache[ i ] && evicted_orig_acc[ i ]!=UINT_MAX ) ) {
    2586           0 :         fd_accdb_accmeta_t * evicted = &accdb->acc_pool[ evicted_orig_acc[ i ] ];
    2587           0 :         ulong entry_sz = sizeof(fd_accdb_disk_meta_t) + (ulong)FD_ACCDB_SIZE_DATA( evicted->executable_size );
    2588           0 :         ulong old_off = fd_accdb_acc_xchg_offset( evicted, FD_ACCDB_OFF_INVAL );
    2589           0 :         if( FD_LIKELY( old_off!=FD_ACCDB_OFF_INVAL ) ) {
    2590           0 :           fd_accdb_shmem_bytes_freed( accdb->shmem, old_off, entry_sz );
    2591           0 :           FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->shmetrics->disk_used_bytes, entry_sz );
    2592           0 :         }
    2593           0 :         FD_TEST( pending_cnt<(int)(sizeof(pending_accs)/sizeof(pending_accs[0])) );
    2594           0 :         pending_accs [ pending_cnt ] = evicted;
    2595           0 :         if( FD_LIKELY( batch_contiguous ) ) pending_offs[ pending_cnt ] = file_offset + cumulative_offset;
    2596           0 :         else                                pending_offs[ pending_cnt ] = allocate_next_write( accdb, entry_sz );
    2597           0 :         pending_lines[ pending_cnt ] = original_cache_line[ i ];
    2598           0 :         pending_cnt++;
    2599           0 :         cumulative_offset += entry_sz;
    2600           0 :         FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->shmetrics->disk_used_bytes, entry_sz );
    2601           0 :       }
    2602      116562 :     } else {
    2603       82677 :       if( FD_LIKELY( exists_in_cache[ i ] || evicted_orig_acc[ i ]==UINT_MAX ) ) continue;
    2604             : 
    2605           0 :       fd_accdb_accmeta_t * evicted = &accdb->acc_pool[ evicted_orig_acc[ i ] ];
    2606           0 :       ulong entry_sz = sizeof(fd_accdb_disk_meta_t) + (ulong)FD_ACCDB_SIZE_DATA( evicted->executable_size );
    2607           0 :       ulong old_off = fd_accdb_acc_xchg_offset( evicted, FD_ACCDB_OFF_INVAL );
    2608           0 :       if( FD_LIKELY( old_off!=FD_ACCDB_OFF_INVAL ) ) {
    2609           0 :         fd_accdb_shmem_bytes_freed( accdb->shmem, old_off, entry_sz );
    2610           0 :         FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->shmetrics->disk_used_bytes, entry_sz );
    2611           0 :       }
    2612           0 :       FD_TEST( pending_cnt<(int)(sizeof(pending_accs)/sizeof(pending_accs[0])) );
    2613           0 :       pending_accs [ pending_cnt ] = evicted;
    2614           0 :       if( FD_LIKELY( batch_contiguous ) ) pending_offs[ pending_cnt ] = file_offset + cumulative_offset;
    2615           0 :       else                                pending_offs[ pending_cnt ] = allocate_next_write( accdb, entry_sz );
    2616           0 :       pending_lines[ pending_cnt ] = original_cache_line[ i ];
    2617           0 :       pending_cnt++;
    2618           0 :       cumulative_offset += entry_sz;
    2619           0 :       FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->shmetrics->disk_used_bytes, entry_sz );
    2620           0 :     }
    2621      199239 :   }
    2622             : 
    2623             :   // STEP 8.
    2624             :   //   Fill the output entries with cache pointers and metadata based on
    2625             :   //   the accounts we have located and the cache lines we have
    2626             :   //   reserved.
    2627             : 
    2628      624351 :   for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
    2629      312597 :     if( FD_UNLIKELY( !accmetas[ i ] && !writable[ i ] ) ) {
    2630      113358 :       out_accs[ i ].data = NULL;
    2631      113358 :       out_accs[ i ].data_len = 0UL;
    2632      113358 :       out_accs[ i ].lamports = 0UL;
    2633      113358 :       out_accs[ i ].executable = 0;
    2634      113358 :       memset( out_accs[ i ].owner, 0, 32UL );
    2635      113358 :       fd_memcpy( out_accs[ i ].pubkey, pubkeys[ i ], 32UL );
    2636      113358 :       out_accs[ i ].prior_lamports = 0UL;
    2637      113358 :       out_accs[ i ].prior_data_len = 0UL;
    2638      113358 :       out_accs[ i ].prior_executable = 0;
    2639      113358 :       memset( out_accs[ i ].prior_owner, 0, 32UL );
    2640      113358 :       out_accs[ i ].prior_data = NULL;
    2641      113358 :       out_accs[ i ].commit = 0;
    2642      113358 :       out_accs[ i ].pd_write = 0;
    2643      113358 :       out_accs[ i ]._writable = 0;
    2644      113358 :       out_accs[ i ]._original_size_class = ULONG_MAX;
    2645      113358 :       out_accs[ i ]._original_cache_idx = ULONG_MAX;
    2646      113358 :       continue;
    2647      113358 :     }
    2648             : 
    2649      199239 :     if( FD_LIKELY( !writable[ i ] ) ) out_accs[ i ].data = (uchar *)(original_cache_line[ i ]+1UL);
    2650      116562 :     else                              out_accs[ i ].data = (uchar *)(destination_cache_lines[ i ][ 7UL ]+1UL);
    2651             :     /* Tombstone reset: agave's account loader returns AccountSharedData::default()
    2652             :        (System owner, empty data, exec=0) for any account with lamports==0.
    2653             :        https://github.com/anza-xyz/agave/blob/v2.3.1/svm/src/account_loader.rs#L199-L228 */
    2654      199239 :     fd_racesan_hook( "accdb_acquire:pre_step7_meta" );
    2655      199239 :     int tombstone = accmetas[ i ] && accmetas[ i ]->lamports==0UL;
    2656      199239 :     out_accs[ i ].data_len = ( accmetas[ i ] && !tombstone ) ? FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size ) : 0UL;
    2657      199239 :     out_accs[ i ].executable = ( accmetas[ i ] && !tombstone ) ? FD_ACCDB_SIZE_EXEC( accmetas[ i ]->executable_size ) : 0;
    2658      199239 :     fd_racesan_hook( "accdb_acquire:mid_step7_meta" );
    2659      199239 :     out_accs[ i ].lamports = accmetas[ i ] ? accmetas[ i ]->lamports : 0UL;
    2660      199239 :     if( FD_UNLIKELY( !accmetas[ i ] ) ) {
    2661      101667 :       memset( out_accs[ i ].owner,       0, 32UL );
    2662      101667 :       memset( out_accs[ i ].prior_owner, 0, 32UL );
    2663      101667 :     }
    2664             :     /* For accmetas[i] != NULL, both owners are copied from the cache
    2665             :        line below in step 15, after step 12 has populated it from disk
    2666             :        for cold loads. */
    2667             : 
    2668      199239 :     out_accs[ i ].prior_lamports   = out_accs[ i ].lamports;
    2669      199239 :     out_accs[ i ].prior_data_len   = out_accs[ i ].data_len;
    2670      199239 :     out_accs[ i ].prior_executable = out_accs[ i ].executable;
    2671      199239 :     out_accs[ i ].prior_data       = (uchar *)(original_cache_line[ i ] ? (original_cache_line[ i ]+1UL) : NULL);
    2672             : 
    2673      199239 :     out_accs[ i ].commit = 0;
    2674      199239 :     out_accs[ i ].pd_write = 0;
    2675      199239 :     out_accs[ i ]._writable = writable[ i ];
    2676      199239 :     if( FD_UNLIKELY( writable[ i ] && accmetas[ i ] ) ) out_accs[ i ]._overwrite = accdb->fork_pool[ fork_id.val ].shmem->generation==accmetas[ i ]->key.generation;
    2677      184344 :     else                                            out_accs[ i ]._overwrite = 0;
    2678             : 
    2679      199239 :     FD_TEST( out_accs[ i ].data_len<=(10UL<<20) );
    2680      199239 :     FD_TEST( !out_accs[ i ]._overwrite || accdb->fork_pool[ fork_id.val ].shmem->generation==accmetas[ i ]->key.generation );
    2681             : 
    2682             : #if FD_TMPL_USE_HANDHOLDING
    2683             :     if( FD_UNLIKELY( !writable[ i ] && accmetas[ i ] && !tombstone ) ) {
    2684             :       ulong cls = fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size ) );
    2685             :       FD_TEST( fd_accdb_ptr_in_region( accdb, cls, out_accs[ i ].data ) );
    2686             :     }
    2687             : #endif
    2688             : 
    2689      199239 :     if( FD_UNLIKELY( writable[ i ] ) ) {
    2690      116562 :       out_accs[ i ]._fork_id = fork_id.val;
    2691      116562 :       out_accs[ i ]._generation = fork->shmem->generation;
    2692      116562 :       out_accs[ i ]._acc_map_idx = acc_map_idxs[ i ];
    2693      116562 :     }
    2694      199239 :     fd_memcpy( out_accs[ i ].pubkey, pubkeys[ i ], 32UL );
    2695             : 
    2696      199239 :     if( FD_UNLIKELY( !accmetas[ i ] ) ) {
    2697      101667 :       out_accs[ i ]._original_size_class = ULONG_MAX;
    2698      101667 :       out_accs[ i ]._original_cache_idx = ULONG_MAX;
    2699      101667 :     } else {
    2700       97572 :       out_accs[ i ]._original_size_class = fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size ) );
    2701       97572 :       out_accs[ i ]._original_cache_idx = cache_line_idx( accdb, out_accs[ i ]._original_size_class, original_cache_line[ i ] );
    2702       97572 :     }
    2703             : 
    2704      199239 :     if( FD_UNLIKELY( writable[ i ] ) ) {
    2705     1049058 :       for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) {
    2706      932496 :         out_accs[ i ]._write.destination_cache_idx[ j ] = cache_line_idx( accdb, j, destination_cache_lines[ i ][ j ] );
    2707      932496 :       }
    2708      116562 :     }
    2709      199239 :   }
    2710             : 
    2711             :   // STEP 9.
    2712             :   //   Write the dirty eviction data to disk and publish the new offsets
    2713             :   //   BEFORE constructing read iovecs.  This is critical: step 4 may
    2714             :   //   have evicted a dirty cache line belonging to another account in
    2715             :   //   the same batch whose acc->offset is still FD_ACCDB_OFF_INVAL.
    2716             :   //   The read-iovec loop below spin-waits on
    2717             :   //   offset!=FD_ACCDB_OFF_INVAL, so publishing evicted offsets first
    2718             :   //   prevents an intra-batch deadlock where the thread waits on an
    2719             :   //   offset that only it can resolve.
    2720      311754 :   if( FD_LIKELY( batch_contiguous ) ) {
    2721             :     /* Fast path: all evictions fit in one contiguous region.  Use the
    2722             :        pre-built iovec array for a single batched pwritev2 call. */
    2723           0 :     ulong bytes_written = 0UL;
    2724           0 :     struct iovec * write_ptr = write_ops;
    2725           0 :     while( FD_LIKELY( bytes_written<total_write_sz ) ) {
    2726           0 :       long result = pwritev2( accdb->fd, write_ptr, fd_int_min( write_ops_cnt, IOV_MAX ), (long)(file_offset+bytes_written), 0 );
    2727           0 :       if( FD_UNLIKELY( -1==result && (errno==EINTR || errno==EAGAIN || errno==EWOULDBLOCK ) ) ) continue;
    2728           0 :       else if( FD_UNLIKELY( -1==result ) ) FD_LOG_ERR(( "pwritev2() failed (%d-%s)", errno, fd_io_strerror( errno ) ));
    2729           0 :       else if( FD_UNLIKELY( !result ) ) FD_LOG_ERR(( "accounts database is corrupt, pwritev2() returned 0 at offset %lu with %lu bytes remaining",
    2730           0 :                                                      file_offset+bytes_written, total_write_sz-bytes_written ));
    2731           0 :       bytes_written += (ulong)result;
    2732           0 :       accdb->metrics->bytes_written += (ulong)result;
    2733           0 :       accdb->metrics->write_ops++;
    2734             : 
    2735           0 :       while( write_ops_cnt && (ulong)result>=(ulong)write_ptr[ 0 ].iov_len ) {
    2736           0 :         result -= (long)write_ptr[ 0 ].iov_len;
    2737           0 :         write_ptr++;
    2738           0 :         write_ops_cnt--;
    2739           0 :       }
    2740           0 :       if( FD_LIKELY( write_ops_cnt ) ) {
    2741           0 :         write_ptr[ 0 ].iov_base = (uchar *)write_ptr[ 0 ].iov_base + result;
    2742           0 :         write_ptr[ 0 ].iov_len -= (ulong)result;
    2743           0 :       }
    2744           0 :     }
    2745      311754 :   } else {
    2746             :     /* Slow path: total eviction batch exceeds a single partition.
    2747             :        Write each entry individually using its own allocated offset.
    2748             :        This path is only taken in extreme edge cases (many concurrent
    2749             :        dirty 10 MiB evictions). */
    2750      311754 :     struct iovec * wp = write_ops;
    2751      311754 :     for( int k=0; k<pending_cnt; k++ ) {
    2752           0 :       ulong entry_sz = sizeof(fd_accdb_disk_meta_t) + (ulong)FD_ACCDB_SIZE_DATA( pending_accs[ k ]->executable_size );
    2753           0 :       ulong entry_off = pending_offs[ k ];
    2754           0 :       struct iovec entry_iovs[2] = { wp[0], wp[1] };
    2755           0 :       wp += 2;
    2756             : 
    2757           0 :       ulong written = 0UL;
    2758           0 :       while( FD_LIKELY( written<entry_sz ) ) {
    2759           0 :         long result = pwritev2( accdb->fd, entry_iovs, 2, (long)(entry_off+written), 0 );
    2760           0 :         if( FD_UNLIKELY( -1==result && (errno==EINTR || errno==EAGAIN || errno==EWOULDBLOCK ) ) ) continue;
    2761           0 :         else if( FD_UNLIKELY( -1==result ) ) FD_LOG_ERR(( "pwritev2() failed (%d-%s)", errno, fd_io_strerror( errno ) ));
    2762           0 :         else if( FD_UNLIKELY( !result ) ) FD_LOG_ERR(( "accounts database is corrupt, pwritev2() returned 0 at offset %lu with %lu bytes remaining", entry_off+written, entry_sz-written ));
    2763           0 :         written += (ulong)result;
    2764           0 :         accdb->metrics->bytes_written += (ulong)result;
    2765           0 :         accdb->metrics->write_ops++;
    2766             : 
    2767           0 :         for( int v=0; v<2; v++ ) {
    2768           0 :           if( (ulong)result>=(ulong)entry_iovs[ v ].iov_len ) {
    2769           0 :             result -= (long)entry_iovs[ v ].iov_len;
    2770           0 :             entry_iovs[ v ].iov_len = 0UL;
    2771           0 :           } else {
    2772           0 :             entry_iovs[ v ].iov_base = (uchar *)entry_iovs[ v ].iov_base + result;
    2773           0 :             entry_iovs[ v ].iov_len -= (ulong)result;
    2774           0 :             break;
    2775           0 :           }
    2776           0 :         }
    2777           0 :       }
    2778           0 :     }
    2779      311754 :   }
    2780             : 
    2781             :   // STEP 10.
    2782             :   //   Now that the data is on disk, publish the evicted account offsets
    2783             :   //   so concurrent acquire threads spinning on
    2784             :   //   offset==FD_ACCDB_OFF_INVAL can proceed.  The fence ensures
    2785             :   //   pwritev2 data is globally visible before the offset stores.
    2786      311754 :   FD_COMPILER_MFENCE();
    2787      311754 :   for( int k=0; k<pending_cnt; k++ ) {
    2788           0 :     pending_accs[ k ]->offset_fork = fd_accdb_acc_pack_offset_fork( pending_offs[ k ], fd_accdb_acc_fork_id(pending_accs[ k ]) );
    2789           0 :     pending_lines[ k ]->persisted = 1;
    2790           0 :   }
    2791             : 
    2792             :   // STEP 11.
    2793             :   //   Now construct iovecs for any reads we need to do of accounts into
    2794             :   //   the cache.  For reading accounts, we read them directly into the
    2795             :   //   sole cache line we took (and maybe just evicted).  For writing
    2796             :   //   accounts, we read them into the right sized cache line, and later
    2797             :   //   it will be copied to the staging buffer.  This is to prevent
    2798             :   //   repeatedly reading the same account off disk into cache, if it is
    2799             :   //   being written cold multiple times and every write fails.
    2800             : 
    2801      311754 :   ulong read_ops_cnt = 0UL;
    2802      311754 :   ulong read_offsets[ FD_ACCDB_CACHE_CLASS_CNT*FD_ACCDB_MAX_ACQUIRE_CNT ];
    2803      311754 :   uchar * read_bases[ FD_ACCDB_CACHE_CLASS_CNT*FD_ACCDB_MAX_ACQUIRE_CNT ];
    2804      311754 :   ulong read_sizes[ FD_ACCDB_CACHE_CLASS_CNT*FD_ACCDB_MAX_ACQUIRE_CNT ];
    2805      311754 :   struct iovec read_ops[ FD_ACCDB_CACHE_CLASS_CNT*FD_ACCDB_MAX_ACQUIRE_CNT ];
    2806             : 
    2807      624351 :   for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
    2808      312597 :     if( FD_UNLIKELY( !accmetas[ i ] || exists_in_cache[ i ] ) ) continue;
    2809             : 
    2810          30 :     accdb->metrics->accounts_not_found_per_class[ fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size ) ) ]++;
    2811             : 
    2812             :     /* Tombstones (lamports==0) have no on-disk payload to read, and
    2813             :        background_advance_root may unlink the acc and never assign it a
    2814             :        disk offset, so the offset_fork spin below would hang forever.
    2815             :        Step 15's tombstone reset zeros the owner for these accounts. */
    2816          30 :     if( FD_UNLIKELY( !accmetas[ i ]->lamports ) ) continue;
    2817             : 
    2818             :     /* We are guaranteed that if an account is in the cache, the bytes
    2819             :        are available (all cache operations are atomic via refcnt CAS),
    2820             :        but we are not guaranteed that if something is _not_ in the cache
    2821             :        that it has been written back to disk yet.  In particular, if we
    2822             :        are trying to read an account that another thread is in the
    2823             :        process of evicting, we know they removed it from the cache, but
    2824             :        we don't know exactly when they will have written it back fully
    2825             :        to disk, so we may need to wait for that here.
    2826             : 
    2827             :        Compaction may concurrently relocate this record, but
    2828             :        epoch-based safe reclamation guarantees the source partition
    2829             :        is not freed until all epoch-protected operations that could
    2830             :        have snapshotted the old offset have exited.  So the data at the
    2831             :        snapshotted offset remains stable for the duration of our
    2832             :        read and no post-read validation is needed. */
    2833          30 :     ulong off_packed = FD_VOLATILE_CONST( accmetas[ i ]->offset_fork );
    2834          30 :     while( FD_UNLIKELY( (off_packed & FD_ACCDB_OFF_MASK)==FD_ACCDB_OFF_INVAL ) ) {
    2835           0 :       FD_SPIN_PAUSE();
    2836           0 :       off_packed = FD_VOLATILE_CONST( accmetas[ i ]->offset_fork );
    2837           0 :     }
    2838          30 :     fd_racesan_hook( "accdb_coldload:pre_iovec" );
    2839             : 
    2840          30 :     read_offsets[ read_ops_cnt ] = fd_accdb_acc_offset(accmetas[ i ]) + offsetof(fd_accdb_disk_meta_t, owner);
    2841          30 :     read_bases[ read_ops_cnt ]   = original_cache_line[ i ]->owner;
    2842          30 :     read_sizes[ read_ops_cnt ]   = 32UL + FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size );
    2843          30 :     read_ops[ read_ops_cnt++ ]   = (struct iovec){ .iov_base = original_cache_line[ i ]->owner, .iov_len = 32UL + FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size ) };
    2844          30 :   }
    2845             : 
    2846             :   // STEP 12.
    2847             :   //   Almost done... now do the actual reads of accounts into cache,
    2848             :   //   using the iovecs we constructed.  This is basically the same loop
    2849             :   //   as the writes, but with preadv2 instead of pwritev2, and that the
    2850             :   //   reads are not necessarily all contiguous, but occur at random
    2851             :   //   offsets.
    2852             :   //
    2853             :   //   CONCURRENCY: The compaction tile may concurrently relocate a
    2854             :   //   record we are about to read (both are epoch-protected).  Epoch-
    2855             :   //   based safe reclamation guarantees the source partition is not
    2856             :   //   freed until all epoch-protected operations that could have
    2857             :   //   snapshotted the old offset have exited, so the data at the
    2858             :   //   remains stable for the duration of this read — no post-read
    2859             :   //   validation or retry is needed.
    2860      311784 :   for( ulong i=0UL; i<read_ops_cnt; i++ ) {
    2861          30 :     ulong bytes_read = 0UL;
    2862          60 :     while( FD_LIKELY( bytes_read<read_sizes[ i ] ) ) {
    2863          30 :       long result = preadv2( accdb->fd, &read_ops[ i ], 1, (long)(read_offsets[ i ]+bytes_read), 0 );
    2864          30 :       if( FD_UNLIKELY( -1==result && (errno==EINTR || errno==EAGAIN || errno==EWOULDBLOCK ) ) ) continue;
    2865          30 :       else if( FD_UNLIKELY( -1==result ) ) FD_LOG_ERR(( "preadv2() failed (%d-%s)", errno, fd_io_strerror( errno ) ));
    2866          30 :       else if( FD_UNLIKELY( !result ) ) FD_LOG_ERR(( "accounts database is corrupt, data expected at offset %lu with size %lu exceeded file extents",
    2867          30 :                                                      read_offsets[ i ]+bytes_read, read_sizes[ i ] ));
    2868          30 :       fd_accdb_partition_read_bump( accdb, read_offsets[ i ]+bytes_read, (ulong)result );
    2869          30 :       bytes_read += (ulong)result;
    2870          30 :       accdb->metrics->bytes_read += (ulong)result;
    2871          30 :       accdb->metrics->read_ops++;
    2872             : 
    2873          30 :       read_ops[ i ].iov_base = read_bases[ i ] + bytes_read;
    2874          30 :       read_ops[ i ].iov_len  = read_sizes[ i ] - bytes_read;
    2875          30 :     }
    2876          30 :   }
    2877             : 
    2878             :   // STEP 13.
    2879             :   //   Publish the real acc index for any cache lines we just loaded
    2880             :   //   from disk, so concurrent threads spinning on acc_idx==UINT_MAX
    2881             :   //   can proceed.  The fence ensures all preadv2 data is visible
    2882             :   //   before the sentinel is cleared.
    2883      311754 :   FD_COMPILER_MFENCE();
    2884      624351 :   for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
    2885      312597 :     if( FD_UNLIKELY( !accmetas[ i ] || exists_in_cache[ i ] ) ) continue;
    2886          30 :     FD_VOLATILE( original_cache_line[ i ]->acc_idx ) = (uint)( accmetas[ i ] - accdb->acc_pool );
    2887          30 :     FD_TEST( FD_VOLATILE_CONST( original_cache_line[ i ]->acc_idx )==(uint)( accmetas[ i ] - accdb->acc_pool ) );
    2888          30 :   }
    2889             : 
    2890             :   // STEP 14.
    2891             :   //   Spin-wait for any cache lines found via acc->cache_idx that are
    2892             :   //   still being loaded by another thread's preadv2.  The loading
    2893             :   //   thread sets acc_idx to UINT_MAX before publishing cache_idx
    2894             :   //   and publishes the real acc index after its read completes.
    2895             :   //   This step is placed as late as possible to give the loading
    2896             :   //   thread maximum time to finish before we need to spin.
    2897      624351 :   for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
    2898      312597 :     if( FD_UNLIKELY( !accmetas[ i ] && !writable[ i ] ) ) continue;
    2899             : 
    2900      199239 :     if( FD_UNLIKELY( !original_cache_line[ i ] ) ) continue;
    2901       97572 :     if( FD_LIKELY( FD_VOLATILE_CONST( original_cache_line[ i ]->acc_idx )!=UINT_MAX ) ) goto step13_check;
    2902           0 :     accdb->metrics->accounts_waited++;
    2903           0 :     while( FD_UNLIKELY( FD_VOLATILE_CONST( original_cache_line[ i ]->acc_idx )==UINT_MAX ) ) {
    2904           0 :       fd_racesan_hook( "accdb_acquire:step14_load_wait" );
    2905           0 :       FD_SPIN_PAUSE();
    2906           0 :     }
    2907       97572 :   step13_check:;
    2908             : #if FD_TMPL_USE_HANDHOLDING
    2909             :     FD_TEST( original_cache_line[ i ]->key.generation==accmetas[ i ]->key.generation &&
    2910             :              !memcmp( original_cache_line[ i ]->key.pubkey, pubkeys[ i ], 32UL ) );
    2911             : #endif
    2912       97572 :   }
    2913             : 
    2914             :   // STEP 15.
    2915             :   //   Now that all reads from disk into original_cache_line have
    2916             :   //   completed (and any concurrent loaders have published their
    2917             :   //   acc_idx in step 14), copy the owner into the output entries.
    2918             :   //   This must happen here rather than in step 8 because the cache
    2919             :   //   line owner is only valid post-read for cold loads.
    2920      624351 :   for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
    2921      312597 :     if( FD_UNLIKELY( !accmetas[ i ] ) ) continue;
    2922       97572 :     fd_racesan_hook( "accdb_acquire:pre_step14_owner" );
    2923             :     /* Tombstone reset: see STEP 7 comment. */
    2924       97572 :     if( FD_UNLIKELY( accmetas[ i ]->lamports==0UL ) ) {
    2925          30 :       memset( out_accs[ i ].owner,       0, 32UL );
    2926          30 :       memset( out_accs[ i ].prior_owner, 0, 32UL );
    2927       97542 :     } else {
    2928       97542 :       fd_memcpy( out_accs[ i ].owner,       original_cache_line[ i ]->owner, 32UL );
    2929       97542 :       fd_memcpy( out_accs[ i ].prior_owner, original_cache_line[ i ]->owner, 32UL );
    2930       97542 :     }
    2931       97572 :   }
    2932             : 
    2933             :   // STEP 16.
    2934             :   //   Finally, copy any accounts we are writing into the staging
    2935             :   //   buffers, so they occupy a 10MiB cache line for the execution
    2936             :   //   system.
    2937      624351 :   for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
    2938      312597 :     if( FD_UNLIKELY( !accmetas[ i ] || !writable[ i ] ) ) continue;
    2939             : 
    2940       14895 :     ulong copy_sz = (ulong)FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size );
    2941       14895 :     fd_memcpy( destination_cache_lines[ i ][ 7UL ]+1UL, original_cache_line[ i ]+1UL, copy_sz );
    2942       14895 :     accdb->metrics->bytes_copied += copy_sz;
    2943       14895 :   }
    2944             : 
    2945      311754 :   FD_COMPILER_MFENCE();
    2946      311754 :   FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
    2947      311754 : }
    2948             : 
    2949             : void
    2950             : fd_accdb_acquire( fd_accdb_t *          accdb,
    2951             :                   fd_accdb_fork_id_t    fork_id,
    2952             :                   ulong                 pubkeys_cnt,
    2953             :                   uchar const * const * pubkeys,
    2954             :                   int *                 writable,
    2955      311052 :                   fd_acc_t *            out_accs ) {
    2956      311052 :   FD_TEST( accdb->acquire_state==FD_ACCDB_ACQUIRE_STATE_IDLE );
    2957      311052 :   accdb->acquire_state = FD_ACCDB_ACQUIRE_STATE_OPEN;
    2958      311052 :   fd_accdb_acquire_inner( accdb, fork_id, RESERVATION_TYPE_SIMPLE, 0UL, pubkeys_cnt, pubkeys, writable, out_accs );
    2959      311052 : }
    2960             : 
    2961             : void
    2962             : fd_accdb_acquire_a( fd_accdb_t *             accdb,
    2963             :                        fd_accdb_fork_id_t    fork_id,
    2964             :                        ulong                 pubkeys_cnt,
    2965             :                        uchar const * const * pubkeys,
    2966             :                        int *                 writable,
    2967         351 :                        fd_acc_t *            out_accs ) {
    2968         351 :   FD_TEST( accdb->acquire_state==FD_ACCDB_ACQUIRE_STATE_IDLE );
    2969         351 :   accdb->acquire_state = FD_ACCDB_ACQUIRE_STATE_PHASE_A;
    2970         351 :   fd_accdb_acquire_inner( accdb, fork_id, RESERVATION_TYPE_MAYBE_PROGRAMDATA, 0UL, pubkeys_cnt, pubkeys, writable, out_accs );
    2971         351 : }
    2972             : 
    2973             : void
    2974             : fd_accdb_acquire_b( fd_accdb_t *          accdb,
    2975             :                     fd_accdb_fork_id_t    fork_id,
    2976             :                     ulong                 reserved_cnt,
    2977             :                     ulong                 pubkeys_cnt,
    2978             :                     uchar const * const * pubkeys,
    2979             :                     int *                 writable,
    2980         351 :                     fd_acc_t *            out_accs ) {
    2981         351 :   FD_TEST( accdb->acquire_state==FD_ACCDB_ACQUIRE_STATE_PHASE_A );
    2982         351 :   accdb->acquire_state = FD_ACCDB_ACQUIRE_STATE_OPEN;
    2983         351 :   fd_accdb_acquire_inner( accdb, fork_id, RESERVATION_TYPE_ALREADY_RESERVED, reserved_cnt, pubkeys_cnt, pubkeys, writable, out_accs );
    2984         351 : }
    2985             : 
    2986             : /* release_inner drains one group of acquired accs but does NOT change the
    2987             :    handle's acquire_state.  The public fd_accdb_release / fd_accdb_release_ab
    2988             :    wrappers below own the state transition (a single-phase release closes
    2989             :    the bracket; release_ab drains both phase groups then closes). */
    2990             : static void
    2991             : release_inner( fd_accdb_t * accdb,
    2992             :                ulong        accs_cnt,
    2993      311322 :                fd_acc_t *   accs ) {
    2994      311322 :   FD_TEST( accdb->acquire_state==FD_ACCDB_ACQUIRE_STATE_OPEN );
    2995             : 
    2996      311322 :   {
    2997      311322 :     ulong prev = FD_VOLATILE_CONST( *accdb->my_epoch_slot );
    2998      311322 :     FD_TEST( prev==ULONG_MAX || prev<=FD_VOLATILE_CONST( accdb->shmem->epoch ) );
    2999      311322 :   }
    3000             : 
    3001      311322 :   FD_COMPILER_MFENCE();
    3002      311322 :   FD_VOLATILE( *accdb->my_epoch_slot ) = FD_VOLATILE_CONST( accdb->shmem->epoch );
    3003      311322 :   FD_HW_MFENCE(); /* StoreLoad: epoch store must be globally visible
    3004             :                      before any subsequent loads so the deferred
    3005             :                      reclamation scan does not miss us. */
    3006             : 
    3007             :   // STEP 1.
    3008             :   //   For each cache line which was written to in the 10MiB staging
    3009             :   //   buffer, we may need to copy to the data out to a right sized
    3010             :   //   cache line.  Figuring out the target cache line is non-obvious,
    3011             :   //   but follows the more complete logic below this, we just pull the
    3012             :   //   memcpy out so they are not done inside the cache lock.
    3013             : 
    3014      623490 :   for( ulong i=0UL; i<accs_cnt; i++ ) {
    3015      312168 :     if( FD_UNLIKELY( accs[ i ]._original_size_class==ULONG_MAX && !accs[ i ]._writable ) ) continue;
    3016             : 
    3017             : #if FD_TMPL_USE_HANDHOLDING
    3018             :     if( FD_LIKELY( accs[ i ]._original_size_class!=ULONG_MAX ) ) {
    3019             :       FD_TEST( accs[ i ]._original_cache_idx<accdb->shmem->cache_class_max[ accs[ i ]._original_size_class ] );
    3020             :     }
    3021             :     if( FD_UNLIKELY( accs[ i ].commit ) ) FD_TEST( accs[ i ]._writable );
    3022             : #endif
    3023             : 
    3024      198837 :     if( FD_LIKELY( !accs[ i ]._writable || !accs[ i ].commit ) ) continue;
    3025             : #if FD_TMPL_USE_HANDHOLDING
    3026             :     if( FD_UNLIKELY( accs[ i ]._overwrite ) ) {
    3027             :       FD_TEST( accs[ i ]._writable );
    3028             :       FD_TEST( accs[ i ]._original_cache_idx!=ULONG_MAX );
    3029             :       FD_TEST( accs[ i ]._original_size_class!=ULONG_MAX );
    3030             :     }
    3031             : #endif
    3032             : 
    3033      115629 :     ulong original_size_class = accs[ i ]._original_size_class;
    3034      115629 :     ulong new_size_class = fd_accdb_cache_class( accs[ i ].data_len );
    3035      115629 :     if( FD_UNLIKELY( new_size_class==7UL ) ) continue;
    3036             : 
    3037      115317 :     fd_accdb_cache_line_t * target_cache_line;
    3038      115317 :     if( FD_LIKELY( original_size_class==new_size_class && accs[ i ]._overwrite ) ) target_cache_line = cache_line( accdb, original_size_class, accs[ i ]._original_cache_idx );
    3039      112134 :     else                                                                              target_cache_line = cache_line( accdb, new_size_class, accs[ i ]._write.destination_cache_idx[ new_size_class ] );
    3040             : 
    3041      115317 :     fd_accdb_cache_line_t * staging_line = cache_line( accdb, 7UL, accs[ i ]._write.destination_cache_idx[ 7UL ] );
    3042             : 
    3043      115317 :     fd_racesan_hook( "accdb_commit:pre_owner_write" );
    3044             : 
    3045             : #if FD_TMPL_USE_HANDHOLDING
    3046             :     if( FD_UNLIKELY( original_size_class==new_size_class && accs[ i ]._overwrite ) ) {
    3047             :       uint rc = FD_VOLATILE_CONST( target_cache_line->refcnt );
    3048             :       FD_TEST( target_cache_line->key.generation==accs[ i ]._generation &&
    3049             :                !memcmp( target_cache_line->key.pubkey, accs[ i ].pubkey, 32UL ) &&
    3050             :                rc>0U &&
    3051             :               rc!=FD_ACCDB_EVICT_SENTINEL );
    3052             :     }
    3053             : #endif
    3054             : 
    3055      115317 :     fd_memcpy( target_cache_line->owner, accs[ i ].owner, 32UL );
    3056      115317 :     fd_memcpy( target_cache_line+1UL, staging_line+1UL, accs[ i ].data_len );
    3057      115317 :     accdb->metrics->bytes_copied += accs[ i ].data_len;
    3058      115317 :   }
    3059             : 
    3060             :   // STEP 2.
    3061             :   //   Now update the metadata structures and free lists to reflect the
    3062             :   //   fact that we are done with these cache lines.  This is fully
    3063             :   //   atomic with CLOCK.
    3064             : 
    3065      623490 :   for( ulong i=0UL; i<accs_cnt; i++ ) {
    3066      312168 :     if( FD_UNLIKELY( accs[ i ]._original_size_class==ULONG_MAX && !accs[ i ]._writable ) ) continue;
    3067             : 
    3068      198837 :     ulong original_size_class = accs[ i ]._original_size_class;
    3069      198837 :     fd_accdb_cache_line_t * original_cache_line = accs[ i ]._original_cache_idx==ULONG_MAX ? NULL : cache_line( accdb, original_size_class, accs[ i ]._original_cache_idx );
    3070             :     /* For overwrite commits, defer the refcnt decrement on
    3071             :        original_cache_line until after invalidation completes.  If
    3072             :        we dropped refcnt to 0 here, a concurrent CLOCK sweep could
    3073             :        CAS(refcnt, 0, EVICT_SENTINEL) and steal the line before we
    3074             :        get to invalidate it, causing data corruption.
    3075             :        Non-overwrite and non-commit paths unpin
    3076             :        immediately because they never invalidate the original line. */
    3077      198837 :     if( FD_LIKELY( original_cache_line ) ) {
    3078             : #if FD_TMPL_USE_HANDHOLDING
    3079             :       FD_TEST( original_cache_line->refcnt>0U );
    3080             : #endif
    3081       97170 :       if( FD_LIKELY( !accs[ i ]._writable || !accs[ i ].commit || !accs[ i ]._overwrite ) ) {
    3082       93633 :         FD_ATOMIC_FETCH_AND_SUB( &original_cache_line->refcnt, 1U );
    3083       93633 :       }
    3084       97170 :     }
    3085             : 
    3086      198837 :     if( FD_LIKELY( !accs[ i ]._writable ) ) {
    3087             :       /* For readonly accounts, mark as recently used so the CLOCK
    3088             :          algorithm gives it a second chance before eviction. */
    3089             : #if FD_TMPL_USE_HANDHOLDING
    3090             :       FD_TEST( original_cache_line );
    3091             : #endif
    3092       82509 :       original_cache_line->referenced = 1;
    3093       82509 :       continue;
    3094       82509 :     }
    3095             : 
    3096      116328 :     fd_accdb_cache_line_t * destination_cache_lines[ FD_ACCDB_CACHE_CLASS_CNT ];
    3097     1046952 :     for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) destination_cache_lines[ j ] = cache_line( accdb, j, accs[ i ]._write.destination_cache_idx[ j ] );
    3098      116328 :     int destination_committed[ FD_ACCDB_CACHE_CLASS_CNT ] = {0};
    3099             : 
    3100      116328 :     if( FD_LIKELY( !accs[ i ].commit ) ) {
    3101             :       /* If it's writable but it didn't commit, all of the destination
    3102             :          cache lines (including the staging buffer which is trashed) are
    3103             :          unused and can be pushed to the CAS free list for immediate
    3104             :          reuse.  Whatever buffer it was accessing also gets marked as
    3105             :          recently used. */
    3106         699 :       if( FD_LIKELY( original_cache_line ) ) original_cache_line->referenced = 1;
    3107        6291 :       for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) {
    3108             :         /* acquire_cache_line via CLOCK leaves line->acc_idx pointing
    3109             :            at the prior owner.  cache_free_push consumers (CLOCK,
    3110             :            background_preevict) skip lines only when acc_idx==UINT_MAX
    3111             :            AND gen==UINT_MAX; if we leave the stale acc_idx, a future
    3112             :            CLOCK pick would call line 849/853 against the wrong acc
    3113             :            and corrupt its cache_idx/valid. */
    3114        5592 :         destination_cache_lines[ j ]->acc_idx        = UINT_MAX;
    3115        5592 :         destination_cache_lines[ j ]->key.generation = UINT_MAX;
    3116        5592 :         destination_cache_lines[ j ]->persisted = 1;
    3117        5592 :         while( FD_UNLIKELY( FD_ATOMIC_CAS( &destination_cache_lines[ j ]->refcnt, 1U, 0U )!=1U ) ) {
    3118           0 :           fd_racesan_hook( "accdb_release:dest_refcnt_wait" );
    3119           0 :           FD_SPIN_PAUSE();
    3120           0 :         }
    3121        5592 :         cache_free_push( accdb, j, destination_cache_lines[ j ] );
    3122        5592 :       }
    3123         699 :       continue;
    3124         699 :     }
    3125             : 
    3126      115629 :     ulong new_size_class = fd_accdb_cache_class( accs[ i ].data_len );
    3127      115629 :     uint original_acc_idx = original_cache_line ? original_cache_line->acc_idx : UINT_MAX;
    3128      115629 :     fd_accdb_cache_line_t * committed_line;
    3129             : 
    3130             :     /* For overwrites, invalidate the on-disk offset BEFORE removing
    3131             :        the cache acc.  This ensures a concurrent acquire that misses
    3132             :        the cache will see offset==FD_ACCDB_OFF_INVAL and spin-wait,
    3133             :        rather than reading stale on-disk bytes from the old location.
    3134             :        The CAS-loop exchange also serializes with a concurrent
    3135             :        compaction CAS (old_offset -> dest_offset). */
    3136      115629 :     ulong old_offset = FD_ACCDB_OFF_INVAL;
    3137      115629 :     if( FD_LIKELY( accs[ i ]._overwrite ) ) {
    3138        3537 :       fd_accdb_accmeta_t * ow_accmeta = &accdb->acc_pool[ original_acc_idx ];
    3139        3537 :       fd_racesan_hook( "accdb_overwrite:pre_xchg_offset" );
    3140        3537 :       old_offset = fd_accdb_acc_xchg_offset( ow_accmeta, FD_ACCDB_OFF_INVAL );
    3141        3537 :       if( FD_LIKELY( old_offset!=FD_ACCDB_OFF_INVAL ) ) {
    3142           0 :         fd_accdb_shmem_bytes_freed( accdb->shmem, old_offset, (ulong)FD_ACCDB_SIZE_DATA(ow_accmeta->executable_size)+sizeof(fd_accdb_disk_meta_t) );
    3143           0 :         FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->shmetrics->disk_used_bytes, (ulong)FD_ACCDB_SIZE_DATA(ow_accmeta->executable_size)+sizeof(fd_accdb_disk_meta_t) );
    3144           0 :       }
    3145        3537 :     }
    3146             : 
    3147      115629 :     if( FD_UNLIKELY( new_size_class==7UL ) ) {
    3148             :       /* The account belongs in the largest size class, and we already
    3149             :          have it resident in a 10MiB buffer anyway, so no need to copy
    3150             :          back.  If we are "overwriting" (same generation as the account
    3151             :          came from), then the original can be discarded (pushed to
    3152             :          the CAS free list) and removed from the cache. */
    3153         312 :       destination_cache_lines[ 7UL ]->persisted = 0;
    3154         312 :       destination_committed[ 7UL ] = 1;
    3155         312 :       if( FD_LIKELY( accs[ i ]._overwrite ) ) {
    3156             :         /* Atomically clear acc.VALID and acc.cache_idx BEFORE freeing
    3157             :            the line, so a reader cannot observe acc.VALID=1 with
    3158             :            acc.cache_idx pointing at a line that has been recycled to
    3159             :            another acc.  evict_clear_acc_cache_ref uses the CLAIM
    3160             :            protocol to serialize with cold_load_acc. */
    3161         303 :         evict_clear_acc_cache_ref( &accdb->acc_pool[ original_acc_idx ], original_size_class, accs[ i ]._original_cache_idx );
    3162             : 
    3163             :         /* Convert our own pin directly into the eviction claim
    3164             :            (CAS refcnt 1 -> EVICT_SENTINEL) so refcnt never passes
    3165             :            through 0 while acc_idx/key.generation are still valid:
    3166             :            in such a window background_preevict could claim and free
    3167             :            the line, and a claim of our own would then succeed on
    3168             :            the free-listed line and push it a second time.  The CAS
    3169             :            fails if a concurrent reader that pinned the line via
    3170             :            cache_try_pin BEFORE evict_clear_acc_cache_ref completed
    3171             :            still holds a reference (its ABA check on
    3172             :            line->key.generation is not synchronized with our writes
    3173             :            to that field); then just drop our pin and leave
    3174             :            acc_idx/key.generation intact so CLOCK reclaims the line
    3175             :            once the reader unpins.  That reclaim's
    3176             :            evict_clear_acc_cache_ref is a no-op, but its dirty-
    3177             :            writeback gate keys on persisted/acc_idx and would
    3178             :            republish the pre-overwrite bytes over the committed
    3179             :            version, so set persisted while still pinned. */
    3180         303 :         original_cache_line->persisted = 1;
    3181         303 :         fd_racesan_hook( "accdb_release:pre_discard_claim" );
    3182         303 :         if( FD_LIKELY( FD_ATOMIC_CAS( &original_cache_line->refcnt, 1U, FD_ACCDB_EVICT_SENTINEL )==1U ) ) {
    3183         303 :           original_cache_line->acc_idx   = UINT_MAX;
    3184         303 :           original_cache_line->key.generation = UINT_MAX;
    3185         303 :           FD_COMPILER_MFENCE();
    3186         303 :           FD_VOLATILE( original_cache_line->refcnt ) = 0;
    3187         303 :           cache_free_push( accdb, original_size_class, original_cache_line );
    3188         303 :         } else {
    3189           0 :           FD_ATOMIC_FETCH_AND_SUB( &original_cache_line->refcnt, 1U );
    3190           0 :         }
    3191         303 :       }
    3192         312 :       committed_line = destination_cache_lines[ 7UL ];
    3193      115317 :     } else {
    3194             :       /* The account started in some arbitrary size class, transited
    3195             :          through a 10MiB staging buffer, and is now being written back
    3196             :          to some arbitrary (non-10MiB) size class, so we need to copy it
    3197             :          there.  The staging buffer is discarded.  If we are going to
    3198             :          a different size class, and we are "overwriting" (same
    3199             :          generation), then the original can also be discarded, but if
    3200             :          we are staying in the same size class, we can reuse the cache
    3201             :          line in place. */
    3202      115317 :       fd_accdb_cache_line_t * target_cache_line;
    3203      115317 :       if( FD_LIKELY( original_size_class==new_size_class ) ) {
    3204       13944 :         if( FD_LIKELY( accs[ i ]._overwrite ) ) {
    3205             :           /* a reader holding a stale acc->cache_idx from this line's
    3206             :              previous life may hold a transient cache_try_pin pin, so
    3207             :              our own pin only bounds refcnt from below. */
    3208        3183 :           uint ow_rc = FD_VOLATILE_CONST( original_cache_line->refcnt );
    3209        3183 :           FD_TEST( ow_rc>0U && ow_rc!=FD_ACCDB_EVICT_SENTINEL );
    3210        3183 :           original_cache_line->key.generation = UINT_MAX;
    3211             :           /* Keep refcnt>=1 through the reuse window so CLOCK cannot
    3212             :              steal the line between invalidation and re-publish. The
    3213             :              pin is released in the destination cleanup loop after
    3214             :              acc->cache_idx has been republished. */
    3215        3183 :           original_cache_line->acc_idx = UINT_MAX;
    3216        3183 :           target_cache_line = original_cache_line;
    3217       10761 :         } else {
    3218       10761 :           target_cache_line = destination_cache_lines[ new_size_class ];
    3219       10761 :           destination_committed[ new_size_class ] = 1;
    3220       10761 :         }
    3221      101373 :       } else {
    3222      101373 :         if( FD_LIKELY( accs[ i ]._overwrite ) ) {
    3223             :           /* Atomically clear acc.VALID and acc.cache_idx BEFORE freeing
    3224             :              the line, so a reader cannot observe acc.VALID=1 with
    3225             :              acc.cache_idx pointing at a line that has been recycled to
    3226             :              another acc.  evict_clear_acc_cache_ref uses the CLAIM
    3227             :              protocol to serialize with cold_load_acc.  See the
    3228             :              size_class==7 path above for the refcnt claim and
    3229             :              persisted rationale. */
    3230          51 :           evict_clear_acc_cache_ref( &accdb->acc_pool[ original_acc_idx ], original_size_class, accs[ i ]._original_cache_idx );
    3231          51 :           original_cache_line->persisted = 1;
    3232          51 :           fd_racesan_hook( "accdb_release:pre_discard_claim" );
    3233          51 :           if( FD_LIKELY( FD_ATOMIC_CAS( &original_cache_line->refcnt, 1U, FD_ACCDB_EVICT_SENTINEL )==1U ) ) {
    3234          51 :             original_cache_line->acc_idx   = UINT_MAX;
    3235          51 :             original_cache_line->key.generation = UINT_MAX;
    3236          51 :             FD_COMPILER_MFENCE();
    3237          51 :             FD_VOLATILE( original_cache_line->refcnt ) = 0;
    3238          51 :             cache_free_push( accdb, original_size_class, original_cache_line );
    3239          51 :           } else {
    3240           0 :             FD_ATOMIC_FETCH_AND_SUB( &original_cache_line->refcnt, 1U );
    3241           0 :           }
    3242          51 :         }
    3243             : 
    3244      101373 :         destination_committed[ new_size_class ] = 1;
    3245      101373 :         target_cache_line = destination_cache_lines[ new_size_class ];
    3246      101373 :       }
    3247             : 
    3248      115317 :       target_cache_line->persisted = 0;
    3249             :       /* If target is the original cache line (overwrite, same size
    3250             :          class), mark as referenced directly since the cleanup loop
    3251             :          only handles destination lines. */
    3252      115317 :       if( FD_LIKELY( !destination_committed[ new_size_class ] ) ) target_cache_line->referenced = 1;
    3253      115317 :       committed_line = target_cache_line;
    3254      115317 :     }
    3255             : 
    3256             :     /* For non-overwrite commits, the original cache line (if any) still
    3257             :        holds valid ancestor data but is no longer pinned.  Mark it as
    3258             :        recently used so the CLOCK algorithm retains it. */
    3259      115629 :     if( FD_UNLIKELY( !accs[ i ]._overwrite && original_cache_line ) ) {
    3260       10806 :       original_cache_line->referenced = 1;
    3261       10806 :     }
    3262             : 
    3263             :     /* Handle every destination cache line: committed ones keep
    3264             :        refcnt>=1 until acc->cache_idx is published (the deferred
    3265             :        unpin happens after the publish below), uncommitted ones are
    3266             :        fully freed to the CAS free list. */
    3267     1040661 :     for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) {
    3268      925032 :       if( destination_committed[ j ] ) {
    3269      112446 :         destination_cache_lines[ j ]->referenced = 1;
    3270      812586 :       } else {
    3271             :         /* See note above (no-commit path): clear stale acc_idx/gen
    3272             :            before pushing, otherwise CLOCK can pick this line and
    3273             :            stomp the prior owner's cache_idx/valid.  CAS on refcnt for
    3274             :            the same stray-pin reason as the no-commit path. */
    3275      812586 :         destination_cache_lines[ j ]->acc_idx        = UINT_MAX;
    3276      812586 :         destination_cache_lines[ j ]->key.generation = UINT_MAX;
    3277      812586 :         destination_cache_lines[ j ]->persisted = 1;
    3278      812586 :         while( FD_UNLIKELY( FD_ATOMIC_CAS( &destination_cache_lines[ j ]->refcnt, 1U, 0U )!=1U ) ) {
    3279           0 :           fd_racesan_hook( "accdb_release:dest_refcnt_wait" );
    3280           0 :           FD_SPIN_PAUSE();
    3281           0 :         }
    3282      812586 :         cache_free_push( accdb, j, destination_cache_lines[ j ] );
    3283      812586 :       }
    3284      925032 :     }
    3285             : 
    3286             :     /* Update the accounts index for this committed write.  For an
    3287             :        overwrite (same fork+generation), update the existing acc
    3288             :        in place.  Otherwise allocate a new acc, prepend it
    3289             :        to the hash chain, and record the write in a txn linked to
    3290             :        the fork so advance_root can clean up old versions. */
    3291      115629 :     if( FD_LIKELY( accs[ i ]._overwrite ) ) {
    3292        3537 :       accdb->metrics->accounts_committed_overwrite_per_class[ new_size_class ]++;
    3293        3537 :       committed_line->acc_idx = original_acc_idx;
    3294             : 
    3295        3537 :       fd_accdb_accmeta_t * accmeta = &accdb->acc_pool[ original_acc_idx ];
    3296             :       /* The offset was already atomically swapped to FD_ACCDB_OFF_INVAL
    3297             :          and bytes freed above, so just update the metadata and
    3298             :          re-publish the cache location.  CAS-loop preserves CLAIM bit
    3299             :          (a concurrent evict_clear_acc_cache_ref or acc_unlink may
    3300             :          hold it) and clears VALID; a plain store would clobber CLAIM
    3301             :          and break those protocols. */
    3302        3537 :       uint pd = accs[ i ].pd_write ? FD_ACCDB_SIZE_PD_WRITE_BIT : 0U;
    3303        3537 :       for(;;) {
    3304        3537 :         uint cur = FD_VOLATILE_CONST( accmeta->executable_size );
    3305        3537 :         uint nxt = (cur & (FD_ACCDB_SIZE_CACHE_CLAIM_BIT|FD_ACCDB_SIZE_PD_WRITE_BIT))
    3306        3537 :                  | FD_ACCDB_SIZE_PACK( (uint)accs[ i ].data_len, accs[ i ].executable )
    3307        3537 :                  | pd;
    3308        3537 :         if( FD_LIKELY( FD_ATOMIC_CAS( &accmeta->executable_size, cur, nxt )==cur ) ) break;
    3309           0 :         FD_SPIN_PAUSE();
    3310           0 :       }
    3311        3537 :       accmeta->lamports = accs[ i ].lamports;
    3312        3537 :       fd_racesan_hook( "accdb_overwrite:mid_inplace" );
    3313             : 
    3314        3537 :       fd_memcpy( committed_line->owner, accs[ i ].owner, 32UL );
    3315        3537 :       fd_memcpy( committed_line->key.pubkey, accmeta->key.pubkey, 32UL );
    3316        3537 :       committed_line->key.generation = accmeta->key.generation;
    3317        3537 :       committed_line->acc_idx = original_acc_idx;
    3318        3537 :       FD_VOLATILE( accmeta->cache_idx ) = FD_ACCDB_ACC_CIDX_PACK( (uint)new_size_class, (uint)cache_line_idx( accdb, new_size_class, committed_line ) );
    3319             :       /* Atomic OR so a concurrent evict_clear_acc_cache_ref's CLAIM
    3320             :          clear (FETCH_AND_AND with ~CLAIM) cannot be lost by an RMW
    3321             :          race with a plain |= store. */
    3322        3537 :       FD_ATOMIC_FETCH_AND_OR( &accmeta->executable_size, FD_ACCDB_SIZE_CACHE_VALID_BIT );
    3323             : 
    3324             :       /* Now that acc->cache_idx is published, unpin so CLOCK can
    3325             :          eventually evict it.  For same-size overwrites, committed_line
    3326             :          IS the reused original_cache_line.  For cross-size overwrites,
    3327             :          committed_line is a destination line whose refcnt decrement was
    3328             :          deferred from the cleanup loop. */
    3329        3537 :       FD_ATOMIC_FETCH_AND_SUB( &committed_line->refcnt, 1U );
    3330        3537 :       committed_line->referenced = 1;
    3331      112092 :     } else {
    3332      112092 :       accdb->metrics->accounts_committed_new_per_class[ new_size_class ]++;
    3333      112092 :       fd_accdb_accmeta_t * accmeta = acc_pool_acquire( accdb->acc_pool_join );
    3334      112092 :       FD_TEST( accmeta );
    3335      112092 :       ulong acc_idx = acc_pool_idx( accdb->acc_pool_join, accmeta );
    3336      112092 :       fd_memcpy( accmeta->key.pubkey, accs[ i ].pubkey, 32UL );
    3337      112092 :       accmeta->lamports        = accs[ i ].lamports;
    3338      112092 :       accmeta->executable_size = FD_ACCDB_SIZE_PACK( (uint)accs[ i ].data_len, accs[ i ].executable )
    3339      112092 :                                | (accs[ i ].pd_write ? FD_ACCDB_SIZE_PD_WRITE_BIT : 0U);
    3340      112092 :       accmeta->key.generation  = accs[ i ]._generation;
    3341      112092 :       accmeta->offset_fork     = fd_accdb_acc_pack_offset_fork( FD_ACCDB_OFF_INVAL, accs[ i ]._fork_id );
    3342             : 
    3343             :       /* Publish in the cache BEFORE the acc_map head so that a
    3344             :          concurrent acquire that finds this acc in the hash chain will
    3345             :          also find a cache hit, rather than inserting a conflicting
    3346             :          placeholder cache acc. */
    3347      112092 :       committed_line->acc_idx = (uint)acc_idx;
    3348      112092 :       fd_memcpy( committed_line->owner, accs[ i ].owner, 32UL );
    3349      112092 :       fd_memcpy( committed_line->key.pubkey, accmeta->key.pubkey, 32UL );
    3350      112092 :       committed_line->key.generation = accmeta->key.generation;
    3351      112092 :       FD_VOLATILE( accmeta->cache_idx ) = FD_ACCDB_ACC_CIDX_PACK( (uint)new_size_class, (uint)cache_line_idx( accdb, new_size_class, committed_line ) );
    3352             :       /* Atomic OR so a concurrent evict_clear_acc_cache_ref's CLAIM
    3353             :          clear (FETCH_AND_AND with ~CLAIM) cannot be lost by an RMW
    3354             :          race with a plain |= store. */
    3355      112092 :       FD_ATOMIC_FETCH_AND_OR( &accmeta->executable_size, FD_ACCDB_SIZE_CACHE_VALID_BIT );
    3356             : 
    3357             :       /* Now that acc->cache_idx is published, unpin it so
    3358             :          CLOCK can eventually evict it. */
    3359      112092 :       FD_ATOMIC_FETCH_AND_SUB( &committed_line->refcnt, 1U );
    3360      112092 :       committed_line->referenced = 1;
    3361             : 
    3362             :       /* CAS loop to prepend to the hash chain.  Succeeds on the first
    3363             :          try in most cases, but a concurrent acc_unlink CAS removing
    3364             :          the old head can change acc_map[idx] between our load and
    3365             :          CAS.  Multiple concurrent releases may also race on the head
    3366             :          pointer — the CAS retry handles this. */
    3367      112092 :       for(;;) {
    3368      112092 :         uint old_head = FD_VOLATILE_CONST( accdb->acc_map[ accs[ i ]._acc_map_idx ] );
    3369      112092 :         accmeta->map.next = old_head;
    3370      112092 :         FD_COMPILER_MFENCE();
    3371      112092 :         fd_racesan_hook( "accdb_release:pre_chain_cas" );
    3372      112092 :         if( FD_LIKELY( FD_ATOMIC_CAS( &accdb->acc_map[ accs[ i ]._acc_map_idx ], old_head, (uint)acc_idx )==old_head ) ) break;
    3373           0 :         FD_SPIN_PAUSE();
    3374           0 :       }
    3375             : 
    3376             :       /* CONCURRENCY: The cache acc is published before the acc_map
    3377             :          head so that a concurrent fd_accdb_acquire reader that
    3378             :          observes the new head also finds a cache hit, preventing
    3379             :          duplicate cache insertion.
    3380             : 
    3381             :          (1) The CAS on acc_map[idx] serializes head-pointer mutations
    3382             :              from concurrent releases onto the same chain without any
    3383             :              external lock.
    3384             : 
    3385             :          (2) The FD_COMPILER_MFENCE above ensures stores to the acc node
    3386             :              fields (pubkey, lamports, size, generation, fork_id,
    3387             :              offset, map.next) are ordered before the CAS that publishes
    3388             :              the new head.  On x86-64 (TSO), hardware also guarantees
    3389             :              this, but the compiler fence is needed to prevent the
    3390             :              compiler from reordering the stores.  A reader that
    3391             :              observes the new head is guaranteed to see a fully
    3392             :              initialized node.  A reader that has not yet seen the new
    3393             :              head simply traverses the previous (still valid) chain.
    3394             : 
    3395             :          (3) A concurrent acc_unlink (advance_root / purge) may CAS the
    3396             :              head away between our load and CAS here.  The CAS retry
    3397             :              loop handles this. */
    3398             : 
    3399      112092 :       fd_accdb_txn_t * txn = txn_pool_acquire( accdb->txn_pool );
    3400      112092 :       FD_TEST( txn ); /* Sized so it always succeeds */
    3401      112092 :       txn->acc_pool_idx = (uint)acc_idx;
    3402      112092 :       uint txn_idx = (uint)txn_pool_idx( accdb->txn_pool, txn );
    3403      112092 :       for(;;) {
    3404      112092 :         uint old_head = FD_VOLATILE_CONST( accdb->fork_pool[ accs[ i ]._fork_id ].shmem->txn_head );
    3405      112092 :         txn->fork.next = old_head;
    3406      112092 :         if( FD_LIKELY( FD_ATOMIC_CAS( &accdb->fork_pool[ accs[ i ]._fork_id ].shmem->txn_head, old_head, txn_idx )==old_head ) ) break;
    3407           0 :         FD_SPIN_PAUSE();
    3408           0 :       }
    3409             : 
    3410      112092 :       FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->shmetrics->accounts_total, 1UL );
    3411      112092 :     }
    3412      115629 :   }
    3413             : 
    3414             :   // STEP 3.
    3415             :   //   Finally, we release the cache class reservations we took at the
    3416             :   //   beginning when we acquired these cache lines.  Credits return
    3417             :   //   directly to the shared pool so other threads can use them
    3418             :   //   immediately.
    3419             : 
    3420      311322 :   ulong refund[ FD_ACCDB_CACHE_CLASS_CNT ] = {0};
    3421      623490 :   for( ulong i=0UL; i<accs_cnt; i++ ) {
    3422      312168 :     if( FD_LIKELY( accs[ i ]._original_size_class!=ULONG_MAX ) ) {
    3423       97170 :       if( FD_UNLIKELY( accdb->shmem->cache_class_used[ accs[ i ]._original_size_class ].val!=ULONG_MAX ) ) {
    3424           3 :         refund[ accs[ i ]._original_size_class ]++;
    3425           3 :       }
    3426       97170 :     }
    3427      312168 :     if( FD_UNLIKELY( accs[ i ]._writable ) ) {
    3428     1046952 :       for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) {
    3429      930624 :         if( FD_UNLIKELY( accdb->shmem->cache_class_used[ j ].val!=ULONG_MAX ) ) {
    3430          54 :           refund[ j ]++;
    3431          54 :         }
    3432      930624 :       }
    3433      116328 :     }
    3434      312168 :   }
    3435     2801898 :   for( ulong k=0UL; k<FD_ACCDB_CACHE_CLASS_CNT; k++ ) {
    3436     2490576 :     if( FD_UNLIKELY( refund[ k ] ) ) FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->cache_class_used[ k ].val, refund[ k ] );
    3437     2490576 :   }
    3438             : 
    3439      311322 :   FD_COMPILER_MFENCE();
    3440      311322 :   FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
    3441      311322 : }
    3442             : 
    3443             : void
    3444             : fd_accdb_release( fd_accdb_t * accdb,
    3445             :                   ulong        accs_cnt,
    3446      311052 :                   fd_acc_t *   accs ) {
    3447      311052 :   FD_TEST( accdb->acquire_state==FD_ACCDB_ACQUIRE_STATE_OPEN );
    3448      311052 :   release_inner( accdb, accs_cnt, accs );
    3449      311052 :   accdb->acquire_state = FD_ACCDB_ACQUIRE_STATE_IDLE;
    3450      311052 : }
    3451             : 
    3452             : void
    3453             : fd_accdb_release_ab( fd_accdb_t * accdb,
    3454             :                      ulong        accs_cnt,
    3455             :                      fd_acc_t *   accs,
    3456             :                      ulong        execs_cnt,
    3457         234 :                      fd_acc_t *   execs ) {
    3458         234 :   FD_TEST( accdb->acquire_state==FD_ACCDB_ACQUIRE_STATE_OPEN );
    3459         234 :   release_inner( accdb, accs_cnt, accs );
    3460         234 :   if( FD_LIKELY( execs_cnt ) ) release_inner( accdb, execs_cnt, execs );
    3461         234 :   accdb->acquire_state = FD_ACCDB_ACQUIRE_STATE_IDLE;
    3462         234 : }
    3463             : 
    3464             : fd_acc_t
    3465             : fd_accdb_read_one( fd_accdb_t *       accdb,
    3466             :                    fd_accdb_fork_id_t fork_id,
    3467      194028 :                    uchar const *      pubkey ) {
    3468      194028 :   fd_acc_t acc;
    3469      194028 :   fd_accdb_acquire( accdb, fork_id, 1UL, &pubkey, (int[]){0}, &acc );
    3470      194028 :   return acc;
    3471      194028 : }
    3472             : 
    3473             : void
    3474             : fd_accdb_unread_one( fd_accdb_t * accdb,
    3475      194028 :                      fd_acc_t *   acc ) {
    3476      194028 :   fd_accdb_release( accdb, 1UL, acc );
    3477      194028 : }
    3478             : 
    3479             : fd_acc_t
    3480             : fd_accdb_write_one( fd_accdb_t *       accdb,
    3481             :                     fd_accdb_fork_id_t fork_id,
    3482      114204 :                     uchar const *      pubkey ) {
    3483      114204 :   fd_acc_t acc;
    3484      114204 :   fd_accdb_acquire( accdb, fork_id, 1UL, &pubkey, (int[]){1}, &acc );
    3485      114204 :   return acc;
    3486      114204 : }
    3487             : 
    3488             : void
    3489             : fd_accdb_unwrite_one( fd_accdb_t * accdb,
    3490      114204 :                       fd_acc_t *   acc ) {
    3491      114204 :   fd_accdb_release( accdb, 1UL, acc );
    3492      114204 : }
    3493             : 
    3494             : int
    3495             : fd_accdb_read_one_nocache( fd_accdb_t *       accdb,
    3496             :                            fd_accdb_fork_id_t fork_id,
    3497             :                            uchar const *      pubkey,
    3498             :                            ulong *            out_lamports,
    3499             :                            int *              out_executable,
    3500             :                            uchar *            out_owner,
    3501             :                            uchar *            out_data,
    3502           0 :                            ulong *            out_data_len ) {
    3503             :   /* Publish epoch — protects against compaction freeing the partition
    3504             :      under us during the preadv2 path.  This is the only write the
    3505             :      readonly joiner makes into accdb shmem (and the pointer it stores
    3506             :      through is mapped through a separately-mmap'd writable page that
    3507             :      aliases shmem->joiner_epochs[idx]). */
    3508           0 :   FD_COMPILER_MFENCE();
    3509           0 :   FD_VOLATILE( *accdb->my_epoch_slot ) = FD_VOLATILE_CONST( accdb->shmem->epoch );
    3510           0 :   FD_HW_MFENCE();
    3511             : 
    3512             :   /// STEP 1.
    3513             :   ///   Walk the hash chain at acc_map[hash(pubkey)] using the same
    3514             :   //    visibility test as fd_accdb_acquire_inner.  See that function
    3515             :   //    for the detailed safety argument under concurrent prepend.
    3516           0 :   uint root_generation = accdb->fork_pool[ accdb->shmem->root_fork_id.val ].shmem->generation;
    3517           0 :   fd_accdb_fork_t * fork = &accdb->fork_pool[ fork_id.val ];
    3518           0 :   ulong hash = fd_hash32( pubkey, accdb->shmem->seed )&(accdb->shmem->chain_cnt-1UL);
    3519           0 :   uint acc_idx = FD_VOLATILE_CONST( accdb->acc_map[ hash ] );
    3520           0 :   fd_accdb_accmeta_t const * accmeta = NULL;
    3521           0 :   while( acc_idx!=UINT_MAX ) {
    3522           0 :     fd_accdb_accmeta_t const * candidate = &accdb->acc_pool[ acc_idx ];
    3523           0 :     uint next_idx = FD_VOLATILE_CONST( candidate->map.next );
    3524           0 :     if( FD_UNLIKELY( (candidate->key.generation>root_generation &&
    3525           0 :                       fd_accdb_acc_fork_id(candidate)!=fork_id.val &&
    3526           0 :                       !descends_set_test( fork->descends, fd_accdb_acc_fork_id(candidate) )) ) ||
    3527           0 :                      memcmp( pubkey, candidate->key.pubkey, 32UL ) ) {
    3528           0 :       acc_idx = next_idx;
    3529           0 :       continue;
    3530           0 :     }
    3531           0 :     accmeta = candidate;
    3532           0 :     break;
    3533           0 :   }
    3534             : 
    3535           0 :   if( FD_UNLIKELY( !accmeta ) ) {
    3536           0 :     accdb->metrics->accounts_acquired_per_class[ 0 ]++;
    3537           0 :     *out_lamports = 0UL;
    3538           0 :     FD_COMPILER_MFENCE();
    3539           0 :     FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
    3540           0 :     return FD_ACCDB_READ_ONE_NOCACHE_MISS;
    3541           0 :   }
    3542             : 
    3543             :   /// STEP 2.
    3544             :   ///   Snapshot acc fields.  The acc element's metadata is effectively
    3545             :   ///   immutable from the perspective of cross-fork readers (see the
    3546             :   ///   comment block in fd_accdb.h about cross-fork reads). */
    3547           0 :   uint  snap_es       = FD_VOLATILE_CONST( accmeta->executable_size );
    3548           0 :   uint  snap_gen      = accmeta->key.generation;
    3549           0 :   ulong snap_lamports = accmeta->lamports;
    3550           0 :   uint  snap_cidx     = FD_VOLATILE_CONST( accmeta->cache_idx );
    3551           0 :   ulong data_len      = (ulong)FD_ACCDB_SIZE_DATA( snap_es );
    3552           0 :   int   executable    = FD_ACCDB_SIZE_EXEC( snap_es );
    3553             : 
    3554           0 :   accdb->metrics->accounts_acquired_per_class[ fd_accdb_cache_class( data_len ) ]++;
    3555             : 
    3556           0 :   if( FD_UNLIKELY( !snap_lamports ) ) {
    3557           0 :     *out_lamports = 0UL;
    3558           0 :     FD_COMPILER_MFENCE();
    3559           0 :     FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
    3560           0 :     return FD_ACCDB_READ_ONE_NOCACHE_MISS;
    3561           0 :   }
    3562             : 
    3563             :   /// STEP 3.
    3564             :   ///    Cache hit fast path with try-read-test (ABA) loop.  Same
    3565             :   ///    primitives as cache_try_pin: re-check key.generation + pubkey
    3566             :   ///    before and after the bulk copy, and bail to the disk path if the
    3567             :   ///    line was claimed for eviction (refcnt ==
    3568             :   ///    FD_ACCDB_EVICT_SENTINEL).  No CAS on refcnt, we never pin the
    3569             :   ///    line.
    3570           0 :   if( FD_LIKELY( FD_ACCDB_SIZE_CACHE_VALID( snap_es ) && snap_cidx!=FD_ACCDB_ACC_CIDX_INVAL ) ) {
    3571           0 :     ulong cls = FD_ACCDB_ACC_CIDX_CLASS( snap_cidx );
    3572           0 :     ulong idx = FD_ACCDB_ACC_CIDX_IDX  ( snap_cidx );
    3573           0 :     fd_accdb_cache_line_t * line = cache_line( accdb, cls, idx );
    3574             : 
    3575           0 :     for(;;) {
    3576           0 :       uint gen0 = FD_VOLATILE_CONST( line->key.generation );
    3577           0 :       uint rc0  = FD_VOLATILE_CONST( line->refcnt );
    3578           0 :       uint ai0  = FD_VOLATILE_CONST( line->acc_idx );
    3579           0 :       if( FD_UNLIKELY( rc0==FD_ACCDB_EVICT_SENTINEL ) ) goto miss;
    3580           0 :       if( FD_UNLIKELY( gen0!=snap_gen ) ) goto miss;
    3581           0 :       if( FD_UNLIKELY( memcmp( line->key.pubkey, pubkey, 32UL ) ) ) goto miss;
    3582             :       /* acc_idx==UINT_MAX is the "loading" sentinel set by cold_load_acc
    3583             :          before the preadv2 fills the line.  CACHE_VALID can be observed
    3584             :          set while the bytes are still stale, so fall to the disk path
    3585             :          (which spins on offset_fork and reads from the file) rather
    3586             :          than copying garbage. */
    3587           0 :       if( FD_UNLIKELY( ai0==UINT_MAX ) ) goto miss;
    3588             : 
    3589           0 :       FD_COMPILER_MFENCE();
    3590           0 :       memcpy( out_owner, line->owner, 32UL );
    3591           0 :       memcpy( out_data,  (uchar const *)(line+1UL), data_len );
    3592           0 :       FD_COMPILER_MFENCE();
    3593             : 
    3594           0 :       uint gen1 = FD_VOLATILE_CONST( line->key.generation );
    3595           0 :       uint rc1  = FD_VOLATILE_CONST( line->refcnt );
    3596           0 :       uint ai1  = FD_VOLATILE_CONST( line->acc_idx );
    3597           0 :       if( FD_UNLIKELY( rc1==FD_ACCDB_EVICT_SENTINEL ) ) goto miss;
    3598           0 :       if( FD_UNLIKELY( gen1!=snap_gen ) ) goto miss;
    3599           0 :       if( FD_UNLIKELY( memcmp( line->key.pubkey, pubkey, 32UL ) ) ) goto miss;
    3600           0 :       if( FD_UNLIKELY( ai1==UINT_MAX ) ) goto miss;
    3601             : 
    3602           0 :       *out_lamports   = snap_lamports;
    3603           0 :       *out_executable = executable;
    3604           0 :       *out_data_len   = data_len;
    3605           0 :       accdb->metrics->bytes_copied += data_len;
    3606           0 :       FD_COMPILER_MFENCE();
    3607           0 :       FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
    3608           0 :       return FD_ACCDB_READ_ONE_NOCACHE_CACHE;
    3609           0 :     }
    3610           0 :   }
    3611             : 
    3612           0 : miss:;
    3613           0 :   accdb->metrics->accounts_not_found_per_class[ fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( snap_es ) ) ]++;
    3614             : 
    3615             :   /// STEP 4.
    3616             :   ///   Disk path.  Spin until the writer publishes a real offset
    3617             :   ///   (matches STEP 10 of fd_accdb_acquire_inner).  Compaction may
    3618             :   ///   concurrently relocate the record, but our published epoch
    3619             :   ///   prevents the source partition from being freed until we exit
    3620             :   ///   our critical section, so the bytes at the snapshotted offset
    3621             :   ///   remain stable for the duration of the read.
    3622           0 :   fd_racesan_hook( "accdb_nocache:pre_offset" );
    3623           0 :   ulong off_packed = FD_VOLATILE_CONST( accmeta->offset_fork );
    3624           0 :   if( FD_UNLIKELY( (off_packed & FD_ACCDB_OFF_MASK)==FD_ACCDB_OFF_INVAL ) ) {
    3625           0 :     accdb->metrics->accounts_waited++;
    3626           0 :     while( FD_UNLIKELY( ((off_packed=FD_VOLATILE_CONST( accmeta->offset_fork )) & FD_ACCDB_OFF_MASK)==FD_ACCDB_OFF_INVAL ) ) FD_SPIN_PAUSE();
    3627           0 :   }
    3628           0 :   ulong off = off_packed & FD_ACCDB_OFF_MASK;
    3629           0 :   fd_racesan_hook( "accdb_nocache:pre_preadv2" );
    3630             : 
    3631           0 :   struct iovec iovs[ 2 ] = {
    3632           0 :     { .iov_base = out_owner, .iov_len = 32UL     },
    3633           0 :     { .iov_base = out_data,  .iov_len = data_len },
    3634           0 :   };
    3635           0 :   ulong total = 32UL+data_len;
    3636           0 :   ulong start = off+offsetof( fd_accdb_disk_meta_t, owner );
    3637           0 :   ulong got   = 0UL;
    3638           0 :   int   nio   = data_len ? 2 : 1;
    3639           0 :   while( FD_LIKELY( got<total ) ) {
    3640           0 :     long result = preadv2( accdb->fd, iovs, nio, (long)(start+got), 0 );
    3641           0 :     if( FD_UNLIKELY( -1==result && (errno==EINTR || errno==EAGAIN || errno==EWOULDBLOCK) ) ) continue;
    3642           0 :     else if( FD_UNLIKELY( -1==result ) ) FD_LOG_ERR(( "preadv2() failed (%d-%s)", errno, fd_io_strerror( errno ) ));
    3643           0 :     else if( FD_UNLIKELY( !result ) ) FD_LOG_ERR(( "accounts database is corrupt, data expected at offset %lu with size %lu exceeded file extents", start+got, total ));
    3644           0 :     fd_accdb_partition_read_bump( accdb, start+got, (ulong)result );
    3645           0 :     got += (ulong)result;
    3646           0 :     accdb->metrics->bytes_read += (ulong)result;
    3647           0 :     accdb->metrics->read_ops++;
    3648             : 
    3649           0 :     long r = result;
    3650           0 :     for( int v=0; v<nio; v++ ) {
    3651           0 :       if( (ulong)r>=iovs[ v ].iov_len ) {
    3652           0 :         r -= (long)iovs[ v ].iov_len;
    3653           0 :         iovs[ v ].iov_len = 0UL;
    3654           0 :       } else {
    3655           0 :         iovs[ v ].iov_base = (uchar *)iovs[ v ].iov_base + r;
    3656           0 :         iovs[ v ].iov_len -= (ulong)r;
    3657           0 :         break;
    3658           0 :       }
    3659           0 :     }
    3660           0 :   }
    3661             : 
    3662           0 :   *out_lamports   = snap_lamports;
    3663           0 :   *out_executable = executable;
    3664           0 :   *out_data_len   = data_len;
    3665             : 
    3666           0 :   FD_COMPILER_MFENCE();
    3667           0 :   FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
    3668           0 :   return FD_ACCDB_READ_ONE_NOCACHE_DISK;
    3669           0 : }
    3670             : 
    3671             : int
    3672             : fd_accdb_exists( fd_accdb_t *       accdb,
    3673             :                  fd_accdb_fork_id_t fork_id,
    3674         111 :                  uchar const *      pubkey ) {
    3675         111 :   FD_COMPILER_MFENCE();
    3676         111 :   FD_VOLATILE( *accdb->my_epoch_slot ) = FD_VOLATILE_CONST( accdb->shmem->epoch );
    3677         111 :   FD_HW_MFENCE();
    3678             : 
    3679         111 :   uint root_generation = accdb->fork_pool[ accdb->shmem->root_fork_id.val ].shmem->generation;
    3680         111 :   fd_accdb_fork_t * fork = &accdb->fork_pool[ fork_id.val ];
    3681         111 :   ulong hash = fd_hash32( pubkey, accdb->shmem->seed )&(accdb->shmem->chain_cnt-1UL);
    3682         111 :   uint acc = FD_VOLATILE_CONST( accdb->acc_map[ hash ] );
    3683         180 :   while( acc!=UINT_MAX ) {
    3684         174 :     fd_accdb_accmeta_t const * candidate_acc = &accdb->acc_pool[ acc ];
    3685         174 :     uint next_acc = FD_VOLATILE_CONST( candidate_acc->map.next );
    3686             : 
    3687         174 :     if( FD_UNLIKELY( (candidate_acc->key.generation>root_generation && fd_accdb_acc_fork_id(candidate_acc)!=fork_id.val && !descends_set_test( fork->descends, fd_accdb_acc_fork_id(candidate_acc) )) ) || memcmp( pubkey, candidate_acc->key.pubkey, 32UL ) ) {
    3688          69 :       acc = next_acc;
    3689          69 :       continue;
    3690          69 :     }
    3691             : 
    3692         105 :     break;
    3693         174 :   }
    3694             : 
    3695         111 :   int result;
    3696         111 :   if( FD_UNLIKELY( acc==UINT_MAX ) ) result = 0;
    3697         105 :   else                               result = !!FD_VOLATILE_CONST( accdb->acc_pool[ acc ].lamports );
    3698             : 
    3699         111 :   FD_COMPILER_MFENCE();
    3700         111 :   FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
    3701         111 :   return result;
    3702         111 : }
    3703             : 
    3704             : int
    3705             : fd_accdb_probe_pd_this_fork( fd_accdb_t *       accdb,
    3706             :                              fd_accdb_fork_id_t fork_id,
    3707             :                              uchar const *      pubkey,
    3708             :                              int *              out_pd_write,
    3709             :                              ulong *            out_data_len,
    3710         114 :                              ulong *            out_lamports ) {
    3711         114 :   FD_COMPILER_MFENCE();
    3712         114 :   FD_VOLATILE( *accdb->my_epoch_slot ) = FD_VOLATILE_CONST( accdb->shmem->epoch );
    3713         114 :   FD_HW_MFENCE();
    3714             : 
    3715         114 :   uint root_generation = accdb->fork_pool[ accdb->shmem->root_fork_id.val ].shmem->generation;
    3716         114 :   fd_accdb_fork_t * fork = &accdb->fork_pool[ fork_id.val ];
    3717         114 :   ulong hash = fd_hash32( pubkey, accdb->shmem->seed )&(accdb->shmem->chain_cnt-1UL);
    3718         114 :   uint acc = FD_VOLATILE_CONST( accdb->acc_map[ hash ] );
    3719         114 :   while( acc!=UINT_MAX ) {
    3720         111 :     fd_accdb_accmeta_t const * candidate_acc = &accdb->acc_pool[ acc ];
    3721         111 :     uint next_acc = FD_VOLATILE_CONST( candidate_acc->map.next );
    3722             : 
    3723         111 :     if( FD_UNLIKELY( (candidate_acc->key.generation>root_generation && fd_accdb_acc_fork_id(candidate_acc)!=fork_id.val && !descends_set_test( fork->descends, fd_accdb_acc_fork_id(candidate_acc) )) ) || memcmp( pubkey, candidate_acc->key.pubkey, 32UL ) ) {
    3724           0 :       acc = next_acc;
    3725           0 :       continue;
    3726           0 :     }
    3727             : 
    3728         111 :     break;
    3729         111 :   }
    3730             : 
    3731         114 :   int   pd        = 0;
    3732         114 :   int   gen_match = 0;
    3733         114 :   ulong len       = 0UL;
    3734         114 :   ulong lamports  = 0UL;
    3735         114 :   if( FD_LIKELY( acc!=UINT_MAX ) ) {
    3736         111 :     fd_accdb_accmeta_t const * m = &accdb->acc_pool[ acc ];
    3737         111 :     uint es   = FD_VOLATILE_CONST( m->executable_size );
    3738         111 :     gen_match = ( m->key.generation==fork->shmem->generation );
    3739         111 :     pd        = gen_match && FD_ACCDB_SIZE_PD_WRITE( es );
    3740         111 :     len       = FD_ACCDB_SIZE_DATA( es );
    3741         111 :     lamports  = FD_VOLATILE_CONST( m->lamports );
    3742         111 :   }
    3743             : 
    3744         114 :   FD_COMPILER_MFENCE();
    3745         114 :   FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
    3746             : 
    3747         114 :   *out_pd_write = pd;
    3748         114 :   if( gen_match ) {
    3749          99 :     *out_data_len = len;
    3750          99 :     *out_lamports = lamports;
    3751          99 :   }
    3752         114 :   return gen_match;
    3753         114 : }
    3754             : 
    3755             : ulong
    3756             : fd_accdb_lamports( fd_accdb_t *       accdb,
    3757             :                    fd_accdb_fork_id_t fork_id,
    3758        8754 :                    uchar const *      pubkey ) {
    3759        8754 :   FD_COMPILER_MFENCE();
    3760        8754 :   FD_VOLATILE( *accdb->my_epoch_slot ) = FD_VOLATILE_CONST( accdb->shmem->epoch );
    3761        8754 :   FD_HW_MFENCE();
    3762             : 
    3763        8754 :   uint root_generation = accdb->fork_pool[ accdb->shmem->root_fork_id.val ].shmem->generation;
    3764        8754 :   fd_accdb_fork_t * fork = &accdb->fork_pool[ fork_id.val ];
    3765        8754 :   ulong hash = fd_hash32( pubkey, accdb->shmem->seed )&(accdb->shmem->chain_cnt-1UL);
    3766        8754 :   uint acc = FD_VOLATILE_CONST( accdb->acc_map[ hash ] );
    3767       10092 :   while( acc!=UINT_MAX ) {
    3768        2313 :     fd_accdb_accmeta_t const * candidate_acc = &accdb->acc_pool[ acc ];
    3769        2313 :     uint next_acc = FD_VOLATILE_CONST( candidate_acc->map.next );
    3770             : 
    3771        2313 :     if( FD_UNLIKELY( (candidate_acc->key.generation>root_generation && fd_accdb_acc_fork_id(candidate_acc)!=fork_id.val && !descends_set_test( fork->descends, fd_accdb_acc_fork_id(candidate_acc) )) ) || memcmp( pubkey, candidate_acc->key.pubkey, 32UL ) ) {
    3772        1338 :       acc = next_acc;
    3773        1338 :       continue;
    3774        1338 :     }
    3775             : 
    3776         975 :     break;
    3777        2313 :   }
    3778             : 
    3779        8754 :   ulong result;
    3780        8754 :   if( FD_UNLIKELY( acc==UINT_MAX ) ) result = 0UL;
    3781         975 :   else                               result = FD_VOLATILE_CONST( accdb->acc_pool[ acc ].lamports );
    3782             : 
    3783        8754 :   FD_COMPILER_MFENCE();
    3784        8754 :   FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
    3785        8754 :   return result;
    3786        8754 : }
    3787             : 
    3788             : /* cache_bg_evict pre-evicts cache lines in the background to keep the
    3789             :    per-class CAS free lists populated ahead of demand.  For each class
    3790             :   whose immediately available capacity has dropped below low_water,
    3791             :   a bounded CLOCK sweep claims lines, writes dirty ones to disk, and
    3792             :   pushes them onto the free list until available capacity reaches
    3793             :   target.  Immediately available capacity includes both the CAS free
    3794             :   list and the never-initialized tail of the class, since foreground
    3795             :   allocators can consume either path without evicting resident data.
    3796             : 
    3797             :   Budget: at most 256 CLOCK ticks per class per invocation to keep the
    3798             :   background loop responsive.  The function is called every tick of
    3799             :   fd_accdb_background, so large refills happen across several ticks
    3800             :   rather than blocking.  The low_water / target thresholds are static
    3801             :   per-class watermarks computed at initialization; pre-eviction only
    3802             :   converts resident lines into free-list entries and does not consume
    3803             :   cache-slot reservations.
    3804             : 
    3805             :   force: when non-zero, ignore the watermark and sweep every line in
    3806             :   every class.  Always 0 in normal operation; used only by
    3807             :   test_accdb_racesan to deterministically exercise the writeback path
    3808             :   without manufacturing real cache pressure. */
    3809             : 
    3810             : static void
    3811             : background_preevict( fd_accdb_t * accdb,
    3812             :                      int *        charge_busy,
    3813     3268415 :                      int          force ) {
    3814     3268415 :   fd_accdb_shmem_t * shmem = accdb->shmem;
    3815             : 
    3816    29415735 :   for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) {
    3817    26147320 :     ulong target = shmem->cache_free_target[ c ];
    3818    26147320 :     ulong max_c  = shmem->cache_class_max[ c ];
    3819    26147320 :     ulong init   = fd_ulong_min( FD_VOLATILE_CONST( shmem->cache_class_init[ c ].val ), max_c );
    3820    26147320 :     ulong freec  = FD_VOLATILE_CONST( shmem->cache_free_cnt[ c ].val );
    3821    26147320 :     ulong live   = init>freec ? init-freec : 0UL;
    3822    26147320 :     ulong avail  = max_c-live;
    3823    26147320 :     if( FD_LIKELY( !force && avail>=shmem->cache_free_low_water[ c ] ) ) continue;
    3824             : 
    3825           0 :     *charge_busy = 1;
    3826             : 
    3827           0 :     ulong budget  = force ? init : 256UL;
    3828           0 :     ulong evicted = 0UL;
    3829           0 :     if( FD_UNLIKELY( force ) ) target = max_c; /* sweep everything */
    3830             : 
    3831           0 :     for( ulong tick=0UL; tick<budget && avail+evicted<target; tick++ ) {
    3832             :       /* Only sweep the lazily initialized prefix.  cache_class_init
    3833             :          may transiently exceed max_c during the acquire_cache_line
    3834             :          overflow/undo path, so clamp it before using it as the wrap
    3835             :          bound. */
    3836           0 :       init = fd_ulong_min( FD_VOLATILE_CONST( shmem->cache_class_init[ c ].val ), max_c );
    3837           0 :       if( FD_UNLIKELY( !init ) ) break;
    3838             : 
    3839           0 :       ulong hand = FD_ATOMIC_FETCH_AND_ADD( &shmem->clock_hand[ c ].val, 1UL ) % init;
    3840             : 
    3841           0 :       fd_accdb_cache_line_t * line = cache_line( accdb, c, hand );
    3842             : 
    3843           0 :       if( FD_UNLIKELY( line->key.generation==UINT_MAX && line->acc_idx==UINT_MAX ) ) continue;
    3844             : 
    3845           0 :       uint rc = FD_VOLATILE_CONST( line->refcnt );
    3846           0 :       if( FD_UNLIKELY( rc ) ) continue;
    3847             : 
    3848           0 :       if( FD_UNLIKELY( line->referenced ) ) {
    3849           0 :         line->referenced = 0;
    3850           0 :         continue;
    3851           0 :       }
    3852             : 
    3853           0 :       if( FD_UNLIKELY( FD_ATOMIC_CAS( &line->refcnt, 0U, FD_ACCDB_EVICT_SENTINEL )!=0U ) ) continue;
    3854             : 
    3855           0 :       if( FD_UNLIKELY( line->acc_idx==UINT_MAX && line->key.generation==UINT_MAX ) ) {
    3856           0 :         FD_VOLATILE( line->refcnt ) = 0;
    3857           0 :         continue;
    3858           0 :       }
    3859             : 
    3860           0 :       uint acc_idx = line->acc_idx;
    3861             : #if FD_TMPL_USE_HANDHOLDING
    3862             :       uint line_gen FD_FN_UNUSED = line->key.generation;
    3863             : #endif
    3864           0 :       if( FD_LIKELY( acc_idx!=UINT_MAX ) ) {
    3865           0 :         evict_clear_acc_cache_ref( &accdb->acc_pool[ acc_idx ], c, hand );
    3866           0 :       }
    3867           0 :       line->key.generation = UINT_MAX;
    3868           0 :       if( FD_UNLIKELY( !line->persisted && acc_idx!=UINT_MAX ) ) {
    3869           0 :         fd_accdb_accmeta_t * accmeta = &accdb->acc_pool[ acc_idx ];
    3870             : 
    3871             :         /* advertise to external observers that write is in progress */
    3872           0 :         FD_COMPILER_MFENCE();
    3873           0 :         FD_VOLATILE( *accdb->my_epoch_slot ) = FD_VOLATILE_CONST( accdb->shmem->epoch );
    3874           0 :         FD_HW_MFENCE();
    3875             : 
    3876           0 :         fd_racesan_hook( "preevict:pre_synth" );
    3877             : #if FD_TMPL_USE_HANDHOLDING
    3878             :         FD_TEST( line_gen==accmeta->key.generation &&
    3879             :                  !memcmp( line->key.pubkey, accmeta->key.pubkey, 32UL ) );
    3880             : #endif
    3881           0 :         ulong entry_sz = sizeof(fd_accdb_disk_meta_t)+(ulong)FD_ACCDB_SIZE_DATA( accmeta->executable_size );
    3882             : 
    3883             :         /* Atomically swap the old offset to FD_ACCDB_OFF_INVAL so that
    3884             :            a concurrent compaction CAS (old_offset -> dest_offset)
    3885             :            cannot succeed between our read and our later store of
    3886             :            the new file_off.  Without the exchange, compaction could
    3887             :            relocate the record, then our plain store would overwrite
    3888             :            the relocated offset, leaving the compaction destination
    3889             :            as unreachable dead space whose bytes are never freed. */
    3890           0 :         ulong old_offset = fd_accdb_acc_xchg_offset( accmeta, FD_ACCDB_OFF_INVAL );
    3891           0 :         if( FD_LIKELY( old_offset!=FD_ACCDB_OFF_INVAL ) ) {
    3892           0 :           fd_accdb_shmem_bytes_freed( shmem, old_offset, entry_sz );
    3893           0 :           FD_ATOMIC_FETCH_AND_SUB( &shmem->shmetrics->disk_used_bytes, entry_sz );
    3894           0 :         }
    3895             : 
    3896           0 :         fd_accdb_disk_meta_t meta;
    3897           0 :         fd_memcpy( meta.pubkey, accmeta->key.pubkey, 32UL );
    3898           0 :         meta.size       = FD_ACCDB_SIZE_DATA( accmeta->executable_size );
    3899           0 :         meta.generation = accmeta->key.generation;
    3900           0 :         fd_memcpy( meta.owner, line->owner, 32UL );
    3901             : 
    3902           0 :         struct iovec iovs[ 2UL ] = {
    3903           0 :           { .iov_base = &meta,              .iov_len = sizeof(fd_accdb_disk_meta_t) },
    3904           0 :           { .iov_base = (void *)(line+1UL), .iov_len = FD_ACCDB_SIZE_DATA( accmeta->executable_size ) }
    3905           0 :         };
    3906             : 
    3907           0 :         ulong file_off = allocate_next_write( accdb, entry_sz );
    3908           0 :         ulong written = 0UL;
    3909           0 :         while( written<entry_sz ) {
    3910           0 :           long result = pwritev2( accdb->fd, iovs, 2, (long)(file_off+written), 0 );
    3911           0 :           if( FD_UNLIKELY( result==-1 && errno==EINTR ) ) continue;
    3912           0 :           else if( FD_UNLIKELY( result<=0 ) ) FD_LOG_ERR(( "pwritev2() failed (%d-%s)", errno, fd_io_strerror( errno ) ));
    3913           0 :           written += (ulong)result;
    3914           0 :           accdb->metrics->bytes_written += (ulong)result;
    3915           0 :           accdb->metrics->write_ops++;
    3916             : 
    3917           0 :           for( int v=0; v<2; v++ ) {
    3918           0 :             if( (ulong)result>=iovs[ v ].iov_len ) {
    3919           0 :               result -= (long)iovs[ v ].iov_len;
    3920           0 :               iovs[ v ].iov_len = 0UL;
    3921           0 :             } else {
    3922           0 :               iovs[ v ].iov_base = (uchar *)iovs[ v ].iov_base + result;
    3923           0 :               iovs[ v ].iov_len -= (ulong)result;
    3924           0 :               break;
    3925           0 :             }
    3926           0 :           }
    3927           0 :         }
    3928             : 
    3929           0 :         FD_COMPILER_MFENCE();
    3930           0 :         accmeta->offset_fork = fd_accdb_acc_pack_offset_fork( file_off, fd_accdb_acc_fork_id(accmeta) );
    3931           0 :         FD_ATOMIC_FETCH_AND_ADD( &shmem->shmetrics->disk_used_bytes, entry_sz );
    3932             : 
    3933           0 :         accdb->metrics->accounts_preevicted++;
    3934           0 :         accdb->metrics->accounts_preevicted_per_class[ c ]++;
    3935             : 
    3936           0 :         FD_COMPILER_MFENCE();
    3937           0 :         FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
    3938           0 :       }
    3939             : 
    3940           0 :       line->persisted      = 1;
    3941           0 :       line->acc_idx        = UINT_MAX;
    3942           0 :       line->key.generation = UINT_MAX;
    3943           0 :       FD_COMPILER_MFENCE();
    3944           0 :       FD_VOLATILE( line->refcnt ) = 0;
    3945           0 :       cache_free_push( accdb, c, line );
    3946           0 :       evicted++;
    3947           0 :     }
    3948           0 :   }
    3949     3268415 : }
    3950             : 
    3951             : int
    3952             : fd_accdb_snapshot_write_one( fd_accdb_t *       accdb,
    3953             :                              fd_accdb_fork_id_t fork_id,
    3954             :                              uchar const *      pubkey,
    3955             :                              ulong              slot,
    3956             :                              ulong              lamports,
    3957             :                              ulong              data_len,
    3958             :                              int                executable,
    3959          69 :                              ulong *            out_replaced_lamports ) {
    3960             :   /* Snapshot slots are stored in the 32-bit cache_idx scratch field
    3961             :      during loading.  Reject anything that would truncate. */
    3962          69 :   if( FD_UNLIKELY( slot>UINT_MAX ) ) FD_LOG_ERR(( "snapshot slot %lu exceeds 2^32-1, accdb format must be widened", slot ));
    3963             : 
    3964          69 :   int incremental = fork_id.val!=USHORT_MAX;
    3965             : 
    3966          69 :   fd_accdb_fork_t * fork     = NULL;
    3967          69 :   uint              fork_gen = 0U;
    3968          69 :   if( FD_UNLIKELY( incremental ) ) {
    3969          33 :     fork     = &accdb->fork_pool[ fork_id.val ];
    3970          33 :     fork_gen = fork->shmem->generation;
    3971          33 :   }
    3972             : 
    3973          69 :   ulong hash = fd_hash32( pubkey, accdb->shmem->seed )&(accdb->shmem->chain_cnt-1UL);
    3974             : 
    3975          69 :   *out_replaced_lamports = 0UL;
    3976             : 
    3977          69 :   fd_accdb_accmeta_t * accmeta = NULL;
    3978          69 :   int cross_fork = 0; /* incremental only: existing entry from different fork */
    3979             : 
    3980          69 :   ulong next_acc = accdb->acc_map[ hash ];
    3981          75 :   while( next_acc!=UINT_MAX ) {
    3982          12 :     fd_accdb_accmeta_t * candidate_acc = &accdb->acc_pool[ next_acc ];
    3983          12 :     if( FD_UNLIKELY( !memcmp( pubkey, candidate_acc->key.pubkey, 32UL ) ) ) {
    3984           6 :       if( FD_LIKELY( (ulong)candidate_acc->cache_idx>slot ) ) {
    3985             :         /* Still advance the write head so snapwr and snapin stay in
    3986             :            sync — snapwr unconditionally writes every account to disk.
    3987             :            Mark the space as immediately freed since it is dead on
    3988             :            arrival. */
    3989           0 :         ulong dead_sz  = sizeof(fd_accdb_disk_meta_t)+data_len;
    3990           0 :         ulong dead_off = allocate_next_write( accdb, dead_sz );
    3991           0 :         fd_accdb_shmem_bytes_freed( accdb->shmem, dead_off, dead_sz );
    3992           0 :         return -1;
    3993           0 :       }
    3994           6 :       if( FD_UNLIKELY( incremental ) && candidate_acc->key.generation!=fork_gen ) {
    3995             :         /* Cross-snapshot override: don't replace in-place; insert a
    3996             :            new entry alongside the old one so purge can revert. */
    3997           6 :         cross_fork = 1;
    3998           6 :         *out_replaced_lamports = candidate_acc->lamports;
    3999           6 :       } else {
    4000             :         /* Same-fork duplicate (or full-snapshot mode): replace in-place */
    4001           0 :         accmeta = candidate_acc;
    4002           0 :       }
    4003           6 :       break;
    4004           6 :     }
    4005           6 :     next_acc = candidate_acc->map.next;
    4006           6 :   }
    4007             : 
    4008          69 :   int replace = !!accmeta;
    4009             : 
    4010          69 :   if( FD_UNLIKELY( !accmeta ) ) {
    4011          69 :     accmeta = acc_pool_acquire_nolock( accdb->acc_pool_join );
    4012          69 :     if( FD_UNLIKELY( !accmeta ) ) FD_LOG_ERR(( "accounts database ran out of space during snapshot loading, increase [accounts.max_accounts], current value is %lu", acc_pool_ele_max( accdb->acc_pool_join ) ));
    4013             : 
    4014          69 :     uint acc_idx = (uint)acc_pool_idx( accdb->acc_pool_join, accmeta );
    4015             : 
    4016          69 :     fd_memcpy( accmeta->key.pubkey, pubkey, 32UL );
    4017          69 :     if( FD_UNLIKELY( !incremental && accdb->shmem->root_fork_id.val==USHORT_MAX ) ) {
    4018           0 :       FD_LOG_ERR(( "snapshot_write_one called without a root fork attached" ));
    4019           0 :     }
    4020          69 :     accmeta->key.generation = incremental ? fork_gen : accdb->fork_pool[ accdb->shmem->root_fork_id.val ].shmem->generation;
    4021          69 :     accmeta->map.next = accdb->acc_map[ hash ];
    4022          69 :     accdb->acc_map[ hash ] = acc_idx;
    4023             : 
    4024             :     /* In incremental mode, record this insert in the fork's txn list
    4025             :        so purge can find and unlink it on failure. */
    4026          69 :     if( FD_UNLIKELY( incremental ) ) {
    4027          33 :       fd_accdb_txn_t * txn = txn_pool_acquire( accdb->txn_pool );
    4028          33 :       if( FD_UNLIKELY( !txn ) ) FD_LOG_ERR(( "txn pool exhausted during incremental snapshot loading" ));
    4029          33 :       txn->acc_pool_idx = acc_idx;
    4030          33 :       uint txn_idx      = (uint)txn_pool_idx( accdb->txn_pool, txn );
    4031          33 :       txn->fork.next          = fork->shmem->txn_head;
    4032          33 :       fork->shmem->txn_head   = txn_idx;
    4033          33 :     }
    4034          69 :   }
    4035             : 
    4036          69 :   if( FD_UNLIKELY( replace ) ) {
    4037             :     /* The old version's disk space is now dead. */
    4038           0 :     ulong old_sz = sizeof(fd_accdb_disk_meta_t) + FD_ACCDB_SIZE_DATA( accmeta->executable_size );
    4039           0 :     fd_accdb_shmem_bytes_freed( accdb->shmem, fd_accdb_acc_offset( accmeta ), old_sz );
    4040           0 :     accdb->shmem->shmetrics->disk_used_bytes -= old_sz;
    4041           0 :     *out_replaced_lamports = accmeta->lamports;
    4042           0 :   }
    4043             : 
    4044          69 :   accmeta->cache_idx = (uint)slot;
    4045          69 :   accmeta->lamports = lamports;
    4046          69 :   accmeta->executable_size = FD_ACCDB_SIZE_PACK( (uint)data_len, executable );
    4047          69 :   ulong entry_sz = sizeof(fd_accdb_disk_meta_t)+data_len;
    4048          69 :   ulong file_off = allocate_next_write( accdb, entry_sz );
    4049          69 :   accmeta->offset_fork = incremental ? fd_accdb_acc_pack_offset_fork( file_off, fork_id.val ) : file_off;
    4050          69 :   accdb->shmem->shmetrics->disk_used_bytes += entry_sz;
    4051          69 :   if( !replace ) accdb->shmem->shmetrics->accounts_total++;
    4052             : 
    4053          69 :   return ( replace || cross_fork ) ? 2 : 1;
    4054          69 : }
    4055             : 
    4056             : int
    4057             : fd_accdb_snapshot_write_batch( fd_accdb_t *        accdb,
    4058             :                                fd_accdb_fork_id_t  fork_id,
    4059             :                                ulong               cnt,
    4060             :                                uchar const * const pubkeys[],
    4061             :                                ulong  const        slots[],
    4062             :                                ulong  const        lamports[],
    4063             :                                ulong  const        data_lens[],
    4064             :                                int    const        executables[],
    4065             :                                ulong *             accounts_ignored,
    4066             :                                ulong *             accounts_replaced,
    4067             :                                ulong *             accounts_loaded,
    4068             :                                ulong *             out_replaced_lamports,
    4069          12 :                                ulong *             out_ignored_lamports ) {
    4070          12 :   int incremental = fork_id.val!=USHORT_MAX;
    4071             : 
    4072          12 :   fd_accdb_fork_t * fork     = NULL;
    4073          12 :   uint              fork_gen = 0U;
    4074          12 :   if( FD_UNLIKELY( incremental ) ) {
    4075           3 :     fork     = &accdb->fork_pool[ fork_id.val ];
    4076           3 :     fork_gen = fork->shmem->generation;
    4077           3 :   }
    4078             : 
    4079          12 :   ulong seed      = accdb->shmem->seed;
    4080          12 :   ulong chain_msk = accdb->shmem->chain_cnt - 1UL;
    4081          12 :   if( FD_UNLIKELY( !incremental && accdb->shmem->root_fork_id.val==USHORT_MAX ) ) {
    4082           0 :     FD_LOG_ERR(( "snapshot_write_batch called without a root fork attached" ));
    4083           0 :   }
    4084          12 :   uint  gen       = incremental ? 0U : accdb->fork_pool[ accdb->shmem->root_fork_id.val ].shmem->generation;
    4085             : 
    4086          12 :   ulong ignored          = 0UL;
    4087          12 :   ulong replaced         = 0UL;
    4088          12 :   ulong loaded           = 0UL;
    4089          12 :   ulong cross_replaced   = 0UL; /* cross-fork overrides (subset of replaced) */
    4090          12 :   ulong replaced_lamports = 0UL;
    4091          12 :   ulong ignored_lamports  = 0UL;
    4092             : 
    4093             :   /* Snapshot slots are stored in the 32-bit cache_idx scratch field
    4094             :      during loading.  Reject anything that would truncate. */
    4095          42 :   for( ulong i=0UL; i<cnt; i++ ) {
    4096          30 :     if( FD_UNLIKELY( slots[ i ]>UINT_MAX ) ) FD_LOG_ERR(( "snapshot slot %lu exceeds 2^32-1, accdb format must be widened", slots[ i ] ));
    4097          30 :   }
    4098             : 
    4099             :   /* Phase 1: compute hashes and prefetch chain heads. */
    4100             : 
    4101          12 :   ulong                hashes[ 8 ];
    4102          12 :   fd_accdb_accmeta_t * existing[ 8 ];       /* same-fork dup or full-snapshot replace */
    4103          12 :   fd_accdb_accmeta_t * cross_existing[ 8 ]; /* cross-fork dup (incremental only) */
    4104          12 :   int                  skip[ 8 ];
    4105             : 
    4106          42 :   for( ulong i=0UL; i<cnt; i++ ) {
    4107          30 :     hashes[ i ]          = fd_hash32( pubkeys[ i ], seed ) & chain_msk;
    4108          30 :     existing[ i ]        = NULL;
    4109          30 :     cross_existing[ i ]  = NULL;
    4110          30 :     skip[ i ]            = 0;
    4111             : 
    4112             :     /* Prefetch the chain head and first pool element on the chain */
    4113          30 :     __builtin_prefetch( &accdb->acc_map[ hashes[ i ] ], 1, 1 );
    4114          30 :   }
    4115             : 
    4116             :   /* Phase 2: walk chains looking for duplicates.  By now the chain
    4117             :      heads prefetched above should be warm in L1/L2.  If the existing
    4118             :      entry has a higher slot, mark skip.  Otherwise, save the existing
    4119             :      entry pointer for in-place update (matching write_one semantics).
    4120             :      In incremental mode, cross-fork entries are saved separately so
    4121             :      they can be left in place while a new entry is inserted. */
    4122             : 
    4123          42 :   for( ulong i=0UL; i<cnt; i++ ) {
    4124          30 :     ulong next_acc = accdb->acc_map[ hashes[ i ] ];
    4125             : 
    4126          30 :     if( FD_LIKELY( next_acc!=UINT_MAX ) ) {
    4127           9 :       __builtin_prefetch( &accdb->acc_pool[ next_acc ], 0, 1 );
    4128           9 :     }
    4129             : 
    4130          30 :     while( next_acc!=UINT_MAX ) {
    4131           9 :       fd_accdb_accmeta_t * candidate = &accdb->acc_pool[ next_acc ];
    4132             : 
    4133           9 :       if( FD_LIKELY( candidate->map.next!=UINT_MAX ) ) {
    4134           0 :         __builtin_prefetch( &accdb->acc_pool[ candidate->map.next ], 0, 1 );
    4135           0 :       }
    4136             : 
    4137           9 :       if( FD_UNLIKELY( !memcmp( pubkeys[ i ], candidate->key.pubkey, 32UL ) ) ) {
    4138           9 :         if( FD_LIKELY( (ulong)candidate->cache_idx>slots[ i ] ) ) {
    4139           3 :           skip[ i ] = 1;
    4140           6 :         } else if( FD_UNLIKELY( incremental ) && candidate->key.generation!=fork_gen ) {
    4141           3 :           cross_existing[ i ] = candidate;
    4142           3 :         } else {
    4143           3 :           existing[ i ] = candidate;
    4144           3 :         }
    4145           9 :         break;
    4146           9 :       }
    4147           0 :       next_acc = candidate->map.next;
    4148           0 :     }
    4149          30 :   }
    4150             : 
    4151             :   /* Phase 2b: reject intra-batch duplicate pubkeys.  Snapin always
    4152             :      populates a batch from a single AppendVec, so every slot in the
    4153             :      batch is identical and a duplicate pubkey means the same account
    4154             :      appears twice at the same slot — i.e. a corrupt snapshot per the
    4155             :      Agave spec.  We have no principled way to pick a winner; return
    4156             :      -1 so the caller can flag the snapshot malformed.  Batches are
    4157             :      bounded (<=8) so the O(n^2) scan is trivial. */
    4158             : 
    4159          30 :   for( ulong i=1UL; i<cnt; i++ ) {
    4160          45 :     for( ulong j=0UL; j<i; j++ ) {
    4161          27 :       if( hashes[ j ]!=hashes[ i ] ) continue;
    4162           0 :       if( FD_UNLIKELY( !memcmp( pubkeys[ j ], pubkeys[ i ], 32UL ) ) ) {
    4163           0 :         FD_LOG_WARNING(( "corrupt snapshot: duplicate pubkey within a single batch (entries %lu and %lu, slots %lu and %lu)", j, i, slots[ j ], slots[ i ] ));
    4164           0 :         return -1;
    4165           0 :       }
    4166           0 :     }
    4167          18 :   }
    4168             : 
    4169             :   /* Phase 3: commit.  For each account either update the existing
    4170             :      entry in-place (replace), allocate and insert at the chain head
    4171             :      (new), or skip entirely (ignore).  This matches the
    4172             :      insert/replace/ignore semantics of write_one. */
    4173             : 
    4174          12 :   ulong used_bytes_added   = 0UL;
    4175          12 :   ulong used_bytes_removed = 0UL;
    4176             : 
    4177          42 :   for( ulong i=0UL; i<cnt; i++ ) {
    4178          30 :     if( FD_UNLIKELY( skip[ i ] ) ) {
    4179             :       /* Still advance the write head so snapwr and snapin stay in
    4180             :          sync — snapwr unconditionally writes every account to disk.
    4181             :          Mark the space as immediately freed since it is dead on
    4182             :          arrival. */
    4183           3 :       ulong dead_sz  = sizeof(fd_accdb_disk_meta_t)+data_lens[ i ];
    4184           3 :       ulong dead_off = allocate_next_write( accdb, dead_sz );
    4185           3 :       fd_accdb_shmem_bytes_freed( accdb->shmem, dead_off, dead_sz );
    4186           3 :       ignored_lamports += lamports[ i ];
    4187           3 :       ignored++;
    4188           3 :       continue;
    4189           3 :     }
    4190             : 
    4191          27 :     fd_accdb_accmeta_t * accmeta;
    4192             : 
    4193          27 :     if( FD_UNLIKELY( existing[ i ] ) ) {
    4194           3 :       accmeta = existing[ i ];
    4195             :       /* The old version's disk space is now dead. */
    4196           3 :       ulong old_sz = sizeof(fd_accdb_disk_meta_t) + FD_ACCDB_SIZE_DATA( accmeta->executable_size );
    4197           3 :       fd_accdb_shmem_bytes_freed( accdb->shmem, fd_accdb_acc_offset( accmeta ), old_sz );
    4198           3 :       used_bytes_removed += old_sz;
    4199           3 :       replaced_lamports += accmeta->lamports;
    4200           3 :       replaced++;
    4201          24 :     } else {
    4202          24 :       accmeta = acc_pool_acquire_nolock( accdb->acc_pool_join );
    4203          24 :       if( FD_UNLIKELY( !accmeta ) ) FD_LOG_ERR(( "accounts database ran out of space during snapshot loading" ));
    4204             : 
    4205          24 :       uint acc_idx = (uint)acc_pool_idx( accdb->acc_pool_join, accmeta );
    4206             : 
    4207          24 :       fd_memcpy( accmeta->key.pubkey, pubkeys[ i ], 32UL );
    4208          24 :       accmeta->key.generation = incremental ? fork_gen : gen;
    4209          24 :       accmeta->map.next = accdb->acc_map[ hashes[ i ] ];
    4210          24 :       accdb->acc_map[ hashes[ i ] ] = acc_idx;
    4211             : 
    4212          24 :       if( FD_UNLIKELY( incremental ) ) {
    4213           6 :         fd_accdb_txn_t * txn = txn_pool_acquire( accdb->txn_pool );
    4214           6 :         if( FD_UNLIKELY( !txn ) ) FD_LOG_ERR(( "txn pool exhausted during incremental snapshot loading" ));
    4215           6 :         txn->acc_pool_idx = acc_idx;
    4216           6 :         uint txn_idx      = (uint)txn_pool_idx( accdb->txn_pool, txn );
    4217           6 :         txn->fork.next          = fork->shmem->txn_head;
    4218           6 :         fork->shmem->txn_head   = txn_idx;
    4219           6 :       }
    4220             : 
    4221          24 :       if( cross_existing[ i ] ) {
    4222           3 :         replaced_lamports += cross_existing[ i ]->lamports;
    4223           3 :         replaced++;
    4224           3 :         cross_replaced++;
    4225          21 :       } else {
    4226          21 :         loaded++;
    4227          21 :       }
    4228          24 :     }
    4229             : 
    4230          27 :     accmeta->cache_idx       = (uint)slots[ i ];
    4231          27 :     accmeta->lamports        = lamports[ i ];
    4232          27 :     accmeta->executable_size = FD_ACCDB_SIZE_PACK( (uint)data_lens[ i ], executables[ i ] );
    4233          27 :     ulong entry_sz       = sizeof(fd_accdb_disk_meta_t)+data_lens[ i ];
    4234          27 :     ulong file_off       = allocate_next_write( accdb, entry_sz );
    4235          27 :     accmeta->offset_fork = incremental ? fd_accdb_acc_pack_offset_fork( file_off, fork_id.val ) : file_off;
    4236          27 :     used_bytes_added    += entry_sz;
    4237          27 :   }
    4238             : 
    4239          12 :   accdb->shmem->shmetrics->disk_used_bytes += used_bytes_added;
    4240          12 :   accdb->shmem->shmetrics->disk_used_bytes -= used_bytes_removed;
    4241             : 
    4242             :   /* accounts_total tracks acc_pool entries: increment for every new
    4243             :      allocation (both genuinely new accounts and cross-fork overrides
    4244             :      that insert a second pool entry).  The output counter
    4245             :      *accounts_loaded excludes cross-fork overrides to match
    4246             :      snapshot_write_one semantics (cross-fork returns 2 = replaced). */
    4247          12 :   accdb->shmem->shmetrics->accounts_total += loaded + cross_replaced;
    4248             : 
    4249          12 :   *accounts_ignored      = ignored;
    4250          12 :   *accounts_replaced     = replaced;
    4251          12 :   *accounts_loaded       = loaded;
    4252          12 :   *out_replaced_lamports = replaced_lamports;
    4253          12 :   *out_ignored_lamports  = ignored_lamports;
    4254             : 
    4255          12 :   return 0;
    4256          12 : }
    4257             : 
    4258             : static void
    4259           0 : delta_reset( fd_accdb_t * accdb ) {
    4260           0 :   if( !accdb->shmem->delta.head ) return; /* clean */
    4261           0 :   uint * chains    = accdb->delta.chains;
    4262           0 :   ulong  chain_cnt = accdb->shmem->delta.chain_cnt;
    4263           0 :   for( ulong i=0UL; i<chain_cnt; i++ ) chains[ i ] = UINT_MAX;
    4264           0 :   accdb->shmem->delta.head = 0UL;
    4265           0 : }
    4266             : 
    4267             : static int
    4268           0 : delta_is_valid( fd_accdb_shmem_t const * accdb ) {
    4269           0 :   return accdb->delta.head < accdb->delta.ele_max;
    4270           0 : }
    4271             : 
    4272             : void
    4273             : fd_accdb_background( fd_accdb_t * accdb,
    4274     3268943 :                      int *        charge_busy ) {
    4275     3268943 :   fd_accdb_shmem_t * shmem = accdb->shmem;
    4276             : 
    4277     3268943 :   ulong * snap_sync_p = &shmem->snapshot_sync;
    4278     3268943 :   ulong   snap_sync   = fd_accdb_snapshot_sync_state( snap_sync_p );
    4279             : 
    4280     3268943 :   uint op = FD_VOLATILE_CONST( shmem->cmd_op );
    4281     3268943 :   if( FD_UNLIKELY( op!=FD_ACCDB_CMD_IDLE ) ) {
    4282         528 :     fd_accdb_fork_id_t fork_id = { .val = FD_VOLATILE_CONST( shmem->cmd_fork_id ) };
    4283             : 
    4284         528 :     switch( op ) {
    4285         501 :       case FD_ACCDB_CMD_ADVANCE_ROOT:
    4286         501 :         background_advance_root( accdb, fork_id );
    4287         501 :         break;
    4288          21 :       case FD_ACCDB_CMD_PURGE:
    4289          21 :         background_purge( accdb, fork_id );
    4290          21 :         break;
    4291           3 :       case FD_ACCDB_CMD_DRAIN_DEFERRED:
    4292           3 :         drain_deferred_frees( accdb );
    4293           3 :         break;
    4294           3 :       case FD_ACCDB_CMD_CLEAR_DEFERRED: {
    4295             :         /* Posted by fd_accdb_reset after it clobbers shared pools.
    4296             :            T2's deferred fork chain now points at recycled elements;
    4297             :            discard the stale pointers.  Epoch slots are preserved
    4298             :            across reset so no re-join is needed. */
    4299           3 :         accdb->deferred_fork_head  = NULL;
    4300           3 :         accdb->deferred_fork_tail  = NULL;
    4301           3 :         accdb->deferred_fork_epoch = 0UL;
    4302           3 :         break;
    4303           0 :       }
    4304           0 :       default:
    4305           0 :         FD_LOG_ERR(( "unexpected accdb cmd_op %u", op ));
    4306         528 :     }
    4307             : 
    4308         528 :     FD_COMPILER_MFENCE();
    4309         528 :     FD_VOLATILE( shmem->cmd_op ) = FD_ACCDB_CMD_IDLE;
    4310         528 :     *charge_busy = 1;
    4311         528 :     return;
    4312         528 :   }
    4313             : 
    4314     3268415 :   if( FD_UNLIKELY( snap_sync!=FD_ACCDB_SNAPSHOT_SYNC_IDLE ) ) {
    4315           0 :     switch( snap_sync ) {
    4316           0 :     case FD_ACCDB_SNAPSHOT_SYNC_RUNNING:
    4317             :       /* while producing a snapshot, don't do compaction work */
    4318           0 :       background_preevict( accdb, charge_busy, 0 );
    4319           0 :       return;
    4320           0 :     case FD_ACCDB_SNAPSHOT_SYNC_DONE:
    4321           0 :       fd_accdb_snapshot_sync_advance( snap_sync_p, FD_ACCDB_SNAPSHOT_SYNC_IDLE );
    4322           0 :       break;
    4323           0 :     case FD_ACCDB_SNAPSHOT_SYNC_START_FULL:
    4324           0 :       FD_CHECK_CRIT( !FD_VOLATILE_CONST( shmem->snapshot_loading ),
    4325           0 :                      "snapshot production requested during snapshot load" );
    4326           0 :       delta_reset( accdb );
    4327           0 :       fd_accdb_snapshot_sync_advance( snap_sync_p, FD_ACCDB_SNAPSHOT_SYNC_RUNNING );
    4328           0 :       *charge_busy = 1;
    4329           0 :       return;
    4330           0 :     case FD_ACCDB_SNAPSHOT_SYNC_START_INCR:
    4331           0 :       FD_CHECK_CRIT( !FD_VOLATILE_CONST( shmem->snapshot_loading ),
    4332           0 :                      "snapshot production requested during snapshot load" );
    4333           0 :       if( delta_is_valid( accdb->shmem ) ) {
    4334           0 :         fd_accdb_snapshot_sync_advance( snap_sync_p, FD_ACCDB_SNAPSHOT_SYNC_RUNNING );
    4335           0 :       } else {
    4336             :         /* cannot produce incrementals because delta ran out of space,
    4337             :            therefore don't know which accounts changed */
    4338           0 :         fd_accdb_snapshot_sync_advance( snap_sync_p, FD_ACCDB_SNAPSHOT_SYNC_FAIL );
    4339           0 :       }
    4340           0 :       *charge_busy = 1;
    4341           0 :       return;
    4342           0 :     case FD_ACCDB_SNAPSHOT_SYNC_FAIL:
    4343             :       /* wait for client to acknowledge */
    4344           0 :       break;
    4345           0 :     default:
    4346           0 :       FD_LOG_CRIT(( "corrupt snapshot_sync state %lu", snap_sync ));
    4347           0 :     }
    4348           0 :   }
    4349             : 
    4350     3268415 :   background_preevict( accdb, charge_busy, 0 );
    4351             : 
    4352    13073660 :   for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
    4353     9805245 :     background_compact( accdb, k, charge_busy );
    4354     9805245 :   }
    4355     3268415 : }
    4356             : 
    4357             : fd_accdb_shmem_metrics_t const *
    4358          57 : fd_accdb_shmetrics( fd_accdb_t * accdb ) {
    4359          57 :   return accdb->shmem->shmetrics;
    4360          57 : }
    4361             : 
    4362             : fd_accdb_metrics_t const *
    4363           9 : fd_accdb_metrics( fd_accdb_t * accdb ) {
    4364           9 :   return accdb->metrics;
    4365           9 : }
    4366             : 
    4367             : void
    4368             : fd_accdb_cache_class_occupancy( fd_accdb_t * accdb,
    4369             :                                 ulong *      used,
    4370             :                                 ulong *      max,
    4371          12 :                                 ulong *      reserved ) {
    4372         108 :   for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) {
    4373          96 :     ulong cap   = accdb->shmem->cache_class_max[ c ];
    4374          96 :     ulong init  = FD_VOLATILE_CONST( accdb->shmem->cache_class_init[ c ].val );
    4375          96 :     ulong freec = FD_VOLATILE_CONST( accdb->shmem->cache_free_cnt  [ c ].val );
    4376          96 :     ulong live  = init>freec ? init-freec : 0UL;
    4377          96 :     if( live>cap ) live = cap;
    4378          96 :     max     [ c ] = cap;
    4379          96 :     used    [ c ] = live;
    4380          96 :     reserved[ c ] = FD_VOLATILE_CONST( accdb->shmem->cache_class_used[ c ].val );
    4381          96 :   }
    4382          12 : }
    4383             : 
    4384             : void
    4385             : fd_accdb_cache_class_thresholds( fd_accdb_t * accdb,
    4386             :                                  ulong *      target_used,
    4387           0 :                                  ulong *      low_water_used ) {
    4388           0 :   for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) {
    4389           0 :     ulong max_c    = accdb->shmem->cache_class_max     [ c ];
    4390           0 :     ulong free_tgt = accdb->shmem->cache_free_target   [ c ];
    4391           0 :     ulong free_lwm = accdb->shmem->cache_free_low_water[ c ];
    4392           0 :     target_used   [ c ] = max_c>free_tgt ? max_c-free_tgt : 0UL;
    4393           0 :     low_water_used[ c ] = max_c>free_lwm ? max_c-free_lwm : 0UL;
    4394           0 :   }
    4395           0 : }
    4396             : 
    4397             : #if FD_HAS_RACESAN
    4398             : 
    4399             : /* Force pre-eviction (ignore the watermark) so a deterministic
    4400             :    single-threaded test can exercise the writeback path without
    4401             :    manufacturing real cache pressure.  Sweeps several times: CLOCK needs
    4402             :    two visits to evict a recently-touched line (clear the "referenced"
    4403             :    bit, then evict), and the clock hand position carries across calls, so
    4404             :    one or two sweeps is not enough to guarantee every eligible line is
    4405             :    flushed back. */
    4406             : void
    4407             : fd_accdb_debug_force_preevict( fd_accdb_t * accdb ) {
    4408             :   for( ulong iter=0UL; iter<8UL; iter++ ) {
    4409             :     int charge_busy = 0;
    4410             :     background_preevict( accdb, &charge_busy, 1 );
    4411             :   }
    4412             : }
    4413             : 
    4414             : /* Locate the resident cache line currently holding `pubkey` (most recent
    4415             :    generation if multiple).  Returns 1 and fills out_class/out_idx on a
    4416             :    hit, 0 if no resident line matches.  Test-only helper so the test can
    4417             :    target a specific line without seeing the opaque fd_accdb struct. */
    4418             : 
    4419             : int
    4420             : fd_accdb_debug_find_line( fd_accdb_t *  accdb,
    4421             :                           uchar const * pubkey,
    4422             :                           ulong *       out_class,
    4423             :                           ulong *       out_idx ) {
    4424             :   int   found      = 0;
    4425             :   uint  best_gen   = 0U;
    4426             :   for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) {
    4427             :     ulong init  = FD_VOLATILE_CONST( accdb->shmem->cache_class_init[ c ].val );
    4428             :     ulong max_c = accdb->shmem->cache_class_max[ c ];
    4429             :     if( init>max_c ) init = max_c;
    4430             :     for( ulong idx=0UL; idx<init; idx++ ) {
    4431             :       fd_accdb_cache_line_t * line = cache_line( accdb, c, idx );
    4432             :       if( line->key.generation==UINT_MAX ) continue;
    4433             :       if( memcmp( line->key.pubkey, pubkey, 32UL ) ) continue;
    4434             :       if( !found || line->key.generation>=best_gen ) {
    4435             :         best_gen   = line->key.generation;
    4436             :         *out_class = c;
    4437             :         *out_idx   = idx;
    4438             :         found      = 1;
    4439             :       }
    4440             :     }
    4441             :   }
    4442             :   return found;
    4443             : }
    4444             : 
    4445             : /* Address of one cache line.  struct fd_accdb_private is local to this
    4446             :    translation unit, so a test cannot reach accdb->cache[] itself. */
    4447             : 
    4448             : void *
    4449             : fd_accdb_debug_line_addr( fd_accdb_t * accdb,
    4450             :                           ulong        size_class,
    4451             :                           ulong        line_idx ) {
    4452             :   return cache_line( accdb, size_class, line_idx );
    4453             : }
    4454             : 
    4455             : /* Deterministically evict a single specified cache line via the
    4456             :    foreground evictor's claim sequence (CAS refcnt 0->EVICT_SENTINEL),
    4457             :    then write the dirty line back exactly as fd_accdb_acquire_inner's
    4458             :    STEP-4 / background_ preevict do (pubkey from accmeta, owner+data
    4459             :    from the line).  Mirrors acquire_cache_line's CLOCK-claim path
    4460             :    (fd_accdb.c) so a racesan test can reproduce, without a 640+-slot
    4461             :    cache-pressure rig, the interleaving where acc_unlink observes
    4462             :    EVICT_SENTINEL on the line it is unlinking.
    4463             : 
    4464             :    The fd_racesan_hook("clock_evict:post_sentinel") fires right after
    4465             :    the sentinel is installed (matching the production foreground path),
    4466             :    so the test can suspend this fiber holding the sentinel while another
    4467             :    fiber drives acc_unlink to its reclaim CAS.  Returns the captured
    4468             :    evicted acc_idx (UINT_MAX if the line was clean / unbound). */
    4469             : 
    4470             : uint
    4471             : fd_accdb_debug_clock_evict_line( fd_accdb_t * accdb,
    4472             :                                  ulong        size_class,
    4473             :                                  ulong        line_idx ) {
    4474             :   fd_accdb_shmem_t *      shmem = accdb->shmem;
    4475             :   fd_accdb_cache_line_t * line  = cache_line( accdb, size_class, line_idx );
    4476             : 
    4477             :   /* Claim for eviction, same as acquire_cache_line's CLOCK path. */
    4478             :   if( FD_UNLIKELY( FD_ATOMIC_CAS( &line->refcnt, 0U, FD_ACCDB_EVICT_SENTINEL )!=0U ) ) return UINT_MAX;
    4479             : 
    4480             :   if( FD_UNLIKELY( line->acc_idx==UINT_MAX && line->key.generation==UINT_MAX ) ) {
    4481             :     FD_VOLATILE( line->refcnt ) = 0;
    4482             :     return UINT_MAX;
    4483             :   }
    4484             : 
    4485             :   fd_racesan_hook( "clock_evict:post_sentinel" );
    4486             : 
    4487             :   uint acc_idx = line->acc_idx;
    4488             :   if( FD_LIKELY( acc_idx!=UINT_MAX ) ) {
    4489             :     evict_clear_acc_cache_ref( &accdb->acc_pool[ acc_idx ], size_class, line_idx );
    4490             :   }
    4491             :   uint evicted_acc_idx = line->persisted ? UINT_MAX : acc_idx;
    4492             :   line->key.generation = UINT_MAX;
    4493             : 
    4494             :   /* Write back the dirty line, exactly like the production writeback
    4495             :      sites: this is the synthesis that would emit a pubkey=NEW/owner=OLD
    4496             :      poison record if the accmeta slot had been recycled out from under
    4497             :      us.  In the SENTINEL-vs-acc_unlink race this proves no poison: the
    4498             :      epoch the evictor holds blocks drain_deferred_frees, so the slot is
    4499             :      never recycled while we are here. */
    4500             :   if( FD_UNLIKELY( !line->persisted && acc_idx!=UINT_MAX ) ) {
    4501             :     fd_accdb_accmeta_t * accmeta = &accdb->acc_pool[ acc_idx ];
    4502             :     ulong entry_sz = sizeof(fd_accdb_disk_meta_t)+(ulong)FD_ACCDB_SIZE_DATA( accmeta->executable_size );
    4503             : 
    4504             :     ulong old_offset = fd_accdb_acc_xchg_offset( accmeta, FD_ACCDB_OFF_INVAL );
    4505             :     if( FD_LIKELY( old_offset!=FD_ACCDB_OFF_INVAL ) ) {
    4506             :       fd_accdb_shmem_bytes_freed( shmem, old_offset, entry_sz );
    4507             :       FD_ATOMIC_FETCH_AND_SUB( &shmem->shmetrics->disk_used_bytes, entry_sz );
    4508             :     }
    4509             : 
    4510             :     fd_accdb_disk_meta_t meta;
    4511             :     fd_memcpy( meta.pubkey, accmeta->key.pubkey, 32UL );
    4512             :     meta.size       = FD_ACCDB_SIZE_DATA( accmeta->executable_size );
    4513             :     meta.generation = accmeta->key.generation;
    4514             :     fd_memcpy( meta.owner, line->owner, 32UL );
    4515             : 
    4516             :     struct iovec iovs[ 2UL ] = {
    4517             :       { .iov_base = &meta,              .iov_len = sizeof(fd_accdb_disk_meta_t) },
    4518             :       { .iov_base = (void *)(line+1UL), .iov_len = FD_ACCDB_SIZE_DATA( accmeta->executable_size ) }
    4519             :     };
    4520             :     ulong file_off = allocate_next_write( accdb, entry_sz );
    4521             :     ulong written  = 0UL;
    4522             :     while( written<entry_sz ) {
    4523             :       long result = pwritev2( accdb->fd, iovs, 2, (long)(file_off+written), 0 );
    4524             :       if( FD_UNLIKELY( result==-1 && errno==EINTR ) ) continue;
    4525             :       else if( FD_UNLIKELY( result<=0 ) ) FD_LOG_ERR(( "pwritev2() failed (%d-%s)", errno, fd_io_strerror( errno ) ));
    4526             :       written += (ulong)result;
    4527             :       for( int v=0; v<2; v++ ) {
    4528             :         if( (ulong)result>=iovs[ v ].iov_len ) { result -= (long)iovs[ v ].iov_len; iovs[ v ].iov_len = 0UL; }
    4529             :         else { iovs[ v ].iov_base = (uchar *)iovs[ v ].iov_base + result; iovs[ v ].iov_len -= (ulong)result; break; }
    4530             :       }
    4531             :     }
    4532             :     FD_COMPILER_MFENCE();
    4533             :     accmeta->offset_fork = fd_accdb_acc_pack_offset_fork( file_off, fd_accdb_acc_fork_id(accmeta) );
    4534             :     FD_ATOMIC_FETCH_AND_ADD( &shmem->shmetrics->disk_used_bytes, entry_sz );
    4535             :   }
    4536             : 
    4537             :   line->persisted      = 1;
    4538             :   line->acc_idx        = UINT_MAX;
    4539             :   line->key.generation = UINT_MAX;
    4540             :   FD_COMPILER_MFENCE();
    4541             :   FD_VOLATILE( line->refcnt ) = 0;
    4542             :   cache_free_push( accdb, size_class, line );
    4543             :   return evicted_acc_idx;
    4544             : }
    4545             : 
    4546             : #endif

Generated by: LCOV version 1.14