Line data Source code
1 : #define _GNU_SOURCE
2 : #include "fd_accdb.h"
3 : #include "fd_accdb_shmem.h"
4 : #include "fd_accdb_private.h"
5 : #include "../../util/fd_hash32.h"
6 :
7 : #if FD_TMPL_USE_HANDHOLDING
8 : #include "../../ballet/txn/fd_txn.h"
9 : #include "../../ballet/base58/fd_base58.h"
10 : #endif
11 : #include "../../util/racesan/fd_racesan_target.h"
12 :
13 : #include "../../disco/events/generated/fd_event_gen.h"
14 :
15 : FD_STATIC_ASSERT( sizeof(fd_accdb_cache_line_t)==FD_ACCDB_CACHE_META_SZ, cache_meta_sz );
16 :
17 : #if FD_HAS_RACESAN
18 : /* Test-only telemetry: background_compact publishes the pubkey + dest
19 : offset of the record it is about to relocation-CAS at the
20 : accdb_compact:pre_offset_cas hook, so test_accdb_racesan can PROVE the
21 : parked relocation is the account it set up (avoiding a vacuous test).
22 : Zero-cost / absent in production (racesan off). */
23 : uchar fd_accdb_dbg_reloc_pubkey[ 32UL ];
24 : ulong fd_accdb_dbg_reloc_dest;
25 : ulong fd_accdb_dbg_reloc_cnt;
26 : #endif
27 :
28 : #include <stddef.h>
29 : #include <unistd.h>
30 : #include <fcntl.h>
31 : #include <errno.h>
32 : #include <sys/uio.h>
33 :
34 : struct fd_accdb_fork {
35 : fd_accdb_fork_shmem_t * shmem;
36 : descends_set_t * descends;
37 : };
38 :
39 : typedef struct fd_accdb_fork fd_accdb_fork_t;
40 :
41 315459 : #define FD_ACCDB_ACQUIRE_STATE_IDLE (0)
42 351 : #define FD_ACCDB_ACQUIRE_STATE_PHASE_A (1)
43 311403 : #define FD_ACCDB_ACQUIRE_STATE_OPEN (2)
44 :
45 : struct __attribute__((aligned(FD_ACCDB_ALIGN))) fd_accdb_private {
46 : int fd;
47 :
48 : int acquire_state;
49 :
50 : fd_accdb_shmem_t * shmem;
51 :
52 : fd_accdb_fork_t * fork_pool;
53 : fork_pool_t fork_shmem_pool[1];
54 :
55 : fd_accdb_accmeta_t * acc_pool;
56 : acc_pool_t acc_pool_join[1];
57 : uint * acc_map;
58 :
59 : uchar * cache [ FD_ACCDB_CACHE_CLASS_CNT ];
60 :
61 : fd_accdb_partition_t * partition_pool;
62 : compaction_dlist_t * compaction_dlist[ FD_ACCDB_COMPACTION_LAYER_CNT ];
63 : deferred_free_dlist_t * deferred_free_dlist;
64 :
65 : txn_pool_t txn_pool[1];
66 :
67 : /* Pointer into shmem->joiner_epochs[ my_slot ].val for writer
68 : joiners, or into a private per-tile fseq for read-only joiners.
69 : Set to the current global epoch on entry to an epoch-protected
70 : operation, and ULONG_MAX on exit. Used to determine when
71 : deferred frees are safe. */
72 : ulong * my_epoch_slot;
73 :
74 : /* Read-only pointers to external epoch slots (e.g. fseqs owned by
75 : RO consumer tiles like the rpc tile). Scanned in addition to
76 : shmem->joiner_epochs[] by compaction's deferred-free
77 : reclamation. Borrowed; the caller of fd_accdb_new owns the
78 : storage. */
79 : ulong const * const * external_epoch_slots;
80 : ulong external_epoch_cnt;
81 :
82 : /* Side buffer of acc pool indices that have been CAS-unlinked from
83 : their hash chains but cannot be released back to acc_pool yet,
84 : because concurrent readers (acquire / compact) may still be
85 : traversing the removed nodes via map.next. The batch is released
86 : once all joiner_epochs exceed shmem->deferred_acc_epoch. Indices
87 : are written here (not into pool.next) until after the epoch drain
88 : because pool.next is union-aliased to cache_idx, which a concurrent
89 : cold_load_acc may still write through a captured pointer. Backed
90 : by shmem->deferred_acc_buf_off; cnt and epoch live in shmem too. */
91 : uint * deferred_acc_buf;
92 :
93 : /* Chain of fork pool slots whose IDs are still potentially
94 : referenced by concurrent readers (via descends_set_test or
95 : root_fork_id snapshot). The chain is released back to fork_pool
96 : once all joiner_epochs exceed deferred_fork_epoch. NULL head
97 : means no deferred forks. */
98 : fd_accdb_fork_shmem_t * deferred_fork_head;
99 : fd_accdb_fork_shmem_t * deferred_fork_tail;
100 : ulong deferred_fork_epoch;
101 :
102 : fd_accdb_metrics_t metrics[1];
103 :
104 : /* Set by fd_accdb_snapshot_load_begin/end. When non-zero, layer-0
105 : partition handoffs (in change_partition) re-tier the partitions
106 : that fell out of the snapshot-load working set: P-2 to Warm and
107 : P-3 to Cold. This backfills tiering for snapshot-loaded data
108 : that never gets a second write (and therefore would otherwise
109 : never be promoted by compaction). */
110 : int snapshot_loading;
111 :
112 : /* Track account addresses changed since a full snap.
113 : Used to determine which accounts should be packed into a full
114 : snapshot (including tombstones for accounts no longer present in
115 : accdb, but present in the full snapshot). */
116 : struct {
117 : uint * chains;
118 : fd_accdb_delta_t * pool;
119 : struct {
120 : uchar const * pubkey;
121 : uint chain;
122 : } scratch[ FD_ACCDB_MAX_ACQUIRE_CNT ];
123 : } delta;
124 :
125 : /* Write counters that are not published yet. Metrics are aggregated
126 : in batches to avoid expensive atomic operations on each write.
127 : 64-byte aligned to fit in a single cache line. */
128 : struct {
129 : ulong bytes; /* bytes reserved on partition_idx */
130 : ulong num_ops; /* reservations behind those bytes */
131 : ulong partition_idx; /* set while num_ops>0 */
132 : } write_stats __attribute__((aligned(64)));
133 : };
134 :
135 : static inline fd_accdb_cache_line_t *
136 : cache_line( fd_accdb_t * accdb,
137 : ulong cls,
138 2289933 : ulong idx ) {
139 2289933 : return (fd_accdb_cache_line_t *)( accdb->cache[ cls ] + idx * fd_accdb_cache_slot_sz[ cls ] );
140 2289933 : }
141 :
142 : /* Bump the per-partition read counters for the partition that contains
143 : file_offset. Called at preadv2 sites. Writes are counted at
144 : allocate time (see fd_accdb_partition_write_bump) so that they reflect
145 : bytes committed to a partition rather than syscalls — the snapshot
146 : loader bypasses pwritev2 entirely, but every write still goes through
147 : reserve_next_write. */
148 : static inline void
149 : fd_accdb_partition_read_bump( fd_accdb_t * accdb,
150 : ulong file_offset,
151 42 : ulong bytes ) {
152 42 : if( FD_UNLIKELY( !bytes ) ) return;
153 : /* Readonly joiners have no partition_pool join (see
154 : fd_accdb_join_readonly) and do not contribute to per-partition
155 : read telemetry today; their disk reads still show up in the
156 : joiner-local fd_accdb_metrics_t bytes_read/read_ops. */
157 42 : if( FD_UNLIKELY( !accdb->partition_pool ) ) return;
158 42 : ulong partition_idx = file_offset / accdb->shmem->partition_sz;
159 42 : ulong partition_cnt = partition_pool_max( accdb->partition_pool );
160 42 : FD_CHECK_CRIT( partition_idx<partition_cnt, "read partition index out of range" );
161 42 : fd_accdb_partition_t * p = partition_pool_ele( accdb->partition_pool, partition_idx );
162 42 : FD_ATOMIC_FETCH_AND_ADD( &p->bytes_read, bytes );
163 42 : FD_ATOMIC_FETCH_AND_ADD( &p->read_ops, 1UL );
164 42 : }
165 :
166 : /* Bump the per-partition write counters. bytes is how much landed on
167 : this partition, over num_ops reservations. */
168 : static inline void
169 : fd_accdb_partition_write_bump( fd_accdb_t * accdb,
170 : ulong partition_idx,
171 : ulong bytes,
172 57 : ulong num_ops ) {
173 57 : if( FD_UNLIKELY( !bytes ) ) return;
174 57 : ulong partition_cnt = partition_pool_max( accdb->partition_pool );
175 57 : FD_CHECK_CRIT( partition_idx<partition_cnt, "write partition index out of range" );
176 57 : fd_accdb_partition_t * p = partition_pool_ele( accdb->partition_pool, partition_idx );
177 57 : FD_ATOMIC_FETCH_AND_ADD( &p->bytes_written, bytes );
178 57 : FD_ATOMIC_FETCH_AND_ADD( &p->write_ops, num_ops );
179 57 : }
180 :
181 : void
182 69 : fd_accdb_flush_metrics( fd_accdb_t * accdb ) {
183 69 : ulong bytes = accdb->write_stats.bytes;
184 69 : ulong num_ops = accdb->write_stats.num_ops;
185 69 : ulong part_idx = accdb->write_stats.partition_idx;
186 :
187 69 : if( !num_ops ) return;
188 :
189 54 : memset( &accdb->write_stats, 0, sizeof(accdb->write_stats) );
190 :
191 54 : FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->shmetrics->disk_current_bytes, bytes );
192 54 : fd_accdb_partition_write_bump( accdb, part_idx, bytes, num_ops );
193 54 : }
194 :
195 : static inline ulong
196 : cache_line_idx( fd_accdb_t * accdb,
197 : ulong cls,
198 1965678 : fd_accdb_cache_line_t const * line ) {
199 1965678 : return (ulong)( (uchar const *)line - accdb->cache[ cls ] ) / fd_accdb_cache_slot_sz[ cls ];
200 1965678 : }
201 :
202 : #if FD_TMPL_USE_HANDHOLDING
203 : static inline int
204 : fd_accdb_ptr_in_region( fd_accdb_t const * accdb,
205 : ulong cls,
206 : void const * ptr ) {
207 : if( FD_UNLIKELY( cls>=FD_ACCDB_CACHE_CLASS_CNT ) ) return 0;
208 :
209 : uchar const * base = accdb->cache[ cls ];
210 : if( FD_UNLIKELY( !base ) ) return 0;
211 :
212 : ulong slot_sz = fd_accdb_cache_slot_sz[ cls ];
213 : ulong region_sz = accdb->shmem->cache_class_max[ cls ] * slot_sz;
214 : uchar const * p = (uchar const *)ptr;
215 :
216 : if( FD_UNLIKELY( p<base || p>=base+region_sz ) ) return 0;
217 : return ( (ulong)( p - base ) % slot_sz )==FD_ACCDB_CACHE_META_SZ;
218 : }
219 : #endif
220 :
221 : FD_FN_CONST ulong
222 12792 : fd_accdb_align( void ) {
223 12792 : return FD_ACCDB_ALIGN;
224 12792 : }
225 :
226 : FD_FN_CONST ulong
227 279 : fd_accdb_footprint( ulong max_live_slots ) {
228 279 : ulong l;
229 279 : l = FD_LAYOUT_INIT;
230 279 : l = FD_LAYOUT_APPEND( l, FD_ACCDB_ALIGN, sizeof(fd_accdb_t) );
231 279 : l = FD_LAYOUT_APPEND( l, alignof(fd_accdb_fork_t), max_live_slots*sizeof(fd_accdb_fork_t) );
232 279 : return FD_LAYOUT_FINI( l, FD_ACCDB_ALIGN );
233 279 : }
234 :
235 : void *
236 : fd_accdb_new( void * ljoin,
237 : fd_accdb_shmem_t * shmem,
238 : int fd,
239 : ulong external_epoch_cnt,
240 4170 : ulong const ** external_epoch_slots ) {
241 4170 : if( FD_UNLIKELY( !ljoin ) ) {
242 0 : FD_LOG_WARNING(( "NULL ljoin" ));
243 0 : return NULL;
244 0 : }
245 :
246 4170 : if( FD_UNLIKELY( !fd_ulong_is_aligned( (ulong)ljoin, fd_accdb_align() ) ) ) {
247 0 : FD_LOG_WARNING(( "misaligned ljoin" ));
248 0 : return NULL;
249 0 : }
250 :
251 4170 : if( FD_UNLIKELY( fd<0 ) ) {
252 0 : FD_LOG_WARNING(( "fd must be a valid file descriptor" ));
253 0 : return NULL;
254 0 : }
255 :
256 4170 : ulong max_live_slots = shmem->max_live_slots;
257 4170 : ulong max_accounts = shmem->max_accounts;
258 4170 : ulong max_account_writes_per_slot = shmem->max_account_writes_per_slot;
259 4170 : ulong partition_cnt = shmem->partition_cnt;
260 :
261 4170 : ulong chain_cnt = fd_ulong_pow2_up( (max_accounts>>1) + (max_accounts&1UL) );
262 4170 : ulong txn_max = max_live_slots * max_account_writes_per_slot;
263 :
264 4170 : FD_SCRATCH_ALLOC_INIT( l, shmem );
265 4170 : FD_SCRATCH_ALLOC_APPEND( l, FD_ACCDB_SHMEM_ALIGN, sizeof(fd_accdb_shmem_t) );
266 4170 : void * _fork_pool_ele = FD_SCRATCH_ALLOC_APPEND( l, alignof(fd_accdb_fork_shmem_t), max_live_slots*sizeof(fd_accdb_fork_shmem_t) );
267 4170 : void * _descends_sets = FD_SCRATCH_ALLOC_APPEND( l, descends_set_align(), max_live_slots*descends_set_footprint( max_live_slots ) );
268 4170 : void * _acc_map = FD_SCRATCH_ALLOC_APPEND( l, alignof(uint), chain_cnt*sizeof(uint) );
269 4170 : void * _acc_pool_ele = FD_SCRATCH_ALLOC_APPEND( l, alignof(fd_accdb_accmeta_t), max_accounts*sizeof(fd_accdb_accmeta_t) );
270 4170 : void * _txn_pool_ele = FD_SCRATCH_ALLOC_APPEND( l, alignof(fd_accdb_txn_t), txn_max*sizeof(fd_accdb_txn_t) );
271 4170 : void * _partition_pool = FD_SCRATCH_ALLOC_APPEND( l, partition_pool_align(), partition_pool_footprint( partition_cnt ) );
272 4170 : void * _compaction_dlists[ FD_ACCDB_COMPACTION_LAYER_CNT ];
273 16680 : for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
274 12510 : _compaction_dlists[ k ] = FD_SCRATCH_ALLOC_APPEND( l, compaction_dlist_align(), compaction_dlist_footprint() );
275 12510 : }
276 4170 : void * _deferred_free_dlist = FD_SCRATCH_ALLOC_APPEND( l, deferred_free_dlist_align(), deferred_free_dlist_footprint() );
277 :
278 4170 : FD_SCRATCH_ALLOC_INIT( l2, ljoin );
279 4170 : fd_accdb_t * accdb = FD_SCRATCH_ALLOC_APPEND( l2, fd_accdb_align(), sizeof(fd_accdb_t) );
280 4170 : void * _local_fork_pool = FD_SCRATCH_ALLOC_APPEND( l2, alignof(fd_accdb_fork_t), max_live_slots*sizeof(fd_accdb_fork_t) );
281 :
282 4170 : accdb->fd = fd;
283 4170 : accdb->acquire_state = FD_ACCDB_ACQUIRE_STATE_IDLE;
284 4170 : accdb->snapshot_loading = 0;
285 :
286 4170 : accdb->shmem = (fd_accdb_shmem_t *)shmem;
287 4170 : FD_TEST( acc_pool_join( accdb->acc_pool_join, shmem->acc_pool, _acc_pool_ele, max_accounts ) );
288 4170 : accdb->acc_pool = accdb->acc_pool_join->ele;
289 4170 : accdb->acc_map = _acc_map;
290 4170 : FD_TEST( txn_pool_join( accdb->txn_pool, shmem->txn_pool, _txn_pool_ele, txn_max ) );
291 37530 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) accdb->cache[ c ] = (uchar *)shmem + shmem->cache_region_off[ c ];
292 4170 : accdb->partition_pool = partition_pool_join( _partition_pool );
293 4170 : FD_TEST( accdb->partition_pool );
294 16680 : for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
295 12510 : accdb->compaction_dlist[ k ] = compaction_dlist_join( _compaction_dlists[ k ] );
296 12510 : FD_TEST( accdb->compaction_dlist[ k ] );
297 12510 : }
298 4170 : accdb->deferred_free_dlist = deferred_free_dlist_join( _deferred_free_dlist );
299 4170 : FD_TEST( accdb->deferred_free_dlist );
300 :
301 4170 : FD_TEST( fork_pool_join( accdb->fork_shmem_pool, shmem->fork_pool, _fork_pool_ele, max_live_slots ) );
302 4170 : accdb->fork_pool = _local_fork_pool;
303 74547 : for( ulong i=0UL; i<max_live_slots; i++ ) {
304 70377 : fd_accdb_fork_t * fork = &accdb->fork_pool[ i ];
305 70377 : fork->shmem = fork_pool_ele( accdb->fork_shmem_pool, i );
306 70377 : fork->descends = descends_set_join( (uchar *)_descends_sets + i*descends_set_footprint( max_live_slots ) );
307 70377 : FD_TEST( fork->shmem );
308 70377 : FD_TEST( fork->descends );
309 70377 : }
310 :
311 4170 : ulong epoch_idx = FD_ATOMIC_FETCH_AND_ADD( &shmem->joiner_cnt, 1UL );
312 4170 : FD_TEST( epoch_idx<shmem->joiner_cnt_max );
313 4170 : accdb->my_epoch_slot = &shmem->joiner_epochs[ epoch_idx ].val;
314 :
315 4170 : accdb->external_epoch_slots = external_epoch_slots;
316 4170 : accdb->external_epoch_cnt = external_epoch_cnt;
317 :
318 4170 : accdb->deferred_acc_buf = (uint *)( (uchar *)shmem + shmem->deferred_acc_buf_off );
319 :
320 4170 : accdb->delta.chains = (uint *) ( (uchar *)shmem + shmem->delta.chain_off );
321 4170 : accdb->delta.pool = (fd_accdb_delta_t *) ( (uchar *)shmem + shmem->delta.ele_off );
322 :
323 4170 : accdb->deferred_fork_head = NULL;
324 4170 : accdb->deferred_fork_tail = NULL;
325 4170 : accdb->deferred_fork_epoch = 0UL;
326 :
327 4170 : memset( accdb->metrics, 0, sizeof(fd_accdb_metrics_t) );
328 4170 : memset( &accdb->write_stats, 0, sizeof(accdb->write_stats) );
329 :
330 4170 : return accdb;
331 4170 : }
332 :
333 : static inline void wait_cmd( fd_accdb_t * accdb );
334 : static inline void submit_cmd( fd_accdb_t * accdb, uint op, ushort fork_id );
335 :
336 : void
337 3 : fd_accdb_reset( fd_accdb_t * accdb ) {
338 3 : fd_accdb_shmem_t * shmem = accdb->shmem;
339 :
340 : /* Wait for any pending background command (advance_root / purge) on
341 : T2 to finish before clobbering shared state. */
342 3 : wait_cmd( accdb );
343 :
344 : /* Reset pools through the joiner's existing pointers. acc_pool and
345 : txn_pool use POOL_LAZY=1 so reset is O(1). fork_pool and
346 : partition_pool rebuild their free lists in O(max_live_slots) and
347 : O(partition_cnt), both small. */
348 3 : acc_pool_reset( accdb->acc_pool_join );
349 3 : txn_pool_reset( accdb->txn_pool );
350 3 : fork_pool_reset( accdb->fork_shmem_pool );
351 3 : partition_pool_reset( accdb->partition_pool );
352 :
353 : /* Clear hash chains */
354 3 : fd_memset( accdb->acc_map, 0xFF, shmem->chain_cnt*sizeof(uint) );
355 :
356 : /* Empty dlists */
357 12 : for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
358 9 : compaction_dlist_remove_all( accdb->compaction_dlist[ k ], accdb->partition_pool );
359 9 : }
360 3 : deferred_free_dlist_remove_all( accdb->deferred_free_dlist, accdb->partition_pool );
361 :
362 : /* Null descends_sets. */
363 195 : for( ulong i=0UL; i<shmem->max_live_slots; i++ ) {
364 192 : descends_set_null( accdb->fork_pool[ i ].descends );
365 192 : }
366 :
367 : /* Reset shmem scalar fields. */
368 3 : shmem->root_fork_id = (fd_accdb_fork_id_t){ .val = USHORT_MAX };
369 3 : shmem->generation = 0U;
370 3 : shmem->partition_lock = 0;
371 3 : shmem->partition_max = 0UL;
372 :
373 : /* Write heads: sentinel values that force partition-switch on first
374 : write. */
375 12 : for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
376 9 : shmem->whead[ k ] = accdb_offset( shmem->partition_cnt, shmem->partition_sz );
377 9 : shmem->has_partition[ k ] = 0;
378 9 : }
379 :
380 : /* Cache state */
381 27 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) {
382 24 : shmem->clock_hand[ c ].val = 0UL;
383 24 : shmem->cache_free[ c ].ver_top = (ulong)UINT_MAX;
384 24 : shmem->cache_free_cnt[ c ].val = 0UL;
385 24 : shmem->cache_class_init[ c ].val = 0UL;
386 24 : if( shmem->cache_class_max[ c ]>=shmem->cache_min_reserved*shmem->joiner_cnt_max )
387 24 : shmem->cache_class_used[ c ].val = ULONG_MAX;
388 0 : else
389 0 : shmem->cache_class_used[ c ].val = 0UL;
390 24 : }
391 :
392 : /* Reset every cache slot's metadata to empty sentinels. */
393 27 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) {
394 24 : ulong slot_sz = fd_accdb_cache_slot_sz[ c ];
395 82200 : for( ulong i=0UL; i<shmem->cache_class_max[ c ]; i++ ) {
396 82176 : fd_accdb_cache_line_t * line = (fd_accdb_cache_line_t *)( accdb->cache[ c ] + i*slot_sz );
397 82176 : line->key.generation = UINT_MAX;
398 82176 : line->acc_idx = UINT_MAX;
399 82176 : line->refcnt = 0U;
400 82176 : line->referenced = 0;
401 82176 : line->persisted = 1;
402 82176 : }
403 24 : }
404 :
405 : /* Epoch system: reset epoch and all slot values to idle, but
406 : preserve joiner_cnt and each tile's my_epoch_slot pointer so that
407 : tiles which joined during init keep their original slot indices. */
408 3 : shmem->epoch = 1UL;
409 771 : for( ulong i=0UL; i<FD_ACCDB_MAX_JOINERS; i++ ) shmem->joiner_epochs[ i ].val = ULONG_MAX;
410 :
411 : /* Deferred acc buffer. */
412 3 : shmem->deferred_acc_buf_cnt = 0UL;
413 3 : shmem->deferred_acc_epoch = 0UL;
414 :
415 : /* Shared metrics: zero gauges that reflect current state (now empty)
416 : but preserve counters and accounts_capacity. */
417 3 : shmem->shmetrics->accounts_total = 0UL;
418 3 : shmem->shmetrics->disk_allocated_bytes = 0UL;
419 3 : shmem->shmetrics->disk_current_bytes = 0UL;
420 3 : shmem->shmetrics->disk_used_bytes = 0UL;
421 3 : shmem->shmetrics->in_compaction = 0;
422 :
423 : /* Command slot */
424 3 : shmem->cmd_op = FD_ACCDB_CMD_IDLE;
425 3 : shmem->cmd_fork_id = USHORT_MAX;
426 :
427 3 : shmem->snapshot_loading = 0;
428 :
429 3 : FD_COMPILER_MFENCE();
430 :
431 : /* Tell the accdb tile to clear its stale deferred fork chain.
432 : Its deferred_fork_head/tail now reference recycled pool elements;
433 : it must discard them before processing any future advance_root or
434 : purge command. The command is asynchronous; the next advance_root
435 : or purge call will wait for it to complete via wait_cmd. */
436 3 : submit_cmd( accdb, FD_ACCDB_CMD_CLEAR_DEFERRED, 0 );
437 :
438 : /* Reset local state */
439 3 : accdb->deferred_fork_head = NULL;
440 3 : accdb->deferred_fork_tail = NULL;
441 3 : accdb->deferred_fork_epoch = 0UL;
442 3 : accdb->snapshot_loading = 0;
443 3 : accdb->acquire_state = FD_ACCDB_ACQUIRE_STATE_IDLE;
444 3 : memset( &accdb->write_stats, 0, sizeof(accdb->write_stats) );
445 3 : }
446 :
447 : void
448 21 : fd_accdb_snapshot_load_begin( fd_accdb_t * accdb ) {
449 21 : FD_CHECK_CRIT( fd_accdb_snapshot_sync_state( &accdb->shmem->snapshot_sync )==FD_ACCDB_SNAPSHOT_SYNC_IDLE,
450 21 : "snapshot load started while snapshot production active" );
451 21 : accdb->snapshot_loading = 1;
452 21 : FD_VOLATILE( accdb->shmem->snapshot_loading ) = 1;
453 21 : }
454 :
455 : static inline void
456 : change_partition( fd_accdb_t * accdb,
457 : accdb_offset_t const * offset_before,
458 : accdb_offset_t * out_offset,
459 : int * has_partition,
460 : uchar layer );
461 :
462 : void
463 21 : fd_accdb_snapshot_load_end( fd_accdb_t * accdb ) {
464 21 : fd_accdb_flush_metrics( accdb );
465 21 : spin_lock_acquire( &accdb->shmem->partition_lock );
466 :
467 : /* Force the next layer-0 write onto a fresh Hot partition so we do
468 : not keep appending live execution writes to the tail of a partition
469 : that was tagged Cold during snapshot load. Must run while
470 : snapshot_loading is still set so the partition we just closed
471 : (the snapshot-tagged Cold one) is not enqueued for compaction by
472 : change_partition's tail-credit try_enqueue. change_partition will
473 : retag the newly-allocated partition as Cold (because the flag is
474 : still set), so we fix it back to Hot below. */
475 21 : if( FD_LIKELY( accdb->shmem->has_partition[ 0 ] ) ) {
476 21 : change_partition( accdb, &accdb->shmem->whead[ 0 ], &accdb->shmem->whead[ 0 ], &accdb->shmem->has_partition[ 0 ], 0 );
477 21 : ulong new_idx = packed_partition_idx( &accdb->shmem->whead[ 0 ] );
478 21 : fd_accdb_partition_t * newp = partition_pool_ele( accdb->partition_pool, new_idx );
479 21 : FD_VOLATILE( newp->layer ) = 0;
480 21 : }
481 :
482 21 : accdb->snapshot_loading = 0;
483 21 : FD_VOLATILE( accdb->shmem->snapshot_loading ) = 0;
484 :
485 : /* Sweep all partitions written during the load — any that crossed
486 : the fragmentation threshold while enqueue was suppressed are
487 : re-checked now and pushed onto the compaction queue. */
488 21 : ulong partition_max = accdb->shmem->partition_max;
489 90 : for( ulong p=0UL; p<partition_max; p++ ) {
490 69 : fd_accdb_shmem_try_enqueue_compaction( accdb->shmem, p );
491 69 : }
492 :
493 21 : spin_lock_release( &accdb->shmem->partition_lock );
494 21 : }
495 :
496 : static inline uint
497 : delta_chain( fd_accdb_shmem_t const * accdb,
498 0 : uchar const pubkey[ 32 ] ) {
499 0 : uint hash = (uint)fd_hash32( pubkey, accdb->delta.seed );
500 0 : return hash & accdb->delta.chain_mask;
501 0 : }
502 :
503 : static int
504 : delta_insert( fd_accdb_t * accdb,
505 1458 : uchar const pubkey[ 32 ] ) {
506 : /* FIXME consider batch inserting */
507 1458 : fd_accdb_shmem_t * shmem = accdb->shmem;
508 1458 : if( FD_UNLIKELY( shmem->delta.head >= shmem->delta.ele_max ) ) return 0;
509 :
510 0 : uint * chains = accdb->delta.chains;
511 0 : uint * chain = &chains[ delta_chain( shmem, pubkey ) ];
512 0 : fd_accdb_delta_t * pool = accdb->delta.pool;
513 :
514 0 : uint head = *chain;
515 0 : for( uint cur=head; cur!=UINT_MAX; cur=pool[ cur ].next ) {
516 0 : if( FD_UNLIKELY( !memcmp( pool[ cur ].pubkey, pubkey, 32UL ) ) ) return 1;
517 0 : }
518 :
519 0 : ulong idx = shmem->delta.head++;
520 0 : fd_accdb_delta_t * delta = &pool[ idx ];
521 0 : delta->next = head;
522 0 : memcpy( delta->pubkey, pubkey, 32UL );
523 0 : *chain = (uint)idx;
524 0 : return 1;
525 0 : }
526 :
527 : int
528 : fd_accdb_snapshot_recover_delta( fd_accdb_t * accdb,
529 0 : fd_accdb_fork_id_t fork_id ) {
530 0 : if( FD_UNLIKELY( fork_id.val>=fork_pool_ele_max( accdb->fork_shmem_pool ) ) ) {
531 0 : FD_LOG_CRIT(( "fd_accdb_snapshot_populate_delta: invalid fork id %u (capacity %lu)",
532 0 : (uint)fork_id.val, fork_pool_ele_max( accdb->fork_shmem_pool ) ));
533 0 : }
534 :
535 0 : uint txn_idx = accdb->fork_pool[ fork_id.val ].shmem->txn_head;
536 0 : while( txn_idx!=UINT_MAX ) {
537 0 : fd_accdb_txn_t const * txn = txn_pool_ele( accdb->txn_pool, (ulong)txn_idx );
538 0 : fd_accdb_accmeta_t const * acc = &accdb->acc_pool[ txn->acc_pool_idx ];
539 0 : if( FD_UNLIKELY( !delta_insert( accdb, acc->key.pubkey ) ) ) return -1;
540 0 : txn_idx = txn->fork.next;
541 0 : }
542 0 : return 0;
543 0 : }
544 :
545 : void
546 : fd_accdb_snapshot_save_whead( fd_accdb_t * accdb,
547 12 : fd_accdb_snapshot_recovery_t * out ) {
548 : /* Flush metrics to update disk_current_bytes. */
549 12 : fd_accdb_flush_metrics( accdb );
550 :
551 12 : out->whead_val = FD_VOLATILE_CONST( accdb->shmem->whead[ 0 ].val );
552 12 : out->has_partition = FD_VOLATILE_CONST( accdb->shmem->has_partition[ 0 ] );
553 12 : out->partition_max = FD_VOLATILE_CONST( accdb->shmem->partition_max );
554 12 : out->disk_current_bytes = FD_VOLATILE_CONST( accdb->shmem->shmetrics->disk_current_bytes );
555 :
556 12 : if( out->has_partition ) {
557 12 : accdb_offset_t whead = { .val = out->whead_val };
558 12 : ulong idx = packed_partition_idx( &whead );
559 12 : fd_accdb_partition_t * part = partition_pool_ele( accdb->partition_pool, idx );
560 12 : out->savepoint_bytes_freed = FD_VOLATILE_CONST( part->bytes_freed );
561 12 : } else {
562 0 : out->savepoint_bytes_freed = 0UL;
563 0 : }
564 12 : }
565 :
566 : void
567 : fd_accdb_snapshot_revert_whead( fd_accdb_t * accdb,
568 12 : fd_accdb_snapshot_recovery_t const * recover ) {
569 12 : fd_accdb_shmem_t * shmem = accdb->shmem;
570 :
571 : /* Partitions are about to be released, so flush metrics first. */
572 12 : fd_accdb_flush_metrics( accdb );
573 :
574 : /* Wait for any pending background command (purge) on T2 to finish
575 : before releasing partitions. */
576 12 : wait_cmd( accdb );
577 :
578 12 : ulong cur_partition_max = shmem->partition_max;
579 :
580 : /* Materialize the active partition's write_offset from the whead
581 : before releasing. Closed partitions have write_offset set by
582 : change_partition, but the last active partition still has
583 : write_offset == 0 from its initialization. The real byte offset
584 : is encoded in whead[0]. */
585 12 : if( shmem->has_partition[ 0 ] && cur_partition_max>recover->partition_max ) {
586 6 : ulong active_idx = packed_partition_idx( &shmem->whead[ 0 ] );
587 6 : if( active_idx>=recover->partition_max && active_idx<cur_partition_max ) {
588 6 : fd_accdb_partition_t * active = partition_pool_ele( accdb->partition_pool, active_idx );
589 6 : active->write_offset = packed_partition_offset( &shmem->whead[ 0 ] );
590 6 : }
591 6 : }
592 :
593 : /* Release partitions that have been previously allocated. Must hold
594 : partition_lock because partition_pool_ele_release mutates the
595 : pool free list. Before releasing, unlink any partition that sits
596 : on a compaction dlist (queued flag).
597 :
598 : Release in descending index order so that the LIFO free list
599 : re-acquires them in ascending order (P, P+1, P+2, ...). This
600 : keeps reserve_next_write in sync with snapwr, which advances
601 : its flat file offset sequentially. */
602 12 : spin_lock_acquire( &shmem->partition_lock );
603 24 : for( ulong p=cur_partition_max; p>recover->partition_max; p-- ) {
604 12 : fd_accdb_partition_t * part = partition_pool_ele( accdb->partition_pool, p-1UL );
605 12 : if( FD_UNLIKELY( part->queued ) ) {
606 6 : compaction_dlist_ele_remove( accdb->compaction_dlist[ part->layer ], part, accdb->partition_pool );
607 6 : }
608 12 : FD_VOLATILE( part->write_offset ) = 0UL;
609 12 : partition_pool_ele_release( accdb->partition_pool, part );
610 12 : }
611 :
612 12 : shmem->whead[ 0 ].val = recover->whead_val;
613 12 : shmem->has_partition[ 0 ] = recover->has_partition;
614 12 : shmem->partition_max = recover->partition_max;
615 :
616 : /* disk_used_bytes is NOT saved/restored here. It is implicitly
617 : reverted by purge_inner -> acc_unlink, which decrements
618 : disk_used_bytes for each unlinked entry. The caller must
619 : complete the purge before calling revert_whead. */
620 :
621 12 : shmem->shmetrics->disk_current_bytes = recover->disk_current_bytes;
622 12 : shmem->shmetrics->disk_allocated_bytes = recover->partition_max * shmem->partition_sz;
623 :
624 12 : if( recover->has_partition ) {
625 12 : accdb_offset_t sp_off = (accdb_offset_t){ .val = recover->whead_val };
626 12 : ulong sp_idx = packed_partition_idx( &sp_off );
627 12 : fd_accdb_partition_t * sp = partition_pool_ele( accdb->partition_pool, sp_idx );
628 12 : sp->bytes_freed = recover->savepoint_bytes_freed;
629 12 : sp->write_offset = 0UL;
630 12 : }
631 :
632 12 : spin_lock_release( &shmem->partition_lock );
633 12 : }
634 :
635 : fd_accdb_t *
636 4170 : fd_accdb_join( void * shaccdb ) {
637 4170 : if( FD_UNLIKELY( !shaccdb ) ) {
638 0 : FD_LOG_WARNING(( "NULL shaccdb" ));
639 0 : return NULL;
640 0 : }
641 :
642 4170 : if( FD_UNLIKELY( !fd_ulong_is_aligned( (ulong)shaccdb, fd_accdb_align() ) ) ) {
643 0 : FD_LOG_WARNING(( "misaligned shaccdb" ));
644 0 : return NULL;
645 0 : }
646 :
647 4170 : return (fd_accdb_t*)shaccdb;
648 4170 : }
649 :
650 : fd_accdb_t *
651 : fd_accdb_join_readonly( void * ljoin,
652 : fd_accdb_shmem_t * shmem,
653 : ulong * my_epoch_slot_rw,
654 0 : int fd_ro ) {
655 0 : if( FD_UNLIKELY( !ljoin ) ) {
656 0 : FD_LOG_WARNING(( "NULL ljoin" ));
657 0 : return NULL;
658 0 : }
659 :
660 0 : if( FD_UNLIKELY( !fd_ulong_is_aligned( (ulong)ljoin, fd_accdb_align() ) ) ) {
661 0 : FD_LOG_WARNING(( "misaligned ljoin" ));
662 0 : return NULL;
663 0 : }
664 :
665 0 : if( FD_UNLIKELY( !my_epoch_slot_rw ) ) {
666 0 : FD_LOG_WARNING(( "NULL my_epoch_slot_rw" ));
667 0 : return NULL;
668 0 : }
669 :
670 0 : ulong max_live_slots = shmem->max_live_slots;
671 0 : ulong max_accounts = shmem->max_accounts;
672 0 : ulong max_account_writes_per_slot = shmem->max_account_writes_per_slot;
673 0 : ulong partition_cnt = shmem->partition_cnt;
674 :
675 0 : ulong chain_cnt = fd_ulong_pow2_up( (max_accounts>>1) + (max_accounts&1UL) );
676 0 : ulong txn_max = max_live_slots * max_account_writes_per_slot;
677 :
678 : /* Recompute the same shmem scratch layout that fd_accdb_shmem_new
679 : used. All FD_SCRATCH_ALLOC_APPEND calls here only compute pointer
680 : offsets — they do not write to shmem. */
681 0 : FD_SCRATCH_ALLOC_INIT( l, shmem );
682 0 : FD_SCRATCH_ALLOC_APPEND( l, FD_ACCDB_SHMEM_ALIGN, sizeof(fd_accdb_shmem_t) );
683 0 : void * _fork_pool_ele = FD_SCRATCH_ALLOC_APPEND( l, alignof(fd_accdb_fork_shmem_t), max_live_slots*sizeof(fd_accdb_fork_shmem_t) );
684 0 : void * _descends_sets = FD_SCRATCH_ALLOC_APPEND( l, descends_set_align(), max_live_slots*descends_set_footprint( max_live_slots ) );
685 0 : void * _acc_map = FD_SCRATCH_ALLOC_APPEND( l, alignof(uint), chain_cnt*sizeof(uint) );
686 0 : void * _acc_pool_ele = FD_SCRATCH_ALLOC_APPEND( l, alignof(fd_accdb_accmeta_t), max_accounts*sizeof(fd_accdb_accmeta_t) );
687 0 : FD_SCRATCH_ALLOC_APPEND( l, alignof(fd_accdb_txn_t), txn_max*sizeof(fd_accdb_txn_t) );
688 0 : FD_SCRATCH_ALLOC_APPEND( l, partition_pool_align(), partition_pool_footprint( partition_cnt ) );
689 0 : for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
690 0 : FD_SCRATCH_ALLOC_APPEND( l, compaction_dlist_align(), compaction_dlist_footprint() );
691 0 : }
692 0 : FD_SCRATCH_ALLOC_APPEND( l, deferred_free_dlist_align(), deferred_free_dlist_footprint() );
693 :
694 0 : FD_SCRATCH_ALLOC_INIT( l2, ljoin );
695 0 : fd_accdb_t * accdb = FD_SCRATCH_ALLOC_APPEND( l2, fd_accdb_align(), sizeof(fd_accdb_t) );
696 0 : void * _local_fork_pool = FD_SCRATCH_ALLOC_APPEND( l2, alignof(fd_accdb_fork_t), max_live_slots*sizeof(fd_accdb_fork_t) );
697 :
698 0 : accdb->fd = fd_ro;
699 0 : accdb->acquire_state = FD_ACCDB_ACQUIRE_STATE_IDLE;
700 0 : accdb->shmem = shmem;
701 0 : FD_TEST( acc_pool_join( accdb->acc_pool_join, shmem->acc_pool, _acc_pool_ele, max_accounts ) );
702 0 : accdb->acc_pool = accdb->acc_pool_join->ele;
703 0 : accdb->acc_map = _acc_map;
704 0 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) accdb->cache[ c ] = (uchar *)shmem + shmem->cache_region_off[ c ];
705 :
706 : /* Writer-only structures: leave NULL so any accidental writer-path
707 : call from a readonly joiner crashes loudly rather than corrupting
708 : state. */
709 0 : accdb->partition_pool = NULL;
710 0 : for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) accdb->compaction_dlist[ k ] = NULL;
711 0 : accdb->deferred_free_dlist = NULL;
712 0 : accdb->delta.chains = NULL;
713 0 : accdb->delta.pool = NULL;
714 :
715 0 : FD_TEST( fork_pool_join( accdb->fork_shmem_pool, shmem->fork_pool, _fork_pool_ele, max_live_slots ) );
716 0 : accdb->fork_pool = _local_fork_pool;
717 0 : for( ulong i=0UL; i<max_live_slots; i++ ) {
718 0 : fd_accdb_fork_t * fork = &accdb->fork_pool[ i ];
719 0 : fork->shmem = fork_pool_ele( accdb->fork_shmem_pool, i );
720 0 : fork->descends = descends_set_join( (uchar *)_descends_sets + i*descends_set_footprint( max_live_slots ) );
721 0 : FD_TEST( fork->shmem );
722 0 : FD_TEST( fork->descends );
723 0 : }
724 :
725 : /* my_epoch_slot_rw points at memory owned by this joiner (e.g. a
726 : private per-tile fseq) that the joiner can write to. The
727 : accdb tile sees it via its external_epoch_slots[] array (mapped
728 : read-only) and includes it in its compaction epoch scan.
729 : Storing through this pointer is the only side effect a readonly
730 : joiner has on shared state. */
731 0 : accdb->my_epoch_slot = my_epoch_slot_rw;
732 :
733 : /* Readonly joiners do not own external slots themselves; only the
734 : compaction tile / writer joiners do. */
735 0 : accdb->external_epoch_slots = NULL;
736 0 : accdb->external_epoch_cnt = 0UL;
737 :
738 0 : accdb->deferred_acc_buf = NULL;
739 0 : accdb->deferred_fork_head = NULL;
740 0 : accdb->deferred_fork_tail = NULL;
741 0 : accdb->deferred_fork_epoch = 0UL;
742 :
743 0 : memset( accdb->metrics, 0, sizeof(fd_accdb_metrics_t) );
744 0 : memset( &accdb->write_stats, 0, sizeof(accdb->write_stats) );
745 :
746 0 : return accdb;
747 0 : }
748 :
749 : /* T1 -> T2 cmd channel. Two states on cmd_op:
750 :
751 : IDLE - no cmd in flight
752 : non-IDLE - cmd pending; T2 will process it then flip back to IDLE
753 :
754 : T1 submits by writing fork_id then cmd_op (non-IDLE). T2 processes
755 : by reading fork_id then writing cmd_op = IDLE. T1 waits for IDLE
756 : before submitting again, so T2 never sees a half-written cmd and
757 : never re-processes the same cmd. */
758 :
759 : static inline void
760 9369 : wait_cmd( fd_accdb_t * accdb ) {
761 9369 : fd_accdb_shmem_t * shmem = accdb->shmem;
762 11559410 : while( FD_VOLATILE_CONST( shmem->cmd_op )!=FD_ACCDB_CMD_IDLE ) FD_SPIN_PAUSE();
763 9369 : FD_COMPILER_MFENCE();
764 9369 : }
765 :
766 : static inline void
767 : submit_cmd( fd_accdb_t * accdb,
768 : uint op,
769 528 : ushort fork_id ) {
770 528 : fd_accdb_shmem_t * shmem = accdb->shmem;
771 528 : FD_VOLATILE( shmem->cmd_fork_id ) = fork_id;
772 528 : FD_COMPILER_MFENCE();
773 528 : FD_VOLATILE( shmem->cmd_op ) = op;
774 528 : }
775 :
776 : fd_accdb_fork_id_t
777 : fd_accdb_attach_child( fd_accdb_t * accdb,
778 8829 : fd_accdb_fork_id_t parent_fork_id ) {
779 : /* fork_pool_acquire is not NULL-checked: replay gates attaches on
780 : fd_banks_can_start_bank, and wait_cmd ensures the prior
781 : advance_root has fully run on T2, so
782 : live + deferred forks <= max_live_slots. */
783 8829 : wait_cmd( accdb );
784 :
785 8829 : if( FD_UNLIKELY( accdb->shmem->generation==UINT_MAX ) ) FD_LOG_ERR(( "accdb ran out of generation sequence numbers, restart required" ));
786 :
787 8829 : fd_accdb_fork_shmem_t * acquired = fork_pool_acquire( accdb->fork_shmem_pool );
788 8829 : if( FD_UNLIKELY( !acquired ) ) {
789 3 : submit_cmd( accdb, FD_ACCDB_CMD_DRAIN_DEFERRED, USHORT_MAX );
790 3 : wait_cmd( accdb );
791 3 : acquired = fork_pool_acquire( accdb->fork_shmem_pool );
792 3 : FD_CHECK_CRIT( acquired, "accdb fork pool exhausted after deferred drain" );
793 3 : }
794 8829 : ulong idx = fork_pool_idx( accdb->fork_shmem_pool, acquired );
795 :
796 8829 : fd_accdb_fork_t * fork = &accdb->fork_pool[ idx ];
797 8829 : fd_accdb_fork_id_t fork_id = { .val = (ushort)idx };
798 :
799 8829 : fork->shmem->child_id = (fd_accdb_fork_id_t){ .val = USHORT_MAX };
800 :
801 8829 : if( FD_LIKELY( parent_fork_id.val==USHORT_MAX ) ) {
802 4110 : fork->shmem->parent_id = (fd_accdb_fork_id_t){ .val = USHORT_MAX };
803 4110 : fork->shmem->sibling_id = (fd_accdb_fork_id_t){ .val = USHORT_MAX };
804 :
805 4110 : descends_set_null( fork->descends );
806 4110 : accdb->shmem->root_fork_id = fork_id;
807 4719 : } else {
808 4719 : fd_accdb_fork_t * parent = &accdb->fork_pool[ parent_fork_id.val ];
809 4719 : fork->shmem->parent_id = parent_fork_id;
810 :
811 4719 : descends_set_copy( fork->descends, parent->descends );
812 4719 : descends_set_insert( fork->descends, parent_fork_id.val );
813 :
814 : /* Atomically prepend to parent's child list. T2 (background_purge)
815 : may concurrently unlink a different child from the same list, so
816 : we must CAS here. */
817 4719 : FD_COMPILER_MFENCE();
818 4719 : for(;;) {
819 4719 : ushort old_head = FD_VOLATILE_CONST( parent->shmem->child_id.val );
820 4719 : fork->shmem->sibling_id = (fd_accdb_fork_id_t){ .val = old_head };
821 4719 : FD_COMPILER_MFENCE();
822 4719 : if( FD_LIKELY( FD_ATOMIC_CAS( &parent->shmem->child_id.val, old_head, fork_id.val )==old_head ) ) break;
823 0 : FD_SPIN_PAUSE();
824 0 : }
825 4719 : }
826 :
827 8829 : fork->shmem->generation = accdb->shmem->generation++;
828 8829 : fork->shmem->txn_head = UINT_MAX;
829 :
830 8829 : FD_TEST( !descends_set_test( fork->descends, fork_id.val ) );
831 :
832 8829 : return fork_id;
833 8829 : }
834 :
835 : /* evict_clear_acc_cache_ref atomically tears down acc->cache_idx and
836 : acc->executable_size.CACHE_VALID for an acc that is being evicted
837 : from cache line (size_class, line_idx). The caller must already
838 : hold an exclusive claim on the line (line->refcnt ==
839 : FD_ACCDB_EVICT_SENTINEL) so that no concurrent thread can pin the
840 : line.
841 :
842 : The naive sequence (clear cache_idx, clear VALID) lets a reader in
843 : cold_load_acc see VALID=1 and read a stale INVAL cache_idx, which
844 : decodes to an OOB cache_line pointer. The reverse sequence (clear
845 : VALID, clear cache_idx) lets a concurrent cold_load_acc observe
846 : VALID=0/CLAIM=0 and start publishing a *new* cache_idx + VALID=1
847 : between our two stores; our later cache_idx=INVAL would then
848 : stomp on the cold-loader's published idx.
849 :
850 : We close both races by acquiring CACHE_CLAIM_BIT before mutating
851 : acc->cache_idx. cold_load_acc spins while CLAIM is held, so it
852 : cannot enter the publish path concurrently. If CLAIM is already
853 : held, a cold-loader is already mid-publish; in that case
854 : acc->cache_idx is being repointed away from our line, and we must
855 : not touch it. After mutation we release CLAIM.
856 :
857 : Verifies acc->cache_idx still encodes (size_class, line_idx) before
858 : clobbering, in case the acc was concurrently re-published into a
859 : different line (e.g. by a previous cold_load_acc completing before
860 : we arrived). */
861 :
862 : static inline void
863 : evict_clear_acc_cache_ref( fd_accdb_accmeta_t * accmeta,
864 : ulong size_class,
865 366 : ulong line_idx ) {
866 366 : uint expected_cidx = FD_ACCDB_ACC_CIDX_PACK( (uint)size_class, (uint)line_idx );
867 :
868 : /* CAS-acquire CLAIM. If a cold-loader already holds CLAIM, they
869 : own the publish path; bail without touching accmeta fields (their
870 : republish is repointing accmeta->cache_idx away from our line). */
871 366 : for(;;) {
872 366 : uint cur = FD_VOLATILE_CONST( accmeta->executable_size );
873 366 : if( FD_UNLIKELY( cur & FD_ACCDB_SIZE_CACHE_CLAIM_BIT ) ) return;
874 366 : uint nxt = cur | FD_ACCDB_SIZE_CACHE_CLAIM_BIT;
875 366 : if( FD_LIKELY( FD_ATOMIC_CAS( &accmeta->executable_size, cur, nxt )==cur ) ) break;
876 0 : fd_racesan_hook( "accdb_evict_clear:claim_wait" );
877 0 : FD_SPIN_PAUSE();
878 0 : }
879 :
880 366 : fd_racesan_hook( "accdb_evict_clear:post_claim" );
881 :
882 : /* CLAIM held. If accmeta->cache_idx still points at our line, clear
883 : VALID and INVAL the cache_idx. Otherwise the accmeta was already
884 : re-published into a different line; leave it alone. */
885 366 : if( FD_LIKELY( FD_VOLATILE_CONST( accmeta->cache_idx )==expected_cidx ) ) {
886 366 : FD_ATOMIC_FETCH_AND_AND( &accmeta->executable_size, ~FD_ACCDB_SIZE_CACHE_VALID_BIT );
887 366 : FD_VOLATILE( accmeta->cache_idx ) = FD_ACCDB_ACC_CIDX_INVAL;
888 366 : }
889 :
890 : /* Release CLAIM. */
891 366 : FD_ATOMIC_FETCH_AND_AND( &accmeta->executable_size, ~FD_ACCDB_SIZE_CACHE_CLAIM_BIT );
892 366 : }
893 :
894 : /* cache_free_push pushes a fully-freed cache line onto the per-class
895 : CAS free list (Treiber stack). The caller must have already
896 : invalidated the line (key.generation==UINT_MAX, acc_idx==UINT_MAX)
897 : and set persisted=1 before pushing. Those stores must also be
898 : ordered before the store that releases refcnt to 0. */
899 :
900 : static inline void
901 : cache_free_push( fd_accdb_t * accdb,
902 : ulong size_class,
903 819951 : fd_accdb_cache_line_t * line ) {
904 819951 : ulong line_idx = cache_line_idx( accdb, size_class, line );
905 819951 : for(;;) {
906 819951 : ulong old_vt = FD_VOLATILE_CONST( accdb->shmem->cache_free[ size_class ].ver_top );
907 819951 : uint old_top = (uint)( old_vt & (ulong)UINT_MAX );
908 819951 : uint old_ver = (uint)( old_vt >> 32 );
909 819951 : line->next = old_top;
910 819951 : FD_COMPILER_MFENCE();
911 819951 : ulong new_vt = ((ulong)(uint)( old_ver+1U ) << 32) | (ulong)(uint)line_idx;
912 819951 : if( FD_LIKELY( FD_ATOMIC_CAS( &accdb->shmem->cache_free[ size_class ].ver_top, old_vt, new_vt )==old_vt ) ) {
913 819951 : FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->cache_free_cnt[ size_class ].val, 1UL );
914 819951 : return;
915 819951 : }
916 0 : FD_SPIN_PAUSE();
917 0 : }
918 819951 : }
919 :
920 : /* cache_free_pop pops a line from the per-class CAS free list. Returns
921 : NULL if the list is empty. */
922 :
923 : static inline fd_accdb_cache_line_t *
924 : cache_free_pop( fd_accdb_t * accdb,
925 932526 : ulong size_class ) {
926 932526 : for(;;) {
927 932526 : ulong old_vt = FD_VOLATILE_CONST( accdb->shmem->cache_free[ size_class ].ver_top );
928 932526 : uint old_top = (uint)( old_vt & (ulong)UINT_MAX );
929 932526 : if( FD_UNLIKELY( old_top==UINT_MAX ) ) return NULL;
930 788325 : uint old_ver = (uint)( old_vt >> 32 );
931 788325 : fd_accdb_cache_line_t * top = cache_line( accdb, size_class, (ulong)old_top );
932 788325 : uint next = FD_VOLATILE_CONST( top->next );
933 788325 : ulong new_vt = ((ulong)(uint)( old_ver+1U ) << 32) | (ulong)next;
934 788325 : if( FD_LIKELY( FD_ATOMIC_CAS( &accdb->shmem->cache_free[ size_class ].ver_top, old_vt, new_vt )==old_vt ) ) {
935 788325 : FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->cache_free_cnt[ size_class ].val, 1UL );
936 788325 : return top;
937 788325 : }
938 0 : FD_SPIN_PAUSE();
939 0 : }
940 932526 : }
941 :
942 : /* cache_try_pin attempts a lock-free pin of a cache-hit line. Returns
943 : the line if successfully pinned, or NULL if the line is being evicted
944 : or was recycled (ABA). */
945 :
946 : static inline fd_accdb_cache_line_t *
947 : cache_try_pin( fd_accdb_cache_line_t * line,
948 : uchar const pubkey[ 32 ],
949 97542 : uint generation ) {
950 97542 : for(;;) {
951 97542 : uint old_rc = FD_VOLATILE_CONST( line->refcnt );
952 97542 : if( FD_UNLIKELY( old_rc==FD_ACCDB_EVICT_SENTINEL ) ) return NULL;
953 : /* No saturation guard needed: refcnt is a uint and at most
954 : FD_ACCDB_MAX_JOINERS (256) threads can pin concurrently,
955 : so old_rc+1 can never reach FD_ACCDB_EVICT_SENTINEL
956 : (UINT_MAX) or wrap. */
957 97542 : if( FD_LIKELY( FD_ATOMIC_CAS( &line->refcnt, old_rc, old_rc+1U )==old_rc ) ) {
958 : /* Pinned. ABA check: verify the key hasn't changed under us. */
959 97542 : fd_racesan_hook( "accdb_try_pin:post_cas" );
960 97542 : FD_COMPILER_MFENCE();
961 97542 : if( FD_UNLIKELY( line->key.generation!=generation ||
962 97542 : memcmp( line->key.pubkey, pubkey, 32UL ) ) ) {
963 0 : FD_ATOMIC_FETCH_AND_SUB( &line->refcnt, 1U );
964 0 : return NULL;
965 0 : }
966 97542 : line->referenced = 1;
967 97542 : fd_racesan_hook( "cache_try_pin:pinned" );
968 97542 : return line;
969 97542 : }
970 0 : FD_SPIN_PAUSE();
971 0 : }
972 97542 : }
973 :
974 : /* wait_for_epoch_drain spins until every joiner's published epoch
975 : exceeds tag, meaning all readers that were active at epoch=tag have
976 : since exited their critical sections. */
977 :
978 : static void
979 : wait_for_epoch_drain( fd_accdb_t * accdb,
980 864 : ulong tag ) {
981 864 : for(;;) {
982 864 : ulong min_epoch = ULONG_MAX;
983 864 : ulong joiner_cnt = FD_VOLATILE_CONST( accdb->shmem->joiner_cnt );
984 1746 : for( ulong t=0UL; t<joiner_cnt; t++ ) {
985 882 : ulong e = FD_VOLATILE_CONST( accdb->shmem->joiner_epochs[ t ].val );
986 882 : if( FD_LIKELY( e<min_epoch ) ) min_epoch = e;
987 882 : }
988 864 : for( ulong t=0UL; t<accdb->external_epoch_cnt; t++ ) {
989 0 : ulong e = FD_VOLATILE_CONST( *accdb->external_epoch_slots[ t ] );
990 0 : if( FD_LIKELY( e<min_epoch ) ) min_epoch = e;
991 0 : }
992 864 : if( FD_LIKELY( tag<min_epoch ) ) break;
993 0 : fd_racesan_hook( "accdb_epoch_drain:wait" );
994 0 : FD_SPIN_PAUSE();
995 0 : }
996 864 : }
997 :
998 : /* drain_deferred_frees releases back to their respective pools any acc
999 : batch and/or fork slots that were unlinked in a prior advance_root /
1000 : purge call. The resources cannot be released immediately because
1001 : concurrent readers may still reference them. We wait until every
1002 : joiner's published epoch exceeds the tag stamped when each resource
1003 : was unlinked.
1004 :
1005 : Must be called before creating new deferred batches (there is at most
1006 : one of each outstanding at a time). */
1007 :
1008 : static void
1009 525 : drain_deferred_frees( fd_accdb_t * accdb ) {
1010 525 : if( FD_UNLIKELY( accdb->deferred_fork_head ) ) {
1011 435 : wait_for_epoch_drain( accdb, accdb->deferred_fork_epoch );
1012 435 : fork_pool_release_chain( accdb->fork_shmem_pool, accdb->deferred_fork_head, accdb->deferred_fork_tail );
1013 435 : accdb->deferred_fork_head = NULL;
1014 435 : accdb->deferred_fork_tail = NULL;
1015 435 : }
1016 :
1017 525 : ulong n = accdb->shmem->deferred_acc_buf_cnt;
1018 525 : if( FD_LIKELY( !n ) ) return;
1019 429 : wait_for_epoch_drain( accdb, accdb->shmem->deferred_acc_epoch );
1020 :
1021 : /* All readers that could have been holding a captured pointer to any
1022 : of these accs at unlink time have now exited their epoch sections.
1023 : It is safe to materialize pool.next links and hand the chain to
1024 : acc_pool_release_chain. */
1025 429 : uint * buf = accdb->deferred_acc_buf;
1026 429 : fd_accdb_accmeta_t * acc_pool = accdb->acc_pool;
1027 :
1028 : /* Late-publish sweep: a concurrent acquire evictor may have published
1029 : a new offset into one of these accmetas after acc_unlink's
1030 : xchg-to-INVAL but before exiting its epoch. Now that the epoch has
1031 : drained, any such publish is complete and visible. Free the
1032 : orphaned disk bytes here, before the accmeta is released to the
1033 : pool and its fields recycled. */
1034 429 : ulong acc_pool_cap = acc_pool_ele_max( accdb->acc_pool_join );
1035 1515 : for( ulong i=0UL; i<n; i++ ) {
1036 1086 : FD_TEST( (ulong)buf[ i ]<acc_pool_cap );
1037 1086 : fd_accdb_accmeta_t * accmeta = &acc_pool[ buf[ i ] ];
1038 1086 : ulong off = fd_accdb_acc_offset( accmeta );
1039 1086 : if( FD_UNLIKELY( off!=FD_ACCDB_OFF_INVAL ) ) {
1040 0 : ulong entry_sz = (ulong)FD_ACCDB_SIZE_DATA(accmeta->executable_size)+sizeof(fd_accdb_disk_meta_t);
1041 0 : fd_accdb_shmem_bytes_freed( accdb->shmem, off, entry_sz );
1042 0 : FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->shmetrics->disk_used_bytes, entry_sz );
1043 0 : }
1044 1086 : }
1045 :
1046 1086 : for( ulong i=0UL; i+1UL<n; i++ ) {
1047 657 : acc_pool[ buf[ i ] ].pool.next = acc_pool_private_cidx( (ulong)buf[ i+1UL ] );
1048 657 : }
1049 429 : fd_accdb_accmeta_t * head = &acc_pool[ buf[ 0UL ] ];
1050 429 : fd_accdb_accmeta_t * tail = &acc_pool[ buf[ n-1UL ] ];
1051 429 : acc_pool_release_chain( accdb->acc_pool_join, head, tail );
1052 429 : accdb->shmem->deferred_acc_buf_cnt = 0UL;
1053 429 : }
1054 :
1055 : /* deferred_acc_append records an unlinked acc index in the side buffer
1056 : for later release after wait_for_epoch_drain. T2 is the sole writer.
1057 : The chain link from acc->pool.next is NOT laid down here: pool.next
1058 : is union-aliased to cache_idx, and a concurrent cold_load_acc may
1059 : still publish through a captured pointer until the epoch drains.
1060 : Materialization of the chain happens in drain_deferred_frees. */
1061 :
1062 : static inline void
1063 : deferred_acc_append( fd_accdb_t * accdb,
1064 1455 : uint acc_idx ) {
1065 1455 : fd_accdb_shmem_t * shmem = accdb->shmem;
1066 1455 : FD_TEST( shmem->deferred_acc_buf_cnt<shmem->deferred_acc_buf_max );
1067 1455 : accdb->deferred_acc_buf[ shmem->deferred_acc_buf_cnt++ ] = acc_idx;
1068 1455 : }
1069 :
1070 : /* acc_unlink unlinks an account from its hash map chain, frees any
1071 : associated disk bytes, and invalidates a stale cache reference. Does
1072 : NOT release the acc pool slot — the caller is responsible for that
1073 : (or for batching releases).
1074 :
1075 : prev is the previous element in the map chain (UINT_MAX if acc_idx is
1076 : the head).
1077 :
1078 : CONCURRENCY: The chain link being removed is swapped out with a CAS
1079 : so that a concurrent fd_accdb_release prepending to the same chain
1080 : cannot lose its update. If a head-removal CAS fails (a new node was
1081 : prepended since we loaded the head), we re-walk from the new head to
1082 : find the target as an interior node. Interior CAS cannot fail from
1083 : inserts (inserts only touch the head) and only one remover exists at
1084 : a time (advance_root / purge are serialized). */
1085 :
1086 : static inline void
1087 : acc_unlink( fd_accdb_t * accdb,
1088 : uint map_idx,
1089 : uint prev,
1090 1455 : uint acc_idx ) {
1091 1455 : fd_accdb_accmeta_t * accmeta = &accdb->acc_pool[ acc_idx ];
1092 :
1093 : /* Atomically capture and clear the offset. Two races to defuse:
1094 :
1095 : (1) A concurrent fd_accdb_acquire_inner that is CLOCK-evicting the
1096 : cache line currently holding this acc's data may have already
1097 : xchg'd the offset to INVAL in step 5-6 and freed the old disk
1098 : bytes. Without atomicity we would re-read the old offset and
1099 : free those same bytes a second time. The xchg here serializes:
1100 : whoever wins sees the real offset and frees; the loser sees
1101 : INVAL and skips.
1102 :
1103 : (2) That same evictor may also be mid-flight to publish a NEW
1104 : offset in step 9 (after step 5-6's free but before step 9's
1105 : store). That late publish lands on an accmeta that is about
1106 : to be chain-unlinked and deferred-released. drain_deferred_
1107 : frees sweeps the deferred buffer after epoch drain to catch
1108 : the late publish and free the orphaned bytes. */
1109 1455 : ulong entry_sz = (ulong)FD_ACCDB_SIZE_DATA(accmeta->executable_size)+sizeof(fd_accdb_disk_meta_t);
1110 1455 : ulong old_offset = fd_accdb_acc_xchg_offset( accmeta, FD_ACCDB_OFF_INVAL );
1111 1455 : if( FD_LIKELY( old_offset!=FD_ACCDB_OFF_INVAL ) ) {
1112 36 : fd_accdb_shmem_bytes_freed( accdb->shmem, old_offset, entry_sz );
1113 36 : FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->shmetrics->disk_used_bytes, entry_sz );
1114 36 : }
1115 1455 : FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->shmetrics->accounts_total, 1UL );
1116 1455 : accdb->metrics->accounts_deleted++;
1117 :
1118 1455 : if( FD_LIKELY( prev==UINT_MAX ) ) {
1119 : /* Head removal — CAS may fail if a concurrent insert prepended a
1120 : new node. On failure the target is now interior. */
1121 48 : for(;;) {
1122 48 : uint old_head = FD_VOLATILE_CONST( accdb->acc_map[ map_idx ] );
1123 48 : if( FD_LIKELY( old_head==acc_idx ) ) {
1124 48 : if( FD_LIKELY( FD_ATOMIC_CAS( &accdb->acc_map[ map_idx ], acc_idx, accmeta->map.next )==acc_idx ) ) break;
1125 0 : FD_SPIN_PAUSE();
1126 0 : continue;
1127 48 : }
1128 : /* Head changed — walk from new head to find prev for interior
1129 : removal. The target must still be in the chain because only
1130 : this thread removes elements. */
1131 0 : prev = old_head;
1132 0 : while( FD_VOLATILE_CONST( accdb->acc_pool[ prev ].map.next )!=acc_idx ) prev = FD_VOLATILE_CONST( accdb->acc_pool[ prev ].map.next );
1133 0 : FD_ATOMIC_CAS( &accdb->acc_pool[ prev ].map.next, acc_idx, accmeta->map.next );
1134 0 : break;
1135 48 : }
1136 1407 : } else {
1137 1407 : FD_ATOMIC_CAS( &accdb->acc_pool[ prev ].map.next, acc_idx, accmeta->map.next );
1138 1407 : }
1139 :
1140 1455 : fd_racesan_hook( "accdb_acc_unlink:post_splice" );
1141 :
1142 : /* If the freed acc still has a cached location, invalidate it and
1143 : try to reclaim the cache line so the eviction path does not try
1144 : to write back stale data from a recycled pool slot. Lock-free:
1145 : CAS the refcnt 0 -> EVICT_SENTINEL to claim it exclusively, then
1146 : push to the CAS free list. If the line is pinned (refcnt>0),
1147 : skip, the pinner's release will handle it.
1148 :
1149 : Acquire CACHE_CLAIM_BIT before touching acc->cache_idx /
1150 : CACHE_VALID — see evict_clear_acc_cache_ref for the protocol.
1151 : Without CLAIM, a concurrent cold_load_acc can publish a fresh
1152 : (cache_idx, VALID=1) pair into this acc between our two stores,
1153 : and our subsequent cache_idx=INVAL stomps onto the freelist
1154 : pool.next field (the union sibling of cache_idx), corrupting the
1155 : pool. Unlike evict_clear_acc_cache_ref, we cannot bail when CLAIM
1156 : is held: this acc is being permanently unlinked, so we must
1157 : spin-wait for the cold-loader to release CLAIM and then invalidate
1158 : whatever cache_idx is current. */
1159 1455 : uint cur_es;
1160 1455 : for(;;) {
1161 1455 : cur_es = FD_VOLATILE_CONST( accmeta->executable_size );
1162 1455 : if( FD_UNLIKELY( cur_es & FD_ACCDB_SIZE_CACHE_CLAIM_BIT ) ) { FD_SPIN_PAUSE(); continue; }
1163 1455 : uint nxt_es = cur_es | FD_ACCDB_SIZE_CACHE_CLAIM_BIT;
1164 1455 : if( FD_LIKELY( FD_ATOMIC_CAS( &accmeta->executable_size, cur_es, nxt_es )==cur_es ) ) break;
1165 0 : FD_SPIN_PAUSE();
1166 0 : }
1167 :
1168 1455 : uint cidx = FD_ACCDB_ACC_CIDX_INVAL;
1169 1455 : int had_valid = FD_ACCDB_SIZE_CACHE_VALID( cur_es );
1170 1455 : if( FD_UNLIKELY( had_valid ) ) {
1171 1419 : cidx = FD_VOLATILE_CONST( accmeta->cache_idx );
1172 : /* Clear VALID before INVAL'ing cache_idx — matches the order in
1173 : evict_clear_acc_cache_ref so cold_load_acc's "VALID=1 +
1174 : cidx=INVAL" spin path resolves on the next iteration when it
1175 : observes VALID=0. */
1176 1419 : FD_ATOMIC_FETCH_AND_AND( &accmeta->executable_size, ~FD_ACCDB_SIZE_CACHE_VALID_BIT );
1177 1419 : FD_VOLATILE( accmeta->cache_idx ) = FD_ACCDB_ACC_CIDX_INVAL;
1178 1419 : }
1179 :
1180 : /* Release CLAIM. */
1181 1455 : FD_ATOMIC_FETCH_AND_AND( &accmeta->executable_size, ~FD_ACCDB_SIZE_CACHE_CLAIM_BIT );
1182 :
1183 1455 : if( FD_UNLIKELY( had_valid ) ) {
1184 1419 : fd_accdb_cache_line_t * stale = cache_line( accdb, FD_ACCDB_ACC_CIDX_CLASS( cidx ), FD_ACCDB_ACC_CIDX_IDX( cidx ) );
1185 1419 : fd_racesan_hook( "acc_unlink:pre_reclaim_cas" );
1186 1419 : uint old_rc = FD_ATOMIC_CAS( &stale->refcnt, 0U, FD_ACCDB_EVICT_SENTINEL );
1187 1419 : fd_racesan_hook( "acc_unlink:post_reclaim_cas" );
1188 1419 : if( FD_LIKELY( !old_rc ) ) {
1189 : /* Claimed. Validate key (ABA, slot could have been recycled
1190 : between our read of cache_idx and the CAS). */
1191 1419 : if( FD_LIKELY( stale->key.generation==accmeta->key.generation &&
1192 1419 : !memcmp( stale->key.pubkey, accmeta->key.pubkey, 32UL ) ) ) {
1193 1419 : ulong sc = FD_ACCDB_ACC_CIDX_CLASS( cidx );
1194 1419 : stale->key.generation = UINT_MAX;
1195 1419 : stale->persisted = 1;
1196 1419 : stale->acc_idx = UINT_MAX;
1197 1419 : FD_COMPILER_MFENCE();
1198 1419 : FD_VOLATILE( stale->refcnt ) = 0;
1199 1419 : cache_free_push( accdb, sc, stale );
1200 1419 : } else {
1201 : /* Wrong line (ABA). Release claim. */
1202 0 : FD_VOLATILE( stale->refcnt ) = 0;
1203 0 : }
1204 1419 : }
1205 0 : else if( FD_LIKELY( old_rc!=FD_ACCDB_EVICT_SENTINEL ) ) {
1206 : /* The CAS lost to a non-sentinel refcnt, but that does not prove
1207 : `stale` is still our line. Between capturing cidx and here we
1208 : released the claim, so we could have evicted `stale` and
1209 : recycled it to an unrelated account. */
1210 0 : fd_accdb_cache_line_t * mine = cache_try_pin( stale, accmeta->key.pubkey, accmeta->key.generation );
1211 0 : if( FD_LIKELY( mine ) ) {
1212 : /* Genuinely our line, still pinned by a reader. The accmeta
1213 : slot is about to be deferred-released and recycled; if a
1214 : later writeback of this dirty line fires, it would pair the
1215 : recycled accmeta's pubkey with the old owner/data. Set
1216 : persisted so the writeback gate never fires. */
1217 0 : FD_VOLATILE( mine->persisted ) = 1;
1218 :
1219 : /* Only the tombstone self-unlink may be pinned here old-version
1220 : and purge unlinks are never pinned, because a reader on a
1221 : live fork resolves to the newest version, not the one these
1222 : unlink. */
1223 0 : FD_TEST( accmeta->lamports==0UL );
1224 :
1225 0 : FD_ATOMIC_FETCH_AND_SUB( &mine->refcnt, 1U );
1226 0 : }
1227 : /* Else was recycled to a foreign account. Nothing to neutralize,
1228 : leave the line alone. */
1229 0 : } else {
1230 : /* A foreground evictor already claimed this line. It holds its
1231 : epoch acquire and writeback, so drain_deferred_frees cannot
1232 : recycle the slot before it finishes. Its writeback names the
1233 : old account correctly, no poison. */
1234 0 : }
1235 1419 : }
1236 1455 : }
1237 :
1238 : /* fork_slot_defer removes fork_id from every descends_set and chains
1239 : the fork pool slot onto the deferred fork chain for later release.
1240 : The slot must not be released immediately because concurrent readers
1241 : may still reference the fork ID via descends_set or stale chain
1242 : walks.
1243 :
1244 : The eager descends_set_remove here is safe despite being a
1245 : non-atomic RMW that races with concurrent descends_set_test in
1246 : fd_accdb_acquire, for two reasons:
1247 :
1248 : (a) Rooted parent forks: after advance_root publishes the new
1249 : root_fork_id, any acquire loads root_generation >=
1250 : parent->generation. Every account from the old parent has
1251 : generation <= parent->generation, so the
1252 : "generation > root_generation" gate in the chain walk is
1253 : never satisfied and the parent's bit is never tested.
1254 :
1255 : (b) Purged / pruned sibling forks: a purged fork is by
1256 : definition not an ancestor of any live fork, so its bit
1257 : was never set in any live fork's descends_set. Clearing
1258 : it is a literal no-op.
1259 :
1260 : Fork-id ABA after slot reuse is also safe: the fork pool slot
1261 : is not released until drain_deferred_frees, which waits until
1262 : all epoch-protected readers have exited. On x86 (TSO), the
1263 : synchronization chain (T2: bit clear -> epoch FAA; reader:
1264 : epoch load -> epoch_slot store -> mfence -> bit read) guarantees
1265 : that any reader entering a new epoch section after the drain
1266 : will observe the cleared bit before the slot is recycled by
1267 : attach_child. */
1268 :
1269 : static inline void
1270 : fork_slot_defer( fd_accdb_t * accdb,
1271 : fd_accdb_fork_id_t fork_id,
1272 : fd_accdb_fork_shmem_t ** fork_head,
1273 534 : fd_accdb_fork_shmem_t ** fork_tail ) {
1274 11391 : for( ulong i=0UL; i<accdb->shmem->max_live_slots; i++ ) descends_set_remove( accdb->fork_pool[ i ].descends, fork_id.val );
1275 534 : fd_accdb_fork_shmem_t * shmem = fork_pool_ele( accdb->fork_shmem_pool, (ulong)fork_id.val );
1276 534 : if( *fork_tail ) (*fork_tail)->pool.next = fork_pool_private_cidx( (ulong)fork_id.val );
1277 522 : else *fork_head = shmem;
1278 534 : *fork_tail = shmem;
1279 534 : }
1280 :
1281 : static void
1282 : purge_inner( fd_accdb_t * accdb,
1283 : fd_accdb_fork_id_t fork_id,
1284 : fd_accdb_fork_shmem_t ** fork_head,
1285 33 : fd_accdb_fork_shmem_t ** fork_tail ) {
1286 33 : fd_accdb_fork_t * fork = &accdb->fork_pool[ fork_id.val ];
1287 :
1288 33 : fd_accdb_fork_id_t child = fork->shmem->child_id;
1289 42 : while( child.val!=USHORT_MAX ) {
1290 9 : fd_accdb_fork_id_t next = accdb->fork_pool[ child.val ].shmem->sibling_id;
1291 9 : purge_inner( accdb, child, fork_head, fork_tail );
1292 9 : child = next;
1293 9 : }
1294 :
1295 33 : uint txn = fork->shmem->txn_head;
1296 33 : if( txn!=UINT_MAX ) {
1297 30 : fd_accdb_txn_t * txn_head = txn_pool_ele( accdb->txn_pool, (ulong)txn );
1298 30 : fd_accdb_txn_t * txn_tail = NULL;
1299 81 : while( txn!=UINT_MAX ) {
1300 51 : fd_accdb_txn_t * txne = txn_pool_ele( accdb->txn_pool, (ulong)txn );
1301 :
1302 51 : uint acc_idx = txne->acc_pool_idx;
1303 51 : uint acc_map_idx = (uint)(fd_hash32( accdb->acc_pool[ acc_idx ].key.pubkey, accdb->shmem->seed ) &
1304 51 : (accdb->shmem->chain_cnt-1UL));
1305 :
1306 51 : uint prev = UINT_MAX;
1307 51 : uint cur = FD_VOLATILE_CONST( accdb->acc_map[ acc_map_idx ] );
1308 54 : while( cur!=acc_idx ) {
1309 3 : prev = cur;
1310 3 : cur = FD_VOLATILE_CONST( accdb->acc_pool[ cur ].map.next );
1311 3 : }
1312 :
1313 51 : fd_racesan_hook( "accdb_purge:pre_unlink" );
1314 51 : acc_unlink( accdb, acc_map_idx, prev, acc_idx );
1315 51 : deferred_acc_append( accdb, acc_idx );
1316 :
1317 51 : txn_tail = txne;
1318 51 : txn = txne->fork.next;
1319 51 : }
1320 30 : txn_pool_release_chain( accdb->txn_pool, txn_head, txn_tail );
1321 30 : }
1322 :
1323 33 : fork_slot_defer( accdb, fork_id, fork_head, fork_tail );
1324 33 : }
1325 :
1326 : static inline void
1327 : remove_children( fd_accdb_t * accdb,
1328 : fd_accdb_fork_t * fork,
1329 : fd_accdb_fork_t * except,
1330 : fd_accdb_fork_shmem_t ** fork_head,
1331 501 : fd_accdb_fork_shmem_t ** fork_tail ) {
1332 501 : fd_accdb_fork_id_t sibling_idx = fork->shmem->child_id;
1333 1005 : while( sibling_idx.val!=USHORT_MAX ) {
1334 504 : fd_accdb_fork_t * sibling = &accdb->fork_pool[ sibling_idx.val ];
1335 504 : fd_accdb_fork_id_t cur_idx = sibling_idx;
1336 :
1337 504 : sibling_idx = sibling->shmem->sibling_id;
1338 504 : if( FD_UNLIKELY( sibling==except ) ) continue;
1339 :
1340 3 : purge_inner( accdb, cur_idx, fork_head, fork_tail );
1341 3 : }
1342 501 : }
1343 :
1344 : static void
1345 : background_advance_root( fd_accdb_t * accdb,
1346 501 : fd_accdb_fork_id_t fork_id ) {
1347 501 : drain_deferred_frees( accdb );
1348 :
1349 : /* The caller guarantees that rooting is sequential: each call
1350 : advances the root by exactly one slot (the immediate child of the
1351 : current root). Skipping levels is not supported. */
1352 501 : fd_accdb_fork_t * fork = &accdb->fork_pool[ fork_id.val ];
1353 501 : FD_TEST( fork->shmem->parent_id.val==accdb->shmem->root_fork_id.val );
1354 501 : FD_TEST( fork->shmem->parent_id.val!=USHORT_MAX );
1355 :
1356 501 : fd_accdb_fork_t * parent_fork = &accdb->fork_pool[ fork->shmem->parent_id.val ];
1357 :
1358 : /* Accumulate freed fork pool slots across remove_children and the
1359 : old-version cleanup below into a chain that will be deferred-
1360 : released after the epoch bump. Freed acc pool slots are recorded
1361 : in the shmem side buffer via deferred_acc_append (they cannot be
1362 : chained via pool.next yet — see comment on the side buffer). */
1363 501 : fd_accdb_fork_shmem_t * fork_head = NULL;
1364 501 : fd_accdb_fork_shmem_t * fork_tail = NULL;
1365 :
1366 : /* When a fork is rooted, any competing forks can be immediately
1367 : removed as they will not be needed again. This includes child
1368 : forks of the pruned siblings as well. */
1369 501 : remove_children( accdb, parent_fork, fork, &fork_head, &fork_tail );
1370 :
1371 : /* And for any accounts which were updated in the newly rooted slot,
1372 : we will now never need to access any older version, so we can
1373 : discard any slots earlier than the one we are rooting. */
1374 501 : uint txn = fork->shmem->txn_head;
1375 501 : if( txn!=UINT_MAX ) {
1376 498 : fd_accdb_txn_t * txn_head = txn_pool_ele( accdb->txn_pool, (ulong)txn );
1377 498 : fd_accdb_txn_t * txn_tail = NULL;
1378 1956 : while( txn!=UINT_MAX ) {
1379 1458 : fd_accdb_txn_t * txne = txn_pool_ele( accdb->txn_pool, (ulong)txn );
1380 :
1381 1458 : fd_accdb_accmeta_t const * new_acc = &accdb->acc_pool[ txne->acc_pool_idx ];
1382 1458 : uint acc_map_idx = (uint)(fd_hash32( new_acc->key.pubkey, accdb->shmem->seed ) &
1383 1458 : (accdb->shmem->chain_cnt-1UL));
1384 :
1385 1458 : delta_insert( accdb, new_acc->key.pubkey );
1386 :
1387 1458 : uint prev = UINT_MAX;
1388 1458 : uint new_acc_prev = UINT_MAX; /* prev of new_acc on the chain when we encounter it (UINT_MAX if head or never seen) */
1389 1458 : int new_acc_seen = 0;
1390 1458 : uint acc = FD_VOLATILE_CONST( accdb->acc_map[ acc_map_idx ] );
1391 1458 : FD_TEST( acc!=UINT_MAX );
1392 4644 : while( acc!=UINT_MAX ) {
1393 3186 : fd_accdb_accmeta_t const * cur_acc = &accdb->acc_pool[ acc ];
1394 3186 : uint cur_next = FD_VOLATILE_CONST( cur_acc->map.next );
1395 :
1396 3186 : if( FD_LIKELY( acc==txne->acc_pool_idx ) ) {
1397 1458 : new_acc_prev = prev;
1398 1458 : new_acc_seen = 1;
1399 1458 : prev = acc;
1400 1458 : acc = cur_next;
1401 1458 : continue;
1402 1458 : }
1403 :
1404 1728 : if( FD_LIKELY( (cur_acc->key.generation<=parent_fork->shmem->generation || descends_set_test( fork->descends, fd_accdb_acc_fork_id(cur_acc) ) ) && !memcmp( new_acc->key.pubkey, cur_acc->key.pubkey, 32UL ) ) ) {
1405 1404 : uint next = cur_next;
1406 1404 : fd_racesan_hook( "accdb_advance:pre_unlink" );
1407 1404 : acc_unlink( accdb, acc_map_idx, prev, acc );
1408 1404 : deferred_acc_append( accdb, acc );
1409 1404 : acc = next;
1410 1404 : } else {
1411 324 : prev = acc;
1412 324 : acc = cur_next;
1413 324 : }
1414 1728 : }
1415 :
1416 : /* If the newly rooted version is a tombstone (lamports==0, e.g.
1417 : account was closed), drop it from the index too: no fork can
1418 : reach it anymore, and keeping it around just wastes a hash
1419 : slot and the disk bytes it occupies.
1420 :
1421 : If a later txn on this same fork wrote the same pubkey, that
1422 : txn's inner walk above would have already unlinked this txn's
1423 : new_acc as an "older version" - in that case new_acc_seen=0
1424 : and we skip, since the freelist cleanup is already done. */
1425 1458 : if( FD_UNLIKELY( new_acc_seen && new_acc->lamports==0UL ) ) {
1426 0 : uint new_acc_idx = (uint)txne->acc_pool_idx;
1427 0 : acc_unlink( accdb, acc_map_idx, new_acc_prev, new_acc_idx );
1428 0 : deferred_acc_append( accdb, new_acc_idx );
1429 0 : }
1430 :
1431 1458 : txn_tail = txne;
1432 1458 : txn = txne->fork.next;
1433 1458 : }
1434 498 : txn_pool_release_chain( accdb->txn_pool, txn_head, txn_tail );
1435 498 : }
1436 :
1437 501 : uint parent_txn = parent_fork->shmem->txn_head;
1438 501 : if( parent_txn!=UINT_MAX ) {
1439 69 : fd_accdb_txn_t * parent_head = txn_pool_ele( accdb->txn_pool, (ulong)parent_txn );
1440 69 : fd_accdb_txn_t * parent_tail = NULL;
1441 1437 : while( parent_txn!=UINT_MAX ) {
1442 1368 : fd_accdb_txn_t * t = txn_pool_ele( accdb->txn_pool, (ulong)parent_txn );
1443 1368 : parent_tail = t;
1444 1368 : parent_txn = t->fork.next;
1445 1368 : }
1446 69 : txn_pool_release_chain( accdb->txn_pool, parent_head, parent_tail );
1447 69 : }
1448 :
1449 : /* Remove the parent from all descends_sets and chain it for deferred
1450 : release, so that when the slot is eventually recycled to a new
1451 : fork, no concurrent reader can mistake the new fork for the old
1452 : ancestor. Entries from the freed parent are still visible via the
1453 : generation <= root_generation fast path in reads. */
1454 501 : fd_accdb_fork_id_t old_parent_id = fork->shmem->parent_id;
1455 501 : fork_slot_defer( accdb, old_parent_id, &fork_head, &fork_tail );
1456 :
1457 501 : fork->shmem->parent_id = (fd_accdb_fork_id_t){ .val = USHORT_MAX };
1458 501 : fork->shmem->sibling_id = (fd_accdb_fork_id_t){ .val = USHORT_MAX };
1459 501 : fork->shmem->txn_head = UINT_MAX;
1460 501 : descends_set_null( fork->descends );
1461 :
1462 : /* Publish the new root_fork_id BEFORE bumping the epoch and deferring
1463 : the parent slot. On x86-64 (TSO) a concurrent reader that still
1464 : loads the old root_fork_id is guaranteed to see the parent shmem in
1465 : its original (not-yet-recycled) state because the slot has not been
1466 : released yet. A reader that loads the new root_fork_id uses the
1467 : new fork. */
1468 501 : fd_racesan_hook( "accdb_advance:pre_publish_root" );
1469 501 : accdb->shmem->root_fork_id = fork_id;
1470 501 : FD_COMPILER_MFENCE();
1471 501 : fd_racesan_hook( "accdb_advance:post_publish_root" );
1472 :
1473 : /* Bump epoch and defer both the acc batch and parent fork slot. They
1474 : will be released at the next drain_deferred_frees call once all
1475 : concurrent readers have exited. The acc batch lives in the shmem
1476 : side buffer; only its epoch tag needs setting here. */
1477 501 : ulong tag = FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->epoch, 1UL );
1478 501 : if( FD_LIKELY( accdb->shmem->deferred_acc_buf_cnt ) ) {
1479 495 : accdb->shmem->deferred_acc_epoch = tag;
1480 495 : }
1481 501 : if( FD_LIKELY( fork_head ) ) {
1482 501 : accdb->deferred_fork_head = fork_head;
1483 501 : accdb->deferred_fork_tail = fork_tail;
1484 501 : accdb->deferred_fork_epoch = tag;
1485 501 : }
1486 501 : }
1487 :
1488 : void
1489 : fd_accdb_advance_root( fd_accdb_t * accdb,
1490 501 : fd_accdb_fork_id_t fork_id ) {
1491 501 : FD_CHECK_CRIT( fd_accdb_snapshot_sync_state( &accdb->shmem->snapshot_sync )!=FD_ACCDB_SNAPSHOT_SYNC_RUNNING,
1492 501 : "fd_accdb_advance_root called during snapshot production" );
1493 501 : wait_cmd( accdb );
1494 501 : submit_cmd( accdb, FD_ACCDB_CMD_ADVANCE_ROOT, fork_id.val );
1495 501 : }
1496 :
1497 : /* background_purge does the heavy lifting of purge on T2: unlink the
1498 : fork from the parent's child list, drain deferred frees, recursively
1499 : purge the fork subtree, and defer-release the freed acc pool
1500 : elements. The sibling-list unlink is done here (not on T1) because
1501 : advance_root / remove_children also mutate sibling lists on T2, and
1502 : T2 is single-threaded so plain stores are safe. */
1503 :
1504 : static void
1505 : background_purge( fd_accdb_t * accdb,
1506 21 : fd_accdb_fork_id_t fork_id ) {
1507 : /* Unlink fork_id from its parent's child list. This runs on T2
1508 : which is the sole mutator of sibling lists (advance_root and
1509 : remove_children also run on T2), so plain stores are safe. */
1510 21 : fd_accdb_fork_t * fork = &accdb->fork_pool[ fork_id.val ];
1511 21 : fd_accdb_fork_id_t parent_id = fork->shmem->parent_id;
1512 21 : if( FD_LIKELY( parent_id.val!=USHORT_MAX ) ) {
1513 21 : fd_accdb_fork_t * parent = &accdb->fork_pool[ parent_id.val ];
1514 21 : if( FD_UNLIKELY( parent->shmem->child_id.val==fork_id.val ) ) {
1515 21 : parent->shmem->child_id = fork->shmem->sibling_id;
1516 21 : } else {
1517 0 : fd_accdb_fork_id_t prev_id = parent->shmem->child_id;
1518 0 : while( prev_id.val!=USHORT_MAX ) {
1519 0 : fd_accdb_fork_t * prev = &accdb->fork_pool[ prev_id.val ];
1520 0 : if( prev->shmem->sibling_id.val==fork_id.val ) {
1521 0 : prev->shmem->sibling_id = fork->shmem->sibling_id;
1522 0 : break;
1523 0 : }
1524 0 : prev_id = prev->shmem->sibling_id;
1525 0 : }
1526 0 : }
1527 21 : }
1528 :
1529 21 : drain_deferred_frees( accdb );
1530 :
1531 21 : fd_accdb_fork_shmem_t * fork_head = NULL;
1532 21 : fd_accdb_fork_shmem_t * fork_tail = NULL;
1533 21 : purge_inner( accdb, fork_id, &fork_head, &fork_tail );
1534 :
1535 21 : ulong tag = FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->epoch, 1UL );
1536 21 : if( FD_LIKELY( accdb->shmem->deferred_acc_buf_cnt ) ) {
1537 18 : accdb->shmem->deferred_acc_epoch = tag;
1538 18 : }
1539 21 : if( FD_LIKELY( fork_head ) ) {
1540 21 : accdb->deferred_fork_head = fork_head;
1541 21 : accdb->deferred_fork_tail = fork_tail;
1542 21 : accdb->deferred_fork_epoch = tag;
1543 21 : }
1544 21 : }
1545 :
1546 : void
1547 : fd_accdb_purge( fd_accdb_t * accdb,
1548 21 : fd_accdb_fork_id_t fork_id ) {
1549 21 : FD_TEST( fork_id.val!=accdb->shmem->root_fork_id.val );
1550 :
1551 21 : wait_cmd( accdb );
1552 21 : submit_cmd( accdb, FD_ACCDB_CMD_PURGE, fork_id.val );
1553 21 : }
1554 :
1555 : static inline fd_accdb_cache_line_t *
1556 : acquire_cache_line( fd_accdb_t * accdb,
1557 : ulong size_class,
1558 932526 : uint * out_evicted_acc_idx ) {
1559 : /* Priority 1: CAS free list — already invalidated,
1560 : persisted==1, generation==UINT_MAX. Cheapest path. */
1561 932526 : fd_accdb_cache_line_t * result = cache_free_pop( accdb, size_class );
1562 932526 : if( FD_LIKELY( result ) ) {
1563 788325 : while( FD_UNLIKELY( FD_ATOMIC_CAS( &result->refcnt, 0U, 1U )!=0U ) ) {
1564 0 : fd_racesan_hook( "accdb_freepop:refcnt_wait" );
1565 0 : FD_SPIN_PAUSE();
1566 0 : }
1567 788325 : result->referenced = 0;
1568 788325 : *out_evicted_acc_idx = UINT_MAX;
1569 788325 : return result;
1570 788325 : }
1571 :
1572 : /* Priority 2: Lazy initial allocation — atomic FAA with undo on
1573 : overflow. Safe for concurrent callers. */
1574 144201 : ulong old_init = FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->cache_class_init[ size_class ].val, 1UL );
1575 144201 : if( FD_LIKELY( old_init<accdb->shmem->cache_class_max[ size_class ] ) ) {
1576 144189 : result = cache_line( accdb, size_class, old_init );
1577 144189 : result->refcnt = 1;
1578 144189 : result->persisted = 1;
1579 144189 : result->referenced = 0;
1580 144189 : result->acc_idx = UINT_MAX;
1581 144189 : result->key.generation = UINT_MAX;
1582 144189 : *out_evicted_acc_idx = UINT_MAX;
1583 144189 : return result;
1584 144189 : }
1585 12 : FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->cache_class_init[ size_class ].val, 1UL );
1586 :
1587 : /* Priority 3: CLOCK sweep ... scan forward giving second chances. */
1588 30 : for(;;) {
1589 30 : ulong hand = FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->clock_hand[ size_class ].val, 1UL ) % accdb->shmem->cache_class_max[ size_class ];
1590 30 : fd_accdb_cache_line_t * line = cache_line( accdb, size_class, hand );
1591 :
1592 30 : if( FD_UNLIKELY( line->key.generation==UINT_MAX && line->acc_idx==UINT_MAX ) ) continue;
1593 :
1594 30 : fd_racesan_hook( "accdb_clock:pre_refcnt" );
1595 30 : uint rc = FD_VOLATILE_CONST( line->refcnt );
1596 30 : if( FD_UNLIKELY( rc!=0U ) ) continue; /* Pinned or being evicted */
1597 :
1598 30 : if( FD_UNLIKELY( line->referenced ) ) {
1599 18 : line->referenced = 0;
1600 18 : continue; /* Second chance */
1601 18 : }
1602 :
1603 12 : if( FD_UNLIKELY( FD_ATOMIC_CAS( &line->refcnt, 0U, FD_ACCDB_EVICT_SENTINEL )!=0U ) ) continue;
1604 :
1605 12 : if( FD_UNLIKELY( line->acc_idx==UINT_MAX && line->key.generation==UINT_MAX ) ) {
1606 0 : FD_VOLATILE( line->refcnt ) = 0;
1607 0 : continue;
1608 0 : }
1609 :
1610 : /* The line is now claimed for eviction (refcnt==EVICT_SENTINEL). A
1611 : concurrent acc_unlink that targets this same line's accmeta will
1612 : observe the sentinel here and take its do-nothing branch — see the
1613 : test_accdb_racesan SENTINEL case. */
1614 12 : fd_racesan_hook( "clock_evict:post_sentinel" );
1615 :
1616 12 : if( FD_LIKELY( line->acc_idx!=UINT_MAX ) ) {
1617 12 : evict_clear_acc_cache_ref( &accdb->acc_pool[ line->acc_idx ], size_class, hand );
1618 12 : }
1619 12 : *out_evicted_acc_idx = line->persisted ? UINT_MAX : line->acc_idx;
1620 12 : line->key.generation = UINT_MAX;
1621 12 : line->refcnt = 1;
1622 12 : line->referenced = 0;
1623 12 : return line;
1624 12 : }
1625 :
1626 0 : FD_TEST( 0 );
1627 0 : return NULL;
1628 0 : }
1629 :
1630 : static inline void
1631 : change_partition( fd_accdb_t * accdb,
1632 : accdb_offset_t const * offset_before,
1633 : accdb_offset_t * out_offset,
1634 : int * has_partition,
1635 63 : uchar layer ) {
1636 : /* New data will not fit in the current partition, so we need to
1637 : move to the next one. */
1638 63 : ulong partition_idx_before = packed_partition_idx( offset_before );
1639 63 : ulong partition_offset_before = packed_partition_offset( offset_before );
1640 63 : if( FD_LIKELY( *has_partition ) ) {
1641 39 : fd_accdb_partition_t * before = partition_pool_ele( accdb->partition_pool, partition_idx_before );
1642 39 : before->write_offset = partition_offset_before;
1643 39 : }
1644 :
1645 : /* Single rdtsc per partition lifecycle event: stamp the closing
1646 : partition's filled time and the new partition's created time off
1647 : the same sample. */
1648 63 : long now_ticks = (long)fd_tickcount();
1649 :
1650 63 : ulong free_size = accdb->shmem->partition_sz - partition_offset_before;
1651 63 : if( FD_LIKELY( *has_partition ) ) {
1652 39 : fd_accdb_partition_t * old = partition_pool_ele( accdb->partition_pool, partition_idx_before );
1653 39 : FD_ATOMIC_FETCH_AND_ADD( &old->bytes_freed, free_size );
1654 39 : FD_VOLATILE( old->filled_ticks ) = now_ticks;
1655 : /* The tail slack is now committed dead — count it as current
1656 : (written-through) so fragmentation reflects it. */
1657 39 : FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->shmetrics->disk_current_bytes, free_size );
1658 39 : }
1659 :
1660 63 : if( FD_UNLIKELY( !partition_pool_free( accdb->partition_pool ) ) ) FD_LOG_ERR(( "accounts database file is at capacity" ));
1661 63 : fd_accdb_partition_t * partition = partition_pool_ele_acquire( accdb->partition_pool );
1662 63 : partition->bytes_freed = 0UL;
1663 63 : partition->marked_compaction = 0;
1664 63 : partition->layer = layer;
1665 63 : partition->read_ops = 0UL;
1666 63 : partition->bytes_read = 0UL;
1667 63 : partition->write_ops = 0UL;
1668 63 : partition->bytes_written = 0UL;
1669 63 : partition->write_offset = 0UL;
1670 63 : partition->compaction_offset = 0UL;
1671 63 : partition->created_ticks = now_ticks;
1672 63 : partition->filled_ticks = 0L;
1673 63 : partition->queued = 0;
1674 63 : partition->compacting_now = 0;
1675 :
1676 63 : ulong new_partition_idx = partition_pool_idx( accdb->partition_pool, partition );
1677 63 : int had_partition = *has_partition;
1678 63 : *out_offset = accdb_offset( new_partition_idx, 0UL );
1679 63 : FD_COMPILER_MFENCE();
1680 63 : *has_partition = 1;
1681 :
1682 : /* Now that the write head has been rotated away from the old
1683 : partition, check if it should be enqueued for compaction. We call
1684 : try_enqueue directly because the caller already holds
1685 : partition_lock (calling fd_accdb_shmem_bytes_freed here would
1686 : deadlock on the non-reentrant lock). Skip when
1687 : has_partition was 0, because the sentinel partition_idx is
1688 : not a valid pool element. */
1689 63 : if( FD_LIKELY( had_partition && partition_idx_before!=new_partition_idx ) ) {
1690 39 : fd_accdb_shmem_try_enqueue_compaction( accdb->shmem, partition_idx_before );
1691 39 : }
1692 :
1693 : /* Snapshot-load tiering: accounts loaded from a snapshot never get
1694 : a second write, so compaction-driven promotion never fires and
1695 : they would otherwise live in Hot forever. When snapshot_loading
1696 : is set, tag the new partition as Cold up front. We do not set
1697 : has_partition[Cold] / whead[Cold] — those are owned by the
1698 : compaction tile and represent the live Cold write head, which is
1699 : independent of snapshot-loaded partitions that happen to be
1700 : labeled Cold. */
1701 63 : if( FD_UNLIKELY( accdb->snapshot_loading && layer==0 ) ) {
1702 51 : FD_VOLATILE( partition->layer ) = FD_ACCDB_COMPACTION_LAYER_CNT-1UL;
1703 51 : }
1704 :
1705 63 : if( FD_UNLIKELY( new_partition_idx>=accdb->shmem->partition_max ) ) {
1706 63 : FD_LOG_INFO(( "growing accounts database from %lu GiB to %lu GiB", accdb->shmem->partition_max*accdb->shmem->partition_sz/(1UL<<30UL), (new_partition_idx+1UL)*accdb->shmem->partition_sz/(1UL<<30UL) ));
1707 :
1708 63 : int result = fallocate( accdb->fd, 0, (long)(new_partition_idx*accdb->shmem->partition_sz), (long)accdb->shmem->partition_sz );
1709 63 : if( FD_UNLIKELY( -1==result ) ) {
1710 0 : if( FD_LIKELY( errno==ENOSPC ) ) FD_LOG_ERR(( "fallocate() failed (%d-%s). The accounts database filled "
1711 0 : "the disk it is on, trying to grow from %lu GiB to %lu GiB. Please "
1712 0 : "free up disk space and restart the validator.",
1713 0 : errno, fd_io_strerror( errno ), accdb->shmem->partition_max*accdb->shmem->partition_sz/(1UL<<30UL), (new_partition_idx+1UL)*accdb->shmem->partition_sz/(1UL<<30UL) ));
1714 0 : else FD_LOG_ERR(( "fallocate() failed (%d-%s)", errno, fd_io_strerror( errno ) ));
1715 0 : }
1716 :
1717 : /* CAS loop: the compaction tile may also be growing the file
1718 : concurrently, so neither path may clobber the other. */
1719 63 : for(;;) {
1720 63 : ulong cur = accdb->shmem->partition_max;
1721 63 : if( FD_LIKELY( new_partition_idx+1UL<=cur ) ) break;
1722 63 : if( FD_LIKELY( FD_ATOMIC_CAS( &accdb->shmem->partition_max, cur, new_partition_idx+1UL )==cur ) ) {
1723 63 : fd_event_accdb_partition_added_t ev = {
1724 63 : .partition_idx = new_partition_idx,
1725 63 : .prior_partition_idx = had_partition ? partition_idx_before : ULONG_MAX,
1726 63 : .layer = layer,
1727 63 : .old_partition_max = cur,
1728 63 : .new_partition_max = new_partition_idx+1UL,
1729 63 : .partition_sz = accdb->shmem->partition_sz,
1730 63 : .disk_allocated_bytes = (new_partition_idx+1UL)*accdb->shmem->partition_sz,
1731 63 : };
1732 63 : fd_event_report_accdb_partition_added( &ev );
1733 63 : break;
1734 63 : }
1735 63 : }
1736 63 : accdb->shmem->shmetrics->disk_allocated_bytes = accdb->shmem->partition_max*accdb->shmem->partition_sz;
1737 63 : }
1738 63 : }
1739 :
1740 : /* Reserve sz bytes in the layer-0 write head and set
1741 : out_partition_idx to where they landed. Bumps no counters. */
1742 :
1743 : static inline ulong
1744 : reserve_next_write( fd_accdb_t * accdb,
1745 : ulong sz,
1746 99 : ulong * out_partition_idx ) {
1747 138 : for(;;) {
1748 138 : accdb_offset_t offset = { .val = FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->whead[ 0 ].val, sz ) };
1749 138 : if( FD_LIKELY( packed_partition_offset( &offset )+sz<=accdb->shmem->partition_sz ) ) {
1750 99 : *out_partition_idx = packed_partition_idx( &offset );
1751 99 : return packed_partition_file_offset( &offset, accdb->shmem->partition_sz );
1752 99 : }
1753 :
1754 39 : if( FD_UNLIKELY( packed_partition_offset( &offset )>accdb->shmem->partition_sz ) ) {
1755 : /* This can happen if another thread also raced to allocate the
1756 : next write and won. Wait for the partition switch to finish
1757 : before retrying, so we do not keep doing fetch-and-adds that
1758 : advance the offset further past the boundary.
1759 :
1760 : A switch is detected by the head moving to a different
1761 : partition index OR its offset dropping back to a valid position
1762 : (a switch resets the offset to 0). We must not key the wait
1763 : solely on the index changing: the initial write head is a
1764 : sentinel whose packed index can coincide with a real pool
1765 : index. */
1766 0 : ulong stale_partition = packed_partition_idx( &offset );
1767 0 : for(;;) {
1768 0 : accdb_offset_t cur = { .val = FD_VOLATILE_CONST( accdb->shmem->whead[ 0 ].val ) };
1769 0 : if( packed_partition_idx( &cur )!=stale_partition ) break;
1770 0 : if( packed_partition_offset( &cur )<=accdb->shmem->partition_sz ) break;
1771 0 : FD_SPIN_PAUSE();
1772 0 : }
1773 0 : continue;
1774 0 : }
1775 :
1776 39 : spin_lock_acquire( &accdb->shmem->partition_lock );
1777 39 : change_partition( accdb, &offset, &accdb->shmem->whead[ 0 ], &accdb->shmem->has_partition[ 0 ], 0 );
1778 39 : spin_lock_release( &accdb->shmem->partition_lock );
1779 39 : }
1780 99 : }
1781 :
1782 : /* Reserve sz bytes. Layer-0 write metrics are deferred until
1783 : explicitly flushed. */
1784 :
1785 : static inline ulong
1786 : allocate_next_write( fd_accdb_t * accdb,
1787 99 : ulong sz ) {
1788 99 : ulong partition_idx;
1789 99 : ulong file_offset = reserve_next_write( accdb, sz, &partition_idx );
1790 :
1791 : /* Very rarely the reservation crosses into a new partition. Since
1792 : stats are aggregated per-partition, any accumulated stats are
1793 : flushed and reset to accommodate the new partition. */
1794 99 : if( FD_LIKELY( accdb->write_stats.num_ops ) &&
1795 99 : FD_UNLIKELY( accdb->write_stats.partition_idx!=partition_idx ) ) {
1796 15 : fd_accdb_flush_metrics( accdb );
1797 15 : }
1798 :
1799 99 : accdb->write_stats.partition_idx = partition_idx;
1800 99 : accdb->write_stats.bytes += sz;
1801 99 : accdb->write_stats.num_ops++;
1802 99 : return file_offset;
1803 99 : }
1804 :
1805 : /* Compaction write allocation. Single-threaded: only the compaction
1806 : tile calls these, so the compaction write heads do not need atomic
1807 : fetch-and-add. dest_layer is the target layer (1..N-1). */
1808 :
1809 : static inline ulong
1810 : allocate_next_compaction_write( fd_accdb_t * accdb,
1811 : ulong sz,
1812 3 : ulong dest_layer ) {
1813 3 : accdb_offset_t offset = accdb->shmem->whead[ dest_layer ];
1814 3 : if( FD_UNLIKELY( !accdb->shmem->has_partition[ dest_layer ] ||
1815 3 : packed_partition_offset( &offset )+sz>accdb->shmem->partition_sz ) ) {
1816 3 : spin_lock_acquire( &accdb->shmem->partition_lock );
1817 3 : change_partition( accdb, &offset, &accdb->shmem->whead[ dest_layer ], &accdb->shmem->has_partition[ dest_layer ], (uchar)dest_layer );
1818 3 : spin_lock_release( &accdb->shmem->partition_lock );
1819 3 : offset = accdb->shmem->whead[ dest_layer ];
1820 3 : }
1821 3 : accdb->shmem->whead[ dest_layer ].val += sz;
1822 3 : FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->shmetrics->disk_current_bytes, sz );
1823 3 : ulong file_offset = packed_partition_file_offset( &offset, accdb->shmem->partition_sz );
1824 3 : fd_accdb_partition_write_bump( accdb, packed_partition_idx( &offset ), sz, 1UL );
1825 3 : return file_offset;
1826 3 : }
1827 :
1828 : /* fd_accdb_compact relocates one record from the oldest partition
1829 : queued for compaction at src_layer into the write head for the
1830 : next colder tier, or the same tier for the deepest layer. It is
1831 : designed to be called repeatedly from a dedicated compaction tile.
1832 : If there is work to do, *charge_busy is set to 1; otherwise 0 is
1833 : left unchanged and the call returns immediately.
1834 :
1835 : src_layer must be in 0..FD_ACCDB_COMPACTION_LAYER_CNT-1. */
1836 :
1837 : static void
1838 : background_compact( fd_accdb_t * accdb,
1839 : ulong src_layer,
1840 9805245 : int * charge_busy ) {
1841 9805245 : FD_COMPILER_MFENCE();
1842 9805245 : FD_VOLATILE( *accdb->my_epoch_slot ) = FD_VOLATILE_CONST( accdb->shmem->epoch );
1843 9805245 : FD_HW_MFENCE(); /* StoreLoad: epoch store must be globally visible
1844 : before any subsequent loads so the deferred
1845 : reclamation scan does not miss us. */
1846 :
1847 : /* Reclaim any deferred-free partitions whose epoch has been observed
1848 : by all joiners (i.e. no epoch-publishing joiner could still be
1849 : referencing data in them). Scan writer slots [0, joiner_cnt)
1850 : plus each external (read-only) joiner's private epoch fseq. */
1851 9805245 : ulong min_epoch = ULONG_MAX;
1852 9805245 : ulong joiner_cnt = FD_VOLATILE_CONST( accdb->shmem->joiner_cnt );
1853 24134193 : for( ulong t=0UL; t<joiner_cnt; t++ ) {
1854 14328948 : ulong e = FD_VOLATILE_CONST( accdb->shmem->joiner_epochs[ t ].val );
1855 14328948 : if( FD_LIKELY( e<min_epoch ) ) min_epoch = e;
1856 14328948 : }
1857 9805245 : for( ulong t=0UL; t<accdb->external_epoch_cnt; t++ ) {
1858 0 : ulong e = FD_VOLATILE_CONST( *accdb->external_epoch_slots[ t ] );
1859 0 : if( FD_LIKELY( e<min_epoch ) ) min_epoch = e;
1860 0 : }
1861 9805248 : for(;;) {
1862 9805248 : if( FD_LIKELY( deferred_free_dlist_is_empty( accdb->deferred_free_dlist, accdb->partition_pool ) ) ) break;
1863 3 : fd_accdb_partition_t * p = deferred_free_dlist_ele_peek_head( accdb->deferred_free_dlist, accdb->partition_pool );
1864 3 : if( FD_LIKELY( p->epoch_tag>=min_epoch ) ) break;
1865 :
1866 3 : fd_racesan_hook( "accdb_reclaim:pre_free_partition" );
1867 :
1868 3 : spin_lock_acquire( &accdb->shmem->partition_lock );
1869 3 : deferred_free_dlist_ele_pop_head( accdb->deferred_free_dlist, accdb->partition_pool );
1870 3 : FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->shmetrics->disk_current_bytes, accdb->shmem->partition_sz );
1871 3 : FD_VOLATILE( p->write_offset ) = 0UL;
1872 3 : partition_pool_ele_release( accdb->partition_pool, p );
1873 3 : spin_lock_release( &accdb->shmem->partition_lock );
1874 3 : }
1875 :
1876 9805245 : if( FD_LIKELY( compaction_dlist_is_empty( accdb->compaction_dlist[ src_layer ], accdb->partition_pool ) ) ) {
1877 9805236 : FD_COMPILER_MFENCE();
1878 9805236 : FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
1879 9805236 : return;
1880 9805236 : }
1881 9 : fd_accdb_partition_t * compact = compaction_dlist_ele_peek_head( accdb->compaction_dlist[ src_layer ], accdb->partition_pool );
1882 9 : if( FD_UNLIKELY( !compact ) ) {
1883 0 : FD_COMPILER_MFENCE();
1884 0 : FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
1885 0 : return;
1886 0 : }
1887 :
1888 : /* Wait until all epoch-publishing joiners that were active when this
1889 : partition was enqueued for compaction have exited, ensuring any
1890 : in-flight pwritev2 to this partition has completed before we start
1891 : reading from it. */
1892 9 : if( FD_UNLIKELY( compact->compaction_ready_epoch>=min_epoch ) ) {
1893 0 : FD_COMPILER_MFENCE();
1894 0 : FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
1895 0 : return;
1896 0 : }
1897 :
1898 9 : *charge_busy = 1;
1899 :
1900 9 : if( FD_UNLIKELY( !compact->compacting_now ) ) {
1901 3 : compact->compaction_start_wallclock = fd_log_wallclock();
1902 3 : compact->compaction_accounts_relocated = 0UL;
1903 3 : compact->compaction_bytes_relocated = 0UL;
1904 3 : compact->compaction_dead_records = 0UL;
1905 3 : }
1906 9 : FD_VOLATILE( compact->queued ) = 0;
1907 9 : FD_VOLATILE( compact->compacting_now ) = 1;
1908 :
1909 9 : fd_accdb_disk_meta_t meta[1];
1910 :
1911 9 : ulong compact_base = partition_pool_idx( accdb->partition_pool, compact )*accdb->shmem->partition_sz;
1912 :
1913 : /* Read the on-disk metadata header at the current compaction
1914 : cursor within the partition being compacted. */
1915 9 : ulong bytes_read = 0UL;
1916 18 : while( FD_UNLIKELY( bytes_read<sizeof(fd_accdb_disk_meta_t) ) ) {
1917 9 : long result = pread( accdb->fd, ((uchar *)meta)+bytes_read, sizeof(fd_accdb_disk_meta_t)-bytes_read, (long)(compact_base+compact->compaction_offset+bytes_read) );
1918 9 : if( FD_UNLIKELY( -1==result && (errno==EINTR || errno==EAGAIN || errno==EWOULDBLOCK ) ) ) continue;
1919 9 : else if( FD_UNLIKELY( -1==result ) ) FD_LOG_ERR(( "pread() failed (%d-%s)", errno, fd_io_strerror( errno ) ));
1920 9 : else if( FD_UNLIKELY( !result ) ) FD_LOG_ERR(( "accounts database is corrupt, data expected at offset %lu with size %lu exceeded file extents",
1921 9 : compact_base+compact->compaction_offset+bytes_read, sizeof(fd_accdb_disk_meta_t) ));
1922 9 : fd_accdb_partition_read_bump( accdb, compact_base+compact->compaction_offset, (ulong)result );
1923 9 : bytes_read += (ulong)result;
1924 9 : }
1925 :
1926 : /* Walk the hash chain to find a live index entry whose on-disk
1927 : offset matches the record we are compacting. */
1928 9 : fd_accdb_accmeta_t * accmeta = NULL;
1929 9 : ulong source_packed = 0UL;
1930 9 : uint acc_idx = FD_VOLATILE_CONST( accdb->acc_map[ fd_hash32( meta->pubkey, accdb->shmem->seed )&(accdb->shmem->chain_cnt-1UL) ] );
1931 15 : while( acc_idx!=UINT_MAX ) {
1932 9 : fd_accdb_accmeta_t * candidate = &accdb->acc_pool[ acc_idx ];
1933 9 : uint next_idx = FD_VOLATILE_CONST( candidate->map.next );
1934 9 : ulong candidate_packed = FD_VOLATILE_CONST( candidate->offset_fork );
1935 9 : if( FD_LIKELY( (candidate_packed & FD_ACCDB_OFF_MASK)==compact_base+compact->compaction_offset ) ) {
1936 3 : accmeta = candidate;
1937 3 : source_packed = candidate_packed;
1938 3 : break;
1939 3 : }
1940 6 : acc_idx = next_idx;
1941 6 : }
1942 :
1943 9 : ulong record_sz = sizeof(fd_accdb_disk_meta_t) + (ulong)meta->size;
1944 9 : ulong bytes_copied = 0UL;
1945 9 : if( FD_UNLIKELY( !accmeta ) ) {
1946 : /* Dead record — the index entry was already removed, so this
1947 : on-disk extent is garbage. Nothing to relocate. */
1948 6 : compact->compaction_dead_records++;
1949 6 : } else {
1950 3 : ulong dest_layer = fd_ulong_min( src_layer+1UL, FD_ACCDB_COMPACTION_LAYER_CNT-1UL );
1951 3 : ulong dest_offset = allocate_next_compaction_write( accdb, record_sz, dest_layer );
1952 :
1953 6 : while( FD_UNLIKELY( bytes_copied<record_sz ) ) {
1954 3 : long in_off = (long)(compact_base + compact->compaction_offset + bytes_copied);
1955 3 : long out_off = (long)(dest_offset + bytes_copied);
1956 :
1957 3 : long result = copy_file_range( accdb->fd, &in_off, accdb->fd, &out_off, record_sz-bytes_copied, 0 );
1958 3 : if( FD_UNLIKELY( -1==result && (errno==EINTR || errno==EAGAIN || errno==EWOULDBLOCK ) ) ) continue;
1959 3 : else if( FD_UNLIKELY( -1==result ) ) FD_LOG_ERR(( "copy_file_range() failed (%d-%s)", errno, fd_io_strerror( errno ) ));
1960 3 : else if( FD_UNLIKELY( !result ) ) FD_LOG_ERR(( "accounts database is corrupt, data expected at offset %lu with size %lu exceeded file extents",
1961 3 : compact_base+compact->compaction_offset+bytes_copied, record_sz ));
1962 3 : fd_accdb_partition_read_bump( accdb, compact_base+compact->compaction_offset+bytes_copied, (ulong)result );
1963 3 : bytes_copied += (ulong)result;
1964 3 : accdb->metrics->copy_ops++;
1965 3 : }
1966 :
1967 3 : accdb->shmem->shmetrics->accounts_relocated++;
1968 3 : accdb->shmem->shmetrics->accounts_relocated_bytes += bytes_copied;
1969 3 : compact->compaction_accounts_relocated++;
1970 3 : compact->compaction_bytes_relocated += bytes_copied;
1971 :
1972 : /* Ensure the data is on disk before publishing the new offset,
1973 : so concurrent acquire threads do not preadv2 from a location
1974 : that hasn't been written yet. */
1975 3 : FD_COMPILER_MFENCE();
1976 :
1977 : /* CAS the offset from the exact source record we copied to the new
1978 : destination. If a concurrent release overwrote the offset to
1979 : FD_ACCDB_OFF_INVAL (dirty sentinel for a new commit), or later
1980 : published a newer on-disk location, the CAS fails and we treat
1981 : the relocated copy as stale. We CAS the full packed
1982 : offset_fork so the fork_id is preserved and so we only publish
1983 : the relocation if the copied source record is still current. */
1984 3 : ulong new_packed = ( source_packed & ~FD_ACCDB_OFF_MASK ) | ( dest_offset & FD_ACCDB_OFF_MASK );
1985 :
1986 : #if FD_HAS_RACESAN
1987 : fd_memcpy( fd_accdb_dbg_reloc_pubkey, accmeta->key.pubkey, 32UL );
1988 : fd_accdb_dbg_reloc_dest = dest_offset;
1989 : fd_accdb_dbg_reloc_cnt++;
1990 : #endif
1991 :
1992 3 : fd_racesan_hook( "accdb_compact:pre_offset_cas" );
1993 3 : if( FD_UNLIKELY( FD_ATOMIC_CAS( &accmeta->offset_fork, source_packed, new_packed )!=source_packed ) ) {
1994 : /* Record was superseded by a concurrent overwrite commit.
1995 : The disk space we just wrote is dead on arrival — account
1996 : it as freed so compaction can reclaim it later. */
1997 0 : fd_accdb_shmem_bytes_freed( accdb->shmem, dest_offset, record_sz );
1998 0 : bytes_copied = 0UL;
1999 0 : }
2000 3 : }
2001 :
2002 9 : fd_racesan_hook( "accdb_compact:post_relocate" );
2003 :
2004 9 : compact->compaction_offset += record_sz;
2005 :
2006 9 : if( FD_UNLIKELY( compact->compaction_offset>=compact->write_offset ) ) {
2007 3 : FD_LOG_INFO(( "compaction of partition %lu completed", partition_pool_idx( accdb->partition_pool, compact ) ));
2008 :
2009 3 : fd_event_accdb_compaction_completed_t ev = {
2010 3 : .partition_idx = partition_pool_idx( accdb->partition_pool, compact ),
2011 3 : .src_layer = (uchar)src_layer,
2012 3 : .dest_layer = (uchar)fd_ulong_min( src_layer+1UL, FD_ACCDB_COMPACTION_LAYER_CNT-1UL ),
2013 3 : .bytes_scanned = compact->write_offset,
2014 3 : .bytes_freed = compact->bytes_freed,
2015 3 : .accounts_relocated = compact->compaction_accounts_relocated,
2016 3 : .bytes_relocated = compact->compaction_bytes_relocated,
2017 3 : .dead_records = compact->compaction_dead_records,
2018 3 : .start_time = (ulong)compact->compaction_start_wallclock,
2019 3 : .end_time = (ulong)fd_log_wallclock(),
2020 3 : };
2021 3 : fd_event_report_accdb_compaction_completed( &ev );
2022 :
2023 : /* Ensure the new acc->offset_fork stores above are visible to other
2024 : cores before the source partition is moved to the deferred-free
2025 : list. On x86 (TSO) hardware store ordering already guarantees
2026 : this, but the compiler fence prevents the compiler from sinking
2027 : the offset store past the inlined pool/dlist mutations below. */
2028 3 : FD_COMPILER_MFENCE();
2029 :
2030 : /* Bump the global epoch and tag this partition so the reclamation
2031 : scan knows when all epoch-publishing joiners that could reference
2032 : data in this partition have exited. */
2033 3 : ulong tag = FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->epoch, 1UL );
2034 3 : compact->epoch_tag = tag;
2035 :
2036 : /* partition_lock serializes these dlist/pool mutations with
2037 : concurrent push_tail in fd_accdb_shmem_bytes_freed and
2038 : partition_pool_ele_acquire in change_partition. Neither fd_dlist
2039 : nor fd_pool are thread-safe, so all mutations must be under the
2040 : same lock. */
2041 3 : spin_lock_acquire( &accdb->shmem->partition_lock );
2042 :
2043 3 : accdb->shmem->shmetrics->partitions_freed++;
2044 3 : compaction_dlist_ele_pop_head( accdb->compaction_dlist[ src_layer ], accdb->partition_pool );
2045 3 : FD_VOLATILE( compact->compacting_now ) = 0;
2046 3 : FD_VOLATILE( compact->queued ) = 0;
2047 3 : deferred_free_dlist_ele_push_tail( accdb->deferred_free_dlist, compact, accdb->partition_pool );
2048 :
2049 3 : accdb->shmem->shmetrics->compactions_completed++;
2050 3 : if( FD_LIKELY( compaction_dlist_is_empty( accdb->compaction_dlist[ src_layer ], accdb->partition_pool ) ) ) {
2051 3 : accdb->shmem->shmetrics->in_compaction = 0;
2052 3 : } else {
2053 0 : fd_accdb_partition_t * next = compaction_dlist_ele_peek_head( accdb->compaction_dlist[ src_layer ], accdb->partition_pool );
2054 0 : FD_LOG_INFO(( "compaction of layer %lu partition %lu started", src_layer, partition_pool_idx( accdb->partition_pool, next ) ));
2055 0 : }
2056 :
2057 3 : spin_lock_release( &accdb->shmem->partition_lock );
2058 3 : }
2059 :
2060 9 : accdb->metrics->bytes_read += bytes_read + bytes_copied;
2061 9 : accdb->metrics->bytes_written += bytes_copied;
2062 :
2063 9 : FD_COMPILER_MFENCE();
2064 9 : FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
2065 9 : }
2066 :
2067 : /* cold_load_acc resolves the cache slot for `acc` when STEP 1's
2068 : cache_try_pin failed. It uses bit 29 of executable_size as a
2069 : single-claimer lock so that two concurrent acquirers cannot each
2070 : install their own cache slot for the same acc (which would orphan
2071 : one slot with a dangling line->acc_idx and eventually corrupt
2072 : acc->cache_valid via CLOCK).
2073 :
2074 : Protocol per acc:
2075 : - If cache_valid is set, retry cache_try_pin (another thread
2076 : finished the cold-load while we were here). On success, mark
2077 : exists_in_cache so STEP 4 will not write back the slot.
2078 : - If claim is set, spin (another thread is mid-cold-load).
2079 : - Otherwise CAS-set the claim bit. Winner allocates a cache
2080 : line, populates the placeholder (acc_idx=UINT_MAX), publishes
2081 : cache_idx, then atomically (CAS-loop) sets cache_valid and
2082 : clears claim.
2083 :
2084 : The eviction sites that clear cache_valid must use FETCH_AND with
2085 : ~CACHE_VALID_BIT (preserving the claim bit) to interact correctly
2086 : with this protocol. */
2087 :
2088 : static fd_accdb_cache_line_t *
2089 : cold_load_acc( fd_accdb_t * accdb,
2090 : fd_accdb_accmeta_t * accmeta,
2091 : uchar const * pubkey,
2092 : int * out_exists_in_cache,
2093 30 : uint * out_evicted_acc_idx ) {
2094 30 : for(;;) {
2095 30 : uint old_es = FD_VOLATILE_CONST( accmeta->executable_size );
2096 30 : int valid = FD_ACCDB_SIZE_CACHE_VALID( old_es );
2097 30 : int claimed = FD_ACCDB_SIZE_CACHE_CLAIM( old_es );
2098 :
2099 30 : if( FD_UNLIKELY( valid ) ) {
2100 : /* old_es snapshot saw VALID=1 but a concurrent
2101 : evict_clear_acc_cache_ref may have cleared VALID and stored
2102 : cache_idx=INVAL between our snapshot and this load. Decoding
2103 : INVAL would yield a wild cache_line pointer; retry the loop
2104 : instead (next iteration will see VALID=0). */
2105 0 : uint cidx = FD_VOLATILE_CONST( accmeta->cache_idx );
2106 0 : if( FD_UNLIKELY( cidx==FD_ACCDB_ACC_CIDX_INVAL ) ) { FD_SPIN_PAUSE(); continue; }
2107 0 : fd_accdb_cache_line_t * hit = cache_line( accdb, FD_ACCDB_ACC_CIDX_CLASS( cidx ), FD_ACCDB_ACC_CIDX_IDX( cidx ) );
2108 0 : fd_racesan_hook( "accdb_cold_load:pre_try_pin" );
2109 0 : fd_accdb_cache_line_t * pinned = cache_try_pin( hit, pubkey, accmeta->key.generation );
2110 0 : if( FD_LIKELY( pinned ) ) {
2111 0 : *out_exists_in_cache = 1;
2112 0 : *out_evicted_acc_idx = UINT_MAX;
2113 0 : return pinned;
2114 0 : }
2115 0 : FD_SPIN_PAUSE();
2116 0 : continue;
2117 0 : }
2118 :
2119 30 : if( FD_UNLIKELY( claimed ) ) {
2120 0 : fd_racesan_hook( "accdb_cold_load:claim_wait" );
2121 0 : FD_SPIN_PAUSE();
2122 0 : continue;
2123 0 : }
2124 :
2125 30 : if( FD_UNLIKELY( FD_ATOMIC_CAS( &accmeta->executable_size, old_es, old_es | FD_ACCDB_SIZE_CACHE_CLAIM_BIT )!=old_es ) ) {
2126 0 : FD_SPIN_PAUSE();
2127 0 : continue;
2128 0 : }
2129 :
2130 : /* We hold the claim. Allocate a cache line and publish. */
2131 30 : ulong size_class = fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( old_es ) );
2132 30 : fd_accdb_cache_line_t * line = acquire_cache_line( accdb, size_class, out_evicted_acc_idx );
2133 30 : fd_memcpy( line->key.pubkey, accmeta->key.pubkey, 32UL );
2134 30 : line->key.generation = accmeta->key.generation;
2135 : /* Leave acc_idx at UINT_MAX (the "loading" sentinel) until step 12
2136 : publishes it after the preadv2 fence. Concurrent threads that
2137 : pin via cache_idx will spin on this in step 13. */
2138 30 : line->acc_idx = UINT_MAX;
2139 30 : FD_COMPILER_MFENCE();
2140 30 : FD_VOLATILE( accmeta->cache_idx ) = FD_ACCDB_ACC_CIDX_PACK( (uint)size_class, (uint)cache_line_idx( accdb, size_class, line ) );
2141 30 : FD_COMPILER_MFENCE();
2142 :
2143 30 : fd_racesan_hook( "accdb_cold_load:pre_valid" );
2144 :
2145 : /* Atomically set CACHE_VALID_BIT and clear CACHE_CLAIM_BIT.
2146 : Eviction may have flipped CACHE_VALID_BIT on us between our
2147 : claim and now (it preserves CLAIM but can clear VALID); the
2148 : CAS loop tolerates that. The data length and exec bits stay
2149 : unchanged. */
2150 30 : for(;;) {
2151 30 : uint cur = FD_VOLATILE_CONST( accmeta->executable_size );
2152 30 : uint nxt = (cur & ~FD_ACCDB_SIZE_CACHE_CLAIM_BIT) | FD_ACCDB_SIZE_CACHE_VALID_BIT;
2153 30 : if( FD_LIKELY( FD_ATOMIC_CAS( &accmeta->executable_size, cur, nxt )==cur ) ) break;
2154 0 : FD_SPIN_PAUSE();
2155 0 : }
2156 :
2157 30 : *out_exists_in_cache = 0;
2158 30 : return line;
2159 30 : }
2160 30 : }
2161 :
2162 311052 : #define RESERVATION_TYPE_SIMPLE (0)
2163 351 : #define RESERVATION_TYPE_MAYBE_PROGRAMDATA (1)
2164 351 : #define RESERVATION_TYPE_ALREADY_RESERVED (2)
2165 :
2166 : static void
2167 : fd_accdb_acquire_inner( fd_accdb_t * accdb,
2168 : fd_accdb_fork_id_t fork_id,
2169 : int reservation_type,
2170 : ulong reserved_cnt,
2171 : ulong pubkeys_cnt,
2172 : uchar const * const * pubkeys,
2173 : int * writable,
2174 311754 : fd_acc_t * out_accs ) {
2175 311754 : accdb->metrics->acquire_calls++;
2176 :
2177 311754 : ulong max_acquire_cnt = accdb->shmem->bundle_enabled ? FD_ACCDB_MAX_ACQUIRE_CNT : FD_ACCDB_MAX_TX_ACCOUNT_LOCKS;
2178 311754 : FD_TEST( pubkeys_cnt<=max_acquire_cnt );
2179 :
2180 311754 : FD_TEST( FD_VOLATILE_CONST( *accdb->my_epoch_slot )==ULONG_MAX );
2181 :
2182 311754 : FD_COMPILER_MFENCE();
2183 311754 : FD_VOLATILE( *accdb->my_epoch_slot ) = FD_VOLATILE_CONST( accdb->shmem->epoch );
2184 311754 : FD_HW_MFENCE(); /* StoreLoad: epoch store must be globally visible
2185 : before any subsequent loads so the deferred
2186 : reclamation scan does not miss us */
2187 :
2188 : // STEP 1.
2189 : // Locate each account in the fork and index structure, to determine
2190 : // if it already exists, its size and other metadata, and which
2191 : // specific slot (generation) it was last written in.
2192 :
2193 311754 : fd_accdb_fork_t * fork = &accdb->fork_pool[ fork_id.val ];
2194 311754 : uint root_generation = accdb->fork_pool[ accdb->shmem->root_fork_id.val ].shmem->generation;
2195 :
2196 311754 : fd_racesan_hook( "accdb_acquire:post_root_gen" );
2197 :
2198 311754 : fd_accdb_accmeta_t * accmetas[ FD_ACCDB_MAX_ACQUIRE_CNT ];
2199 311754 : ulong acc_map_idxs[ FD_ACCDB_MAX_ACQUIRE_CNT ];
2200 :
2201 : /* Walk the hash chain for each pubkey and take the first visible
2202 : match. Correctness relies on newer entries always being prepended
2203 : to the chain head, which is guaranteed because replay processes
2204 : writes in slot order and release always inserts at the head.
2205 :
2206 : CONCURRENCY: This chain walk runs epoch-protected. A concurrent
2207 : fd_accdb_release may prepend a new node to the same chain while
2208 : we walk it. This is safe on x86-64 (TSO): the releasing thread
2209 : stores all acc fields (pubkey, generation, map.next, ...) before
2210 : publishing the new head via a CAS on acc_map[idx], and TSO
2211 : guarantees a reading core that observes the new head also observes
2212 : all prior stores to the node. A reader that does not yet see the
2213 : new head simply sees an older (still valid) version of the chain.
2214 : On weakly-ordered architectures an explicit acquire fence would be
2215 : needed before the chain walk and a release fence in
2216 : fd_accdb_release before the head-pointer store. Multiple
2217 : concurrent releases serialize on the CAS of the chain head. */
2218 624351 : for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
2219 312597 : acc_map_idxs[ i ] = fd_hash32( pubkeys[ i ], accdb->shmem->seed )&(accdb->shmem->chain_cnt-1UL);
2220 312597 : uint acc = FD_VOLATILE_CONST( accdb->acc_map[ acc_map_idxs[ i ] ] );
2221 389556 : while( acc!=UINT_MAX ) {
2222 174531 : fd_accdb_accmeta_t const * candidate_acc = &accdb->acc_pool[ acc ];
2223 174531 : uint next_acc = FD_VOLATILE_CONST( candidate_acc->map.next );
2224 :
2225 174531 : fd_racesan_hook( "accdb_acquire:post_next" );
2226 :
2227 174531 : if( FD_UNLIKELY( (candidate_acc->key.generation>root_generation &&
2228 174531 : fd_accdb_acc_fork_id(candidate_acc)!=fork_id.val &&
2229 174531 : !descends_set_test( fork->descends, fd_accdb_acc_fork_id(candidate_acc) )) ) ||
2230 174531 : memcmp( pubkeys[ i ], candidate_acc->key.pubkey, 32UL ) ) {
2231 76959 : acc = next_acc;
2232 76959 : continue;
2233 76959 : }
2234 :
2235 97572 : break;
2236 174531 : }
2237 312597 : if( FD_UNLIKELY( acc==UINT_MAX ) ) accmetas[ i ] = NULL;
2238 97572 : else accmetas[ i ] = &accdb->acc_pool[ acc ];
2239 :
2240 : #if FD_TMPL_USE_HANDHOLDING
2241 : if( FD_UNLIKELY( accmetas[ i ] ) ) {
2242 : fd_accdb_accmeta_t const * sel = accmetas[ i ];
2243 : FD_TEST( !memcmp( sel->key.pubkey, pubkeys[ i ], 32UL ) );
2244 : FD_TEST( sel->key.generation<=root_generation ||
2245 : fd_accdb_acc_fork_id( sel )==fork_id.val ||
2246 : descends_set_test( fork->descends, fd_accdb_acc_fork_id( sel ) ) );
2247 : FD_TEST( sel->key.generation<=FD_VOLATILE_CONST( accdb->shmem->generation ) );
2248 : }
2249 : #endif
2250 :
2251 312597 : if( FD_UNLIKELY( accmetas[ i ] && !writable[ i ] && !accmetas[ i ]->lamports ) ) accmetas[ i ] = NULL;
2252 :
2253 : /* Attribute this acquired account to a size class for per-class
2254 : rate metrics. Use the account's current size class when known;
2255 : otherwise (new account) bucket as class 0. */
2256 312597 : ulong acq_class = 0UL;
2257 312597 : if( FD_LIKELY( accmetas[ i ] ) ) acq_class = fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size ) );
2258 312597 : if( FD_LIKELY( writable[ i ] ) ) accdb->metrics->writable_accounts_acquired_per_class[ acq_class ]++;
2259 196035 : else accdb->metrics->accounts_acquired_per_class[ acq_class ]++;
2260 312597 : }
2261 :
2262 : // STEP 2.
2263 : // The two-phase programdata acquire (acquire_a then acquire_b)
2264 : // works as follows: acquire_a (RESERVATION_TYPE_MAYBE_PROGRAMDATA)
2265 : // over-reserves one slot in every live size class per candidate
2266 : // account (reserved_cnt total per class), because it does not yet
2267 : // know which accounts have programdata or what size class it lands
2268 : // in. acquire_b then resolves the actual programdata pubkeys and
2269 : // re-enters here with RESERVATION_TYPE_ALREADY_RESERVED to refund
2270 : // the surplus. Keep one reservation per found programdata account
2271 : // in its own size class (consumed later by release) and give the
2272 : // rest back.
2273 311754 : if( FD_UNLIKELY( reservation_type==RESERVATION_TYPE_ALREADY_RESERVED ) ) {
2274 351 : ulong refund[ FD_ACCDB_CACHE_CLASS_CNT ] = {0};
2275 3159 : for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) {
2276 2808 : if( FD_LIKELY( accdb->shmem->cache_class_used[ j ].val!=ULONG_MAX ) ) refund[ j ] = reserved_cnt;
2277 2808 : }
2278 390 : for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
2279 39 : if( FD_LIKELY( accmetas[ i ] ) ) {
2280 36 : ulong cls = fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size ) );
2281 36 : if( FD_LIKELY( accdb->shmem->cache_class_used[ cls ].val!=ULONG_MAX ) ) {
2282 3 : FD_TEST( refund[ cls ]>0UL );
2283 3 : refund[ cls ]--;
2284 3 : }
2285 36 : }
2286 39 : }
2287 3159 : for( ulong k=0UL; k<FD_ACCDB_CACHE_CLASS_CNT; k++ ) {
2288 2808 : if( FD_UNLIKELY( refund[ k ] ) ) FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->cache_class_used[ k ].val, refund[ k ] );
2289 2808 : }
2290 351 : }
2291 :
2292 : // STEP 3.
2293 : // We are potentially going to need to read the account data off of
2294 : // disk into the cache, if the account(s) are not in the cache so
2295 : // reserve the necessary cache space. This is done with an "atomic
2296 : // subtract" spin loop on the cache class counters, which is
2297 : // actually faster than doing a real CAS on a packed ulong.
2298 : //
2299 : // For reads, we only need space to copy the account data into a
2300 : // single right-sized cache line, but for writes ... we need to
2301 : // reserve one of every size class. The reason is we are going to
2302 : // need a 10MiB staging buffer for the executor to write to (it may
2303 : // grow the account, so needs the max size class). Even if the
2304 : // account is already in the 10MiB cache class, we need another one
2305 : // because a transaction can fail half way, so we need scratch space
2306 : // to be able to unwind.
2307 : //
2308 : // So we acquire one of each size class. Then when the transaction
2309 : // finishes, if it succeeded, we will copy the data back to the
2310 : // whichever size-class is now right-sized post execution.
2311 311754 : if( FD_LIKELY( reservation_type==RESERVATION_TYPE_SIMPLE || reservation_type==RESERVATION_TYPE_MAYBE_PROGRAMDATA ) ) {
2312 311403 : ulong requested_buckets[ FD_ACCDB_CACHE_CLASS_CNT ] = {0};
2313 623961 : for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
2314 312558 : if( FD_LIKELY( accmetas[ i ] || writable[ i ] ) ) {
2315 199203 : if( FD_LIKELY( accmetas[ i ] ) ) {
2316 97536 : if( FD_UNLIKELY( accdb->shmem->cache_class_used[ fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size ) ) ].val!=ULONG_MAX ) ) {
2317 0 : requested_buckets[ fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size ) ) ]++;
2318 0 : }
2319 97536 : }
2320 199203 : if( FD_UNLIKELY( writable[ i ] ) ) {
2321 1049058 : for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) {
2322 932496 : if( FD_UNLIKELY( accdb->shmem->cache_class_used[ j ].val!=ULONG_MAX ) ) {
2323 54 : requested_buckets[ j ]++;
2324 54 : }
2325 932496 : }
2326 116562 : }
2327 199203 : }
2328 :
2329 312558 : if( FD_LIKELY( reservation_type==RESERVATION_TYPE_MAYBE_PROGRAMDATA ) ) {
2330 : /* Any account could also have an implied reference to a
2331 : programdata account, which we don't know yet ... so we need to
2332 : reserve worst case space if they all went to the same size
2333 : class. This reservation runs unconditionally per pubkey (not
2334 : gated on accmetas/writable) so that acquire_b can refund based on
2335 : pubkeys_cnt without needing to re-derive the live-account set. */
2336 13284 : for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) {
2337 11808 : if( FD_UNLIKELY( accdb->shmem->cache_class_used[ j ].val!=ULONG_MAX ) ) {
2338 36 : requested_buckets[ j ]++;
2339 36 : }
2340 11808 : }
2341 1476 : }
2342 312558 : }
2343 :
2344 : /* TODO: This over-reserves cache slots for writable accounts that
2345 : already exist. For each such account we reserve one line in the
2346 : account's size class (for the read into cache) AND one line in
2347 : every size class (for the write destination buffers). But if the
2348 : account is already resident in cache (which is the common case
2349 : for hot accounts), the read-into-cache line is unnecessary — we
2350 : will get a cache hit in step 4 and never use it. The fix is to
2351 : probe acc->cache_idx here and skip the per-account size class
2352 : reservation per-account size class reservation when a hit is
2353 : found. This would reduce peak reservation by up to one line per
2354 : writable account per acquire batch, lowering contention on the
2355 : cache class counters and allowing smaller cache provisioning. */
2356 :
2357 : /* Reserve cache slots by atomically incrementing the shared used
2358 : counters. If any class exceeds its max, the reservation
2359 : overflowed — subtract back partial grabs and retry. */
2360 311403 : for(;;) {
2361 311403 : int acquire_failed = 0;
2362 311403 : ulong grabbed[ FD_ACCDB_CACHE_CLASS_CNT ] = {0};
2363 2802627 : for( ulong i=0UL; i<FD_ACCDB_CACHE_CLASS_CNT; i++ ) {
2364 2491224 : if( FD_LIKELY( !requested_buckets[ i ] ) ) continue;
2365 72 : ulong new_used = FD_ATOMIC_ADD_AND_FETCH( &accdb->shmem->cache_class_used[ i ].val, requested_buckets[ i ] );
2366 72 : if( FD_UNLIKELY( new_used>accdb->shmem->cache_class_max[ i ] ) ) {
2367 0 : FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->cache_class_used[ i ].val, requested_buckets[ i ] );
2368 0 : acquire_failed = 1;
2369 72 : } else {
2370 72 : grabbed[ i ] = requested_buckets[ i ];
2371 72 : }
2372 72 : if( FD_UNLIKELY( acquire_failed ) ) {
2373 0 : accdb->metrics->acquire_failed++;
2374 0 : for( ulong j=0UL; j<i; j++ ) {
2375 0 : if( grabbed[ j ] ) FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->cache_class_used[ j ].val, grabbed[ j ] );
2376 0 : }
2377 0 : FD_SPIN_PAUSE();
2378 0 : break;
2379 0 : }
2380 72 : }
2381 311403 : if( FD_LIKELY( !acquire_failed ) ) break;
2382 311403 : }
2383 311403 : }
2384 :
2385 : // STEP 4.
2386 : // For any accounts that are not in cache, we now need to actually
2387 : // retrieve the cache pointers from our structures. Space has been
2388 : // reserved already, so this step is guaranteed to succeed, and is
2389 : // just pulling the cache lines out of the free lists and marking
2390 : // them as in-use.
2391 : //
2392 : // This step is fully lock-free. Cache hits are pinned with an
2393 : // atomic CAS on refcnt (cache_try_pin). Eviction uses the CLOCK
2394 : // algorithm. The CAS free list provides immediate recycling of
2395 : // fully-freed lines.
2396 :
2397 311754 : int exists_in_cache[ FD_ACCDB_MAX_ACQUIRE_CNT ];
2398 311754 : fd_accdb_cache_line_t * original_cache_line[ FD_ACCDB_MAX_ACQUIRE_CNT ];
2399 311754 : fd_accdb_cache_line_t * destination_cache_lines[ FD_ACCDB_MAX_ACQUIRE_CNT ][ FD_ACCDB_CACHE_CLASS_CNT ];
2400 :
2401 : /* Saved acc_pool indices of evicted dirty cache lines. These are
2402 : captured before clearing acc_idx to UINT_MAX on the line struct, so
2403 : that the sentinel protocol (step 14) works correctly while the
2404 : evicted account metadata is still available for writeback in steps
2405 : 4 and 6. */
2406 311754 : uint evicted_dest_acc[ FD_ACCDB_MAX_ACQUIRE_CNT ][ FD_ACCDB_CACHE_CLASS_CNT ];
2407 311754 : uint evicted_orig_acc[ FD_ACCDB_MAX_ACQUIRE_CNT ];
2408 :
2409 624351 : for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
2410 312597 : if( FD_UNLIKELY( !accmetas[ i ] && !writable[ i ] ) ) continue;
2411 :
2412 199239 : original_cache_line[ i ] = NULL;
2413 199239 : if( FD_LIKELY( accmetas[ i ] ) ) {
2414 97572 : if( FD_LIKELY( FD_ACCDB_SIZE_CACHE_VALID( FD_VOLATILE_CONST( accmetas[ i ]->executable_size ) ) ) ) {
2415 : /* Concurrent evict_clear_acc_cache_ref clears VALID then stores
2416 : cache_idx=INVAL. We may have observed VALID=1 just before the
2417 : writer cleared it, so cidx can read as INVAL here; decoding it
2418 : would yield a wild cache_line pointer. Skip on INVAL. Any
2419 : other stale cidx is harmless: cache_try_pin's ABA generation
2420 : check rejects a recycled line. */
2421 97542 : uint cidx = FD_VOLATILE_CONST( accmetas[ i ]->cache_idx );
2422 97542 : if( FD_LIKELY( cidx!=FD_ACCDB_ACC_CIDX_INVAL ) ) {
2423 97542 : fd_accdb_cache_line_t * hit = cache_line( accdb, FD_ACCDB_ACC_CIDX_CLASS( cidx ), FD_ACCDB_ACC_CIDX_IDX( cidx ) );
2424 97542 : fd_racesan_hook( "accdb_acquire:pre_try_pin" );
2425 97542 : original_cache_line[ i ] = cache_try_pin( hit, pubkeys[ i ], accmetas[ i ]->key.generation );
2426 : #if FD_TMPL_USE_HANDHOLDING
2427 : if( FD_LIKELY( original_cache_line[ i ] ) ) {
2428 : FD_TEST( original_cache_line[ i ]->key.generation==accmetas[ i ]->key.generation &&
2429 : !memcmp( original_cache_line[ i ]->key.pubkey, pubkeys[ i ], 32UL ) );
2430 : uint rc = FD_VOLATILE_CONST( original_cache_line[ i ]->refcnt );
2431 : FD_TEST( rc>0U && rc!=FD_ACCDB_EVICT_SENTINEL );
2432 : }
2433 : #endif
2434 97542 : }
2435 97542 : }
2436 97572 : }
2437 199239 : exists_in_cache[ i ] = original_cache_line[ i ]!=NULL;
2438 :
2439 199239 : if( FD_UNLIKELY( writable[ i ] ) ) {
2440 1049058 : for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) destination_cache_lines[ i ][ j ] = acquire_cache_line( accdb, j, &evicted_dest_acc[ i ][ j ] );
2441 116562 : if( FD_UNLIKELY( accmetas[ i ] && !original_cache_line[ i ] ) ) {
2442 0 : original_cache_line[ i ] = cold_load_acc( accdb, accmetas[ i ], pubkeys[ i ], &exists_in_cache[ i ], &evicted_orig_acc[ i ] );
2443 0 : }
2444 116562 : } else {
2445 82677 : if( FD_UNLIKELY( !original_cache_line[ i ] ) ) {
2446 30 : original_cache_line[ i ] = cold_load_acc( accdb, accmetas[ i ], pubkeys[ i ], &exists_in_cache[ i ], &evicted_orig_acc[ i ] );
2447 30 : }
2448 82677 : }
2449 199239 : }
2450 :
2451 : // STEP 5.
2452 : // For any cache lines we have retrieved, which we might potentially
2453 : // be about to trash (by writing stuff in there), we need to write
2454 : // them back to disk first if they are dirty. This is the process of
2455 : // "persisting" (a/k/a evicting) whatever was previously in the
2456 : // cache line we are about to use.
2457 : //
2458 : // This step does not actually persist the data to disk, it just
2459 : // constructs a series of iovecs (write instructions) which will be
2460 : // used later to do the actual write. The reason is that we want to
2461 : // batch all the writes together into a single writev call, to
2462 : // minimize overhead, and also keep the actual writes at the end of
2463 : // the function and independent of the specific control flow, so
2464 : // that they could be offloaded to another thread of made
2465 : // asynchronous (e.g. with io_uring) in the future without needing
2466 : // to change the rest of the logic.
2467 :
2468 311754 : int write_ops_cnt = 0;
2469 311754 : int write_meta_cnt = 0;
2470 311754 : ulong total_write_sz = 0UL;
2471 311754 : fd_accdb_disk_meta_t write_metas[ (FD_ACCDB_CACHE_CLASS_CNT+1UL)*FD_ACCDB_MAX_ACQUIRE_CNT ];
2472 311754 : struct iovec write_ops[ 2UL*(FD_ACCDB_CACHE_CLASS_CNT+1UL)*FD_ACCDB_MAX_ACQUIRE_CNT ];
2473 :
2474 624351 : for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
2475 312597 : if( FD_UNLIKELY( !accmetas[ i ] && !writable[ i ] ) ) continue;
2476 :
2477 199239 : if( FD_UNLIKELY( writable[ i ] ) ) {
2478 1049058 : for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) {
2479 932496 : if( FD_LIKELY( evicted_dest_acc[ i ][ j ]==UINT_MAX ) ) continue;
2480 0 : accdb->metrics->accounts_evicted++;
2481 0 : accdb->metrics->accounts_evicted_per_class[ j ]++;
2482 :
2483 0 : fd_accdb_accmeta_t const * evicted = &accdb->acc_pool[ evicted_dest_acc[ i ][ j ] ];
2484 0 : fd_racesan_hook( "writeback:pre_synth" );
2485 0 : total_write_sz += sizeof(fd_accdb_disk_meta_t) + FD_ACCDB_SIZE_DATA( evicted->executable_size );
2486 0 : FD_TEST( write_meta_cnt<(int)(sizeof(write_metas)/sizeof(write_metas[0])) );
2487 0 : fd_memcpy( write_metas[ write_meta_cnt ].pubkey, evicted->key.pubkey, 32UL );
2488 0 : write_metas[ write_meta_cnt ].size = FD_ACCDB_SIZE_DATA( evicted->executable_size );
2489 0 : write_metas[ write_meta_cnt ].generation = evicted->key.generation;
2490 0 : fd_memcpy( write_metas[ write_meta_cnt ].owner, destination_cache_lines[ i ][ j ]->owner, 32UL );
2491 0 : write_ops[ write_ops_cnt++ ] = (struct iovec){ .iov_base = &write_metas[ write_meta_cnt ], .iov_len = sizeof(fd_accdb_disk_meta_t) };
2492 0 : write_meta_cnt++;
2493 0 : write_ops[ write_ops_cnt++ ] = (struct iovec){ .iov_base = destination_cache_lines[ i ][ j ]+1UL, .iov_len = FD_ACCDB_SIZE_DATA( evicted->executable_size ) };
2494 0 : }
2495 116562 : if( FD_UNLIKELY( accmetas[ i ] && !exists_in_cache[ i ] && evicted_orig_acc[ i ]!=UINT_MAX ) ) {
2496 0 : fd_accdb_accmeta_t const * evicted = &accdb->acc_pool[ evicted_orig_acc[ i ] ];
2497 0 : accdb->metrics->accounts_evicted++;
2498 0 : accdb->metrics->accounts_evicted_per_class[ fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( evicted->executable_size ) ) ]++;
2499 :
2500 0 : total_write_sz += sizeof(fd_accdb_disk_meta_t) + FD_ACCDB_SIZE_DATA( evicted->executable_size );
2501 0 : FD_TEST( write_meta_cnt<(int)(sizeof(write_metas)/sizeof(write_metas[0])) );
2502 0 : fd_memcpy( write_metas[ write_meta_cnt ].pubkey, evicted->key.pubkey, 32UL );
2503 0 : write_metas[ write_meta_cnt ].size = FD_ACCDB_SIZE_DATA( evicted->executable_size );
2504 0 : write_metas[ write_meta_cnt ].generation = evicted->key.generation;
2505 0 : fd_memcpy( write_metas[ write_meta_cnt ].owner, original_cache_line[ i ]->owner, 32UL );
2506 0 : write_ops[ write_ops_cnt++ ] = (struct iovec){ .iov_base = &write_metas[ write_meta_cnt ], .iov_len = sizeof(fd_accdb_disk_meta_t) };
2507 0 : write_meta_cnt++;
2508 0 : write_ops[ write_ops_cnt++ ] = (struct iovec){ .iov_base = original_cache_line[ i ]+1UL, .iov_len = FD_ACCDB_SIZE_DATA( evicted->executable_size ) };
2509 0 : }
2510 116562 : } else {
2511 82677 : if( FD_LIKELY( exists_in_cache[ i ] || evicted_orig_acc[ i ]==UINT_MAX ) ) continue;
2512 0 : fd_accdb_accmeta_t const * evicted = &accdb->acc_pool[ evicted_orig_acc[ i ] ];
2513 0 : accdb->metrics->accounts_evicted++;
2514 0 : accdb->metrics->accounts_evicted_per_class[ fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( evicted->executable_size ) ) ]++;
2515 0 : total_write_sz += sizeof(fd_accdb_disk_meta_t) + FD_ACCDB_SIZE_DATA( evicted->executable_size );
2516 0 : FD_TEST( write_meta_cnt<(int)(sizeof(write_metas)/sizeof(write_metas[0])) );
2517 0 : fd_memcpy( write_metas[ write_meta_cnt ].pubkey, evicted->key.pubkey, 32UL );
2518 0 : write_metas[ write_meta_cnt ].size = FD_ACCDB_SIZE_DATA( evicted->executable_size );
2519 0 : write_metas[ write_meta_cnt ].generation = evicted->key.generation;
2520 0 : fd_memcpy( write_metas[ write_meta_cnt ].owner, original_cache_line[ i ]->owner, 32UL );
2521 0 : write_ops[ write_ops_cnt++ ] = (struct iovec){ .iov_base = &write_metas[ write_meta_cnt ], .iov_len = sizeof(fd_accdb_disk_meta_t) };
2522 0 : write_meta_cnt++;
2523 0 : write_ops[ write_ops_cnt++ ] = (struct iovec){ .iov_base = original_cache_line[ i ]+1UL, .iov_len = FD_ACCDB_SIZE_DATA( evicted->executable_size ) };
2524 0 : }
2525 199239 : }
2526 :
2527 : // STEP 6-7.
2528 : // Compute the file offset for the writes we are about to do and
2529 : // build the pending offset table. The common case is a single
2530 : // atomic fetch-add on the write head, reserving a contiguous
2531 : // region. If the total eviction batch is too large to fit in one
2532 : // partition (extremely unlikely — requires many dirty 10MiB
2533 : // evictions), fall back to per-entry allocation so that each
2534 : // individual write fits in a single partition.
2535 : //
2536 : // The actual stores to evicted->offset_fork and line->persisted
2537 : // are deferred until after pwritev2 completes (Step 9-10), so
2538 : // a concurrent acquire spinning on offset==FD_ACCDB_OFF_INVAL
2539 : // does not proceed to preadv2 from a location that hasn't been
2540 : // written.
2541 311754 : int pending_cnt = 0;
2542 311754 : fd_accdb_accmeta_t * pending_accs [ (FD_ACCDB_CACHE_CLASS_CNT+1UL)*FD_ACCDB_MAX_ACQUIRE_CNT ];
2543 311754 : ulong pending_offs [ (FD_ACCDB_CACHE_CLASS_CNT+1UL)*FD_ACCDB_MAX_ACQUIRE_CNT ];
2544 311754 : fd_accdb_cache_line_t * pending_lines[ (FD_ACCDB_CACHE_CLASS_CNT+1UL)*FD_ACCDB_MAX_ACQUIRE_CNT ];
2545 :
2546 311754 : ulong file_offset;
2547 311754 : int batch_contiguous;
2548 311754 : if( FD_LIKELY( total_write_sz && total_write_sz<=accdb->shmem->partition_sz ) ) {
2549 0 : file_offset = allocate_next_write( accdb, total_write_sz );
2550 0 : batch_contiguous = 1;
2551 311754 : } else {
2552 311754 : file_offset = 0UL;
2553 311754 : batch_contiguous = 0;
2554 311754 : }
2555 :
2556 311754 : ulong cumulative_offset = 0UL;
2557 624351 : for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
2558 312597 : if( FD_UNLIKELY( !accmetas[ i ] && !writable[ i ] ) ) continue;
2559 :
2560 199239 : if( FD_UNLIKELY( writable[ i ] ) ) {
2561 1049058 : for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) {
2562 932496 : if( FD_LIKELY( evicted_dest_acc[ i ][ j ]==UINT_MAX ) ) continue;
2563 :
2564 0 : fd_accdb_accmeta_t * evicted = &accdb->acc_pool[ evicted_dest_acc[ i ][ j ] ];
2565 0 : ulong entry_sz = sizeof(fd_accdb_disk_meta_t) + (ulong)FD_ACCDB_SIZE_DATA( evicted->executable_size );
2566 : /* xchg-to-INVAL atomically captures the old offset and prevents
2567 : a concurrent acc_unlink from also reading and freeing it (the
2568 : xchg there will see INVAL and skip). Step 10 republishes the
2569 : new offset; the spinner at line ~2082 tolerates the transient
2570 : INVAL. Same pattern as the overwrite path at line ~2388. */
2571 0 : ulong old_off = fd_accdb_acc_xchg_offset( evicted, FD_ACCDB_OFF_INVAL );
2572 0 : if( FD_LIKELY( old_off!=FD_ACCDB_OFF_INVAL ) ) {
2573 0 : fd_accdb_shmem_bytes_freed( accdb->shmem, old_off, entry_sz );
2574 0 : FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->shmetrics->disk_used_bytes, entry_sz );
2575 0 : }
2576 0 : FD_TEST( pending_cnt<(int)(sizeof(pending_accs)/sizeof(pending_accs[0])) );
2577 0 : pending_accs [ pending_cnt ] = evicted;
2578 0 : if( FD_LIKELY( batch_contiguous ) ) pending_offs[ pending_cnt ] = file_offset + cumulative_offset;
2579 0 : else pending_offs[ pending_cnt ] = allocate_next_write( accdb, entry_sz );
2580 0 : pending_lines[ pending_cnt ] = destination_cache_lines[ i ][ j ];
2581 0 : pending_cnt++;
2582 0 : cumulative_offset += entry_sz;
2583 0 : FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->shmetrics->disk_used_bytes, entry_sz );
2584 0 : }
2585 116562 : if( FD_UNLIKELY( accmetas[ i ] && !exists_in_cache[ i ] && evicted_orig_acc[ i ]!=UINT_MAX ) ) {
2586 0 : fd_accdb_accmeta_t * evicted = &accdb->acc_pool[ evicted_orig_acc[ i ] ];
2587 0 : ulong entry_sz = sizeof(fd_accdb_disk_meta_t) + (ulong)FD_ACCDB_SIZE_DATA( evicted->executable_size );
2588 0 : ulong old_off = fd_accdb_acc_xchg_offset( evicted, FD_ACCDB_OFF_INVAL );
2589 0 : if( FD_LIKELY( old_off!=FD_ACCDB_OFF_INVAL ) ) {
2590 0 : fd_accdb_shmem_bytes_freed( accdb->shmem, old_off, entry_sz );
2591 0 : FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->shmetrics->disk_used_bytes, entry_sz );
2592 0 : }
2593 0 : FD_TEST( pending_cnt<(int)(sizeof(pending_accs)/sizeof(pending_accs[0])) );
2594 0 : pending_accs [ pending_cnt ] = evicted;
2595 0 : if( FD_LIKELY( batch_contiguous ) ) pending_offs[ pending_cnt ] = file_offset + cumulative_offset;
2596 0 : else pending_offs[ pending_cnt ] = allocate_next_write( accdb, entry_sz );
2597 0 : pending_lines[ pending_cnt ] = original_cache_line[ i ];
2598 0 : pending_cnt++;
2599 0 : cumulative_offset += entry_sz;
2600 0 : FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->shmetrics->disk_used_bytes, entry_sz );
2601 0 : }
2602 116562 : } else {
2603 82677 : if( FD_LIKELY( exists_in_cache[ i ] || evicted_orig_acc[ i ]==UINT_MAX ) ) continue;
2604 :
2605 0 : fd_accdb_accmeta_t * evicted = &accdb->acc_pool[ evicted_orig_acc[ i ] ];
2606 0 : ulong entry_sz = sizeof(fd_accdb_disk_meta_t) + (ulong)FD_ACCDB_SIZE_DATA( evicted->executable_size );
2607 0 : ulong old_off = fd_accdb_acc_xchg_offset( evicted, FD_ACCDB_OFF_INVAL );
2608 0 : if( FD_LIKELY( old_off!=FD_ACCDB_OFF_INVAL ) ) {
2609 0 : fd_accdb_shmem_bytes_freed( accdb->shmem, old_off, entry_sz );
2610 0 : FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->shmetrics->disk_used_bytes, entry_sz );
2611 0 : }
2612 0 : FD_TEST( pending_cnt<(int)(sizeof(pending_accs)/sizeof(pending_accs[0])) );
2613 0 : pending_accs [ pending_cnt ] = evicted;
2614 0 : if( FD_LIKELY( batch_contiguous ) ) pending_offs[ pending_cnt ] = file_offset + cumulative_offset;
2615 0 : else pending_offs[ pending_cnt ] = allocate_next_write( accdb, entry_sz );
2616 0 : pending_lines[ pending_cnt ] = original_cache_line[ i ];
2617 0 : pending_cnt++;
2618 0 : cumulative_offset += entry_sz;
2619 0 : FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->shmetrics->disk_used_bytes, entry_sz );
2620 0 : }
2621 199239 : }
2622 :
2623 : // STEP 8.
2624 : // Fill the output entries with cache pointers and metadata based on
2625 : // the accounts we have located and the cache lines we have
2626 : // reserved.
2627 :
2628 624351 : for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
2629 312597 : if( FD_UNLIKELY( !accmetas[ i ] && !writable[ i ] ) ) {
2630 113358 : out_accs[ i ].data = NULL;
2631 113358 : out_accs[ i ].data_len = 0UL;
2632 113358 : out_accs[ i ].lamports = 0UL;
2633 113358 : out_accs[ i ].executable = 0;
2634 113358 : memset( out_accs[ i ].owner, 0, 32UL );
2635 113358 : fd_memcpy( out_accs[ i ].pubkey, pubkeys[ i ], 32UL );
2636 113358 : out_accs[ i ].prior_lamports = 0UL;
2637 113358 : out_accs[ i ].prior_data_len = 0UL;
2638 113358 : out_accs[ i ].prior_executable = 0;
2639 113358 : memset( out_accs[ i ].prior_owner, 0, 32UL );
2640 113358 : out_accs[ i ].prior_data = NULL;
2641 113358 : out_accs[ i ].commit = 0;
2642 113358 : out_accs[ i ].pd_write = 0;
2643 113358 : out_accs[ i ]._writable = 0;
2644 113358 : out_accs[ i ]._original_size_class = ULONG_MAX;
2645 113358 : out_accs[ i ]._original_cache_idx = ULONG_MAX;
2646 113358 : continue;
2647 113358 : }
2648 :
2649 199239 : if( FD_LIKELY( !writable[ i ] ) ) out_accs[ i ].data = (uchar *)(original_cache_line[ i ]+1UL);
2650 116562 : else out_accs[ i ].data = (uchar *)(destination_cache_lines[ i ][ 7UL ]+1UL);
2651 : /* Tombstone reset: agave's account loader returns AccountSharedData::default()
2652 : (System owner, empty data, exec=0) for any account with lamports==0.
2653 : https://github.com/anza-xyz/agave/blob/v2.3.1/svm/src/account_loader.rs#L199-L228 */
2654 199239 : fd_racesan_hook( "accdb_acquire:pre_step7_meta" );
2655 199239 : int tombstone = accmetas[ i ] && accmetas[ i ]->lamports==0UL;
2656 199239 : out_accs[ i ].data_len = ( accmetas[ i ] && !tombstone ) ? FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size ) : 0UL;
2657 199239 : out_accs[ i ].executable = ( accmetas[ i ] && !tombstone ) ? FD_ACCDB_SIZE_EXEC( accmetas[ i ]->executable_size ) : 0;
2658 199239 : fd_racesan_hook( "accdb_acquire:mid_step7_meta" );
2659 199239 : out_accs[ i ].lamports = accmetas[ i ] ? accmetas[ i ]->lamports : 0UL;
2660 199239 : if( FD_UNLIKELY( !accmetas[ i ] ) ) {
2661 101667 : memset( out_accs[ i ].owner, 0, 32UL );
2662 101667 : memset( out_accs[ i ].prior_owner, 0, 32UL );
2663 101667 : }
2664 : /* For accmetas[i] != NULL, both owners are copied from the cache
2665 : line below in step 15, after step 12 has populated it from disk
2666 : for cold loads. */
2667 :
2668 199239 : out_accs[ i ].prior_lamports = out_accs[ i ].lamports;
2669 199239 : out_accs[ i ].prior_data_len = out_accs[ i ].data_len;
2670 199239 : out_accs[ i ].prior_executable = out_accs[ i ].executable;
2671 199239 : out_accs[ i ].prior_data = (uchar *)(original_cache_line[ i ] ? (original_cache_line[ i ]+1UL) : NULL);
2672 :
2673 199239 : out_accs[ i ].commit = 0;
2674 199239 : out_accs[ i ].pd_write = 0;
2675 199239 : out_accs[ i ]._writable = writable[ i ];
2676 199239 : if( FD_UNLIKELY( writable[ i ] && accmetas[ i ] ) ) out_accs[ i ]._overwrite = accdb->fork_pool[ fork_id.val ].shmem->generation==accmetas[ i ]->key.generation;
2677 184344 : else out_accs[ i ]._overwrite = 0;
2678 :
2679 199239 : FD_TEST( out_accs[ i ].data_len<=(10UL<<20) );
2680 199239 : FD_TEST( !out_accs[ i ]._overwrite || accdb->fork_pool[ fork_id.val ].shmem->generation==accmetas[ i ]->key.generation );
2681 :
2682 : #if FD_TMPL_USE_HANDHOLDING
2683 : if( FD_UNLIKELY( !writable[ i ] && accmetas[ i ] && !tombstone ) ) {
2684 : ulong cls = fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size ) );
2685 : FD_TEST( fd_accdb_ptr_in_region( accdb, cls, out_accs[ i ].data ) );
2686 : }
2687 : #endif
2688 :
2689 199239 : if( FD_UNLIKELY( writable[ i ] ) ) {
2690 116562 : out_accs[ i ]._fork_id = fork_id.val;
2691 116562 : out_accs[ i ]._generation = fork->shmem->generation;
2692 116562 : out_accs[ i ]._acc_map_idx = acc_map_idxs[ i ];
2693 116562 : }
2694 199239 : fd_memcpy( out_accs[ i ].pubkey, pubkeys[ i ], 32UL );
2695 :
2696 199239 : if( FD_UNLIKELY( !accmetas[ i ] ) ) {
2697 101667 : out_accs[ i ]._original_size_class = ULONG_MAX;
2698 101667 : out_accs[ i ]._original_cache_idx = ULONG_MAX;
2699 101667 : } else {
2700 97572 : out_accs[ i ]._original_size_class = fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size ) );
2701 97572 : out_accs[ i ]._original_cache_idx = cache_line_idx( accdb, out_accs[ i ]._original_size_class, original_cache_line[ i ] );
2702 97572 : }
2703 :
2704 199239 : if( FD_UNLIKELY( writable[ i ] ) ) {
2705 1049058 : for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) {
2706 932496 : out_accs[ i ]._write.destination_cache_idx[ j ] = cache_line_idx( accdb, j, destination_cache_lines[ i ][ j ] );
2707 932496 : }
2708 116562 : }
2709 199239 : }
2710 :
2711 : // STEP 9.
2712 : // Write the dirty eviction data to disk and publish the new offsets
2713 : // BEFORE constructing read iovecs. This is critical: step 4 may
2714 : // have evicted a dirty cache line belonging to another account in
2715 : // the same batch whose acc->offset is still FD_ACCDB_OFF_INVAL.
2716 : // The read-iovec loop below spin-waits on
2717 : // offset!=FD_ACCDB_OFF_INVAL, so publishing evicted offsets first
2718 : // prevents an intra-batch deadlock where the thread waits on an
2719 : // offset that only it can resolve.
2720 311754 : if( FD_LIKELY( batch_contiguous ) ) {
2721 : /* Fast path: all evictions fit in one contiguous region. Use the
2722 : pre-built iovec array for a single batched pwritev2 call. */
2723 0 : ulong bytes_written = 0UL;
2724 0 : struct iovec * write_ptr = write_ops;
2725 0 : while( FD_LIKELY( bytes_written<total_write_sz ) ) {
2726 0 : long result = pwritev2( accdb->fd, write_ptr, fd_int_min( write_ops_cnt, IOV_MAX ), (long)(file_offset+bytes_written), 0 );
2727 0 : if( FD_UNLIKELY( -1==result && (errno==EINTR || errno==EAGAIN || errno==EWOULDBLOCK ) ) ) continue;
2728 0 : else if( FD_UNLIKELY( -1==result ) ) FD_LOG_ERR(( "pwritev2() failed (%d-%s)", errno, fd_io_strerror( errno ) ));
2729 0 : else if( FD_UNLIKELY( !result ) ) FD_LOG_ERR(( "accounts database is corrupt, pwritev2() returned 0 at offset %lu with %lu bytes remaining",
2730 0 : file_offset+bytes_written, total_write_sz-bytes_written ));
2731 0 : bytes_written += (ulong)result;
2732 0 : accdb->metrics->bytes_written += (ulong)result;
2733 0 : accdb->metrics->write_ops++;
2734 :
2735 0 : while( write_ops_cnt && (ulong)result>=(ulong)write_ptr[ 0 ].iov_len ) {
2736 0 : result -= (long)write_ptr[ 0 ].iov_len;
2737 0 : write_ptr++;
2738 0 : write_ops_cnt--;
2739 0 : }
2740 0 : if( FD_LIKELY( write_ops_cnt ) ) {
2741 0 : write_ptr[ 0 ].iov_base = (uchar *)write_ptr[ 0 ].iov_base + result;
2742 0 : write_ptr[ 0 ].iov_len -= (ulong)result;
2743 0 : }
2744 0 : }
2745 311754 : } else {
2746 : /* Slow path: total eviction batch exceeds a single partition.
2747 : Write each entry individually using its own allocated offset.
2748 : This path is only taken in extreme edge cases (many concurrent
2749 : dirty 10 MiB evictions). */
2750 311754 : struct iovec * wp = write_ops;
2751 311754 : for( int k=0; k<pending_cnt; k++ ) {
2752 0 : ulong entry_sz = sizeof(fd_accdb_disk_meta_t) + (ulong)FD_ACCDB_SIZE_DATA( pending_accs[ k ]->executable_size );
2753 0 : ulong entry_off = pending_offs[ k ];
2754 0 : struct iovec entry_iovs[2] = { wp[0], wp[1] };
2755 0 : wp += 2;
2756 :
2757 0 : ulong written = 0UL;
2758 0 : while( FD_LIKELY( written<entry_sz ) ) {
2759 0 : long result = pwritev2( accdb->fd, entry_iovs, 2, (long)(entry_off+written), 0 );
2760 0 : if( FD_UNLIKELY( -1==result && (errno==EINTR || errno==EAGAIN || errno==EWOULDBLOCK ) ) ) continue;
2761 0 : else if( FD_UNLIKELY( -1==result ) ) FD_LOG_ERR(( "pwritev2() failed (%d-%s)", errno, fd_io_strerror( errno ) ));
2762 0 : else if( FD_UNLIKELY( !result ) ) FD_LOG_ERR(( "accounts database is corrupt, pwritev2() returned 0 at offset %lu with %lu bytes remaining", entry_off+written, entry_sz-written ));
2763 0 : written += (ulong)result;
2764 0 : accdb->metrics->bytes_written += (ulong)result;
2765 0 : accdb->metrics->write_ops++;
2766 :
2767 0 : for( int v=0; v<2; v++ ) {
2768 0 : if( (ulong)result>=(ulong)entry_iovs[ v ].iov_len ) {
2769 0 : result -= (long)entry_iovs[ v ].iov_len;
2770 0 : entry_iovs[ v ].iov_len = 0UL;
2771 0 : } else {
2772 0 : entry_iovs[ v ].iov_base = (uchar *)entry_iovs[ v ].iov_base + result;
2773 0 : entry_iovs[ v ].iov_len -= (ulong)result;
2774 0 : break;
2775 0 : }
2776 0 : }
2777 0 : }
2778 0 : }
2779 311754 : }
2780 :
2781 : // STEP 10.
2782 : // Now that the data is on disk, publish the evicted account offsets
2783 : // so concurrent acquire threads spinning on
2784 : // offset==FD_ACCDB_OFF_INVAL can proceed. The fence ensures
2785 : // pwritev2 data is globally visible before the offset stores.
2786 311754 : FD_COMPILER_MFENCE();
2787 311754 : for( int k=0; k<pending_cnt; k++ ) {
2788 0 : pending_accs[ k ]->offset_fork = fd_accdb_acc_pack_offset_fork( pending_offs[ k ], fd_accdb_acc_fork_id(pending_accs[ k ]) );
2789 0 : pending_lines[ k ]->persisted = 1;
2790 0 : }
2791 :
2792 : // STEP 11.
2793 : // Now construct iovecs for any reads we need to do of accounts into
2794 : // the cache. For reading accounts, we read them directly into the
2795 : // sole cache line we took (and maybe just evicted). For writing
2796 : // accounts, we read them into the right sized cache line, and later
2797 : // it will be copied to the staging buffer. This is to prevent
2798 : // repeatedly reading the same account off disk into cache, if it is
2799 : // being written cold multiple times and every write fails.
2800 :
2801 311754 : ulong read_ops_cnt = 0UL;
2802 311754 : ulong read_offsets[ FD_ACCDB_CACHE_CLASS_CNT*FD_ACCDB_MAX_ACQUIRE_CNT ];
2803 311754 : uchar * read_bases[ FD_ACCDB_CACHE_CLASS_CNT*FD_ACCDB_MAX_ACQUIRE_CNT ];
2804 311754 : ulong read_sizes[ FD_ACCDB_CACHE_CLASS_CNT*FD_ACCDB_MAX_ACQUIRE_CNT ];
2805 311754 : struct iovec read_ops[ FD_ACCDB_CACHE_CLASS_CNT*FD_ACCDB_MAX_ACQUIRE_CNT ];
2806 :
2807 624351 : for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
2808 312597 : if( FD_UNLIKELY( !accmetas[ i ] || exists_in_cache[ i ] ) ) continue;
2809 :
2810 30 : accdb->metrics->accounts_not_found_per_class[ fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size ) ) ]++;
2811 :
2812 : /* Tombstones (lamports==0) have no on-disk payload to read, and
2813 : background_advance_root may unlink the acc and never assign it a
2814 : disk offset, so the offset_fork spin below would hang forever.
2815 : Step 15's tombstone reset zeros the owner for these accounts. */
2816 30 : if( FD_UNLIKELY( !accmetas[ i ]->lamports ) ) continue;
2817 :
2818 : /* We are guaranteed that if an account is in the cache, the bytes
2819 : are available (all cache operations are atomic via refcnt CAS),
2820 : but we are not guaranteed that if something is _not_ in the cache
2821 : that it has been written back to disk yet. In particular, if we
2822 : are trying to read an account that another thread is in the
2823 : process of evicting, we know they removed it from the cache, but
2824 : we don't know exactly when they will have written it back fully
2825 : to disk, so we may need to wait for that here.
2826 :
2827 : Compaction may concurrently relocate this record, but
2828 : epoch-based safe reclamation guarantees the source partition
2829 : is not freed until all epoch-protected operations that could
2830 : have snapshotted the old offset have exited. So the data at the
2831 : snapshotted offset remains stable for the duration of our
2832 : read and no post-read validation is needed. */
2833 30 : ulong off_packed = FD_VOLATILE_CONST( accmetas[ i ]->offset_fork );
2834 30 : while( FD_UNLIKELY( (off_packed & FD_ACCDB_OFF_MASK)==FD_ACCDB_OFF_INVAL ) ) {
2835 0 : FD_SPIN_PAUSE();
2836 0 : off_packed = FD_VOLATILE_CONST( accmetas[ i ]->offset_fork );
2837 0 : }
2838 30 : fd_racesan_hook( "accdb_coldload:pre_iovec" );
2839 :
2840 30 : read_offsets[ read_ops_cnt ] = fd_accdb_acc_offset(accmetas[ i ]) + offsetof(fd_accdb_disk_meta_t, owner);
2841 30 : read_bases[ read_ops_cnt ] = original_cache_line[ i ]->owner;
2842 30 : read_sizes[ read_ops_cnt ] = 32UL + FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size );
2843 30 : read_ops[ read_ops_cnt++ ] = (struct iovec){ .iov_base = original_cache_line[ i ]->owner, .iov_len = 32UL + FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size ) };
2844 30 : }
2845 :
2846 : // STEP 12.
2847 : // Almost done... now do the actual reads of accounts into cache,
2848 : // using the iovecs we constructed. This is basically the same loop
2849 : // as the writes, but with preadv2 instead of pwritev2, and that the
2850 : // reads are not necessarily all contiguous, but occur at random
2851 : // offsets.
2852 : //
2853 : // CONCURRENCY: The compaction tile may concurrently relocate a
2854 : // record we are about to read (both are epoch-protected). Epoch-
2855 : // based safe reclamation guarantees the source partition is not
2856 : // freed until all epoch-protected operations that could have
2857 : // snapshotted the old offset have exited, so the data at the
2858 : // remains stable for the duration of this read — no post-read
2859 : // validation or retry is needed.
2860 311784 : for( ulong i=0UL; i<read_ops_cnt; i++ ) {
2861 30 : ulong bytes_read = 0UL;
2862 60 : while( FD_LIKELY( bytes_read<read_sizes[ i ] ) ) {
2863 30 : long result = preadv2( accdb->fd, &read_ops[ i ], 1, (long)(read_offsets[ i ]+bytes_read), 0 );
2864 30 : if( FD_UNLIKELY( -1==result && (errno==EINTR || errno==EAGAIN || errno==EWOULDBLOCK ) ) ) continue;
2865 30 : else if( FD_UNLIKELY( -1==result ) ) FD_LOG_ERR(( "preadv2() failed (%d-%s)", errno, fd_io_strerror( errno ) ));
2866 30 : else if( FD_UNLIKELY( !result ) ) FD_LOG_ERR(( "accounts database is corrupt, data expected at offset %lu with size %lu exceeded file extents",
2867 30 : read_offsets[ i ]+bytes_read, read_sizes[ i ] ));
2868 30 : fd_accdb_partition_read_bump( accdb, read_offsets[ i ]+bytes_read, (ulong)result );
2869 30 : bytes_read += (ulong)result;
2870 30 : accdb->metrics->bytes_read += (ulong)result;
2871 30 : accdb->metrics->read_ops++;
2872 :
2873 30 : read_ops[ i ].iov_base = read_bases[ i ] + bytes_read;
2874 30 : read_ops[ i ].iov_len = read_sizes[ i ] - bytes_read;
2875 30 : }
2876 30 : }
2877 :
2878 : // STEP 13.
2879 : // Publish the real acc index for any cache lines we just loaded
2880 : // from disk, so concurrent threads spinning on acc_idx==UINT_MAX
2881 : // can proceed. The fence ensures all preadv2 data is visible
2882 : // before the sentinel is cleared.
2883 311754 : FD_COMPILER_MFENCE();
2884 624351 : for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
2885 312597 : if( FD_UNLIKELY( !accmetas[ i ] || exists_in_cache[ i ] ) ) continue;
2886 30 : FD_VOLATILE( original_cache_line[ i ]->acc_idx ) = (uint)( accmetas[ i ] - accdb->acc_pool );
2887 30 : FD_TEST( FD_VOLATILE_CONST( original_cache_line[ i ]->acc_idx )==(uint)( accmetas[ i ] - accdb->acc_pool ) );
2888 30 : }
2889 :
2890 : // STEP 14.
2891 : // Spin-wait for any cache lines found via acc->cache_idx that are
2892 : // still being loaded by another thread's preadv2. The loading
2893 : // thread sets acc_idx to UINT_MAX before publishing cache_idx
2894 : // and publishes the real acc index after its read completes.
2895 : // This step is placed as late as possible to give the loading
2896 : // thread maximum time to finish before we need to spin.
2897 624351 : for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
2898 312597 : if( FD_UNLIKELY( !accmetas[ i ] && !writable[ i ] ) ) continue;
2899 :
2900 199239 : if( FD_UNLIKELY( !original_cache_line[ i ] ) ) continue;
2901 97572 : if( FD_LIKELY( FD_VOLATILE_CONST( original_cache_line[ i ]->acc_idx )!=UINT_MAX ) ) goto step13_check;
2902 0 : accdb->metrics->accounts_waited++;
2903 0 : while( FD_UNLIKELY( FD_VOLATILE_CONST( original_cache_line[ i ]->acc_idx )==UINT_MAX ) ) {
2904 0 : fd_racesan_hook( "accdb_acquire:step14_load_wait" );
2905 0 : FD_SPIN_PAUSE();
2906 0 : }
2907 97572 : step13_check:;
2908 : #if FD_TMPL_USE_HANDHOLDING
2909 : FD_TEST( original_cache_line[ i ]->key.generation==accmetas[ i ]->key.generation &&
2910 : !memcmp( original_cache_line[ i ]->key.pubkey, pubkeys[ i ], 32UL ) );
2911 : #endif
2912 97572 : }
2913 :
2914 : // STEP 15.
2915 : // Now that all reads from disk into original_cache_line have
2916 : // completed (and any concurrent loaders have published their
2917 : // acc_idx in step 14), copy the owner into the output entries.
2918 : // This must happen here rather than in step 8 because the cache
2919 : // line owner is only valid post-read for cold loads.
2920 624351 : for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
2921 312597 : if( FD_UNLIKELY( !accmetas[ i ] ) ) continue;
2922 97572 : fd_racesan_hook( "accdb_acquire:pre_step14_owner" );
2923 : /* Tombstone reset: see STEP 7 comment. */
2924 97572 : if( FD_UNLIKELY( accmetas[ i ]->lamports==0UL ) ) {
2925 30 : memset( out_accs[ i ].owner, 0, 32UL );
2926 30 : memset( out_accs[ i ].prior_owner, 0, 32UL );
2927 97542 : } else {
2928 97542 : fd_memcpy( out_accs[ i ].owner, original_cache_line[ i ]->owner, 32UL );
2929 97542 : fd_memcpy( out_accs[ i ].prior_owner, original_cache_line[ i ]->owner, 32UL );
2930 97542 : }
2931 97572 : }
2932 :
2933 : // STEP 16.
2934 : // Finally, copy any accounts we are writing into the staging
2935 : // buffers, so they occupy a 10MiB cache line for the execution
2936 : // system.
2937 624351 : for( ulong i=0UL; i<pubkeys_cnt; i++ ) {
2938 312597 : if( FD_UNLIKELY( !accmetas[ i ] || !writable[ i ] ) ) continue;
2939 :
2940 14895 : ulong copy_sz = (ulong)FD_ACCDB_SIZE_DATA( accmetas[ i ]->executable_size );
2941 14895 : fd_memcpy( destination_cache_lines[ i ][ 7UL ]+1UL, original_cache_line[ i ]+1UL, copy_sz );
2942 14895 : accdb->metrics->bytes_copied += copy_sz;
2943 14895 : }
2944 :
2945 311754 : FD_COMPILER_MFENCE();
2946 311754 : FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
2947 311754 : }
2948 :
2949 : void
2950 : fd_accdb_acquire( fd_accdb_t * accdb,
2951 : fd_accdb_fork_id_t fork_id,
2952 : ulong pubkeys_cnt,
2953 : uchar const * const * pubkeys,
2954 : int * writable,
2955 311052 : fd_acc_t * out_accs ) {
2956 311052 : FD_TEST( accdb->acquire_state==FD_ACCDB_ACQUIRE_STATE_IDLE );
2957 311052 : accdb->acquire_state = FD_ACCDB_ACQUIRE_STATE_OPEN;
2958 311052 : fd_accdb_acquire_inner( accdb, fork_id, RESERVATION_TYPE_SIMPLE, 0UL, pubkeys_cnt, pubkeys, writable, out_accs );
2959 311052 : }
2960 :
2961 : void
2962 : fd_accdb_acquire_a( fd_accdb_t * accdb,
2963 : fd_accdb_fork_id_t fork_id,
2964 : ulong pubkeys_cnt,
2965 : uchar const * const * pubkeys,
2966 : int * writable,
2967 351 : fd_acc_t * out_accs ) {
2968 351 : FD_TEST( accdb->acquire_state==FD_ACCDB_ACQUIRE_STATE_IDLE );
2969 351 : accdb->acquire_state = FD_ACCDB_ACQUIRE_STATE_PHASE_A;
2970 351 : fd_accdb_acquire_inner( accdb, fork_id, RESERVATION_TYPE_MAYBE_PROGRAMDATA, 0UL, pubkeys_cnt, pubkeys, writable, out_accs );
2971 351 : }
2972 :
2973 : void
2974 : fd_accdb_acquire_b( fd_accdb_t * accdb,
2975 : fd_accdb_fork_id_t fork_id,
2976 : ulong reserved_cnt,
2977 : ulong pubkeys_cnt,
2978 : uchar const * const * pubkeys,
2979 : int * writable,
2980 351 : fd_acc_t * out_accs ) {
2981 351 : FD_TEST( accdb->acquire_state==FD_ACCDB_ACQUIRE_STATE_PHASE_A );
2982 351 : accdb->acquire_state = FD_ACCDB_ACQUIRE_STATE_OPEN;
2983 351 : fd_accdb_acquire_inner( accdb, fork_id, RESERVATION_TYPE_ALREADY_RESERVED, reserved_cnt, pubkeys_cnt, pubkeys, writable, out_accs );
2984 351 : }
2985 :
2986 : /* release_inner drains one group of acquired accs but does NOT change the
2987 : handle's acquire_state. The public fd_accdb_release / fd_accdb_release_ab
2988 : wrappers below own the state transition (a single-phase release closes
2989 : the bracket; release_ab drains both phase groups then closes). */
2990 : static void
2991 : release_inner( fd_accdb_t * accdb,
2992 : ulong accs_cnt,
2993 311322 : fd_acc_t * accs ) {
2994 311322 : FD_TEST( accdb->acquire_state==FD_ACCDB_ACQUIRE_STATE_OPEN );
2995 :
2996 311322 : {
2997 311322 : ulong prev = FD_VOLATILE_CONST( *accdb->my_epoch_slot );
2998 311322 : FD_TEST( prev==ULONG_MAX || prev<=FD_VOLATILE_CONST( accdb->shmem->epoch ) );
2999 311322 : }
3000 :
3001 311322 : FD_COMPILER_MFENCE();
3002 311322 : FD_VOLATILE( *accdb->my_epoch_slot ) = FD_VOLATILE_CONST( accdb->shmem->epoch );
3003 311322 : FD_HW_MFENCE(); /* StoreLoad: epoch store must be globally visible
3004 : before any subsequent loads so the deferred
3005 : reclamation scan does not miss us. */
3006 :
3007 : // STEP 1.
3008 : // For each cache line which was written to in the 10MiB staging
3009 : // buffer, we may need to copy to the data out to a right sized
3010 : // cache line. Figuring out the target cache line is non-obvious,
3011 : // but follows the more complete logic below this, we just pull the
3012 : // memcpy out so they are not done inside the cache lock.
3013 :
3014 623490 : for( ulong i=0UL; i<accs_cnt; i++ ) {
3015 312168 : if( FD_UNLIKELY( accs[ i ]._original_size_class==ULONG_MAX && !accs[ i ]._writable ) ) continue;
3016 :
3017 : #if FD_TMPL_USE_HANDHOLDING
3018 : if( FD_LIKELY( accs[ i ]._original_size_class!=ULONG_MAX ) ) {
3019 : FD_TEST( accs[ i ]._original_cache_idx<accdb->shmem->cache_class_max[ accs[ i ]._original_size_class ] );
3020 : }
3021 : if( FD_UNLIKELY( accs[ i ].commit ) ) FD_TEST( accs[ i ]._writable );
3022 : #endif
3023 :
3024 198837 : if( FD_LIKELY( !accs[ i ]._writable || !accs[ i ].commit ) ) continue;
3025 : #if FD_TMPL_USE_HANDHOLDING
3026 : if( FD_UNLIKELY( accs[ i ]._overwrite ) ) {
3027 : FD_TEST( accs[ i ]._writable );
3028 : FD_TEST( accs[ i ]._original_cache_idx!=ULONG_MAX );
3029 : FD_TEST( accs[ i ]._original_size_class!=ULONG_MAX );
3030 : }
3031 : #endif
3032 :
3033 115629 : ulong original_size_class = accs[ i ]._original_size_class;
3034 115629 : ulong new_size_class = fd_accdb_cache_class( accs[ i ].data_len );
3035 115629 : if( FD_UNLIKELY( new_size_class==7UL ) ) continue;
3036 :
3037 115317 : fd_accdb_cache_line_t * target_cache_line;
3038 115317 : if( FD_LIKELY( original_size_class==new_size_class && accs[ i ]._overwrite ) ) target_cache_line = cache_line( accdb, original_size_class, accs[ i ]._original_cache_idx );
3039 112134 : else target_cache_line = cache_line( accdb, new_size_class, accs[ i ]._write.destination_cache_idx[ new_size_class ] );
3040 :
3041 115317 : fd_accdb_cache_line_t * staging_line = cache_line( accdb, 7UL, accs[ i ]._write.destination_cache_idx[ 7UL ] );
3042 :
3043 115317 : fd_racesan_hook( "accdb_commit:pre_owner_write" );
3044 :
3045 : #if FD_TMPL_USE_HANDHOLDING
3046 : if( FD_UNLIKELY( original_size_class==new_size_class && accs[ i ]._overwrite ) ) {
3047 : uint rc = FD_VOLATILE_CONST( target_cache_line->refcnt );
3048 : FD_TEST( target_cache_line->key.generation==accs[ i ]._generation &&
3049 : !memcmp( target_cache_line->key.pubkey, accs[ i ].pubkey, 32UL ) &&
3050 : rc>0U &&
3051 : rc!=FD_ACCDB_EVICT_SENTINEL );
3052 : }
3053 : #endif
3054 :
3055 115317 : fd_memcpy( target_cache_line->owner, accs[ i ].owner, 32UL );
3056 115317 : fd_memcpy( target_cache_line+1UL, staging_line+1UL, accs[ i ].data_len );
3057 115317 : accdb->metrics->bytes_copied += accs[ i ].data_len;
3058 115317 : }
3059 :
3060 : // STEP 2.
3061 : // Now update the metadata structures and free lists to reflect the
3062 : // fact that we are done with these cache lines. This is fully
3063 : // atomic with CLOCK.
3064 :
3065 623490 : for( ulong i=0UL; i<accs_cnt; i++ ) {
3066 312168 : if( FD_UNLIKELY( accs[ i ]._original_size_class==ULONG_MAX && !accs[ i ]._writable ) ) continue;
3067 :
3068 198837 : ulong original_size_class = accs[ i ]._original_size_class;
3069 198837 : fd_accdb_cache_line_t * original_cache_line = accs[ i ]._original_cache_idx==ULONG_MAX ? NULL : cache_line( accdb, original_size_class, accs[ i ]._original_cache_idx );
3070 : /* For overwrite commits, defer the refcnt decrement on
3071 : original_cache_line until after invalidation completes. If
3072 : we dropped refcnt to 0 here, a concurrent CLOCK sweep could
3073 : CAS(refcnt, 0, EVICT_SENTINEL) and steal the line before we
3074 : get to invalidate it, causing data corruption.
3075 : Non-overwrite and non-commit paths unpin
3076 : immediately because they never invalidate the original line. */
3077 198837 : if( FD_LIKELY( original_cache_line ) ) {
3078 : #if FD_TMPL_USE_HANDHOLDING
3079 : FD_TEST( original_cache_line->refcnt>0U );
3080 : #endif
3081 97170 : if( FD_LIKELY( !accs[ i ]._writable || !accs[ i ].commit || !accs[ i ]._overwrite ) ) {
3082 93633 : FD_ATOMIC_FETCH_AND_SUB( &original_cache_line->refcnt, 1U );
3083 93633 : }
3084 97170 : }
3085 :
3086 198837 : if( FD_LIKELY( !accs[ i ]._writable ) ) {
3087 : /* For readonly accounts, mark as recently used so the CLOCK
3088 : algorithm gives it a second chance before eviction. */
3089 : #if FD_TMPL_USE_HANDHOLDING
3090 : FD_TEST( original_cache_line );
3091 : #endif
3092 82509 : original_cache_line->referenced = 1;
3093 82509 : continue;
3094 82509 : }
3095 :
3096 116328 : fd_accdb_cache_line_t * destination_cache_lines[ FD_ACCDB_CACHE_CLASS_CNT ];
3097 1046952 : for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) destination_cache_lines[ j ] = cache_line( accdb, j, accs[ i ]._write.destination_cache_idx[ j ] );
3098 116328 : int destination_committed[ FD_ACCDB_CACHE_CLASS_CNT ] = {0};
3099 :
3100 116328 : if( FD_LIKELY( !accs[ i ].commit ) ) {
3101 : /* If it's writable but it didn't commit, all of the destination
3102 : cache lines (including the staging buffer which is trashed) are
3103 : unused and can be pushed to the CAS free list for immediate
3104 : reuse. Whatever buffer it was accessing also gets marked as
3105 : recently used. */
3106 699 : if( FD_LIKELY( original_cache_line ) ) original_cache_line->referenced = 1;
3107 6291 : for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) {
3108 : /* acquire_cache_line via CLOCK leaves line->acc_idx pointing
3109 : at the prior owner. cache_free_push consumers (CLOCK,
3110 : background_preevict) skip lines only when acc_idx==UINT_MAX
3111 : AND gen==UINT_MAX; if we leave the stale acc_idx, a future
3112 : CLOCK pick would call line 849/853 against the wrong acc
3113 : and corrupt its cache_idx/valid. */
3114 5592 : destination_cache_lines[ j ]->acc_idx = UINT_MAX;
3115 5592 : destination_cache_lines[ j ]->key.generation = UINT_MAX;
3116 5592 : destination_cache_lines[ j ]->persisted = 1;
3117 5592 : while( FD_UNLIKELY( FD_ATOMIC_CAS( &destination_cache_lines[ j ]->refcnt, 1U, 0U )!=1U ) ) {
3118 0 : fd_racesan_hook( "accdb_release:dest_refcnt_wait" );
3119 0 : FD_SPIN_PAUSE();
3120 0 : }
3121 5592 : cache_free_push( accdb, j, destination_cache_lines[ j ] );
3122 5592 : }
3123 699 : continue;
3124 699 : }
3125 :
3126 115629 : ulong new_size_class = fd_accdb_cache_class( accs[ i ].data_len );
3127 115629 : uint original_acc_idx = original_cache_line ? original_cache_line->acc_idx : UINT_MAX;
3128 115629 : fd_accdb_cache_line_t * committed_line;
3129 :
3130 : /* For overwrites, invalidate the on-disk offset BEFORE removing
3131 : the cache acc. This ensures a concurrent acquire that misses
3132 : the cache will see offset==FD_ACCDB_OFF_INVAL and spin-wait,
3133 : rather than reading stale on-disk bytes from the old location.
3134 : The CAS-loop exchange also serializes with a concurrent
3135 : compaction CAS (old_offset -> dest_offset). */
3136 115629 : ulong old_offset = FD_ACCDB_OFF_INVAL;
3137 115629 : if( FD_LIKELY( accs[ i ]._overwrite ) ) {
3138 3537 : fd_accdb_accmeta_t * ow_accmeta = &accdb->acc_pool[ original_acc_idx ];
3139 3537 : fd_racesan_hook( "accdb_overwrite:pre_xchg_offset" );
3140 3537 : old_offset = fd_accdb_acc_xchg_offset( ow_accmeta, FD_ACCDB_OFF_INVAL );
3141 3537 : if( FD_LIKELY( old_offset!=FD_ACCDB_OFF_INVAL ) ) {
3142 0 : fd_accdb_shmem_bytes_freed( accdb->shmem, old_offset, (ulong)FD_ACCDB_SIZE_DATA(ow_accmeta->executable_size)+sizeof(fd_accdb_disk_meta_t) );
3143 0 : FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->shmetrics->disk_used_bytes, (ulong)FD_ACCDB_SIZE_DATA(ow_accmeta->executable_size)+sizeof(fd_accdb_disk_meta_t) );
3144 0 : }
3145 3537 : }
3146 :
3147 115629 : if( FD_UNLIKELY( new_size_class==7UL ) ) {
3148 : /* The account belongs in the largest size class, and we already
3149 : have it resident in a 10MiB buffer anyway, so no need to copy
3150 : back. If we are "overwriting" (same generation as the account
3151 : came from), then the original can be discarded (pushed to
3152 : the CAS free list) and removed from the cache. */
3153 312 : destination_cache_lines[ 7UL ]->persisted = 0;
3154 312 : destination_committed[ 7UL ] = 1;
3155 312 : if( FD_LIKELY( accs[ i ]._overwrite ) ) {
3156 : /* Atomically clear acc.VALID and acc.cache_idx BEFORE freeing
3157 : the line, so a reader cannot observe acc.VALID=1 with
3158 : acc.cache_idx pointing at a line that has been recycled to
3159 : another acc. evict_clear_acc_cache_ref uses the CLAIM
3160 : protocol to serialize with cold_load_acc. */
3161 303 : evict_clear_acc_cache_ref( &accdb->acc_pool[ original_acc_idx ], original_size_class, accs[ i ]._original_cache_idx );
3162 :
3163 : /* Convert our own pin directly into the eviction claim
3164 : (CAS refcnt 1 -> EVICT_SENTINEL) so refcnt never passes
3165 : through 0 while acc_idx/key.generation are still valid:
3166 : in such a window background_preevict could claim and free
3167 : the line, and a claim of our own would then succeed on
3168 : the free-listed line and push it a second time. The CAS
3169 : fails if a concurrent reader that pinned the line via
3170 : cache_try_pin BEFORE evict_clear_acc_cache_ref completed
3171 : still holds a reference (its ABA check on
3172 : line->key.generation is not synchronized with our writes
3173 : to that field); then just drop our pin and leave
3174 : acc_idx/key.generation intact so CLOCK reclaims the line
3175 : once the reader unpins. That reclaim's
3176 : evict_clear_acc_cache_ref is a no-op, but its dirty-
3177 : writeback gate keys on persisted/acc_idx and would
3178 : republish the pre-overwrite bytes over the committed
3179 : version, so set persisted while still pinned. */
3180 303 : original_cache_line->persisted = 1;
3181 303 : fd_racesan_hook( "accdb_release:pre_discard_claim" );
3182 303 : if( FD_LIKELY( FD_ATOMIC_CAS( &original_cache_line->refcnt, 1U, FD_ACCDB_EVICT_SENTINEL )==1U ) ) {
3183 303 : original_cache_line->acc_idx = UINT_MAX;
3184 303 : original_cache_line->key.generation = UINT_MAX;
3185 303 : FD_COMPILER_MFENCE();
3186 303 : FD_VOLATILE( original_cache_line->refcnt ) = 0;
3187 303 : cache_free_push( accdb, original_size_class, original_cache_line );
3188 303 : } else {
3189 0 : FD_ATOMIC_FETCH_AND_SUB( &original_cache_line->refcnt, 1U );
3190 0 : }
3191 303 : }
3192 312 : committed_line = destination_cache_lines[ 7UL ];
3193 115317 : } else {
3194 : /* The account started in some arbitrary size class, transited
3195 : through a 10MiB staging buffer, and is now being written back
3196 : to some arbitrary (non-10MiB) size class, so we need to copy it
3197 : there. The staging buffer is discarded. If we are going to
3198 : a different size class, and we are "overwriting" (same
3199 : generation), then the original can also be discarded, but if
3200 : we are staying in the same size class, we can reuse the cache
3201 : line in place. */
3202 115317 : fd_accdb_cache_line_t * target_cache_line;
3203 115317 : if( FD_LIKELY( original_size_class==new_size_class ) ) {
3204 13944 : if( FD_LIKELY( accs[ i ]._overwrite ) ) {
3205 : /* a reader holding a stale acc->cache_idx from this line's
3206 : previous life may hold a transient cache_try_pin pin, so
3207 : our own pin only bounds refcnt from below. */
3208 3183 : uint ow_rc = FD_VOLATILE_CONST( original_cache_line->refcnt );
3209 3183 : FD_TEST( ow_rc>0U && ow_rc!=FD_ACCDB_EVICT_SENTINEL );
3210 3183 : original_cache_line->key.generation = UINT_MAX;
3211 : /* Keep refcnt>=1 through the reuse window so CLOCK cannot
3212 : steal the line between invalidation and re-publish. The
3213 : pin is released in the destination cleanup loop after
3214 : acc->cache_idx has been republished. */
3215 3183 : original_cache_line->acc_idx = UINT_MAX;
3216 3183 : target_cache_line = original_cache_line;
3217 10761 : } else {
3218 10761 : target_cache_line = destination_cache_lines[ new_size_class ];
3219 10761 : destination_committed[ new_size_class ] = 1;
3220 10761 : }
3221 101373 : } else {
3222 101373 : if( FD_LIKELY( accs[ i ]._overwrite ) ) {
3223 : /* Atomically clear acc.VALID and acc.cache_idx BEFORE freeing
3224 : the line, so a reader cannot observe acc.VALID=1 with
3225 : acc.cache_idx pointing at a line that has been recycled to
3226 : another acc. evict_clear_acc_cache_ref uses the CLAIM
3227 : protocol to serialize with cold_load_acc. See the
3228 : size_class==7 path above for the refcnt claim and
3229 : persisted rationale. */
3230 51 : evict_clear_acc_cache_ref( &accdb->acc_pool[ original_acc_idx ], original_size_class, accs[ i ]._original_cache_idx );
3231 51 : original_cache_line->persisted = 1;
3232 51 : fd_racesan_hook( "accdb_release:pre_discard_claim" );
3233 51 : if( FD_LIKELY( FD_ATOMIC_CAS( &original_cache_line->refcnt, 1U, FD_ACCDB_EVICT_SENTINEL )==1U ) ) {
3234 51 : original_cache_line->acc_idx = UINT_MAX;
3235 51 : original_cache_line->key.generation = UINT_MAX;
3236 51 : FD_COMPILER_MFENCE();
3237 51 : FD_VOLATILE( original_cache_line->refcnt ) = 0;
3238 51 : cache_free_push( accdb, original_size_class, original_cache_line );
3239 51 : } else {
3240 0 : FD_ATOMIC_FETCH_AND_SUB( &original_cache_line->refcnt, 1U );
3241 0 : }
3242 51 : }
3243 :
3244 101373 : destination_committed[ new_size_class ] = 1;
3245 101373 : target_cache_line = destination_cache_lines[ new_size_class ];
3246 101373 : }
3247 :
3248 115317 : target_cache_line->persisted = 0;
3249 : /* If target is the original cache line (overwrite, same size
3250 : class), mark as referenced directly since the cleanup loop
3251 : only handles destination lines. */
3252 115317 : if( FD_LIKELY( !destination_committed[ new_size_class ] ) ) target_cache_line->referenced = 1;
3253 115317 : committed_line = target_cache_line;
3254 115317 : }
3255 :
3256 : /* For non-overwrite commits, the original cache line (if any) still
3257 : holds valid ancestor data but is no longer pinned. Mark it as
3258 : recently used so the CLOCK algorithm retains it. */
3259 115629 : if( FD_UNLIKELY( !accs[ i ]._overwrite && original_cache_line ) ) {
3260 10806 : original_cache_line->referenced = 1;
3261 10806 : }
3262 :
3263 : /* Handle every destination cache line: committed ones keep
3264 : refcnt>=1 until acc->cache_idx is published (the deferred
3265 : unpin happens after the publish below), uncommitted ones are
3266 : fully freed to the CAS free list. */
3267 1040661 : for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) {
3268 925032 : if( destination_committed[ j ] ) {
3269 112446 : destination_cache_lines[ j ]->referenced = 1;
3270 812586 : } else {
3271 : /* See note above (no-commit path): clear stale acc_idx/gen
3272 : before pushing, otherwise CLOCK can pick this line and
3273 : stomp the prior owner's cache_idx/valid. CAS on refcnt for
3274 : the same stray-pin reason as the no-commit path. */
3275 812586 : destination_cache_lines[ j ]->acc_idx = UINT_MAX;
3276 812586 : destination_cache_lines[ j ]->key.generation = UINT_MAX;
3277 812586 : destination_cache_lines[ j ]->persisted = 1;
3278 812586 : while( FD_UNLIKELY( FD_ATOMIC_CAS( &destination_cache_lines[ j ]->refcnt, 1U, 0U )!=1U ) ) {
3279 0 : fd_racesan_hook( "accdb_release:dest_refcnt_wait" );
3280 0 : FD_SPIN_PAUSE();
3281 0 : }
3282 812586 : cache_free_push( accdb, j, destination_cache_lines[ j ] );
3283 812586 : }
3284 925032 : }
3285 :
3286 : /* Update the accounts index for this committed write. For an
3287 : overwrite (same fork+generation), update the existing acc
3288 : in place. Otherwise allocate a new acc, prepend it
3289 : to the hash chain, and record the write in a txn linked to
3290 : the fork so advance_root can clean up old versions. */
3291 115629 : if( FD_LIKELY( accs[ i ]._overwrite ) ) {
3292 3537 : accdb->metrics->accounts_committed_overwrite_per_class[ new_size_class ]++;
3293 3537 : committed_line->acc_idx = original_acc_idx;
3294 :
3295 3537 : fd_accdb_accmeta_t * accmeta = &accdb->acc_pool[ original_acc_idx ];
3296 : /* The offset was already atomically swapped to FD_ACCDB_OFF_INVAL
3297 : and bytes freed above, so just update the metadata and
3298 : re-publish the cache location. CAS-loop preserves CLAIM bit
3299 : (a concurrent evict_clear_acc_cache_ref or acc_unlink may
3300 : hold it) and clears VALID; a plain store would clobber CLAIM
3301 : and break those protocols. */
3302 3537 : uint pd = accs[ i ].pd_write ? FD_ACCDB_SIZE_PD_WRITE_BIT : 0U;
3303 3537 : for(;;) {
3304 3537 : uint cur = FD_VOLATILE_CONST( accmeta->executable_size );
3305 3537 : uint nxt = (cur & (FD_ACCDB_SIZE_CACHE_CLAIM_BIT|FD_ACCDB_SIZE_PD_WRITE_BIT))
3306 3537 : | FD_ACCDB_SIZE_PACK( (uint)accs[ i ].data_len, accs[ i ].executable )
3307 3537 : | pd;
3308 3537 : if( FD_LIKELY( FD_ATOMIC_CAS( &accmeta->executable_size, cur, nxt )==cur ) ) break;
3309 0 : FD_SPIN_PAUSE();
3310 0 : }
3311 3537 : accmeta->lamports = accs[ i ].lamports;
3312 3537 : fd_racesan_hook( "accdb_overwrite:mid_inplace" );
3313 :
3314 3537 : fd_memcpy( committed_line->owner, accs[ i ].owner, 32UL );
3315 3537 : fd_memcpy( committed_line->key.pubkey, accmeta->key.pubkey, 32UL );
3316 3537 : committed_line->key.generation = accmeta->key.generation;
3317 3537 : committed_line->acc_idx = original_acc_idx;
3318 3537 : FD_VOLATILE( accmeta->cache_idx ) = FD_ACCDB_ACC_CIDX_PACK( (uint)new_size_class, (uint)cache_line_idx( accdb, new_size_class, committed_line ) );
3319 : /* Atomic OR so a concurrent evict_clear_acc_cache_ref's CLAIM
3320 : clear (FETCH_AND_AND with ~CLAIM) cannot be lost by an RMW
3321 : race with a plain |= store. */
3322 3537 : FD_ATOMIC_FETCH_AND_OR( &accmeta->executable_size, FD_ACCDB_SIZE_CACHE_VALID_BIT );
3323 :
3324 : /* Now that acc->cache_idx is published, unpin so CLOCK can
3325 : eventually evict it. For same-size overwrites, committed_line
3326 : IS the reused original_cache_line. For cross-size overwrites,
3327 : committed_line is a destination line whose refcnt decrement was
3328 : deferred from the cleanup loop. */
3329 3537 : FD_ATOMIC_FETCH_AND_SUB( &committed_line->refcnt, 1U );
3330 3537 : committed_line->referenced = 1;
3331 112092 : } else {
3332 112092 : accdb->metrics->accounts_committed_new_per_class[ new_size_class ]++;
3333 112092 : fd_accdb_accmeta_t * accmeta = acc_pool_acquire( accdb->acc_pool_join );
3334 112092 : FD_TEST( accmeta );
3335 112092 : ulong acc_idx = acc_pool_idx( accdb->acc_pool_join, accmeta );
3336 112092 : fd_memcpy( accmeta->key.pubkey, accs[ i ].pubkey, 32UL );
3337 112092 : accmeta->lamports = accs[ i ].lamports;
3338 112092 : accmeta->executable_size = FD_ACCDB_SIZE_PACK( (uint)accs[ i ].data_len, accs[ i ].executable )
3339 112092 : | (accs[ i ].pd_write ? FD_ACCDB_SIZE_PD_WRITE_BIT : 0U);
3340 112092 : accmeta->key.generation = accs[ i ]._generation;
3341 112092 : accmeta->offset_fork = fd_accdb_acc_pack_offset_fork( FD_ACCDB_OFF_INVAL, accs[ i ]._fork_id );
3342 :
3343 : /* Publish in the cache BEFORE the acc_map head so that a
3344 : concurrent acquire that finds this acc in the hash chain will
3345 : also find a cache hit, rather than inserting a conflicting
3346 : placeholder cache acc. */
3347 112092 : committed_line->acc_idx = (uint)acc_idx;
3348 112092 : fd_memcpy( committed_line->owner, accs[ i ].owner, 32UL );
3349 112092 : fd_memcpy( committed_line->key.pubkey, accmeta->key.pubkey, 32UL );
3350 112092 : committed_line->key.generation = accmeta->key.generation;
3351 112092 : FD_VOLATILE( accmeta->cache_idx ) = FD_ACCDB_ACC_CIDX_PACK( (uint)new_size_class, (uint)cache_line_idx( accdb, new_size_class, committed_line ) );
3352 : /* Atomic OR so a concurrent evict_clear_acc_cache_ref's CLAIM
3353 : clear (FETCH_AND_AND with ~CLAIM) cannot be lost by an RMW
3354 : race with a plain |= store. */
3355 112092 : FD_ATOMIC_FETCH_AND_OR( &accmeta->executable_size, FD_ACCDB_SIZE_CACHE_VALID_BIT );
3356 :
3357 : /* Now that acc->cache_idx is published, unpin it so
3358 : CLOCK can eventually evict it. */
3359 112092 : FD_ATOMIC_FETCH_AND_SUB( &committed_line->refcnt, 1U );
3360 112092 : committed_line->referenced = 1;
3361 :
3362 : /* CAS loop to prepend to the hash chain. Succeeds on the first
3363 : try in most cases, but a concurrent acc_unlink CAS removing
3364 : the old head can change acc_map[idx] between our load and
3365 : CAS. Multiple concurrent releases may also race on the head
3366 : pointer — the CAS retry handles this. */
3367 112092 : for(;;) {
3368 112092 : uint old_head = FD_VOLATILE_CONST( accdb->acc_map[ accs[ i ]._acc_map_idx ] );
3369 112092 : accmeta->map.next = old_head;
3370 112092 : FD_COMPILER_MFENCE();
3371 112092 : fd_racesan_hook( "accdb_release:pre_chain_cas" );
3372 112092 : if( FD_LIKELY( FD_ATOMIC_CAS( &accdb->acc_map[ accs[ i ]._acc_map_idx ], old_head, (uint)acc_idx )==old_head ) ) break;
3373 0 : FD_SPIN_PAUSE();
3374 0 : }
3375 :
3376 : /* CONCURRENCY: The cache acc is published before the acc_map
3377 : head so that a concurrent fd_accdb_acquire reader that
3378 : observes the new head also finds a cache hit, preventing
3379 : duplicate cache insertion.
3380 :
3381 : (1) The CAS on acc_map[idx] serializes head-pointer mutations
3382 : from concurrent releases onto the same chain without any
3383 : external lock.
3384 :
3385 : (2) The FD_COMPILER_MFENCE above ensures stores to the acc node
3386 : fields (pubkey, lamports, size, generation, fork_id,
3387 : offset, map.next) are ordered before the CAS that publishes
3388 : the new head. On x86-64 (TSO), hardware also guarantees
3389 : this, but the compiler fence is needed to prevent the
3390 : compiler from reordering the stores. A reader that
3391 : observes the new head is guaranteed to see a fully
3392 : initialized node. A reader that has not yet seen the new
3393 : head simply traverses the previous (still valid) chain.
3394 :
3395 : (3) A concurrent acc_unlink (advance_root / purge) may CAS the
3396 : head away between our load and CAS here. The CAS retry
3397 : loop handles this. */
3398 :
3399 112092 : fd_accdb_txn_t * txn = txn_pool_acquire( accdb->txn_pool );
3400 112092 : FD_TEST( txn ); /* Sized so it always succeeds */
3401 112092 : txn->acc_pool_idx = (uint)acc_idx;
3402 112092 : uint txn_idx = (uint)txn_pool_idx( accdb->txn_pool, txn );
3403 112092 : for(;;) {
3404 112092 : uint old_head = FD_VOLATILE_CONST( accdb->fork_pool[ accs[ i ]._fork_id ].shmem->txn_head );
3405 112092 : txn->fork.next = old_head;
3406 112092 : if( FD_LIKELY( FD_ATOMIC_CAS( &accdb->fork_pool[ accs[ i ]._fork_id ].shmem->txn_head, old_head, txn_idx )==old_head ) ) break;
3407 0 : FD_SPIN_PAUSE();
3408 0 : }
3409 :
3410 112092 : FD_ATOMIC_FETCH_AND_ADD( &accdb->shmem->shmetrics->accounts_total, 1UL );
3411 112092 : }
3412 115629 : }
3413 :
3414 : // STEP 3.
3415 : // Finally, we release the cache class reservations we took at the
3416 : // beginning when we acquired these cache lines. Credits return
3417 : // directly to the shared pool so other threads can use them
3418 : // immediately.
3419 :
3420 311322 : ulong refund[ FD_ACCDB_CACHE_CLASS_CNT ] = {0};
3421 623490 : for( ulong i=0UL; i<accs_cnt; i++ ) {
3422 312168 : if( FD_LIKELY( accs[ i ]._original_size_class!=ULONG_MAX ) ) {
3423 97170 : if( FD_UNLIKELY( accdb->shmem->cache_class_used[ accs[ i ]._original_size_class ].val!=ULONG_MAX ) ) {
3424 3 : refund[ accs[ i ]._original_size_class ]++;
3425 3 : }
3426 97170 : }
3427 312168 : if( FD_UNLIKELY( accs[ i ]._writable ) ) {
3428 1046952 : for( ulong j=0UL; j<FD_ACCDB_CACHE_CLASS_CNT; j++ ) {
3429 930624 : if( FD_UNLIKELY( accdb->shmem->cache_class_used[ j ].val!=ULONG_MAX ) ) {
3430 54 : refund[ j ]++;
3431 54 : }
3432 930624 : }
3433 116328 : }
3434 312168 : }
3435 2801898 : for( ulong k=0UL; k<FD_ACCDB_CACHE_CLASS_CNT; k++ ) {
3436 2490576 : if( FD_UNLIKELY( refund[ k ] ) ) FD_ATOMIC_FETCH_AND_SUB( &accdb->shmem->cache_class_used[ k ].val, refund[ k ] );
3437 2490576 : }
3438 :
3439 311322 : FD_COMPILER_MFENCE();
3440 311322 : FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
3441 311322 : }
3442 :
3443 : void
3444 : fd_accdb_release( fd_accdb_t * accdb,
3445 : ulong accs_cnt,
3446 311052 : fd_acc_t * accs ) {
3447 311052 : FD_TEST( accdb->acquire_state==FD_ACCDB_ACQUIRE_STATE_OPEN );
3448 311052 : release_inner( accdb, accs_cnt, accs );
3449 311052 : accdb->acquire_state = FD_ACCDB_ACQUIRE_STATE_IDLE;
3450 311052 : }
3451 :
3452 : void
3453 : fd_accdb_release_ab( fd_accdb_t * accdb,
3454 : ulong accs_cnt,
3455 : fd_acc_t * accs,
3456 : ulong execs_cnt,
3457 234 : fd_acc_t * execs ) {
3458 234 : FD_TEST( accdb->acquire_state==FD_ACCDB_ACQUIRE_STATE_OPEN );
3459 234 : release_inner( accdb, accs_cnt, accs );
3460 234 : if( FD_LIKELY( execs_cnt ) ) release_inner( accdb, execs_cnt, execs );
3461 234 : accdb->acquire_state = FD_ACCDB_ACQUIRE_STATE_IDLE;
3462 234 : }
3463 :
3464 : fd_acc_t
3465 : fd_accdb_read_one( fd_accdb_t * accdb,
3466 : fd_accdb_fork_id_t fork_id,
3467 194028 : uchar const * pubkey ) {
3468 194028 : fd_acc_t acc;
3469 194028 : fd_accdb_acquire( accdb, fork_id, 1UL, &pubkey, (int[]){0}, &acc );
3470 194028 : return acc;
3471 194028 : }
3472 :
3473 : void
3474 : fd_accdb_unread_one( fd_accdb_t * accdb,
3475 194028 : fd_acc_t * acc ) {
3476 194028 : fd_accdb_release( accdb, 1UL, acc );
3477 194028 : }
3478 :
3479 : fd_acc_t
3480 : fd_accdb_write_one( fd_accdb_t * accdb,
3481 : fd_accdb_fork_id_t fork_id,
3482 114204 : uchar const * pubkey ) {
3483 114204 : fd_acc_t acc;
3484 114204 : fd_accdb_acquire( accdb, fork_id, 1UL, &pubkey, (int[]){1}, &acc );
3485 114204 : return acc;
3486 114204 : }
3487 :
3488 : void
3489 : fd_accdb_unwrite_one( fd_accdb_t * accdb,
3490 114204 : fd_acc_t * acc ) {
3491 114204 : fd_accdb_release( accdb, 1UL, acc );
3492 114204 : }
3493 :
3494 : int
3495 : fd_accdb_read_one_nocache( fd_accdb_t * accdb,
3496 : fd_accdb_fork_id_t fork_id,
3497 : uchar const * pubkey,
3498 : ulong * out_lamports,
3499 : int * out_executable,
3500 : uchar * out_owner,
3501 : uchar * out_data,
3502 0 : ulong * out_data_len ) {
3503 : /* Publish epoch — protects against compaction freeing the partition
3504 : under us during the preadv2 path. This is the only write the
3505 : readonly joiner makes into accdb shmem (and the pointer it stores
3506 : through is mapped through a separately-mmap'd writable page that
3507 : aliases shmem->joiner_epochs[idx]). */
3508 0 : FD_COMPILER_MFENCE();
3509 0 : FD_VOLATILE( *accdb->my_epoch_slot ) = FD_VOLATILE_CONST( accdb->shmem->epoch );
3510 0 : FD_HW_MFENCE();
3511 :
3512 : /// STEP 1.
3513 : /// Walk the hash chain at acc_map[hash(pubkey)] using the same
3514 : // visibility test as fd_accdb_acquire_inner. See that function
3515 : // for the detailed safety argument under concurrent prepend.
3516 0 : uint root_generation = accdb->fork_pool[ accdb->shmem->root_fork_id.val ].shmem->generation;
3517 0 : fd_accdb_fork_t * fork = &accdb->fork_pool[ fork_id.val ];
3518 0 : ulong hash = fd_hash32( pubkey, accdb->shmem->seed )&(accdb->shmem->chain_cnt-1UL);
3519 0 : uint acc_idx = FD_VOLATILE_CONST( accdb->acc_map[ hash ] );
3520 0 : fd_accdb_accmeta_t const * accmeta = NULL;
3521 0 : while( acc_idx!=UINT_MAX ) {
3522 0 : fd_accdb_accmeta_t const * candidate = &accdb->acc_pool[ acc_idx ];
3523 0 : uint next_idx = FD_VOLATILE_CONST( candidate->map.next );
3524 0 : if( FD_UNLIKELY( (candidate->key.generation>root_generation &&
3525 0 : fd_accdb_acc_fork_id(candidate)!=fork_id.val &&
3526 0 : !descends_set_test( fork->descends, fd_accdb_acc_fork_id(candidate) )) ) ||
3527 0 : memcmp( pubkey, candidate->key.pubkey, 32UL ) ) {
3528 0 : acc_idx = next_idx;
3529 0 : continue;
3530 0 : }
3531 0 : accmeta = candidate;
3532 0 : break;
3533 0 : }
3534 :
3535 0 : if( FD_UNLIKELY( !accmeta ) ) {
3536 0 : accdb->metrics->accounts_acquired_per_class[ 0 ]++;
3537 0 : *out_lamports = 0UL;
3538 0 : FD_COMPILER_MFENCE();
3539 0 : FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
3540 0 : return FD_ACCDB_READ_ONE_NOCACHE_MISS;
3541 0 : }
3542 :
3543 : /// STEP 2.
3544 : /// Snapshot acc fields. The acc element's metadata is effectively
3545 : /// immutable from the perspective of cross-fork readers (see the
3546 : /// comment block in fd_accdb.h about cross-fork reads). */
3547 0 : uint snap_es = FD_VOLATILE_CONST( accmeta->executable_size );
3548 0 : uint snap_gen = accmeta->key.generation;
3549 0 : ulong snap_lamports = accmeta->lamports;
3550 0 : uint snap_cidx = FD_VOLATILE_CONST( accmeta->cache_idx );
3551 0 : ulong data_len = (ulong)FD_ACCDB_SIZE_DATA( snap_es );
3552 0 : int executable = FD_ACCDB_SIZE_EXEC( snap_es );
3553 :
3554 0 : accdb->metrics->accounts_acquired_per_class[ fd_accdb_cache_class( data_len ) ]++;
3555 :
3556 0 : if( FD_UNLIKELY( !snap_lamports ) ) {
3557 0 : *out_lamports = 0UL;
3558 0 : FD_COMPILER_MFENCE();
3559 0 : FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
3560 0 : return FD_ACCDB_READ_ONE_NOCACHE_MISS;
3561 0 : }
3562 :
3563 : /// STEP 3.
3564 : /// Cache hit fast path with try-read-test (ABA) loop. Same
3565 : /// primitives as cache_try_pin: re-check key.generation + pubkey
3566 : /// before and after the bulk copy, and bail to the disk path if the
3567 : /// line was claimed for eviction (refcnt ==
3568 : /// FD_ACCDB_EVICT_SENTINEL). No CAS on refcnt, we never pin the
3569 : /// line.
3570 0 : if( FD_LIKELY( FD_ACCDB_SIZE_CACHE_VALID( snap_es ) && snap_cidx!=FD_ACCDB_ACC_CIDX_INVAL ) ) {
3571 0 : ulong cls = FD_ACCDB_ACC_CIDX_CLASS( snap_cidx );
3572 0 : ulong idx = FD_ACCDB_ACC_CIDX_IDX ( snap_cidx );
3573 0 : fd_accdb_cache_line_t * line = cache_line( accdb, cls, idx );
3574 :
3575 0 : for(;;) {
3576 0 : uint gen0 = FD_VOLATILE_CONST( line->key.generation );
3577 0 : uint rc0 = FD_VOLATILE_CONST( line->refcnt );
3578 0 : uint ai0 = FD_VOLATILE_CONST( line->acc_idx );
3579 0 : if( FD_UNLIKELY( rc0==FD_ACCDB_EVICT_SENTINEL ) ) goto miss;
3580 0 : if( FD_UNLIKELY( gen0!=snap_gen ) ) goto miss;
3581 0 : if( FD_UNLIKELY( memcmp( line->key.pubkey, pubkey, 32UL ) ) ) goto miss;
3582 : /* acc_idx==UINT_MAX is the "loading" sentinel set by cold_load_acc
3583 : before the preadv2 fills the line. CACHE_VALID can be observed
3584 : set while the bytes are still stale, so fall to the disk path
3585 : (which spins on offset_fork and reads from the file) rather
3586 : than copying garbage. */
3587 0 : if( FD_UNLIKELY( ai0==UINT_MAX ) ) goto miss;
3588 :
3589 0 : FD_COMPILER_MFENCE();
3590 0 : memcpy( out_owner, line->owner, 32UL );
3591 0 : memcpy( out_data, (uchar const *)(line+1UL), data_len );
3592 0 : FD_COMPILER_MFENCE();
3593 :
3594 0 : uint gen1 = FD_VOLATILE_CONST( line->key.generation );
3595 0 : uint rc1 = FD_VOLATILE_CONST( line->refcnt );
3596 0 : uint ai1 = FD_VOLATILE_CONST( line->acc_idx );
3597 0 : if( FD_UNLIKELY( rc1==FD_ACCDB_EVICT_SENTINEL ) ) goto miss;
3598 0 : if( FD_UNLIKELY( gen1!=snap_gen ) ) goto miss;
3599 0 : if( FD_UNLIKELY( memcmp( line->key.pubkey, pubkey, 32UL ) ) ) goto miss;
3600 0 : if( FD_UNLIKELY( ai1==UINT_MAX ) ) goto miss;
3601 :
3602 0 : *out_lamports = snap_lamports;
3603 0 : *out_executable = executable;
3604 0 : *out_data_len = data_len;
3605 0 : accdb->metrics->bytes_copied += data_len;
3606 0 : FD_COMPILER_MFENCE();
3607 0 : FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
3608 0 : return FD_ACCDB_READ_ONE_NOCACHE_CACHE;
3609 0 : }
3610 0 : }
3611 :
3612 0 : miss:;
3613 0 : accdb->metrics->accounts_not_found_per_class[ fd_accdb_cache_class( FD_ACCDB_SIZE_DATA( snap_es ) ) ]++;
3614 :
3615 : /// STEP 4.
3616 : /// Disk path. Spin until the writer publishes a real offset
3617 : /// (matches STEP 10 of fd_accdb_acquire_inner). Compaction may
3618 : /// concurrently relocate the record, but our published epoch
3619 : /// prevents the source partition from being freed until we exit
3620 : /// our critical section, so the bytes at the snapshotted offset
3621 : /// remain stable for the duration of the read.
3622 0 : fd_racesan_hook( "accdb_nocache:pre_offset" );
3623 0 : ulong off_packed = FD_VOLATILE_CONST( accmeta->offset_fork );
3624 0 : if( FD_UNLIKELY( (off_packed & FD_ACCDB_OFF_MASK)==FD_ACCDB_OFF_INVAL ) ) {
3625 0 : accdb->metrics->accounts_waited++;
3626 0 : while( FD_UNLIKELY( ((off_packed=FD_VOLATILE_CONST( accmeta->offset_fork )) & FD_ACCDB_OFF_MASK)==FD_ACCDB_OFF_INVAL ) ) FD_SPIN_PAUSE();
3627 0 : }
3628 0 : ulong off = off_packed & FD_ACCDB_OFF_MASK;
3629 0 : fd_racesan_hook( "accdb_nocache:pre_preadv2" );
3630 :
3631 0 : struct iovec iovs[ 2 ] = {
3632 0 : { .iov_base = out_owner, .iov_len = 32UL },
3633 0 : { .iov_base = out_data, .iov_len = data_len },
3634 0 : };
3635 0 : ulong total = 32UL+data_len;
3636 0 : ulong start = off+offsetof( fd_accdb_disk_meta_t, owner );
3637 0 : ulong got = 0UL;
3638 0 : int nio = data_len ? 2 : 1;
3639 0 : while( FD_LIKELY( got<total ) ) {
3640 0 : long result = preadv2( accdb->fd, iovs, nio, (long)(start+got), 0 );
3641 0 : if( FD_UNLIKELY( -1==result && (errno==EINTR || errno==EAGAIN || errno==EWOULDBLOCK) ) ) continue;
3642 0 : else if( FD_UNLIKELY( -1==result ) ) FD_LOG_ERR(( "preadv2() failed (%d-%s)", errno, fd_io_strerror( errno ) ));
3643 0 : else if( FD_UNLIKELY( !result ) ) FD_LOG_ERR(( "accounts database is corrupt, data expected at offset %lu with size %lu exceeded file extents", start+got, total ));
3644 0 : fd_accdb_partition_read_bump( accdb, start+got, (ulong)result );
3645 0 : got += (ulong)result;
3646 0 : accdb->metrics->bytes_read += (ulong)result;
3647 0 : accdb->metrics->read_ops++;
3648 :
3649 0 : long r = result;
3650 0 : for( int v=0; v<nio; v++ ) {
3651 0 : if( (ulong)r>=iovs[ v ].iov_len ) {
3652 0 : r -= (long)iovs[ v ].iov_len;
3653 0 : iovs[ v ].iov_len = 0UL;
3654 0 : } else {
3655 0 : iovs[ v ].iov_base = (uchar *)iovs[ v ].iov_base + r;
3656 0 : iovs[ v ].iov_len -= (ulong)r;
3657 0 : break;
3658 0 : }
3659 0 : }
3660 0 : }
3661 :
3662 0 : *out_lamports = snap_lamports;
3663 0 : *out_executable = executable;
3664 0 : *out_data_len = data_len;
3665 :
3666 0 : FD_COMPILER_MFENCE();
3667 0 : FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
3668 0 : return FD_ACCDB_READ_ONE_NOCACHE_DISK;
3669 0 : }
3670 :
3671 : int
3672 : fd_accdb_exists( fd_accdb_t * accdb,
3673 : fd_accdb_fork_id_t fork_id,
3674 111 : uchar const * pubkey ) {
3675 111 : FD_COMPILER_MFENCE();
3676 111 : FD_VOLATILE( *accdb->my_epoch_slot ) = FD_VOLATILE_CONST( accdb->shmem->epoch );
3677 111 : FD_HW_MFENCE();
3678 :
3679 111 : uint root_generation = accdb->fork_pool[ accdb->shmem->root_fork_id.val ].shmem->generation;
3680 111 : fd_accdb_fork_t * fork = &accdb->fork_pool[ fork_id.val ];
3681 111 : ulong hash = fd_hash32( pubkey, accdb->shmem->seed )&(accdb->shmem->chain_cnt-1UL);
3682 111 : uint acc = FD_VOLATILE_CONST( accdb->acc_map[ hash ] );
3683 180 : while( acc!=UINT_MAX ) {
3684 174 : fd_accdb_accmeta_t const * candidate_acc = &accdb->acc_pool[ acc ];
3685 174 : uint next_acc = FD_VOLATILE_CONST( candidate_acc->map.next );
3686 :
3687 174 : if( FD_UNLIKELY( (candidate_acc->key.generation>root_generation && fd_accdb_acc_fork_id(candidate_acc)!=fork_id.val && !descends_set_test( fork->descends, fd_accdb_acc_fork_id(candidate_acc) )) ) || memcmp( pubkey, candidate_acc->key.pubkey, 32UL ) ) {
3688 69 : acc = next_acc;
3689 69 : continue;
3690 69 : }
3691 :
3692 105 : break;
3693 174 : }
3694 :
3695 111 : int result;
3696 111 : if( FD_UNLIKELY( acc==UINT_MAX ) ) result = 0;
3697 105 : else result = !!FD_VOLATILE_CONST( accdb->acc_pool[ acc ].lamports );
3698 :
3699 111 : FD_COMPILER_MFENCE();
3700 111 : FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
3701 111 : return result;
3702 111 : }
3703 :
3704 : int
3705 : fd_accdb_probe_pd_this_fork( fd_accdb_t * accdb,
3706 : fd_accdb_fork_id_t fork_id,
3707 : uchar const * pubkey,
3708 : int * out_pd_write,
3709 : ulong * out_data_len,
3710 114 : ulong * out_lamports ) {
3711 114 : FD_COMPILER_MFENCE();
3712 114 : FD_VOLATILE( *accdb->my_epoch_slot ) = FD_VOLATILE_CONST( accdb->shmem->epoch );
3713 114 : FD_HW_MFENCE();
3714 :
3715 114 : uint root_generation = accdb->fork_pool[ accdb->shmem->root_fork_id.val ].shmem->generation;
3716 114 : fd_accdb_fork_t * fork = &accdb->fork_pool[ fork_id.val ];
3717 114 : ulong hash = fd_hash32( pubkey, accdb->shmem->seed )&(accdb->shmem->chain_cnt-1UL);
3718 114 : uint acc = FD_VOLATILE_CONST( accdb->acc_map[ hash ] );
3719 114 : while( acc!=UINT_MAX ) {
3720 111 : fd_accdb_accmeta_t const * candidate_acc = &accdb->acc_pool[ acc ];
3721 111 : uint next_acc = FD_VOLATILE_CONST( candidate_acc->map.next );
3722 :
3723 111 : if( FD_UNLIKELY( (candidate_acc->key.generation>root_generation && fd_accdb_acc_fork_id(candidate_acc)!=fork_id.val && !descends_set_test( fork->descends, fd_accdb_acc_fork_id(candidate_acc) )) ) || memcmp( pubkey, candidate_acc->key.pubkey, 32UL ) ) {
3724 0 : acc = next_acc;
3725 0 : continue;
3726 0 : }
3727 :
3728 111 : break;
3729 111 : }
3730 :
3731 114 : int pd = 0;
3732 114 : int gen_match = 0;
3733 114 : ulong len = 0UL;
3734 114 : ulong lamports = 0UL;
3735 114 : if( FD_LIKELY( acc!=UINT_MAX ) ) {
3736 111 : fd_accdb_accmeta_t const * m = &accdb->acc_pool[ acc ];
3737 111 : uint es = FD_VOLATILE_CONST( m->executable_size );
3738 111 : gen_match = ( m->key.generation==fork->shmem->generation );
3739 111 : pd = gen_match && FD_ACCDB_SIZE_PD_WRITE( es );
3740 111 : len = FD_ACCDB_SIZE_DATA( es );
3741 111 : lamports = FD_VOLATILE_CONST( m->lamports );
3742 111 : }
3743 :
3744 114 : FD_COMPILER_MFENCE();
3745 114 : FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
3746 :
3747 114 : *out_pd_write = pd;
3748 114 : if( gen_match ) {
3749 99 : *out_data_len = len;
3750 99 : *out_lamports = lamports;
3751 99 : }
3752 114 : return gen_match;
3753 114 : }
3754 :
3755 : ulong
3756 : fd_accdb_lamports( fd_accdb_t * accdb,
3757 : fd_accdb_fork_id_t fork_id,
3758 8754 : uchar const * pubkey ) {
3759 8754 : FD_COMPILER_MFENCE();
3760 8754 : FD_VOLATILE( *accdb->my_epoch_slot ) = FD_VOLATILE_CONST( accdb->shmem->epoch );
3761 8754 : FD_HW_MFENCE();
3762 :
3763 8754 : uint root_generation = accdb->fork_pool[ accdb->shmem->root_fork_id.val ].shmem->generation;
3764 8754 : fd_accdb_fork_t * fork = &accdb->fork_pool[ fork_id.val ];
3765 8754 : ulong hash = fd_hash32( pubkey, accdb->shmem->seed )&(accdb->shmem->chain_cnt-1UL);
3766 8754 : uint acc = FD_VOLATILE_CONST( accdb->acc_map[ hash ] );
3767 10092 : while( acc!=UINT_MAX ) {
3768 2313 : fd_accdb_accmeta_t const * candidate_acc = &accdb->acc_pool[ acc ];
3769 2313 : uint next_acc = FD_VOLATILE_CONST( candidate_acc->map.next );
3770 :
3771 2313 : if( FD_UNLIKELY( (candidate_acc->key.generation>root_generation && fd_accdb_acc_fork_id(candidate_acc)!=fork_id.val && !descends_set_test( fork->descends, fd_accdb_acc_fork_id(candidate_acc) )) ) || memcmp( pubkey, candidate_acc->key.pubkey, 32UL ) ) {
3772 1338 : acc = next_acc;
3773 1338 : continue;
3774 1338 : }
3775 :
3776 975 : break;
3777 2313 : }
3778 :
3779 8754 : ulong result;
3780 8754 : if( FD_UNLIKELY( acc==UINT_MAX ) ) result = 0UL;
3781 975 : else result = FD_VOLATILE_CONST( accdb->acc_pool[ acc ].lamports );
3782 :
3783 8754 : FD_COMPILER_MFENCE();
3784 8754 : FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
3785 8754 : return result;
3786 8754 : }
3787 :
3788 : /* cache_bg_evict pre-evicts cache lines in the background to keep the
3789 : per-class CAS free lists populated ahead of demand. For each class
3790 : whose immediately available capacity has dropped below low_water,
3791 : a bounded CLOCK sweep claims lines, writes dirty ones to disk, and
3792 : pushes them onto the free list until available capacity reaches
3793 : target. Immediately available capacity includes both the CAS free
3794 : list and the never-initialized tail of the class, since foreground
3795 : allocators can consume either path without evicting resident data.
3796 :
3797 : Budget: at most 256 CLOCK ticks per class per invocation to keep the
3798 : background loop responsive. The function is called every tick of
3799 : fd_accdb_background, so large refills happen across several ticks
3800 : rather than blocking. The low_water / target thresholds are static
3801 : per-class watermarks computed at initialization; pre-eviction only
3802 : converts resident lines into free-list entries and does not consume
3803 : cache-slot reservations.
3804 :
3805 : force: when non-zero, ignore the watermark and sweep every line in
3806 : every class. Always 0 in normal operation; used only by
3807 : test_accdb_racesan to deterministically exercise the writeback path
3808 : without manufacturing real cache pressure. */
3809 :
3810 : static void
3811 : background_preevict( fd_accdb_t * accdb,
3812 : int * charge_busy,
3813 3268415 : int force ) {
3814 3268415 : fd_accdb_shmem_t * shmem = accdb->shmem;
3815 :
3816 29415735 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) {
3817 26147320 : ulong target = shmem->cache_free_target[ c ];
3818 26147320 : ulong max_c = shmem->cache_class_max[ c ];
3819 26147320 : ulong init = fd_ulong_min( FD_VOLATILE_CONST( shmem->cache_class_init[ c ].val ), max_c );
3820 26147320 : ulong freec = FD_VOLATILE_CONST( shmem->cache_free_cnt[ c ].val );
3821 26147320 : ulong live = init>freec ? init-freec : 0UL;
3822 26147320 : ulong avail = max_c-live;
3823 26147320 : if( FD_LIKELY( !force && avail>=shmem->cache_free_low_water[ c ] ) ) continue;
3824 :
3825 0 : *charge_busy = 1;
3826 :
3827 0 : ulong budget = force ? init : 256UL;
3828 0 : ulong evicted = 0UL;
3829 0 : if( FD_UNLIKELY( force ) ) target = max_c; /* sweep everything */
3830 :
3831 0 : for( ulong tick=0UL; tick<budget && avail+evicted<target; tick++ ) {
3832 : /* Only sweep the lazily initialized prefix. cache_class_init
3833 : may transiently exceed max_c during the acquire_cache_line
3834 : overflow/undo path, so clamp it before using it as the wrap
3835 : bound. */
3836 0 : init = fd_ulong_min( FD_VOLATILE_CONST( shmem->cache_class_init[ c ].val ), max_c );
3837 0 : if( FD_UNLIKELY( !init ) ) break;
3838 :
3839 0 : ulong hand = FD_ATOMIC_FETCH_AND_ADD( &shmem->clock_hand[ c ].val, 1UL ) % init;
3840 :
3841 0 : fd_accdb_cache_line_t * line = cache_line( accdb, c, hand );
3842 :
3843 0 : if( FD_UNLIKELY( line->key.generation==UINT_MAX && line->acc_idx==UINT_MAX ) ) continue;
3844 :
3845 0 : uint rc = FD_VOLATILE_CONST( line->refcnt );
3846 0 : if( FD_UNLIKELY( rc ) ) continue;
3847 :
3848 0 : if( FD_UNLIKELY( line->referenced ) ) {
3849 0 : line->referenced = 0;
3850 0 : continue;
3851 0 : }
3852 :
3853 0 : if( FD_UNLIKELY( FD_ATOMIC_CAS( &line->refcnt, 0U, FD_ACCDB_EVICT_SENTINEL )!=0U ) ) continue;
3854 :
3855 0 : if( FD_UNLIKELY( line->acc_idx==UINT_MAX && line->key.generation==UINT_MAX ) ) {
3856 0 : FD_VOLATILE( line->refcnt ) = 0;
3857 0 : continue;
3858 0 : }
3859 :
3860 0 : uint acc_idx = line->acc_idx;
3861 : #if FD_TMPL_USE_HANDHOLDING
3862 : uint line_gen FD_FN_UNUSED = line->key.generation;
3863 : #endif
3864 0 : if( FD_LIKELY( acc_idx!=UINT_MAX ) ) {
3865 0 : evict_clear_acc_cache_ref( &accdb->acc_pool[ acc_idx ], c, hand );
3866 0 : }
3867 0 : line->key.generation = UINT_MAX;
3868 0 : if( FD_UNLIKELY( !line->persisted && acc_idx!=UINT_MAX ) ) {
3869 0 : fd_accdb_accmeta_t * accmeta = &accdb->acc_pool[ acc_idx ];
3870 :
3871 : /* advertise to external observers that write is in progress */
3872 0 : FD_COMPILER_MFENCE();
3873 0 : FD_VOLATILE( *accdb->my_epoch_slot ) = FD_VOLATILE_CONST( accdb->shmem->epoch );
3874 0 : FD_HW_MFENCE();
3875 :
3876 0 : fd_racesan_hook( "preevict:pre_synth" );
3877 : #if FD_TMPL_USE_HANDHOLDING
3878 : FD_TEST( line_gen==accmeta->key.generation &&
3879 : !memcmp( line->key.pubkey, accmeta->key.pubkey, 32UL ) );
3880 : #endif
3881 0 : ulong entry_sz = sizeof(fd_accdb_disk_meta_t)+(ulong)FD_ACCDB_SIZE_DATA( accmeta->executable_size );
3882 :
3883 : /* Atomically swap the old offset to FD_ACCDB_OFF_INVAL so that
3884 : a concurrent compaction CAS (old_offset -> dest_offset)
3885 : cannot succeed between our read and our later store of
3886 : the new file_off. Without the exchange, compaction could
3887 : relocate the record, then our plain store would overwrite
3888 : the relocated offset, leaving the compaction destination
3889 : as unreachable dead space whose bytes are never freed. */
3890 0 : ulong old_offset = fd_accdb_acc_xchg_offset( accmeta, FD_ACCDB_OFF_INVAL );
3891 0 : if( FD_LIKELY( old_offset!=FD_ACCDB_OFF_INVAL ) ) {
3892 0 : fd_accdb_shmem_bytes_freed( shmem, old_offset, entry_sz );
3893 0 : FD_ATOMIC_FETCH_AND_SUB( &shmem->shmetrics->disk_used_bytes, entry_sz );
3894 0 : }
3895 :
3896 0 : fd_accdb_disk_meta_t meta;
3897 0 : fd_memcpy( meta.pubkey, accmeta->key.pubkey, 32UL );
3898 0 : meta.size = FD_ACCDB_SIZE_DATA( accmeta->executable_size );
3899 0 : meta.generation = accmeta->key.generation;
3900 0 : fd_memcpy( meta.owner, line->owner, 32UL );
3901 :
3902 0 : struct iovec iovs[ 2UL ] = {
3903 0 : { .iov_base = &meta, .iov_len = sizeof(fd_accdb_disk_meta_t) },
3904 0 : { .iov_base = (void *)(line+1UL), .iov_len = FD_ACCDB_SIZE_DATA( accmeta->executable_size ) }
3905 0 : };
3906 :
3907 0 : ulong file_off = allocate_next_write( accdb, entry_sz );
3908 0 : ulong written = 0UL;
3909 0 : while( written<entry_sz ) {
3910 0 : long result = pwritev2( accdb->fd, iovs, 2, (long)(file_off+written), 0 );
3911 0 : if( FD_UNLIKELY( result==-1 && errno==EINTR ) ) continue;
3912 0 : else if( FD_UNLIKELY( result<=0 ) ) FD_LOG_ERR(( "pwritev2() failed (%d-%s)", errno, fd_io_strerror( errno ) ));
3913 0 : written += (ulong)result;
3914 0 : accdb->metrics->bytes_written += (ulong)result;
3915 0 : accdb->metrics->write_ops++;
3916 :
3917 0 : for( int v=0; v<2; v++ ) {
3918 0 : if( (ulong)result>=iovs[ v ].iov_len ) {
3919 0 : result -= (long)iovs[ v ].iov_len;
3920 0 : iovs[ v ].iov_len = 0UL;
3921 0 : } else {
3922 0 : iovs[ v ].iov_base = (uchar *)iovs[ v ].iov_base + result;
3923 0 : iovs[ v ].iov_len -= (ulong)result;
3924 0 : break;
3925 0 : }
3926 0 : }
3927 0 : }
3928 :
3929 0 : FD_COMPILER_MFENCE();
3930 0 : accmeta->offset_fork = fd_accdb_acc_pack_offset_fork( file_off, fd_accdb_acc_fork_id(accmeta) );
3931 0 : FD_ATOMIC_FETCH_AND_ADD( &shmem->shmetrics->disk_used_bytes, entry_sz );
3932 :
3933 0 : accdb->metrics->accounts_preevicted++;
3934 0 : accdb->metrics->accounts_preevicted_per_class[ c ]++;
3935 :
3936 0 : FD_COMPILER_MFENCE();
3937 0 : FD_VOLATILE( *accdb->my_epoch_slot ) = ULONG_MAX;
3938 0 : }
3939 :
3940 0 : line->persisted = 1;
3941 0 : line->acc_idx = UINT_MAX;
3942 0 : line->key.generation = UINT_MAX;
3943 0 : FD_COMPILER_MFENCE();
3944 0 : FD_VOLATILE( line->refcnt ) = 0;
3945 0 : cache_free_push( accdb, c, line );
3946 0 : evicted++;
3947 0 : }
3948 0 : }
3949 3268415 : }
3950 :
3951 : int
3952 : fd_accdb_snapshot_write_one( fd_accdb_t * accdb,
3953 : fd_accdb_fork_id_t fork_id,
3954 : uchar const * pubkey,
3955 : ulong slot,
3956 : ulong lamports,
3957 : ulong data_len,
3958 : int executable,
3959 69 : ulong * out_replaced_lamports ) {
3960 : /* Snapshot slots are stored in the 32-bit cache_idx scratch field
3961 : during loading. Reject anything that would truncate. */
3962 69 : if( FD_UNLIKELY( slot>UINT_MAX ) ) FD_LOG_ERR(( "snapshot slot %lu exceeds 2^32-1, accdb format must be widened", slot ));
3963 :
3964 69 : int incremental = fork_id.val!=USHORT_MAX;
3965 :
3966 69 : fd_accdb_fork_t * fork = NULL;
3967 69 : uint fork_gen = 0U;
3968 69 : if( FD_UNLIKELY( incremental ) ) {
3969 33 : fork = &accdb->fork_pool[ fork_id.val ];
3970 33 : fork_gen = fork->shmem->generation;
3971 33 : }
3972 :
3973 69 : ulong hash = fd_hash32( pubkey, accdb->shmem->seed )&(accdb->shmem->chain_cnt-1UL);
3974 :
3975 69 : *out_replaced_lamports = 0UL;
3976 :
3977 69 : fd_accdb_accmeta_t * accmeta = NULL;
3978 69 : int cross_fork = 0; /* incremental only: existing entry from different fork */
3979 :
3980 69 : ulong next_acc = accdb->acc_map[ hash ];
3981 75 : while( next_acc!=UINT_MAX ) {
3982 12 : fd_accdb_accmeta_t * candidate_acc = &accdb->acc_pool[ next_acc ];
3983 12 : if( FD_UNLIKELY( !memcmp( pubkey, candidate_acc->key.pubkey, 32UL ) ) ) {
3984 6 : if( FD_LIKELY( (ulong)candidate_acc->cache_idx>slot ) ) {
3985 : /* Still advance the write head so snapwr and snapin stay in
3986 : sync — snapwr unconditionally writes every account to disk.
3987 : Mark the space as immediately freed since it is dead on
3988 : arrival. */
3989 0 : ulong dead_sz = sizeof(fd_accdb_disk_meta_t)+data_len;
3990 0 : ulong dead_off = allocate_next_write( accdb, dead_sz );
3991 0 : fd_accdb_shmem_bytes_freed( accdb->shmem, dead_off, dead_sz );
3992 0 : return -1;
3993 0 : }
3994 6 : if( FD_UNLIKELY( incremental ) && candidate_acc->key.generation!=fork_gen ) {
3995 : /* Cross-snapshot override: don't replace in-place; insert a
3996 : new entry alongside the old one so purge can revert. */
3997 6 : cross_fork = 1;
3998 6 : *out_replaced_lamports = candidate_acc->lamports;
3999 6 : } else {
4000 : /* Same-fork duplicate (or full-snapshot mode): replace in-place */
4001 0 : accmeta = candidate_acc;
4002 0 : }
4003 6 : break;
4004 6 : }
4005 6 : next_acc = candidate_acc->map.next;
4006 6 : }
4007 :
4008 69 : int replace = !!accmeta;
4009 :
4010 69 : if( FD_UNLIKELY( !accmeta ) ) {
4011 69 : accmeta = acc_pool_acquire_nolock( accdb->acc_pool_join );
4012 69 : if( FD_UNLIKELY( !accmeta ) ) FD_LOG_ERR(( "accounts database ran out of space during snapshot loading, increase [accounts.max_accounts], current value is %lu", acc_pool_ele_max( accdb->acc_pool_join ) ));
4013 :
4014 69 : uint acc_idx = (uint)acc_pool_idx( accdb->acc_pool_join, accmeta );
4015 :
4016 69 : fd_memcpy( accmeta->key.pubkey, pubkey, 32UL );
4017 69 : if( FD_UNLIKELY( !incremental && accdb->shmem->root_fork_id.val==USHORT_MAX ) ) {
4018 0 : FD_LOG_ERR(( "snapshot_write_one called without a root fork attached" ));
4019 0 : }
4020 69 : accmeta->key.generation = incremental ? fork_gen : accdb->fork_pool[ accdb->shmem->root_fork_id.val ].shmem->generation;
4021 69 : accmeta->map.next = accdb->acc_map[ hash ];
4022 69 : accdb->acc_map[ hash ] = acc_idx;
4023 :
4024 : /* In incremental mode, record this insert in the fork's txn list
4025 : so purge can find and unlink it on failure. */
4026 69 : if( FD_UNLIKELY( incremental ) ) {
4027 33 : fd_accdb_txn_t * txn = txn_pool_acquire( accdb->txn_pool );
4028 33 : if( FD_UNLIKELY( !txn ) ) FD_LOG_ERR(( "txn pool exhausted during incremental snapshot loading" ));
4029 33 : txn->acc_pool_idx = acc_idx;
4030 33 : uint txn_idx = (uint)txn_pool_idx( accdb->txn_pool, txn );
4031 33 : txn->fork.next = fork->shmem->txn_head;
4032 33 : fork->shmem->txn_head = txn_idx;
4033 33 : }
4034 69 : }
4035 :
4036 69 : if( FD_UNLIKELY( replace ) ) {
4037 : /* The old version's disk space is now dead. */
4038 0 : ulong old_sz = sizeof(fd_accdb_disk_meta_t) + FD_ACCDB_SIZE_DATA( accmeta->executable_size );
4039 0 : fd_accdb_shmem_bytes_freed( accdb->shmem, fd_accdb_acc_offset( accmeta ), old_sz );
4040 0 : accdb->shmem->shmetrics->disk_used_bytes -= old_sz;
4041 0 : *out_replaced_lamports = accmeta->lamports;
4042 0 : }
4043 :
4044 69 : accmeta->cache_idx = (uint)slot;
4045 69 : accmeta->lamports = lamports;
4046 69 : accmeta->executable_size = FD_ACCDB_SIZE_PACK( (uint)data_len, executable );
4047 69 : ulong entry_sz = sizeof(fd_accdb_disk_meta_t)+data_len;
4048 69 : ulong file_off = allocate_next_write( accdb, entry_sz );
4049 69 : accmeta->offset_fork = incremental ? fd_accdb_acc_pack_offset_fork( file_off, fork_id.val ) : file_off;
4050 69 : accdb->shmem->shmetrics->disk_used_bytes += entry_sz;
4051 69 : if( !replace ) accdb->shmem->shmetrics->accounts_total++;
4052 :
4053 69 : return ( replace || cross_fork ) ? 2 : 1;
4054 69 : }
4055 :
4056 : int
4057 : fd_accdb_snapshot_write_batch( fd_accdb_t * accdb,
4058 : fd_accdb_fork_id_t fork_id,
4059 : ulong cnt,
4060 : uchar const * const pubkeys[],
4061 : ulong const slots[],
4062 : ulong const lamports[],
4063 : ulong const data_lens[],
4064 : int const executables[],
4065 : ulong * accounts_ignored,
4066 : ulong * accounts_replaced,
4067 : ulong * accounts_loaded,
4068 : ulong * out_replaced_lamports,
4069 12 : ulong * out_ignored_lamports ) {
4070 12 : int incremental = fork_id.val!=USHORT_MAX;
4071 :
4072 12 : fd_accdb_fork_t * fork = NULL;
4073 12 : uint fork_gen = 0U;
4074 12 : if( FD_UNLIKELY( incremental ) ) {
4075 3 : fork = &accdb->fork_pool[ fork_id.val ];
4076 3 : fork_gen = fork->shmem->generation;
4077 3 : }
4078 :
4079 12 : ulong seed = accdb->shmem->seed;
4080 12 : ulong chain_msk = accdb->shmem->chain_cnt - 1UL;
4081 12 : if( FD_UNLIKELY( !incremental && accdb->shmem->root_fork_id.val==USHORT_MAX ) ) {
4082 0 : FD_LOG_ERR(( "snapshot_write_batch called without a root fork attached" ));
4083 0 : }
4084 12 : uint gen = incremental ? 0U : accdb->fork_pool[ accdb->shmem->root_fork_id.val ].shmem->generation;
4085 :
4086 12 : ulong ignored = 0UL;
4087 12 : ulong replaced = 0UL;
4088 12 : ulong loaded = 0UL;
4089 12 : ulong cross_replaced = 0UL; /* cross-fork overrides (subset of replaced) */
4090 12 : ulong replaced_lamports = 0UL;
4091 12 : ulong ignored_lamports = 0UL;
4092 :
4093 : /* Snapshot slots are stored in the 32-bit cache_idx scratch field
4094 : during loading. Reject anything that would truncate. */
4095 42 : for( ulong i=0UL; i<cnt; i++ ) {
4096 30 : if( FD_UNLIKELY( slots[ i ]>UINT_MAX ) ) FD_LOG_ERR(( "snapshot slot %lu exceeds 2^32-1, accdb format must be widened", slots[ i ] ));
4097 30 : }
4098 :
4099 : /* Phase 1: compute hashes and prefetch chain heads. */
4100 :
4101 12 : ulong hashes[ 8 ];
4102 12 : fd_accdb_accmeta_t * existing[ 8 ]; /* same-fork dup or full-snapshot replace */
4103 12 : fd_accdb_accmeta_t * cross_existing[ 8 ]; /* cross-fork dup (incremental only) */
4104 12 : int skip[ 8 ];
4105 :
4106 42 : for( ulong i=0UL; i<cnt; i++ ) {
4107 30 : hashes[ i ] = fd_hash32( pubkeys[ i ], seed ) & chain_msk;
4108 30 : existing[ i ] = NULL;
4109 30 : cross_existing[ i ] = NULL;
4110 30 : skip[ i ] = 0;
4111 :
4112 : /* Prefetch the chain head and first pool element on the chain */
4113 30 : __builtin_prefetch( &accdb->acc_map[ hashes[ i ] ], 1, 1 );
4114 30 : }
4115 :
4116 : /* Phase 2: walk chains looking for duplicates. By now the chain
4117 : heads prefetched above should be warm in L1/L2. If the existing
4118 : entry has a higher slot, mark skip. Otherwise, save the existing
4119 : entry pointer for in-place update (matching write_one semantics).
4120 : In incremental mode, cross-fork entries are saved separately so
4121 : they can be left in place while a new entry is inserted. */
4122 :
4123 42 : for( ulong i=0UL; i<cnt; i++ ) {
4124 30 : ulong next_acc = accdb->acc_map[ hashes[ i ] ];
4125 :
4126 30 : if( FD_LIKELY( next_acc!=UINT_MAX ) ) {
4127 9 : __builtin_prefetch( &accdb->acc_pool[ next_acc ], 0, 1 );
4128 9 : }
4129 :
4130 30 : while( next_acc!=UINT_MAX ) {
4131 9 : fd_accdb_accmeta_t * candidate = &accdb->acc_pool[ next_acc ];
4132 :
4133 9 : if( FD_LIKELY( candidate->map.next!=UINT_MAX ) ) {
4134 0 : __builtin_prefetch( &accdb->acc_pool[ candidate->map.next ], 0, 1 );
4135 0 : }
4136 :
4137 9 : if( FD_UNLIKELY( !memcmp( pubkeys[ i ], candidate->key.pubkey, 32UL ) ) ) {
4138 9 : if( FD_LIKELY( (ulong)candidate->cache_idx>slots[ i ] ) ) {
4139 3 : skip[ i ] = 1;
4140 6 : } else if( FD_UNLIKELY( incremental ) && candidate->key.generation!=fork_gen ) {
4141 3 : cross_existing[ i ] = candidate;
4142 3 : } else {
4143 3 : existing[ i ] = candidate;
4144 3 : }
4145 9 : break;
4146 9 : }
4147 0 : next_acc = candidate->map.next;
4148 0 : }
4149 30 : }
4150 :
4151 : /* Phase 2b: reject intra-batch duplicate pubkeys. Snapin always
4152 : populates a batch from a single AppendVec, so every slot in the
4153 : batch is identical and a duplicate pubkey means the same account
4154 : appears twice at the same slot — i.e. a corrupt snapshot per the
4155 : Agave spec. We have no principled way to pick a winner; return
4156 : -1 so the caller can flag the snapshot malformed. Batches are
4157 : bounded (<=8) so the O(n^2) scan is trivial. */
4158 :
4159 30 : for( ulong i=1UL; i<cnt; i++ ) {
4160 45 : for( ulong j=0UL; j<i; j++ ) {
4161 27 : if( hashes[ j ]!=hashes[ i ] ) continue;
4162 0 : if( FD_UNLIKELY( !memcmp( pubkeys[ j ], pubkeys[ i ], 32UL ) ) ) {
4163 0 : FD_LOG_WARNING(( "corrupt snapshot: duplicate pubkey within a single batch (entries %lu and %lu, slots %lu and %lu)", j, i, slots[ j ], slots[ i ] ));
4164 0 : return -1;
4165 0 : }
4166 0 : }
4167 18 : }
4168 :
4169 : /* Phase 3: commit. For each account either update the existing
4170 : entry in-place (replace), allocate and insert at the chain head
4171 : (new), or skip entirely (ignore). This matches the
4172 : insert/replace/ignore semantics of write_one. */
4173 :
4174 12 : ulong used_bytes_added = 0UL;
4175 12 : ulong used_bytes_removed = 0UL;
4176 :
4177 42 : for( ulong i=0UL; i<cnt; i++ ) {
4178 30 : if( FD_UNLIKELY( skip[ i ] ) ) {
4179 : /* Still advance the write head so snapwr and snapin stay in
4180 : sync — snapwr unconditionally writes every account to disk.
4181 : Mark the space as immediately freed since it is dead on
4182 : arrival. */
4183 3 : ulong dead_sz = sizeof(fd_accdb_disk_meta_t)+data_lens[ i ];
4184 3 : ulong dead_off = allocate_next_write( accdb, dead_sz );
4185 3 : fd_accdb_shmem_bytes_freed( accdb->shmem, dead_off, dead_sz );
4186 3 : ignored_lamports += lamports[ i ];
4187 3 : ignored++;
4188 3 : continue;
4189 3 : }
4190 :
4191 27 : fd_accdb_accmeta_t * accmeta;
4192 :
4193 27 : if( FD_UNLIKELY( existing[ i ] ) ) {
4194 3 : accmeta = existing[ i ];
4195 : /* The old version's disk space is now dead. */
4196 3 : ulong old_sz = sizeof(fd_accdb_disk_meta_t) + FD_ACCDB_SIZE_DATA( accmeta->executable_size );
4197 3 : fd_accdb_shmem_bytes_freed( accdb->shmem, fd_accdb_acc_offset( accmeta ), old_sz );
4198 3 : used_bytes_removed += old_sz;
4199 3 : replaced_lamports += accmeta->lamports;
4200 3 : replaced++;
4201 24 : } else {
4202 24 : accmeta = acc_pool_acquire_nolock( accdb->acc_pool_join );
4203 24 : if( FD_UNLIKELY( !accmeta ) ) FD_LOG_ERR(( "accounts database ran out of space during snapshot loading" ));
4204 :
4205 24 : uint acc_idx = (uint)acc_pool_idx( accdb->acc_pool_join, accmeta );
4206 :
4207 24 : fd_memcpy( accmeta->key.pubkey, pubkeys[ i ], 32UL );
4208 24 : accmeta->key.generation = incremental ? fork_gen : gen;
4209 24 : accmeta->map.next = accdb->acc_map[ hashes[ i ] ];
4210 24 : accdb->acc_map[ hashes[ i ] ] = acc_idx;
4211 :
4212 24 : if( FD_UNLIKELY( incremental ) ) {
4213 6 : fd_accdb_txn_t * txn = txn_pool_acquire( accdb->txn_pool );
4214 6 : if( FD_UNLIKELY( !txn ) ) FD_LOG_ERR(( "txn pool exhausted during incremental snapshot loading" ));
4215 6 : txn->acc_pool_idx = acc_idx;
4216 6 : uint txn_idx = (uint)txn_pool_idx( accdb->txn_pool, txn );
4217 6 : txn->fork.next = fork->shmem->txn_head;
4218 6 : fork->shmem->txn_head = txn_idx;
4219 6 : }
4220 :
4221 24 : if( cross_existing[ i ] ) {
4222 3 : replaced_lamports += cross_existing[ i ]->lamports;
4223 3 : replaced++;
4224 3 : cross_replaced++;
4225 21 : } else {
4226 21 : loaded++;
4227 21 : }
4228 24 : }
4229 :
4230 27 : accmeta->cache_idx = (uint)slots[ i ];
4231 27 : accmeta->lamports = lamports[ i ];
4232 27 : accmeta->executable_size = FD_ACCDB_SIZE_PACK( (uint)data_lens[ i ], executables[ i ] );
4233 27 : ulong entry_sz = sizeof(fd_accdb_disk_meta_t)+data_lens[ i ];
4234 27 : ulong file_off = allocate_next_write( accdb, entry_sz );
4235 27 : accmeta->offset_fork = incremental ? fd_accdb_acc_pack_offset_fork( file_off, fork_id.val ) : file_off;
4236 27 : used_bytes_added += entry_sz;
4237 27 : }
4238 :
4239 12 : accdb->shmem->shmetrics->disk_used_bytes += used_bytes_added;
4240 12 : accdb->shmem->shmetrics->disk_used_bytes -= used_bytes_removed;
4241 :
4242 : /* accounts_total tracks acc_pool entries: increment for every new
4243 : allocation (both genuinely new accounts and cross-fork overrides
4244 : that insert a second pool entry). The output counter
4245 : *accounts_loaded excludes cross-fork overrides to match
4246 : snapshot_write_one semantics (cross-fork returns 2 = replaced). */
4247 12 : accdb->shmem->shmetrics->accounts_total += loaded + cross_replaced;
4248 :
4249 12 : *accounts_ignored = ignored;
4250 12 : *accounts_replaced = replaced;
4251 12 : *accounts_loaded = loaded;
4252 12 : *out_replaced_lamports = replaced_lamports;
4253 12 : *out_ignored_lamports = ignored_lamports;
4254 :
4255 12 : return 0;
4256 12 : }
4257 :
4258 : static void
4259 0 : delta_reset( fd_accdb_t * accdb ) {
4260 0 : if( !accdb->shmem->delta.head ) return; /* clean */
4261 0 : uint * chains = accdb->delta.chains;
4262 0 : ulong chain_cnt = accdb->shmem->delta.chain_cnt;
4263 0 : for( ulong i=0UL; i<chain_cnt; i++ ) chains[ i ] = UINT_MAX;
4264 0 : accdb->shmem->delta.head = 0UL;
4265 0 : }
4266 :
4267 : static int
4268 0 : delta_is_valid( fd_accdb_shmem_t const * accdb ) {
4269 0 : return accdb->delta.head < accdb->delta.ele_max;
4270 0 : }
4271 :
4272 : void
4273 : fd_accdb_background( fd_accdb_t * accdb,
4274 3268943 : int * charge_busy ) {
4275 3268943 : fd_accdb_shmem_t * shmem = accdb->shmem;
4276 :
4277 3268943 : ulong * snap_sync_p = &shmem->snapshot_sync;
4278 3268943 : ulong snap_sync = fd_accdb_snapshot_sync_state( snap_sync_p );
4279 :
4280 3268943 : uint op = FD_VOLATILE_CONST( shmem->cmd_op );
4281 3268943 : if( FD_UNLIKELY( op!=FD_ACCDB_CMD_IDLE ) ) {
4282 528 : fd_accdb_fork_id_t fork_id = { .val = FD_VOLATILE_CONST( shmem->cmd_fork_id ) };
4283 :
4284 528 : switch( op ) {
4285 501 : case FD_ACCDB_CMD_ADVANCE_ROOT:
4286 501 : background_advance_root( accdb, fork_id );
4287 501 : break;
4288 21 : case FD_ACCDB_CMD_PURGE:
4289 21 : background_purge( accdb, fork_id );
4290 21 : break;
4291 3 : case FD_ACCDB_CMD_DRAIN_DEFERRED:
4292 3 : drain_deferred_frees( accdb );
4293 3 : break;
4294 3 : case FD_ACCDB_CMD_CLEAR_DEFERRED: {
4295 : /* Posted by fd_accdb_reset after it clobbers shared pools.
4296 : T2's deferred fork chain now points at recycled elements;
4297 : discard the stale pointers. Epoch slots are preserved
4298 : across reset so no re-join is needed. */
4299 3 : accdb->deferred_fork_head = NULL;
4300 3 : accdb->deferred_fork_tail = NULL;
4301 3 : accdb->deferred_fork_epoch = 0UL;
4302 3 : break;
4303 0 : }
4304 0 : default:
4305 0 : FD_LOG_ERR(( "unexpected accdb cmd_op %u", op ));
4306 528 : }
4307 :
4308 528 : FD_COMPILER_MFENCE();
4309 528 : FD_VOLATILE( shmem->cmd_op ) = FD_ACCDB_CMD_IDLE;
4310 528 : *charge_busy = 1;
4311 528 : return;
4312 528 : }
4313 :
4314 3268415 : if( FD_UNLIKELY( snap_sync!=FD_ACCDB_SNAPSHOT_SYNC_IDLE ) ) {
4315 0 : switch( snap_sync ) {
4316 0 : case FD_ACCDB_SNAPSHOT_SYNC_RUNNING:
4317 : /* while producing a snapshot, don't do compaction work */
4318 0 : background_preevict( accdb, charge_busy, 0 );
4319 0 : return;
4320 0 : case FD_ACCDB_SNAPSHOT_SYNC_DONE:
4321 0 : fd_accdb_snapshot_sync_advance( snap_sync_p, FD_ACCDB_SNAPSHOT_SYNC_IDLE );
4322 0 : break;
4323 0 : case FD_ACCDB_SNAPSHOT_SYNC_START_FULL:
4324 0 : FD_CHECK_CRIT( !FD_VOLATILE_CONST( shmem->snapshot_loading ),
4325 0 : "snapshot production requested during snapshot load" );
4326 0 : delta_reset( accdb );
4327 0 : fd_accdb_snapshot_sync_advance( snap_sync_p, FD_ACCDB_SNAPSHOT_SYNC_RUNNING );
4328 0 : *charge_busy = 1;
4329 0 : return;
4330 0 : case FD_ACCDB_SNAPSHOT_SYNC_START_INCR:
4331 0 : FD_CHECK_CRIT( !FD_VOLATILE_CONST( shmem->snapshot_loading ),
4332 0 : "snapshot production requested during snapshot load" );
4333 0 : if( delta_is_valid( accdb->shmem ) ) {
4334 0 : fd_accdb_snapshot_sync_advance( snap_sync_p, FD_ACCDB_SNAPSHOT_SYNC_RUNNING );
4335 0 : } else {
4336 : /* cannot produce incrementals because delta ran out of space,
4337 : therefore don't know which accounts changed */
4338 0 : fd_accdb_snapshot_sync_advance( snap_sync_p, FD_ACCDB_SNAPSHOT_SYNC_FAIL );
4339 0 : }
4340 0 : *charge_busy = 1;
4341 0 : return;
4342 0 : case FD_ACCDB_SNAPSHOT_SYNC_FAIL:
4343 : /* wait for client to acknowledge */
4344 0 : break;
4345 0 : default:
4346 0 : FD_LOG_CRIT(( "corrupt snapshot_sync state %lu", snap_sync ));
4347 0 : }
4348 0 : }
4349 :
4350 3268415 : background_preevict( accdb, charge_busy, 0 );
4351 :
4352 13073660 : for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
4353 9805245 : background_compact( accdb, k, charge_busy );
4354 9805245 : }
4355 3268415 : }
4356 :
4357 : fd_accdb_shmem_metrics_t const *
4358 57 : fd_accdb_shmetrics( fd_accdb_t * accdb ) {
4359 57 : return accdb->shmem->shmetrics;
4360 57 : }
4361 :
4362 : fd_accdb_metrics_t const *
4363 9 : fd_accdb_metrics( fd_accdb_t * accdb ) {
4364 9 : return accdb->metrics;
4365 9 : }
4366 :
4367 : void
4368 : fd_accdb_cache_class_occupancy( fd_accdb_t * accdb,
4369 : ulong * used,
4370 : ulong * max,
4371 12 : ulong * reserved ) {
4372 108 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) {
4373 96 : ulong cap = accdb->shmem->cache_class_max[ c ];
4374 96 : ulong init = FD_VOLATILE_CONST( accdb->shmem->cache_class_init[ c ].val );
4375 96 : ulong freec = FD_VOLATILE_CONST( accdb->shmem->cache_free_cnt [ c ].val );
4376 96 : ulong live = init>freec ? init-freec : 0UL;
4377 96 : if( live>cap ) live = cap;
4378 96 : max [ c ] = cap;
4379 96 : used [ c ] = live;
4380 96 : reserved[ c ] = FD_VOLATILE_CONST( accdb->shmem->cache_class_used[ c ].val );
4381 96 : }
4382 12 : }
4383 :
4384 : void
4385 : fd_accdb_cache_class_thresholds( fd_accdb_t * accdb,
4386 : ulong * target_used,
4387 0 : ulong * low_water_used ) {
4388 0 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) {
4389 0 : ulong max_c = accdb->shmem->cache_class_max [ c ];
4390 0 : ulong free_tgt = accdb->shmem->cache_free_target [ c ];
4391 0 : ulong free_lwm = accdb->shmem->cache_free_low_water[ c ];
4392 0 : target_used [ c ] = max_c>free_tgt ? max_c-free_tgt : 0UL;
4393 0 : low_water_used[ c ] = max_c>free_lwm ? max_c-free_lwm : 0UL;
4394 0 : }
4395 0 : }
4396 :
4397 : #if FD_HAS_RACESAN
4398 :
4399 : /* Force pre-eviction (ignore the watermark) so a deterministic
4400 : single-threaded test can exercise the writeback path without
4401 : manufacturing real cache pressure. Sweeps several times: CLOCK needs
4402 : two visits to evict a recently-touched line (clear the "referenced"
4403 : bit, then evict), and the clock hand position carries across calls, so
4404 : one or two sweeps is not enough to guarantee every eligible line is
4405 : flushed back. */
4406 : void
4407 : fd_accdb_debug_force_preevict( fd_accdb_t * accdb ) {
4408 : for( ulong iter=0UL; iter<8UL; iter++ ) {
4409 : int charge_busy = 0;
4410 : background_preevict( accdb, &charge_busy, 1 );
4411 : }
4412 : }
4413 :
4414 : /* Locate the resident cache line currently holding `pubkey` (most recent
4415 : generation if multiple). Returns 1 and fills out_class/out_idx on a
4416 : hit, 0 if no resident line matches. Test-only helper so the test can
4417 : target a specific line without seeing the opaque fd_accdb struct. */
4418 :
4419 : int
4420 : fd_accdb_debug_find_line( fd_accdb_t * accdb,
4421 : uchar const * pubkey,
4422 : ulong * out_class,
4423 : ulong * out_idx ) {
4424 : int found = 0;
4425 : uint best_gen = 0U;
4426 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) {
4427 : ulong init = FD_VOLATILE_CONST( accdb->shmem->cache_class_init[ c ].val );
4428 : ulong max_c = accdb->shmem->cache_class_max[ c ];
4429 : if( init>max_c ) init = max_c;
4430 : for( ulong idx=0UL; idx<init; idx++ ) {
4431 : fd_accdb_cache_line_t * line = cache_line( accdb, c, idx );
4432 : if( line->key.generation==UINT_MAX ) continue;
4433 : if( memcmp( line->key.pubkey, pubkey, 32UL ) ) continue;
4434 : if( !found || line->key.generation>=best_gen ) {
4435 : best_gen = line->key.generation;
4436 : *out_class = c;
4437 : *out_idx = idx;
4438 : found = 1;
4439 : }
4440 : }
4441 : }
4442 : return found;
4443 : }
4444 :
4445 : /* Address of one cache line. struct fd_accdb_private is local to this
4446 : translation unit, so a test cannot reach accdb->cache[] itself. */
4447 :
4448 : void *
4449 : fd_accdb_debug_line_addr( fd_accdb_t * accdb,
4450 : ulong size_class,
4451 : ulong line_idx ) {
4452 : return cache_line( accdb, size_class, line_idx );
4453 : }
4454 :
4455 : /* Deterministically evict a single specified cache line via the
4456 : foreground evictor's claim sequence (CAS refcnt 0->EVICT_SENTINEL),
4457 : then write the dirty line back exactly as fd_accdb_acquire_inner's
4458 : STEP-4 / background_ preevict do (pubkey from accmeta, owner+data
4459 : from the line). Mirrors acquire_cache_line's CLOCK-claim path
4460 : (fd_accdb.c) so a racesan test can reproduce, without a 640+-slot
4461 : cache-pressure rig, the interleaving where acc_unlink observes
4462 : EVICT_SENTINEL on the line it is unlinking.
4463 :
4464 : The fd_racesan_hook("clock_evict:post_sentinel") fires right after
4465 : the sentinel is installed (matching the production foreground path),
4466 : so the test can suspend this fiber holding the sentinel while another
4467 : fiber drives acc_unlink to its reclaim CAS. Returns the captured
4468 : evicted acc_idx (UINT_MAX if the line was clean / unbound). */
4469 :
4470 : uint
4471 : fd_accdb_debug_clock_evict_line( fd_accdb_t * accdb,
4472 : ulong size_class,
4473 : ulong line_idx ) {
4474 : fd_accdb_shmem_t * shmem = accdb->shmem;
4475 : fd_accdb_cache_line_t * line = cache_line( accdb, size_class, line_idx );
4476 :
4477 : /* Claim for eviction, same as acquire_cache_line's CLOCK path. */
4478 : if( FD_UNLIKELY( FD_ATOMIC_CAS( &line->refcnt, 0U, FD_ACCDB_EVICT_SENTINEL )!=0U ) ) return UINT_MAX;
4479 :
4480 : if( FD_UNLIKELY( line->acc_idx==UINT_MAX && line->key.generation==UINT_MAX ) ) {
4481 : FD_VOLATILE( line->refcnt ) = 0;
4482 : return UINT_MAX;
4483 : }
4484 :
4485 : fd_racesan_hook( "clock_evict:post_sentinel" );
4486 :
4487 : uint acc_idx = line->acc_idx;
4488 : if( FD_LIKELY( acc_idx!=UINT_MAX ) ) {
4489 : evict_clear_acc_cache_ref( &accdb->acc_pool[ acc_idx ], size_class, line_idx );
4490 : }
4491 : uint evicted_acc_idx = line->persisted ? UINT_MAX : acc_idx;
4492 : line->key.generation = UINT_MAX;
4493 :
4494 : /* Write back the dirty line, exactly like the production writeback
4495 : sites: this is the synthesis that would emit a pubkey=NEW/owner=OLD
4496 : poison record if the accmeta slot had been recycled out from under
4497 : us. In the SENTINEL-vs-acc_unlink race this proves no poison: the
4498 : epoch the evictor holds blocks drain_deferred_frees, so the slot is
4499 : never recycled while we are here. */
4500 : if( FD_UNLIKELY( !line->persisted && acc_idx!=UINT_MAX ) ) {
4501 : fd_accdb_accmeta_t * accmeta = &accdb->acc_pool[ acc_idx ];
4502 : ulong entry_sz = sizeof(fd_accdb_disk_meta_t)+(ulong)FD_ACCDB_SIZE_DATA( accmeta->executable_size );
4503 :
4504 : ulong old_offset = fd_accdb_acc_xchg_offset( accmeta, FD_ACCDB_OFF_INVAL );
4505 : if( FD_LIKELY( old_offset!=FD_ACCDB_OFF_INVAL ) ) {
4506 : fd_accdb_shmem_bytes_freed( shmem, old_offset, entry_sz );
4507 : FD_ATOMIC_FETCH_AND_SUB( &shmem->shmetrics->disk_used_bytes, entry_sz );
4508 : }
4509 :
4510 : fd_accdb_disk_meta_t meta;
4511 : fd_memcpy( meta.pubkey, accmeta->key.pubkey, 32UL );
4512 : meta.size = FD_ACCDB_SIZE_DATA( accmeta->executable_size );
4513 : meta.generation = accmeta->key.generation;
4514 : fd_memcpy( meta.owner, line->owner, 32UL );
4515 :
4516 : struct iovec iovs[ 2UL ] = {
4517 : { .iov_base = &meta, .iov_len = sizeof(fd_accdb_disk_meta_t) },
4518 : { .iov_base = (void *)(line+1UL), .iov_len = FD_ACCDB_SIZE_DATA( accmeta->executable_size ) }
4519 : };
4520 : ulong file_off = allocate_next_write( accdb, entry_sz );
4521 : ulong written = 0UL;
4522 : while( written<entry_sz ) {
4523 : long result = pwritev2( accdb->fd, iovs, 2, (long)(file_off+written), 0 );
4524 : if( FD_UNLIKELY( result==-1 && errno==EINTR ) ) continue;
4525 : else if( FD_UNLIKELY( result<=0 ) ) FD_LOG_ERR(( "pwritev2() failed (%d-%s)", errno, fd_io_strerror( errno ) ));
4526 : written += (ulong)result;
4527 : for( int v=0; v<2; v++ ) {
4528 : if( (ulong)result>=iovs[ v ].iov_len ) { result -= (long)iovs[ v ].iov_len; iovs[ v ].iov_len = 0UL; }
4529 : else { iovs[ v ].iov_base = (uchar *)iovs[ v ].iov_base + result; iovs[ v ].iov_len -= (ulong)result; break; }
4530 : }
4531 : }
4532 : FD_COMPILER_MFENCE();
4533 : accmeta->offset_fork = fd_accdb_acc_pack_offset_fork( file_off, fd_accdb_acc_fork_id(accmeta) );
4534 : FD_ATOMIC_FETCH_AND_ADD( &shmem->shmetrics->disk_used_bytes, entry_sz );
4535 : }
4536 :
4537 : line->persisted = 1;
4538 : line->acc_idx = UINT_MAX;
4539 : line->key.generation = UINT_MAX;
4540 : FD_COMPILER_MFENCE();
4541 : FD_VOLATILE( line->refcnt ) = 0;
4542 : cache_free_push( accdb, size_class, line );
4543 : return evicted_acc_idx;
4544 : }
4545 :
4546 : #endif
|