Line data Source code
1 : #include "fd_accdb_shmem.h"
2 : #include "fd_accdb_private.h"
3 :
4 : #include "../../util/log/fd_log.h"
5 :
6 : #define POOL_NAME partition_pool
7 12498 : #define POOL_T fd_accdb_partition_t
8 33501096 : #define POOL_NEXT pool_next
9 : #define POOL_IDX_T ulong
10 : #define POOL_IMPL_STYLE 2
11 :
12 : #include "../../util/tmpl/fd_pool.c"
13 :
14 : #define DLIST_NAME compaction_dlist
15 : #define DLIST_ELE_T fd_accdb_partition_t
16 : #define DLIST_PREV dlist_prev
17 0 : #define DLIST_NEXT dlist_next
18 : #define DLIST_IMPL_STYLE 2
19 :
20 : #include "../../util/tmpl/fd_dlist.c"
21 :
22 : #define DLIST_NAME deferred_free_dlist
23 : #define DLIST_ELE_T fd_accdb_partition_t
24 : #define DLIST_PREV dlist_prev
25 0 : #define DLIST_NEXT dlist_next
26 : #define DLIST_IMPL_STYLE 2
27 :
28 : #include "../../util/tmpl/fd_dlist.c"
29 :
30 : FD_FN_CONST ulong
31 8592 : fd_accdb_shmem_align( void ) {
32 8592 : return FD_ACCDB_SHMEM_ALIGN;
33 8592 : }
34 :
35 : fd_accdb_shmem_t *
36 4164 : fd_accdb_shmem_join( void * shtc ) {
37 4164 : if( FD_UNLIKELY( !shtc ) ) {
38 0 : FD_LOG_WARNING(( "NULL shtc" ));
39 0 : return NULL;
40 0 : }
41 :
42 4164 : if( FD_UNLIKELY( !fd_ulong_is_aligned( (ulong)shtc, fd_accdb_shmem_align() ) ) ) {
43 0 : FD_LOG_WARNING(( "misaligned shtc" ));
44 0 : return NULL;
45 0 : }
46 :
47 4164 : fd_accdb_shmem_t * accdb = (fd_accdb_shmem_t *)shtc;
48 :
49 4164 : if( FD_UNLIKELY( accdb->magic!=FD_ACCDB_SHMEM_MAGIC ) ) {
50 0 : FD_LOG_WARNING(( "bad magic" ));
51 0 : return NULL;
52 0 : }
53 4164 : return accdb;
54 4164 : }
55 :
56 : ulong
57 : fd_accdb_shmem_footprint( ulong max_accounts,
58 : ulong max_live_slots,
59 : ulong max_account_writes_per_slot,
60 : ulong partition_cnt,
61 : ulong cache_footprint,
62 : ulong cache_min_reserved,
63 : ulong joiner_cnt,
64 276 : ulong max_incremental_accounts ) {
65 276 : if( FD_UNLIKELY( !max_accounts ) ) return 0UL;
66 276 : if( FD_UNLIKELY( !max_live_slots ) ) return 0UL;
67 276 : if( FD_UNLIKELY( !max_account_writes_per_slot) ) return 0UL;
68 276 : if( FD_UNLIKELY( !partition_cnt ) ) return 0UL;
69 276 : if( FD_UNLIKELY( !cache_min_reserved ) ) return 0UL;
70 : /* Partition indices are packed into 13 bits of accdb_offset_t
71 : (bits 63..51), so partition_cnt==8192 uses indices 0..8191, the
72 : full 13-bit range. The initial write-head sentinel encodes its
73 : invalidity in the offset bits (partition_offset==partition_sz),
74 : not the index, so it remains distinguishable even when no spare
75 : index value is left. */
76 276 : if( FD_UNLIKELY( partition_cnt>(1UL<<13) ) ) return 0UL;
77 276 : if( FD_UNLIKELY( !joiner_cnt || joiner_cnt>FD_ACCDB_MAX_JOINERS ) ) return 0UL;
78 :
79 276 : if( FD_UNLIKELY( max_accounts>=UINT_MAX ) ) return 0UL;
80 :
81 276 : if( FD_UNLIKELY( max_live_slots>=USHORT_MAX ) ) return 0UL;
82 :
83 276 : ulong txn_max = max_live_slots * max_account_writes_per_slot;
84 276 : if( FD_UNLIKELY( txn_max/max_account_writes_per_slot!=max_live_slots ) ) return 0UL;
85 276 : if( FD_UNLIKELY( txn_max>=UINT_MAX ) ) return 0UL;
86 :
87 276 : ulong descends_fp = descends_set_footprint( max_live_slots );
88 276 : if( FD_UNLIKELY( !descends_fp ) ) return 0UL;
89 276 : if( FD_UNLIKELY( max_live_slots>ULONG_MAX/descends_fp ) ) return 0UL;
90 :
91 276 : ulong chain_cnt = fd_ulong_pow2_up( (max_accounts>>1) + (max_accounts&1UL) );
92 :
93 276 : if( FD_UNLIKELY( chain_cnt>ULONG_MAX/sizeof(uint) ) ) return 0UL;
94 :
95 276 : if( FD_UNLIKELY( !cache_footprint ) ) return 0UL;
96 276 : ulong cache_class_max[ FD_ACCDB_CACHE_CLASS_CNT ];
97 276 : if( FD_UNLIKELY( !fd_accdb_cache_class_cnt( cache_footprint, cache_min_reserved, cache_class_max ) ) ) return 0UL;
98 :
99 276 : if( FD_UNLIKELY( max_incremental_accounts>UINT_MAX ) ) return 0UL;
100 276 : ulong delta_chain_cnt = fd_ulong_pow2_up( (max_incremental_accounts>>1) + (max_incremental_accounts&1UL) );
101 :
102 276 : ulong l;
103 276 : l = FD_LAYOUT_INIT;
104 276 : l = FD_LAYOUT_APPEND( l, FD_ACCDB_SHMEM_ALIGN, sizeof(fd_accdb_shmem_t) );
105 276 : l = FD_LAYOUT_APPEND( l, alignof(fd_accdb_fork_shmem_t), max_live_slots*sizeof(fd_accdb_fork_shmem_t) );
106 276 : l = FD_LAYOUT_APPEND( l, descends_set_align(), max_live_slots*descends_set_footprint( max_live_slots ) );
107 276 : l = FD_LAYOUT_APPEND( l, alignof(uint), chain_cnt*sizeof(uint) );
108 276 : l = FD_LAYOUT_APPEND( l, alignof(fd_accdb_accmeta_t), max_accounts*sizeof(fd_accdb_accmeta_t) );
109 276 : l = FD_LAYOUT_APPEND( l, alignof(fd_accdb_txn_t), txn_max*sizeof(fd_accdb_txn_t) );
110 276 : l = FD_LAYOUT_APPEND( l, partition_pool_align(), partition_pool_footprint( partition_cnt ) );
111 1104 : for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
112 828 : l = FD_LAYOUT_APPEND( l, compaction_dlist_align(), compaction_dlist_footprint() );
113 828 : }
114 276 : l = FD_LAYOUT_APPEND( l, deferred_free_dlist_align(), deferred_free_dlist_footprint() );
115 276 : l = FD_LAYOUT_APPEND( l, alignof(uint), txn_max*sizeof(uint) );
116 2484 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) {
117 2208 : l = FD_LAYOUT_APPEND( l, FD_ACCDB_CACHE_META_SZ, cache_class_max[c]*fd_accdb_cache_slot_sz[c] );
118 2208 : }
119 276 : l = FD_LAYOUT_APPEND( l, alignof(uint), delta_chain_cnt*sizeof(uint) );
120 276 : l = FD_LAYOUT_APPEND( l, alignof(fd_accdb_delta_t),max_incremental_accounts*sizeof(fd_accdb_delta_t) );
121 276 : return FD_LAYOUT_FINI( l, FD_ACCDB_SHMEM_ALIGN );
122 276 : }
123 :
124 : void *
125 : fd_accdb_shmem_new( void * shmem,
126 : ulong max_accounts,
127 : ulong max_live_slots,
128 : ulong max_account_writes_per_slot,
129 : ulong partition_cnt,
130 : ulong partition_sz,
131 : ulong cache_footprint,
132 : ulong cache_min_reserved,
133 : int bundle_enabled,
134 : ulong seed,
135 : ulong joiner_cnt,
136 4164 : ulong max_incremental_accounts ) {
137 4164 : if( FD_UNLIKELY( !shmem ) ) {
138 0 : FD_LOG_WARNING(( "NULL shmem" ));
139 0 : return NULL;
140 0 : }
141 :
142 4164 : if( FD_UNLIKELY( !fd_ulong_is_aligned( (ulong)shmem, fd_accdb_shmem_align() ) ) ) {
143 0 : FD_LOG_WARNING(( "misaligned shmem" ));
144 0 : return NULL;
145 0 : }
146 :
147 4164 : if( FD_UNLIKELY( !max_accounts ) ) {
148 0 : FD_LOG_WARNING(( "max_accounts must be non-zero" ));
149 0 : return NULL;
150 0 : }
151 :
152 4164 : if( FD_UNLIKELY( !max_live_slots ) ) {
153 0 : FD_LOG_WARNING(( "max_live_slots must be non-zero" ));
154 0 : return NULL;
155 0 : }
156 :
157 4164 : if( FD_UNLIKELY( !max_account_writes_per_slot ) ) {
158 0 : FD_LOG_WARNING(( "max_account_writes_per_slot must be non-zero" ));
159 0 : return NULL;
160 0 : }
161 :
162 4164 : if( FD_UNLIKELY( !joiner_cnt || joiner_cnt>FD_ACCDB_MAX_JOINERS ) ) {
163 0 : FD_LOG_WARNING(( "joiner_cnt must be in [1, %lu]", FD_ACCDB_MAX_JOINERS ));
164 0 : return NULL;
165 0 : }
166 :
167 4164 : if( FD_UNLIKELY( max_live_slots>=USHORT_MAX ) ) {
168 0 : FD_LOG_WARNING(( "max_live_slots must be less than %u", (uint)USHORT_MAX ));
169 0 : return NULL;
170 0 : }
171 :
172 4164 : if( FD_UNLIKELY( !partition_cnt ) ) {
173 0 : FD_LOG_WARNING(( "partition_cnt must be non-zero" ));
174 0 : return NULL;
175 0 : }
176 :
177 4164 : if( FD_UNLIKELY( partition_cnt>(1UL<<13) ) ) {
178 0 : FD_LOG_WARNING(( "partition_cnt must be at most %lu", 1UL<<13 ));
179 0 : return NULL;
180 0 : }
181 :
182 4164 : if( FD_UNLIKELY( !partition_sz ) ) {
183 0 : FD_LOG_WARNING(( "partition_sz must be non-zero" ));
184 0 : return NULL;
185 0 : }
186 :
187 : /* Partition offsets are packed into the low 51 bits of accdb_offset_t
188 : (see FD_ACCDB_PARTITION_OFF_BITS in fd_accdb.c). partition_sz must
189 : be small enough that speculative fetch-and-adds from up to
190 : FD_ACCDB_MAX_JOINERS concurrent threads in allocate_next_write
191 : can never carry the offset field into the partition_idx bits.
192 : Worst case: all joiners each do one FETCH_AND_ADD of partition_sz
193 : before the partition switch completes, starting from an offset of
194 : at most partition_sz-1. */
195 4164 : if( FD_UNLIKELY( partition_sz>(1UL<<51)/(FD_ACCDB_MAX_JOINERS+1UL) ) ) {
196 0 : FD_LOG_WARNING(( "partition_sz must be at most %lu", (1UL<<51)/(FD_ACCDB_MAX_JOINERS+1UL) ));
197 0 : return NULL;
198 0 : }
199 :
200 : /* The maximum file offset is (partition_cnt-1)*partition_sz +
201 : partition_sz - 1, which must fit in a signed long (off_t) because
202 : pwritev2, preadv2, and fallocate all take signed offsets. */
203 4164 : if( FD_UNLIKELY( partition_cnt>=(ulong)LONG_MAX/partition_sz ) ) {
204 0 : FD_LOG_WARNING(( "partition_cnt*partition_sz must be at most LONG_MAX" ));
205 0 : return NULL;
206 0 : }
207 :
208 : /* The total addressable file space (partition_cnt * partition_sz)
209 : must not exceed 2^FD_ACCDB_OFF_BITS. File offsets are stored in
210 : the 48-bit offset portion of acc->offset_fork, and the all-ones
211 : value FD_ACCDB_OFF_INVAL is reserved as a dirty sentinel. The
212 : allocator guarantees record start offsets are always at least
213 : sizeof(fd_accdb_disk_meta_t) below a partition boundary, so a
214 : total of exactly 2^48 is safe (no valid offset reaches the
215 : sentinel), but exceeding it is not. */
216 4164 : if( FD_UNLIKELY( partition_cnt>((1UL<<FD_ACCDB_OFF_BITS)/partition_sz) ) ) {
217 0 : FD_LOG_WARNING(( "partition_cnt*partition_sz must be at most %lu", 1UL<<FD_ACCDB_OFF_BITS ));
218 0 : return NULL;
219 0 : }
220 :
221 : /* partition_sz must be large enough to hold at least one worst-case
222 : account write (disk metadata header + largest cache class payload).
223 : Without this, allocate_next_write can never fit the entry in a
224 : single partition. */
225 4164 : ulong min_partition_sz = sizeof(fd_accdb_disk_meta_t) + fd_accdb_cache_slot_sz[ FD_ACCDB_CACHE_CLASS_CNT-1UL ] - FD_ACCDB_CACHE_META_SZ;
226 4164 : if( FD_UNLIKELY( partition_sz<min_partition_sz ) ) {
227 0 : FD_LOG_WARNING(( "partition_sz must be at least %lu to fit worst-case account write", min_partition_sz ));
228 0 : return NULL;
229 0 : }
230 :
231 4164 : if( FD_UNLIKELY( max_accounts>=UINT_MAX ) ) {
232 0 : FD_LOG_WARNING(( "max_accounts must be less than UINT_MAX" ));
233 0 : return NULL;
234 0 : }
235 :
236 4164 : ulong txn_max = max_live_slots * max_account_writes_per_slot;
237 4164 : if( FD_UNLIKELY( txn_max/max_account_writes_per_slot!=max_live_slots ) ) {
238 0 : FD_LOG_WARNING(( "max_live_slots*max_account_writes_per_slot overflows" ));
239 0 : return NULL;
240 0 : }
241 4164 : if( FD_UNLIKELY( txn_max>=UINT_MAX ) ) {
242 0 : FD_LOG_WARNING(( "max_live_slots*max_account_writes_per_slot must be less than UINT_MAX" ));
243 0 : return NULL;
244 0 : }
245 :
246 4164 : ulong descends_fp = descends_set_footprint( max_live_slots );
247 4164 : if( FD_UNLIKELY( !descends_fp || max_live_slots>ULONG_MAX/descends_fp ) ) {
248 0 : FD_LOG_WARNING(( "max_live_slots*descends_set_footprint overflows" ));
249 0 : return NULL;
250 0 : }
251 :
252 4164 : ulong chain_cnt = fd_ulong_pow2_up( (max_accounts>>1) + (max_accounts&1UL) );
253 :
254 4164 : if( FD_UNLIKELY( chain_cnt>ULONG_MAX/sizeof(uint) ) ) {
255 0 : FD_LOG_WARNING(( "chain_cnt*sizeof(uint) overflows" ));
256 0 : return NULL;
257 0 : }
258 :
259 4164 : if( FD_UNLIKELY( !cache_min_reserved ) ) {
260 0 : FD_LOG_WARNING(( "cache_min_reserved must be non-zero" ));
261 0 : return NULL;
262 0 : }
263 :
264 4164 : ulong cache_class_max[ FD_ACCDB_CACHE_CLASS_CNT ];
265 4164 : if( FD_UNLIKELY( !fd_accdb_cache_class_cnt( cache_footprint, cache_min_reserved, cache_class_max ) ) ) {
266 0 : FD_LOG_WARNING(( "invalid cache_footprint" ));
267 0 : return NULL;
268 0 : }
269 : /* cidx packs only FD_ACCDB_CACHE_LINE_BITS bits of line index, so
270 : cache_class_max[c]>FD_ACCDB_CACHE_LINE_MAX would let line indices
271 : alias. fd_accdb_cache_class_cnt clamps this; assert here so any
272 : future regression in the allocator is caught at shmem-new time
273 : rather than as silent cache corruption at runtime. */
274 37476 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) FD_TEST( cache_class_max[ c ]<=FD_ACCDB_CACHE_LINE_MAX );
275 :
276 4164 : if( FD_UNLIKELY( max_incremental_accounts>UINT_MAX ) ) {
277 0 : FD_LOG_WARNING(( "max_incremental_accounts must be at most %u", UINT_MAX ));
278 0 : return NULL;
279 0 : }
280 :
281 4164 : ulong delta_chain_cnt = fd_ulong_pow2_up( (max_incremental_accounts>>1) + (max_incremental_accounts&1UL) );
282 4164 : if( FD_UNLIKELY( delta_chain_cnt>UINT_MAX ) ) {
283 0 : FD_LOG_WARNING(( "max_incremental_accounts must be at most %u", UINT_MAX ));
284 0 : return NULL;
285 0 : }
286 :
287 4164 : FD_SCRATCH_ALLOC_INIT( l, shmem );
288 4164 : fd_accdb_shmem_t * accdb = FD_SCRATCH_ALLOC_APPEND( l, FD_ACCDB_SHMEM_ALIGN, sizeof(fd_accdb_shmem_t) );
289 4164 : void * _fork_pool_ele = FD_SCRATCH_ALLOC_APPEND( l, alignof(fd_accdb_fork_shmem_t), max_live_slots*sizeof(fd_accdb_fork_shmem_t) );
290 4164 : void * _descends_sets = FD_SCRATCH_ALLOC_APPEND( l, descends_set_align(), max_live_slots*descends_set_footprint( max_live_slots ) );
291 4164 : void * _acc_map = FD_SCRATCH_ALLOC_APPEND( l, alignof(uint), chain_cnt*sizeof(uint) );
292 4164 : void * _acc_pool_ele = FD_SCRATCH_ALLOC_APPEND( l, alignof(fd_accdb_accmeta_t), max_accounts*sizeof(fd_accdb_accmeta_t) );
293 4164 : void * _txn_pool_ele = FD_SCRATCH_ALLOC_APPEND( l, alignof(fd_accdb_txn_t), txn_max*sizeof(fd_accdb_txn_t) );
294 4164 : void * _partition_pool = FD_SCRATCH_ALLOC_APPEND( l, partition_pool_align(), partition_pool_footprint( partition_cnt ) );
295 4164 : void * _compaction_dlists[ FD_ACCDB_COMPACTION_LAYER_CNT ];
296 16656 : for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
297 12492 : _compaction_dlists[ k ] = FD_SCRATCH_ALLOC_APPEND( l, compaction_dlist_align(), compaction_dlist_footprint() );
298 12492 : }
299 4164 : void * _deferred_free_dlist = FD_SCRATCH_ALLOC_APPEND( l, deferred_free_dlist_align(), deferred_free_dlist_footprint() );
300 4164 : void * _deferred_acc_buf = FD_SCRATCH_ALLOC_APPEND( l, alignof(uint), txn_max*sizeof(uint) );
301 4164 : void * _cache_regions[ FD_ACCDB_CACHE_CLASS_CNT ];
302 37476 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) {
303 33312 : _cache_regions[ c ] = FD_SCRATCH_ALLOC_APPEND( l, FD_ACCDB_CACHE_META_SZ, cache_class_max[c]*fd_accdb_cache_slot_sz[c] );
304 33312 : }
305 4164 : void * _delta_map = FD_SCRATCH_ALLOC_APPEND( l, alignof(uint), delta_chain_cnt*sizeof(uint) );
306 4164 : void * _delta_pool = FD_SCRATCH_ALLOC_APPEND( l, alignof(fd_accdb_delta_t), max_incremental_accounts*sizeof(fd_accdb_delta_t) );
307 :
308 4164 : fd_memset( _acc_map, 0xFF, chain_cnt*sizeof(uint) );
309 :
310 4164 : FD_TEST( acc_pool_new( accdb->acc_pool ) );
311 4164 : acc_pool_t _acc_pool_join[1];
312 4164 : FD_TEST( acc_pool_join( _acc_pool_join, accdb->acc_pool, _acc_pool_ele, max_accounts ) );
313 4164 : acc_pool_reset( _acc_pool_join );
314 4164 : acc_pool_leave( _acc_pool_join );
315 :
316 4164 : FD_TEST( fork_pool_new( accdb->fork_pool ) );
317 4164 : fork_pool_t _fork_pool_join[1];
318 4164 : FD_TEST( fork_pool_join( _fork_pool_join, accdb->fork_pool, _fork_pool_ele, max_live_slots ) );
319 4164 : fork_pool_reset( _fork_pool_join );
320 4164 : fork_pool_leave( _fork_pool_join );
321 :
322 4164 : ulong descends_set_fp = descends_set_footprint( max_live_slots );
323 74325 : for( ulong i=0UL; i<max_live_slots; i++ ) {
324 70161 : descends_set_t * descends_set = descends_set_join( descends_set_new( (uchar *)_descends_sets + i*descends_set_fp, max_live_slots ) );
325 70161 : FD_TEST( descends_set );
326 70161 : }
327 :
328 4164 : FD_TEST( txn_pool_new( accdb->txn_pool ) );
329 4164 : txn_pool_t _txn_pool_join[1];
330 4164 : FD_TEST( txn_pool_join( _txn_pool_join, accdb->txn_pool, _txn_pool_ele, txn_max ) );
331 4164 : txn_pool_reset( _txn_pool_join );
332 4164 : txn_pool_leave( _txn_pool_join );
333 :
334 4164 : fd_accdb_partition_t * partition_pool = partition_pool_join( partition_pool_new( _partition_pool, partition_cnt ) );
335 4164 : FD_TEST( partition_pool );
336 33480684 : for( ulong i=0UL; i<partition_cnt; i++ ) {
337 33476520 : partition_pool_ele( partition_pool, i )->write_offset = 0UL;
338 33476520 : }
339 :
340 16656 : for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
341 12492 : compaction_dlist_t * dlist = compaction_dlist_join( compaction_dlist_new( _compaction_dlists[ k ] ) );
342 12492 : FD_TEST( dlist );
343 12492 : }
344 :
345 4164 : deferred_free_dlist_t * deferred_free = deferred_free_dlist_join( deferred_free_dlist_new( _deferred_free_dlist ) );
346 4164 : FD_TEST( deferred_free );
347 :
348 4164 : fd_memset( _delta_map, 0xFF, delta_chain_cnt*sizeof(uint) );
349 :
350 4164 : accdb->seed = seed;
351 4164 : accdb->root_fork_id = (fd_accdb_fork_id_t){ .val = USHORT_MAX };
352 4164 : accdb->generation = 0U;
353 :
354 4164 : accdb->partition_lock = 0;
355 4164 : accdb->snapshot_loading = 0;
356 4164 : accdb->bundle_enabled = bundle_enabled;
357 :
358 37476 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) accdb->clock_hand[ c ].val = 0UL;
359 37476 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) accdb->cache_free[ c ].ver_top = (ulong)UINT_MAX;
360 37476 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) accdb->cache_free_cnt[ c ].val = 0UL;
361 :
362 37476 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) {
363 33312 : ulong max_c = cache_class_max[ c ];
364 33312 : ulong floor_c = fd_ulong_min( cache_min_reserved, max_c );
365 33312 : ulong headroom = ( max_c>floor_c ) ? ( max_c - floor_c ) : 0UL;
366 33312 : ulong cap = fd_ulong_min( 8192UL, (64UL<<20) / fd_accdb_cache_slot_sz[ c ] );
367 33312 : ulong burst_floor = fd_ulong_min( 512UL, headroom/2UL );
368 33312 : ulong target = fd_ulong_min( cap, fd_ulong_max( headroom/10UL, burst_floor ) );
369 33312 : accdb->cache_free_target [ c ] = target;
370 33312 : accdb->cache_free_low_water[ c ] = (target * 3UL) / 4UL;
371 33312 : }
372 :
373 16656 : for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
374 : /* Sentinel: partition_offset == partition_sz forces the first
375 : allocate_next_write to fall into the partition-switch slow path,
376 : which acquires a real partition from the pool.
377 :
378 : The invalidity lives in the offset bits, not the index bits. The
379 : index here (partition_cnt) is only nominally invalid: at the
380 : maximum partition_cnt==8192 it does not fit in the 13-bit index
381 : field and wraps to 0, a perfectly valid pool index. */
382 12492 : accdb->whead[ k ] = accdb_offset( partition_cnt, partition_sz );
383 12492 : accdb->has_partition[ k ] = 0;
384 12492 : }
385 :
386 4164 : accdb->chain_cnt = chain_cnt;
387 4164 : accdb->max_live_slots = max_live_slots;
388 4164 : accdb->max_accounts = max_accounts;
389 4164 : accdb->max_account_writes_per_slot = max_account_writes_per_slot;
390 4164 : accdb->joiner_cnt_max = joiner_cnt;
391 4164 : accdb->cache_min_reserved = cache_min_reserved;
392 4164 : accdb->partition_cnt = partition_cnt;
393 4164 : accdb->partition_sz = partition_sz;
394 4164 : accdb->partition_max = 0UL;
395 :
396 4164 : accdb->partition_pool_off = (ulong)partition_pool - (ulong)shmem;
397 16656 : for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
398 12492 : accdb->compaction_dlist_off[ k ] = (ulong)_compaction_dlists[ k ] - (ulong)shmem;
399 12492 : }
400 4164 : accdb->deferred_free_dlist_off = (ulong)_deferred_free_dlist - (ulong)shmem;
401 :
402 4164 : accdb->deferred_acc_buf_off = (ulong)_deferred_acc_buf - (ulong)shmem;
403 4164 : accdb->deferred_acc_buf_cnt = 0UL;
404 4164 : accdb->deferred_acc_buf_max = txn_max;
405 4164 : accdb->deferred_acc_epoch = 0UL;
406 :
407 4164 : accdb->epoch = 1UL;
408 4164 : accdb->snapshot_sync = FD_ACCDB_SNAPSHOT_SYNC_IDLE;
409 4164 : accdb->joiner_cnt = 0UL;
410 1070148 : for( ulong i=0UL; i<FD_ACCDB_MAX_JOINERS; i++ ) accdb->joiner_epochs[ i ].val = ULONG_MAX;
411 :
412 37476 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) accdb->cache_class_init[ c ].val = 0UL;
413 37476 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) accdb->cache_class_max[ c ] = cache_class_max[ c ];
414 37476 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) accdb->cache_region_off[ c ] = (ulong)_cache_regions[ c ] - (ulong)shmem;
415 :
416 : /* Pre-initialize every cache slot's metadata to the "empty" sentinel
417 : (gen=UINT_MAX, acc_idx=UINT_MAX, refcnt=0). Without this, the
418 : lazy-init path in acquire_cache_line bumps cache_class_init before
419 : writing the sentinels into the freshly-claimed line; a concurrent
420 : background_preevict reading the bumped init counter could then sweep
421 : a slot whose memory still reads as zero, see (gen=0, acc_idx=0)
422 : instead of the skip predicate, CAS refcnt 0->EVICT_SENTINEL, and
423 : "evict" a line the lazy-init owner is about to publish. */
424 37476 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) {
425 33312 : ulong slot_sz = fd_accdb_cache_slot_sz[ c ];
426 283146222 : for( ulong i=0UL; i<cache_class_max[ c ]; i++ ) {
427 283112910 : fd_accdb_cache_line_t * line = (fd_accdb_cache_line_t *)( (uchar *)_cache_regions[ c ] + i*slot_sz );
428 283112910 : line->key.generation = UINT_MAX;
429 283112910 : line->acc_idx = UINT_MAX;
430 283112910 : line->refcnt = 0U;
431 283112910 : line->referenced = 0;
432 283112910 : line->persisted = 1;
433 283112910 : }
434 33312 : }
435 :
436 : /* If a class has enough slots for every joiner's worst case
437 : simultaneously (cache_min_reserved per joiner), no reservation can
438 : ever overflow. Sentinel ULONG_MAX tells acquire/release to skip
439 : the atomic counters entirely. */
440 37476 : for( ulong c=0UL; c<FD_ACCDB_CACHE_CLASS_CNT; c++ ) {
441 33312 : if( cache_class_max[ c ]>=cache_min_reserved*joiner_cnt ) accdb->cache_class_used[ c ].val = ULONG_MAX;
442 60 : else accdb->cache_class_used[ c ].val = 0UL;
443 33312 : }
444 :
445 4164 : accdb->delta.seed = seed+1UL;
446 4164 : accdb->delta.chain_off = (ulong)_delta_map - (ulong)shmem;
447 4164 : accdb->delta.chain_cnt = (uint)delta_chain_cnt;
448 4164 : accdb->delta.chain_mask = (uint)delta_chain_cnt - 1U;
449 4164 : accdb->delta.ele_off = (ulong)_delta_pool - (ulong)shmem;
450 4164 : accdb->delta.ele_max = max_incremental_accounts;
451 4164 : accdb->delta.head = 0UL;
452 :
453 4164 : memset( accdb->shmetrics, 0, sizeof( fd_accdb_shmem_metrics_t ) );
454 4164 : accdb->shmetrics->accounts_capacity = max_accounts;
455 :
456 4164 : accdb->cmd_op = FD_ACCDB_CMD_IDLE;
457 4164 : accdb->cmd_fork_id = USHORT_MAX;
458 :
459 4164 : FD_COMPILER_MFENCE();
460 4164 : FD_VOLATILE( accdb->magic ) = FD_ACCDB_SHMEM_MAGIC;
461 4164 : FD_COMPILER_MFENCE();
462 :
463 4164 : return (void *)accdb;
464 4164 : }
465 :
466 : void
467 : fd_accdb_shmem_try_enqueue_compaction( fd_accdb_shmem_t * accdb,
468 117 : ulong partition_idx ) {
469 : /* Caller must hold partition_lock. */
470 :
471 117 : fd_accdb_partition_t * partition_pool = (fd_accdb_partition_t *)( (uchar *)accdb + accdb->partition_pool_off );
472 117 : fd_accdb_partition_t * partition = partition_pool_ele( partition_pool, partition_idx );
473 :
474 117 : if( FD_UNLIKELY( partition->bytes_freed<(accdb->partition_sz*FD_ACCDB_COMPACTION_THRESHOLD_PCT/100UL) ) ) return;
475 54 : if( FD_UNLIKELY( partition->marked_compaction ) ) return;
476 :
477 : /* While a snapshot load is in flight, defer all compaction so the
478 : compaction tile cannot race with the bulk loader. Anything that
479 : crosses the threshold here will be re-checked by
480 : fd_accdb_snapshot_load_end's sweep when loading completes. */
481 45 : if( FD_UNLIKELY( FD_VOLATILE_CONST( accdb->snapshot_loading ) ) ) return;
482 :
483 : /* Do not enqueue any currently active write-head partition. Its
484 : write_offset is not yet finalized, so compaction cannot determine
485 : the valid data range. The partition_lock serializes this check
486 : with change_partition, so it is not racy. */
487 108 : for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
488 81 : if( FD_UNLIKELY( accdb->has_partition[ k ] && packed_partition_idx( &accdb->whead[ k ] )==partition_idx ) ) return;
489 81 : }
490 :
491 27 : uchar layer = partition->layer;
492 27 : compaction_dlist_t * compaction_dlist = (compaction_dlist_t *)( (uchar *)accdb + accdb->compaction_dlist_off[ layer ] );
493 :
494 27 : partition->marked_compaction = 1;
495 27 : partition->compaction_offset = 0UL;
496 27 : partition->compaction_ready_epoch = FD_ATOMIC_FETCH_AND_ADD( &accdb->epoch, 1UL );
497 27 : partition->queued = 1;
498 27 : if( FD_LIKELY( compaction_dlist_is_empty( compaction_dlist, partition_pool ) ) ) {
499 21 : FD_LOG_INFO(( "compaction of layer %u partition %lu started", (uint)layer, partition_pool_idx( partition_pool, partition ) ));
500 21 : }
501 27 : compaction_dlist_ele_push_tail( compaction_dlist, partition, partition_pool );
502 27 : accdb->shmetrics->in_compaction = 1;
503 27 : accdb->shmetrics->compactions_requested++;
504 27 : }
505 :
506 : void
507 : fd_accdb_shmem_bytes_freed( fd_accdb_shmem_t * accdb,
508 : ulong offset,
509 42 : ulong sz ) {
510 42 : fd_accdb_partition_t * partition_pool = (fd_accdb_partition_t *)( (uchar *)accdb + accdb->partition_pool_off );
511 :
512 42 : ulong partition_idx = offset/accdb->partition_sz;
513 42 : fd_accdb_partition_t * partition = partition_pool_ele( partition_pool, partition_idx );
514 : /* Launder the pointer: GCC derives partition from (accdb + off) and so
515 : believes __builtin_object_size( &partition->bytes_freed )==0, which
516 : trips a spurious -Wstringop-overflow on the atomic add below. */
517 42 : FD_COMPILER_FORGET( partition );
518 42 : FD_ATOMIC_FETCH_AND_ADD( &partition->bytes_freed, sz );
519 :
520 : /* Fast-path exit: skip the lock if clearly below threshold or
521 : already enqueued. */
522 42 : if( FD_LIKELY( partition->bytes_freed<(accdb->partition_sz*FD_ACCDB_COMPACTION_THRESHOLD_PCT/100UL) ) ) return;
523 24 : if( FD_UNLIKELY( partition->marked_compaction ) ) return;
524 :
525 9 : spin_lock_acquire( &accdb->partition_lock );
526 9 : fd_accdb_shmem_try_enqueue_compaction( accdb, partition_idx );
527 9 : spin_lock_release( &accdb->partition_lock );
528 9 : }
529 :
530 : ulong
531 9 : fd_accdb_shmem_partition_max( fd_accdb_shmem_t const * accdb ) {
532 9 : return accdb->partition_max;
533 9 : }
534 :
535 : ulong
536 0 : fd_accdb_shmem_partition_sz( fd_accdb_shmem_t const * accdb ) {
537 0 : return accdb->partition_sz;
538 0 : }
539 :
540 : void
541 : fd_accdb_shmem_partition_info( fd_accdb_shmem_t const * accdb,
542 : ulong partition_idx,
543 33 : fd_accdb_shmem_partition_info_t * out ) {
544 33 : fd_accdb_partition_t const * partition_pool = (fd_accdb_partition_t const *)( (uchar const *)accdb + accdb->partition_pool_off );
545 33 : fd_accdb_partition_t const * p = partition_pool_ele_const( partition_pool, partition_idx );
546 :
547 33 : out->file_offset = partition_idx * accdb->partition_sz;
548 33 : out->is_write_head = 0;
549 : /* If this partition is currently the active write head for any
550 : layer, partition->write_offset is stale (it's only updated at
551 : handoff in change_partition). The live tip lives in whead[layer].
552 : Surface the live value so the GUI shows real-time fill, not the
553 : "0 until rolled" snapshot. The tip is a reservation that can
554 : briefly overrun the partition, so clamp rather than show >100%. */
555 33 : ulong head_off = ULONG_MAX;
556 78 : for( ulong k=0UL; k<FD_ACCDB_COMPACTION_LAYER_CNT; k++ ) {
557 63 : if( !FD_VOLATILE_CONST( accdb->has_partition[ k ] ) ) continue;
558 33 : accdb_offset_t whead = { .val = FD_VOLATILE_CONST( accdb->whead[ k ].val ) };
559 33 : if( packed_partition_idx( &whead )==partition_idx ) {
560 18 : head_off = packed_partition_offset( &whead );
561 18 : out->is_write_head = 1;
562 18 : break;
563 18 : }
564 33 : }
565 33 : if( FD_UNLIKELY( out->is_write_head ) ) {
566 18 : out->write_offset_raw = head_off;
567 18 : out->write_offset = fd_ulong_min( head_off, accdb->partition_sz );
568 18 : } else {
569 15 : out->write_offset_raw = FD_VOLATILE_CONST( p->write_offset );
570 15 : out->write_offset = out->write_offset_raw;
571 15 : }
572 33 : out->bytes_freed = FD_VOLATILE_CONST( p->bytes_freed );
573 33 : out->compaction_offset = FD_VOLATILE_CONST( p->compaction_offset );
574 33 : out->read_ops = FD_VOLATILE_CONST( p->read_ops );
575 33 : out->bytes_read = FD_VOLATILE_CONST( p->bytes_read );
576 33 : out->write_ops = FD_VOLATILE_CONST( p->write_ops );
577 33 : out->bytes_written = FD_VOLATILE_CONST( p->bytes_written );
578 33 : out->created_ticks = (long)FD_VOLATILE_CONST( p->created_ticks );
579 33 : out->filled_ticks = (long)FD_VOLATILE_CONST( p->filled_ticks );
580 33 : out->layer = p->layer;
581 33 : uchar compacting = FD_VOLATILE_CONST( p->compacting_now );
582 33 : uchar queued = FD_VOLATILE_CONST( p->queued );
583 33 : out->compaction_state = compacting ? 2 : ( queued ? 1 : 0 );
584 33 : }
585 :
586 : FD_STATIC_ASSERT( sizeof(((fd_accdb_shmem_writer_barrier_t *)0)->bits)*8UL==FD_ACCDB_MAX_JOINERS, barrier_width );
587 :
588 : void
589 : fd_accdb_shmem_writer_barrier_capture( fd_accdb_shmem_t const * accdb,
590 0 : fd_accdb_shmem_writer_barrier_t * barrier ) {
591 0 : memset( barrier->bits, 0, sizeof(barrier->bits) );
592 0 : ulong joiner_cnt = FD_VOLATILE_CONST( accdb->joiner_cnt );
593 0 : for( ulong t=0UL; t<joiner_cnt; t++ ) {
594 0 : if( FD_VOLATILE_CONST( accdb->joiner_epochs[ t ].val )==ULONG_MAX ) continue;
595 0 : barrier->bits[ t/64UL ] |= 1UL<<(t%64UL);
596 0 : }
597 0 : }
598 :
599 : ulong
600 : fd_accdb_shmem_writer_barrier_poll( fd_accdb_shmem_t const * accdb,
601 0 : fd_accdb_shmem_writer_barrier_t * barrier ) {
602 0 : ulong remain = 0UL;
603 0 : for( ulong w=0UL; w<sizeof(barrier->bits)/sizeof(ulong); w++ ) {
604 0 : ulong bits = barrier->bits[ w ];
605 0 : while( bits ) {
606 0 : ulong b = (ulong)fd_ulong_find_lsb( bits );
607 0 : bits &= bits-1UL;
608 0 : if( FD_VOLATILE_CONST( accdb->joiner_epochs[ w*64UL+b ].val )==ULONG_MAX ) {
609 0 : barrier->bits[ w ] &= ~(1UL<<b);
610 0 : }
611 0 : }
612 0 : remain |= barrier->bits[ w ];
613 0 : }
614 0 : return remain;
615 0 : }
616 :
617 : ulong const *
618 0 : fd_accdb_shmem_snapshot_sync( fd_accdb_shmem_t const * accdb ) {
619 0 : return &accdb->snapshot_sync;
620 0 : }
|