Line data Source code
1 : #ifndef HEADER_fd_src_disco_topo_fd_topo_h
2 : #define HEADER_fd_src_disco_topo_fd_topo_h
3 :
4 : #include "../stem/fd_stem.h"
5 : #include "../../tango/fd_tango.h"
6 : #include "../../waltz/xdp/fd_xdp1.h"
7 : #include "../../waltz/http/fd_url.h"
8 : #include "../../ballet/base58/fd_base58.h"
9 : #include "../../flamenco/fd_flamenco_base.h"
10 : #include "../../util/net/fd_net_headers.h"
11 : #include "../../util/net/fd_ip6.h"
12 : #include "../pack/fd_pack_acct_blocklist.h"
13 :
14 : /* Maximum number of workspaces that may be present in a topology. */
15 : #define FD_TOPO_MAX_WKSPS (256UL)
16 : /* Maximum number of links that may be present in a topology. */
17 0 : #define FD_TOPO_MAX_LINKS (256UL)
18 : /* Maximum number of tiles that may be present in a topology. */
19 0 : #define FD_TOPO_MAX_TILES (256UL)
20 : /* Maximum number of objects that may be present in a topology. */
21 : #define FD_TOPO_MAX_OBJS (4096UL)
22 : /* Maximum number of links that may go into any one tile in the
23 : topology. */
24 387 : #define FD_TOPO_MAX_TILE_IN_LINKS ( 128UL)
25 : /* Maximum number of links that a tile may write to. */
26 : #define FD_TOPO_MAX_TILE_OUT_LINKS ( 32UL)
27 : /* Maximum number of objects that a tile can use. */
28 : #define FD_TOPO_MAX_TILE_OBJS ( 256UL)
29 :
30 : /* Maximum number of additional ip addresses */
31 : #define FD_NET_MAX_SRC_ADDR 4
32 :
33 : /* Maximum number of additional destinations for leader shreds and for retransmitted shreds */
34 : #define FD_TOPO_ADTL_DESTS_MAX ( 32UL)
35 :
36 3 : #define FD_TOPO_CORE_DUMP_LEVEL_DISABLED (0)
37 0 : #define FD_TOPO_CORE_DUMP_LEVEL_MINIMAL (1)
38 93 : #define FD_TOPO_CORE_DUMP_LEVEL_REGULAR (2)
39 0 : #define FD_TOPO_CORE_DUMP_LEVEL_FULL (3)
40 0 : #define FD_TOPO_CORE_DUMP_LEVEL_NEVER (4)
41 :
42 : /* A workspace is a Firedancer specific memory management structure that
43 : sits on top of 1 or more memory mapped gigantic or huge pages mounted
44 : to the hugetlbfs. */
45 : typedef struct {
46 : ulong id; /* The ID of this workspace. Indexed from [0, wksp_cnt). When placed in a topology, the ID must be the index of the workspace in the workspaces list. */
47 : char name[ 14UL ]; /* The name of this workspace, like "pack". There can be at most one of each workspace name in a topology. */
48 :
49 : ulong numa_idx; /* The index of the NUMA node on the system that this workspace should be allocated from. */
50 :
51 : ulong min_part_max; /* Artificially raise part_max */
52 : ulong min_loose_sz; /* Artificially raise loose footprint */
53 :
54 : /* Computed fields. These are not supplied as configuration but calculated as needed. */
55 : struct {
56 : ulong page_sz; /* The size of the pages that this workspace is backed by. One of FD_PAGE_SIZE_*. */
57 : ulong page_cnt; /* The number of pages that must be mapped to this workspace to store all the data needed by consumers. */
58 : ulong part_max; /* The maximum number of partitions in the underlying workspace. There can only be this many allocations made at any one time. */
59 :
60 : int core_dump_level; /* The core dump level required to be set in the application configuration to have this workspace appear in core dumps. */
61 :
62 : fd_wksp_t * wksp; /* The workspace memory in the local process. */
63 : ulong known_footprint; /* Total size in bytes of all data in Firedancer that will be stored in this workspace at startup. */
64 : ulong total_footprint; /* Total size in bytes of all data in Firedancer that could be stored in this workspace, includes known data and loose data. */
65 : };
66 : } fd_topo_wksp_t;
67 :
68 : /* A link is an mcache in a workspace that has one producer and one or
69 : more consumers. A link may optionally also have a dcache, that holds
70 : fragments referred to by the mcache entries.
71 :
72 : A link belongs to exactly one workspace. A link has exactly one
73 : producer, and 1 or more consumers. Each consumer is either reliable
74 : or not reliable. A link has a depth and a MTU, which correspond to
75 : the depth and MTU of the mcache and dcache respectively. A MTU of
76 : zero means no dcache is needed, as there is no data. */
77 : typedef struct {
78 : ulong id; /* The ID of this link. Indexed from [0, link_cnt). When placed in a topology, the ID must be the index of the link in the links list. */
79 : char name[ 14UL ]; /* The name of this link, like "pack_execle". There can be multiple of each link name in a topology. */
80 : ulong kind_id; /* The ID of this link within its name. If there are N links of a particular name, they have IDs [0, N). The pair (name, kind_id) uniquely identifies a link, as does "id" on its own. */
81 :
82 : ulong depth; /* The depth of the mcache representing the link. */
83 : ulong mtu; /* The MTU of data fragments in the mcache. A value of 0 means there is no dcache. */
84 : ulong burst; /* The max amount of MTU sized data fragments that might be bursted to the dcache. */
85 :
86 : ulong mcache_obj_id;
87 : ulong dcache_obj_id;
88 :
89 : /* Computed fields. These are not supplied as configuration but calculated as needed. */
90 : struct {
91 : fd_frag_meta_t * mcache; /* The mcache of this link. */
92 : void * dcache; /* The dcache of this link, if it has one. */
93 : };
94 :
95 : uint permit_no_consumers : 1; /* Permit a topology where this link has no consumers */
96 : uint permit_no_producers : 1; /* Permit a topology where this link has no producers */
97 : } fd_topo_link_t;
98 :
99 : /* Be careful: ip and host are in different byte order */
100 : typedef struct {
101 : uint ip; /* in network byte order */
102 : ushort port; /* in host byte order */
103 : } fd_topo_ip_port_t;
104 :
105 : struct fd_topo_net_tile {
106 : ulong umem_dcache_obj_id; /* dcache for network UMEM frames */
107 : uint bind_address;
108 :
109 : ushort shred_listen_port;
110 : ushort quic_transaction_listen_port;
111 : ushort legacy_transaction_listen_port;
112 : ushort gossip_listen_port;
113 : ushort repair_client_listen_port;
114 : ushort repair_serve_listen_port;
115 : ushort txsend_src_port;
116 : ushort votor_quic_client_listen_port;
117 : ushort votor_quic_server_listen_port;
118 : };
119 : typedef struct fd_topo_net_tile fd_topo_net_tile_t;
120 :
121 : /* A tile is a unique process that is spawned by Firedancer to represent
122 : one thread of execution. Firedancer sandboxes all tiles to their own
123 : process for security reasons.
124 :
125 : A tile belongs to exactly one workspace. A tile is a consumer of 0
126 : or more links, it's inputs. A tile is a producer of 0 or more output
127 : links.
128 :
129 : All input links will be automatically polled by the tile
130 : infrastructure, and output links will automatically source and manage
131 : credits from consumers. */
132 : struct fd_topo_tile {
133 : ulong id; /* The ID of this tile. Indexed from [0, tile_cnt). When placed in a topology, the ID must be the index of the tile in the tiles list. */
134 : char name[ 7UL ]; /* The name of this tile. There can be multiple of each tile name in a topology. */
135 : ulong kind_id; /* The ID of this tile within its name. If there are n tile of a particular name, they have IDs [0, N). The pair (name, kind_id) uniquely identifies a tile, as does "id" on its own. */
136 :
137 : int is_waker_client; /* Tile has file descriptors which must be serviced by the external waker tile. */
138 : int is_agave; /* If the tile needs to run in the Agave (Anza) address space or not. */
139 : int allow_shutdown; /* If the tile is allowed to shutdown gracefully. If false, when the tile exits it will tear down the entire application. */
140 :
141 : ulong cpu_idx; /* The CPU index to pin the tile on. A value of ULONG_MAX or more indicates the tile should be floating and not pinned to a core. */
142 : int floats; /* Scheduled by the kernel over the CPUs of the floating tiles on its NUMA node, never a pinned tile's CPU, instead of pinned to cpu_idx (efficient mode). cpu_idx still places memory and isolation, and is the fallback when no such CPU remains. */
143 :
144 : ulong waker_client_idx; /* Client slot in the fixed inherited fd range (inner epoll fd FD_WAKER_INNER_FD( idx )), or ULONG_MAX if not a waker client */
145 : ulong waker_fseq_obj_id; /* fseq object holding the tile's waker readiness word or ULONG_MAX */
146 :
147 : ulong in_cnt; /* The number of links that this tile reads from. */
148 : ulong in_link_id[ FD_TOPO_MAX_TILE_IN_LINKS ]; /* The link_id of each link that this tile reads from, indexed in [0, in_cnt). */
149 : int in_link_reliable[ FD_TOPO_MAX_TILE_IN_LINKS ]; /* If each link that this tile reads from is a reliable or unreliable consumer, indexed in [0, in_cnt). */
150 : int in_link_poll[ FD_TOPO_MAX_TILE_IN_LINKS ]; /* If each link that this tile reads from should be polled by the tile infrastructure, indexed in [0, in_cnt).
151 : If the link is not polled, the tile will not receive frags for it and the tile writer is responsible for
152 : reading from the link. The link must be marked as unreliable as it is not flow controlled. */
153 :
154 : ulong out_cnt; /* The number of links that this tile writes to. */
155 : ulong out_link_id[ FD_TOPO_MAX_TILE_OUT_LINKS ]; /* The link_id of each link that this tile writes to, indexed in [0, link_cnt). */
156 :
157 : ulong event_link_id; /* If not ULONG_MAX, the link_id of a dedicated unreliable link to the event tile that this tile reports
158 : telemetry events on via the thread-local fd_event_report_* macros. This link is deliberately NOT part
159 : of out_link_id[] / out_cnt: it is written directly (outside fd_stem) by the thread-local reporter. */
160 :
161 : ulong tile_obj_id;
162 : ulong metrics_obj_id;
163 : ulong id_keyswitch_obj_id; /* keyswitch object id for identity key updates */
164 : ulong av_keyswitch_obj_id; /* keyswitch object id for authority key updates */
165 : ulong in_link_fseq_obj_id[ FD_TOPO_MAX_TILE_IN_LINKS ];
166 :
167 : ulong uses_obj_cnt;
168 : ulong uses_obj_id[ FD_TOPO_MAX_TILE_OBJS ];
169 : int uses_obj_mode[ FD_TOPO_MAX_TILE_OBJS ];
170 :
171 : /* Computed fields. These are not supplied as configuration but calculated as needed. */
172 : struct {
173 : ulong * metrics; /* The shared memory for metrics that this tile should write. Consumer by monitoring and metrics writing tiles. */
174 :
175 : /* The fseq of each link that this tile reads from. Multiple fseqs
176 : may point to the link, if there are multiple consumers. An fseq
177 : can be uniquely identified via (link_id, tile_id), or (link_kind,
178 : link_kind_id, tile_kind, tile_kind_id) */
179 : ulong * in_link_fseq[ FD_TOPO_MAX_TILE_IN_LINKS ];
180 : };
181 :
182 : /* Configuration fields. These are required to be known by the topology so it can determine the
183 : total size of Firedancer in memory. */
184 : union {
185 : fd_topo_net_tile_t net;
186 :
187 : struct {
188 : fd_topo_net_tile_t net;
189 :
190 : char if_virt[ 16 ]; /* device name (virtual, for routing) */
191 : char if_phys[ 16 ]; /* device name (physical, for RX/TX) */
192 : uint if_queue; /* device queue index */
193 :
194 : /* xdp specific options */
195 : ulong xdp_rx_queue_size;
196 : ulong xdp_tx_queue_size;
197 : ulong free_ring_depth;
198 : long tx_flush_timeout_ns;
199 : char xdp_mode[8];
200 : int zero_copy;
201 :
202 : char poll_mode[ 16 ]; /* "softirq" or "prefbusy" */
203 :
204 : ulong netdev_tbl_obj_id;
205 :
206 : ulong route_max;
207 : ulong route_peer_max;
208 : ulong route_peer_seed;
209 : ulong neigh4_obj_id; /* neigh4 hash map */
210 :
211 : int xsk_core_dump;
212 : } xdp;
213 :
214 : struct {
215 : fd_topo_net_tile_t net;
216 :
217 : char if_name[ 16 ];
218 : uint rx_queue_size;
219 : uint tx_queue_size;
220 : uint batch_size;
221 :
222 : ulong netdev_tbl_obj_id;
223 : ulong route_max;
224 : ulong route_peer_max;
225 : ulong route_peer_seed;
226 : ulong neigh4_obj_id;
227 : } mlx5;
228 :
229 : struct {
230 : fd_topo_net_tile_t net;
231 : /* sock specific options */
232 : int so_sndbuf;
233 : int so_rcvbuf;
234 : } sock;
235 :
236 : struct {
237 : ulong netdev_tbl_obj_id;
238 : ulong route_max;
239 : ulong route_peer_max;
240 : char neigh_if[ 16 ]; /* neigh4 interface name */
241 : ulong neigh4_obj_id; /* neigh4 hash map */
242 : } netlink;
243 :
244 : struct {
245 : char identity_key_path[ PATH_MAX ];
246 : } admin;
247 :
248 0 : #define FD_TOPO_GOSSIP_ENTRYPOINTS_MAX 16UL
249 :
250 : struct {
251 : char identity_key_path[ PATH_MAX ];
252 :
253 : ulong entrypoints_cnt;
254 : char entrypoints[ FD_TOPO_GOSSIP_ENTRYPOINTS_MAX ][ FD_HOSTPORT_BUF_MAX ];
255 :
256 : long boot_timestamp_nanos;
257 :
258 : ulong tcache_depth;
259 :
260 : ushort shred_version;
261 : int allow_private_address;
262 :
263 : char gossip_host[ FD_FQDN_BUF_MAX ];
264 : fd_ip4_port_t gossip_addr;
265 : fd_ip4_port_t src_addr;
266 : } gossvf;
267 :
268 : struct {
269 : char identity_key_path[ PATH_MAX ];
270 :
271 : ulong entrypoints_cnt;
272 : char entrypoints[ FD_TOPO_GOSSIP_ENTRYPOINTS_MAX ][ FD_HOSTPORT_BUF_MAX ];
273 :
274 : long boot_timestamp_nanos;
275 :
276 : char gossip_host[ FD_FQDN_BUF_MAX ];
277 : uint net_ip_addr; /* net.ip_addr fallback when gossip_host empty */
278 : uint ip_addr;
279 : uint bind_ip_addr;
280 : ushort shred_version;
281 :
282 : ulong max_entries;
283 : ulong max_purged;
284 : ulong max_failed;
285 :
286 : fd_hash_t wait_for_supermajority_with_bank_hash;
287 :
288 : struct {
289 : ushort gossip;
290 : ushort tvu;
291 : ushort tvu_quic;
292 : ushort tpu;
293 : ushort tpu_quic;
294 : ushort repair;
295 : ushort rserve;
296 : ushort votor;
297 : } ports;
298 : } gossip;
299 :
300 : struct {
301 : uint out_depth;
302 : uint reasm_cnt;
303 : ulong max_concurrent_connections;
304 : ulong max_concurrent_handshakes;
305 : ushort quic_transaction_listen_port;
306 : long idle_timeout_millis;
307 : uint ack_delay_millis;
308 : int retry;
309 : char key_log_path[ PATH_MAX ];
310 : } quic;
311 :
312 : struct {
313 : ulong tcache_depth;
314 : } verify;
315 :
316 : struct {
317 : ulong tcache_depth;
318 : } dedup;
319 :
320 : struct {
321 : char url[ FD_URL_MAX ];
322 : ulong url_len;
323 : char sni[ FD_SNI_BUF_MAX ];
324 : ulong sni_len;
325 : char identity_key_path[ PATH_MAX ];
326 : char key_log_path[ PATH_MAX ];
327 : ulong buf_sz;
328 : ulong out_depth;
329 : ulong keepalive_interval_nanos;
330 : uchar tls_cert_verify : 1;
331 : } bundle;
332 :
333 : struct {
334 : char url[ FD_URL_MAX ];
335 : char identity_key_path[ PATH_MAX ];
336 : char action[ 16 ];
337 : uchar genesis_hash[ 32 ];
338 : ushort shred_version;
339 :
340 : char accounts_path [ PATH_MAX ];
341 : char snapshots_path[ PATH_MAX ];
342 : char log_path [ PATH_MAX ];
343 : char shredb_path [ PATH_MAX ];
344 : char guidb_path [ PATH_MAX ];
345 : char net_interface [ 16 ];
346 : long boot_timestamp_nanos;
347 : } event;
348 :
349 : struct {
350 : ulong max_pending_transactions;
351 : ulong execle_tile_count;
352 : ulong max_cost_per_block;
353 : ulong max_shreds_per_block;
354 : ulong bench_max_shreds_per_block; /* [development.bench], floors the leader's per-slot shred limit */
355 : int use_consumed_cus;
356 : int schedule_strategy;
357 : struct {
358 : int enabled;
359 : uchar tip_distribution_program_addr[ 32 ];
360 : uchar tip_payment_program_addr[ 32 ];
361 : uchar tip_distribution_authority[ 32 ];
362 : ulong commission_bps;
363 : char identity_key_path[ PATH_MAX ];
364 : char vote_account_path[ PATH_MAX ]; /* or pubkey is okay */
365 : } bundle;
366 : ulong acct_blocklist_cnt;
367 : fd_pubkey_t acct_blocklist[ FD_PACK_ACCT_BLOCKLIST_MAX ];
368 : } pack;
369 :
370 : struct {
371 : int lagged_consecutive_leader_start;
372 : int plugins_enabled;
373 : ulong execle_cnt;
374 : char identity_key_path[ PATH_MAX ];
375 : struct {
376 : int enabled;
377 : uchar tip_payment_program_addr[ 32 ];
378 : uchar tip_distribution_program_addr[ 32 ];
379 : char vote_account_path[ PATH_MAX ];
380 : } bundle;
381 : } pohh;
382 :
383 : struct {
384 : ulong execle_cnt;
385 : char identity_key_path[ PATH_MAX ];
386 : ulong max_txn_per_slot;
387 : } poh;
388 :
389 : struct {
390 : ulong execle_cnt;
391 : char identity_key_path[ PATH_MAX ];
392 : ulong max_txn_per_slot;
393 : } motor;
394 :
395 : struct {
396 : ulong fec_exposure;
397 : ulong fec_resolver_depth;
398 : char identity_key_path[ PATH_MAX ];
399 : ushort shred_listen_port;
400 : ulong max_shreds_per_block;
401 : ulong bench_max_shreds_per_block; /* [development.bench], floors the chain's per-slot limit */
402 : ushort expected_shred_version;
403 : ulong adtl_dests_retransmit_cnt;
404 : fd_topo_ip_port_t adtl_dests_retransmit[ FD_TOPO_ADTL_DESTS_MAX ];
405 : ulong adtl_dests_leader_cnt;
406 : fd_topo_ip_port_t adtl_dests_leader[ FD_TOPO_ADTL_DESTS_MAX ];
407 : } shred;
408 :
409 : struct {
410 : ulong disable_blockstore_from_slot;
411 : } store;
412 :
413 : struct {
414 : char identity_key_path[ PATH_MAX ];
415 : ulong authorized_voter_paths_cnt;
416 : char authorized_voter_paths[ 16 ][ PATH_MAX ];
417 : struct {
418 : uchar tip_payment_program_addr[ 32 ];
419 : uchar tip_distribution_program_addr[ 32 ];
420 : } bundle;
421 : } sign;
422 :
423 : struct {
424 : uint listen_addr;
425 : ushort listen_port;
426 :
427 : int is_voting;
428 : int is_alpenglow;
429 :
430 : char cluster[ 32 ];
431 : char identity_key_path[ PATH_MAX ];
432 : char vote_key_path[ PATH_MAX ];
433 : char accounts_database_path[ PATH_MAX ];
434 : char gui_database_path[ PATH_MAX ];
435 :
436 : ulong max_http_connections;
437 : ulong max_websocket_connections;
438 : ulong max_http_request_length;
439 : ulong send_buffer_size_mb;
440 : ulong db_size_gib;
441 : int schedule_strategy;
442 :
443 : int websocket_compression;
444 : ulong tile_cnt;
445 : ulong max_live_slots;
446 :
447 : char wfs_bank_hash[ FD_BASE58_ENCODED_32_SZ ];
448 : ushort expected_shred_version;
449 : ulong cache_size_gib;
450 : ulong accdb_obj_id;
451 : ulong max_txn_per_slot;
452 : } gui;
453 :
454 : struct {
455 : fd_ip6_addr_t listen_addr;
456 : ushort listen_port;
457 :
458 : ulong max_http_connections;
459 : ulong max_websocket_connections;
460 : ulong send_buffer_size_mb;
461 : ulong max_http_request_length;
462 :
463 : ulong max_live_slots;
464 : ulong genesis_max_message_size;
465 :
466 : ulong accdb_obj_id;
467 : ulong accdb_epoch_fseq_obj_id;
468 :
469 : char identity_key_path[ PATH_MAX ];
470 : int delay_startup;
471 :
472 : int snapshot_server_enabled;
473 : char snapshot_server_host[ FD_FQDN_BUF_MAX ];
474 : ushort snapshot_server_port;
475 : } rpc;
476 :
477 : struct {
478 : uint prometheus_listen_addr;
479 : ushort prometheus_listen_port;
480 : } metric;
481 :
482 : struct {
483 : int is_voting;
484 :
485 : char accounts_path [ PATH_MAX ];
486 : char shreds_path [ PATH_MAX ];
487 : char snapshots_path[ PATH_MAX ];
488 : char gui_path [ PATH_MAX ];
489 : char log_path [ PATH_MAX ];
490 : } diag;
491 :
492 : struct {
493 : ulong fec_max;
494 :
495 : ulong accdb_obj_id;
496 : ulong txncache_obj_id;
497 :
498 : char shred_cap[ PATH_MAX ];
499 :
500 : char identity_key_path[ PATH_MAX ];
501 : uint ip_addr;
502 : char vote_account_path[ PATH_MAX ];
503 :
504 : fd_hash_t wait_for_supermajority_with_bank_hash;
505 : ushort expected_shred_version;
506 : int wait_for_vote_to_start_leader;
507 :
508 : ulong heap_size_gib;
509 : ulong sched_depth;
510 : ulong max_live_slots;
511 : ulong full_snapshot_interval_blocks;
512 : ulong incremental_snapshot_interval_blocks;
513 :
514 : /* not specified in TOML */
515 :
516 : long boot_timestamp_nanos;
517 :
518 : ulong enable_features_cnt;
519 : char enable_features[ 16 ][ FD_BASE58_ENCODED_32_SZ ];
520 :
521 : char genesis_path[ PATH_MAX ];
522 :
523 : ulong max_txn_per_slot; /* config->limits */
524 : ulong max_shreds_per_block;
525 :
526 : ulong capture_start_slot;
527 : char solcap_capture[ PATH_MAX ];
528 : char dump_proto_dir[ PATH_MAX ];
529 : int dump_block_to_pb;
530 : int report_runtime_diffs;
531 :
532 : struct {
533 : int enabled;
534 : uchar tip_payment_program_addr[ 32 ];
535 : uchar tip_distribution_program_addr[ 32 ];
536 : char vote_account_path[ PATH_MAX ];
537 : } bundle;
538 :
539 : int alpenglow;
540 : } replay;
541 :
542 : struct {
543 : ulong txncache_obj_id;
544 : ulong progcache_obj_id;
545 : ulong accdb_obj_id;
546 :
547 : ulong max_live_slots;
548 :
549 : ulong capture_start_slot;
550 : char solcap_capture[ PATH_MAX ];
551 : char dump_proto_dir[ PATH_MAX ];
552 : char dump_syscall_name_filter[ PATH_MAX ];
553 : char dump_instr_program_id_filter[ FD_BASE58_ENCODED_32_SZ ];
554 : int dump_instr_to_pb;
555 : int dump_txn_to_pb;
556 : int dump_txn_as_fixture;
557 : int dump_syscall_to_pb;
558 : int report_runtime_diffs;
559 : } execrp;
560 :
561 : struct {
562 : ushort send_to_port;
563 : uint send_to_ip_addr;
564 : ulong conn_cnt;
565 : int no_quic;
566 : } benchs;
567 :
568 : struct {
569 : ushort rpc_port;
570 : uint rpc_ip_addr;
571 : ulong duration_s;
572 : } bencho;
573 :
574 : struct {
575 : ulong accounts_cnt;
576 : int mode;
577 : float contending_fraction;
578 : float cu_price_spread;
579 : } benchg;
580 :
581 : struct {
582 : ushort repair_client_listen_port;
583 : char identity_key_path[ PATH_MAX ];
584 : ulong max_pending_shred_sets;
585 : ulong slot_max;
586 : ulong max_shreds_per_block;
587 :
588 : /* non-config */
589 :
590 : ulong repair_sign_depth;
591 : ulong repair_sign_cnt;
592 : } repair;
593 :
594 : struct {
595 : ushort repair_client_listen_port;
596 : char identity_key_path[ PATH_MAX ];
597 : ulong slot_max;
598 : ulong max_shreds_per_block;
599 :
600 : ulong repair_sign_depth;
601 : ulong repair_sign_cnt;
602 : } rotor;
603 :
604 : struct {
605 : ushort repair_serve_listen_port;
606 : char identity_key_path[ PATH_MAX ];
607 : ulong ping_cache_entries;
608 : ulong max_shreds_per_block;
609 : } rserve;
610 :
611 : struct {
612 : ushort txsend_src_port;
613 :
614 : /* non-config */
615 :
616 : uint ip_addr;
617 : char identity_key_path[ PATH_MAX ];
618 : } txsend;
619 :
620 : struct {
621 : uint fake_dst_ip;
622 : } pktgen;
623 :
624 : struct {
625 : char ledger_format[ 16 ];
626 : char ledger_path[ PATH_MAX ];
627 : ulong end_slot;
628 : ulong root_distance;
629 : int alpenglow;
630 : long boot_timestamp_nanos;
631 : } backtest;
632 :
633 : struct {
634 : char ledger_format[ 16 ];
635 : char ledger_path[ PATH_MAX ];
636 : ulong end_slot;
637 : ushort shred_listen_port;
638 : } forktest;
639 :
640 : struct {
641 : ulong accdb_obj_id;
642 :
643 : ulong authorized_voter_paths_cnt;
644 : char authorized_voter_paths[ 16 ][ PATH_MAX ];
645 : int hard_fork_fatal;
646 : int wait_for_supermajority;
647 : ulong max_live_slots;
648 : char identity_key[ PATH_MAX ];
649 : char vote_account[ PATH_MAX ];
650 : char base_path[PATH_MAX];
651 : ulong max_shreds_per_block;
652 : } tower;
653 :
654 : struct {
655 : char identity_key_path[ PATH_MAX ];
656 : ushort quic_client_listen_port;
657 : ushort quic_server_listen_port;
658 : uint ip_addr;
659 : ulong max_live_slots;
660 : } votor;
661 :
662 : struct {
663 : ulong accdb_obj_id;
664 : ulong max_live_slots;
665 :
666 : ulong rpc_epoch_obj_id;
667 : ulong resolv_epoch_obj_ids[ 16 ];
668 : ulong resolv_epoch_obj_cnt;
669 : ulong snapmk_epoch_obj_id;
670 : ulong snapzp_epoch_obj_ids[ 64 ];
671 : ulong snapzp_epoch_obj_cnt;
672 : } accdb;
673 :
674 : struct {
675 : ulong max_live_slots;
676 : ulong accdb_obj_id;
677 : ulong accdb_epoch_fseq_obj_id;
678 : } resolv;
679 :
680 :
681 : #define FD_TOPO_SNAPSHOTS_GOSSIP_LIST_MAX (32UL)
682 63 : #define FD_TOPO_SNAPSHOTS_SERVERS_MAX (16UL)
683 63 : #define FD_TOPO_MAX_RESOLVED_ADDRS ( 4UL)
684 63 : #define FD_TOPO_SNAPSHOTS_SERVERS_MAX_RESOLVED (FD_TOPO_MAX_RESOLVED_ADDRS*FD_TOPO_SNAPSHOTS_SERVERS_MAX)
685 :
686 : struct fd_topo_tile_snapct {
687 : char snapshots_path[ PATH_MAX ];
688 :
689 : ulong entrypoints_cnt;
690 : char entrypoints[ FD_TOPO_GOSSIP_ENTRYPOINTS_MAX ][ FD_HOSTPORT_BUF_MAX ];
691 :
692 : struct {
693 : uint max_local_full_effective_age;
694 : uint max_local_incremental_age;
695 :
696 : struct {
697 : int allow_any;
698 : ulong allow_list_cnt;
699 : fd_pubkey_t allow_list[ FD_TOPO_SNAPSHOTS_GOSSIP_LIST_MAX ];
700 : ulong block_list_cnt;
701 : fd_pubkey_t block_list[ FD_TOPO_SNAPSHOTS_GOSSIP_LIST_MAX ];
702 : } gossip;
703 :
704 : ulong servers_cnt;
705 : char servers[ FD_TOPO_SNAPSHOTS_SERVERS_MAX ][ FD_URL_MAX ];
706 : } sources;
707 :
708 : int incremental_snapshots;
709 : uint max_full_snapshots_to_keep;
710 : uint max_incremental_snapshots_to_keep;
711 : uint max_retry_abort;
712 : long wait_for_peers_timeout_nanos;
713 : } snapct;
714 :
715 : struct {
716 : char snapshots_path[ PATH_MAX ];
717 : int incremental_snapshots;
718 : uint min_download_speed_mibs;
719 : } snapld;
720 :
721 : struct {
722 : ulong max_live_slots;
723 : ulong accdb_obj_id;
724 : ulong txncache_obj_id;
725 : ulong banks_obj_id;
726 : ulong max_txn_per_slot;
727 : } snapin;
728 :
729 : struct {
730 : ulong partition_sz;
731 : } snapwr;
732 :
733 : struct {
734 :
735 : uint bind_address;
736 : ushort bind_port;
737 :
738 : ushort expected_shred_version;
739 : ulong entrypoints_cnt;
740 : char entrypoints[ FD_TOPO_GOSSIP_ENTRYPOINTS_MAX ][ FD_HOSTPORT_BUF_MAX ];
741 : } ipecho;
742 :
743 : struct {
744 : ulong max_live_slots;
745 : ulong txncache_obj_id;
746 : ulong progcache_obj_id;
747 : ulong accdb_obj_id;
748 : int report_runtime_diffs;
749 : } execle;
750 :
751 : struct {
752 : int validate_genesis_hash;
753 : int allow_download;
754 :
755 : ushort expected_shred_version;
756 : ulong entrypoints_cnt;
757 : char entrypoints[ FD_TOPO_GOSSIP_ENTRYPOINTS_MAX ][ FD_HOSTPORT_BUF_MAX ];
758 :
759 : int has_expected_genesis_hash;
760 : uchar expected_genesis_hash[ 32UL ];
761 :
762 : char genesis_path[ PATH_MAX ];
763 :
764 : uint target_gid;
765 : uint target_uid;
766 :
767 : ulong max_live_slots;
768 : ulong accdb_obj_id;
769 : ulong max_message_size;
770 : } genesi;
771 :
772 : struct {
773 : ulong capture_start_slot;
774 : char solcap_capture[ PATH_MAX ];
775 : int recent_only;
776 : ulong recent_slots_per_file;
777 : } solcap;
778 :
779 : struct {
780 : ulong accdb_obj_id;
781 : ulong accdb_epoch_obj_id;
782 : ulong visited_set_obj_id;
783 : ulong banks_obj_id;
784 : ulong zp_fseq_id;
785 : ulong txncache_obj_id;
786 : ulong max_accounts;
787 : ulong max_live_slots;
788 : ulong max_txn_per_slot;
789 : uint max_full_snapshots_to_keep;
790 : char snapshots_path[ PATH_MAX ];
791 : uint max_incremental_snapshots_to_keep;
792 : } snapmk;
793 :
794 : struct {
795 : ulong accdb_obj_id;
796 : ulong accdb_epoch_obj_id;
797 : ulong visited_set_obj_id;
798 : ulong zp_fseq_id;
799 : ulong max_live_slots;
800 : uint snap_fd_cnt;
801 : } snapzp;
802 :
803 : struct {
804 : ulong accdb_obj_id;
805 : } snaprd;
806 :
807 : struct {
808 : ulong snap_max;
809 : ulong conn_max;
810 : uint io_worker_cnt;
811 : ulong idle_timeout_millis;
812 : ulong send_timeout_millis;
813 : ulong send_buffer_size_kib;
814 : fd_ip6_addr_t listen_addr;
815 : ushort listen_port;
816 : } snapsv;
817 : };
818 : };
819 :
820 : typedef struct fd_topo_tile fd_topo_tile_t;
821 :
822 : typedef struct {
823 : ulong id;
824 : char name[ 13UL ]; /* object type */
825 : ulong wksp_id;
826 :
827 : /* Optional label for object */
828 : char label[ 13UL ]; /* object label */
829 : ulong label_idx; /* index of object for this label (ULONG_MAX if not labelled) */
830 :
831 : ulong offset;
832 : ulong footprint;
833 : } fd_topo_obj_t;
834 :
835 : /* An fd_topo_t represents the overall structure of a Firedancer
836 : configuration, describing all the workspaces, tiles, and links
837 : between them. */
838 : struct fd_topo {
839 : char app_name[ 256UL ];
840 : uchar props[ 32768UL ];
841 :
842 : ulong sleep_obj_id;
843 :
844 : ulong wksp_cnt;
845 : ulong link_cnt;
846 : ulong tile_cnt;
847 : ulong obj_cnt;
848 :
849 : fd_topo_wksp_t workspaces[ FD_TOPO_MAX_WKSPS ];
850 : fd_topo_link_t links[ FD_TOPO_MAX_LINKS ];
851 : fd_topo_tile_t tiles[ FD_TOPO_MAX_TILES ];
852 : fd_topo_obj_t objs[ FD_TOPO_MAX_OBJS ];
853 :
854 : ulong agave_affinity_cnt;
855 : ulong agave_affinity_cpu_idx[ FD_TILE_MAX ];
856 : ulong blocklist_cores_cnt;
857 : ulong blocklist_cores_cpu_idx[ FD_TILE_MAX ];
858 :
859 : ulong max_page_size; /* 2^21 or 2^30 */
860 : ulong gigantic_page_threshold; /* see [hugetlbfs.gigantic_page_threshold_mib]*/
861 :
862 : /* Rendered configuration documents for the boot telemetry event,
863 : filled by the app during topology construction. */
864 : ulong resolved_config_json_len;
865 : char resolved_config_json[ 262144UL ];
866 : ulong user_config_json_len;
867 : char user_config_json[ 131072UL ];
868 :
869 : ulong layout_hash;
870 : };
871 : typedef struct fd_topo fd_topo_t;
872 :
873 : typedef struct {
874 : char const * name;
875 :
876 : int keep_host_networking;
877 : int allow_connect;
878 : int allow_renameat;
879 : ulong rlimit_file_cnt;
880 : ulong rlimit_address_space;
881 : ulong rlimit_data;
882 : ulong rlimit_nproc;
883 : int for_tpool;
884 :
885 : ulong (*max_event_sz )( fd_topo_tile_t const * tile );
886 : ulong (*populate_allowed_seccomp)( fd_topo_t const * topo, fd_topo_tile_t const * tile, ulong out_cnt, struct sock_filter * out );
887 : ulong (*populate_allowed_fds )( fd_topo_t const * topo, fd_topo_tile_t const * tile, ulong out_fds_sz, int * out_fds );
888 : ulong (*scratch_align )( void );
889 : ulong (*scratch_footprint )( fd_topo_tile_t const * tile );
890 : ulong (*loose_footprint )( fd_topo_tile_t const * tile );
891 : void (*privileged_init )( fd_topo_t const * topo, fd_topo_tile_t const * tile );
892 : void (*unprivileged_init )( fd_topo_t const * topo, fd_topo_tile_t const * tile );
893 : void (*run )( fd_topo_t * topo, fd_topo_tile_t * tile );
894 : ulong (*rlimit_file_cnt_fn )( fd_topo_t const * topo, fd_topo_tile_t const * tile );
895 : } fd_topo_run_tile_t;
896 :
897 : struct fd_topo_obj_callbacks {
898 : char const * name;
899 : ulong (* footprint )( fd_topo_t const * topo, fd_topo_obj_t const * obj );
900 : ulong (* align )( fd_topo_t const * topo, fd_topo_obj_t const * obj );
901 : ulong (* loose )( fd_topo_t const * topo, fd_topo_obj_t const * obj );
902 : void (* new )( fd_topo_t const * topo, fd_topo_obj_t const * obj );
903 : };
904 :
905 : typedef struct fd_topo_obj_callbacks fd_topo_obj_callbacks_t;
906 :
907 : FD_PROTOTYPES_BEGIN
908 :
909 : FD_FN_CONST static inline ulong
910 252 : fd_topo_workspace_align( void ) {
911 : /* This needs to be the max( align ) of all the child members that
912 : could be aligned into this workspace, otherwise our footprint
913 : calculation will not be correct. For now just set to 4096 but this
914 : should probably be calculated dynamically, or we should reduce
915 : those child aligns if we can. */
916 252 : return 4096UL;
917 252 : }
918 :
919 : void *
920 : fd_topo_obj_laddr( fd_topo_t const * topo,
921 : ulong obj_id );
922 :
923 : /* Returns a pointer in the local address space to the base address of
924 : the workspace out of which the given object was allocated. */
925 :
926 : static inline void *
927 : fd_topo_obj_wksp_base( fd_topo_t const * topo,
928 0 : ulong obj_id ) {
929 0 : FD_TEST( obj_id<FD_TOPO_MAX_OBJS );
930 0 : fd_topo_obj_t const * obj = &topo->objs[ obj_id ];
931 0 : FD_TEST( obj->id == obj_id );
932 0 : ulong const wksp_id = obj->wksp_id;
933 :
934 0 : FD_TEST( wksp_id<FD_TOPO_MAX_WKSPS );
935 0 : fd_topo_wksp_t const * wksp = &topo->workspaces[ wksp_id ];
936 0 : FD_TEST( wksp->id == wksp_id );
937 0 : return wksp->wksp;
938 0 : }
939 :
940 : FD_FN_PURE static inline ulong
941 : fd_topo_tile_name_cnt( fd_topo_t const * topo,
942 15 : char const * name ) {
943 15 : ulong cnt = 0;
944 30 : for( ulong i=0; i<topo->tile_cnt; i++ ) {
945 15 : if( FD_UNLIKELY( !strcmp( topo->tiles[ i ].name, name ) ) ) cnt++;
946 15 : }
947 15 : return cnt;
948 15 : }
949 :
950 : /* Finds the workspace of a given name in the topology. Returns
951 : ULONG_MAX if there is no such workspace. There can be at most one
952 : workspace of a given name. */
953 :
954 : FD_FN_PURE static inline ulong
955 : fd_topo_find_wksp( fd_topo_t const * topo,
956 609 : char const * name ) {
957 657 : for( ulong i=0; i<topo->wksp_cnt; i++ ) {
958 657 : if( FD_UNLIKELY( !strcmp( topo->workspaces[ i ].name, name ) ) ) return i;
959 657 : }
960 0 : return ULONG_MAX;
961 609 : }
962 :
963 : /* Find the tile of a given name and kind_id in the topology, there will
964 : be at most one such tile, since kind_id is unique among the name.
965 : Returns ULONG_MAX if there is no such tile. */
966 :
967 : FD_FN_PURE static inline ulong
968 : fd_topo_find_tile( fd_topo_t const * topo,
969 : char const * name,
970 417 : ulong kind_id ) {
971 525 : for( ulong i=0; i<topo->tile_cnt; i++ ) {
972 507 : if( FD_UNLIKELY( !strcmp( topo->tiles[ i ].name, name ) ) && topo->tiles[ i ].kind_id == kind_id ) return i;
973 507 : }
974 18 : return ULONG_MAX;
975 417 : }
976 :
977 : /* Find the link of a given name and kind_id in the topology, there will
978 : be at most one such link, since kind_id is unique among the name.
979 : Returns ULONG_MAX if there is no such link. */
980 :
981 : FD_FN_PURE static inline ulong
982 : fd_topo_find_link( fd_topo_t const * topo,
983 : char const * name,
984 234 : ulong kind_id ) {
985 948 : for( ulong i=0; i<topo->link_cnt; i++ ) {
986 948 : if( FD_UNLIKELY( !strcmp( topo->links[ i ].name, name ) ) && topo->links[ i ].kind_id == kind_id ) return i;
987 948 : }
988 0 : return ULONG_MAX;
989 234 : }
990 :
991 : FD_FN_PURE static inline ulong
992 : fd_topo_find_tile_in_link( fd_topo_t const * topo,
993 : fd_topo_tile_t const * tile,
994 : char const * name,
995 0 : ulong kind_id ) {
996 0 : for( ulong i=0; i<tile->in_cnt; i++ ) {
997 0 : if( FD_UNLIKELY( !strcmp( topo->links[ tile->in_link_id[ i ] ].name, name ) )
998 0 : && topo->links[ tile->in_link_id[ i ] ].kind_id == kind_id ) return i;
999 0 : }
1000 0 : return ULONG_MAX;
1001 0 : }
1002 :
1003 : FD_FN_PURE static inline ulong
1004 : fd_topo_find_tile_out_link( fd_topo_t const * topo,
1005 : fd_topo_tile_t const * tile,
1006 : char const * name,
1007 318 : ulong kind_id ) {
1008 984 : for( ulong i=0; i<tile->out_cnt; i++ ) {
1009 984 : if( FD_UNLIKELY( !strcmp( topo->links[ tile->out_link_id[ i ] ].name, name ) )
1010 984 : && topo->links[ tile->out_link_id[ i ] ].kind_id == kind_id ) return i;
1011 984 : }
1012 0 : return ULONG_MAX;
1013 318 : }
1014 :
1015 : /* Find the id of the tile which is a producer for the given link. If
1016 : no tile is a producer for the link, returns ULONG_MAX. This should
1017 : not be possible for a well formed and validated topology. */
1018 : FD_FN_PURE static inline ulong
1019 : fd_topo_find_link_producer( fd_topo_t const * topo,
1020 0 : fd_topo_link_t const * link ) {
1021 0 : for( ulong i=0; i<topo->tile_cnt; i++ ) {
1022 0 : fd_topo_tile_t const * tile = &topo->tiles[ i ];
1023 :
1024 0 : for( ulong j=0; j<tile->out_cnt; j++ ) {
1025 0 : if( FD_UNLIKELY( tile->out_link_id[ j ] == link->id ) ) return i;
1026 0 : }
1027 0 : }
1028 0 : return ULONG_MAX;
1029 0 : }
1030 :
1031 : /* Given a link, count the number of consumers of that link among all
1032 : the tiles in the topology. */
1033 : FD_FN_PURE static inline ulong
1034 : fd_topo_link_consumer_cnt( fd_topo_t const * topo,
1035 0 : fd_topo_link_t const * link ) {
1036 0 : ulong cnt = 0;
1037 0 : for( ulong i=0; i<topo->tile_cnt; i++ ) {
1038 0 : fd_topo_tile_t const * tile = &topo->tiles[ i ];
1039 0 : for( ulong j=0; j<tile->in_cnt; j++ ) {
1040 0 : if( FD_UNLIKELY( tile->in_link_id[ j ] == link->id ) ) cnt++;
1041 0 : }
1042 0 : }
1043 :
1044 0 : return cnt;
1045 0 : }
1046 :
1047 : /* Given a link, count the number of reliable consumers of that link
1048 : among all the tiles in the topology. */
1049 : FD_FN_PURE static inline ulong
1050 : fd_topo_link_reliable_consumer_cnt( fd_topo_t const * topo,
1051 0 : fd_topo_link_t const * link ) {
1052 0 : ulong cnt = 0;
1053 0 : for( ulong i=0; i<topo->tile_cnt; i++ ) {
1054 0 : fd_topo_tile_t const * tile = &topo->tiles[ i ];
1055 0 : for( ulong j=0; j<tile->in_cnt; j++ ) {
1056 0 : if( FD_UNLIKELY( tile->in_link_id[ j ] == link->id && tile->in_link_reliable[ j ] ) ) cnt++;
1057 0 : }
1058 0 : }
1059 0 :
1060 0 : return cnt;
1061 0 : }
1062 :
1063 : FD_FN_PURE static inline ulong
1064 : fd_topo_tile_consumer_cnt( fd_topo_t const * topo,
1065 0 : fd_topo_tile_t const * tile ) {
1066 0 : (void)topo;
1067 0 : return tile->out_cnt;
1068 0 : }
1069 :
1070 : FD_FN_PURE static inline ulong
1071 : fd_topo_tile_reliable_consumer_cnt( fd_topo_t const * topo,
1072 0 : fd_topo_tile_t const * tile ) {
1073 0 : ulong reliable_cons_cnt = 0UL;
1074 0 : for( ulong i=0UL; i<topo->tile_cnt; i++ ) {
1075 0 : fd_topo_tile_t const * consumer_tile = &topo->tiles[ i ];
1076 0 : for( ulong j=0UL; j<consumer_tile->in_cnt; j++ ) {
1077 0 : for( ulong k=0UL; k<tile->out_cnt; k++ ) {
1078 0 : if( FD_UNLIKELY( consumer_tile->in_link_id[ j ]==tile->out_link_id[ k ] && consumer_tile->in_link_reliable[ j ] ) ) {
1079 0 : reliable_cons_cnt++;
1080 0 : }
1081 0 : }
1082 0 : }
1083 0 : }
1084 0 : return reliable_cons_cnt;
1085 0 : }
1086 :
1087 : FD_FN_PURE static inline ulong
1088 : fd_topo_tile_producer_cnt( fd_topo_t const * topo,
1089 0 : fd_topo_tile_t const * tile ) {
1090 0 : (void)topo;
1091 0 : ulong in_cnt = 0UL;
1092 0 : for( ulong i=0UL; i<tile->in_cnt; i++ ) {
1093 0 : if( FD_UNLIKELY( !tile->in_link_poll[ i ] ) ) continue;
1094 0 : in_cnt++;
1095 0 : }
1096 0 : return in_cnt;
1097 0 : }
1098 :
1099 : FD_FN_PURE FD_FN_UNUSED static ulong
1100 : fd_topo_obj_cnt( fd_topo_t const * topo,
1101 : char const * obj_type,
1102 0 : char const * label ) {
1103 0 : ulong cnt = 0UL;
1104 0 : for( ulong i=0UL; i<topo->obj_cnt; i++ ) {
1105 0 : fd_topo_obj_t const * obj = &topo->objs[ i ];
1106 0 : if( strncmp( obj->name, obj_type, sizeof(obj->name) ) ) continue;
1107 0 : if( label &&
1108 0 : strncmp( obj->label, label, sizeof(obj->label) ) ) continue;
1109 0 : cnt++;
1110 0 : }
1111 0 : return cnt;
1112 0 : }
1113 :
1114 : FD_FN_PURE FD_FN_UNUSED static fd_topo_obj_t const *
1115 : fd_topo_find_obj( fd_topo_t const * topo,
1116 : char const * obj_type,
1117 : char const * label,
1118 0 : ulong label_idx ) {
1119 0 : for( ulong i=0UL; i<topo->obj_cnt; i++ ) {
1120 0 : fd_topo_obj_t const * obj = &topo->objs[ i ];
1121 0 : if( strncmp( obj->name, obj_type, sizeof(obj->name) ) ) continue;
1122 0 : if( label &&
1123 0 : strncmp( obj->label, label, sizeof(obj->label) ) ) continue;
1124 0 : if( label_idx != ULONG_MAX && obj->label_idx != label_idx ) continue;
1125 0 : return obj;
1126 0 : }
1127 0 : return NULL;
1128 0 : }
1129 :
1130 : FD_FN_PURE FD_FN_UNUSED static fd_topo_obj_t const *
1131 : fd_topo_find_tile_obj( fd_topo_t const * topo,
1132 : fd_topo_tile_t const * tile,
1133 0 : char const * obj_type ) {
1134 0 : for( ulong i=0UL; i<(tile->uses_obj_cnt); i++ ) {
1135 0 : fd_topo_obj_t const * obj = &topo->objs[ tile->uses_obj_id[ i ] ];
1136 0 : if( strncmp( obj->name, obj_type, sizeof(obj->name) ) ) continue;
1137 0 : return obj;
1138 0 : }
1139 0 : return NULL;
1140 0 : }
1141 :
1142 : /* Join (map into the process) all shared memory (huge/gigantic pages)
1143 : needed by the tile, in the given topology. All memory associated
1144 : with the tile (aka. used by links that the tile either produces to or
1145 : consumes from, or used by the tile itself for its cnc) will be
1146 : attached (mapped into the process).
1147 :
1148 : This is needed to play nicely with the sandbox. Once a process is
1149 : sandboxed we can no longer map any memory. */
1150 : void
1151 : fd_topo_join_tile_workspaces( fd_topo_t * topo,
1152 : fd_topo_tile_t * tile,
1153 : int core_dump_level );
1154 :
1155 : /* Join (map into the process) the shared memory (huge/gigantic pages)
1156 : for the given workspace. Mode is one of
1157 : FD_SHMEM_JOIN_MODE_READ_WRITE or FD_SHMEM_JOIN_MODE_READ_ONLY and
1158 : determines the prot argument that will be passed to mmap when mapping
1159 : the pages in (PROT_WRITE or PROT_READ respectively).
1160 :
1161 : Dump should be set to 1 if the workspace memory should be dumpable
1162 : when the process crashes, or 0 if not. */
1163 : void
1164 : fd_topo_join_workspace( fd_topo_t * topo,
1165 : fd_topo_wksp_t * wksp,
1166 : int mode,
1167 : int dump );
1168 :
1169 : /* Join (map into the process) all shared memory (huge/gigantic pages)
1170 : needed by all tiles in the topology. Mode is one of
1171 : FD_SHMEM_JOIN_MODE_READ_WRITE or FD_SHMEM_JOIN_MODE_READ_ONLY and
1172 : determines the prot argument that will be passed to mmap when
1173 : mapping the pages in (PROT_WRITE or PROT_READ respectively). */
1174 : void
1175 : fd_topo_join_workspaces( fd_topo_t * topo,
1176 : int mode,
1177 : int core_dump_level );
1178 :
1179 : /* Leave (unmap from the process) the shared memory needed for the
1180 : given workspace in the topology, if it was previously mapped.
1181 :
1182 : topo and wksp are assumed non-NULL. It is OK if the workspace
1183 : has not been previously joined, in which case this is a no-op. */
1184 :
1185 : void
1186 : fd_topo_leave_workspace( fd_topo_t * topo,
1187 : fd_topo_wksp_t * wksp );
1188 :
1189 : /* Leave (unmap from the process) all shared memory needed by all
1190 : tiles in the topology, if each of them was mapped.
1191 :
1192 : topo is assumed non-NULL. Only workspaces which were previously
1193 : joined are unmapped. */
1194 :
1195 : void
1196 : fd_topo_leave_workspaces( fd_topo_t * topo );
1197 :
1198 : /* Create the given workspace needed by the topology on the system.
1199 : This does not "join" the workspaces (map their memory into the
1200 : process), but only creates the .wksp file and formats it correctly
1201 : as a workspace.
1202 :
1203 : Returns 0 on success and -1 on failure, with errno set to the error.
1204 : The only reason for failure currently that will be returned is
1205 : ENOMEM, as other unexpected errors will cause the program to exit.
1206 :
1207 : If update_existing is 1, the workspace will not be created from
1208 : scratch but it will be assumed that it already exists from a prior
1209 : run and needs to be maybe resized and then have the header
1210 : structures reinitialized. This can save a very expensive operation
1211 : of zeroing all of the workspace pages. This is dangerous in
1212 : production because it can leave stray memory from prior runs around,
1213 : and should only be used in development environments. */
1214 :
1215 : int
1216 : fd_topo_create_workspace( fd_topo_t * topo,
1217 : fd_topo_wksp_t * wksp,
1218 : int update_existing );
1219 :
1220 : /* Join the standard IPC objects needed by the topology of this particular
1221 : tile */
1222 :
1223 : void
1224 : fd_topo_fill_tile( fd_topo_t * topo,
1225 : fd_topo_tile_t * tile );
1226 :
1227 : /* Same as fd_topo_fill_tile but fills in all the objects for a
1228 : particular workspace with the given mode. */
1229 : void
1230 : fd_topo_workspace_fill( fd_topo_t * topo,
1231 : fd_topo_wksp_t * wksp );
1232 :
1233 : /* Apply a new function to every object that is resident in the given
1234 : workspace in the topology. */
1235 :
1236 : void
1237 : fd_topo_wksp_new( fd_topo_t const * topo,
1238 : fd_topo_wksp_t const * wksp,
1239 : fd_topo_obj_callbacks_t ** callbacks );
1240 :
1241 : /* Same as fd_topo_fill_tile but fills in all tiles in the topology. */
1242 :
1243 : void
1244 : fd_topo_fill( fd_topo_t * topo );
1245 :
1246 : /* fd_topo_tile_stack_join joins a huge page optimized stack for the
1247 : provided tile. The stack is assumed to already exist at a known
1248 : path in the hugetlbfs mount. */
1249 :
1250 : void *
1251 : fd_topo_tile_stack_join( char const * app_name,
1252 : char const * tile_name,
1253 : ulong tile_kind_id );
1254 :
1255 : /* fd_topo_run_single_process runs all the tiles in a single process
1256 : (the calling process). This spawns a thread for each tile, switches
1257 : that thread to the given UID and GID and then runs the tile in it.
1258 : Each thread will never exit, as tiles are expected to run forever.
1259 : An error is logged and the application will exit if a tile exits.
1260 : The function itself does return after spawning all the threads.
1261 :
1262 : The threads will not be sandboxed in any way, except switching to the
1263 : provided UID and GID, so they will share the same address space, and
1264 : not have any seccomp restrictions or use any Linux namespaces. The
1265 : calling thread will also switch to the provided UID and GID before
1266 : it returns.
1267 :
1268 : In production, when running with an Agave child process this is
1269 : used for spawning certain tiles inside the Agave address space.
1270 : It's also useful for tooling and debugging, but is not how the main
1271 : production Firedancer process runs. For production, each tile is run
1272 : in its own address space with a separate process and full security
1273 : sandbox.
1274 :
1275 : The agave argument determines which tiles are started. If the
1276 : argument is 0 or 1, only non-agave (or only agave) tiles are started.
1277 : If the argument is any other value, all tiles in the topology are
1278 : started regardless of if they are Agave tiles or not. */
1279 :
1280 : void
1281 : fd_topo_run_single_process( fd_topo_t * topo,
1282 : int agave,
1283 : uint uid,
1284 : uint gid,
1285 : fd_topo_run_tile_t (* tile_run )( fd_topo_tile_t const * tile ) );
1286 :
1287 : /* fd_topo_run_tile runs the given tile directly within the current
1288 : process (and thread). The function will never return, as tiles are
1289 : expected to run forever. An error is logged and the application will
1290 : exit if the tile exits.
1291 :
1292 : The sandbox argument determines if the current process will be
1293 : sandboxed fully before starting the tile. The thread will switch to
1294 : the UID and GID provided before starting the tile, even if the thread
1295 : is not being sandboxed. Although POSIX specifies that all threads in
1296 : a process must share a UID and GID, this is not the case on Linux.
1297 : The thread will switch to the provided UID and GID without switching
1298 : the other threads in the process.
1299 :
1300 : If keep_controlling_terminal is set to 0, and the sandbox is enabled
1301 : the controlling terminal will be detached as an additional sandbox
1302 : measure, but you will not be able to send Ctrl+C or other signals
1303 : from the terminal. See fd_sandbox.h for more information.
1304 :
1305 : The allow_fd argument is only used if sandbox is true, and is a file
1306 : descriptor which will be allowed to exist in the process. Normally
1307 : the sandbox code rejects and aborts if there is an unexpected file
1308 : descriptor present on boot. This is helpful to allow a parent
1309 : process to be notified on termination of the tile by waiting for a
1310 : pipe file descriptor to get closed.
1311 :
1312 : wait and debugger are both used in debugging. If wait is non-NULL,
1313 : the runner will wait until the value pointed to by wait is non-zero
1314 : before launching the tile. Likewise, if debugger is non-NULL, the
1315 : runner will wait until a debugger is attached before setting the
1316 : value pointed to by debugger to non-zero. These are intended to be
1317 : used as a pair, where many tiles share a waiting reference, and then
1318 : one of the tiles (a tile you want to attach the debugger to) has the
1319 : same reference provided as the debugger, so all tiles will stop and
1320 : wait for the debugger to attach to it before proceeding. */
1321 :
1322 : void
1323 : fd_topo_run_tile( fd_topo_t * topo,
1324 : fd_topo_tile_t * tile,
1325 : int sandbox,
1326 : int keep_controlling_terminal,
1327 : int dumpable,
1328 : uint uid,
1329 : uint gid,
1330 : int allow_fd,
1331 : fd_topo_run_tile_t * tile_run );
1332 :
1333 : /* This is for determining the value of RLIMIT_MLOCK that we need to
1334 : successfully run all tiles in separate processes. The value returned
1335 : is the maximum amount of memory that will be locked with mlock() by
1336 : any individual process in the tree. Specifically, if we have three
1337 : tile processes, and they each need to lock 5, 9, and 2 MiB of memory
1338 : respectively, RLIMIT_MLOCK needs to be 9 MiB to allow all three
1339 : process mlock() calls to succeed.
1340 :
1341 : Tiles lock memory in three ways. Any workspace they are using, they
1342 : lock the entire workspace. Then each tile uses huge pages for the
1343 : stack which are also locked, and finally some tiles use private
1344 : locked mmaps outside the workspace for storing key material. The
1345 : results here include all of this memory together.
1346 :
1347 : The result is not necessarily the amount of memory used by the tile
1348 : process, although it will be quite close. Tiles could potentially
1349 : allocate memory (eg, with brk) without needing to lock it, which
1350 : would not need to included, and some kernel memory that tiles cause
1351 : to be allocated (for example XSK buffers) is also not included. The
1352 : actual amount of memory used will not be less than this value. */
1353 : FD_FN_PURE ulong
1354 : fd_topo_mlock_max_tile( fd_topo_t const * topo );
1355 :
1356 : /* Same as fd_topo_mlock_max_tile, but for loading the entire topology
1357 : into one process, rather than a separate process per tile. This is
1358 : used, for example, by the configuration code when it creates all the
1359 : workspaces, or the monitor that maps the entire system into one
1360 : address space. */
1361 : FD_FN_PURE ulong
1362 : fd_topo_mlock( fd_topo_t const * topo );
1363 :
1364 : /* This returns the number of gigantic pages needed by the topology on
1365 : the provided numa node. It includes pages needed by the workspaces,
1366 : as well as additional allocations like huge pages for process stacks
1367 : and private key storage. */
1368 :
1369 : FD_FN_PURE ulong
1370 : fd_topo_gigantic_page_cnt( fd_topo_t const * topo,
1371 : ulong numa_idx );
1372 :
1373 : /* This returns the number of huge pages in the application needed by
1374 : the topology on the provided numa node. It includes pages needed by
1375 : things placed in the hugetlbfs (workspaces, process stacks). If
1376 : include_anonymous is true, it also includes anonymous hugepages which
1377 : are needed but are not placed in the hugetlbfs. */
1378 :
1379 : FD_FN_PURE ulong
1380 : fd_topo_huge_page_cnt( fd_topo_t const * topo,
1381 : ulong numa_idx,
1382 : int include_anonymous );
1383 :
1384 : /* Returns the number of normal (4 KiB) pages needed by the topology
1385 : for extra allocations like private key storage and XSK rings. */
1386 :
1387 : FD_FN_PURE ulong
1388 : fd_topo_normal_page_cnt( fd_topo_t const * topo );
1389 :
1390 : /* Prints a message describing the topology to an output stream. If
1391 : stdout is true, will be written to stdout, otherwise will be written
1392 : as a NOTICE log message to the log file. */
1393 : void
1394 : fd_topo_print_log( int stdout,
1395 : fd_topo_t * topo );
1396 :
1397 : /* fd_topo_print_json prints the same topology description as
1398 : fd_topo_print_log to stdout, as a JSON document. */
1399 : void
1400 : fd_topo_print_json( fd_topo_t * topo );
1401 :
1402 : FD_PROTOTYPES_END
1403 :
1404 : #endif /* HEADER_fd_src_disco_topo_fd_topo_h */
|