Line data Source code
1 : #define _GNU_SOURCE
2 : #include "run.h"
3 : #include "../../../../flamenco/accdb/fd_accdb.h"
4 : #include "../../../../disco/store/fd_store.h"
5 :
6 : #include <sys/wait.h>
7 : #include "generated/main_seccomp.h"
8 : #if defined(__aarch64__)
9 : #include "generated/pidns_arm64_seccomp.h"
10 : #else
11 : #include "generated/pidns_seccomp.h"
12 : #endif
13 :
14 : #include "../../fd_bootinfo.h"
15 : #include "../../../platform/fd_sys_util.h"
16 : #include "../../../platform/fd_file_util.h"
17 : #include "../../../platform/fd_net_util.h"
18 : #include "../../../../disco/net/fd_net_tile.h"
19 : #include "../../../../discof/backup/fd_backup.h"
20 : #include "../../../../discof/backup/fd_snap_pool.h"
21 : #include "../../../../discof/restore/utils/fd_ssarchive.h"
22 : #include "../../../../disco/waker/fd_waker.h"
23 : #include "../../../../util/pod/fd_pod_format.h"
24 :
25 : #include "../configure/configure.h"
26 : #include "../configure/fd_cpu_isolation.h"
27 :
28 : #include <dirent.h>
29 : #include <sched.h>
30 : #include <stdio.h>
31 : #include <stdlib.h> /* getenv */
32 : #include <poll.h>
33 : #include <unistd.h>
34 : #include <errno.h>
35 : #include <fcntl.h>
36 : #include <sys/prctl.h>
37 : #include <sys/resource.h>
38 : #include <sys/mman.h>
39 : #include <sys/stat.h>
40 : #include <linux/capability.h>
41 :
42 : #include "../../../../util/tile/fd_tile_private.h"
43 :
44 : extern fd_topo_obj_callbacks_t * CALLBACKS[];
45 :
46 0 : #define NAME "run"
47 :
48 : void
49 : run_cmd_perm( args_t * args,
50 : fd_cap_chk_t * chk,
51 0 : config_t const * config ) {
52 0 : (void)args;
53 :
54 0 : ulong mlock_limit = fd_topo_mlock_max_tile( &config->topo );
55 :
56 0 : fd_cap_chk_raise_rlimit( chk, NAME, RLIMIT_MEMLOCK, mlock_limit, "call `rlimit(2)` to increase `RLIMIT_MEMLOCK` so all memory can be locked with `mlock(2)`" );
57 0 : fd_cap_chk_raise_rlimit( chk, NAME, RLIMIT_NICE, 40, "call `setpriority(2)` to increase thread priorities" );
58 0 : fd_cap_chk_raise_rlimit( chk, NAME, RLIMIT_NOFILE, CONFIGURE_NR_OPEN_FILES,
59 0 : "call `rlimit(2) to increase `RLIMIT_NOFILE` to allow more open files for Agave" );
60 0 : fd_cap_chk_cap( chk, NAME, CAP_NET_RAW, "call `socket(2)` to bind to a raw socket for use by XDP" );
61 0 : fd_cap_chk_cap( chk, NAME, CAP_SYS_ADMIN, "call `bpf(2)` with the `BPF_OBJ_GET` command to initialize XDP" );
62 0 : if( fd_sandbox_requires_cap_sys_admin( config->uid, config->gid ) )
63 0 : fd_cap_chk_cap( chk, NAME, CAP_SYS_ADMIN, "call `unshare(2)` with `CLONE_NEWUSER` to sandbox the process in a user namespace" );
64 0 : if( FD_LIKELY( getuid() != config->uid ) )
65 0 : fd_cap_chk_cap( chk, NAME, CAP_SETUID, "call `setresuid(2)` to switch uid to the sandbox user" );
66 0 : if( FD_LIKELY( getgid()!=config->gid ) )
67 0 : fd_cap_chk_cap( chk, NAME, CAP_SETGID, "call `setresgid(2)` to switch gid to the sandbox user" );
68 0 : if( FD_UNLIKELY( config->tiles.metric.prometheus_listen_port<1024 ) )
69 0 : fd_cap_chk_cap( chk, NAME, CAP_NET_BIND_SERVICE, "call `bind(2)` to bind to a privileged port for serving metrics" );
70 0 : if( FD_UNLIKELY( config->tiles.gui.gui_listen_port<1024 ) )
71 0 : fd_cap_chk_cap( chk, NAME, CAP_NET_BIND_SERVICE, "call `bind(2)` to bind to a privileged port for serving the GUI" );
72 0 : }
73 :
74 : struct pidns_clone_args {
75 : config_t const * config;
76 : int * pipefd;
77 : int closefd;
78 : };
79 :
80 : extern char fd_log_private_path[ 1024 ]; /* empty string on start */
81 :
82 : static pid_t pid_namespace;
83 :
84 0 : #define FD_LOG_ERR_NOEXIT(a) do { long _fd_log_msg_now = fd_log_wallclock(); fd_log_private_1( 4, _fd_log_msg_now, __FILE__, __LINE__, __func__, fd_log_private_0 a ); } while(0)
85 :
86 : static void
87 0 : parent_signal( int sig ) {
88 0 : if( FD_LIKELY( pid_namespace ) ) kill( pid_namespace, SIGKILL );
89 :
90 0 : if( -1!=fd_log_private_logfile_fd() ) FD_LOG_ERR_NOEXIT(( "Received signal %s%s%s %s(%s)%s\n%sLog at \"%s\"%s", fd_log_style_bold(), fd_io_strsignal_name( sig ), fd_log_style_normal(), fd_log_style_dim(), fd_io_strsignal_desc( sig ), fd_log_style_normal(), fd_log_style_dim(), fd_log_private_path, fd_log_style_normal() ));
91 0 : else FD_LOG_ERR_NOEXIT(( "Received signal %s%s%s %s(%s)%s", fd_log_style_bold(), fd_io_strsignal_name( sig ), fd_log_style_normal(), fd_log_style_dim(), fd_io_strsignal_desc( sig ), fd_log_style_normal() ));
92 :
93 0 : if( FD_LIKELY( sig==SIGINT ) ) fd_sys_util_exit_group( 128+SIGINT );
94 0 : else fd_sys_util_exit_group( 0 );
95 0 : }
96 :
97 : static void
98 0 : install_parent_signals( void ) {
99 0 : struct sigaction sa = {
100 0 : .sa_handler = parent_signal,
101 0 : .sa_flags = 0,
102 0 : };
103 0 : if( FD_UNLIKELY( sigaction( SIGTERM, &sa, NULL ) ) )
104 0 : FD_LOG_ERR(( "sigaction(SIGTERM) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
105 0 : if( FD_UNLIKELY( sigaction( SIGINT, &sa, NULL ) ) )
106 0 : FD_LOG_ERR(( "sigaction(SIGINT) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
107 :
108 0 : sa.sa_handler = SIG_IGN;
109 0 : if( FD_UNLIKELY( sigaction( SIGUSR1, &sa, NULL ) ) )
110 0 : FD_LOG_ERR(( "sigaction(SIGUSR1) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
111 0 : if( FD_UNLIKELY( sigaction( SIGUSR2, &sa, NULL ) ) )
112 0 : FD_LOG_ERR(( "sigaction(SIGUSR2) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
113 0 : }
114 :
115 : void *
116 0 : create_clone_stack( void ) {
117 0 : ulong mmap_sz = FD_TILE_PRIVATE_STACK_SZ + 2UL*FD_SHMEM_NORMAL_PAGE_SZ;
118 0 : uchar * stack = (uchar *)mmap( NULL, mmap_sz, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, (off_t)0 );
119 0 : if( FD_UNLIKELY( stack==MAP_FAILED ) )
120 0 : FD_LOG_ERR(( "mmap() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
121 :
122 : /* Make space for guard lo and guard hi */
123 0 : if( FD_UNLIKELY( munmap( stack, FD_SHMEM_NORMAL_PAGE_SZ ) ) )
124 0 : FD_LOG_ERR(( "munmap failed (%i-%s)", errno, fd_io_strerror( errno ) ));
125 0 : stack += FD_SHMEM_NORMAL_PAGE_SZ;
126 0 : if( FD_UNLIKELY( munmap( stack + FD_TILE_PRIVATE_STACK_SZ, FD_SHMEM_NORMAL_PAGE_SZ ) ) )
127 0 : FD_LOG_ERR(( "munmap failed (%i-%s)", errno, fd_io_strerror( errno ) ));
128 :
129 : /* Create the guard regions in the extra space */
130 0 : void * guard_lo = (void *)(stack - FD_SHMEM_NORMAL_PAGE_SZ );
131 0 : if( FD_UNLIKELY( mmap( guard_lo, FD_SHMEM_NORMAL_PAGE_SZ, PROT_NONE,
132 0 : MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, (off_t)0 )!=guard_lo ) )
133 0 : FD_LOG_ERR(( "mmap failed (%i-%s)", errno, fd_io_strerror( errno ) ));
134 :
135 0 : void * guard_hi = (void *)(stack + FD_TILE_PRIVATE_STACK_SZ);
136 0 : if( FD_UNLIKELY( mmap( guard_hi, FD_SHMEM_NORMAL_PAGE_SZ, PROT_NONE,
137 0 : MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, (off_t)0 )!=guard_hi ) )
138 0 : FD_LOG_ERR(( "mmap failed (%i-%s)", errno, fd_io_strerror( errno ) ));
139 :
140 0 : return stack;
141 0 : }
142 :
143 :
144 : static int
145 : execve_agave( int config_memfd,
146 0 : int pipefd ) {
147 0 : if( FD_UNLIKELY( -1==fcntl( pipefd, F_SETFD, 0 ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,0) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
148 0 : pid_t child = fork();
149 0 : if( FD_UNLIKELY( -1==child ) ) FD_LOG_ERR(( "fork() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
150 0 : if( FD_LIKELY( !child ) ) {
151 0 : char _current_executable_path[ PATH_MAX ];
152 0 : FD_TEST( -1!=fd_file_util_self_exe( _current_executable_path ) );
153 :
154 0 : char config_fd[ 32 ];
155 0 : FD_TEST( fd_cstr_printf_check( config_fd, sizeof( config_fd ), NULL, "%d", config_memfd ) );
156 0 : char * args[ 5 ] = { _current_executable_path, "run-agave", "--config-fd", config_fd, NULL };
157 :
158 0 : char * envp[] = { NULL, NULL };
159 0 : char * google_creds = getenv( "GOOGLE_APPLICATION_CREDENTIALS" );
160 0 : char provide_creds[ PATH_MAX+30UL ];
161 0 : if( FD_UNLIKELY( google_creds ) ) {
162 0 : FD_TEST( fd_cstr_printf_check( provide_creds, sizeof( provide_creds ), NULL, "GOOGLE_APPLICATION_CREDENTIALS=%s", google_creds ) );
163 0 : envp[ 0 ] = provide_creds;
164 0 : }
165 :
166 0 : if( FD_UNLIKELY( -1==execve( _current_executable_path, args, envp ) ) ) FD_LOG_ERR(( "execve() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
167 0 : } else {
168 0 : if( FD_UNLIKELY( -1==fcntl( pipefd, F_SETFD, FD_CLOEXEC ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,FD_CLOEXEC) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
169 0 : return child;
170 0 : }
171 0 : return 0;
172 0 : }
173 :
174 : static int
175 0 : cgroup_procs_write( char const * path ) {
176 0 : int fd = open( path, O_WRONLY );
177 0 : if( FD_UNLIKELY( fd<0 ) ) {
178 0 : if( FD_LIKELY( errno==ENOENT ) ) return 0;
179 0 : FD_LOG_ERR(( "open(%s) failed (%i-%s)", path, errno, fd_io_strerror( errno ) ));
180 0 : }
181 :
182 0 : char pid[ 32 ];
183 0 : ulong pid_len;
184 0 : FD_TEST( fd_cstr_printf_check( pid, sizeof(pid), &pid_len, "%ld", (long)getpid() ) );
185 0 : if( FD_UNLIKELY( write( fd, pid, pid_len )!=(long)pid_len ) )
186 0 : FD_LOG_ERR(( "write(%s,\"%s\") failed (%i-%s)", path, pid, errno, fd_io_strerror( errno ) ));
187 0 : if( FD_UNLIKELY( close( fd ) ) ) FD_LOG_ERR(( "close(%s) failed (%i-%s)", path, errno, fd_io_strerror( errno ) ));
188 0 : return 1;
189 0 : }
190 :
191 : struct spawn_cgroup {
192 : int probed; /* isolation cgroup existence checked, original saved */
193 : int present; /* isolation cgroup exists */
194 : int joined; /* currently a member of the isolation cgroup */
195 : char isolation[ PATH_MAX ];
196 : char original[ PATH_MAX ];
197 : };
198 :
199 : static void
200 : join_isolation_cgroup( char const * name,
201 0 : struct spawn_cgroup * cg ) {
202 0 : if( FD_UNLIKELY( !cg->probed ) ) {
203 0 : cg->probed = 1;
204 0 : FD_TEST( fd_cstr_printf_check( cg->isolation, sizeof(cg->isolation), NULL, "/sys/fs/cgroup/%s/cgroup.procs", name ) );
205 :
206 : /* The cpuset stage not being configured (no cgroup) is the common
207 : case and must be decided FIRST: on cgroup v1-only or hybrid
208 : hosts /proc/self/cgroup does not have the v2 format, and
209 : validating it before knowing the stage is even in use would
210 : turn every tile launch on such hosts into a fatal error. */
211 0 : cg->present = !access( cg->isolation, F_OK );
212 0 : if( FD_LIKELY( !cg->present ) ) return;
213 :
214 : /* Remember where we came from. The v2 entry in /proc/self/cgroup
215 : is the line "0::<path>"; on a pure v2 hierarchy it is the only
216 : line, but on hybrid systems v1 controller lines precede it, so
217 : search rather than assume. */
218 0 : char buf[ 4096 ];
219 0 : int fd = open( "/proc/self/cgroup", O_RDONLY );
220 0 : if( FD_UNLIKELY( fd<0 ) ) FD_LOG_ERR(( "open(/proc/self/cgroup) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
221 0 : long n = read( fd, buf, sizeof(buf)-1UL );
222 0 : if( FD_UNLIKELY( n<0L ) ) FD_LOG_ERR(( "read(/proc/self/cgroup) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
223 0 : if( FD_UNLIKELY( close( fd ) ) ) FD_LOG_ERR(( "close(/proc/self/cgroup) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
224 0 : buf[ n ] = '\0';
225 :
226 0 : char * line = buf;
227 0 : while( line && strncmp( line, "0::", 3UL ) ) {
228 0 : line = strchr( line, '\n' );
229 0 : if( FD_LIKELY( line ) ) line++;
230 0 : }
231 0 : if( FD_UNLIKELY( !line || !line[ 0 ] ) )
232 0 : FD_LOG_ERR(( "no cgroup v2 entry in /proc/self/cgroup while the cpuset isolation cgroup `/sys/fs/cgroup/%s` "
233 0 : "exists. Remove it with `%s configure fini cpuset`", name, FD_BINARY_NAME ));
234 0 : char * nl = strchr( line, '\n' ); if( FD_LIKELY( nl ) ) *nl = '\0';
235 0 : FD_TEST( fd_cstr_printf_check( cg->original, PATH_MAX, NULL,
236 0 : "/sys/fs/cgroup%s/cgroup.procs", line+3UL ) );
237 0 : }
238 :
239 0 : if( FD_UNLIKELY( !cg->present || cg->joined ) ) return;
240 0 : cg->joined = cgroup_procs_write( cg->isolation );
241 0 : }
242 :
243 : static void
244 0 : leave_isolation_cgroup( struct spawn_cgroup * cg ) {
245 0 : if( FD_LIKELY( !cg->joined ) ) return;
246 0 : if( FD_UNLIKELY( !cgroup_procs_write( cg->original ) ) ) FD_LOG_ERR(( "could not return to original cgroup `%s`", cg->original ));
247 0 : cg->joined = 0;
248 0 : }
249 :
250 : static pid_t
251 : execve_tile( char const * name,
252 : fd_topo_tile_t const * tile,
253 : fd_cpuset_t const * float_cpu_set,
254 : fd_cpuset_t const * floating_cpu_set,
255 : int floating_priority,
256 : int config_memfd,
257 : int pipefd,
258 0 : struct spawn_cgroup * cg ) {
259 0 : FD_CPUSET_DECL( cpu_set );
260 0 : if( FD_LIKELY( tile->cpu_idx!=ULONG_MAX ) ) {
261 : /* Join the cpuset isolation cgroup (if configured) and set the
262 : thread affinity before we clone the new process, to ensure
263 : kernel first touch happens on the desired thread. The child
264 : inherits both. */
265 0 : join_isolation_cgroup( name, cg );
266 0 : if( FD_UNLIKELY( tile->floats ) ) {
267 : /* the floating CPUs on this tile's NUMA node: memory was placed
268 : by cpu_idx */
269 0 : ulong numa_idx = fd_shmem_numa_idx( tile->cpu_idx );
270 0 : for( ulong cpu=0UL; cpu<FD_TILE_MAX; cpu++ )
271 0 : if( fd_cpuset_test( float_cpu_set, cpu ) && fd_shmem_numa_idx( cpu )==numa_idx ) fd_cpuset_insert( cpu_set, cpu );
272 0 : }
273 0 : if( FD_UNLIKELY( !fd_cpuset_cnt( cpu_set ) ) ) fd_cpuset_insert( cpu_set, tile->cpu_idx );
274 0 : if( FD_UNLIKELY( -1==setpriority( PRIO_PROCESS, 0, -19 ) ) ) FD_LOG_ERR(( "setpriority() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
275 0 : } else {
276 0 : leave_isolation_cgroup( cg );
277 0 : fd_memcpy( cpu_set, floating_cpu_set, fd_cpuset_footprint() );
278 0 : if( FD_UNLIKELY( -1==setpriority( PRIO_PROCESS, 0, floating_priority ) ) ) FD_LOG_ERR(( "setpriority() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
279 0 : }
280 :
281 0 : if( FD_UNLIKELY( fd_cpuset_setaffinity( 0, cpu_set ) ) ) {
282 0 : if( FD_LIKELY( errno==EINVAL ) ) {
283 0 : FD_LOG_ERR(( "Unable to set the thread affinity for tile %s:%lu on cpu %lu. It is likely that the affinity "
284 0 : "you have specified for this tile in [layout.affinity] of your configuration file contains a "
285 0 : "CPU (%lu) which does not exist on this machine.",
286 0 : tile->name, tile->kind_id, tile->cpu_idx, tile->cpu_idx ));
287 0 : } else {
288 0 : FD_LOG_ERR(( "sched_setaffinity failed (%i-%s)", errno, fd_io_strerror( errno ) ));
289 0 : }
290 0 : }
291 :
292 : /* Clear CLOEXEC on the side of the pipe we want to pass to the tile. */
293 0 : if( FD_UNLIKELY( -1==fcntl( pipefd, F_SETFD, 0 ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,0) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
294 0 : pid_t child = fork();
295 0 : if( FD_UNLIKELY( -1==child ) ) FD_LOG_ERR(( "fork() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
296 0 : if( FD_LIKELY( !child ) ) {
297 0 : char _current_executable_path[ PATH_MAX ];
298 0 : FD_TEST( -1!=fd_file_util_self_exe( _current_executable_path ) );
299 :
300 0 : char kind_id[ 32 ], config_fd[ 32 ], pipe_fd[ 32 ];
301 0 : FD_TEST( fd_cstr_printf_check( kind_id, sizeof( kind_id ), NULL, "%lu", tile->kind_id ) );
302 0 : FD_TEST( fd_cstr_printf_check( config_fd, sizeof( config_fd ), NULL, "%d", config_memfd ) );
303 0 : FD_TEST( fd_cstr_printf_check( pipe_fd, sizeof( pipe_fd ), NULL, "%d", pipefd ) );
304 0 : char const * args[ 9 ] = { _current_executable_path, "run1", tile->name, kind_id, "--pipe-fd", pipe_fd, "--config-fd", config_fd, NULL };
305 0 : if( FD_UNLIKELY( -1==execve( _current_executable_path, (char **)args, NULL ) ) ) FD_LOG_ERR(( "execve() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
306 0 : } else {
307 0 : if( FD_UNLIKELY( -1==fcntl( pipefd, F_SETFD, FD_CLOEXEC ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,FD_CLOEXEC) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
308 0 : return child;
309 0 : }
310 0 : return 0;
311 0 : }
312 :
313 : int
314 0 : main_pid_namespace( void * _args ) {
315 0 : struct pidns_clone_args * args = _args;
316 0 : if( FD_UNLIKELY( close( args->pipefd[ 0 ] ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
317 0 : if( FD_UNLIKELY( -1!=args->closefd ) ) {
318 0 : if( FD_UNLIKELY( close( args->closefd ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
319 0 : }
320 :
321 0 : config_t const * config = args->config;
322 :
323 0 : fd_log_thread_set( "pidns" );
324 0 : ulong pid = fd_sandbox_getpid(); /* Need to read /proc again.. we got a new PID from clone */
325 0 : fd_log_private_group_id_set( pid );
326 0 : fd_log_private_thread_id_set( pid );
327 0 : fd_log_private_stack_discover( FD_TILE_PRIVATE_STACK_SZ,
328 0 : &fd_tile_private_stack0, &fd_tile_private_stack1 );
329 :
330 0 : if( FD_UNLIKELY( !config->development.sandbox ) ) {
331 : /* If no sandbox, then there's no actual PID namespace so we can't
332 : wait() grandchildren for the exit code. Do this as a workaround. */
333 0 : if( FD_UNLIKELY( -1==prctl( PR_SET_CHILD_SUBREAPER, 1, 0, 0, 0 ) ) )
334 0 : FD_LOG_ERR(( "prctl(PR_SET_CHILD_SUBREAPER) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
335 0 : }
336 :
337 : /* Save the current affinity, it will be restored after creating any child tiles */
338 0 : FD_CPUSET_DECL( floating_cpu_set );
339 0 : if( FD_UNLIKELY( fd_cpuset_getaffinity( 0, floating_cpu_set ) ) )
340 0 : FD_LOG_ERR(( "fd_cpuset_getaffinity failed (%i-%s)", errno, fd_io_strerror( errno ) ));
341 :
342 : /* The CPUs of the floating tiles (efficient mode): floaters share
343 : these among themselves and never a pinned tile's CPU */
344 0 : FD_CPUSET_DECL( float_cpu_set );
345 0 : int any_floats = 0;
346 0 : for( ulong i=0UL; i<config->topo.tile_cnt; i++ ) {
347 0 : if( FD_LIKELY( !config->topo.tiles[ i ].floats ) ) continue;
348 0 : fd_cpuset_insert( float_cpu_set, config->topo.tiles[ i ].cpu_idx );
349 0 : any_floats = 1;
350 0 : }
351 0 : for( ulong i=0UL; i<config->topo.tile_cnt; i++ )
352 0 : if( FD_LIKELY( !config->topo.tiles[ i ].floats && config->topo.tiles[ i ].cpu_idx!=ULONG_MAX ) ) fd_cpuset_remove( float_cpu_set, config->topo.tiles[ i ].cpu_idx );
353 :
354 0 : pid_t child_pids[ FD_TOPO_MAX_TILES+1 ];
355 0 : ulong actual_pids[ FD_TOPO_MAX_TILES+1 ];
356 0 : for( ulong i=0UL; i<FD_TOPO_MAX_TILES+1; i++ ) actual_pids[ i ] = ULONG_MAX;
357 0 : char child_names[ FD_TOPO_MAX_TILES+1 ][ 32 ];
358 0 : ulong child_idxs[ FD_TOPO_MAX_TILES+1 ];
359 0 : struct pollfd fds[ FD_TOPO_MAX_TILES+2 ];
360 :
361 0 : int config_memfd = fd_config_to_memfd( config );
362 0 : if( FD_UNLIKELY( -1==config_memfd ) ) FD_LOG_ERR(( "fd_config_to_memfd() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
363 :
364 0 : int need_mlx5 = 0==strcmp( config->net.provider, "mlx5" );
365 0 : fd_mlx5_fds_t mlx5_fds = { .cmd_fd=-1, .async_fd=-1 };
366 0 : if( need_mlx5 ) {
367 0 : fd_topo_install_mlx5( (fd_topo_t *)&config->topo, &mlx5_fds );
368 0 : }
369 :
370 0 : ulong child_cnt = 0UL;
371 0 : if( FD_LIKELY( !config->is_firedancer && !config->development.no_agave ) ) {
372 0 : int pipefd[ 2 ];
373 0 : if( FD_UNLIKELY( pipe2( pipefd, O_CLOEXEC ) ) ) FD_LOG_ERR(( "pipe2() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
374 0 : fds[ child_cnt ] = (struct pollfd){ .fd = pipefd[ 0 ], .events = 0 };
375 0 : child_pids[ child_cnt ] = execve_agave( config_memfd, pipefd[ 1 ] );
376 0 : FD_TEST( child_pids[ child_cnt ]>0 );
377 0 : actual_pids[ child_cnt ] = (ulong)child_pids[ child_cnt ];
378 0 : child_idxs[ child_cnt ] = ULONG_MAX;
379 0 : if( FD_UNLIKELY( close( pipefd[ 1 ] ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
380 0 : strncpy( child_names[ child_cnt ], "agave", 32 );
381 0 : child_cnt++;
382 0 : }
383 :
384 0 : errno = 0;
385 0 : int save_priority = getpriority( PRIO_PROCESS, 0 );
386 0 : if( FD_UNLIKELY( -1==save_priority && errno ) ) FD_LOG_ERR(( "getpriority() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
387 :
388 0 : int need_xdp = 0==strcmp( config->net.provider, "xdp" );
389 0 : fd_xdp_fds_t xdp_fds[ FD_TOPO_XDP_FDS_MAX ];
390 0 : uint xdp_fds_cnt = FD_TOPO_XDP_FDS_MAX;
391 0 : if( need_xdp ) {
392 0 : fd_topo_install_xdp( &config->topo, xdp_fds, &xdp_fds_cnt, config->net.bind_address_parsed, 0 );
393 0 : }
394 :
395 0 : initialize_accdb_fd( config );
396 0 : initialize_store_fds( config );
397 0 : ulong store_obj_id = fd_pod_query_ulong( config->topo.props, "store", ULONG_MAX );
398 0 : int has_store = store_obj_id!=ULONG_MAX;
399 0 : ulong snap_max = 0UL;
400 0 : int snapshot_upload_enabled = 0;
401 0 : int snapshot_dio_enabled = 0;
402 0 : if( config->is_firedancer ) {
403 0 : snap_max = initialize_snapshot_fds( config );
404 0 : snapshot_upload_enabled = fd_topo_find_tile( &config->topo, "snapsv", 0UL )!=ULONG_MAX;
405 0 : snapshot_dio_enabled = fd_topo_find_tile( &config->topo, "snapzp", 0UL )!=ULONG_MAX;
406 0 : }
407 :
408 0 : ulong waker_client_cnt = 0UL;
409 0 : for( ulong i=0UL; i<config->topo.tile_cnt; i++ ) {
410 0 : ulong idx = config->topo.tiles[ i ].waker_client_idx;
411 0 : if( FD_UNLIKELY( idx!=ULONG_MAX ) ) waker_client_cnt = fd_ulong_max( waker_client_cnt, idx+1UL );
412 0 : }
413 0 : fd_waker_install( waker_client_cnt );
414 :
415 0 : struct spawn_cgroup spawn_cg = {0};
416 :
417 0 : for( ulong pass=0UL; pass<2UL; pass++ ) {
418 0 : for( ulong i=0UL; i<config->topo.tile_cnt; i++ ) {
419 0 : fd_topo_tile_t const * tile = &config->topo.tiles[ i ];
420 0 : if( FD_UNLIKELY( tile->is_agave ) ) continue;
421 0 : if( FD_UNLIKELY( (tile->cpu_idx!=ULONG_MAX)!=pass ) ) continue;
422 :
423 0 : if( need_xdp ) {
424 0 : if( FD_UNLIKELY( strcmp( tile->name, "net" ) ) ) {
425 0 : for( uint i=0U; i<xdp_fds_cnt; i++ ) {
426 : /* close XDP related file descriptors */
427 0 : if( FD_UNLIKELY( -1==fcntl( xdp_fds[i].xsk_map_fd, F_SETFD, FD_CLOEXEC ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,FD_CLOEXEC) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
428 0 : if( FD_UNLIKELY( -1==fcntl( xdp_fds[i].prog_link_fd, F_SETFD, FD_CLOEXEC ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,FD_CLOEXEC) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
429 0 : }
430 0 : } else {
431 0 : for( uint i=0U; i<xdp_fds_cnt; i++ ) {
432 0 : if( FD_UNLIKELY( -1==fcntl( xdp_fds[i].xsk_map_fd, F_SETFD, 0 ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,0) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
433 0 : if( FD_UNLIKELY( -1==fcntl( xdp_fds[i].prog_link_fd, F_SETFD, 0 ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,0) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
434 0 : }
435 0 : }
436 0 : }
437 :
438 0 : if( need_mlx5 ) {
439 0 : int const fd_flags = strcmp( tile->name, "mlx5" ) ? FD_CLOEXEC : 0;
440 0 : if( FD_UNLIKELY( -1==fcntl( mlx5_fds.cmd_fd, F_SETFD, fd_flags ) ) ) {
441 0 : FD_LOG_ERR(( "fcntl(F_SETFD) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
442 0 : }
443 0 : if( FD_UNLIKELY( -1==fcntl( mlx5_fds.async_fd, F_SETFD, fd_flags ) ) ) {
444 0 : FD_LOG_ERR(( "fcntl(F_SETFD) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
445 0 : }
446 0 : }
447 :
448 0 : if( FD_LIKELY( config->is_firedancer ) ) {
449 0 : int tile_uses_accdb = 0;
450 0 : int tile_uses_accdb_ro = 0;
451 0 : for( ulong i=0UL; i<tile->uses_obj_cnt; i++ ) {
452 0 : fd_topo_obj_t const * obj = &config->topo.objs[ tile->uses_obj_id[ i ] ];
453 0 : if( FD_UNLIKELY( !strcmp( obj->name, "accdb" ) ) ) {
454 0 : if( FD_UNLIKELY( tile->uses_obj_mode[ i ]==FD_SHMEM_JOIN_MODE_READ_ONLY ) ) tile_uses_accdb_ro = 1;
455 0 : else tile_uses_accdb = 1;
456 0 : break;
457 0 : }
458 0 : }
459 :
460 : /* The gui joins the accdb shmem read-only (for partition stats)
461 : but never reads account data from the on-disk file, so it does
462 : not need the accounts.db fd. Withhold it to keep the gui at
463 : least privilege. */
464 0 : if( FD_UNLIKELY( !strcmp( tile->name, "gui" ) ) ) tile_uses_accdb_ro = 0;
465 0 : if( FD_UNLIKELY( !strcmp( tile->name, "snapmk" ) ) ) tile_uses_accdb = tile_uses_accdb_ro = 0;
466 :
467 : /* snapwr writes accdb pwrite()s without joining accdb shmem, so
468 : it needs the RW fd despite not appearing as an accdb obj user
469 : in the topology. */
470 0 : if( FD_UNLIKELY( tile_uses_accdb || !strcmp( tile->name, "snapwr" ) ) ) {
471 0 : if( FD_UNLIKELY( -1==fcntl( FD_ACCDB_FD_RW, F_SETFD, 0 ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,0) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
472 0 : } else {
473 0 : if( FD_UNLIKELY( -1==fcntl( FD_ACCDB_FD_RW, F_SETFD, FD_CLOEXEC ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,FD_CLOEXEC) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
474 0 : }
475 :
476 0 : if( FD_UNLIKELY( tile_uses_accdb_ro ) ) {
477 0 : if( FD_UNLIKELY( -1==fcntl( FD_ACCDB_FD_RO, F_SETFD, 0 ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,0) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
478 0 : } else {
479 0 : if( FD_UNLIKELY( -1==fcntl( FD_ACCDB_FD_RO, F_SETFD, FD_CLOEXEC ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,FD_CLOEXEC) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
480 0 : }
481 :
482 0 : if( FD_LIKELY( has_store ) ) {
483 0 : int tile_uses_store = 0;
484 0 : for( ulong i=0UL; i<tile->uses_obj_cnt; i++ ) tile_uses_store |= tile->uses_obj_id[ i ]==store_obj_id;
485 0 : int tile_uses_store_rw = tile_uses_store &&
486 0 : (!strcmp( tile->name, "shred" ) ||
487 0 : !strcmp( tile->name, "backt" ) ||
488 0 : !strcmp( tile->name, "rserve" ));
489 0 : int tile_uses_store_ro = tile_uses_store && !strcmp( tile->name, "replay" );
490 0 : if( FD_UNLIKELY( fcntl( FD_STORE_FD_RW, F_SETFD, tile_uses_store_rw ? 0 : FD_CLOEXEC )<0 ) )
491 0 : FD_LOG_ERR(( "fcntl(FD_STORE_FD_RW,F_SETFD) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
492 0 : if( FD_UNLIKELY( fcntl( FD_STORE_FD_RO, F_SETFD, tile_uses_store_ro ? 0 : FD_CLOEXEC )<0 ) )
493 0 : FD_LOG_ERR(( "fcntl(FD_STORE_FD_RO,F_SETFD) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
494 0 : }
495 :
496 0 : int tile_uses_snap_fd = !strcmp( tile->name, "snapct" ) ||
497 0 : !strcmp( tile->name, "snapmk" );
498 0 : int tile_uses_snap_dio_fd = !strcmp( tile->name, "snapzp" );
499 0 : int tile_uses_snap_rd_fd = !strcmp( tile->name, "snapsv" );
500 0 : for( ulong j=0UL; j<snap_max; j++ ) {
501 0 : if( FD_UNLIKELY( -1==fcntl( FD_SNAP_FD( j ), F_SETFD, tile_uses_snap_fd ? 0 : FD_CLOEXEC ) ) )
502 0 : FD_LOG_ERR(( "fcntl(F_SETFD) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
503 0 : if( snapshot_dio_enabled ) {
504 0 : if( FD_UNLIKELY( -1==fcntl( FD_SNAP_DIO_FD( j ), F_SETFD, tile_uses_snap_dio_fd ? 0 : FD_CLOEXEC ) ) )
505 0 : FD_LOG_ERR(( "fcntl(F_SETFD) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
506 0 : }
507 0 : if( snapshot_upload_enabled ) {
508 0 : if( FD_UNLIKELY( -1==fcntl( FD_SNAP_RO_FD( j ), F_SETFD, tile_uses_snap_rd_fd ? 0 : FD_CLOEXEC ) ) )
509 0 : FD_LOG_ERR(( "fcntl(F_SETFD) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
510 0 : }
511 0 : }
512 0 : }
513 :
514 0 : int is_waker = !strcmp( tile->name, "waker" );
515 0 : int outer_entitled = is_waker || tile->waker_client_idx!=ULONG_MAX;
516 0 : if( FD_UNLIKELY( -1==fcntl( FD_WAKER_OUTER_FD, F_SETFD, outer_entitled ? 0 : FD_CLOEXEC ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
517 0 : for( ulong j=0UL; j<waker_client_cnt; j++ ) {
518 0 : int inner_entitled = is_waker || tile->waker_client_idx==j;
519 0 : if( FD_UNLIKELY( -1==fcntl( FD_WAKER_INNER_FD( j ), F_SETFD, inner_entitled ? 0 : FD_CLOEXEC ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
520 0 : }
521 :
522 0 : int pipefd[ 2 ];
523 0 : if( FD_UNLIKELY( pipe2( pipefd, O_CLOEXEC ) ) ) FD_LOG_ERR(( "pipe2() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
524 0 : fds[ child_cnt ] = (struct pollfd){ .fd = pipefd[ 0 ], .events = 0 };
525 :
526 0 : int floating_priority = ( any_floats && !strcmp( tile->name, "waker" ) ) ? -19 : save_priority;
527 0 : child_pids[ child_cnt ] = execve_tile( config->name, tile, float_cpu_set, floating_cpu_set, floating_priority, config_memfd, pipefd[ 1 ], &spawn_cg );
528 0 : child_idxs[ child_cnt ] = i;
529 0 : if( FD_UNLIKELY( close( pipefd[ 1 ] ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
530 0 : strncpy( child_names[ child_cnt ], tile->name, 32 );
531 0 : child_cnt++;
532 0 : }
533 0 : }
534 :
535 0 : leave_isolation_cgroup( &spawn_cg );
536 :
537 : /* Obtain the actual grandchild PID from the pipe */
538 0 : for( ulong i=0UL; i<child_cnt; i++ ) {
539 0 : if( FD_UNLIKELY( actual_pids[ i ]!=ULONG_MAX ) ) continue;
540 0 : FD_TEST( 8UL==read( fds[ i ].fd, &actual_pids[ i ], 8UL ) );
541 0 : }
542 :
543 0 : if( FD_UNLIKELY( -1==setpriority( PRIO_PROCESS, 0, save_priority ) ) ) FD_LOG_ERR(( "setpriority() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
544 0 : if( FD_UNLIKELY( fd_cpuset_setaffinity( 0, floating_cpu_set ) ) )
545 0 : FD_LOG_ERR(( "fd_cpuset_setaffinity failed (%i-%s)", errno, fd_io_strerror( errno ) ));
546 :
547 0 : if( FD_UNLIKELY( close( config_memfd ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
548 0 : if( need_xdp ) {
549 0 : for( uint i=0U; i<xdp_fds_cnt; i++ ) {
550 0 : if( FD_UNLIKELY( close( xdp_fds[i].xsk_map_fd ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
551 0 : if( FD_UNLIKELY( close( xdp_fds[i].prog_link_fd ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
552 0 : }
553 0 : }
554 :
555 0 : if( FD_LIKELY( config->is_firedancer ) ) {
556 0 : if( FD_UNLIKELY( -1==close( FD_ACCDB_FD_RW ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
557 0 : if( FD_UNLIKELY( -1==close( FD_ACCDB_FD_RO ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
558 0 : if( FD_LIKELY( has_store ) ) {
559 0 : if( FD_UNLIKELY( -1==close( FD_STORE_FD_RW ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
560 0 : if( FD_UNLIKELY( -1==close( FD_STORE_FD_RO ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
561 0 : }
562 0 : for( ulong j=0UL; j<snap_max; j++ ) {
563 0 : if( FD_UNLIKELY( -1==close( FD_SNAP_FD( j ) ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
564 0 : if( snapshot_dio_enabled )
565 0 : if( FD_UNLIKELY( -1==close( FD_SNAP_DIO_FD( j ) ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
566 0 : if( snapshot_upload_enabled )
567 0 : if( FD_UNLIKELY( -1==close( FD_SNAP_RO_FD( j ) ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
568 0 : }
569 0 : }
570 :
571 0 : if( FD_UNLIKELY( -1==close( FD_WAKER_OUTER_FD ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
572 0 : for( ulong j=0UL; j<waker_client_cnt; j++ ) {
573 0 : if( FD_UNLIKELY( -1==close( FD_WAKER_INNER_FD( j ) ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
574 0 : }
575 :
576 0 : int allow_fds[ 6+FD_TOPO_MAX_TILES ];
577 0 : ulong allow_fds_cnt = 0;
578 0 : allow_fds[ allow_fds_cnt++ ] = 2; /* stderr */
579 0 : if( FD_LIKELY( fd_log_private_logfile_fd()!=-1 ) )
580 0 : allow_fds[ allow_fds_cnt++ ] = fd_log_private_logfile_fd(); /* logfile */
581 0 : allow_fds[ allow_fds_cnt++ ] = args->pipefd[ 1 ]; /* write end of main pipe */
582 0 : for( ulong i=0UL; i<child_cnt; i++ )
583 0 : allow_fds[ allow_fds_cnt++ ] = fds[ i ].fd; /* read end of child pipes */
584 0 : if( need_mlx5 ) {
585 0 : allow_fds[ allow_fds_cnt++ ] = mlx5_fds.cmd_fd;
586 0 : allow_fds[ allow_fds_cnt++ ] = mlx5_fds.async_fd;
587 0 : }
588 :
589 0 : struct sock_filter seccomp_filter[ 128UL ];
590 0 : unsigned int instr_cnt;
591 : #if defined(__aarch64__)
592 : populate_sock_filter_policy_pidns_arm64( 128UL, seccomp_filter, (uint)fd_log_private_logfile_fd() );
593 : instr_cnt = sock_filter_policy_pidns_arm64_instr_cnt;
594 : #else
595 0 : populate_sock_filter_policy_pidns( 128UL, seccomp_filter, (uint)fd_log_private_logfile_fd() );
596 0 : instr_cnt = sock_filter_policy_pidns_instr_cnt;
597 0 : #endif
598 :
599 0 : if( FD_LIKELY( config->development.sandbox ) ) {
600 0 : fd_sandbox_enter( config->uid,
601 0 : config->gid,
602 0 : 0,
603 0 : 0,
604 0 : 0,
605 0 : 0,
606 0 : 0,
607 0 : 1UL+child_cnt, /* RLIMIT_NOFILE needs to be set to the nfds argument of poll() */
608 0 : 0UL,
609 0 : 0UL,
610 0 : 0UL,
611 0 : allow_fds_cnt,
612 0 : allow_fds,
613 0 : instr_cnt,
614 0 : seccomp_filter );
615 0 : } else {
616 0 : fd_sandbox_switch_uid_gid( config->uid, config->gid );
617 0 : }
618 :
619 : /* Reap child process PIDs so they don't show up in `ps` etc. All of
620 : these children should have exited immediately after clone(2)'ing
621 : another child with a huge page based stack. */
622 0 : for( ulong i=0UL; i<child_cnt; i++ ) {
623 0 : int wstatus;
624 0 : int exited_pid = wait4( child_pids[ i ], &wstatus, (int)__WALL, NULL );
625 0 : if( FD_UNLIKELY( -1==exited_pid ) ) {
626 0 : FD_LOG_ERR(( "pidns wait4() failed (%i-%s) %lu %hu", errno, fd_io_strerror( errno ), i, fds[i].revents ));
627 0 : } else if( FD_UNLIKELY( child_pids[ i ]!=exited_pid ) ) {
628 0 : FD_LOG_ERR(( "pidns wait4() returned unexpected pid %d %d", child_pids[ i ], exited_pid ));
629 0 : } else if( FD_UNLIKELY( !WIFEXITED( wstatus ) ) ) {
630 0 : FD_LOG_ERR_NOEXIT(( "tile %lu (%s) exited while booting with signal %d (%s)", i, child_names[ i ], WTERMSIG( wstatus ), fd_io_strsignal( WTERMSIG( wstatus ) ) ));
631 0 : fd_sys_util_exit_group( WTERMSIG( wstatus ) ? WTERMSIG( wstatus ) : 1 );
632 0 : }
633 0 : if( FD_UNLIKELY( WEXITSTATUS( wstatus ) ) ) {
634 0 : FD_LOG_ERR_NOEXIT(( "tile %lu (%s) exited while booting with code %d", i, child_names[ i ], WEXITSTATUS( wstatus ) ));
635 0 : fd_sys_util_exit_group( WEXITSTATUS( wstatus ) ? WEXITSTATUS( wstatus ) : 1 );
636 0 : }
637 0 : }
638 :
639 0 : fds[ child_cnt ] = (struct pollfd){ .fd = args->pipefd[ 1 ], .events = 0 };
640 0 : strncpy( child_names[ child_cnt ], "parent", 32UL );
641 0 : child_idxs[ child_cnt ] = ULONG_MAX;
642 :
643 : /* We are now the init process of the pid namespace. If the init
644 : process dies, all children are terminated. If any child dies, we
645 : terminate the init process, which will cause the kernel to
646 : terminate all other children bringing all of our processes down as
647 : a group. The parent process will also die if this process dies,
648 : due to getting SIGHUP on the pipe. */
649 0 : while( 1 ) {
650 0 : if( FD_UNLIKELY( -1==poll( fds, 1UL+child_cnt, (int)-1 ) ) ) FD_LOG_ERR(( "poll() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
651 :
652 : /* Parent process died, probably SIGINT, exit gracefully. */
653 0 : if( FD_UNLIKELY( fds[ child_cnt ].revents ) ) fd_sys_util_exit_group( 0 );
654 :
655 : /* Child process died, reap it to figure out exit code. */
656 0 : int wstatus;
657 0 : int exited_pid = wait4( -1, &wstatus, (int)__WALL | (int)WNOHANG, NULL );
658 0 : if( FD_UNLIKELY( -1==exited_pid ) ) {
659 0 : FD_LOG_ERR(( "pidns wait4() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
660 0 : } else if( FD_UNLIKELY( !exited_pid ) ) {
661 : /* Spurious wakeup, no child actually dead yet. */
662 0 : continue;
663 0 : }
664 :
665 : /* Now find the tile corresponding to that PID */
666 0 : FD_TEST( exited_pid>0 );
667 0 : int found = 0;
668 0 : for( ulong i=0UL; i<child_cnt; i++ ) {
669 0 : if( FD_LIKELY( actual_pids[ i ]!=(ulong)exited_pid ) ) continue;
670 :
671 0 : found = 1;
672 0 : fds[ i ].fd = -1; /* Don't poll on this tile anymore */
673 :
674 0 : char * tile_name = child_names[ i ];
675 0 : ulong tile_idx = child_idxs[ i ];
676 0 : ulong tile_id = config->topo.tiles[ tile_idx ].kind_id;
677 :
678 0 : if( FD_UNLIKELY( !WIFEXITED( wstatus ) ) ) {
679 0 : FD_LOG_ERR_NOEXIT(( "tile %s%s:%lu%s exited with signal %d %s(%s)%s", fd_log_style_bold(), tile_name, tile_id, fd_log_style_normal(), WTERMSIG( wstatus ), fd_log_style_dim(), fd_io_strsignal( WTERMSIG( wstatus ) ), fd_log_style_normal() ));
680 0 : fd_sys_util_exit_group( WTERMSIG( wstatus ) ? WTERMSIG( wstatus ) : 1 );
681 0 : } else {
682 0 : int exit_code = WEXITSTATUS( wstatus );
683 0 : if( FD_LIKELY( !exit_code && tile_idx!=ULONG_MAX && config->topo.tiles[ tile_idx ].allow_shutdown ) ) {
684 0 : found = 1;
685 0 : FD_LOG_INFO(( "tile %s:%lu exited gracefully with code %d", tile_name, tile_id, exit_code ));
686 0 : } else {
687 0 : FD_LOG_ERR_NOEXIT(( "tile %s%s:%lu%s exited with code %d", fd_log_style_bold(), tile_name, tile_id, fd_log_style_normal(), exit_code ));
688 0 : fd_sys_util_exit_group( exit_code ? exit_code : 1 );
689 0 : }
690 0 : }
691 0 : }
692 :
693 0 : if( FD_UNLIKELY( !found ) ) FD_LOG_ERR(( "wait4() returned unexpected pid %d", exited_pid ));
694 0 : }
695 :
696 0 : return 0;
697 0 : }
698 :
699 : int
700 : clone_firedancer( config_t const * config,
701 : int close_fd,
702 0 : int * out_pipe ) {
703 : /* This pipe is here just so that the child process knows when the
704 : parent has died (it will get a HUP). */
705 0 : int pipefd[2];
706 0 : if( FD_UNLIKELY( pipe2( pipefd, O_CLOEXEC | O_NONBLOCK ) ) ) FD_LOG_ERR(( "pipe2() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
707 :
708 : /* clone into a pid namespace */
709 0 : int flags = config->development.sandbox ? CLONE_NEWPID : 0;
710 0 : struct pidns_clone_args args = { .config = config, .closefd = close_fd, .pipefd = pipefd, };
711 :
712 0 : void * stack = create_clone_stack();
713 :
714 0 : int pid_namespace = clone( main_pid_namespace, (uchar *)stack + FD_TILE_PRIVATE_STACK_SZ, flags, &args );
715 0 : if( FD_UNLIKELY( pid_namespace<0 ) ) FD_LOG_ERR(( "clone() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
716 :
717 0 : if( FD_UNLIKELY( close( pipefd[ 1 ] ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
718 :
719 0 : *out_pipe = pipefd[ 0 ];
720 0 : return pid_namespace;
721 0 : }
722 :
723 : static void
724 : workspace_path( config_t const * config,
725 : fd_topo_wksp_t const * wksp,
726 0 : char out[ PATH_MAX ] ) {
727 0 : char const * mount_path;
728 0 : switch( wksp->page_sz ) {
729 0 : case FD_SHMEM_HUGE_PAGE_SZ:
730 0 : mount_path = config->hugetlbfs.huge_page_mount_path;
731 0 : break;
732 0 : case FD_SHMEM_GIGANTIC_PAGE_SZ:
733 0 : mount_path = config->hugetlbfs.gigantic_page_mount_path;
734 0 : break;
735 0 : case FD_SHMEM_NORMAL_PAGE_SZ:
736 0 : mount_path = config->hugetlbfs.normal_page_mount_path;
737 0 : break;
738 0 : default:
739 0 : FD_LOG_ERR(( "invalid page size %lu", wksp->page_sz ));
740 0 : }
741 :
742 0 : FD_TEST( fd_cstr_printf_check( out, PATH_MAX, NULL, "%s/%s_%s.wksp", mount_path, config->name, wksp->name ) );
743 0 : }
744 :
745 : static void
746 : warn_unknown_files( config_t const * config,
747 0 : ulong mount_type ) {
748 0 : char const * mount_path;
749 0 : switch( mount_type ) {
750 0 : case 0UL:
751 0 : mount_path = config->hugetlbfs.huge_page_mount_path;
752 0 : break;
753 0 : case 1UL:
754 0 : mount_path = config->hugetlbfs.gigantic_page_mount_path;
755 0 : break;
756 0 : default:
757 0 : FD_LOG_ERR(( "invalid mount type %lu", mount_type ));
758 0 : }
759 :
760 : /* Check if there are any files in mount_path */
761 0 : DIR * dir = opendir( mount_path );
762 0 : if( FD_UNLIKELY( !dir ) ) {
763 0 : if( FD_UNLIKELY( errno!=ENOENT ) ) FD_LOG_ERR(( "error opening `%s` (%i-%s)", mount_path, errno, fd_io_strerror( errno ) ));
764 0 : return;
765 0 : }
766 :
767 0 : struct dirent * entry;
768 0 : for(;;) {
769 0 : errno = 0;
770 0 : entry = readdir( dir );
771 0 : if( FD_UNLIKELY( !entry ) ) break;
772 0 : if( FD_UNLIKELY( !strcmp( entry->d_name, ".") || !strcmp( entry->d_name, ".." ) ) ) continue;
773 :
774 0 : char entry_path[ PATH_MAX ];
775 0 : FD_TEST( fd_cstr_printf_check( entry_path, PATH_MAX, NULL, "%s/%s", mount_path, entry->d_name ));
776 :
777 0 : int known_file = 0;
778 0 : for( ulong i=0UL; i<config->topo.wksp_cnt; i++ ) {
779 0 : fd_topo_wksp_t const * wksp = &config->topo.workspaces[ i ];
780 :
781 0 : char expected_path[ PATH_MAX ];
782 0 : workspace_path( config, wksp, expected_path );
783 :
784 0 : if( !strcmp( entry_path, expected_path ) ) {
785 0 : known_file = 1;
786 0 : break;
787 0 : }
788 0 : }
789 :
790 0 : if( mount_type==0UL ) {
791 0 : for( ulong i=0UL; i<config->topo.tile_cnt; i++ ) {
792 0 : fd_topo_tile_t const * tile = &config->topo.tiles [ i ];
793 :
794 0 : char expected_path[ PATH_MAX ];
795 0 : FD_TEST( fd_cstr_printf_check( expected_path, PATH_MAX, NULL, "%s/%s_stack_%s%lu", config->hugetlbfs.huge_page_mount_path, config->name, tile->name, tile->kind_id ) );
796 :
797 0 : if( !strcmp( entry_path, expected_path ) ) {
798 0 : known_file = 1;
799 0 : break;
800 0 : }
801 0 : }
802 0 : }
803 :
804 0 : if( FD_UNLIKELY( !known_file ) ) FD_LOG_WARNING(( "unknown file `%s` found in `%s`", entry->d_name, mount_path ));
805 0 : }
806 :
807 0 : if( FD_UNLIKELY( errno ) ) FD_LOG_ERR(( "error reading dir `%s` (%i-%s)", mount_path, errno, fd_io_strerror( errno ) ));
808 0 : if( FD_UNLIKELY( closedir( dir ) ) ) FD_LOG_ERR(( "error closing `%s` (%i-%s)", mount_path, errno, fd_io_strerror( errno ) ));
809 0 : }
810 :
811 : void
812 0 : initialize_workspaces( config_t * config ) {
813 : /* Switch to non-root uid/gid for workspace creation. Permissions
814 : checks are still done as the current user. */
815 0 : uint gid = getgid();
816 0 : uint uid = getuid();
817 0 : if( FD_LIKELY( gid!=config->gid && -1==setegid( config->gid ) ) )
818 0 : FD_LOG_ERR(( "setegid() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
819 0 : if( FD_LIKELY( uid!=config->uid && -1==seteuid( config->uid ) ) )
820 0 : FD_LOG_ERR(( "seteuid() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
821 :
822 0 : for( ulong i=0UL; i<config->topo.wksp_cnt; i++ ) {
823 0 : fd_topo_wksp_t * wksp = &config->topo.workspaces[ i ];
824 :
825 0 : char path[ PATH_MAX ];
826 0 : workspace_path( config, wksp, path );
827 :
828 0 : struct stat st;
829 0 : int result = stat( path, &st );
830 :
831 0 : int update_existing;
832 0 : if( FD_UNLIKELY( !result && config->is_live_cluster && !config->is_dev ) ) {
833 0 : if( FD_UNLIKELY( -1==unlink( path ) && errno!=ENOENT ) ) FD_LOG_ERR(( "unlink() failed when trying to create workspace `%s` (%i-%s)", path, errno, fd_io_strerror( errno ) ));
834 0 : update_existing = 0;
835 0 : } else if( FD_UNLIKELY( !result ) ) {
836 : /* Creating all of the workspaces is very expensive because the
837 : kernel has to zero out all of the pages. There can be tens or
838 : hundreds of gigabytes of zeroing to do.
839 :
840 : What would be really nice is if the kernel let us create huge
841 : pages without zeroing them, but it's not possible. The
842 : ftruncate and fallocate calls do not support this type of
843 : resize with the hugetlbfs filesystem.
844 :
845 : Instead.. to prevent repeatedly doing this zeroing every time
846 : we start the validator, we have a small hack here to re-use the
847 : workspace files if they exist. */
848 0 : update_existing = 1;
849 0 : } else if( FD_LIKELY( result && errno==ENOENT ) ) {
850 0 : update_existing = 0;
851 0 : } else {
852 0 : FD_LOG_ERR(( "stat failed when trying to create workspace `%s` (%i-%s)", path, errno, fd_io_strerror( errno ) ));
853 0 : }
854 :
855 0 : if( FD_UNLIKELY( -1==fd_topo_create_workspace( &config->topo, wksp, update_existing ) ) ) {
856 0 : FD_TEST( errno==ENOMEM );
857 :
858 0 : warn_unknown_files( config, wksp->page_sz!=FD_SHMEM_HUGE_PAGE_SZ );
859 :
860 0 : char path[ PATH_MAX ];
861 0 : workspace_path( config, wksp, path );
862 0 : FD_LOG_ERR(( "ENOMEM-Out of memory when trying to create workspace `%s` at `%s` "
863 0 : "with %lu %s pages. Firedancer reserves enough memory for all of its workspaces "
864 0 : "during the `hugetlbfs` configure step, so it is likely you have unknown files "
865 0 : "left over in this directory which are consuming memory, or another program on "
866 0 : "the system is using pages from the same mount.",
867 0 : wksp->name, path, wksp->page_cnt, fd_shmem_page_sz_to_cstr( wksp->page_sz ) ));
868 0 : }
869 0 : fd_topo_join_workspace( &config->topo, wksp, FD_SHMEM_JOIN_MODE_READ_WRITE, 0 );
870 0 : fd_topo_wksp_new( &config->topo, wksp, CALLBACKS );
871 0 : fd_topo_leave_workspace( &config->topo, wksp );
872 0 : }
873 :
874 0 : if( FD_UNLIKELY( seteuid( uid ) ) ) FD_LOG_ERR(( "seteuid() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
875 0 : if( FD_UNLIKELY( setegid( gid ) ) ) FD_LOG_ERR(( "setegid() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
876 0 : }
877 :
878 : void
879 0 : initialize_stacks( config_t const * config ) {
880 : # if FD_HAS_MSAN
881 : /* MSan calls an external symbolizer using fork() on crashes, which is
882 : incompatible with Firedancer's MAP_SHARED stacks. */
883 : (void)config;
884 : return;
885 : # endif
886 :
887 : /* Switch to non-root uid/gid for workspace creation. Permissions
888 : checks are still done as the current user. */
889 0 : uint gid = getgid();
890 0 : uint uid = getuid();
891 0 : if( FD_LIKELY( gid!=config->gid && -1==setegid( config->gid ) ) )
892 0 : FD_LOG_ERR(( "setegid() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
893 0 : if( FD_LIKELY( uid!=config->uid && -1==seteuid( config->uid ) ) )
894 0 : FD_LOG_ERR(( "seteuid() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
895 :
896 0 : for( ulong i=0UL; i<config->topo.tile_cnt; i++ ) {
897 0 : fd_topo_tile_t const * tile = &config->topo.tiles[ i ];
898 :
899 0 : char path[ PATH_MAX ];
900 0 : FD_TEST( fd_cstr_printf_check( path, PATH_MAX, NULL, "%s/%s_stack_%s%lu", config->hugetlbfs.huge_page_mount_path, config->name, tile->name, tile->kind_id ) );
901 :
902 0 : struct stat st;
903 0 : int result = stat( path, &st );
904 :
905 0 : int update_existing;
906 0 : if( FD_UNLIKELY( !result && config->is_live_cluster ) ) {
907 0 : if( FD_UNLIKELY( -1==unlink( path ) && errno!=ENOENT ) ) FD_LOG_ERR(( "unlink() failed when trying to create stack workspace `%s` (%i-%s)", path, errno, fd_io_strerror( errno ) ));
908 0 : update_existing = 0;
909 0 : } else if( FD_UNLIKELY( !result ) ) {
910 : /* See above note about zeroing out pages. */
911 0 : update_existing = 1;
912 0 : } else if( FD_LIKELY( result && errno==ENOENT ) ) {
913 0 : update_existing = 0;
914 0 : } else {
915 0 : FD_LOG_ERR(( "stat failed when trying to create workspace `%s` (%i-%s)", path, errno, fd_io_strerror( errno ) ));
916 0 : }
917 :
918 : /* TODO: Use a better CPU idx for the stack if tile is floating */
919 0 : ulong stack_cpu_idx = 0UL;
920 0 : if( FD_LIKELY( tile->cpu_idx<65535UL ) ) stack_cpu_idx = tile->cpu_idx;
921 :
922 0 : char name[ PATH_MAX ];
923 0 : FD_TEST( fd_cstr_printf_check( name, PATH_MAX, NULL, "%s_stack_%s%lu", config->name, tile->name, tile->kind_id ) );
924 :
925 0 : ulong sub_page_cnt[ 1 ] = { 6 };
926 0 : ulong sub_cpu_idx [ 1 ] = { stack_cpu_idx };
927 0 : int err;
928 0 : if( FD_UNLIKELY( update_existing ) ) {
929 0 : err = fd_shmem_update_multi( name, FD_SHMEM_HUGE_PAGE_SZ, 1, sub_page_cnt, sub_cpu_idx, S_IRUSR | S_IWUSR ); /* logs details */
930 0 : } else {
931 0 : err = fd_shmem_create_multi( name, FD_SHMEM_HUGE_PAGE_SZ, 1, sub_page_cnt, sub_cpu_idx, S_IRUSR | S_IWUSR ); /* logs details */
932 0 : }
933 0 : if( FD_UNLIKELY( err && errno==ENOMEM ) ) {
934 0 : warn_unknown_files( config, 0UL );
935 :
936 0 : char path[ PATH_MAX ];
937 0 : FD_TEST( fd_cstr_printf_check( path, PATH_MAX, NULL, "%s/%s_stack_%s%lu", config->hugetlbfs.huge_page_mount_path, config->name, tile->name, tile->kind_id ) );
938 0 : FD_LOG_ERR(( "ENOMEM-Out of memory when trying to create huge page stack for tile `%s` at `%s`. "
939 0 : "Firedancer reserves enough memory for all of its stacks during the `hugetlbfs` configure "
940 0 : "step, so it is likely you have unknown files left over in this directory which are "
941 0 : "consuming memory, or another program on the system is using pages from the same mount.",
942 0 : tile->name, path ));
943 0 : } else if( FD_UNLIKELY( err ) ) FD_LOG_ERR(( "fd_shmem_create_multi failed" ));
944 0 : }
945 :
946 0 : if( FD_UNLIKELY( seteuid( uid ) ) ) FD_LOG_ERR(( "seteuid() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
947 0 : if( FD_UNLIKELY( setegid( gid ) ) ) FD_LOG_ERR(( "setegid() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
948 0 : }
949 :
950 : void
951 0 : fdctl_check_configure( config_t const * config ) {
952 0 : configure_result_t check = fd_cfg_stage_hugetlbfs.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
953 0 : if( FD_UNLIKELY( check.result!=CONFIGURE_OK ) )
954 0 : FD_LOG_ERR(( "Huge pages are not configured correctly: %s. You can run `%s configure init hugetlbfs` "
955 0 : "to create the mounts correctly. This must be done after every system restart before running "
956 0 : "Firedancer.", check.message, FD_BINARY_NAME ));
957 :
958 0 : if( FD_LIKELY( 0==strcmp( config->net.provider, "xdp" ) ) ) {
959 0 : if( fd_cfg_stage_bonding.enabled( config ) ) {
960 0 : check = fd_cfg_stage_bonding.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
961 0 : if( FD_UNLIKELY( check.result!=CONFIGURE_OK ) )
962 0 : FD_LOG_ERR(( "Bonded network device is not configured correctly: %s. You can run `%s configure init bonding` "
963 0 : "to configure the bonding driver.", check.message, FD_BINARY_NAME ));
964 0 : }
965 :
966 0 : check = fd_cfg_stage_ethtool_channels.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
967 0 : if( FD_UNLIKELY( check.result!=CONFIGURE_OK ) )
968 0 : FD_LOG_ERR(( "Network %s. You can run `%s configure init ethtool-channels` to set the number of channels on the "
969 0 : "network device correctly.", check.message, FD_BINARY_NAME ));
970 :
971 0 : check = fd_cfg_stage_ethtool_offloads.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
972 0 : if( FD_UNLIKELY( check.result!=CONFIGURE_OK ) )
973 0 : FD_LOG_ERR(( "Network %s. You can run `%s configure init ethtool-offloads` to disable features "
974 0 : "as required.", check.message, FD_BINARY_NAME ));
975 :
976 0 : check = fd_cfg_stage_ethtool_loopback.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
977 0 : if( FD_UNLIKELY( check.result!=CONFIGURE_OK ) )
978 0 : FD_LOG_ERR(( "Network %s. You can run `%s configure init ethtool-loopback` to disable tx-udp-segmentation "
979 0 : "on the loopback device.", check.message, FD_BINARY_NAME ));
980 0 : }
981 :
982 0 : check = fd_cfg_stage_sysctl.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
983 0 : if( FD_UNLIKELY( check.result!=CONFIGURE_OK ) )
984 0 : FD_LOG_ERR(( "Kernel parameters are not configured correctly: %s. You can run `%s configure init sysctl` "
985 0 : "to set kernel parameters correctly.", check.message, FD_BINARY_NAME ));
986 :
987 : /* hyperthreads, nohz-full and rcu-nocbs are check-only stages: they
988 : emit warnings themselves and always return OK, so there is no
989 : result to act on (and no init to point the operator at). */
990 0 : (void)fd_cfg_stage_hyperthreads.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
991 0 : (void)fd_cfg_stage_nohz_full.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
992 0 : (void)fd_cfg_stage_rcu_nocbs.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
993 :
994 : /* kworkers and cpuset are optional (but recommended) hardening: an
995 : unconfigured stage only warns. A PARTIALLY_CONFIGURED cpuset is
996 : fatal however: the isolation cgroup exists but covers the wrong
997 : CPUs (e.g. stale from a previous [layout.affinity]), and the tile
998 : launcher would join it and then fail to pin with a confusing
999 : EINVAL. Fail up front with the fix instead. */
1000 0 : if( FD_UNLIKELY( fd_cfg_stage_kworkers.enabled( config ) ) ) {
1001 0 : check = fd_cfg_stage_kworkers.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
1002 0 : if( FD_UNLIKELY( check.result!=CONFIGURE_OK ) )
1003 0 : FD_LOG_WARNING(( "Kernel workqueues may steal CPU time from Firedancer tiles: %s. For lower jitter, run "
1004 0 : "`%s configure init kworkers`.", check.message, FD_BINARY_NAME ));
1005 0 : }
1006 :
1007 : /* Floating tiles need a scheduler domain to be balanced across their
1008 : CPUs; isolcpus= removes it, and every floater would stay on the
1009 : CPU it was forked on. */
1010 0 : FD_CPUSET_DECL( isolated );
1011 0 : if( FD_LIKELY( fd_cpu_isolation_read_list( "/sys/devices/system/cpu/isolated", isolated ) ) ) {
1012 0 : for( ulong i=0UL; i<config->topo.tile_cnt; i++ ) {
1013 0 : fd_topo_tile_t const * tile = &config->topo.tiles[ i ];
1014 0 : if( FD_UNLIKELY( tile->floats && fd_cpuset_test( isolated, tile->cpu_idx ) ) )
1015 0 : FD_LOG_ERR(( "tile %s:%lu floats on CPU %lu, which the isolcpus= boot parameter removed from the kernel scheduler. "
1016 0 : "Floating tiles need their CPUs scheduled: drop them from isolcpus= and isolate them with "
1017 0 : "`%s configure init cpuset` instead.", tile->name, tile->kind_id, tile->cpu_idx, FD_BINARY_NAME ));
1018 0 : }
1019 0 : }
1020 :
1021 0 : if( FD_LIKELY( fd_cfg_stage_cpuset.enabled( config ) ) ) {
1022 0 : check = fd_cfg_stage_cpuset.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
1023 0 : if( FD_UNLIKELY( check.result==CONFIGURE_PARTIALLY_CONFIGURED ) )
1024 0 : FD_LOG_ERR(( "The CPU isolation cgroup exists but does not match the topology: %s. Tiles would fail to pin "
1025 0 : "to their CPUs. Run `%s configure init cpuset` to fix it, or `%s configure fini cpuset` to "
1026 0 : "remove it.", check.message, FD_BINARY_NAME, FD_BINARY_NAME ));
1027 0 : else if( FD_UNLIKELY( check.result!=CONFIGURE_OK ) )
1028 0 : FD_LOG_WARNING(( "Firedancer tile CPUs are not isolated from other processes: %s. For lower jitter, run "
1029 0 : "`%s configure init cpuset`.", check.message, FD_BINARY_NAME ));
1030 0 : }
1031 0 : }
1032 :
1033 : void
1034 : run_firedancer_init( config_t * config,
1035 : int init_workspaces,
1036 0 : int check_configure ) {
1037 0 : struct stat st;
1038 0 : int err = stat( config->paths.identity_key, &st );
1039 0 : if( FD_UNLIKELY( -1==err && errno==ENOENT ) ) FD_LOG_ERR(( "[consensus.identity_path] key does not exist `%s`. You can generate an identity key at this path by running `%s keys new %s --config <toml>`", config->paths.identity_key, FD_BINARY_NAME, config->paths.identity_key ));
1040 0 : else if( FD_UNLIKELY( -1==err ) ) FD_LOG_ERR(( "could not stat [consensus.identity_path] `%s` (%i-%s)", config->paths.identity_key, errno, fd_io_strerror( errno ) ));
1041 :
1042 0 : if( FD_UNLIKELY( !config->is_firedancer ) ) {
1043 0 : for( ulong i=0UL; i<config->frankendancer.paths.authorized_voter_paths_cnt; i++ ) {
1044 0 : err = stat( config->frankendancer.paths.authorized_voter_paths[ i ], &st );
1045 0 : if( FD_UNLIKELY( -1==err && errno==ENOENT ) ) FD_LOG_ERR(( "[consensus.authorized_voter_paths] key does not exist `%s`", config->frankendancer.paths.authorized_voter_paths[ i ] ));
1046 0 : else if( FD_UNLIKELY( -1==err ) ) FD_LOG_ERR(( "could not stat [consensus.authorized_voter_paths] `%s` (%i-%s)", config->frankendancer.paths.authorized_voter_paths[ i ], errno, fd_io_strerror( errno ) ));
1047 0 : }
1048 0 : }
1049 :
1050 : /* FIXME: fdctl_check_configure unconditionally checks for network
1051 : stack prerequisites even if the command being run does not
1052 : require networking. Hack around that here for now. */
1053 0 : if( check_configure ) fdctl_check_configure( config );
1054 0 : if( FD_LIKELY( init_workspaces ) ) initialize_workspaces( config );
1055 0 : initialize_stacks( config );
1056 0 : fd_bootinfo_write( config );
1057 0 : }
1058 :
1059 : void
1060 0 : initialize_accdb_fd( config_t const * config ) {
1061 0 : if( FD_UNLIKELY( !config->is_firedancer ) ) return;
1062 :
1063 : /* O_TRUNC of a previous run's accounts.db frees all its extents
1064 : synchronously. In development skip the truncate to keep reboots
1065 : fast. */
1066 0 : int oflags = O_RDWR|O_CREAT|O_NOATIME;
1067 0 : if( FD_LIKELY( !config->is_dev ) ) oflags |= O_TRUNC;
1068 :
1069 0 : int accounts_fd = open( config->paths.accounts, oflags, S_IRUSR|S_IWUSR );
1070 0 : if( FD_UNLIKELY( -1==accounts_fd ) ) FD_LOG_ERR(( "failed to open accounts.db (%i-%s)", errno, fd_io_strerror( errno ) ));
1071 0 : if( FD_UNLIKELY( -1==dup2( accounts_fd, FD_ACCDB_FD_RW ) ) ) FD_LOG_ERR(( "dup2() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1072 0 : if( FD_UNLIKELY( -1==close( accounts_fd ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1073 :
1074 : /* Read-only fd for tiles (e.g. rpc) that consume accdb but must not
1075 : be able to mutate the on-disk file. Reopen via /proc/self/fd to
1076 : guarantee it refers to the same inode as the RW fd, avoiding any
1077 : race where the file at the path could be replaced between opens. */
1078 0 : char proc_path[ PATH_MAX ];
1079 0 : FD_TEST( fd_cstr_printf_check( proc_path, sizeof(proc_path), NULL, "/proc/self/fd/%d", FD_ACCDB_FD_RW ) );
1080 0 : int accounts_ro_fd = open( proc_path, O_RDONLY|O_NOATIME );
1081 0 : if( FD_UNLIKELY( -1==accounts_ro_fd ) ) FD_LOG_ERR(( "failed to open accounts.db read-only (%i-%s)", errno, fd_io_strerror( errno ) ));
1082 0 : if( FD_UNLIKELY( -1==dup2( accounts_ro_fd, FD_ACCDB_FD_RO ) ) ) FD_LOG_ERR(( "dup2() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1083 0 : if( FD_UNLIKELY( -1==close( accounts_ro_fd ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1084 0 : }
1085 :
1086 : void
1087 0 : initialize_store_fds( config_t const * config ) {
1088 0 : if( FD_UNLIKELY( !config->is_firedancer ) ) return;
1089 :
1090 0 : fd_topo_t const * topo = &config->topo;
1091 0 : ulong store_obj_id = fd_pod_query_ulong( topo->props, "store", ULONG_MAX );
1092 0 : if( FD_UNLIKELY( store_obj_id==ULONG_MAX ) ) return;
1093 :
1094 0 : char const * path = fd_pod_queryf_cstr( topo->props, NULL, "obj.%lu.disk_path", store_obj_id );
1095 0 : ulong fec_max = fd_pod_queryf_ulong( topo->props, 0UL, "obj.%lu.fec_max", store_obj_id );
1096 0 : ulong fec_data_max = fd_pod_queryf_ulong( topo->props, 0UL, "obj.%lu.fec_data_max", store_obj_id );
1097 0 : ulong shred_storage_gib = fd_pod_queryf_ulong( topo->props, ULONG_MAX, "obj.%lu.shred_storage_gib", store_obj_id );
1098 0 : ulong payload_slot_sz = fd_store_payload_slot_sz( fec_data_max );
1099 0 : ulong wire_off;
1100 0 : if( FD_UNLIKELY( !path || !fec_max || !payload_slot_sz || shred_storage_gib>FD_SHREDB_MAX_SIZE_GIB ||
1101 0 : __builtin_umull_overflow( fec_max, payload_slot_sz, &wire_off ) ) )
1102 0 : FD_LOG_ERR(( "invalid Store backing-file configuration" ));
1103 :
1104 0 : int store_fd = fd_store_file_create( path, wire_off, fd_shredb_max_shreds( shred_storage_gib ) );
1105 0 : if( FD_UNLIKELY( store_fd<0 ) )
1106 0 : FD_LOG_ERR(( "failed to create Store backing file `%s` (%i-%s)", path, errno, fd_io_strerror( errno ) ));
1107 0 : if( FD_LIKELY( store_fd!=FD_STORE_FD_RW ) ) {
1108 0 : if( FD_UNLIKELY( dup2( store_fd, FD_STORE_FD_RW )<0 ) ) FD_LOG_ERR(( "dup2(Store RW) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1109 0 : if( FD_UNLIKELY( close( store_fd ) ) ) FD_LOG_ERR(( "close(Store RW source) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1110 0 : }
1111 :
1112 : /* Reopen through procfs so the read-only descriptor is guaranteed to
1113 : name the same inode even if the configured path is replaced. */
1114 0 : char proc_path[ PATH_MAX ];
1115 0 : FD_TEST( fd_cstr_printf_check( proc_path, sizeof(proc_path), NULL, "/proc/self/fd/%d", FD_STORE_FD_RW ) );
1116 0 : int store_ro_fd = open( proc_path, O_RDONLY|O_NOATIME );
1117 0 : if( FD_UNLIKELY( store_ro_fd<0 ) )
1118 0 : FD_LOG_ERR(( "failed to open Store backing file read-only (%i-%s)", errno, fd_io_strerror( errno ) ));
1119 0 : if( FD_LIKELY( store_ro_fd!=FD_STORE_FD_RO ) ) {
1120 0 : if( FD_UNLIKELY( dup2( store_ro_fd, FD_STORE_FD_RO )<0 ) ) FD_LOG_ERR(( "dup2(Store RO) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1121 0 : if( FD_UNLIKELY( close( store_ro_fd ) ) ) FD_LOG_ERR(( "close(Store RO source) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1122 0 : }
1123 0 : }
1124 :
1125 : /* Snapshot production prep
1126 : (On startup, reconcile with files in snapshot dirs, and drop old
1127 : files.) */
1128 :
1129 : static int
1130 0 : matches_partial_snap_filename( char const * name ) {
1131 0 : if( FD_UNLIKELY( !strcmp( name, ".snapshot.tar.bz2-partial" ) ||
1132 0 : !strcmp( name, ".incremental-snapshot.tar.bz2-partial" ) ) ) return 1;
1133 :
1134 0 : if( *name=='.' ) name++;
1135 0 : char const * p;
1136 0 : if( !strncmp( name, "snapshot-x", 10UL ) ) p = name+10UL;
1137 0 : else if( !strncmp( name, "snapshot", 8UL ) ) p = name+8UL;
1138 0 : else return 0;
1139 0 : if( *p<'0' || *p>'9' ) return 0;
1140 0 : while( *p>='0' && *p<='9' ) p++;
1141 0 : return 0==strcmp( p, ".partial" );
1142 0 : }
1143 :
1144 : static int
1145 : snap_inode_desc( void const * _a,
1146 0 : void const * _b ) {
1147 0 : fd_backup_inode_t const * a = (fd_backup_inode_t const *)_a;
1148 0 : fd_backup_inode_t const * b = (fd_backup_inode_t const *)_b;
1149 0 : ulong a_slot = a->incr_slot==ULONG_MAX ? a->full_slot : a->incr_slot;
1150 0 : ulong b_slot = b->incr_slot==ULONG_MAX ? b->full_slot : b->incr_slot;
1151 0 : if( FD_LIKELY( a_slot!=b_slot ) ) return a_slot>b_slot ? -1 : 1;
1152 0 : return 0;
1153 0 : }
1154 :
1155 : static void
1156 : drop_old_snaps( int snap_dir_fd,
1157 : char const * snap_dir,
1158 : fd_backup_inode_t * snaps,
1159 : ulong * snap_cnt_p,
1160 0 : ulong snap_max ) {
1161 0 : ulong snap_cnt = *snap_cnt_p;
1162 0 : if( snap_cnt<=snap_max ) return;
1163 : /* drop oldest */
1164 0 : qsort( snaps, snap_cnt, sizeof(fd_backup_inode_t), snap_inode_desc );
1165 0 : for( ulong i=snap_max; i<snap_cnt; i++ ) {
1166 0 : int err = unlinkat( snap_dir_fd, snaps[ i ].name, 0 );
1167 0 : if( FD_UNLIKELY( err && errno!=ENOENT ) ) {
1168 0 : FD_LOG_WARNING(( "unlinkat(%s/%s) failed (%i-%s)", snap_dir, snaps[ i ].name, errno, fd_io_strerror( errno ) ));
1169 0 : } else if( FD_LIKELY( !err ) ) {
1170 0 : FD_LOG_INFO(( "deleted old snapshot `%s/%s`", snap_dir, snaps[ i ].name ));
1171 0 : }
1172 0 : }
1173 0 : *snap_cnt_p = snap_max;
1174 0 : }
1175 :
1176 : static void
1177 : snap_check_fd( char const * snap_dir,
1178 : char const * name,
1179 0 : int snap_fd ) {
1180 0 : struct stat st;
1181 0 : if( FD_UNLIKELY( -1==fstat( snap_fd, &st ) ) )
1182 0 : FD_LOG_ERR(( "fstat(%s/%s) failed (%i-%s)", snap_dir, name, errno, fd_io_strerror( errno ) ));
1183 :
1184 0 : if( FD_UNLIKELY( !S_ISREG( st.st_mode ) ) )
1185 0 : FD_LOG_ERR(( "snapshot `%s/%s` is not a regular file", snap_dir, name ));
1186 0 : }
1187 :
1188 : static void
1189 : snap_reperm_fd( char const * snap_dir,
1190 : char const * name,
1191 : int snap_fd,
1192 : uint uid,
1193 0 : uint gid ) {
1194 0 : if( FD_UNLIKELY( -1==fchown( snap_fd, uid, gid ) ) )
1195 0 : FD_LOG_ERR(( "fchown(%s/%s) failed (%i-%s)", snap_dir, name, errno, fd_io_strerror( errno ) ));
1196 :
1197 0 : if( FD_UNLIKELY( -1==fchmod( snap_fd, S_IRUSR|S_IWUSR ) ) )
1198 0 : FD_LOG_ERR(( "fchmod(%s/%s) failed (%i-%s)", snap_dir, name, errno, fd_io_strerror( errno ) ));
1199 0 : }
1200 :
1201 : ulong
1202 0 : initialize_snapshot_fds( config_t const * config ) {
1203 :
1204 0 : int download_enabled = fd_topo_find_tile( &config->topo, "snapct", 0UL )!=ULONG_MAX &&
1205 0 : (config->firedancer.snapshots.sources.gossip.allow_any ||
1206 0 : config->firedancer.snapshots.sources.gossip.allow_list_cnt ||
1207 0 : config->firedancer.snapshots.sources.servers_cnt);
1208 0 : int upload_enabled = fd_topo_find_tile( &config->topo, "snapsv", 0UL )!=ULONG_MAX;
1209 0 : int dio_enabled = fd_topo_find_tile( &config->topo, "snapzp", 0UL )!=ULONG_MAX;
1210 0 : fd_snap_pool_layout_t layout = fd_snap_pool_layout( config->firedancer.snapshots.max_full_snapshots_to_keep,
1211 0 : config->firedancer.snapshots.max_incremental_snapshots_to_keep,
1212 0 : config->firedancer.snapshots.incremental_snapshots,
1213 0 : download_enabled );
1214 0 : ulong snap_full_max = layout.full_max;
1215 0 : ulong snap_incr_max = layout.incr_max;
1216 0 : ulong retained_snap_max = layout.retained_max;
1217 0 : ulong snap_max = layout.max;
1218 0 : if( FD_UNLIKELY( !snap_max ) ) return 0UL;
1219 :
1220 0 : char const * snap_dir = config->paths.snapshots;
1221 0 : int dir_fd = open( snap_dir, O_RDONLY|O_DIRECTORY|O_CLOEXEC );
1222 0 : if( FD_UNLIKELY( -1==dir_fd ) ) FD_LOG_ERR(( "open(%s) failed (%i-%s)", snap_dir, errno, fd_io_strerror( errno ) ));
1223 :
1224 : /* This is additionally validated in fd_config_validatef() */
1225 0 : FD_CHECK_ERR( snap_max<=FD_SNAP_MAX, "[snapshots.max_{full,incremental}_snapshots_to_keep] is set too high" );
1226 0 : fd_backup_inode_t snap_full[ FD_SNAP_MAX ];
1227 0 : fd_backup_inode_t snap_incr[ FD_SNAP_MAX ];
1228 0 : fd_backup_inode_t scratch[ 2 ] = {
1229 0 : { .full_slot=ULONG_MAX, .incr_slot=ULONG_MAX },
1230 0 : { .full_slot=ULONG_MAX, .incr_slot=ULONG_MAX }
1231 0 : };
1232 0 : ulong snap_full_cnt = 0UL;
1233 0 : ulong snap_incr_cnt = 0UL;
1234 0 : FD_STATIC_ASSERT( sizeof(snap_full)+sizeof(snap_incr) < 2UL<<20, "stack overflow" );
1235 :
1236 : /* First, reconcile with the existing snapshots */
1237 0 : int dir_fd2 = dup( dir_fd );
1238 0 : if( FD_UNLIKELY( -1==dir_fd2 ) ) FD_LOG_ERR(( "dup(%s) failed (%i-%s)", snap_dir, errno, fd_io_strerror( errno ) ));
1239 0 : DIR * dir = fdopendir( dir_fd2 );
1240 0 : if( FD_UNLIKELY( !dir ) ) FD_LOG_ERR(( "fdopendir(%s) failed (%i-%s)", snap_dir, errno, fd_io_strerror( errno ) ));
1241 0 : struct dirent * entry;
1242 0 : for(;;) {
1243 0 : errno = 0;
1244 0 : entry = readdir( dir );
1245 0 : if( FD_UNLIKELY( !entry ) ) break;
1246 0 : if( FD_UNLIKELY( !strcmp( entry->d_name, ".") || !strcmp( entry->d_name, ".." ) ) ) continue;
1247 :
1248 : /* delete partial files */
1249 0 : if( FD_UNLIKELY( matches_partial_snap_filename( entry->d_name ) ) ) {
1250 0 : if( FD_UNLIKELY( -1==unlinkat( dir_fd, entry->d_name, 0 ) && errno!=ENOENT ) )
1251 0 : FD_LOG_ERR(( "unlinkat(%s/%s) failed (%i-%s)", snap_dir, entry->d_name, errno, fd_io_strerror( errno ) ));
1252 0 : continue;
1253 0 : }
1254 :
1255 : /* decode file name */
1256 0 : int is_zstd;
1257 0 : ulong entry_full_slot, entry_incremental_slot;
1258 0 : uchar decoded_hash[ FD_HASH_FOOTPRINT ];
1259 0 : if( FD_UNLIKELY( -1==fd_ssarchive_parse_filename( entry->d_name, &entry_full_slot, &entry_incremental_slot, decoded_hash, &is_zstd ) ) ) continue;
1260 0 : if( FD_UNLIKELY( strlen( entry->d_name )>=FD_SNAP_NAME_MAX ) ) continue;
1261 0 : fd_backup_inode_t inode = (fd_backup_inode_t){
1262 0 : .full_slot = entry_full_slot,
1263 0 : .incr_slot = entry_incremental_slot
1264 0 : };
1265 0 : fd_cstr_ncpy( inode.name, entry->d_name, FD_SNAP_NAME_MAX );
1266 :
1267 : /* register snapshot (and lazily delete snaps if there are too many)
1268 : the lazy drop must free a slot for the append below, even if
1269 : retention is configured at the hard capacity */
1270 0 : if( entry_incremental_slot==ULONG_MAX ) {
1271 0 : if( FD_UNLIKELY( snap_full_cnt>=FD_SNAP_MAX ) ) {
1272 0 : drop_old_snaps( dir_fd, snap_dir, snap_full, &snap_full_cnt, fd_ulong_min( snap_full_max, FD_SNAP_MAX-1UL ) );
1273 0 : }
1274 0 : snap_full[ snap_full_cnt++ ] = inode;
1275 0 : } else {
1276 0 : if( FD_UNLIKELY( snap_incr_cnt>=FD_SNAP_MAX ) ) {
1277 0 : drop_old_snaps( dir_fd, snap_dir, snap_incr, &snap_incr_cnt, fd_ulong_min( snap_incr_max, FD_SNAP_MAX-1UL ) );
1278 0 : }
1279 0 : snap_incr[ snap_incr_cnt++ ] = inode;
1280 0 : }
1281 0 : }
1282 0 : if( FD_UNLIKELY( errno ) ) FD_LOG_ERR(( "readdir(%s) failed (%i-%s)", snap_dir, errno, fd_io_strerror( errno ) ));
1283 0 : if( FD_UNLIKELY( -1==closedir( dir ) ) ) FD_LOG_ERR(( "closedir(%s) failed (%i-%s)", snap_dir, errno, fd_io_strerror( errno ) ));
1284 :
1285 : /* FIXME this should be smart enough to drop incremental snapshots
1286 : that no longer have a matching full snapshot. */
1287 0 : drop_old_snaps( dir_fd, snap_dir, snap_full, &snap_full_cnt, snap_full_max );
1288 0 : drop_old_snaps( dir_fd, snap_dir, snap_incr, &snap_incr_cnt, snap_incr_max );
1289 :
1290 : /* zero pad */
1291 0 : for( ulong i=snap_full_cnt; i<snap_full_max; i++ ) snap_full[ i ] = (fd_backup_inode_t){ .full_slot=ULONG_MAX, .incr_slot=ULONG_MAX };
1292 0 : for( ulong i=snap_incr_cnt; i<snap_incr_max; i++ ) snap_incr[ i ] = (fd_backup_inode_t){ .full_slot=ULONG_MAX, .incr_slot=ULONG_MAX };
1293 :
1294 0 : for( ulong i=0UL; i<snap_max; i++ ) {
1295 0 : fd_backup_inode_t * inode;
1296 0 : if( i<snap_full_max ) inode = &snap_full[ i ];
1297 0 : else if( i<retained_snap_max ) inode = &snap_incr[ i-snap_full_max ];
1298 0 : else inode = &scratch [ i-retained_snap_max ];
1299 :
1300 0 : int snap_fd;
1301 0 : if( FD_UNLIKELY( !inode->name[ 0 ] ) ) {
1302 0 : fd_snap_pool_partial_name( inode->name, (uint)i );
1303 0 : snap_fd = openat( dir_fd, inode->name, O_RDWR|O_CREAT|O_EXCL|O_CLOEXEC|O_NOFOLLOW, S_IRUSR|S_IWUSR );
1304 0 : if( FD_UNLIKELY( -1==snap_fd ) ) FD_LOG_ERR(( "openat(%s/%s) failed (%i-%s)", snap_dir, inode->name, errno, fd_io_strerror( errno ) ));
1305 0 : } else {
1306 0 : snap_fd = openat( dir_fd, inode->name, O_RDWR|O_CLOEXEC|O_NOFOLLOW );
1307 0 : if( FD_UNLIKELY( -1==snap_fd ) ) FD_LOG_ERR(( "openat(%s/%s) failed (%i-%s)", snap_dir, inode->name, errno, fd_io_strerror( errno ) ));
1308 0 : snap_check_fd( snap_dir, inode->name, snap_fd );
1309 0 : }
1310 :
1311 0 : snap_reperm_fd( snap_dir, inode->name, snap_fd, config->uid, config->gid );
1312 :
1313 0 : if( FD_UNLIKELY( -1==dup2( snap_fd, FD_SNAP_FD( i ) ) ) ) FD_LOG_ERR(( "dup2() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1314 0 : if( FD_UNLIKELY( -1==close( snap_fd ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1315 :
1316 0 : if( dio_enabled ) {
1317 0 : int dio_fd = openat( dir_fd, inode->name, O_WRONLY|O_DIRECT );
1318 0 : if( FD_UNLIKELY( -1==dio_fd ) ) FD_LOG_ERR(( "openat(%s/%s, O_DIRECT) failed (%i-%s)", snap_dir, inode->name, errno, fd_io_strerror( errno ) ));
1319 0 : if( FD_UNLIKELY( -1==dup2( dio_fd, FD_SNAP_DIO_FD( i ) ) ) ) FD_LOG_ERR(( "dup2() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1320 0 : if( FD_UNLIKELY( -1==close( dio_fd ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1321 0 : }
1322 :
1323 0 : if( upload_enabled ) {
1324 0 : int ro_fd = openat( dir_fd, inode->name, O_RDONLY|O_NOFOLLOW );
1325 0 : if( FD_UNLIKELY( -1==ro_fd ) ) FD_LOG_ERR(( "openat(%s/%s, O_RDONLY) failed (%i-%s)", snap_dir, inode->name, errno, fd_io_strerror( errno ) ));
1326 0 : if( FD_UNLIKELY( -1==dup2( ro_fd, FD_SNAP_RO_FD( i ) ) ) ) FD_LOG_ERR(( "dup2() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1327 0 : if( FD_UNLIKELY( -1==close( ro_fd ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1328 0 : }
1329 0 : }
1330 :
1331 0 : if( FD_UNLIKELY( -1==close( dir_fd ) ) ) FD_LOG_ERR(( "close(%s) failed (%i-%s)", snap_dir, errno, fd_io_strerror( errno ) ));
1332 0 : return snap_max;
1333 0 : }
1334 :
1335 : /* The boot sequence is a little bit involved...
1336 :
1337 : A process tree is created that looks like,
1338 :
1339 : + main
1340 : +-- pidns
1341 : +-- agave
1342 : +-- tile 0
1343 : +-- tile 1
1344 : ...
1345 :
1346 : What we want is that if any process in the tree dies, all other
1347 : processes will also die. This is done as follows,
1348 :
1349 : (a) pidns is the init process of a PID namespace, so if it dies the
1350 : kernel will terminate the child processes.
1351 :
1352 : (b) main is the parent of pidns, so it can issue a waitpid() on the
1353 : child PID, and when it completes terminate itself.
1354 :
1355 : (c) pidns is the parent of agave and the tiles, so it could
1356 : issue a waitpid() of -1 to wait for any of them to terminate,
1357 : but how would it know if main has died?
1358 :
1359 : (d) main creates a pipe, and passes the write end to pidns. If main
1360 : dies, the pipe will be closed, and pidns will get a HUP on the
1361 : read end. Then pidns creates a pipe per child and passes the
1362 : write end to the child. If any of the children die, the pipe
1363 : will be closed, and pidns will get a HUP on the read end.
1364 :
1365 : Then pidns can call poll() on both the write end of the main
1366 : pipe and the read end of all the child pipes. If any of them
1367 : raises SIGHUP, then pidns knows that the parent or a child has
1368 : died, and it can terminate itself, which due to (a) and (b)
1369 : will kill all other processes. */
1370 : void
1371 : run_firedancer( config_t * config,
1372 : int parent_pipefd,
1373 0 : int init_workspaces ) {
1374 : /* dump the topology we are using to the output log */
1375 0 : fd_topo_print_log( 0, &config->topo );
1376 :
1377 0 : run_firedancer_init( config, init_workspaces, 1 );
1378 :
1379 0 : if( FD_UNLIKELY( close( 0 ) ) ) FD_LOG_ERR(( "close(0) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1380 0 : if( FD_UNLIKELY( fd_log_private_logfile_fd()!=1 && close( 1 ) ) ) FD_LOG_ERR(( "close(1) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1381 :
1382 0 : int pipefd;
1383 0 : pid_namespace = clone_firedancer( config, parent_pipefd, &pipefd );
1384 :
1385 : /* Print the location of the logfile on SIGINT or SIGTERM, and also
1386 : kill the child. They are connected by a pipe which the child is
1387 : polling so we don't strictly need to kill the child, but its helpful
1388 : to do that before printing the log location line, else it might
1389 : get interleaved due to timing windows in the shutdown. */
1390 0 : install_parent_signals();
1391 :
1392 0 : struct sock_filter seccomp_filter[ 128UL ];
1393 0 : populate_sock_filter_policy_main( 128UL, seccomp_filter, (uint)fd_log_private_logfile_fd(), (uint)pid_namespace );
1394 :
1395 0 : int allow_fds[ 4 ];
1396 0 : ulong allow_fds_cnt = 0;
1397 0 : allow_fds[ allow_fds_cnt++ ] = 2; /* stderr */
1398 0 : if( FD_LIKELY( fd_log_private_logfile_fd()!=-1 ) )
1399 0 : allow_fds[ allow_fds_cnt++ ] = fd_log_private_logfile_fd(); /* logfile, or maybe stdout */
1400 0 : allow_fds[ allow_fds_cnt++ ] = pipefd; /* read end of main pipe */
1401 0 : if( FD_UNLIKELY( parent_pipefd!=-1 ) )
1402 0 : allow_fds[ allow_fds_cnt++ ] = parent_pipefd; /* write end of parent pipe */
1403 :
1404 0 : if( FD_LIKELY( config->development.sandbox ) ) {
1405 0 : fd_sandbox_enter( config->uid,
1406 0 : config->gid,
1407 0 : 0,
1408 0 : 0,
1409 0 : 0,
1410 0 : 1, /* Keep controlling terminal for main so it can receive Ctrl+C */
1411 0 : 0,
1412 0 : 0UL,
1413 0 : 0UL,
1414 0 : 0UL,
1415 0 : 0UL,
1416 0 : allow_fds_cnt,
1417 0 : allow_fds,
1418 0 : sock_filter_policy_main_instr_cnt,
1419 0 : seccomp_filter );
1420 0 : } else {
1421 0 : fd_sandbox_switch_uid_gid( config->uid, config->gid );
1422 0 : }
1423 :
1424 : /* the only clean way to exit is SIGINT or SIGTERM on this parent process,
1425 : so if wait4() completes, it must be an error */
1426 0 : int wstatus;
1427 0 : if( FD_UNLIKELY( -1==wait4( pid_namespace, &wstatus, (int)__WALL, NULL ) ) )
1428 0 : FD_LOG_ERR(( "main wait4() failed (%i-%s)\nLog at \"%s\"", errno, fd_io_strerror( errno ), fd_log_private_path ));
1429 :
1430 0 : if( FD_UNLIKELY( WIFSIGNALED( wstatus ) ) ) fd_sys_util_exit_group( WTERMSIG( wstatus ) ? WTERMSIG( wstatus ) : 1 );
1431 0 : else fd_sys_util_exit_group( WEXITSTATUS( wstatus ) ? WEXITSTATUS( wstatus ) : 1 );
1432 0 : }
1433 :
1434 : void
1435 : run_cmd_fn( args_t * args FD_PARAM_UNUSED,
1436 0 : config_t * config ) {
1437 0 : #define CHECK_PORT_NON_ZERO( field ) \
1438 0 : if( FD_UNLIKELY( config->field==0 ) ) { \
1439 0 : FD_LOG_ERR(( #field " is not set in your configuration file. Please set it to a non-zero value." )); \
1440 0 : }
1441 :
1442 0 : if( FD_UNLIKELY( !config->gossip.entrypoints_cnt && !config->development.bootstrap ) )
1443 0 : FD_LOG_ERR(( "No entrypoints specified in configuration file under [gossip.entrypoints], but "
1444 0 : "at least one is needed to determine how to connect to the Solana cluster. If "
1445 0 : "you want to start a new cluster in a development environment, use `fddev` instead "
1446 0 : "of `fdctl`. If you want to use an existing genesis, set [development.bootstrap] "
1447 0 : "to \"true\" in the configuration file." ));
1448 :
1449 0 : for( ulong i=0; i<config->gossip.entrypoints_cnt; i++ ) {
1450 0 : if( FD_UNLIKELY( !strcmp( config->gossip.entrypoints[ i ], "" ) ) )
1451 0 : FD_LOG_ERR(( "One of the entrypoints in your configuration file under [gossip.entrypoints] is "
1452 0 : "empty. Please remove the empty entrypoint or set it correctly. "));
1453 0 : }
1454 :
1455 0 : CHECK_PORT_NON_ZERO( gossip.port );
1456 0 : CHECK_PORT_NON_ZERO( tiles.quic.quic_transaction_listen_port );
1457 0 : CHECK_PORT_NON_ZERO( tiles.quic.regular_transaction_listen_port );
1458 0 : CHECK_PORT_NON_ZERO( tiles.shred.shred_listen_port );
1459 0 : CHECK_PORT_NON_ZERO( tiles.metric.prometheus_listen_port );
1460 0 : CHECK_PORT_NON_ZERO( tiles.gui.gui_listen_port );
1461 :
1462 0 : #undef CHECK_PORT_NON_ZERO
1463 :
1464 0 : run_firedancer( config, -1, 1 );
1465 0 : }
1466 :
1467 : static void
1468 0 : run1_args_help( fd_action_help_t * help ) {
1469 0 : fd_action_help_arg( help, "<tile-name>", NULL, "Type of tile to run (e.g. `net`, `quic`, `replay`). A tile is a single\n"
1470 0 : "thread pinned to a CPU core that performs one part of the validator's work" );
1471 : fd_action_help_arg( help, "<kind-id>", NULL, "Zero-based index selecting which instance of that tile type to run\n"
1472 0 : "when the topology has more than one" );
1473 0 : fd_action_help_arg( help, "--pipe-fd", "<fd>", "Internal use: file descriptor over which the parent supervisor process\n"
1474 0 : "communicates with this tile (default -1, standalone)" );
1475 0 : }
1476 :
1477 : action_t fd_action_run1 = {
1478 : .name = "run1",
1479 : .args = run1_cmd_args,
1480 : .fn = run1_cmd_fn,
1481 : .perm = NULL,
1482 : .description = "Start up a single Firedancer tile",
1483 : .detail = "Runs one tile of the validator topology in the current process. A tile is a\n"
1484 : "single thread pinned to a CPU core that performs one part of the validator's\n"
1485 : "work. This is primarily an internal command used by `run` to spawn individual\n"
1486 : "tiles; most operators should use `run` instead.",
1487 : .usage = "run1 <tile-name> <kind-id> [OPTIONS]",
1488 : .args_help = run1_args_help,
1489 : };
1490 :
1491 : action_t fd_action_run = {
1492 : .name = "run",
1493 : .args = NULL,
1494 : .fn = run_cmd_fn,
1495 : .require_config = 1,
1496 : .perm = run_cmd_perm,
1497 : .description = "Start up a Firedancer validator",
1498 : .detail = "Boots and runs the full validator described by the configuration file. This\n"
1499 : "is the main command operators use to run Firedancer. It must be started with\n"
1500 : "sufficient privileges to perform boot-time setup, after which it drops\n"
1501 : "privileges to the configured user.",
1502 : .usage = "run [OPTIONS]",
1503 : .permission_err = "insufficient permissions to execute command `%s`. It is recommended "
1504 : "to start Firedancer as the root user, but you can also start it "
1505 : "with the missing capabilities listed above. The program only needs "
1506 : "to start with elevated permissions to do privileged operations at "
1507 : "boot, and will immediately drop permissions and switch to the user "
1508 : "specified in your configuration file once they are complete. Firedancer "
1509 : "will not execute outside of the boot process as root, and will refuse "
1510 : "to start if it cannot drop privileges. Firedancer needs to be started "
1511 : "privileged to configure high performance networking with XDP.",
1512 : };
|