LCOV - code coverage report
Current view: top level - app/shared/commands/run - run.c (source / functions) Hit Total Coverage
Test: cov.lcov Lines: 0 1010 0.0 %
Date: 2026-09-17 04:28:31 Functions: 0 28 0.0 %

          Line data    Source code
       1             : #define _GNU_SOURCE
       2             : #include "run.h"
       3             : #include "../../../../flamenco/accdb/fd_accdb.h"
       4             : #include "../../../../disco/store/fd_store.h"
       5             : 
       6             : #include <sys/wait.h>
       7             : #include "generated/main_seccomp.h"
       8             : #if defined(__aarch64__)
       9             : #include "generated/pidns_arm64_seccomp.h"
      10             : #else
      11             : #include "generated/pidns_seccomp.h"
      12             : #endif
      13             : 
      14             : #include "../../fd_bootinfo.h"
      15             : #include "../../../platform/fd_sys_util.h"
      16             : #include "../../../platform/fd_file_util.h"
      17             : #include "../../../platform/fd_net_util.h"
      18             : #include "../../../../disco/net/fd_net_tile.h"
      19             : #include "../../../../discof/backup/fd_backup.h"
      20             : #include "../../../../discof/backup/fd_snap_pool.h"
      21             : #include "../../../../discof/restore/utils/fd_ssarchive.h"
      22             : #include "../../../../disco/waker/fd_waker.h"
      23             : #include "../../../../util/pod/fd_pod_format.h"
      24             : 
      25             : #include "../configure/configure.h"
      26             : #include "../configure/fd_cpu_isolation.h"
      27             : 
      28             : #include <dirent.h>
      29             : #include <sched.h>
      30             : #include <stdio.h>
      31             : #include <stdlib.h> /* getenv */
      32             : #include <poll.h>
      33             : #include <unistd.h>
      34             : #include <errno.h>
      35             : #include <fcntl.h>
      36             : #include <sys/prctl.h>
      37             : #include <sys/resource.h>
      38             : #include <sys/mman.h>
      39             : #include <sys/stat.h>
      40             : #include <linux/capability.h>
      41             : 
      42             : #include "../../../../util/tile/fd_tile_private.h"
      43             : 
      44             : extern fd_topo_obj_callbacks_t * CALLBACKS[];
      45             : 
      46           0 : #define NAME "run"
      47             : 
      48             : void
      49             : run_cmd_perm( args_t *         args,
      50             :               fd_cap_chk_t *   chk,
      51           0 :               config_t const * config ) {
      52           0 :   (void)args;
      53             : 
      54           0 :   ulong mlock_limit = fd_topo_mlock_max_tile( &config->topo );
      55             : 
      56           0 :   fd_cap_chk_raise_rlimit( chk, NAME, RLIMIT_MEMLOCK, mlock_limit, "call `rlimit(2)` to increase `RLIMIT_MEMLOCK` so all memory can be locked with `mlock(2)`" );
      57           0 :   fd_cap_chk_raise_rlimit( chk, NAME, RLIMIT_NICE,    40,          "call `setpriority(2)` to increase thread priorities" );
      58           0 :   fd_cap_chk_raise_rlimit( chk, NAME, RLIMIT_NOFILE,  CONFIGURE_NR_OPEN_FILES,
      59           0 :                                                                        "call `rlimit(2)  to increase `RLIMIT_NOFILE` to allow more open files for Agave" );
      60           0 :   fd_cap_chk_cap(          chk, NAME, CAP_NET_RAW,                 "call `socket(2)` to bind to a raw socket for use by XDP" );
      61           0 :   fd_cap_chk_cap(          chk, NAME, CAP_SYS_ADMIN,               "call `bpf(2)` with the `BPF_OBJ_GET` command to initialize XDP" );
      62           0 :   if( fd_sandbox_requires_cap_sys_admin( config->uid, config->gid ) )
      63           0 :     fd_cap_chk_cap(        chk, NAME, CAP_SYS_ADMIN,               "call `unshare(2)` with `CLONE_NEWUSER` to sandbox the process in a user namespace" );
      64           0 :   if( FD_LIKELY( getuid() != config->uid ) )
      65           0 :     fd_cap_chk_cap(        chk, NAME, CAP_SETUID,                  "call `setresuid(2)` to switch uid to the sandbox user" );
      66           0 :   if( FD_LIKELY( getgid()!=config->gid ) )
      67           0 :     fd_cap_chk_cap(        chk, NAME, CAP_SETGID,                  "call `setresgid(2)` to switch gid to the sandbox user" );
      68           0 :   if( FD_UNLIKELY( config->tiles.metric.prometheus_listen_port<1024 ) )
      69           0 :     fd_cap_chk_cap(        chk, NAME, CAP_NET_BIND_SERVICE,        "call `bind(2)` to bind to a privileged port for serving metrics" );
      70           0 :   if( FD_UNLIKELY( config->tiles.gui.gui_listen_port<1024 ) )
      71           0 :     fd_cap_chk_cap(        chk, NAME, CAP_NET_BIND_SERVICE,        "call `bind(2)` to bind to a privileged port for serving the GUI" );
      72           0 : }
      73             : 
      74             : struct pidns_clone_args {
      75             :   config_t const * config;
      76             :   int *            pipefd;
      77             :   int              closefd;
      78             : };
      79             : 
      80             : extern char fd_log_private_path[ 1024 ]; /* empty string on start */
      81             : 
      82             : static pid_t pid_namespace;
      83             : 
      84           0 : #define FD_LOG_ERR_NOEXIT(a) do { long _fd_log_msg_now = fd_log_wallclock(); fd_log_private_1( 4, _fd_log_msg_now, __FILE__, __LINE__, __func__, fd_log_private_0 a ); } while(0)
      85             : 
      86             : static void
      87           0 : parent_signal( int sig ) {
      88           0 :   if( FD_LIKELY( pid_namespace ) ) kill( pid_namespace, SIGKILL );
      89             : 
      90           0 :   if( -1!=fd_log_private_logfile_fd() ) FD_LOG_ERR_NOEXIT(( "Received signal %s%s%s %s(%s)%s\n%sLog at \"%s\"%s", fd_log_style_bold(), fd_io_strsignal_name( sig ), fd_log_style_normal(), fd_log_style_dim(), fd_io_strsignal_desc( sig ), fd_log_style_normal(), fd_log_style_dim(), fd_log_private_path, fd_log_style_normal() ));
      91           0 :   else                                  FD_LOG_ERR_NOEXIT(( "Received signal %s%s%s %s(%s)%s",                fd_log_style_bold(), fd_io_strsignal_name( sig ), fd_log_style_normal(), fd_log_style_dim(), fd_io_strsignal_desc( sig ), fd_log_style_normal() ));
      92             : 
      93           0 :   if( FD_LIKELY( sig==SIGINT ) ) fd_sys_util_exit_group( 128+SIGINT );
      94           0 :   else                           fd_sys_util_exit_group( 0          );
      95           0 : }
      96             : 
      97             : static void
      98           0 : install_parent_signals( void ) {
      99           0 :   struct sigaction sa = {
     100           0 :     .sa_handler = parent_signal,
     101           0 :     .sa_flags   = 0,
     102           0 :   };
     103           0 :   if( FD_UNLIKELY( sigaction( SIGTERM, &sa, NULL ) ) )
     104           0 :     FD_LOG_ERR(( "sigaction(SIGTERM) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     105           0 :   if( FD_UNLIKELY( sigaction( SIGINT, &sa, NULL ) ) )
     106           0 :     FD_LOG_ERR(( "sigaction(SIGINT) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     107             : 
     108           0 :   sa.sa_handler = SIG_IGN;
     109           0 :   if( FD_UNLIKELY( sigaction( SIGUSR1, &sa, NULL ) ) )
     110           0 :     FD_LOG_ERR(( "sigaction(SIGUSR1) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     111           0 :   if( FD_UNLIKELY( sigaction( SIGUSR2, &sa, NULL ) ) )
     112           0 :     FD_LOG_ERR(( "sigaction(SIGUSR2) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     113           0 : }
     114             : 
     115             : void *
     116           0 : create_clone_stack( void ) {
     117           0 :   ulong mmap_sz = FD_TILE_PRIVATE_STACK_SZ + 2UL*FD_SHMEM_NORMAL_PAGE_SZ;
     118           0 :   uchar * stack = (uchar *)mmap( NULL, mmap_sz, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, (off_t)0 );
     119           0 :   if( FD_UNLIKELY( stack==MAP_FAILED ) )
     120           0 :     FD_LOG_ERR(( "mmap() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     121             : 
     122             :   /* Make space for guard lo and guard hi */
     123           0 :   if( FD_UNLIKELY( munmap( stack, FD_SHMEM_NORMAL_PAGE_SZ ) ) )
     124           0 :     FD_LOG_ERR(( "munmap failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     125           0 :   stack += FD_SHMEM_NORMAL_PAGE_SZ;
     126           0 :   if( FD_UNLIKELY( munmap( stack + FD_TILE_PRIVATE_STACK_SZ, FD_SHMEM_NORMAL_PAGE_SZ ) ) )
     127           0 :     FD_LOG_ERR(( "munmap failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     128             : 
     129             :   /* Create the guard regions in the extra space */
     130           0 :   void * guard_lo = (void *)(stack - FD_SHMEM_NORMAL_PAGE_SZ );
     131           0 :   if( FD_UNLIKELY( mmap( guard_lo, FD_SHMEM_NORMAL_PAGE_SZ, PROT_NONE,
     132           0 :                          MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, (off_t)0 )!=guard_lo ) )
     133           0 :     FD_LOG_ERR(( "mmap failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     134             : 
     135           0 :   void * guard_hi = (void *)(stack + FD_TILE_PRIVATE_STACK_SZ);
     136           0 :   if( FD_UNLIKELY( mmap( guard_hi, FD_SHMEM_NORMAL_PAGE_SZ, PROT_NONE,
     137           0 :                          MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, (off_t)0 )!=guard_hi ) )
     138           0 :     FD_LOG_ERR(( "mmap failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     139             : 
     140           0 :   return stack;
     141           0 : }
     142             : 
     143             : 
     144             : static int
     145             : execve_agave( int config_memfd,
     146           0 :                     int pipefd ) {
     147           0 :   if( FD_UNLIKELY( -1==fcntl( pipefd, F_SETFD, 0 ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,0) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     148           0 :   pid_t child = fork();
     149           0 :   if( FD_UNLIKELY( -1==child ) ) FD_LOG_ERR(( "fork() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     150           0 :   if( FD_LIKELY( !child ) ) {
     151           0 :     char _current_executable_path[ PATH_MAX ];
     152           0 :     FD_TEST( -1!=fd_file_util_self_exe( _current_executable_path ) );
     153             : 
     154           0 :     char config_fd[ 32 ];
     155           0 :     FD_TEST( fd_cstr_printf_check( config_fd, sizeof( config_fd ), NULL, "%d", config_memfd ) );
     156           0 :     char * args[ 5 ] = { _current_executable_path, "run-agave", "--config-fd", config_fd, NULL };
     157             : 
     158           0 :     char * envp[] = { NULL, NULL };
     159           0 :     char * google_creds = getenv( "GOOGLE_APPLICATION_CREDENTIALS" );
     160           0 :     char provide_creds[ PATH_MAX+30UL ];
     161           0 :     if( FD_UNLIKELY( google_creds ) ) {
     162           0 :       FD_TEST( fd_cstr_printf_check( provide_creds, sizeof( provide_creds ), NULL, "GOOGLE_APPLICATION_CREDENTIALS=%s", google_creds ) );
     163           0 :       envp[ 0 ] = provide_creds;
     164           0 :     }
     165             : 
     166           0 :     if( FD_UNLIKELY( -1==execve( _current_executable_path, args, envp ) ) ) FD_LOG_ERR(( "execve() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     167           0 :   } else {
     168           0 :     if( FD_UNLIKELY( -1==fcntl( pipefd, F_SETFD, FD_CLOEXEC ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,FD_CLOEXEC) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     169           0 :     return child;
     170           0 :   }
     171           0 :   return 0;
     172           0 : }
     173             : 
     174             : static int
     175           0 : cgroup_procs_write( char const * path ) {
     176           0 :   int fd = open( path, O_WRONLY );
     177           0 :   if( FD_UNLIKELY( fd<0 ) ) {
     178           0 :     if( FD_LIKELY( errno==ENOENT ) ) return 0;
     179           0 :     FD_LOG_ERR(( "open(%s) failed (%i-%s)", path, errno, fd_io_strerror( errno ) ));
     180           0 :   }
     181             : 
     182           0 :   char pid[ 32 ];
     183           0 :   ulong pid_len;
     184           0 :   FD_TEST( fd_cstr_printf_check( pid, sizeof(pid), &pid_len, "%ld", (long)getpid() ) );
     185           0 :   if( FD_UNLIKELY( write( fd, pid, pid_len )!=(long)pid_len ) )
     186           0 :     FD_LOG_ERR(( "write(%s,\"%s\") failed (%i-%s)", path, pid, errno, fd_io_strerror( errno ) ));
     187           0 :   if( FD_UNLIKELY( close( fd ) ) ) FD_LOG_ERR(( "close(%s) failed (%i-%s)", path, errno, fd_io_strerror( errno ) ));
     188           0 :   return 1;
     189           0 : }
     190             : 
     191             : struct spawn_cgroup {
     192             :   int  probed;    /* isolation cgroup existence checked, original saved */
     193             :   int  present;   /* isolation cgroup exists */
     194             :   int  joined;    /* currently a member of the isolation cgroup */
     195             :   char isolation[ PATH_MAX ];
     196             :   char original[ PATH_MAX ];
     197             : };
     198             : 
     199             : static void
     200             : join_isolation_cgroup( char const *          name,
     201           0 :                        struct spawn_cgroup * cg ) {
     202           0 :   if( FD_UNLIKELY( !cg->probed ) ) {
     203           0 :     cg->probed = 1;
     204           0 :     FD_TEST( fd_cstr_printf_check( cg->isolation, sizeof(cg->isolation), NULL, "/sys/fs/cgroup/%s/cgroup.procs", name ) );
     205             : 
     206             :     /* The cpuset stage not being configured (no cgroup) is the common
     207             :        case and must be decided FIRST: on cgroup v1-only or hybrid
     208             :        hosts /proc/self/cgroup does not have the v2 format, and
     209             :        validating it before knowing the stage is even in use would
     210             :        turn every tile launch on such hosts into a fatal error. */
     211           0 :     cg->present = !access( cg->isolation, F_OK );
     212           0 :     if( FD_LIKELY( !cg->present ) ) return;
     213             : 
     214             :     /* Remember where we came from.  The v2 entry in /proc/self/cgroup
     215             :        is the line "0::<path>"; on a pure v2 hierarchy it is the only
     216             :        line, but on hybrid systems v1 controller lines precede it, so
     217             :        search rather than assume. */
     218           0 :     char buf[ 4096 ];
     219           0 :     int fd = open( "/proc/self/cgroup", O_RDONLY );
     220           0 :     if( FD_UNLIKELY( fd<0 ) ) FD_LOG_ERR(( "open(/proc/self/cgroup) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     221           0 :     long n = read( fd, buf, sizeof(buf)-1UL );
     222           0 :     if( FD_UNLIKELY( n<0L ) ) FD_LOG_ERR(( "read(/proc/self/cgroup) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     223           0 :     if( FD_UNLIKELY( close( fd ) ) ) FD_LOG_ERR(( "close(/proc/self/cgroup) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     224           0 :     buf[ n ] = '\0';
     225             : 
     226           0 :     char * line = buf;
     227           0 :     while( line && strncmp( line, "0::", 3UL ) ) {
     228           0 :       line = strchr( line, '\n' );
     229           0 :       if( FD_LIKELY( line ) ) line++;
     230           0 :     }
     231           0 :     if( FD_UNLIKELY( !line || !line[ 0 ] ) )
     232           0 :       FD_LOG_ERR(( "no cgroup v2 entry in /proc/self/cgroup while the cpuset isolation cgroup `/sys/fs/cgroup/%s` "
     233           0 :                    "exists. Remove it with `%s configure fini cpuset`", name, FD_BINARY_NAME ));
     234           0 :     char * nl = strchr( line, '\n' ); if( FD_LIKELY( nl ) ) *nl = '\0';
     235           0 :     FD_TEST( fd_cstr_printf_check( cg->original, PATH_MAX, NULL,
     236           0 :                                    "/sys/fs/cgroup%s/cgroup.procs", line+3UL ) );
     237           0 :   }
     238             : 
     239           0 :   if( FD_UNLIKELY( !cg->present || cg->joined ) ) return;
     240           0 :   cg->joined = cgroup_procs_write( cg->isolation );
     241           0 : }
     242             : 
     243             : static void
     244           0 : leave_isolation_cgroup( struct spawn_cgroup * cg ) {
     245           0 :   if( FD_LIKELY( !cg->joined ) ) return;
     246           0 :   if( FD_UNLIKELY( !cgroup_procs_write( cg->original ) ) ) FD_LOG_ERR(( "could not return to original cgroup `%s`", cg->original ));
     247           0 :   cg->joined = 0;
     248           0 : }
     249             : 
     250             : static pid_t
     251             : execve_tile( char const *           name,
     252             :              fd_topo_tile_t const * tile,
     253             :              fd_cpuset_t const *    float_cpu_set,
     254             :              fd_cpuset_t const *    floating_cpu_set,
     255             :              int                    floating_priority,
     256             :              int                    config_memfd,
     257             :              int                    pipefd,
     258           0 :              struct spawn_cgroup *  cg ) {
     259           0 :   FD_CPUSET_DECL( cpu_set );
     260           0 :   if( FD_LIKELY( tile->cpu_idx!=ULONG_MAX ) ) {
     261             :     /* Join the cpuset isolation cgroup (if configured) and set the
     262             :        thread affinity before we clone the new process, to ensure
     263             :        kernel first touch happens on the desired thread.  The child
     264             :        inherits both. */
     265           0 :     join_isolation_cgroup( name, cg );
     266           0 :     if( FD_UNLIKELY( tile->floats ) ) {
     267             :       /* the floating CPUs on this tile's NUMA node: memory was placed
     268             :          by cpu_idx */
     269           0 :       ulong numa_idx = fd_shmem_numa_idx( tile->cpu_idx );
     270           0 :       for( ulong cpu=0UL; cpu<FD_TILE_MAX; cpu++ )
     271           0 :         if( fd_cpuset_test( float_cpu_set, cpu ) && fd_shmem_numa_idx( cpu )==numa_idx ) fd_cpuset_insert( cpu_set, cpu );
     272           0 :     }
     273           0 :     if( FD_UNLIKELY( !fd_cpuset_cnt( cpu_set ) ) ) fd_cpuset_insert( cpu_set, tile->cpu_idx );
     274           0 :     if( FD_UNLIKELY( -1==setpriority( PRIO_PROCESS, 0, -19 ) ) ) FD_LOG_ERR(( "setpriority() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     275           0 :   } else {
     276           0 :     leave_isolation_cgroup( cg );
     277           0 :     fd_memcpy( cpu_set, floating_cpu_set, fd_cpuset_footprint() );
     278           0 :     if( FD_UNLIKELY( -1==setpriority( PRIO_PROCESS, 0, floating_priority ) ) ) FD_LOG_ERR(( "setpriority() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     279           0 :   }
     280             : 
     281           0 :   if( FD_UNLIKELY( fd_cpuset_setaffinity( 0, cpu_set ) ) ) {
     282           0 :     if( FD_LIKELY( errno==EINVAL ) ) {
     283           0 :       FD_LOG_ERR(( "Unable to set the thread affinity for tile %s:%lu on cpu %lu. It is likely that the affinity "
     284           0 :                    "you have specified for this tile in [layout.affinity] of your configuration file contains a "
     285           0 :                    "CPU (%lu) which does not exist on this machine.",
     286           0 :                    tile->name, tile->kind_id, tile->cpu_idx, tile->cpu_idx ));
     287           0 :     } else {
     288           0 :       FD_LOG_ERR(( "sched_setaffinity failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     289           0 :     }
     290           0 :   }
     291             : 
     292             :   /* Clear CLOEXEC on the side of the pipe we want to pass to the tile. */
     293           0 :   if( FD_UNLIKELY( -1==fcntl( pipefd, F_SETFD, 0 ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,0) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     294           0 :   pid_t child = fork();
     295           0 :   if( FD_UNLIKELY( -1==child ) ) FD_LOG_ERR(( "fork() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     296           0 :   if( FD_LIKELY( !child ) ) {
     297           0 :     char _current_executable_path[ PATH_MAX ];
     298           0 :     FD_TEST( -1!=fd_file_util_self_exe( _current_executable_path ) );
     299             : 
     300           0 :     char kind_id[ 32 ], config_fd[ 32 ], pipe_fd[ 32 ];
     301           0 :     FD_TEST( fd_cstr_printf_check( kind_id,   sizeof( kind_id ),   NULL, "%lu", tile->kind_id ) );
     302           0 :     FD_TEST( fd_cstr_printf_check( config_fd, sizeof( config_fd ), NULL, "%d",  config_memfd ) );
     303           0 :     FD_TEST( fd_cstr_printf_check( pipe_fd,   sizeof( pipe_fd ),   NULL, "%d",  pipefd ) );
     304           0 :     char const * args[ 9 ] = { _current_executable_path, "run1", tile->name, kind_id, "--pipe-fd", pipe_fd, "--config-fd", config_fd, NULL };
     305           0 :     if( FD_UNLIKELY( -1==execve( _current_executable_path, (char **)args, NULL ) ) ) FD_LOG_ERR(( "execve() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     306           0 :   } else {
     307           0 :     if( FD_UNLIKELY( -1==fcntl( pipefd, F_SETFD, FD_CLOEXEC ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,FD_CLOEXEC) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     308           0 :     return child;
     309           0 :   }
     310           0 :   return 0;
     311           0 : }
     312             : 
     313             : int
     314           0 : main_pid_namespace( void * _args ) {
     315           0 :   struct pidns_clone_args * args = _args;
     316           0 :   if( FD_UNLIKELY( close( args->pipefd[ 0 ] ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     317           0 :   if( FD_UNLIKELY( -1!=args->closefd ) ) {
     318           0 :     if( FD_UNLIKELY( close( args->closefd ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     319           0 :   }
     320             : 
     321           0 :   config_t const * config = args->config;
     322             : 
     323           0 :   fd_log_thread_set( "pidns" );
     324           0 :   ulong pid = fd_sandbox_getpid(); /* Need to read /proc again.. we got a new PID from clone */
     325           0 :   fd_log_private_group_id_set( pid );
     326           0 :   fd_log_private_thread_id_set( pid );
     327           0 :   fd_log_private_stack_discover( FD_TILE_PRIVATE_STACK_SZ,
     328           0 :                                  &fd_tile_private_stack0, &fd_tile_private_stack1 );
     329             : 
     330           0 :   if( FD_UNLIKELY( !config->development.sandbox ) ) {
     331             :     /* If no sandbox, then there's no actual PID namespace so we can't
     332             :        wait() grandchildren for the exit code.  Do this as a workaround. */
     333           0 :     if( FD_UNLIKELY( -1==prctl( PR_SET_CHILD_SUBREAPER, 1, 0, 0, 0 ) ) )
     334           0 :       FD_LOG_ERR(( "prctl(PR_SET_CHILD_SUBREAPER) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     335           0 :   }
     336             : 
     337             :   /* Save the current affinity, it will be restored after creating any child tiles */
     338           0 :   FD_CPUSET_DECL( floating_cpu_set );
     339           0 :   if( FD_UNLIKELY( fd_cpuset_getaffinity( 0, floating_cpu_set ) ) )
     340           0 :     FD_LOG_ERR(( "fd_cpuset_getaffinity failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     341             : 
     342             :   /* The CPUs of the floating tiles (efficient mode): floaters share
     343             :      these among themselves and never a pinned tile's CPU */
     344           0 :   FD_CPUSET_DECL( float_cpu_set );
     345           0 :   int any_floats = 0;
     346           0 :   for( ulong i=0UL; i<config->topo.tile_cnt; i++ ) {
     347           0 :     if( FD_LIKELY( !config->topo.tiles[ i ].floats ) ) continue;
     348           0 :     fd_cpuset_insert( float_cpu_set, config->topo.tiles[ i ].cpu_idx );
     349           0 :     any_floats = 1;
     350           0 :   }
     351           0 :   for( ulong i=0UL; i<config->topo.tile_cnt; i++ )
     352           0 :     if( FD_LIKELY( !config->topo.tiles[ i ].floats && config->topo.tiles[ i ].cpu_idx!=ULONG_MAX ) ) fd_cpuset_remove( float_cpu_set, config->topo.tiles[ i ].cpu_idx );
     353             : 
     354           0 :   pid_t child_pids[ FD_TOPO_MAX_TILES+1 ];
     355           0 :   ulong actual_pids[ FD_TOPO_MAX_TILES+1 ];
     356           0 :   for( ulong i=0UL; i<FD_TOPO_MAX_TILES+1; i++ ) actual_pids[ i ] = ULONG_MAX;
     357           0 :   char  child_names[ FD_TOPO_MAX_TILES+1 ][ 32 ];
     358           0 :   ulong child_idxs[ FD_TOPO_MAX_TILES+1 ];
     359           0 :   struct pollfd fds[ FD_TOPO_MAX_TILES+2 ];
     360             : 
     361           0 :   int config_memfd = fd_config_to_memfd( config );
     362           0 :   if( FD_UNLIKELY( -1==config_memfd ) ) FD_LOG_ERR(( "fd_config_to_memfd() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     363             : 
     364           0 :   int need_mlx5 = 0==strcmp( config->net.provider, "mlx5" );
     365           0 :   fd_mlx5_fds_t mlx5_fds = { .cmd_fd=-1, .async_fd=-1 };
     366           0 :   if( need_mlx5 ) {
     367           0 :     fd_topo_install_mlx5( (fd_topo_t *)&config->topo, &mlx5_fds );
     368           0 :   }
     369             : 
     370           0 :   ulong child_cnt = 0UL;
     371           0 :   if( FD_LIKELY( !config->is_firedancer && !config->development.no_agave ) ) {
     372           0 :     int pipefd[ 2 ];
     373           0 :     if( FD_UNLIKELY( pipe2( pipefd, O_CLOEXEC ) ) ) FD_LOG_ERR(( "pipe2() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     374           0 :     fds[ child_cnt ] = (struct pollfd){ .fd = pipefd[ 0 ], .events = 0 };
     375           0 :     child_pids[ child_cnt ] = execve_agave( config_memfd, pipefd[ 1 ] );
     376           0 :     FD_TEST( child_pids[ child_cnt ]>0 );
     377           0 :     actual_pids[ child_cnt ] = (ulong)child_pids[ child_cnt ];
     378           0 :     child_idxs[ child_cnt ] = ULONG_MAX;
     379           0 :     if( FD_UNLIKELY( close( pipefd[ 1 ] ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     380           0 :     strncpy( child_names[ child_cnt ], "agave", 32 );
     381           0 :     child_cnt++;
     382           0 :   }
     383             : 
     384           0 :   errno = 0;
     385           0 :   int save_priority = getpriority( PRIO_PROCESS, 0 );
     386           0 :   if( FD_UNLIKELY( -1==save_priority && errno ) ) FD_LOG_ERR(( "getpriority() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     387             : 
     388           0 :   int need_xdp = 0==strcmp( config->net.provider, "xdp" );
     389           0 :   fd_xdp_fds_t xdp_fds[ FD_TOPO_XDP_FDS_MAX ];
     390           0 :   uint         xdp_fds_cnt = FD_TOPO_XDP_FDS_MAX;
     391           0 :   if( need_xdp ) {
     392           0 :     fd_topo_install_xdp( &config->topo, xdp_fds, &xdp_fds_cnt, config->net.bind_address_parsed, 0 );
     393           0 :   }
     394             : 
     395           0 :   initialize_accdb_fd( config );
     396           0 :   initialize_store_fds( config );
     397           0 :   ulong store_obj_id = fd_pod_query_ulong( config->topo.props, "store", ULONG_MAX );
     398           0 :   int   has_store     = store_obj_id!=ULONG_MAX;
     399           0 :   ulong snap_max                = 0UL;
     400           0 :   int   snapshot_upload_enabled = 0;
     401           0 :   int   snapshot_dio_enabled    = 0;
     402           0 :   if( config->is_firedancer ) {
     403           0 :     snap_max                = initialize_snapshot_fds( config );
     404           0 :     snapshot_upload_enabled = fd_topo_find_tile( &config->topo, "snapsv", 0UL )!=ULONG_MAX;
     405           0 :     snapshot_dio_enabled    = fd_topo_find_tile( &config->topo, "snapzp", 0UL )!=ULONG_MAX;
     406           0 :   }
     407             : 
     408           0 :   ulong waker_client_cnt = 0UL;
     409           0 :   for( ulong i=0UL; i<config->topo.tile_cnt; i++ ) {
     410           0 :     ulong idx = config->topo.tiles[ i ].waker_client_idx;
     411           0 :     if( FD_UNLIKELY( idx!=ULONG_MAX ) ) waker_client_cnt = fd_ulong_max( waker_client_cnt, idx+1UL );
     412           0 :   }
     413           0 :   fd_waker_install( waker_client_cnt );
     414             : 
     415           0 :   struct spawn_cgroup spawn_cg = {0};
     416             : 
     417           0 :   for( ulong pass=0UL; pass<2UL; pass++ ) {
     418           0 :     for( ulong i=0UL; i<config->topo.tile_cnt; i++ ) {
     419           0 :       fd_topo_tile_t const * tile = &config->topo.tiles[ i ];
     420           0 :       if( FD_UNLIKELY( tile->is_agave ) ) continue;
     421           0 :       if( FD_UNLIKELY( (tile->cpu_idx!=ULONG_MAX)!=pass ) ) continue;
     422             : 
     423           0 :       if( need_xdp ) {
     424           0 :         if( FD_UNLIKELY( strcmp( tile->name, "net" ) ) ) {
     425           0 :           for( uint i=0U; i<xdp_fds_cnt; i++ ) {
     426             :             /* close XDP related file descriptors */
     427           0 :             if( FD_UNLIKELY( -1==fcntl( xdp_fds[i].xsk_map_fd,   F_SETFD, FD_CLOEXEC ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,FD_CLOEXEC) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     428           0 :             if( FD_UNLIKELY( -1==fcntl( xdp_fds[i].prog_link_fd, F_SETFD, FD_CLOEXEC ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,FD_CLOEXEC) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     429           0 :           }
     430           0 :         } else {
     431           0 :           for( uint i=0U; i<xdp_fds_cnt; i++ ) {
     432           0 :             if( FD_UNLIKELY( -1==fcntl( xdp_fds[i].xsk_map_fd,   F_SETFD, 0 ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,0) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     433           0 :             if( FD_UNLIKELY( -1==fcntl( xdp_fds[i].prog_link_fd, F_SETFD, 0 ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,0) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     434           0 :           }
     435           0 :         }
     436           0 :       }
     437             : 
     438           0 :       if( need_mlx5 ) {
     439           0 :         int const fd_flags = strcmp( tile->name, "mlx5" ) ? FD_CLOEXEC : 0;
     440           0 :         if( FD_UNLIKELY( -1==fcntl( mlx5_fds.cmd_fd, F_SETFD, fd_flags ) ) ) {
     441           0 :           FD_LOG_ERR(( "fcntl(F_SETFD) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     442           0 :         }
     443           0 :         if( FD_UNLIKELY( -1==fcntl( mlx5_fds.async_fd, F_SETFD, fd_flags ) ) ) {
     444           0 :           FD_LOG_ERR(( "fcntl(F_SETFD) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     445           0 :         }
     446           0 :       }
     447             : 
     448           0 :       if( FD_LIKELY( config->is_firedancer ) ) {
     449           0 :         int tile_uses_accdb    = 0;
     450           0 :         int tile_uses_accdb_ro = 0;
     451           0 :         for( ulong i=0UL; i<tile->uses_obj_cnt; i++ ) {
     452           0 :           fd_topo_obj_t const * obj = &config->topo.objs[ tile->uses_obj_id[ i ] ];
     453           0 :           if( FD_UNLIKELY( !strcmp( obj->name, "accdb" ) ) ) {
     454           0 :             if( FD_UNLIKELY( tile->uses_obj_mode[ i ]==FD_SHMEM_JOIN_MODE_READ_ONLY ) ) tile_uses_accdb_ro = 1;
     455           0 :             else                                                                        tile_uses_accdb    = 1;
     456           0 :             break;
     457           0 :           }
     458           0 :         }
     459             : 
     460             :         /* The gui joins the accdb shmem read-only (for partition stats)
     461             :            but never reads account data from the on-disk file, so it does
     462             :            not need the accounts.db fd.  Withhold it to keep the gui at
     463             :            least privilege. */
     464           0 :         if( FD_UNLIKELY( !strcmp( tile->name, "gui" ) ) ) tile_uses_accdb_ro = 0;
     465           0 :         if( FD_UNLIKELY( !strcmp( tile->name, "snapmk" ) ) ) tile_uses_accdb = tile_uses_accdb_ro = 0;
     466             : 
     467             :         /* snapwr writes accdb pwrite()s without joining accdb shmem, so
     468             :            it needs the RW fd despite not appearing as an accdb obj user
     469             :            in the topology. */
     470           0 :         if( FD_UNLIKELY( tile_uses_accdb || !strcmp( tile->name, "snapwr" ) ) ) {
     471           0 :           if( FD_UNLIKELY( -1==fcntl( FD_ACCDB_FD_RW, F_SETFD, 0 ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,0) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     472           0 :         } else {
     473           0 :           if( FD_UNLIKELY( -1==fcntl( FD_ACCDB_FD_RW, F_SETFD, FD_CLOEXEC ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,FD_CLOEXEC) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     474           0 :         }
     475             : 
     476           0 :         if( FD_UNLIKELY( tile_uses_accdb_ro ) ) {
     477           0 :           if( FD_UNLIKELY( -1==fcntl( FD_ACCDB_FD_RO, F_SETFD, 0 ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,0) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     478           0 :         } else {
     479           0 :           if( FD_UNLIKELY( -1==fcntl( FD_ACCDB_FD_RO, F_SETFD, FD_CLOEXEC ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD,FD_CLOEXEC) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     480           0 :         }
     481             : 
     482           0 :         if( FD_LIKELY( has_store ) ) {
     483           0 :           int tile_uses_store = 0;
     484           0 :           for( ulong i=0UL; i<tile->uses_obj_cnt; i++ ) tile_uses_store |= tile->uses_obj_id[ i ]==store_obj_id;
     485           0 :           int tile_uses_store_rw = tile_uses_store &&
     486           0 :                                    (!strcmp( tile->name, "shred" ) ||
     487           0 :                                     !strcmp( tile->name, "backt" ) ||
     488           0 :                                     !strcmp( tile->name, "rserve" ));
     489           0 :           int tile_uses_store_ro = tile_uses_store && !strcmp( tile->name, "replay" );
     490           0 :           if( FD_UNLIKELY( fcntl( FD_STORE_FD_RW, F_SETFD, tile_uses_store_rw ? 0 : FD_CLOEXEC )<0 ) )
     491           0 :             FD_LOG_ERR(( "fcntl(FD_STORE_FD_RW,F_SETFD) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     492           0 :           if( FD_UNLIKELY( fcntl( FD_STORE_FD_RO, F_SETFD, tile_uses_store_ro ? 0 : FD_CLOEXEC )<0 ) )
     493           0 :             FD_LOG_ERR(( "fcntl(FD_STORE_FD_RO,F_SETFD) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     494           0 :         }
     495             : 
     496           0 :         int tile_uses_snap_fd     = !strcmp( tile->name, "snapct" ) ||
     497           0 :                                     !strcmp( tile->name, "snapmk" );
     498           0 :         int tile_uses_snap_dio_fd = !strcmp( tile->name, "snapzp" );
     499           0 :         int tile_uses_snap_rd_fd  = !strcmp( tile->name, "snapsv" );
     500           0 :         for( ulong j=0UL; j<snap_max; j++ ) {
     501           0 :           if( FD_UNLIKELY( -1==fcntl( FD_SNAP_FD( j ), F_SETFD, tile_uses_snap_fd ? 0 : FD_CLOEXEC ) ) )
     502           0 :             FD_LOG_ERR(( "fcntl(F_SETFD) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     503           0 :           if( snapshot_dio_enabled ) {
     504           0 :             if( FD_UNLIKELY( -1==fcntl( FD_SNAP_DIO_FD( j ), F_SETFD, tile_uses_snap_dio_fd ? 0 : FD_CLOEXEC ) ) )
     505           0 :               FD_LOG_ERR(( "fcntl(F_SETFD) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     506           0 :           }
     507           0 :           if( snapshot_upload_enabled ) {
     508           0 :             if( FD_UNLIKELY( -1==fcntl( FD_SNAP_RO_FD( j ), F_SETFD, tile_uses_snap_rd_fd ? 0 : FD_CLOEXEC ) ) )
     509           0 :               FD_LOG_ERR(( "fcntl(F_SETFD) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     510           0 :           }
     511           0 :         }
     512           0 :       }
     513             : 
     514           0 :       int is_waker = !strcmp( tile->name, "waker" );
     515           0 :       int outer_entitled = is_waker || tile->waker_client_idx!=ULONG_MAX;
     516           0 :       if( FD_UNLIKELY( -1==fcntl( FD_WAKER_OUTER_FD, F_SETFD, outer_entitled ? 0 : FD_CLOEXEC ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     517           0 :       for( ulong j=0UL; j<waker_client_cnt; j++ ) {
     518           0 :         int inner_entitled = is_waker || tile->waker_client_idx==j;
     519           0 :         if( FD_UNLIKELY( -1==fcntl( FD_WAKER_INNER_FD( j ), F_SETFD, inner_entitled ? 0 : FD_CLOEXEC ) ) ) FD_LOG_ERR(( "fcntl(F_SETFD) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     520           0 :       }
     521             : 
     522           0 :       int pipefd[ 2 ];
     523           0 :       if( FD_UNLIKELY( pipe2( pipefd, O_CLOEXEC ) ) ) FD_LOG_ERR(( "pipe2() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     524           0 :       fds[ child_cnt ] = (struct pollfd){ .fd = pipefd[ 0 ], .events = 0 };
     525             : 
     526           0 :       int floating_priority = ( any_floats && !strcmp( tile->name, "waker" ) ) ? -19 : save_priority;
     527           0 :       child_pids[ child_cnt ] = execve_tile( config->name, tile, float_cpu_set, floating_cpu_set, floating_priority, config_memfd, pipefd[ 1 ], &spawn_cg );
     528           0 :       child_idxs[ child_cnt ] = i;
     529           0 :       if( FD_UNLIKELY( close( pipefd[ 1 ] ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     530           0 :       strncpy( child_names[ child_cnt ], tile->name, 32 );
     531           0 :       child_cnt++;
     532           0 :     }
     533           0 :   }
     534             : 
     535           0 :   leave_isolation_cgroup( &spawn_cg );
     536             : 
     537             :   /* Obtain the actual grandchild PID from the pipe */
     538           0 :   for( ulong i=0UL; i<child_cnt; i++ ) {
     539           0 :     if( FD_UNLIKELY( actual_pids[ i ]!=ULONG_MAX ) ) continue;
     540           0 :     FD_TEST( 8UL==read( fds[ i ].fd, &actual_pids[ i ], 8UL ) );
     541           0 :   }
     542             : 
     543           0 :   if( FD_UNLIKELY( -1==setpriority( PRIO_PROCESS, 0, save_priority ) ) ) FD_LOG_ERR(( "setpriority() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     544           0 :   if( FD_UNLIKELY( fd_cpuset_setaffinity( 0, floating_cpu_set ) ) )
     545           0 :     FD_LOG_ERR(( "fd_cpuset_setaffinity failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     546             : 
     547           0 :   if( FD_UNLIKELY( close( config_memfd ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     548           0 :   if( need_xdp ) {
     549           0 :     for( uint i=0U; i<xdp_fds_cnt; i++ ) {
     550           0 :       if( FD_UNLIKELY( close( xdp_fds[i].xsk_map_fd ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     551           0 :       if( FD_UNLIKELY( close( xdp_fds[i].prog_link_fd ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     552           0 :     }
     553           0 :   }
     554             : 
     555           0 :   if( FD_LIKELY( config->is_firedancer ) ) {
     556           0 :     if( FD_UNLIKELY( -1==close( FD_ACCDB_FD_RW ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     557           0 :     if( FD_UNLIKELY( -1==close( FD_ACCDB_FD_RO ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     558           0 :     if( FD_LIKELY( has_store ) ) {
     559           0 :       if( FD_UNLIKELY( -1==close( FD_STORE_FD_RW ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     560           0 :       if( FD_UNLIKELY( -1==close( FD_STORE_FD_RO ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     561           0 :     }
     562           0 :     for( ulong j=0UL; j<snap_max; j++ ) {
     563           0 :       if( FD_UNLIKELY( -1==close( FD_SNAP_FD( j ) ) ) )     FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     564           0 :       if( snapshot_dio_enabled )
     565           0 :         if( FD_UNLIKELY( -1==close( FD_SNAP_DIO_FD( j ) ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     566           0 :       if( snapshot_upload_enabled )
     567           0 :         if( FD_UNLIKELY( -1==close( FD_SNAP_RO_FD( j ) ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     568           0 :     }
     569           0 :   }
     570             : 
     571           0 :   if( FD_UNLIKELY( -1==close( FD_WAKER_OUTER_FD ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     572           0 :   for( ulong j=0UL; j<waker_client_cnt; j++ ) {
     573           0 :     if( FD_UNLIKELY( -1==close( FD_WAKER_INNER_FD( j ) ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     574           0 :   }
     575             : 
     576           0 :   int allow_fds[ 6+FD_TOPO_MAX_TILES ];
     577           0 :   ulong allow_fds_cnt = 0;
     578           0 :   allow_fds[ allow_fds_cnt++ ] = 2; /* stderr */
     579           0 :   if( FD_LIKELY( fd_log_private_logfile_fd()!=-1 ) )
     580           0 :     allow_fds[ allow_fds_cnt++ ] = fd_log_private_logfile_fd(); /* logfile */
     581           0 :   allow_fds[ allow_fds_cnt++ ] = args->pipefd[ 1 ]; /* write end of main pipe */
     582           0 :   for( ulong i=0UL; i<child_cnt; i++ )
     583           0 :     allow_fds[ allow_fds_cnt++ ] = fds[ i ].fd; /* read end of child pipes */
     584           0 :   if( need_mlx5 ) {
     585           0 :     allow_fds[ allow_fds_cnt++ ] = mlx5_fds.cmd_fd;
     586           0 :     allow_fds[ allow_fds_cnt++ ] = mlx5_fds.async_fd;
     587           0 :   }
     588             : 
     589           0 :   struct sock_filter seccomp_filter[ 128UL ];
     590           0 :   unsigned int instr_cnt;
     591             :   #if defined(__aarch64__)
     592             :   populate_sock_filter_policy_pidns_arm64( 128UL, seccomp_filter, (uint)fd_log_private_logfile_fd() );
     593             :   instr_cnt = sock_filter_policy_pidns_arm64_instr_cnt;
     594             :   #else
     595           0 :   populate_sock_filter_policy_pidns( 128UL, seccomp_filter, (uint)fd_log_private_logfile_fd() );
     596           0 :   instr_cnt = sock_filter_policy_pidns_instr_cnt;
     597           0 :   #endif
     598             : 
     599           0 :   if( FD_LIKELY( config->development.sandbox ) ) {
     600           0 :     fd_sandbox_enter( config->uid,
     601           0 :                       config->gid,
     602           0 :                       0,
     603           0 :                       0,
     604           0 :                       0,
     605           0 :                       0,
     606           0 :                       0,
     607           0 :                       1UL+child_cnt, /* RLIMIT_NOFILE needs to be set to the nfds argument of poll() */
     608           0 :                       0UL,
     609           0 :                       0UL,
     610           0 :                       0UL,
     611           0 :                       allow_fds_cnt,
     612           0 :                       allow_fds,
     613           0 :                       instr_cnt,
     614           0 :                       seccomp_filter );
     615           0 :   } else {
     616           0 :     fd_sandbox_switch_uid_gid( config->uid, config->gid );
     617           0 :   }
     618             : 
     619             :   /* Reap child process PIDs so they don't show up in `ps` etc.  All of
     620             :      these children should have exited immediately after clone(2)'ing
     621             :      another child with a huge page based stack. */
     622           0 :   for( ulong i=0UL; i<child_cnt; i++ ) {
     623           0 :     int wstatus;
     624           0 :     int exited_pid = wait4( child_pids[ i ], &wstatus, (int)__WALL, NULL );
     625           0 :     if( FD_UNLIKELY( -1==exited_pid ) ) {
     626           0 :       FD_LOG_ERR(( "pidns wait4() failed (%i-%s) %lu %hu", errno, fd_io_strerror( errno ), i, fds[i].revents ));
     627           0 :     } else if( FD_UNLIKELY( child_pids[ i ]!=exited_pid ) ) {
     628           0 :       FD_LOG_ERR(( "pidns wait4() returned unexpected pid %d %d", child_pids[ i ], exited_pid ));
     629           0 :     } else if( FD_UNLIKELY( !WIFEXITED( wstatus ) ) ) {
     630           0 :       FD_LOG_ERR_NOEXIT(( "tile %lu (%s) exited while booting with signal %d (%s)", i, child_names[ i ], WTERMSIG( wstatus ), fd_io_strsignal( WTERMSIG( wstatus ) ) ));
     631           0 :       fd_sys_util_exit_group( WTERMSIG( wstatus ) ? WTERMSIG( wstatus ) : 1 );
     632           0 :     }
     633           0 :     if( FD_UNLIKELY( WEXITSTATUS( wstatus ) ) ) {
     634           0 :       FD_LOG_ERR_NOEXIT(( "tile %lu (%s) exited while booting with code %d", i, child_names[ i ], WEXITSTATUS( wstatus ) ));
     635           0 :       fd_sys_util_exit_group( WEXITSTATUS( wstatus ) ? WEXITSTATUS( wstatus ) : 1 );
     636           0 :     }
     637           0 :   }
     638             : 
     639           0 :   fds[ child_cnt ] = (struct pollfd){ .fd = args->pipefd[ 1 ], .events = 0 };
     640           0 :   strncpy( child_names[ child_cnt ], "parent", 32UL );
     641           0 :   child_idxs[ child_cnt ] = ULONG_MAX;
     642             : 
     643             :   /* We are now the init process of the pid namespace.  If the init
     644             :      process dies, all children are terminated.  If any child dies, we
     645             :      terminate the init process, which will cause the kernel to
     646             :      terminate all other children bringing all of our processes down as
     647             :      a group.  The parent process will also die if this process dies,
     648             :      due to getting SIGHUP on the pipe. */
     649           0 :   while( 1 ) {
     650           0 :     if( FD_UNLIKELY( -1==poll( fds, 1UL+child_cnt, (int)-1 ) ) ) FD_LOG_ERR(( "poll() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     651             : 
     652             :     /* Parent process died, probably SIGINT, exit gracefully. */
     653           0 :     if( FD_UNLIKELY( fds[ child_cnt ].revents ) ) fd_sys_util_exit_group( 0 );
     654             : 
     655             :     /* Child process died, reap it to figure out exit code. */
     656           0 :     int wstatus;
     657           0 :     int exited_pid = wait4( -1, &wstatus, (int)__WALL | (int)WNOHANG, NULL );
     658           0 :     if( FD_UNLIKELY( -1==exited_pid ) ) {
     659           0 :       FD_LOG_ERR(( "pidns wait4() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     660           0 :     } else if( FD_UNLIKELY( !exited_pid ) ) {
     661             :       /* Spurious wakeup, no child actually dead yet. */
     662           0 :       continue;
     663           0 :     }
     664             : 
     665             :     /* Now find the tile corresponding to that PID */
     666           0 :     FD_TEST( exited_pid>0 );
     667           0 :     int found = 0;
     668           0 :     for( ulong i=0UL; i<child_cnt; i++ ) {
     669           0 :       if( FD_LIKELY( actual_pids[ i ]!=(ulong)exited_pid ) ) continue;
     670             : 
     671           0 :       found = 1;
     672           0 :       fds[ i ].fd = -1; /* Don't poll on this tile anymore */
     673             : 
     674           0 :       char * tile_name = child_names[ i ];
     675           0 :       ulong  tile_idx = child_idxs[ i ];
     676           0 :       ulong  tile_id = config->topo.tiles[ tile_idx ].kind_id;
     677             : 
     678           0 :       if( FD_UNLIKELY( !WIFEXITED( wstatus ) ) ) {
     679           0 :         FD_LOG_ERR_NOEXIT(( "tile %s%s:%lu%s exited with signal %d %s(%s)%s", fd_log_style_bold(), tile_name, tile_id, fd_log_style_normal(), WTERMSIG( wstatus ), fd_log_style_dim(), fd_io_strsignal( WTERMSIG( wstatus ) ), fd_log_style_normal() ));
     680           0 :         fd_sys_util_exit_group( WTERMSIG( wstatus ) ? WTERMSIG( wstatus ) : 1 );
     681           0 :       } else {
     682           0 :         int exit_code = WEXITSTATUS( wstatus );
     683           0 :         if( FD_LIKELY( !exit_code && tile_idx!=ULONG_MAX && config->topo.tiles[ tile_idx ].allow_shutdown ) ) {
     684           0 :           found = 1;
     685           0 :           FD_LOG_INFO(( "tile %s:%lu exited gracefully with code %d", tile_name, tile_id, exit_code ));
     686           0 :         } else {
     687           0 :           FD_LOG_ERR_NOEXIT(( "tile %s%s:%lu%s exited with code %d", fd_log_style_bold(), tile_name, tile_id, fd_log_style_normal(), exit_code ));
     688           0 :           fd_sys_util_exit_group( exit_code ? exit_code : 1 );
     689           0 :         }
     690           0 :       }
     691           0 :     }
     692             : 
     693           0 :     if( FD_UNLIKELY( !found ) ) FD_LOG_ERR(( "wait4() returned unexpected pid %d", exited_pid ));
     694           0 :   }
     695             : 
     696           0 :   return 0;
     697           0 : }
     698             : 
     699             : int
     700             : clone_firedancer( config_t const * config,
     701             :                   int              close_fd,
     702           0 :                   int *            out_pipe ) {
     703             :   /* This pipe is here just so that the child process knows when the
     704             :      parent has died (it will get a HUP). */
     705           0 :   int pipefd[2];
     706           0 :   if( FD_UNLIKELY( pipe2( pipefd, O_CLOEXEC | O_NONBLOCK ) ) ) FD_LOG_ERR(( "pipe2() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     707             : 
     708             :   /* clone into a pid namespace */
     709           0 :   int flags = config->development.sandbox ? CLONE_NEWPID : 0;
     710           0 :   struct pidns_clone_args args = { .config = config, .closefd = close_fd, .pipefd = pipefd, };
     711             : 
     712           0 :   void * stack = create_clone_stack();
     713             : 
     714           0 :   int pid_namespace = clone( main_pid_namespace, (uchar *)stack + FD_TILE_PRIVATE_STACK_SZ, flags, &args );
     715           0 :   if( FD_UNLIKELY( pid_namespace<0 ) ) FD_LOG_ERR(( "clone() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     716             : 
     717           0 :   if( FD_UNLIKELY( close( pipefd[ 1 ] ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     718             : 
     719           0 :   *out_pipe = pipefd[ 0 ];
     720           0 :   return pid_namespace;
     721           0 : }
     722             : 
     723             : static void
     724             : workspace_path( config_t const *       config,
     725             :                 fd_topo_wksp_t const * wksp,
     726           0 :                 char                   out[ PATH_MAX ] ) {
     727           0 :   char const * mount_path;
     728           0 :   switch( wksp->page_sz ) {
     729           0 :     case FD_SHMEM_HUGE_PAGE_SZ:
     730           0 :       mount_path = config->hugetlbfs.huge_page_mount_path;
     731           0 :       break;
     732           0 :     case FD_SHMEM_GIGANTIC_PAGE_SZ:
     733           0 :       mount_path = config->hugetlbfs.gigantic_page_mount_path;
     734           0 :       break;
     735           0 :     case FD_SHMEM_NORMAL_PAGE_SZ:
     736           0 :       mount_path = config->hugetlbfs.normal_page_mount_path;
     737           0 :       break;
     738           0 :     default:
     739           0 :       FD_LOG_ERR(( "invalid page size %lu", wksp->page_sz ));
     740           0 :   }
     741             : 
     742           0 :   FD_TEST( fd_cstr_printf_check( out, PATH_MAX, NULL, "%s/%s_%s.wksp", mount_path, config->name, wksp->name ) );
     743           0 : }
     744             : 
     745             : static void
     746             : warn_unknown_files( config_t const * config,
     747           0 :                     ulong            mount_type ) {
     748           0 :   char const * mount_path;
     749           0 :   switch( mount_type ) {
     750           0 :     case 0UL:
     751           0 :       mount_path = config->hugetlbfs.huge_page_mount_path;
     752           0 :       break;
     753           0 :     case 1UL:
     754           0 :       mount_path = config->hugetlbfs.gigantic_page_mount_path;
     755           0 :       break;
     756           0 :     default:
     757           0 :       FD_LOG_ERR(( "invalid mount type %lu", mount_type ));
     758           0 :   }
     759             : 
     760             :   /* Check if there are any files in mount_path */
     761           0 :   DIR * dir = opendir( mount_path );
     762           0 :   if( FD_UNLIKELY( !dir ) ) {
     763           0 :     if( FD_UNLIKELY( errno!=ENOENT ) ) FD_LOG_ERR(( "error opening `%s` (%i-%s)", mount_path, errno, fd_io_strerror( errno ) ));
     764           0 :     return;
     765           0 :   }
     766             : 
     767           0 :   struct dirent * entry;
     768           0 :   for(;;) {
     769           0 :     errno = 0;
     770           0 :     entry = readdir( dir );
     771           0 :     if( FD_UNLIKELY( !entry ) ) break;
     772           0 :     if( FD_UNLIKELY( !strcmp( entry->d_name, ".") || !strcmp( entry->d_name, ".." ) ) ) continue;
     773             : 
     774           0 :     char entry_path[ PATH_MAX ];
     775           0 :     FD_TEST( fd_cstr_printf_check( entry_path, PATH_MAX, NULL, "%s/%s", mount_path, entry->d_name ));
     776             : 
     777           0 :     int known_file = 0;
     778           0 :     for( ulong i=0UL; i<config->topo.wksp_cnt; i++ ) {
     779           0 :       fd_topo_wksp_t const * wksp = &config->topo.workspaces[ i ];
     780             : 
     781           0 :       char expected_path[ PATH_MAX ];
     782           0 :       workspace_path( config, wksp, expected_path );
     783             : 
     784           0 :       if( !strcmp( entry_path, expected_path ) ) {
     785           0 :         known_file = 1;
     786           0 :         break;
     787           0 :       }
     788           0 :     }
     789             : 
     790           0 :     if( mount_type==0UL ) {
     791           0 :       for( ulong i=0UL; i<config->topo.tile_cnt; i++ ) {
     792           0 :         fd_topo_tile_t const * tile = &config->topo.tiles [ i ];
     793             : 
     794           0 :         char expected_path[ PATH_MAX ];
     795           0 :         FD_TEST( fd_cstr_printf_check( expected_path, PATH_MAX, NULL, "%s/%s_stack_%s%lu", config->hugetlbfs.huge_page_mount_path, config->name, tile->name, tile->kind_id ) );
     796             : 
     797           0 :         if( !strcmp( entry_path, expected_path ) ) {
     798           0 :           known_file = 1;
     799           0 :           break;
     800           0 :         }
     801           0 :       }
     802           0 :     }
     803             : 
     804           0 :     if( FD_UNLIKELY( !known_file ) ) FD_LOG_WARNING(( "unknown file `%s` found in `%s`", entry->d_name, mount_path ));
     805           0 :   }
     806             : 
     807           0 :   if( FD_UNLIKELY( errno ) ) FD_LOG_ERR(( "error reading dir `%s` (%i-%s)", mount_path, errno, fd_io_strerror( errno ) ));
     808           0 :   if( FD_UNLIKELY( closedir( dir ) ) ) FD_LOG_ERR(( "error closing `%s` (%i-%s)", mount_path, errno, fd_io_strerror( errno ) ));
     809           0 : }
     810             : 
     811             : void
     812           0 : initialize_workspaces( config_t * config ) {
     813             :   /* Switch to non-root uid/gid for workspace creation.  Permissions
     814             :      checks are still done as the current user. */
     815           0 :   uint gid = getgid();
     816           0 :   uint uid = getuid();
     817           0 :   if( FD_LIKELY( gid!=config->gid && -1==setegid( config->gid ) ) )
     818           0 :     FD_LOG_ERR(( "setegid() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     819           0 :   if( FD_LIKELY( uid!=config->uid && -1==seteuid( config->uid ) ) )
     820           0 :     FD_LOG_ERR(( "seteuid() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     821             : 
     822           0 :   for( ulong i=0UL; i<config->topo.wksp_cnt; i++ ) {
     823           0 :     fd_topo_wksp_t * wksp = &config->topo.workspaces[ i ];
     824             : 
     825           0 :     char path[ PATH_MAX ];
     826           0 :     workspace_path( config, wksp, path );
     827             : 
     828           0 :     struct stat st;
     829           0 :     int result = stat( path, &st );
     830             : 
     831           0 :     int update_existing;
     832           0 :     if( FD_UNLIKELY( !result && config->is_live_cluster && !config->is_dev ) ) {
     833           0 :       if( FD_UNLIKELY( -1==unlink( path ) && errno!=ENOENT ) ) FD_LOG_ERR(( "unlink() failed when trying to create workspace `%s` (%i-%s)", path, errno, fd_io_strerror( errno ) ));
     834           0 :       update_existing = 0;
     835           0 :     } else if( FD_UNLIKELY( !result ) ) {
     836             :       /* Creating all of the workspaces is very expensive because the
     837             :          kernel has to zero out all of the pages.  There can be tens or
     838             :          hundreds of gigabytes of zeroing to do.
     839             : 
     840             :          What would be really nice is if the kernel let us create huge
     841             :          pages without zeroing them, but it's not possible.  The
     842             :          ftruncate and fallocate calls do not support this type of
     843             :          resize with the hugetlbfs filesystem.
     844             : 
     845             :          Instead.. to prevent repeatedly doing this zeroing every time
     846             :          we start the validator, we have a small hack here to re-use the
     847             :          workspace files if they exist. */
     848           0 :       update_existing = 1;
     849           0 :     } else if( FD_LIKELY( result && errno==ENOENT ) ) {
     850           0 :       update_existing = 0;
     851           0 :     } else {
     852           0 :       FD_LOG_ERR(( "stat failed when trying to create workspace `%s` (%i-%s)", path, errno, fd_io_strerror( errno ) ));
     853           0 :     }
     854             : 
     855           0 :     if( FD_UNLIKELY( -1==fd_topo_create_workspace( &config->topo, wksp, update_existing ) ) ) {
     856           0 :       FD_TEST( errno==ENOMEM );
     857             : 
     858           0 :       warn_unknown_files( config, wksp->page_sz!=FD_SHMEM_HUGE_PAGE_SZ );
     859             : 
     860           0 :       char path[ PATH_MAX ];
     861           0 :       workspace_path( config, wksp, path );
     862           0 :       FD_LOG_ERR(( "ENOMEM-Out of memory when trying to create workspace `%s` at `%s` "
     863           0 :                    "with %lu %s pages. Firedancer reserves enough memory for all of its workspaces "
     864           0 :                    "during the `hugetlbfs` configure step, so it is likely you have unknown files "
     865           0 :                    "left over in this directory which are consuming memory, or another program on "
     866           0 :                    "the system is using pages from the same mount.",
     867           0 :                    wksp->name, path, wksp->page_cnt, fd_shmem_page_sz_to_cstr( wksp->page_sz ) ));
     868           0 :     }
     869           0 :     fd_topo_join_workspace( &config->topo, wksp, FD_SHMEM_JOIN_MODE_READ_WRITE, 0 );
     870           0 :     fd_topo_wksp_new( &config->topo, wksp, CALLBACKS );
     871           0 :     fd_topo_leave_workspace( &config->topo, wksp );
     872           0 :   }
     873             : 
     874           0 :   if( FD_UNLIKELY( seteuid( uid ) ) ) FD_LOG_ERR(( "seteuid() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     875           0 :   if( FD_UNLIKELY( setegid( gid ) ) ) FD_LOG_ERR(( "setegid() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     876           0 : }
     877             : 
     878             : void
     879           0 : initialize_stacks( config_t const * config ) {
     880             : # if FD_HAS_MSAN
     881             :   /* MSan calls an external symbolizer using fork() on crashes, which is
     882             :      incompatible with Firedancer's MAP_SHARED stacks. */
     883             :   (void)config;
     884             :   return;
     885             : # endif
     886             : 
     887             :   /* Switch to non-root uid/gid for workspace creation.  Permissions
     888             :      checks are still done as the current user. */
     889           0 :   uint gid = getgid();
     890           0 :   uint uid = getuid();
     891           0 :   if( FD_LIKELY( gid!=config->gid && -1==setegid( config->gid ) ) )
     892           0 :     FD_LOG_ERR(( "setegid() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     893           0 :   if( FD_LIKELY( uid!=config->uid && -1==seteuid( config->uid ) ) )
     894           0 :     FD_LOG_ERR(( "seteuid() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     895             : 
     896           0 :   for( ulong i=0UL; i<config->topo.tile_cnt; i++ ) {
     897           0 :     fd_topo_tile_t const * tile = &config->topo.tiles[ i ];
     898             : 
     899           0 :     char path[ PATH_MAX ];
     900           0 :     FD_TEST( fd_cstr_printf_check( path, PATH_MAX, NULL, "%s/%s_stack_%s%lu", config->hugetlbfs.huge_page_mount_path, config->name, tile->name, tile->kind_id ) );
     901             : 
     902           0 :     struct stat st;
     903           0 :     int result = stat( path, &st );
     904             : 
     905           0 :     int update_existing;
     906           0 :     if( FD_UNLIKELY( !result && config->is_live_cluster ) ) {
     907           0 :       if( FD_UNLIKELY( -1==unlink( path ) && errno!=ENOENT ) ) FD_LOG_ERR(( "unlink() failed when trying to create stack workspace `%s` (%i-%s)", path, errno, fd_io_strerror( errno ) ));
     908           0 :       update_existing = 0;
     909           0 :     } else if( FD_UNLIKELY( !result ) ) {
     910             :       /* See above note about zeroing out pages. */
     911           0 :       update_existing = 1;
     912           0 :     } else if( FD_LIKELY( result && errno==ENOENT ) ) {
     913           0 :       update_existing = 0;
     914           0 :     } else {
     915           0 :       FD_LOG_ERR(( "stat failed when trying to create workspace `%s` (%i-%s)", path, errno, fd_io_strerror( errno ) ));
     916           0 :     }
     917             : 
     918             :     /* TODO: Use a better CPU idx for the stack if tile is floating */
     919           0 :     ulong stack_cpu_idx = 0UL;
     920           0 :     if( FD_LIKELY( tile->cpu_idx<65535UL ) ) stack_cpu_idx = tile->cpu_idx;
     921             : 
     922           0 :     char name[ PATH_MAX ];
     923           0 :     FD_TEST( fd_cstr_printf_check( name, PATH_MAX, NULL, "%s_stack_%s%lu", config->name, tile->name, tile->kind_id ) );
     924             : 
     925           0 :     ulong sub_page_cnt[ 1 ] = { 6 };
     926           0 :     ulong sub_cpu_idx [ 1 ] = { stack_cpu_idx };
     927           0 :     int err;
     928           0 :     if( FD_UNLIKELY( update_existing ) ) {
     929           0 :       err = fd_shmem_update_multi( name, FD_SHMEM_HUGE_PAGE_SZ, 1, sub_page_cnt, sub_cpu_idx, S_IRUSR | S_IWUSR ); /* logs details */
     930           0 :     } else {
     931           0 :       err = fd_shmem_create_multi( name, FD_SHMEM_HUGE_PAGE_SZ, 1, sub_page_cnt, sub_cpu_idx, S_IRUSR | S_IWUSR ); /* logs details */
     932           0 :     }
     933           0 :     if( FD_UNLIKELY( err && errno==ENOMEM ) ) {
     934           0 :       warn_unknown_files( config, 0UL );
     935             : 
     936           0 :       char path[ PATH_MAX ];
     937           0 :       FD_TEST( fd_cstr_printf_check( path, PATH_MAX, NULL, "%s/%s_stack_%s%lu", config->hugetlbfs.huge_page_mount_path, config->name, tile->name, tile->kind_id ) );
     938           0 :       FD_LOG_ERR(( "ENOMEM-Out of memory when trying to create huge page stack for tile `%s` at `%s`. "
     939           0 :                    "Firedancer reserves enough memory for all of its stacks during the `hugetlbfs` configure "
     940           0 :                    "step, so it is likely you have unknown files left over in this directory which are "
     941           0 :                    "consuming memory, or another program on the system is using pages from the same mount.",
     942           0 :                    tile->name, path ));
     943           0 :     } else if( FD_UNLIKELY( err ) ) FD_LOG_ERR(( "fd_shmem_create_multi failed" ));
     944           0 :   }
     945             : 
     946           0 :   if( FD_UNLIKELY( seteuid( uid ) ) ) FD_LOG_ERR(( "seteuid() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     947           0 :   if( FD_UNLIKELY( setegid( gid ) ) ) FD_LOG_ERR(( "setegid() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
     948           0 : }
     949             : 
     950             : void
     951           0 : fdctl_check_configure( config_t const * config ) {
     952           0 :   configure_result_t check = fd_cfg_stage_hugetlbfs.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
     953           0 :   if( FD_UNLIKELY( check.result!=CONFIGURE_OK ) )
     954           0 :     FD_LOG_ERR(( "Huge pages are not configured correctly: %s. You can run `%s configure init hugetlbfs` "
     955           0 :                  "to create the mounts correctly. This must be done after every system restart before running "
     956           0 :                  "Firedancer.", check.message, FD_BINARY_NAME ));
     957             : 
     958           0 :   if( FD_LIKELY( 0==strcmp( config->net.provider, "xdp" ) ) ) {
     959           0 :     if( fd_cfg_stage_bonding.enabled( config ) ) {
     960           0 :       check = fd_cfg_stage_bonding.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
     961           0 :       if( FD_UNLIKELY( check.result!=CONFIGURE_OK ) )
     962           0 :         FD_LOG_ERR(( "Bonded network device is not configured correctly: %s. You can run `%s configure init bonding` "
     963           0 :                     "to configure the bonding driver.", check.message, FD_BINARY_NAME ));
     964           0 :     }
     965             : 
     966           0 :     check = fd_cfg_stage_ethtool_channels.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
     967           0 :     if( FD_UNLIKELY( check.result!=CONFIGURE_OK ) )
     968           0 :       FD_LOG_ERR(( "Network %s. You can run `%s configure init ethtool-channels` to set the number of channels on the "
     969           0 :                   "network device correctly.", check.message, FD_BINARY_NAME ));
     970             : 
     971           0 :     check = fd_cfg_stage_ethtool_offloads.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
     972           0 :     if( FD_UNLIKELY( check.result!=CONFIGURE_OK ) )
     973           0 :       FD_LOG_ERR(( "Network %s. You can run `%s configure init ethtool-offloads` to disable features "
     974           0 :                   "as required.", check.message, FD_BINARY_NAME ));
     975             : 
     976           0 :     check = fd_cfg_stage_ethtool_loopback.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
     977           0 :     if( FD_UNLIKELY( check.result!=CONFIGURE_OK ) )
     978           0 :       FD_LOG_ERR(( "Network %s. You can run `%s configure init ethtool-loopback` to disable tx-udp-segmentation "
     979           0 :                   "on the loopback device.", check.message, FD_BINARY_NAME ));
     980           0 :   }
     981             : 
     982           0 :   check = fd_cfg_stage_sysctl.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
     983           0 :   if( FD_UNLIKELY( check.result!=CONFIGURE_OK ) )
     984           0 :     FD_LOG_ERR(( "Kernel parameters are not configured correctly: %s. You can run `%s configure init sysctl` "
     985           0 :                  "to set kernel parameters correctly.", check.message, FD_BINARY_NAME ));
     986             : 
     987             :   /* hyperthreads, nohz-full and rcu-nocbs are check-only stages: they
     988             :      emit warnings themselves and always return OK, so there is no
     989             :      result to act on (and no init to point the operator at). */
     990           0 :   (void)fd_cfg_stage_hyperthreads.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
     991           0 :   (void)fd_cfg_stage_nohz_full.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
     992           0 :   (void)fd_cfg_stage_rcu_nocbs.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
     993             : 
     994             :   /* kworkers and cpuset are optional (but recommended) hardening: an
     995             :      unconfigured stage only warns.  A PARTIALLY_CONFIGURED cpuset is
     996             :      fatal however: the isolation cgroup exists but covers the wrong
     997             :      CPUs (e.g. stale from a previous [layout.affinity]), and the tile
     998             :      launcher would join it and then fail to pin with a confusing
     999             :      EINVAL.  Fail up front with the fix instead. */
    1000           0 :   if( FD_UNLIKELY( fd_cfg_stage_kworkers.enabled( config ) ) ) {
    1001           0 :     check = fd_cfg_stage_kworkers.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
    1002           0 :     if( FD_UNLIKELY( check.result!=CONFIGURE_OK ) )
    1003           0 :       FD_LOG_WARNING(( "Kernel workqueues may steal CPU time from Firedancer tiles: %s. For lower jitter, run "
    1004           0 :                        "`%s configure init kworkers`.", check.message, FD_BINARY_NAME ));
    1005           0 :   }
    1006             : 
    1007             :   /* Floating tiles need a scheduler domain to be balanced across their
    1008             :      CPUs; isolcpus= removes it, and every floater would stay on the
    1009             :      CPU it was forked on. */
    1010           0 :   FD_CPUSET_DECL( isolated );
    1011           0 :   if( FD_LIKELY( fd_cpu_isolation_read_list( "/sys/devices/system/cpu/isolated", isolated ) ) ) {
    1012           0 :     for( ulong i=0UL; i<config->topo.tile_cnt; i++ ) {
    1013           0 :       fd_topo_tile_t const * tile = &config->topo.tiles[ i ];
    1014           0 :       if( FD_UNLIKELY( tile->floats && fd_cpuset_test( isolated, tile->cpu_idx ) ) )
    1015           0 :         FD_LOG_ERR(( "tile %s:%lu floats on CPU %lu, which the isolcpus= boot parameter removed from the kernel scheduler. "
    1016           0 :                      "Floating tiles need their CPUs scheduled: drop them from isolcpus= and isolate them with "
    1017           0 :                      "`%s configure init cpuset` instead.", tile->name, tile->kind_id, tile->cpu_idx, FD_BINARY_NAME ));
    1018           0 :     }
    1019           0 :   }
    1020             : 
    1021           0 :   if( FD_LIKELY( fd_cfg_stage_cpuset.enabled( config ) ) ) {
    1022           0 :     check = fd_cfg_stage_cpuset.check( config, FD_CONFIGURE_CHECK_TYPE_RUN );
    1023           0 :     if( FD_UNLIKELY( check.result==CONFIGURE_PARTIALLY_CONFIGURED ) )
    1024           0 :       FD_LOG_ERR(( "The CPU isolation cgroup exists but does not match the topology: %s. Tiles would fail to pin "
    1025           0 :                    "to their CPUs. Run `%s configure init cpuset` to fix it, or `%s configure fini cpuset` to "
    1026           0 :                    "remove it.", check.message, FD_BINARY_NAME, FD_BINARY_NAME ));
    1027           0 :     else if( FD_UNLIKELY( check.result!=CONFIGURE_OK ) )
    1028           0 :       FD_LOG_WARNING(( "Firedancer tile CPUs are not isolated from other processes: %s. For lower jitter, run "
    1029           0 :                        "`%s configure init cpuset`.", check.message, FD_BINARY_NAME ));
    1030           0 :   }
    1031           0 : }
    1032             : 
    1033             : void
    1034             : run_firedancer_init( config_t * config,
    1035             :                      int        init_workspaces,
    1036           0 :                      int        check_configure ) {
    1037           0 :   struct stat st;
    1038           0 :   int err = stat( config->paths.identity_key, &st );
    1039           0 :   if( FD_UNLIKELY( -1==err && errno==ENOENT ) ) FD_LOG_ERR(( "[consensus.identity_path] key does not exist `%s`. You can generate an identity key at this path by running `%s keys new %s --config <toml>`", config->paths.identity_key, FD_BINARY_NAME, config->paths.identity_key ));
    1040           0 :   else if( FD_UNLIKELY( -1==err ) )             FD_LOG_ERR(( "could not stat [consensus.identity_path] `%s` (%i-%s)", config->paths.identity_key, errno, fd_io_strerror( errno ) ));
    1041             : 
    1042           0 :   if( FD_UNLIKELY( !config->is_firedancer ) ) {
    1043           0 :     for( ulong i=0UL; i<config->frankendancer.paths.authorized_voter_paths_cnt; i++ ) {
    1044           0 :       err = stat( config->frankendancer.paths.authorized_voter_paths[ i ], &st );
    1045           0 :       if( FD_UNLIKELY( -1==err && errno==ENOENT ) ) FD_LOG_ERR(( "[consensus.authorized_voter_paths] key does not exist `%s`", config->frankendancer.paths.authorized_voter_paths[ i ] ));
    1046           0 :       else if( FD_UNLIKELY( -1==err ) )             FD_LOG_ERR(( "could not stat [consensus.authorized_voter_paths] `%s` (%i-%s)", config->frankendancer.paths.authorized_voter_paths[ i ], errno, fd_io_strerror( errno ) ));
    1047           0 :     }
    1048           0 :   }
    1049             : 
    1050             :   /* FIXME: fdctl_check_configure unconditionally checks for network
    1051             :             stack prerequisites even if the command being run does not
    1052             :             require networking.  Hack around that here for now. */
    1053           0 :   if( check_configure ) fdctl_check_configure( config );
    1054           0 :   if( FD_LIKELY( init_workspaces ) ) initialize_workspaces( config );
    1055           0 :   initialize_stacks( config );
    1056           0 :   fd_bootinfo_write( config );
    1057           0 : }
    1058             : 
    1059             : void
    1060           0 : initialize_accdb_fd( config_t const * config ) {
    1061           0 :   if( FD_UNLIKELY( !config->is_firedancer ) ) return;
    1062             : 
    1063             :   /* O_TRUNC of a previous run's accounts.db frees all its extents
    1064             :      synchronously.  In development skip the truncate to keep reboots
    1065             :      fast. */
    1066           0 :   int oflags = O_RDWR|O_CREAT|O_NOATIME;
    1067           0 :   if( FD_LIKELY( !config->is_dev ) ) oflags |= O_TRUNC;
    1068             : 
    1069           0 :   int accounts_fd = open( config->paths.accounts, oflags, S_IRUSR|S_IWUSR );
    1070           0 :   if( FD_UNLIKELY( -1==accounts_fd ) ) FD_LOG_ERR(( "failed to open accounts.db (%i-%s)", errno, fd_io_strerror( errno ) ));
    1071           0 :   if( FD_UNLIKELY( -1==dup2( accounts_fd, FD_ACCDB_FD_RW ) ) ) FD_LOG_ERR(( "dup2() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
    1072           0 :   if( FD_UNLIKELY( -1==close( accounts_fd ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
    1073             : 
    1074             :   /* Read-only fd for tiles (e.g. rpc) that consume accdb but must not
    1075             :      be able to mutate the on-disk file.  Reopen via /proc/self/fd to
    1076             :      guarantee it refers to the same inode as the RW fd, avoiding any
    1077             :      race where the file at the path could be replaced between opens. */
    1078           0 :   char proc_path[ PATH_MAX ];
    1079           0 :   FD_TEST( fd_cstr_printf_check( proc_path, sizeof(proc_path), NULL, "/proc/self/fd/%d", FD_ACCDB_FD_RW ) );
    1080           0 :   int accounts_ro_fd = open( proc_path, O_RDONLY|O_NOATIME );
    1081           0 :   if( FD_UNLIKELY( -1==accounts_ro_fd ) ) FD_LOG_ERR(( "failed to open accounts.db read-only (%i-%s)", errno, fd_io_strerror( errno ) ));
    1082           0 :   if( FD_UNLIKELY( -1==dup2( accounts_ro_fd, FD_ACCDB_FD_RO ) ) ) FD_LOG_ERR(( "dup2() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
    1083           0 :   if( FD_UNLIKELY( -1==close( accounts_ro_fd ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
    1084           0 : }
    1085             : 
    1086             : void
    1087           0 : initialize_store_fds( config_t const * config ) {
    1088           0 :   if( FD_UNLIKELY( !config->is_firedancer ) ) return;
    1089             : 
    1090           0 :   fd_topo_t const * topo = &config->topo;
    1091           0 :   ulong store_obj_id = fd_pod_query_ulong( topo->props, "store", ULONG_MAX );
    1092           0 :   if( FD_UNLIKELY( store_obj_id==ULONG_MAX ) ) return;
    1093             : 
    1094           0 :   char const * path = fd_pod_queryf_cstr( topo->props, NULL, "obj.%lu.disk_path", store_obj_id );
    1095           0 :   ulong fec_max = fd_pod_queryf_ulong( topo->props, 0UL, "obj.%lu.fec_max", store_obj_id );
    1096           0 :   ulong fec_data_max = fd_pod_queryf_ulong( topo->props, 0UL, "obj.%lu.fec_data_max", store_obj_id );
    1097           0 :   ulong shred_storage_gib = fd_pod_queryf_ulong( topo->props, ULONG_MAX, "obj.%lu.shred_storage_gib", store_obj_id );
    1098           0 :   ulong payload_slot_sz = fd_store_payload_slot_sz( fec_data_max );
    1099           0 :   ulong wire_off;
    1100           0 :   if( FD_UNLIKELY( !path || !fec_max || !payload_slot_sz || shred_storage_gib>FD_SHREDB_MAX_SIZE_GIB ||
    1101           0 :                    __builtin_umull_overflow( fec_max, payload_slot_sz, &wire_off ) ) )
    1102           0 :     FD_LOG_ERR(( "invalid Store backing-file configuration" ));
    1103             : 
    1104           0 :   int store_fd = fd_store_file_create( path, wire_off, fd_shredb_max_shreds( shred_storage_gib ) );
    1105           0 :   if( FD_UNLIKELY( store_fd<0 ) )
    1106           0 :     FD_LOG_ERR(( "failed to create Store backing file `%s` (%i-%s)", path, errno, fd_io_strerror( errno ) ));
    1107           0 :   if( FD_LIKELY( store_fd!=FD_STORE_FD_RW ) ) {
    1108           0 :     if( FD_UNLIKELY( dup2( store_fd, FD_STORE_FD_RW )<0 ) ) FD_LOG_ERR(( "dup2(Store RW) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
    1109           0 :     if( FD_UNLIKELY( close( store_fd ) ) ) FD_LOG_ERR(( "close(Store RW source) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
    1110           0 :   }
    1111             : 
    1112             :   /* Reopen through procfs so the read-only descriptor is guaranteed to
    1113             :      name the same inode even if the configured path is replaced. */
    1114           0 :   char proc_path[ PATH_MAX ];
    1115           0 :   FD_TEST( fd_cstr_printf_check( proc_path, sizeof(proc_path), NULL, "/proc/self/fd/%d", FD_STORE_FD_RW ) );
    1116           0 :   int store_ro_fd = open( proc_path, O_RDONLY|O_NOATIME );
    1117           0 :   if( FD_UNLIKELY( store_ro_fd<0 ) )
    1118           0 :     FD_LOG_ERR(( "failed to open Store backing file read-only (%i-%s)", errno, fd_io_strerror( errno ) ));
    1119           0 :   if( FD_LIKELY( store_ro_fd!=FD_STORE_FD_RO ) ) {
    1120           0 :     if( FD_UNLIKELY( dup2( store_ro_fd, FD_STORE_FD_RO )<0 ) ) FD_LOG_ERR(( "dup2(Store RO) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
    1121           0 :     if( FD_UNLIKELY( close( store_ro_fd ) ) ) FD_LOG_ERR(( "close(Store RO source) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
    1122           0 :   }
    1123           0 : }
    1124             : 
    1125             : /* Snapshot production prep
    1126             :    (On startup, reconcile with files in snapshot dirs, and drop old
    1127             :    files.) */
    1128             : 
    1129             : static int
    1130           0 : matches_partial_snap_filename( char const * name ) {
    1131           0 :   if( FD_UNLIKELY( !strcmp( name, ".snapshot.tar.bz2-partial" ) ||
    1132           0 :                    !strcmp( name, ".incremental-snapshot.tar.bz2-partial" ) ) ) return 1;
    1133             : 
    1134           0 :   if( *name=='.' ) name++;
    1135           0 :   char const * p;
    1136           0 :   if( !strncmp( name, "snapshot-x", 10UL ) ) p = name+10UL;
    1137           0 :   else if( !strncmp( name, "snapshot", 8UL ) ) p = name+8UL;
    1138           0 :   else return 0;
    1139           0 :   if( *p<'0' || *p>'9' ) return 0;
    1140           0 :   while( *p>='0' && *p<='9' ) p++;
    1141           0 :   return 0==strcmp( p, ".partial" );
    1142           0 : }
    1143             : 
    1144             : static int
    1145             : snap_inode_desc( void const * _a,
    1146           0 :                  void const * _b ) {
    1147           0 :   fd_backup_inode_t const * a = (fd_backup_inode_t const *)_a;
    1148           0 :   fd_backup_inode_t const * b = (fd_backup_inode_t const *)_b;
    1149           0 :   ulong a_slot = a->incr_slot==ULONG_MAX ? a->full_slot : a->incr_slot;
    1150           0 :   ulong b_slot = b->incr_slot==ULONG_MAX ? b->full_slot : b->incr_slot;
    1151           0 :   if( FD_LIKELY( a_slot!=b_slot ) ) return a_slot>b_slot ? -1 : 1;
    1152           0 :   return 0;
    1153           0 : }
    1154             : 
    1155             : static void
    1156             : drop_old_snaps( int                 snap_dir_fd,
    1157             :                 char const *        snap_dir,
    1158             :                 fd_backup_inode_t * snaps,
    1159             :                 ulong *             snap_cnt_p,
    1160           0 :                 ulong               snap_max ) {
    1161           0 :   ulong snap_cnt = *snap_cnt_p;
    1162           0 :   if( snap_cnt<=snap_max ) return;
    1163             :   /* drop oldest */
    1164           0 :   qsort( snaps, snap_cnt, sizeof(fd_backup_inode_t), snap_inode_desc );
    1165           0 :   for( ulong i=snap_max; i<snap_cnt; i++ ) {
    1166           0 :     int err = unlinkat( snap_dir_fd, snaps[ i ].name, 0 );
    1167           0 :     if( FD_UNLIKELY( err && errno!=ENOENT ) ) {
    1168           0 :       FD_LOG_WARNING(( "unlinkat(%s/%s) failed (%i-%s)", snap_dir, snaps[ i ].name, errno, fd_io_strerror( errno ) ));
    1169           0 :     } else if( FD_LIKELY( !err ) ) {
    1170           0 :       FD_LOG_INFO(( "deleted old snapshot `%s/%s`", snap_dir, snaps[ i ].name ));
    1171           0 :     }
    1172           0 :   }
    1173           0 :   *snap_cnt_p = snap_max;
    1174           0 : }
    1175             : 
    1176             : static void
    1177             : snap_check_fd( char const * snap_dir,
    1178             :                char const * name,
    1179           0 :                int          snap_fd ) {
    1180           0 :   struct stat st;
    1181           0 :   if( FD_UNLIKELY( -1==fstat( snap_fd, &st ) ) )
    1182           0 :     FD_LOG_ERR(( "fstat(%s/%s) failed (%i-%s)", snap_dir, name, errno, fd_io_strerror( errno ) ));
    1183             : 
    1184           0 :   if( FD_UNLIKELY( !S_ISREG( st.st_mode ) ) )
    1185           0 :     FD_LOG_ERR(( "snapshot `%s/%s` is not a regular file", snap_dir, name ));
    1186           0 : }
    1187             : 
    1188             : static void
    1189             : snap_reperm_fd( char const * snap_dir,
    1190             :                 char const * name,
    1191             :                 int          snap_fd,
    1192             :                 uint         uid,
    1193           0 :                 uint         gid ) {
    1194           0 :   if( FD_UNLIKELY( -1==fchown( snap_fd, uid, gid ) ) )
    1195           0 :     FD_LOG_ERR(( "fchown(%s/%s) failed (%i-%s)", snap_dir, name, errno, fd_io_strerror( errno ) ));
    1196             : 
    1197           0 :   if( FD_UNLIKELY( -1==fchmod( snap_fd, S_IRUSR|S_IWUSR ) ) )
    1198           0 :     FD_LOG_ERR(( "fchmod(%s/%s) failed (%i-%s)", snap_dir, name, errno, fd_io_strerror( errno ) ));
    1199           0 : }
    1200             : 
    1201             : ulong
    1202           0 : initialize_snapshot_fds( config_t const * config ) {
    1203             : 
    1204           0 :   int download_enabled = fd_topo_find_tile( &config->topo, "snapct", 0UL )!=ULONG_MAX &&
    1205           0 :                          (config->firedancer.snapshots.sources.gossip.allow_any ||
    1206           0 :                           config->firedancer.snapshots.sources.gossip.allow_list_cnt ||
    1207           0 :                           config->firedancer.snapshots.sources.servers_cnt);
    1208           0 :   int upload_enabled = fd_topo_find_tile( &config->topo, "snapsv", 0UL )!=ULONG_MAX;
    1209           0 :   int dio_enabled    = fd_topo_find_tile( &config->topo, "snapzp", 0UL )!=ULONG_MAX;
    1210           0 :   fd_snap_pool_layout_t layout = fd_snap_pool_layout( config->firedancer.snapshots.max_full_snapshots_to_keep,
    1211           0 :                                                       config->firedancer.snapshots.max_incremental_snapshots_to_keep,
    1212           0 :                                                       config->firedancer.snapshots.incremental_snapshots,
    1213           0 :                                                       download_enabled );
    1214           0 :   ulong snap_full_max     = layout.full_max;
    1215           0 :   ulong snap_incr_max     = layout.incr_max;
    1216           0 :   ulong retained_snap_max = layout.retained_max;
    1217           0 :   ulong snap_max          = layout.max;
    1218           0 :   if( FD_UNLIKELY( !snap_max ) ) return 0UL;
    1219             : 
    1220           0 :   char const * snap_dir = config->paths.snapshots;
    1221           0 :   int dir_fd = open( snap_dir, O_RDONLY|O_DIRECTORY|O_CLOEXEC );
    1222           0 :   if( FD_UNLIKELY( -1==dir_fd ) ) FD_LOG_ERR(( "open(%s) failed (%i-%s)", snap_dir, errno, fd_io_strerror( errno ) ));
    1223             : 
    1224             :   /* This is additionally validated in fd_config_validatef() */
    1225           0 :   FD_CHECK_ERR( snap_max<=FD_SNAP_MAX, "[snapshots.max_{full,incremental}_snapshots_to_keep] is set too high" );
    1226           0 :   fd_backup_inode_t snap_full[ FD_SNAP_MAX ];
    1227           0 :   fd_backup_inode_t snap_incr[ FD_SNAP_MAX ];
    1228           0 :   fd_backup_inode_t scratch[ 2 ] = {
    1229           0 :     { .full_slot=ULONG_MAX, .incr_slot=ULONG_MAX },
    1230           0 :     { .full_slot=ULONG_MAX, .incr_slot=ULONG_MAX }
    1231           0 :   };
    1232           0 :   ulong             snap_full_cnt = 0UL;
    1233           0 :   ulong             snap_incr_cnt = 0UL;
    1234           0 :   FD_STATIC_ASSERT( sizeof(snap_full)+sizeof(snap_incr) < 2UL<<20, "stack overflow" );
    1235             : 
    1236             :   /* First, reconcile with the existing snapshots */
    1237           0 :   int dir_fd2 = dup( dir_fd );
    1238           0 :   if( FD_UNLIKELY( -1==dir_fd2 ) ) FD_LOG_ERR(( "dup(%s) failed (%i-%s)", snap_dir, errno, fd_io_strerror( errno ) ));
    1239           0 :   DIR * dir = fdopendir( dir_fd2 );
    1240           0 :   if( FD_UNLIKELY( !dir ) ) FD_LOG_ERR(( "fdopendir(%s) failed (%i-%s)", snap_dir, errno, fd_io_strerror( errno ) ));
    1241           0 :   struct dirent * entry;
    1242           0 :   for(;;) {
    1243           0 :     errno = 0;
    1244           0 :     entry = readdir( dir );
    1245           0 :     if( FD_UNLIKELY( !entry ) ) break;
    1246           0 :     if( FD_UNLIKELY( !strcmp( entry->d_name, ".") || !strcmp( entry->d_name, ".." ) ) ) continue;
    1247             : 
    1248             :     /* delete partial files */
    1249           0 :     if( FD_UNLIKELY( matches_partial_snap_filename( entry->d_name ) ) ) {
    1250           0 :       if( FD_UNLIKELY( -1==unlinkat( dir_fd, entry->d_name, 0 ) && errno!=ENOENT ) )
    1251           0 :         FD_LOG_ERR(( "unlinkat(%s/%s) failed (%i-%s)", snap_dir, entry->d_name, errno, fd_io_strerror( errno ) ));
    1252           0 :       continue;
    1253           0 :     }
    1254             : 
    1255             :     /* decode file name */
    1256           0 :     int is_zstd;
    1257           0 :     ulong entry_full_slot, entry_incremental_slot;
    1258           0 :     uchar decoded_hash[ FD_HASH_FOOTPRINT ];
    1259           0 :     if( FD_UNLIKELY( -1==fd_ssarchive_parse_filename( entry->d_name, &entry_full_slot, &entry_incremental_slot, decoded_hash, &is_zstd ) ) ) continue;
    1260           0 :     if( FD_UNLIKELY( strlen( entry->d_name )>=FD_SNAP_NAME_MAX ) ) continue;
    1261           0 :     fd_backup_inode_t inode = (fd_backup_inode_t){
    1262           0 :       .full_slot = entry_full_slot,
    1263           0 :       .incr_slot = entry_incremental_slot
    1264           0 :     };
    1265           0 :     fd_cstr_ncpy( inode.name, entry->d_name, FD_SNAP_NAME_MAX );
    1266             : 
    1267             :     /* register snapshot (and lazily delete snaps if there are too many)
    1268             :        the lazy drop must free a slot for the append below, even if
    1269             :        retention is configured at the hard capacity */
    1270           0 :     if( entry_incremental_slot==ULONG_MAX ) {
    1271           0 :       if( FD_UNLIKELY( snap_full_cnt>=FD_SNAP_MAX ) ) {
    1272           0 :         drop_old_snaps( dir_fd, snap_dir, snap_full, &snap_full_cnt, fd_ulong_min( snap_full_max, FD_SNAP_MAX-1UL ) );
    1273           0 :       }
    1274           0 :       snap_full[ snap_full_cnt++ ] = inode;
    1275           0 :     } else {
    1276           0 :       if( FD_UNLIKELY( snap_incr_cnt>=FD_SNAP_MAX ) ) {
    1277           0 :         drop_old_snaps( dir_fd, snap_dir, snap_incr, &snap_incr_cnt, fd_ulong_min( snap_incr_max, FD_SNAP_MAX-1UL ) );
    1278           0 :       }
    1279           0 :       snap_incr[ snap_incr_cnt++ ] = inode;
    1280           0 :     }
    1281           0 :   }
    1282           0 :   if( FD_UNLIKELY( errno ) ) FD_LOG_ERR(( "readdir(%s) failed (%i-%s)", snap_dir, errno, fd_io_strerror( errno ) ));
    1283           0 :   if( FD_UNLIKELY( -1==closedir( dir ) ) ) FD_LOG_ERR(( "closedir(%s) failed (%i-%s)", snap_dir, errno, fd_io_strerror( errno ) ));
    1284             : 
    1285             :   /* FIXME this should be smart enough to drop incremental snapshots
    1286             :            that no longer have a matching full snapshot. */
    1287           0 :   drop_old_snaps( dir_fd, snap_dir, snap_full, &snap_full_cnt, snap_full_max );
    1288           0 :   drop_old_snaps( dir_fd, snap_dir, snap_incr, &snap_incr_cnt, snap_incr_max );
    1289             : 
    1290             :   /* zero pad */
    1291           0 :   for( ulong i=snap_full_cnt; i<snap_full_max; i++ ) snap_full[ i ] = (fd_backup_inode_t){ .full_slot=ULONG_MAX, .incr_slot=ULONG_MAX };
    1292           0 :   for( ulong i=snap_incr_cnt; i<snap_incr_max; i++ ) snap_incr[ i ] = (fd_backup_inode_t){ .full_slot=ULONG_MAX, .incr_slot=ULONG_MAX };
    1293             : 
    1294           0 :   for( ulong i=0UL; i<snap_max; i++ ) {
    1295           0 :     fd_backup_inode_t * inode;
    1296           0 :     if(      i<snap_full_max     ) inode = &snap_full[ i ];
    1297           0 :     else if( i<retained_snap_max ) inode = &snap_incr[ i-snap_full_max ];
    1298           0 :     else                           inode = &scratch [ i-retained_snap_max ];
    1299             : 
    1300           0 :     int snap_fd;
    1301           0 :     if( FD_UNLIKELY( !inode->name[ 0 ] ) ) {
    1302           0 :       fd_snap_pool_partial_name( inode->name, (uint)i );
    1303           0 :       snap_fd = openat( dir_fd, inode->name, O_RDWR|O_CREAT|O_EXCL|O_CLOEXEC|O_NOFOLLOW, S_IRUSR|S_IWUSR );
    1304           0 :       if( FD_UNLIKELY( -1==snap_fd ) ) FD_LOG_ERR(( "openat(%s/%s) failed (%i-%s)", snap_dir, inode->name, errno, fd_io_strerror( errno ) ));
    1305           0 :     } else {
    1306           0 :       snap_fd = openat( dir_fd, inode->name, O_RDWR|O_CLOEXEC|O_NOFOLLOW );
    1307           0 :       if( FD_UNLIKELY( -1==snap_fd ) ) FD_LOG_ERR(( "openat(%s/%s) failed (%i-%s)", snap_dir, inode->name, errno, fd_io_strerror( errno ) ));
    1308           0 :       snap_check_fd( snap_dir, inode->name, snap_fd );
    1309           0 :     }
    1310             : 
    1311           0 :     snap_reperm_fd( snap_dir, inode->name, snap_fd, config->uid, config->gid );
    1312             : 
    1313           0 :     if( FD_UNLIKELY( -1==dup2( snap_fd, FD_SNAP_FD( i ) ) ) ) FD_LOG_ERR(( "dup2() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
    1314           0 :     if( FD_UNLIKELY( -1==close( snap_fd ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
    1315             : 
    1316           0 :     if( dio_enabled ) {
    1317           0 :       int dio_fd = openat( dir_fd, inode->name, O_WRONLY|O_DIRECT );
    1318           0 :       if( FD_UNLIKELY( -1==dio_fd ) ) FD_LOG_ERR(( "openat(%s/%s, O_DIRECT) failed (%i-%s)", snap_dir, inode->name, errno, fd_io_strerror( errno ) ));
    1319           0 :       if( FD_UNLIKELY( -1==dup2( dio_fd, FD_SNAP_DIO_FD( i ) ) ) ) FD_LOG_ERR(( "dup2() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
    1320           0 :       if( FD_UNLIKELY( -1==close( dio_fd ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
    1321           0 :     }
    1322             : 
    1323           0 :     if( upload_enabled ) {
    1324           0 :       int ro_fd = openat( dir_fd, inode->name, O_RDONLY|O_NOFOLLOW );
    1325           0 :       if( FD_UNLIKELY( -1==ro_fd ) ) FD_LOG_ERR(( "openat(%s/%s, O_RDONLY) failed (%i-%s)", snap_dir, inode->name, errno, fd_io_strerror( errno ) ));
    1326           0 :       if( FD_UNLIKELY( -1==dup2( ro_fd, FD_SNAP_RO_FD( i ) ) ) ) FD_LOG_ERR(( "dup2() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
    1327           0 :       if( FD_UNLIKELY( -1==close( ro_fd ) ) ) FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
    1328           0 :     }
    1329           0 :   }
    1330             : 
    1331           0 :   if( FD_UNLIKELY( -1==close( dir_fd ) ) ) FD_LOG_ERR(( "close(%s) failed (%i-%s)", snap_dir, errno, fd_io_strerror( errno ) ));
    1332           0 :   return snap_max;
    1333           0 : }
    1334             : 
    1335             : /* The boot sequence is a little bit involved...
    1336             : 
    1337             :    A process tree is created that looks like,
    1338             : 
    1339             :    + main
    1340             :    +-- pidns
    1341             :        +-- agave
    1342             :        +-- tile 0
    1343             :        +-- tile 1
    1344             :        ...
    1345             : 
    1346             :    What we want is that if any process in the tree dies, all other
    1347             :    processes will also die.  This is done as follows,
    1348             : 
    1349             :     (a) pidns is the init process of a PID namespace, so if it dies the
    1350             :         kernel will terminate the child processes.
    1351             : 
    1352             :     (b) main is the parent of pidns, so it can issue a waitpid() on the
    1353             :         child PID, and when it completes terminate itself.
    1354             : 
    1355             :     (c) pidns is the parent of agave and the tiles, so it could
    1356             :         issue a waitpid() of -1 to wait for any of them to terminate,
    1357             :         but how would it know if main has died?
    1358             : 
    1359             :     (d) main creates a pipe, and passes the write end to pidns.  If main
    1360             :         dies, the pipe will be closed, and pidns will get a HUP on the
    1361             :         read end.  Then pidns creates a pipe per child and passes the
    1362             :         write end to the child.  If any of the children die, the pipe
    1363             :         will be closed, and pidns will get a HUP on the read end.
    1364             : 
    1365             :         Then pidns can call poll() on both the write end of the main
    1366             :         pipe and the read end of all the child pipes.  If any of them
    1367             :         raises SIGHUP, then pidns knows that the parent or a child has
    1368             :         died, and it can terminate itself, which due to (a) and (b)
    1369             :         will kill all other processes. */
    1370             : void
    1371             : run_firedancer( config_t * config,
    1372             :                 int        parent_pipefd,
    1373           0 :                 int        init_workspaces ) {
    1374             :   /* dump the topology we are using to the output log */
    1375           0 :   fd_topo_print_log( 0, &config->topo );
    1376             : 
    1377           0 :   run_firedancer_init( config, init_workspaces, 1 );
    1378             : 
    1379           0 :   if( FD_UNLIKELY( close( 0 ) ) ) FD_LOG_ERR(( "close(0) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
    1380           0 :   if( FD_UNLIKELY( fd_log_private_logfile_fd()!=1 && close( 1 ) ) ) FD_LOG_ERR(( "close(1) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
    1381             : 
    1382           0 :   int pipefd;
    1383           0 :   pid_namespace = clone_firedancer( config, parent_pipefd, &pipefd );
    1384             : 
    1385             :   /* Print the location of the logfile on SIGINT or SIGTERM, and also
    1386             :      kill the child.  They are connected by a pipe which the child is
    1387             :      polling so we don't strictly need to kill the child, but its helpful
    1388             :      to do that before printing the log location line, else it might
    1389             :      get interleaved due to timing windows in the shutdown. */
    1390           0 :   install_parent_signals();
    1391             : 
    1392           0 :   struct sock_filter seccomp_filter[ 128UL ];
    1393           0 :   populate_sock_filter_policy_main( 128UL, seccomp_filter, (uint)fd_log_private_logfile_fd(), (uint)pid_namespace );
    1394             : 
    1395           0 :   int allow_fds[ 4 ];
    1396           0 :   ulong allow_fds_cnt = 0;
    1397           0 :   allow_fds[ allow_fds_cnt++ ] = 2; /* stderr */
    1398           0 :   if( FD_LIKELY( fd_log_private_logfile_fd()!=-1 ) )
    1399           0 :     allow_fds[ allow_fds_cnt++ ] = fd_log_private_logfile_fd(); /* logfile, or maybe stdout */
    1400           0 :   allow_fds[ allow_fds_cnt++ ] = pipefd; /* read end of main pipe */
    1401           0 :   if( FD_UNLIKELY( parent_pipefd!=-1 ) )
    1402           0 :     allow_fds[ allow_fds_cnt++ ] = parent_pipefd; /* write end of parent pipe */
    1403             : 
    1404           0 :   if( FD_LIKELY( config->development.sandbox ) ) {
    1405           0 :     fd_sandbox_enter( config->uid,
    1406           0 :                       config->gid,
    1407           0 :                       0,
    1408           0 :                       0,
    1409           0 :                       0,
    1410           0 :                       1, /* Keep controlling terminal for main so it can receive Ctrl+C */
    1411           0 :                       0,
    1412           0 :                       0UL,
    1413           0 :                       0UL,
    1414           0 :                       0UL,
    1415           0 :                       0UL,
    1416           0 :                       allow_fds_cnt,
    1417           0 :                       allow_fds,
    1418           0 :                       sock_filter_policy_main_instr_cnt,
    1419           0 :                       seccomp_filter );
    1420           0 :   } else {
    1421           0 :     fd_sandbox_switch_uid_gid( config->uid, config->gid );
    1422           0 :   }
    1423             : 
    1424             :   /* the only clean way to exit is SIGINT or SIGTERM on this parent process,
    1425             :      so if wait4() completes, it must be an error */
    1426           0 :   int wstatus;
    1427           0 :   if( FD_UNLIKELY( -1==wait4( pid_namespace, &wstatus, (int)__WALL, NULL ) ) )
    1428           0 :     FD_LOG_ERR(( "main wait4() failed (%i-%s)\nLog at \"%s\"", errno, fd_io_strerror( errno ), fd_log_private_path ));
    1429             : 
    1430           0 :   if( FD_UNLIKELY( WIFSIGNALED( wstatus ) ) ) fd_sys_util_exit_group( WTERMSIG( wstatus ) ? WTERMSIG( wstatus ) : 1 );
    1431           0 :   else fd_sys_util_exit_group( WEXITSTATUS( wstatus ) ? WEXITSTATUS( wstatus ) : 1 );
    1432           0 : }
    1433             : 
    1434             : void
    1435             : run_cmd_fn( args_t *   args FD_PARAM_UNUSED,
    1436           0 :             config_t * config ) {
    1437           0 :   #define CHECK_PORT_NON_ZERO( field ) \
    1438           0 :     if( FD_UNLIKELY( config->field==0 ) ) { \
    1439           0 :       FD_LOG_ERR(( #field " is not set in your configuration file. Please set it to a non-zero value." )); \
    1440           0 :     }
    1441             : 
    1442           0 :   if( FD_UNLIKELY( !config->gossip.entrypoints_cnt && !config->development.bootstrap ) )
    1443           0 :     FD_LOG_ERR(( "No entrypoints specified in configuration file under [gossip.entrypoints], but "
    1444           0 :                  "at least one is needed to determine how to connect to the Solana cluster. If "
    1445           0 :                  "you want to start a new cluster in a development environment, use `fddev` instead "
    1446           0 :                  "of `fdctl`. If you want to use an existing genesis, set [development.bootstrap] "
    1447           0 :                  "to \"true\" in the configuration file." ));
    1448             : 
    1449           0 :   for( ulong i=0; i<config->gossip.entrypoints_cnt; i++ ) {
    1450           0 :     if( FD_UNLIKELY( !strcmp( config->gossip.entrypoints[ i ], "" ) ) )
    1451           0 :       FD_LOG_ERR(( "One of the entrypoints in your configuration file under [gossip.entrypoints] is "
    1452           0 :                    "empty. Please remove the empty entrypoint or set it correctly. "));
    1453           0 :   }
    1454             : 
    1455           0 :   CHECK_PORT_NON_ZERO( gossip.port );
    1456           0 :   CHECK_PORT_NON_ZERO( tiles.quic.quic_transaction_listen_port );
    1457           0 :   CHECK_PORT_NON_ZERO( tiles.quic.regular_transaction_listen_port );
    1458           0 :   CHECK_PORT_NON_ZERO( tiles.shred.shred_listen_port );
    1459           0 :   CHECK_PORT_NON_ZERO( tiles.metric.prometheus_listen_port );
    1460           0 :   CHECK_PORT_NON_ZERO( tiles.gui.gui_listen_port );
    1461             : 
    1462           0 :   #undef CHECK_PORT_NON_ZERO
    1463             : 
    1464           0 :   run_firedancer( config, -1, 1 );
    1465           0 : }
    1466             : 
    1467             : static void
    1468           0 : run1_args_help( fd_action_help_t * help ) {
    1469           0 :   fd_action_help_arg( help, "<tile-name>", NULL,   "Type of tile to run (e.g. `net`, `quic`, `replay`).  A tile is a single\n"
    1470           0 :                                                   "thread pinned to a CPU core that performs one part of the validator's work" );
    1471             :   fd_action_help_arg( help, "<kind-id>",   NULL,   "Zero-based index selecting which instance of that tile type to run\n"
    1472           0 :                                                   "when the topology has more than one" );
    1473           0 :   fd_action_help_arg( help, "--pipe-fd",   "<fd>", "Internal use: file descriptor over which the parent supervisor process\n"
    1474           0 :                                                   "communicates with this tile (default -1, standalone)" );
    1475           0 : }
    1476             : 
    1477             : action_t fd_action_run1 = {
    1478             :   .name        = "run1",
    1479             :   .args        = run1_cmd_args,
    1480             :   .fn          = run1_cmd_fn,
    1481             :   .perm        = NULL,
    1482             :   .description = "Start up a single Firedancer tile",
    1483             :   .detail      = "Runs one tile of the validator topology in the current process.  A tile is a\n"
    1484             :                  "single thread pinned to a CPU core that performs one part of the validator's\n"
    1485             :                  "work.  This is primarily an internal command used by `run` to spawn individual\n"
    1486             :                  "tiles; most operators should use `run` instead.",
    1487             :   .usage       = "run1 <tile-name> <kind-id> [OPTIONS]",
    1488             :   .args_help   = run1_args_help,
    1489             : };
    1490             : 
    1491             : action_t fd_action_run = {
    1492             :   .name           = "run",
    1493             :   .args           = NULL,
    1494             :   .fn             = run_cmd_fn,
    1495             :   .require_config = 1,
    1496             :   .perm           = run_cmd_perm,
    1497             :   .description    = "Start up a Firedancer validator",
    1498             :   .detail         = "Boots and runs the full validator described by the configuration file.  This\n"
    1499             :                     "is the main command operators use to run Firedancer.  It must be started with\n"
    1500             :                     "sufficient privileges to perform boot-time setup, after which it drops\n"
    1501             :                     "privileges to the configured user.",
    1502             :   .usage          = "run [OPTIONS]",
    1503             :   .permission_err = "insufficient permissions to execute command `%s`. It is recommended "
    1504             :                     "to start Firedancer as the root user, but you can also start it "
    1505             :                     "with the missing capabilities listed above. The program only needs "
    1506             :                     "to start with elevated permissions to do privileged operations at "
    1507             :                     "boot, and will immediately drop permissions and switch to the user "
    1508             :                     "specified in your configuration file once they are complete. Firedancer "
    1509             :                     "will not execute outside of the boot process as root, and will refuse "
    1510             :                     "to start if it cannot drop privileges. Firedancer needs to be started "
    1511             :                     "privileged to configure high performance networking with XDP.",
    1512             : };

Generated by: LCOV version 1.14