Line data Source code
1 : #define _GNU_SOURCE
2 :
3 : #include "fd_diag_tile.h"
4 :
5 : #include "../bundle/fd_bundle_tile.h"
6 : #include "../metrics/fd_metrics.h"
7 : #include "../stem/fd_stem.h"
8 : #include "../topo/fd_topo.h"
9 : #include "../topo/fd_cpu_topo.h"
10 : #include "../../util/tile/fd_tile_private.h"
11 : #include "../../util/io/fd_io.h"
12 :
13 : #include <fcntl.h>
14 : #include <errno.h>
15 : #include <stdlib.h>
16 : #include <sys/types.h> /* SEEK_SET */
17 : #include <sys/stat.h>
18 : #include <sys/vfs.h>
19 : #include <time.h>
20 : #include <unistd.h>
21 :
22 : #include "fd_proc_interrupts.h"
23 : #include "generated/fd_diag_tile_seccomp.h"
24 :
25 0 : #define REPORT_INTERVAL_MILLIS (100L)
26 0 : #define SYSTEM_REPORT_INTERVAL_NANOS (30000000000L)
27 :
28 0 : #define DIAG_WKSP_TILE_IDX_SHARED (ULONG_MAX)
29 :
30 :
31 : struct fd_diag_tile {
32 : long next_report_nanos;
33 :
34 : ulong tile_cnt;
35 : int is_voting;
36 :
37 : struct {
38 : ulong bundle_tile_idx[ FD_TILE_MAX ];
39 : ulong bundle_cnt;
40 : ulong shred_tile_idx[ FD_TILE_MAX ];
41 : ulong shred_cnt;
42 : ulong tower_idx;
43 : ulong votor_idx;
44 : ulong replay_idx;
45 : } tiles;
46 :
47 : ulong starttime_nanos[ FD_TILE_MAX ];
48 : long first_seen_died[ FD_TILE_MAX ];
49 :
50 : int stat_fds[ FD_TILE_MAX ];
51 : int sched_fds[ FD_TILE_MAX ];
52 :
53 : ulong irq_cnt[ FD_METRICS_ENUM_SOFTIRQ_CNT ][ FD_TILE_MAX ];
54 : fd_cpuset_t cpu_has_tile[ fd_cpuset_word_cnt ];
55 : int proc_interrupts_fd;
56 : int proc_softirqs_fd;
57 : int proc_stat_fd;
58 : int proc_meminfo_fd;
59 : ulong device_irq_baseline[ FD_TILE_MAX ];
60 : ulong tlb_baseline[ FD_TILE_MAX ];
61 : ulong loc_baseline[ FD_TILE_MAX ];
62 : ulong irq_ticks_baseline[ FD_TILE_MAX ];
63 : ulong softirq_baseline[ FD_METRICS_ENUM_SOFTIRQ_CNT ][ FD_TILE_MAX ];
64 :
65 : ulong volatile * metrics [ FD_TILE_MAX ];
66 : ushort cpu_to_tile[ FD_TILE_MAX ];
67 :
68 : int gui_enabled;
69 : long next_system_report_nanos;
70 : ushort numa_entry_idx[ FD_DIAG_SYSTEM_TILE_MAX ][ FD_DIAG_SYSTEM_NUMA_MAX ];
71 : struct {
72 : ulong cnt;
73 : struct {
74 : char name[ FD_SHMEM_NAME_MAX ];
75 : ulong tile_idx;
76 : ulong numa_idx;
77 : ulong bytes;
78 : ushort numa_slot;
79 : } wksp[ FD_TOPO_MAX_WKSPS ];
80 : ushort stack_numa_slot[ FD_DIAG_SYSTEM_TILE_MAX ];
81 : } memory;
82 : struct {
83 : ulong cnt;
84 : struct {
85 : ushort idx;
86 : int meminfo_fd;
87 : } node[ FD_DIAG_SYSTEM_NUMA_MAX ];
88 : } numa;
89 :
90 : struct {
91 : void * mem;
92 : ulong idx;
93 : ulong chunk0;
94 : ulong wmark;
95 : ulong chunk;
96 : } system_out;
97 : fd_diag_system_resources_t system_resources;
98 :
99 : struct {
100 : int fd; /* O_PATH reference to an anonymous regular file */
101 : ulong mnt_id; /* mount ID from /proc/self/fdinfo */
102 : char path[ PATH_MAX ];
103 : } mounts[ FD_DIAG_SYSTEM_FILE_MAX ];
104 : ulong mount_cnt;
105 :
106 : struct {
107 : char path[ PATH_MAX ];
108 : uint category;
109 : uint mount_idx;
110 : int data_fd;
111 : ulong volatile * metric;
112 : } files[ FD_DIAG_SYSTEM_FILE_MAX ];
113 : ulong file_cnt;
114 :
115 : struct {
116 : ulong prev_vote_slot;
117 : long vote_slot_changed_ns;
118 : ulong prev_reset_slot;
119 : long reset_slot_changed_ns;
120 : ulong prev_turbine_slot;
121 : long turbine_slot_changed_ns;
122 :
123 : ulong snapshot_turbine_bytes;
124 : ulong snapshot_repair_bytes;
125 : long byte_snapshot_ns;
126 : int repair_outpacing;
127 : } check_engine;
128 : };
129 :
130 : typedef struct fd_diag_tile fd_diag_tile_t;
131 :
132 : FD_FN_CONST static inline ulong
133 0 : scratch_align( void ) {
134 0 : return alignof(fd_diag_tile_t);
135 0 : }
136 :
137 : FD_FN_PURE static inline ulong
138 0 : scratch_footprint( fd_topo_tile_t const * tile ) {
139 0 : (void)tile;
140 0 : return sizeof(fd_diag_tile_t);
141 0 : }
142 :
143 : static int
144 : read_stat_file( int fd,
145 : ulong ns_per_tick,
146 0 : volatile ulong * metrics ) {
147 0 : if( FD_UNLIKELY( -1==lseek( fd, 0, SEEK_SET ) ) ) FD_LOG_ERR(( "lseek failed (%i-%s)", errno, strerror( errno ) ));
148 :
149 0 : char contents[ 4096 ] = {0};
150 0 : ulong contents_len = 0UL;
151 :
152 0 : while( 1 ) {
153 0 : if( FD_UNLIKELY( contents_len>=sizeof( contents ) ) ) FD_LOG_ERR(( "stat contents overflow" ));
154 0 : long n = read( fd, contents + contents_len, sizeof( contents ) - contents_len );
155 0 : if( FD_UNLIKELY( -1==n ) ) {
156 0 : if( FD_UNLIKELY( errno==ESRCH ) ) return 1;
157 0 : FD_LOG_ERR(( "read failed (%i-%s)", errno, strerror( errno ) ));
158 0 : }
159 0 : if( FD_LIKELY( 0==n ) ) break;
160 0 : contents_len += (ulong)n;
161 0 : }
162 :
163 : /* Parse stat file: fields are space-separated.
164 : Field 10 (1-indexed) = minflt, field 12 = majflt,
165 : field 14 = utime, field 15 = stime (all in clock ticks). */
166 0 : char * saveptr;
167 0 : char * token = strtok_r( contents, " ", &saveptr );
168 0 : ulong field_idx = 0UL;
169 :
170 0 : while( token ) {
171 0 : if( FD_UNLIKELY( 9UL==field_idx ) ) {
172 0 : char * endptr;
173 0 : ulong minflt = strtoul( token, &endptr, 10 );
174 0 : if( FD_UNLIKELY( *endptr!='\0' || minflt==ULONG_MAX ) ) FD_LOG_ERR(( "strtoul failed for minflt" ));
175 0 : metrics[ FD_METRICS_COUNTER_TILE_PAGE_FAULT_MINOR_OFF ] = minflt;
176 0 : } else if( FD_UNLIKELY( 11UL==field_idx ) ) {
177 0 : char * endptr;
178 0 : ulong majflt = strtoul( token, &endptr, 10 );
179 0 : if( FD_UNLIKELY( *endptr!='\0' || majflt==ULONG_MAX ) ) FD_LOG_ERR(( "strtoul failed for majflt" ));
180 0 : metrics[ FD_METRICS_COUNTER_TILE_PAGE_FAULT_MAJOR_OFF ] = majflt;
181 0 : } else if( FD_UNLIKELY( 13UL==field_idx ) ) {
182 0 : char * endptr;
183 0 : ulong utime_ticks = strtoul( token, &endptr, 10 );
184 0 : if( FD_UNLIKELY( *endptr!='\0' || utime_ticks==ULONG_MAX ) ) FD_LOG_ERR(( "strtoul failed for utime" ));
185 0 : metrics[ FD_METRICS_COUNTER_TILE_CPU_DURATION_NANOS_USER_OFF ] = utime_ticks*ns_per_tick;
186 0 : } else if( FD_UNLIKELY( 14UL==field_idx ) ) {
187 0 : char * endptr;
188 0 : ulong stime_ticks = strtoul( token, &endptr, 10 );
189 0 : if( FD_UNLIKELY( *endptr!='\0' || stime_ticks==ULONG_MAX ) ) FD_LOG_ERR(( "strtoul failed for stime" ));
190 0 : metrics[ FD_METRICS_COUNTER_TILE_CPU_DURATION_NANOS_SYSTEM_OFF ] = stime_ticks*ns_per_tick;
191 0 : } else if( FD_UNLIKELY( 38UL==field_idx ) ) {
192 0 : char * endptr;
193 0 : ulong last_cpu = strtoul( token, &endptr, 10 );
194 0 : if( FD_UNLIKELY( *endptr!='\0' || last_cpu==ULONG_MAX ) ) FD_LOG_ERR(( "strtoul failed for processor" ));
195 0 : metrics[ FD_METRICS_GAUGE_TILE_LAST_CPU_OFF ] = last_cpu;
196 0 : break; /* No need to parse stat further */
197 0 : }
198 0 : token = strtok_r( NULL, " ", &saveptr );
199 0 : field_idx++;
200 0 : }
201 :
202 0 : if( FD_UNLIKELY( field_idx!=38UL ) ) FD_LOG_ERR(( "failed to parse /proc/<pid>/task/<tid>/stat" ));
203 :
204 0 : return 0;
205 0 : }
206 :
207 : static int
208 : read_sched_file( int fd,
209 0 : volatile ulong * metrics ) {
210 0 : if( FD_UNLIKELY( -1==lseek( fd, 0, SEEK_SET ) ) ) FD_LOG_ERR(( "lseek failed (%i-%s)", errno, strerror( errno ) ));
211 :
212 0 : char contents[ 8192 ] = {0};
213 0 : ulong contents_len = 0UL;
214 :
215 0 : while( 1 ) {
216 0 : if( FD_UNLIKELY( contents_len>=sizeof( contents ) ) ) FD_LOG_ERR(( "sched contents overflow" ));
217 0 : long n = read( fd, contents + contents_len, sizeof( contents ) - contents_len );
218 0 : if( FD_UNLIKELY( -1==n ) ) {
219 0 : if( FD_UNLIKELY( errno==ESRCH ) ) return 1;
220 0 : FD_LOG_ERR(( "read failed (%i-%s)", errno, strerror( errno ) ));
221 0 : }
222 0 : if( FD_LIKELY( 0==n ) ) break;
223 0 : contents_len += (ulong)n;
224 0 : }
225 :
226 0 : int found_wait_sum = 0;
227 0 : int found_voluntary = 0;
228 0 : int found_involuntary = 0;
229 :
230 0 : char * line = contents;
231 0 : while( 1 ) {
232 0 : char * next_line = strchr( line, '\n' );
233 0 : if( FD_UNLIKELY( NULL==next_line ) ) break;
234 0 : *next_line = '\0';
235 :
236 0 : if( FD_UNLIKELY( !strncmp( line, "wait_sum", 8UL ) ) ) {
237 0 : char * colon = strchr( line, ':' );
238 0 : if( FD_LIKELY( colon ) ) {
239 0 : char * value = colon + 1;
240 0 : while( ' '==*value || '\t'==*value ) value++;
241 : /* wait_sum is displayed as seconds.microseconds (e.g., "123.456789").
242 : Parse both components as integers and convert to nanoseconds. */
243 0 : char * endptr;
244 0 : ulong seconds = strtoul( value, &endptr, 10 );
245 0 : if( FD_UNLIKELY( '.'!=*endptr ) ) FD_LOG_ERR(( "expected '.' after seconds in wait_sum" ));
246 0 : if( FD_UNLIKELY( seconds==ULONG_MAX ) ) FD_LOG_ERR(( "strtoul overflow for wait_sum seconds" ));
247 0 : ulong microseconds = strtoul( endptr + 1, &endptr, 10 );
248 0 : if( FD_UNLIKELY( '\0'!=*endptr ) ) FD_LOG_ERR(( "unexpected char after microseconds in wait_sum" ));
249 0 : if( FD_UNLIKELY( microseconds==ULONG_MAX ) ) FD_LOG_ERR(( "strtoul overflow for wait_sum microseconds" ));
250 0 : ulong wait_sum_ns = seconds*1000000000UL + microseconds*1000UL;
251 0 : metrics[ FD_METRICS_COUNTER_TILE_CPU_DURATION_NANOS_WAIT_OFF ] = wait_sum_ns;
252 0 : found_wait_sum = 1;
253 0 : }
254 0 : } else if( FD_UNLIKELY( !strncmp( line, "nr_voluntary_switches", 21UL ) ) ) {
255 0 : char * colon = strchr( line, ':' );
256 0 : if( FD_LIKELY( colon ) ) {
257 0 : char * value = colon + 1;
258 0 : while( ' '==*value || '\t'==*value ) value++;
259 0 : char * endptr;
260 0 : ulong voluntary_switches = strtoul( value, &endptr, 10 );
261 0 : if( FD_UNLIKELY( '\0'!=*endptr ) ) FD_LOG_ERR(( "unexpected char after nr_voluntary_switches" ));
262 0 : if( FD_UNLIKELY( voluntary_switches==ULONG_MAX ) ) FD_LOG_ERR(( "strtoul overflow for nr_voluntary_switches" ));
263 0 : metrics[ FD_METRICS_COUNTER_TILE_CONTEXT_SWITCH_VOLUNTARY_OFF ] = voluntary_switches;
264 0 : found_voluntary = 1;
265 0 : }
266 0 : } else if( FD_UNLIKELY( !strncmp( line, "nr_involuntary_switches", 23UL ) ) ) {
267 0 : char * colon = strchr( line, ':' );
268 0 : if( FD_LIKELY( colon ) ) {
269 0 : char * value = colon + 1;
270 0 : while( ' '==*value || '\t'==*value ) value++;
271 0 : char * endptr;
272 0 : ulong involuntary_switches = strtoul( value, &endptr, 10 );
273 0 : if( FD_UNLIKELY( '\0'!=*endptr ) ) FD_LOG_ERR(( "unexpected char after nr_involuntary_switches" ));
274 0 : if( FD_UNLIKELY( involuntary_switches==ULONG_MAX ) ) FD_LOG_ERR(( "strtoul overflow for nr_involuntary_switches" ));
275 0 : metrics[ FD_METRICS_COUNTER_TILE_CONTEXT_SWITCH_INVOLUNTARY_OFF ] = involuntary_switches;
276 0 : found_involuntary = 1;
277 0 : }
278 0 : }
279 :
280 0 : line = next_line + 1;
281 0 : }
282 :
283 : // wait_sum not present on kernels compiled without CONFIG_SCHEDSTATS=y
284 : // if( FD_UNLIKELY( !found_wait_sum ) ) FD_LOG_ERR(( "wait_sum not found in sched file" ));
285 0 : (void)found_wait_sum;
286 0 : if( FD_UNLIKELY( !found_voluntary ) ) FD_LOG_ERR(( "nr_voluntary_switches not found in sched file" ));
287 0 : if( FD_UNLIKELY( !found_involuntary ) ) FD_LOG_ERR(( "nr_involuntary_switches not found in sched file" ));
288 :
289 0 : return 0;
290 0 : }
291 :
292 : static void
293 0 : check_engine_metric( fd_diag_tile_t * ctx, long now ) {
294 0 : static ulong const vote_distance_threshold = 150UL;
295 0 : static long const vote_stall_threshold_ns = 60L*1000L*1000L*1000L;
296 0 : static ulong const replay_distance_threshold = 12UL;
297 0 : static long const replay_stall_threshold_ns = 12L*1000L*1000L*1000L;
298 0 : static long const turbine_stall_threshold_ns = 12L*1000L*1000L*1000L;
299 0 : static long const turbine_byte_cmp_window_ns = 12L*1000L*1000L*1000L;
300 :
301 0 : ulong bundle_cnt = ctx->tiles.bundle_cnt;
302 0 : ulong bundle_status = FD_DIAG_BUNDLE_STATUS_DISABLED;
303 0 : if( FD_LIKELY( bundle_cnt ) ) {
304 : /* Find the best state across all bundle tiles.
305 : Priority: connected > sleeping > connecting > disconnected */
306 0 : int any_connected = 0;
307 0 : int any_sleeping = 0;
308 0 : int any_connecting = 0;
309 0 : for( ulong i=0UL; i<bundle_cnt; i++ ) {
310 0 : volatile ulong * m = ctx->metrics[ ctx->tiles.bundle_tile_idx[ i ] ];
311 0 : ulong state = m[ FD_METRICS_GAUGE_BUNDLE_STATE_OFF ];
312 0 : if( FD_LIKELY( state==FD_BUNDLE_STATE_CONNECTED ) ) any_connected = 1;
313 0 : else if( state==FD_BUNDLE_STATE_SLEEPING ) any_sleeping = 1;
314 0 : else if( state==FD_BUNDLE_STATE_CONNECTING ) any_connecting = 1;
315 0 : }
316 0 : if( any_connected ) bundle_status = FD_DIAG_BUNDLE_STATUS_CONNECTED;
317 0 : else if( any_sleeping ) bundle_status = FD_DIAG_BUNDLE_STATUS_SLEEPING;
318 0 : else if( any_connecting ) bundle_status = FD_DIAG_BUNDLE_STATUS_CONNECTING;
319 0 : else bundle_status = FD_DIAG_BUNDLE_STATUS_DISCONNECTED;
320 0 : }
321 :
322 0 : ulong tower_idx = ctx->tiles.tower_idx;
323 0 : ulong votor_idx = ctx->tiles.votor_idx;
324 0 : ulong vote_status = FD_DIAG_VOTE_STATUS_DISABLED;
325 :
326 0 : if( FD_UNLIKELY( ctx->is_voting && votor_idx!=ULONG_MAX ) ) {
327 0 : ulong replay_idx_ag = ctx->tiles.replay_idx;
328 0 : if( FD_UNLIKELY( ctx->metrics[ votor_idx ][ FD_METRICS_GAUGE_TILE_STATUS_OFF ]!=1UL || replay_idx_ag==ULONG_MAX ) ) {
329 0 : vote_status = FD_DIAG_VOTE_STATUS_NOT_STARTED;
330 0 : } else {
331 0 : volatile ulong * m = ctx->metrics[ replay_idx_ag ];
332 0 : ulong vote_slot = m[ FD_METRICS_GAUGE_REPLAY_VOTE_SLOT_LAST_REWARDED_OFF ];
333 0 : ulong replay_slot = m[ FD_METRICS_GAUGE_REPLAY_RESET_SLOT_OFF ];
334 0 : if( FD_UNLIKELY( vote_slot==ULONG_MAX || !replay_slot ) ) {
335 0 : vote_status = FD_DIAG_VOTE_STATUS_NOT_STARTED;
336 0 : } else {
337 0 : int current = fd_int_if( replay_slot>=128UL, vote_slot+128UL>replay_slot, vote_slot>0UL );
338 0 : vote_status = fd_ulong_if( current, FD_DIAG_VOTE_STATUS_VOTING, FD_DIAG_VOTE_STATUS_DELINQUENT );
339 0 : }
340 0 : }
341 0 : } else if( FD_LIKELY( ctx->is_voting && tower_idx!=ULONG_MAX ) ) {
342 0 : if( FD_UNLIKELY( ctx->metrics[ tower_idx ][ FD_METRICS_GAUGE_TILE_STATUS_OFF ]!=1UL ) ) {
343 0 : vote_status = FD_DIAG_VOTE_STATUS_NOT_STARTED;
344 0 : } else {
345 0 : volatile ulong * m = ctx->metrics[ tower_idx ];
346 0 : ulong vote_slot = m[ FD_METRICS_GAUGE_TOWER_VOTE_SLOT_OFF ];
347 0 : ulong replay_slot = m[ FD_METRICS_GAUGE_TOWER_REPLAY_SLOT_OFF ];
348 0 : if( FD_UNLIKELY( vote_slot==ULONG_MAX || replay_slot==0UL ) ) {
349 0 : vote_status = FD_DIAG_VOTE_STATUS_NOT_STARTED;
350 0 : } else {
351 0 : if( FD_UNLIKELY( vote_slot!=ctx->check_engine.prev_vote_slot ) ) {
352 0 : ctx->check_engine.prev_vote_slot = vote_slot;
353 0 : ctx->check_engine.vote_slot_changed_ns = now;
354 0 : }
355 0 : int delinquent = (replay_slot>vote_slot && replay_slot-vote_slot>vote_distance_threshold) ||
356 0 : (now-ctx->check_engine.vote_slot_changed_ns>vote_stall_threshold_ns);
357 0 : vote_status = fd_ulong_if( delinquent,
358 0 : FD_DIAG_VOTE_STATUS_DELINQUENT,
359 0 : FD_DIAG_VOTE_STATUS_VOTING );
360 0 : }
361 0 : }
362 0 : }
363 :
364 0 : ulong replay_idx = ctx->tiles.replay_idx;
365 0 : int replay_running = replay_idx!=ULONG_MAX && ctx->metrics[ replay_idx ][ FD_METRICS_GAUGE_TILE_STATUS_OFF ]==1UL;
366 0 : ulong replay_status = FD_DIAG_REPLAY_STATUS_DISABLED;
367 0 : if( FD_LIKELY( replay_idx!=ULONG_MAX ) ) {
368 0 : if( FD_UNLIKELY( !replay_running ) ) {
369 0 : replay_status = FD_DIAG_REPLAY_STATUS_NOT_STARTED;
370 0 : } else {
371 0 : volatile ulong * m = ctx->metrics[ replay_idx ];
372 0 : ulong turbine_slot = m[ FD_METRICS_GAUGE_REPLAY_REASSEMBLY_LATEST_SLOT_OFF ];
373 0 : ulong reset_slot = m[ FD_METRICS_GAUGE_REPLAY_RESET_SLOT_OFF ];
374 0 : if( FD_UNLIKELY( reset_slot!=ctx->check_engine.prev_reset_slot ) ) {
375 0 : ctx->check_engine.prev_reset_slot = reset_slot;
376 0 : ctx->check_engine.reset_slot_changed_ns = now;
377 0 : }
378 0 : if( FD_UNLIKELY( (turbine_slot==0UL) || (reset_slot==0UL) ) ) {
379 0 : replay_status = FD_DIAG_REPLAY_STATUS_NOT_STARTED;
380 0 : } else if( FD_UNLIKELY( ((turbine_slot>reset_slot) && (turbine_slot-reset_slot>replay_distance_threshold)) ||
381 0 : (now-ctx->check_engine.reset_slot_changed_ns>replay_stall_threshold_ns) ) ) {
382 0 : replay_status = FD_DIAG_REPLAY_STATUS_BEHIND;
383 0 : } else {
384 0 : replay_status = FD_DIAG_REPLAY_STATUS_RUNNING;
385 0 : }
386 0 : }
387 0 : }
388 :
389 0 : ulong shred_cnt = ctx->tiles.shred_cnt;
390 0 : ulong turbine_status = FD_DIAG_TURBINE_STATUS_DISABLED;
391 0 : if( FD_LIKELY( replay_idx!=ULONG_MAX && shred_cnt>0UL ) ) {
392 0 : if( FD_UNLIKELY( !replay_running ) ) {
393 0 : turbine_status = FD_DIAG_TURBINE_STATUS_NOT_STARTED;
394 0 : } else {
395 0 : int all_shred_running = 1;
396 0 : ulong cur_turbine_bytes = 0UL, cur_repair_bytes = 0UL;
397 0 : for( ulong i=0UL; i<shred_cnt; i++ ) {
398 0 : volatile ulong * sm = ctx->metrics[ ctx->tiles.shred_tile_idx[ i ] ];
399 0 : cur_turbine_bytes += sm[ FD_METRICS_COUNTER_SHRED_SHRED_TURBINE_RX_BYTES_OFF ];
400 0 : cur_repair_bytes += sm[ FD_METRICS_COUNTER_SHRED_SHRED_REPAIR_RX_BYTES_OFF ];
401 0 : if( FD_UNLIKELY( sm[ FD_METRICS_GAUGE_TILE_STATUS_OFF ]!=1UL ) ) {
402 0 : all_shred_running = 0;
403 0 : break;
404 0 : }
405 0 : }
406 0 : if( FD_UNLIKELY( !all_shred_running ) ) {
407 0 : turbine_status = FD_DIAG_TURBINE_STATUS_NOT_STARTED;
408 0 : } else {
409 0 : ulong turbine_slot = ctx->metrics[ replay_idx ][ FD_METRICS_GAUGE_REPLAY_REASSEMBLY_LATEST_SLOT_OFF ];
410 0 : if( FD_UNLIKELY( turbine_slot!=ctx->check_engine.prev_turbine_slot ) ) {
411 0 : ctx->check_engine.prev_turbine_slot = turbine_slot;
412 0 : ctx->check_engine.turbine_slot_changed_ns = now;
413 0 : }
414 0 : if( FD_UNLIKELY( now-ctx->check_engine.byte_snapshot_ns>=turbine_byte_cmp_window_ns ) ) {
415 0 : ctx->check_engine.repair_outpacing = (cur_repair_bytes-ctx->check_engine.snapshot_repair_bytes)>(cur_turbine_bytes-ctx->check_engine.snapshot_turbine_bytes);
416 0 : ctx->check_engine.snapshot_turbine_bytes = cur_turbine_bytes;
417 0 : ctx->check_engine.snapshot_repair_bytes = cur_repair_bytes;
418 0 : ctx->check_engine.byte_snapshot_ns = now;
419 0 : }
420 :
421 0 : if( FD_UNLIKELY( turbine_slot==0UL ) ) {
422 0 : turbine_status = FD_DIAG_TURBINE_STATUS_NOT_STARTED;
423 0 : } else if( FD_UNLIKELY( now-ctx->check_engine.turbine_slot_changed_ns>turbine_stall_threshold_ns ) ) {
424 0 : turbine_status = FD_DIAG_TURBINE_STATUS_STALLED;
425 0 : } else if( FD_UNLIKELY( ctx->check_engine.repair_outpacing ) ) {
426 0 : turbine_status = FD_DIAG_TURBINE_STATUS_REPAIR_OUTPACING;
427 0 : } else {
428 0 : turbine_status = FD_DIAG_TURBINE_STATUS_RUNNING;
429 0 : }
430 0 : }
431 0 : }
432 0 : }
433 :
434 0 : FD_MGAUGE_SET( DIAG, BUNDLE_STATUS, bundle_status );
435 0 : FD_MGAUGE_SET( DIAG, VOTE_STATUS, vote_status );
436 0 : FD_MGAUGE_SET( DIAG, REPLAY_STATUS, replay_status );
437 0 : FD_MGAUGE_SET( DIAG, TURBINE_STATUS, turbine_status );
438 0 : }
439 :
440 : static void
441 0 : irq_metrics( fd_diag_tile_t * ctx ) {
442 0 : if( FD_UNLIKELY( -1==lseek( ctx->proc_softirqs_fd, 0, SEEK_SET ) ) ) FD_LOG_ERR(( "lseek failed (%i-%s)", errno, strerror( errno ) ));
443 0 : ulong softirq_cpu_cnt = fd_proc_softirqs_sum( ctx->proc_softirqs_fd, ctx->irq_cnt );
444 0 : if( FD_UNLIKELY( !softirq_cpu_cnt ) ) return; /* parse fail */
445 :
446 0 : ulong volatile * softirq_total = &fd_metrics_tl[ MIDX( COUNTER, DIAG, SOFTIRQ ) ];
447 0 : ulong volatile * softirq_undesired = &fd_metrics_tl[ MIDX( COUNTER, DIAG, SOFTIRQ_UNDESIRED ) ];
448 0 : for( ulong j=0UL; j<FD_METRICS_ENUM_SOFTIRQ_CNT; j++ ) {
449 0 : ulong tot_cnt = 0UL;
450 0 : ulong undesired_cnt = 0UL;
451 0 : for( ulong i=0UL; i<softirq_cpu_cnt; i++ ) {
452 0 : ulong since = fd_ulong_sat_sub( ctx->irq_cnt[ j ][ i ], ctx->softirq_baseline[ j ][ i ] );
453 0 : tot_cnt += since;
454 0 : if( fd_cpuset_test( ctx->cpu_has_tile, i ) ) {
455 0 : undesired_cnt += since;
456 0 : }
457 0 : }
458 0 : softirq_total [ j ] = tot_cnt;
459 0 : softirq_undesired[ j ] = undesired_cnt;
460 0 : }
461 :
462 0 : ulong * cpu_irq = ctx->irq_cnt[ 0 ]; /* re-use as scratch memory */
463 0 : ulong * cpu_tlb = ctx->irq_cnt[ 1 ];
464 0 : ulong * cpu_loc = ctx->irq_cnt[ 2 ];
465 0 : if( FD_UNLIKELY( -1==lseek( ctx->proc_interrupts_fd, 0, SEEK_SET ) ) ) FD_LOG_ERR(( "lseek failed (%i-%s)", errno, strerror( errno ) ));
466 0 : ulong cpu_cnt = fd_proc_interrupts_read( ctx->proc_interrupts_fd, cpu_irq, cpu_tlb, cpu_loc );
467 0 : if( FD_UNLIKELY( !cpu_cnt ) ) return; /* parse fail */
468 :
469 0 : ulong tot_cnt = 0UL;
470 0 : ulong undesired_cnt = 0UL;
471 0 : for( ulong i=0UL; i<cpu_cnt; i++ ) {
472 0 : ulong since = fd_ulong_sat_sub( cpu_irq[ i ], ctx->device_irq_baseline[ i ] );
473 0 : tot_cnt += since;
474 0 : if( fd_cpuset_test( ctx->cpu_has_tile, i ) ) {
475 0 : undesired_cnt += since;
476 0 : }
477 0 : ulong tile_id = ctx->cpu_to_tile[ i ];
478 0 : if( tile_id!=USHORT_MAX ) {
479 0 : ctx->metrics[ tile_id ][ FD_METRICS_COUNTER_TILE_IRQ_PREEMPTED_OFF ] = since;
480 0 : }
481 0 : }
482 0 : FD_MCNT_SET( DIAG, DEVICE_IRQ, tot_cnt );
483 0 : FD_MCNT_SET( DIAG, DEVICE_IRQ_UNDESIRED, undesired_cnt );
484 :
485 0 : for( ulong i=0UL; i<cpu_cnt; i++ ) {
486 0 : ulong tile_id = ctx->cpu_to_tile[ i ];
487 0 : if( tile_id!=USHORT_MAX ) {
488 0 : ulong since = fd_ulong_sat_sub( cpu_tlb[ i ], ctx->tlb_baseline[ i ] );
489 0 : ctx->metrics[ tile_id ][ FD_METRICS_COUNTER_TILE_TLB_SHOOTDOWN_OFF ] = since;
490 0 : }
491 0 : }
492 :
493 0 : for( ulong i=0UL; i<cpu_cnt; i++ ) {
494 0 : ulong tile_id = ctx->cpu_to_tile[ i ];
495 0 : if( tile_id!=USHORT_MAX ) {
496 0 : ulong since = fd_ulong_sat_sub( cpu_loc[ i ], ctx->loc_baseline[ i ] );
497 0 : ctx->metrics[ tile_id ][ FD_METRICS_COUNTER_TILE_TIMER_TICK_OFF ] = since;
498 0 : }
499 0 : }
500 0 : }
501 :
502 : /* interrupt_metrics reads per-CPU irq+softirq+steal tick counts from
503 : /proc/stat and publishes them as the INTERRUPT CPU regime for
504 : fixed tiles. On kernels with CONFIG_IRQ_TIME_ACCOUNTING (near
505 : universal), these buckets are disjoint from utime/stime so
506 : idle = lifetime - user - system - wait - interrupt
507 : is exact up to sampling granularity; without it, the irq columns
508 : undercount and interrupt reads near zero (degrades gracefully). */
509 :
510 : static void
511 0 : interrupt_metrics( fd_diag_tile_t * ctx ) {
512 0 : ulong cpu_ticks[ FD_TILE_MAX ];
513 0 : if( FD_UNLIKELY( -1==lseek( ctx->proc_stat_fd, 0, SEEK_SET ) ) ) FD_LOG_ERR(( "lseek failed (%i-%s)", errno, strerror( errno ) ));
514 0 : ulong cpu_cnt = fd_proc_stat_irq_ticks( ctx->proc_stat_fd, cpu_ticks );
515 0 : if( FD_UNLIKELY( !cpu_cnt ) ) return; /* parse fail */
516 :
517 0 : for( ulong i=0UL; i<cpu_cnt; i++ ) {
518 0 : ulong tile_id = ctx->cpu_to_tile[ i ];
519 0 : if( tile_id!=USHORT_MAX ) {
520 : /* CLK_TCK is always 100, so 1 tick = 10ms = 10,000,000 ns */
521 0 : ulong since = fd_ulong_sat_sub( cpu_ticks[ i ], ctx->irq_ticks_baseline[ i ] );
522 0 : ctx->metrics[ tile_id ][ FD_METRICS_COUNTER_TILE_CPU_DURATION_NANOS_INTERRUPT_OFF ] = since*10000000UL;
523 0 : }
524 0 : }
525 0 : }
526 :
527 : static ulong
528 : read_text( int fd,
529 : char * buf,
530 0 : ulong buf_sz ) {
531 0 : if( FD_UNLIKELY( -1==lseek( fd, 0, SEEK_SET ) ) ) FD_LOG_ERR(( "lseek failed (%i-%s)", errno, strerror( errno ) ));
532 0 : ulong len;
533 0 : int err = fd_io_read( fd, buf, buf_sz-1UL, buf_sz-1UL, &len );
534 0 : if( FD_UNLIKELY( err>0 ) ) FD_LOG_ERR(( "fd_io_read failed (%i-%s)", err, fd_io_strerror( err ) ));
535 0 : buf[ len ] = '\0';
536 0 : return len;
537 0 : }
538 :
539 : static void
540 : add_numa_bytes( fd_diag_tile_t * ctx,
541 : ulong tile_idx,
542 : ulong numa_slot,
543 0 : ulong bytes ) {
544 0 : if( FD_UNLIKELY( tile_idx==DIAG_WKSP_TILE_IDX_SHARED ) ) {
545 0 : ctx->system_resources.numa_mem[ numa_slot ].shared_bytes += bytes;
546 0 : return;
547 0 : }
548 :
549 0 : ushort * entry_idx = &ctx->numa_entry_idx[ tile_idx ][ numa_slot ];
550 0 : if( FD_UNLIKELY( *entry_idx==USHORT_MAX ) ) {
551 0 : FD_TEST( ctx->system_resources.tile_mem_cnt<FD_DIAG_SYSTEM_TILE_MEM_MAX );
552 0 : *entry_idx = (ushort)ctx->system_resources.tile_mem_cnt++;
553 0 : fd_diag_system_tile_mem_t * entry = &ctx->system_resources.tile_mem[ *entry_idx ];
554 0 : entry->tile_idx = (ushort)tile_idx;
555 0 : entry->numa_idx = ctx->numa.node[ numa_slot ].idx;
556 0 : }
557 0 : ctx->system_resources.tile_mem[ *entry_idx ].allocated_bytes += bytes;
558 0 : }
559 :
560 : static void
561 0 : add_configured_memory_bytes( fd_diag_tile_t * ctx ) {
562 : /* Workspaces and tile stacks are hugetlbfs-backed, so their configured
563 : sizes are also their resident physical-memory footprint. */
564 0 : for( ulong i=0UL; i<ctx->memory.cnt; i++ ) {
565 0 : if( FD_UNLIKELY( ctx->memory.wksp[ i ].numa_slot==USHORT_MAX ) ) continue;
566 0 : add_numa_bytes( ctx,
567 0 : ctx->memory.wksp[ i ].tile_idx,
568 0 : ctx->memory.wksp[ i ].numa_slot,
569 0 : ctx->memory.wksp[ i ].bytes );
570 0 : }
571 0 : for( ulong tile_idx=0UL; tile_idx<ctx->tile_cnt; tile_idx++ ) {
572 0 : ushort numa_slot = ctx->memory.stack_numa_slot[ tile_idx ];
573 0 : if( FD_UNLIKELY( numa_slot==USHORT_MAX ) ) continue;
574 0 : add_numa_bytes( ctx, tile_idx, numa_slot, FD_TILE_PRIVATE_STACK_SZ );
575 0 : }
576 0 : }
577 :
578 : static void
579 0 : sample_disk( fd_diag_tile_t * ctx ) {
580 0 : ctx->system_resources.mount_cnt = (uint)ctx->mount_cnt;
581 0 : ctx->system_resources.file_cnt = (uint)ctx->file_cnt;
582 :
583 0 : for( ulong i=0UL; i<ctx->mount_cnt; i++ ) {
584 0 : struct statfs st;
585 0 : if( FD_UNLIKELY( fstatfs( ctx->mounts[ i ].fd, &st ) ) ) FD_LOG_ERR(( "fstatfs failed (%i-%s)", errno, strerror( errno ) ));
586 0 : ulong block_sz = (ulong)( st.f_frsize ? st.f_frsize : st.f_bsize );
587 0 : fd_diag_system_mount_t * mount = &ctx->system_resources.mount[ i ];
588 0 : fd_cstr_ncpy( mount->path, ctx->mounts[ i ].path, sizeof(mount->path) );
589 0 : mount->total_bytes = (ulong)st.f_blocks*block_sz;
590 0 : mount->free_bytes = (ulong)st.f_bfree *block_sz;
591 0 : mount->available_bytes = (ulong)st.f_bavail*block_sz;
592 0 : }
593 :
594 0 : for( ulong i=0UL; i<ctx->file_cnt; i++ ) {
595 0 : fd_diag_system_file_t * file = &ctx->system_resources.file[ i ];
596 0 : fd_cstr_ncpy( file->path, ctx->files[ i ].path, sizeof(file->path) );
597 0 : file->category = ctx->files[ i ].category;
598 0 : file->mount_idx = ctx->files[ i ].mount_idx;
599 0 : if( ctx->files[ i ].metric ) file->bytes = *ctx->files[ i ].metric;
600 0 : else if( ctx->files[ i ].data_fd>=0 ) {
601 0 : struct stat st;
602 0 : if( FD_UNLIKELY( fstat( ctx->files[ i ].data_fd, &st ) ) ) FD_LOG_ERR(( "fstat failed (%i-%s)", errno, strerror( errno ) ));
603 0 : file->bytes = (ulong)st.st_size;
604 0 : }
605 0 : }
606 0 : }
607 :
608 : static ulong
609 : read_meminfo_kib( char const * buf,
610 0 : char const * key ) {
611 0 : char const * p = strstr( buf, key );
612 0 : if( FD_UNLIKELY( !p ) ) return 0UL;
613 0 : p += strlen( key );
614 0 : while( *p==' ' || *p=='\t' || *p==':' ) p++;
615 0 : return strtoul( p, NULL, 10 );
616 0 : }
617 :
618 : static void
619 : sample_system( fd_diag_tile_t * ctx,
620 0 : long now ) {
621 0 : ctx->next_system_report_nanos = now + SYSTEM_REPORT_INTERVAL_NANOS;
622 :
623 0 : ctx->system_resources.mem_available_bytes = 0UL;
624 0 : ctx->system_resources.mem_free_bytes = 0UL;
625 0 : ctx->system_resources.numa_mem_cnt = (uint)ctx->numa.cnt;
626 0 : ctx->system_resources.tile_mem_cnt = 0U;
627 0 : ctx->system_resources.mount_cnt = 0U;
628 0 : ctx->system_resources.file_cnt = 0U;
629 0 : fd_memset( ctx->system_resources.numa_mem, 0, sizeof(ctx->system_resources.numa_mem) );
630 0 : fd_memset( ctx->system_resources.tile_mem, 0, sizeof(ctx->system_resources.tile_mem) );
631 0 : fd_memset( ctx->system_resources.mount, 0, sizeof(ctx->system_resources.mount) );
632 0 : fd_memset( ctx->system_resources.file, 0, sizeof(ctx->system_resources.file) );
633 0 : fd_memset( ctx->numa_entry_idx, 0xFF, sizeof(ctx->numa_entry_idx) );
634 0 : char meminfo[ 4096 ];
635 0 : if( FD_LIKELY( read_text( ctx->proc_meminfo_fd, meminfo, sizeof(meminfo) ) ) ) {
636 0 : ctx->system_resources.mem_available_bytes = read_meminfo_kib( meminfo, "MemAvailable" )<<10;
637 0 : ctx->system_resources.mem_free_bytes = read_meminfo_kib( meminfo, "MemFree" )<<10;
638 0 : }
639 0 : for( ulong i=0UL; i<ctx->numa.cnt; i++ ) {
640 0 : ulong numa_idx = ctx->numa.node[ i ].idx;
641 0 : fd_diag_system_numa_mem_t * numa = &ctx->system_resources.numa_mem[ i ];
642 0 : numa->numa_idx = (ushort)numa_idx;
643 :
644 0 : if( FD_UNLIKELY( !read_text( ctx->numa.node[ i ].meminfo_fd, meminfo, sizeof(meminfo) ) ) ) continue;
645 0 : char key[ 64 ];
646 0 : FD_TEST( fd_cstr_printf_check( key, sizeof(key), NULL, "Node %lu MemTotal", numa_idx ) );
647 0 : numa->total_bytes = read_meminfo_kib( meminfo, key )<<10;
648 0 : FD_TEST( fd_cstr_printf_check( key, sizeof(key), NULL, "Node %lu MemFree", numa_idx ) );
649 0 : numa->free_bytes = read_meminfo_kib( meminfo, key )<<10;
650 0 : }
651 0 : add_configured_memory_bytes( ctx );
652 0 : sample_disk( ctx );
653 0 : ctx->system_resources.sample_time_nanos = (ulong)now;
654 0 : }
655 :
656 : static void
657 : publish_system( fd_diag_tile_t * ctx,
658 : fd_stem_context_t * stem,
659 0 : long now ) {
660 0 : if( FD_LIKELY( now<ctx->next_system_report_nanos ) ) return;
661 0 : sample_system( ctx, now );
662 :
663 0 : fd_memcpy( fd_chunk_to_laddr( ctx->system_out.mem, ctx->system_out.chunk ),
664 0 : &ctx->system_resources, sizeof(fd_diag_system_resources_t) );
665 0 : fd_stem_publish( stem, ctx->system_out.idx, sizeof(fd_diag_system_resources_t),
666 0 : ctx->system_out.chunk, sizeof(fd_diag_system_resources_t), 0UL, 0UL,
667 0 : (ulong)fd_frag_meta_ts_comp( fd_tickcount() ) );
668 0 : ctx->system_out.chunk = fd_dcache_compact_next( ctx->system_out.chunk,
669 0 : sizeof(fd_diag_system_resources_t), ctx->system_out.chunk0, ctx->system_out.wmark );
670 0 : }
671 :
672 : static void
673 : before_credit( fd_diag_tile_t * ctx,
674 : fd_stem_context_t * stem,
675 0 : int * charge_busy ) {
676 0 : (void)stem;
677 :
678 0 : long now = fd_log_wallclock();
679 0 : if( now<ctx->next_report_nanos ) {
680 0 : long diff = ctx->next_report_nanos - now;
681 0 : diff = fd_long_min( diff, 2e6 /* 2ms */ );
682 0 : struct timespec const ts = {
683 0 : .tv_sec = diff / (long)1e9,
684 0 : .tv_nsec = diff % (long)1e9
685 0 : };
686 0 : clock_nanosleep( CLOCK_REALTIME, 0, &ts, NULL );
687 0 : return;
688 0 : }
689 0 : ctx->next_report_nanos += REPORT_INTERVAL_MILLIS*1000L*1000L;
690 :
691 0 : *charge_busy = 1;
692 0 : if( FD_UNLIKELY( ctx->gui_enabled ) ) publish_system( ctx, stem, now );
693 :
694 0 : struct timespec boottime;
695 0 : if( FD_UNLIKELY( -1==clock_gettime( CLOCK_BOOTTIME, &boottime ) ) ) FD_LOG_ERR(( "clock_gettime(CLOCK_BOOTTIME) failed (%i-%s)", errno, strerror( errno ) ));
696 0 : ulong now_since_boot_nanos = (ulong)boottime.tv_sec*1000000000UL + (ulong)boottime.tv_nsec;
697 :
698 0 : interrupt_metrics( ctx ); /* before idle computation below, which subtracts it */
699 :
700 0 : for( ulong i=0UL; i<ctx->tile_cnt; i++ ) {
701 0 : if( FD_UNLIKELY( -1==ctx->stat_fds[ i ] ) ) continue;
702 :
703 : /* CLK_TCK is typically 100, so 1 tick = 10ms = 10,000,000 ns */
704 0 : int process_died1 = read_stat_file( ctx->stat_fds[ i ], 10000000UL, ctx->metrics[ i ] );
705 0 : int process_died2 = read_sched_file( ctx->sched_fds[ i ], ctx->metrics[ i ] );
706 :
707 0 : if( FD_UNLIKELY( process_died1 || process_died2 ) ) {
708 0 : ctx->stat_fds[ i ] = -1;
709 0 : continue;
710 0 : }
711 :
712 0 : ulong task_lifetime_nanos = now_since_boot_nanos - ctx->starttime_nanos[ i ];
713 0 : ulong user_nanos = ctx->metrics[ i ][ FD_METRICS_COUNTER_TILE_CPU_DURATION_NANOS_USER_OFF ];
714 0 : ulong system_nanos = ctx->metrics[ i ][ FD_METRICS_COUNTER_TILE_CPU_DURATION_NANOS_SYSTEM_OFF ];
715 0 : ulong wait_nanos = ctx->metrics[ i ][ FD_METRICS_COUNTER_TILE_CPU_DURATION_NANOS_WAIT_OFF ];
716 0 : ulong interrupt_nanos = ctx->metrics[ i ][ FD_METRICS_COUNTER_TILE_CPU_DURATION_NANOS_INTERRUPT_OFF ];
717 0 : ulong busy_nanos = user_nanos+system_nanos+wait_nanos+interrupt_nanos;
718 0 : ulong idle_nanos = (task_lifetime_nanos>busy_nanos) ? (task_lifetime_nanos-busy_nanos) : 0UL;
719 :
720 : /* Counter can't go backwards in Prometheus else it thinks the
721 : application restarted. Use max to ensure monotonicity. */
722 0 : ctx->metrics[ i ][ FD_METRICS_COUNTER_TILE_CPU_DURATION_NANOS_IDLE_OFF ] = fd_ulong_max( idle_nanos, ctx->metrics[ i ][ FD_METRICS_COUNTER_TILE_CPU_DURATION_NANOS_IDLE_OFF ] );
723 0 : }
724 :
725 0 : for( ulong i=0UL; i<ctx->tile_cnt; i++ ) {
726 0 : if( FD_LIKELY( -1!=ctx->stat_fds[ i ] ) ) continue;
727 :
728 : /* The tile died, but it's a tile which is allowed to shutdown, so
729 : just stop updating metrics for it. */
730 0 : if( FD_LIKELY( 2UL==ctx->metrics[ i ][ FD_METRICS_GAUGE_TILE_STATUS_OFF ] ) ) continue;
731 :
732 : /* Supervisor is going to bring the whole process tree down if any
733 : of the target PIDs died, so we can ignore this and wait. */
734 0 : if( FD_UNLIKELY( !ctx->first_seen_died[ i ] ) ) {
735 0 : ctx->first_seen_died[ i ] = now;
736 0 : } else if( FD_LIKELY( ctx->first_seen_died[ i ]==LONG_MAX ) ) {
737 : /* We already reported this, so we can ignore it. */
738 0 : } else if( FD_UNLIKELY( now-ctx->first_seen_died[ i ] < 10L*1000L*1000L*1000L ) ) {
739 : /* Wait 10 seconds for supervisor to kill us before reporting WARNING */
740 0 : } else {
741 0 : FD_LOG_WARNING(( "cannot get metrics for dead tile idx %lu", i ));
742 0 : ctx->first_seen_died[ i ] = LONG_MAX;
743 0 : }
744 0 : }
745 :
746 0 : check_engine_metric( ctx, now );
747 0 : irq_metrics( ctx );
748 0 : }
749 :
750 : /* Disk mount discovery ************************************************/
751 :
752 : static int
753 : fd_mount_id( int fd,
754 0 : ulong * mnt_id ) {
755 0 : char fdinfo_path[ 64 ];
756 0 : if( FD_UNLIKELY( !fd_cstr_printf_check( fdinfo_path, sizeof(fdinfo_path), NULL, "/proc/self/fdinfo/%d", fd ) ) ) return EINVAL;
757 :
758 0 : int fdinfo_fd = open( fdinfo_path, O_RDONLY|O_CLOEXEC );
759 0 : if( FD_UNLIKELY( fdinfo_fd<0 ) ) return errno;
760 :
761 0 : char buf[ 256 ];
762 0 : ulong buf_sz = 0UL;
763 0 : int err = fd_io_read( fdinfo_fd, buf, sizeof(buf)-1UL, sizeof(buf)-1UL, &buf_sz );
764 0 : if( FD_UNLIKELY( err<0 ) ) err = 0; /* EOF before the buffer filled */
765 0 : if( FD_UNLIKELY( close( fdinfo_fd ) && !err ) ) err = errno;
766 0 : if( FD_UNLIKELY( err ) ) return err;
767 0 : buf[ buf_sz ] = '\0';
768 :
769 0 : char const * line = buf;
770 0 : while( line ) {
771 0 : if( FD_UNLIKELY( !strncmp( line, "mnt_id:", 7UL ) ) ) {
772 0 : char * end;
773 0 : ulong id = strtoul( line+7UL, &end, 10 );
774 0 : if( FD_UNLIKELY( end==line+7UL || (*end!='\n' && *end!='\0') || id==ULONG_MAX ) ) return EINVAL;
775 0 : *mnt_id = id;
776 0 : return 0;
777 0 : }
778 0 : line = strchr( line, '\n' );
779 0 : if( line ) line++;
780 0 : }
781 0 : return ENOENT;
782 0 : }
783 :
784 : static int
785 : resolve_path_prefix( char const * path,
786 0 : char resolved[ static PATH_MAX ] ) {
787 0 : char candidate[ PATH_MAX ];
788 0 : fd_cstr_ncpy( candidate, path, sizeof(candidate) );
789 0 : while( !realpath( candidate, resolved ) ) {
790 0 : char * slash = strrchr( candidate, '/' );
791 0 : if( FD_UNLIKELY( !slash ) ) return 0;
792 0 : if( slash==candidate ) slash[ 1 ] = '\0';
793 0 : else *slash = '\0';
794 0 : }
795 0 : return 1;
796 0 : }
797 :
798 : static int
799 : open_disk_probe_dir( char const * resolved,
800 : char probe_dir[ static PATH_MAX ],
801 0 : ulong * mnt_id ) {
802 : /* `probe_dir` is on the target mount used only to create the probe
803 : file; the descriptor for `probe_dir` is not retained. */
804 0 : int path_fd = open( resolved, O_PATH|O_CLOEXEC );
805 0 : if( FD_UNLIKELY( path_fd<0 ) ) return -1;
806 :
807 0 : struct stat path_st;
808 0 : if( FD_UNLIKELY( fstat( path_fd, &path_st ) ) ) {
809 0 : close( path_fd );
810 0 : return -1;
811 0 : }
812 :
813 0 : ulong path_mnt_id;
814 0 : int err = fd_mount_id( path_fd, &path_mnt_id );
815 0 : if( FD_UNLIKELY( err ) ) {
816 0 : close( path_fd );
817 0 : FD_LOG_WARNING(( "reading mount ID for `%s` failed (%i-%s); omitting filesystem from system resource reporting",
818 0 : resolved, err, strerror( err ) ));
819 0 : return -1;
820 0 : }
821 :
822 0 : fd_cstr_ncpy( probe_dir, resolved, PATH_MAX );
823 0 : if( FD_UNLIKELY( !S_ISDIR( path_st.st_mode ) ) ) {
824 0 : char * slash = strrchr( probe_dir, '/' );
825 0 : if( FD_UNLIKELY( !slash ) ) {
826 0 : close( path_fd );
827 0 : return -1;
828 0 : }
829 0 : if( slash==probe_dir ) slash[ 1 ] = '\0';
830 0 : else *slash = '\0';
831 0 : }
832 :
833 0 : int dir_fd = open( probe_dir, O_PATH|O_DIRECTORY|O_CLOEXEC );
834 0 : if( FD_UNLIKELY( dir_fd<0 ) ) {
835 0 : int open_err = errno;
836 0 : close( path_fd );
837 0 : FD_LOG_WARNING(( "open `%s` for disk-capacity probe failed (%i-%s); omitting filesystem from system resource reporting",
838 0 : probe_dir, open_err, strerror( open_err ) ));
839 0 : return -1;
840 0 : }
841 :
842 0 : ulong dir_mnt_id;
843 0 : err = fd_mount_id( dir_fd, &dir_mnt_id );
844 0 : if( FD_UNLIKELY( err ) ) {
845 0 : close( path_fd );
846 0 : close( dir_fd );
847 0 : FD_LOG_WARNING(( "reading mount ID for `%s` failed (%i-%s); omitting filesystem from system resource reporting",
848 0 : probe_dir, err, strerror( err ) ));
849 0 : return -1;
850 0 : }
851 0 : close( path_fd );
852 :
853 : /* A regular file can itself be a bind mount. In that case its
854 : containing directory is not on the mount we need to sample,
855 : and there is nowhere on that mount to create an anonymous file. */
856 0 : if( FD_UNLIKELY( path_mnt_id!=dir_mnt_id ) ) {
857 0 : close( dir_fd );
858 0 : FD_LOG_WARNING(( "cannot create anonymous disk-capacity probe for non-directory mount `%s`; omitting filesystem from system resource reporting",
859 0 : resolved ));
860 0 : return -1;
861 0 : }
862 :
863 0 : *mnt_id = dir_mnt_id;
864 0 : return dir_fd;
865 0 : }
866 :
867 : static int
868 : resolve_mount_path( char const * probe_dir,
869 : ulong mnt_id,
870 0 : char mount_path[ static PATH_MAX ] ) {
871 0 : fd_cstr_ncpy( mount_path, probe_dir, PATH_MAX );
872 0 : while( strcmp( mount_path, "/" ) ) {
873 0 : char parent[ PATH_MAX ];
874 0 : fd_cstr_ncpy( parent, mount_path, sizeof(parent) );
875 0 : char * slash = strrchr( parent, '/' );
876 0 : if( slash==parent ) slash[ 1 ] = '\0';
877 0 : else *slash = '\0';
878 :
879 0 : int parent_fd = open( parent, O_PATH|O_DIRECTORY|O_CLOEXEC );
880 0 : if( FD_UNLIKELY( parent_fd<0 ) ) {
881 0 : int open_err = errno;
882 0 : FD_LOG_WARNING(( "open `%s` while resolving mount point failed (%i-%s); omitting filesystem from system resource reporting",
883 0 : parent, open_err, strerror( open_err ) ));
884 0 : return 0;
885 0 : }
886 0 : ulong parent_mnt_id;
887 0 : int err = fd_mount_id( parent_fd, &parent_mnt_id );
888 0 : close( parent_fd );
889 0 : if( FD_UNLIKELY( err ) ) {
890 0 : FD_LOG_WARNING(( "reading mount ID for `%s` failed (%i-%s); omitting filesystem from system resource reporting",
891 0 : parent, err, strerror( err ) ));
892 0 : return 0;
893 0 : }
894 0 : if( parent_mnt_id!=mnt_id ) break;
895 0 : fd_cstr_ncpy( mount_path, parent, PATH_MAX );
896 0 : }
897 0 : return 1;
898 0 : }
899 :
900 : static int
901 : create_disk_probe( int dir_fd,
902 : char const * probe_dir,
903 0 : ulong mnt_id ) {
904 : /* Retain no directory fds after sandboxing. O_EXCL ensures
905 : that the anonymous inode cannot later be linked into the filesystem;
906 : reopening it O_PATH also removes the temporary read/write authority. */
907 0 : int tmp_fd = openat( dir_fd, ".", O_TMPFILE|O_EXCL|O_RDWR|O_CLOEXEC, S_IRUSR|S_IWUSR );
908 0 : if( FD_UNLIKELY( tmp_fd<0 ) ) {
909 0 : int err = errno;
910 0 : FD_LOG_WARNING(( "anonymous disk-capacity probe in `%s` failed (%i-%s); omitting filesystem from system resource reporting",
911 0 : probe_dir, err, strerror( err ) ));
912 0 : return -1;
913 0 : }
914 :
915 0 : char tmp_path[ 64 ];
916 0 : FD_TEST( fd_cstr_printf_check( tmp_path, sizeof(tmp_path), NULL, "/proc/self/fd/%d", tmp_fd ) );
917 0 : int mount_fd = open( tmp_path, O_PATH|O_CLOEXEC );
918 0 : if( FD_UNLIKELY( mount_fd<0 ) ) {
919 0 : int err = errno;
920 0 : close( tmp_fd );
921 0 : FD_LOG_WARNING(( "reopening anonymous disk-capacity probe in `%s` as O_PATH failed (%i-%s); omitting filesystem from system resource reporting",
922 0 : probe_dir, err, strerror( err ) ));
923 0 : return -1;
924 0 : }
925 :
926 0 : struct stat probe_st;
927 0 : ulong probe_mnt_id;
928 0 : int probe_err = 0;
929 0 : if( FD_UNLIKELY( fstat( mount_fd, &probe_st ) ) ) probe_err = errno;
930 0 : else if( FD_UNLIKELY( !S_ISREG( probe_st.st_mode ) || probe_st.st_nlink ) ) probe_err = EINVAL;
931 0 : else probe_err = fd_mount_id( mount_fd, &probe_mnt_id );
932 0 : if( FD_UNLIKELY( !probe_err && probe_mnt_id!=mnt_id ) ) probe_err = EXDEV;
933 0 : if( FD_UNLIKELY( probe_err ) ) {
934 0 : close( mount_fd );
935 0 : close( tmp_fd );
936 0 : FD_LOG_WARNING(( "anonymous disk-capacity probe in `%s` failed validation (%i-%s); omitting filesystem from system resource reporting",
937 0 : probe_dir, probe_err, strerror( probe_err ) ));
938 0 : return -1;
939 0 : }
940 0 : close( tmp_fd );
941 0 : return mount_fd;
942 0 : }
943 :
944 : static uint
945 : add_disk_mount( fd_diag_tile_t * ctx,
946 0 : char const * path ) {
947 0 : if( FD_UNLIKELY( !path[ 0 ] ) ) return UINT_MAX;
948 :
949 : /* Find longest prefix of `path` that is a real path. */
950 0 : char resolved[ PATH_MAX ];
951 0 : if( FD_UNLIKELY( !resolve_path_prefix( path, resolved ) ) ) return UINT_MAX;
952 :
953 : /* Find candidate dir at or above `resolved` where O_TMPFILE can create a probe file. */
954 0 : char probe_dir[ PATH_MAX ];
955 0 : ulong mnt_id;
956 0 : int dir_fd = open_disk_probe_dir( resolved, probe_dir, &mnt_id );
957 0 : if( FD_UNLIKELY( dir_fd<0 ) ) return UINT_MAX;
958 :
959 : /* Share one probe file among all reported paths on the same mount. */
960 0 : for( ulong i=0UL; i<ctx->mount_cnt; i++ ) {
961 0 : if( ctx->mounts[ i ].mnt_id==mnt_id ) {
962 0 : close( dir_fd );
963 0 : return (uint)i;
964 0 : }
965 0 : }
966 0 : if( FD_UNLIKELY( ctx->mount_cnt>=FD_DIAG_SYSTEM_FILE_MAX ) ) {
967 0 : close( dir_fd );
968 0 : return UINT_MAX;
969 0 : }
970 :
971 : /* Get the path of the mount `path` is on. */
972 0 : char mount_path[ PATH_MAX ];
973 0 : if( FD_UNLIKELY( !resolve_mount_path( probe_dir, mnt_id, mount_path ) ) ) {
974 0 : close( dir_fd );
975 0 : return UINT_MAX;
976 0 : }
977 :
978 : /* Create the probe O_PATH descriptor kept after sandboxing. */
979 0 : int mount_fd = create_disk_probe( dir_fd, probe_dir, mnt_id );
980 0 : close( dir_fd );
981 0 : if( FD_UNLIKELY( mount_fd<0 ) ) return UINT_MAX;
982 :
983 0 : ulong idx = ctx->mount_cnt++;
984 0 : ctx->mounts[ idx ].fd = mount_fd;
985 0 : ctx->mounts[ idx ].mnt_id = mnt_id;
986 0 : fd_cstr_ncpy( ctx->mounts[ idx ].path, mount_path, sizeof(ctx->mounts[ idx ].path) );
987 0 : return (uint)idx;
988 0 : }
989 :
990 : static int
991 : resolve_file_path( char const * path,
992 : int data_fd,
993 0 : char resolved[ static PATH_MAX ] ) {
994 0 : char fd_path[ 64 ];
995 0 : if( FD_LIKELY( data_fd>=0 ) ) {
996 0 : struct stat st;
997 0 : if( FD_UNLIKELY( fstat( data_fd, &st ) || !S_ISREG( st.st_mode ) ) ) return 0;
998 0 : FD_TEST( fd_cstr_printf_check( fd_path, sizeof(fd_path), NULL, "/proc/self/fd/%d", data_fd ) );
999 0 : path = fd_path;
1000 0 : }
1001 0 : if( FD_UNLIKELY( !path || !path[ 0 ] ) ) return 0;
1002 :
1003 0 : if( FD_LIKELY( realpath( path, resolved ) ) ) return 1;
1004 0 : if( FD_LIKELY( path[ 0 ]=='/' ) ) {
1005 0 : if( FD_UNLIKELY( strlen( path )>=PATH_MAX ) ) return 0;
1006 0 : fd_cstr_ncpy( resolved, path, PATH_MAX );
1007 0 : return 1;
1008 0 : }
1009 :
1010 0 : char cwd[ PATH_MAX ];
1011 0 : if( FD_UNLIKELY( !getcwd( cwd, sizeof(cwd) ) ) ) return 0;
1012 0 : return fd_cstr_printf_check( resolved, PATH_MAX, NULL, "%s/%s", cwd, path );
1013 0 : }
1014 :
1015 : static void
1016 : add_file( fd_diag_tile_t * ctx,
1017 : uint category,
1018 : char const * path,
1019 : int data_fd,
1020 0 : ulong volatile * metric ) {
1021 0 : char resolved[ PATH_MAX ];
1022 0 : if( FD_UNLIKELY( !resolve_file_path( path, data_fd, resolved ) ) ) return;
1023 :
1024 0 : uint mount_idx = add_disk_mount( ctx, resolved );
1025 0 : if( FD_UNLIKELY( mount_idx==UINT_MAX || ctx->file_cnt>=FD_DIAG_SYSTEM_FILE_MAX ) ) return;
1026 0 : ulong idx = ctx->file_cnt++;
1027 0 : fd_cstr_ncpy( ctx->files[ idx ].path, resolved, sizeof(ctx->files[ idx ].path) );
1028 0 : ctx->files[ idx ].category = category;
1029 0 : ctx->files[ idx ].mount_idx = mount_idx;
1030 0 : ctx->files[ idx ].data_fd = data_fd;
1031 0 : ctx->files[ idx ].metric = metric;
1032 0 : }
1033 :
1034 : static void
1035 : privileged_init( fd_topo_t const * topo,
1036 0 : fd_topo_tile_t const * tile ) {
1037 0 : void * scratch = fd_topo_obj_laddr( topo, tile->tile_obj_id );
1038 :
1039 0 : FD_SCRATCH_ALLOC_INIT( l, scratch );
1040 0 : fd_diag_tile_t * ctx = FD_SCRATCH_ALLOC_APPEND( l, alignof(fd_diag_tile_t), sizeof(fd_diag_tile_t) );
1041 :
1042 0 : FD_TEST( topo->tile_cnt<=FD_DIAG_SYSTEM_TILE_MAX );
1043 :
1044 0 : FD_TEST( 100L == sysconf( _SC_CLK_TCK ) );
1045 :
1046 0 : ctx->tile_cnt = topo->tile_cnt;
1047 0 : ctx->gui_enabled = fd_topo_find_tile_out_link( topo, tile, "diag_gui", 0UL )!=ULONG_MAX;
1048 0 : ctx->memory.cnt = topo->wksp_cnt;
1049 0 : FD_TEST( ctx->memory.cnt<=FD_TOPO_MAX_WKSPS );
1050 0 : for( ulong wksp_idx=0UL; wksp_idx<ctx->memory.cnt; wksp_idx++ ) {
1051 0 : fd_topo_wksp_t const * wksp = &topo->workspaces[ wksp_idx ];
1052 0 : FD_TEST( fd_cstr_printf_check( ctx->memory.wksp[ wksp_idx ].name,
1053 0 : sizeof(ctx->memory.wksp[ wksp_idx ].name), NULL,
1054 0 : "%s_%s.wksp", topo->app_name, wksp->name ) );
1055 0 : ctx->memory.wksp[ wksp_idx ].numa_idx = wksp->numa_idx;
1056 0 : ctx->memory.wksp[ wksp_idx ].bytes = wksp->page_cnt*wksp->page_sz;
1057 0 : ctx->memory.wksp[ wksp_idx ].numa_slot = USHORT_MAX;
1058 :
1059 0 : ushort owners[ FD_DIAG_SYSTEM_TILE_MAX ];
1060 0 : ulong owner_cnt = 0UL;
1061 0 : for( ulong tile_idx=0UL; tile_idx<topo->tile_cnt; tile_idx++ ) {
1062 0 : fd_topo_tile_t const * owner = &topo->tiles[ tile_idx ];
1063 0 : if( topo->objs[ owner->tile_obj_id ].wksp_id!=wksp_idx ) continue;
1064 0 : owners[ owner_cnt++ ] = (ushort)tile_idx;
1065 0 : }
1066 0 : if( FD_LIKELY( owner_cnt==1UL ) ) {
1067 0 : ctx->memory.wksp[ wksp_idx ].tile_idx = owners[ 0 ];
1068 0 : continue;
1069 0 : }
1070 0 : if( FD_UNLIKELY( owner_cnt>1UL ) ) {
1071 0 : ctx->memory.wksp[ wksp_idx ].tile_idx = DIAG_WKSP_TILE_IDX_SHARED;
1072 0 : continue;
1073 0 : }
1074 :
1075 0 : owner_cnt = 0UL;
1076 0 : for( ulong link_idx=0UL; link_idx<topo->link_cnt; link_idx++ ) {
1077 0 : fd_topo_link_t const * link = &topo->links[ link_idx ];
1078 0 : if( topo->objs[ link->mcache_obj_id ].wksp_id!=wksp_idx &&
1079 0 : ( !link->mtu || topo->objs[ link->dcache_obj_id ].wksp_id!=wksp_idx ) ) continue;
1080 0 : ulong producer_idx = fd_topo_find_link_producer( topo, link );
1081 0 : if( producer_idx==ULONG_MAX ) continue;
1082 0 : int found = 0;
1083 0 : for( ulong owner_idx=0UL; owner_idx<owner_cnt; owner_idx++ )
1084 0 : found |= owners[ owner_idx ]==(ushort)producer_idx;
1085 0 : if( !found ) owners[ owner_cnt++ ] = (ushort)producer_idx;
1086 0 : }
1087 0 : if( FD_LIKELY( owner_cnt==1UL ) ) {
1088 0 : ctx->memory.wksp[ wksp_idx ].tile_idx = owners[ 0 ];
1089 0 : continue;
1090 0 : }
1091 0 : if( FD_UNLIKELY( owner_cnt>1UL ) ) {
1092 0 : ctx->memory.wksp[ wksp_idx ].tile_idx = DIAG_WKSP_TILE_IDX_SHARED;
1093 0 : continue;
1094 0 : }
1095 :
1096 0 : for( ulong tile_idx=0UL; tile_idx<topo->tile_cnt; tile_idx++ ) {
1097 0 : fd_topo_tile_t const * owner = &topo->tiles[ tile_idx ];
1098 0 : int uses_writable = 0;
1099 0 : for( ulong obj_idx=0UL; obj_idx<owner->uses_obj_cnt; obj_idx++ ) {
1100 0 : ulong obj_id = owner->uses_obj_id[ obj_idx ];
1101 0 : uses_writable |= topo->objs[ obj_id ].wksp_id==wksp_idx &&
1102 0 : owner->uses_obj_mode[ obj_idx ]==FD_SHMEM_JOIN_MODE_READ_WRITE;
1103 0 : }
1104 0 : if( uses_writable ) owners[ owner_cnt++ ] = (ushort)tile_idx;
1105 0 : }
1106 0 : ctx->memory.wksp[ wksp_idx ].tile_idx = owner_cnt==1UL ? owners[ 0 ] : DIAG_WKSP_TILE_IDX_SHARED;
1107 0 : }
1108 0 : for( ulong tile_idx=0UL; tile_idx<FD_DIAG_SYSTEM_TILE_MAX; tile_idx++ )
1109 0 : ctx->memory.stack_numa_slot[ tile_idx ] = USHORT_MAX;
1110 0 : ctx->mount_cnt = 0UL;
1111 0 : ctx->file_cnt = 0UL;
1112 0 : for( ulong i=0UL; i<FD_TILE_MAX; i++ ) {
1113 0 : ctx->stat_fds[ i ] = -1;
1114 0 : ctx->sched_fds[ i ] = -1;
1115 0 : }
1116 0 : for( ulong i=0UL; i<FD_DIAG_SYSTEM_NUMA_MAX; i++ ) ctx->numa.node[ i ].meminfo_fd = -1;
1117 :
1118 0 : for( ulong i=0UL; i<topo->tile_cnt; i++ ) {
1119 0 : ulong * metrics = fd_metrics_join( fd_topo_obj_laddr( topo, topo->tiles[ i ].metrics_obj_id ) );
1120 :
1121 0 : for(;;) {
1122 0 : ulong pid, tid;
1123 0 : if( FD_UNLIKELY( tile->id==i ) ) {
1124 0 : pid = fd_sandbox_getpid();
1125 0 : tid = fd_sandbox_gettid();
1126 0 : } else {
1127 0 : pid = fd_metrics_tile( metrics )[ FD_METRICS_GAUGE_TILE_PID_OFF ];
1128 0 : tid = fd_metrics_tile( metrics )[ FD_METRICS_GAUGE_TILE_TID_OFF ];
1129 0 : if( FD_UNLIKELY( !pid || !tid ) ) {
1130 0 : FD_SPIN_PAUSE();
1131 0 : continue;
1132 0 : }
1133 0 : }
1134 :
1135 0 : ctx->metrics[ i ] = fd_metrics_tile( metrics );
1136 :
1137 0 : char path[ 64UL ];
1138 0 : FD_TEST( fd_cstr_printf_check( path, sizeof( path ), NULL, "/proc/%lu/task/%lu/stat", pid, tid ) );
1139 0 : ctx->stat_fds[ i ] = open( path, O_RDONLY );
1140 0 : if( FD_UNLIKELY( -1==ctx->stat_fds[ i ] ) ) {
1141 : /* Might be a tile that's allowed to shutdown already did so
1142 : before we got to here, due to a race condition. Just
1143 : proceed, we will not be able to get metrics for the shut
1144 : down process. */
1145 0 : if( FD_LIKELY( 2UL!=ctx->metrics[ i ][ FD_METRICS_GAUGE_TILE_STATUS_OFF ] ) ) FD_LOG_ERR(( "open stat failed (%i-%s)", errno, strerror( errno ) ));
1146 0 : break;
1147 0 : }
1148 :
1149 0 : FD_TEST( fd_cstr_printf_check( path, sizeof( path ), NULL, "/proc/%lu/task/%lu/sched", pid, tid ) );
1150 0 : ctx->sched_fds[ i ] = open( path, O_RDONLY );
1151 0 : if( FD_UNLIKELY( -1==ctx->sched_fds[ i ] ) ) {
1152 0 : if( FD_LIKELY( 2UL!=ctx->metrics[ i ][ FD_METRICS_GAUGE_TILE_STATUS_OFF ] ) ) FD_LOG_ERR(( "open sched failed (%i-%s)", errno, strerror( errno ) ));
1153 0 : ctx->stat_fds[ i ] = -1;
1154 0 : }
1155 0 : break;
1156 0 : }
1157 0 : }
1158 :
1159 0 : ctx->proc_interrupts_fd = open( "/proc/interrupts", O_RDONLY );
1160 0 : if( FD_UNLIKELY( -1==ctx->proc_interrupts_fd ) ) FD_LOG_ERR(( "open(/proc/interrupts) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1161 :
1162 0 : ctx->proc_softirqs_fd = open( "/proc/softirqs", O_RDONLY );
1163 0 : if( FD_UNLIKELY( -1==ctx->proc_softirqs_fd ) ) FD_LOG_ERR(( "open(/proc/softirqs) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1164 :
1165 0 : ctx->proc_stat_fd = open( "/proc/stat", O_RDONLY );
1166 0 : if( FD_UNLIKELY( -1==ctx->proc_stat_fd ) ) FD_LOG_ERR(( "open(/proc/stat) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1167 :
1168 0 : ctx->proc_meminfo_fd = open( "/proc/meminfo", O_RDONLY );
1169 0 : if( FD_UNLIKELY( -1==ctx->proc_meminfo_fd ) ) FD_LOG_ERR(( "open(/proc/meminfo) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1170 :
1171 0 : ctx->numa.cnt = 0UL;
1172 0 : if( FD_UNLIKELY( !ctx->gui_enabled ) ) return;
1173 :
1174 0 : fd_topo_cpus_t cpus[ 1 ];
1175 0 : fd_topo_cpus_init( cpus );
1176 0 : ctx->system_resources.cpu_cnt = (uint)fd_ulong_min( cpus->cpu_cnt, FD_DIAG_SYSTEM_CPU_MAX );
1177 0 : for( ulong i=0UL; i<ctx->system_resources.cpu_cnt; i++ ) {
1178 0 : fd_diag_system_cpu_t * cpu = &ctx->system_resources.cpu[ i ];
1179 0 : cpu->cpu_idx = (ushort)i;
1180 0 : cpu->numa_idx = (ushort)cpus->cpu[ i ].numa_node;
1181 0 : cpu->sibling_idx = cpus->cpu[ i ].sibling==ULONG_MAX ? USHORT_MAX : (ushort)cpus->cpu[ i ].sibling;
1182 0 : cpu->online = (uchar)cpus->cpu[ i ].online;
1183 0 : }
1184 :
1185 0 : for( ulong numa_idx=0UL; numa_idx<cpus->numa_node_cnt && ctx->numa.cnt<FD_DIAG_SYSTEM_NUMA_MAX; numa_idx++ ) {
1186 0 : char path[ 128 ];
1187 0 : FD_TEST( fd_cstr_printf_check( path, sizeof(path), NULL, "/sys/devices/system/node/node%lu/meminfo", numa_idx ) );
1188 0 : int fd = open( path, O_RDONLY );
1189 0 : if( FD_UNLIKELY( fd<0 ) ) {
1190 0 : if( FD_UNLIKELY( errno!=ENOENT ) ) FD_LOG_WARNING(( "open `%s` failed (%i-%s)", path, errno, strerror( errno ) ));
1191 0 : continue;
1192 0 : }
1193 0 : ctx->numa.node[ ctx->numa.cnt ].idx = (ushort)numa_idx;
1194 0 : ctx->numa.node[ ctx->numa.cnt ].meminfo_fd = fd;
1195 0 : ctx->numa.cnt++;
1196 0 : }
1197 :
1198 0 : for( ulong wksp_idx=0UL; wksp_idx<ctx->memory.cnt; wksp_idx++ ) {
1199 0 : ulong numa_slot = 0UL;
1200 0 : while( numa_slot<ctx->numa.cnt && ctx->numa.node[ numa_slot ].idx!=ctx->memory.wksp[ wksp_idx ].numa_idx ) numa_slot++;
1201 0 : if( FD_UNLIKELY( numa_slot==ctx->numa.cnt ) ) {
1202 0 : FD_LOG_WARNING(( "workspace `%s` is assigned to unavailable NUMA node %lu; omitting it from system memory reporting",
1203 0 : ctx->memory.wksp[ wksp_idx ].name, ctx->memory.wksp[ wksp_idx ].numa_idx ));
1204 0 : continue;
1205 0 : }
1206 0 : ctx->memory.wksp[ wksp_idx ].numa_slot = (ushort)numa_slot;
1207 0 : }
1208 :
1209 0 : for( ulong tile_idx=0UL; tile_idx<topo->tile_cnt; tile_idx++ ) {
1210 : /* Keep this placement rule in sync with initialize_stacks(). */
1211 0 : ulong stack_cpu_idx = topo->tiles[ tile_idx ].cpu_idx<65535UL ? topo->tiles[ tile_idx ].cpu_idx : 0UL;
1212 0 : FD_TEST( stack_cpu_idx<cpus->cpu_cnt );
1213 0 : ulong stack_numa_idx = cpus->cpu[ stack_cpu_idx ].numa_node;
1214 0 : ulong numa_slot = 0UL;
1215 0 : while( numa_slot<ctx->numa.cnt && ctx->numa.node[ numa_slot ].idx!=stack_numa_idx ) numa_slot++;
1216 0 : if( FD_UNLIKELY( numa_slot==ctx->numa.cnt ) ) {
1217 0 : FD_LOG_WARNING(( "stack for tile %s:%lu is assigned to unavailable NUMA node %lu; omitting it from system memory reporting",
1218 0 : topo->tiles[ tile_idx ].name, topo->tiles[ tile_idx ].kind_id, stack_numa_idx ));
1219 0 : continue;
1220 0 : }
1221 0 : ctx->memory.stack_numa_slot[ tile_idx ] = (ushort)numa_slot;
1222 0 : }
1223 :
1224 0 : ulong accdb_idx = fd_topo_find_tile( topo, "accdb", 0UL );
1225 0 : if( tile->diag.accounts_path[ 0 ] && accdb_idx!=ULONG_MAX )
1226 0 : add_file( ctx, FD_DIAG_SYSTEM_FILE_CATEGORY_ACCOUNTS, tile->diag.accounts_path, -1, ctx->metrics[ accdb_idx ] + FD_METRICS_GAUGE_ACCDB_DISK_ALLOCATED_BYTES_OFF );
1227 0 : ulong rserve_idx = fd_topo_find_tile( topo, "rserve", 0UL );
1228 0 : if( tile->diag.shreds_path[ 0 ] && rserve_idx!=ULONG_MAX )
1229 0 : add_file( ctx, FD_DIAG_SYSTEM_FILE_CATEGORY_SHREDS, tile->diag.shreds_path, -1, ctx->metrics[ rserve_idx ] + FD_METRICS_GAUGE_RSERVE_DISK_ALLOCATED_BYTES_OFF );
1230 0 : ulong snapmk_idx = fd_topo_find_tile( topo, "snapmk", 0UL );
1231 0 : if( tile->diag.snapshots_path[ 0 ] && snapmk_idx!=ULONG_MAX )
1232 0 : add_file( ctx, FD_DIAG_SYSTEM_FILE_CATEGORY_SNAPSHOTS, tile->diag.snapshots_path, -1, ctx->metrics[ snapmk_idx ] + FD_METRICS_GAUGE_SNAPMK_DISK_ALLOCATED_BYTES_OFF );
1233 0 : ulong gui_idx = fd_topo_find_tile( topo, "gui", 0UL );
1234 0 : if( gui_idx!=ULONG_MAX )
1235 0 : add_file( ctx, FD_DIAG_SYSTEM_FILE_CATEGORY_GUI, tile->diag.gui_path, -1, ctx->metrics[ gui_idx ] + FD_METRICS_GAUGE_GUI_DISK_ALLOCATED_BYTES_OFF );
1236 0 : int logfile_fd = fd_log_private_logfile_fd();
1237 0 : if( logfile_fd>=0 ) add_file( ctx, FD_DIAG_SYSTEM_FILE_CATEGORY_LOGS, tile->diag.log_path, logfile_fd, NULL );
1238 0 : }
1239 :
1240 : /* Read starttime (field 22) from stat file. Returns 0 on success, 1 if
1241 : process died (ESRCH). */
1242 :
1243 : static int
1244 : read_starttime( int fd,
1245 : ulong ns_per_tick,
1246 0 : ulong * out_starttime_nanos ) {
1247 0 : char contents[ 4096 ] = {0};
1248 0 : ulong contents_len = 0UL;
1249 :
1250 0 : while( 1 ) {
1251 0 : if( FD_UNLIKELY( contents_len>=sizeof( contents ) ) ) FD_LOG_ERR(( "stat contents overflow" ));
1252 0 : long n = read( fd, contents + contents_len, sizeof( contents ) - contents_len );
1253 0 : if( FD_UNLIKELY( -1==n ) ) {
1254 0 : if( FD_UNLIKELY( errno==ESRCH ) ) return 1;
1255 0 : FD_LOG_ERR(( "read stat failed (%i-%s)", errno, strerror( errno ) ));
1256 0 : }
1257 0 : if( FD_LIKELY( 0L==n ) ) break;
1258 0 : contents_len += (ulong)n;
1259 0 : }
1260 :
1261 : /* Parse field 22 (starttime) from stat file */
1262 0 : char * saveptr;
1263 0 : char * token = strtok_r( contents, " ", &saveptr );
1264 0 : ulong field_idx = 0UL;
1265 :
1266 0 : while( token && field_idx<21UL ) {
1267 0 : token = strtok_r( NULL, " ", &saveptr );
1268 0 : field_idx++;
1269 0 : }
1270 :
1271 0 : if( FD_UNLIKELY( !token || field_idx!=21UL ) ) FD_LOG_ERR(( "starttime (field 22) not found in stat" ));
1272 :
1273 0 : char * endptr;
1274 0 : ulong starttime_ticks = strtoul( token, &endptr, 10 );
1275 0 : if( FD_UNLIKELY( *endptr!=' ' && *endptr!='\0' ) ) FD_LOG_ERR(( "strtoul failed for starttime" ));
1276 0 : if( FD_UNLIKELY( starttime_ticks==ULONG_MAX ) ) FD_LOG_ERR(( "strtoul overflow for starttime" ));
1277 :
1278 0 : *out_starttime_nanos = starttime_ticks * ns_per_tick;
1279 0 : return 0;
1280 0 : }
1281 :
1282 : static void
1283 : unprivileged_init( fd_topo_t const * topo,
1284 0 : fd_topo_tile_t const * tile ) {
1285 0 : fd_diag_tile_t * ctx = fd_topo_obj_laddr( topo, tile->tile_obj_id );
1286 :
1287 0 : memset( ctx->first_seen_died, 0, sizeof( ctx->first_seen_died ) );
1288 0 : ctx->next_report_nanos = fd_log_wallclock();
1289 0 : ctx->next_system_report_nanos = ctx->next_report_nanos;
1290 0 : if( FD_UNLIKELY( ctx->gui_enabled ) ) {
1291 0 : ulong out_idx = fd_topo_find_tile_out_link( topo, tile, "diag_gui", 0UL );
1292 0 : FD_TEST( out_idx!=ULONG_MAX );
1293 0 : fd_topo_link_t const * link = &topo->links[ tile->out_link_id[ out_idx ] ];
1294 0 : ctx->system_out.idx = out_idx;
1295 0 : ctx->system_out.mem = topo->workspaces[ topo->objs[ link->dcache_obj_id ].wksp_id ].wksp;
1296 0 : ctx->system_out.chunk0 = fd_dcache_compact_chunk0( ctx->system_out.mem, link->dcache );
1297 0 : ctx->system_out.wmark = fd_dcache_compact_wmark ( ctx->system_out.mem, link->dcache, link->mtu );
1298 0 : ctx->system_out.chunk = ctx->system_out.chunk0;
1299 0 : }
1300 :
1301 : /* Snapshot the cumulative-since-boot /proc interrupt/softirq counters
1302 : so the metrics we report are counted since process startup. */
1303 0 : memset( ctx->softirq_baseline, 0, sizeof( ctx->softirq_baseline ) );
1304 0 : memset( ctx->device_irq_baseline, 0, sizeof( ctx->device_irq_baseline ) );
1305 0 : memset( ctx->tlb_baseline, 0, sizeof( ctx->tlb_baseline ) );
1306 0 : memset( ctx->loc_baseline, 0, sizeof( ctx->loc_baseline ) );
1307 0 : if( FD_UNLIKELY( -1==lseek( ctx->proc_softirqs_fd, 0, SEEK_SET ) ) ) FD_LOG_ERR(( "lseek failed (%i-%s)", errno, strerror( errno ) ));
1308 0 : ulong softirq_cpu_cnt = fd_proc_softirqs_sum( ctx->proc_softirqs_fd, ctx->softirq_baseline );
1309 0 : if( FD_UNLIKELY( !softirq_cpu_cnt ) ) FD_LOG_WARNING(( "failed to read softirq baseline from /proc/softirqs" ));
1310 :
1311 0 : if( FD_UNLIKELY( -1==lseek( ctx->proc_interrupts_fd, 0, SEEK_SET ) ) ) FD_LOG_ERR(( "lseek failed (%i-%s)", errno, strerror( errno ) ));
1312 0 : ulong interrupt_cpu_cnt = fd_proc_interrupts_read( ctx->proc_interrupts_fd,
1313 0 : ctx->device_irq_baseline,
1314 0 : ctx->tlb_baseline,
1315 0 : ctx->loc_baseline );
1316 0 : if( FD_UNLIKELY( !interrupt_cpu_cnt ) ) FD_LOG_WARNING(( "failed to read IRQ baselines from /proc/interrupts" ));
1317 :
1318 0 : memset( ctx->irq_ticks_baseline, 0, sizeof( ctx->irq_ticks_baseline ) );
1319 0 : if( FD_UNLIKELY( -1==lseek( ctx->proc_stat_fd, 0, SEEK_SET ) ) ) FD_LOG_ERR(( "lseek failed (%i-%s)", errno, strerror( errno ) ));
1320 0 : ulong stat_cpu_cnt = fd_proc_stat_irq_ticks( ctx->proc_stat_fd, ctx->irq_ticks_baseline );
1321 0 : if( FD_UNLIKELY( !stat_cpu_cnt ) ) FD_LOG_WARNING(( "failed to read irq tick baseline from /proc/stat" ));
1322 :
1323 : /* Read starttime (field 22) once at init for idle time calculation.
1324 : CLK_TCK is always 100, so 1 tick = 10ms = 10,000,000 ns. */
1325 0 : for( ulong i=0UL; i<ctx->tile_cnt; i++ ) {
1326 0 : if( FD_LIKELY( -1!=ctx->stat_fds[ i ] ) ) {
1327 0 : int died = read_starttime( ctx->stat_fds[ i ], 10000000UL, &ctx->starttime_nanos[ i ] );
1328 0 : if( FD_UNLIKELY( died ) ) ctx->stat_fds[ i ] = -1;
1329 0 : }
1330 0 : }
1331 :
1332 0 : memset( &ctx->check_engine, 0, sizeof(ctx->check_engine) );
1333 :
1334 0 : ctx->tiles.bundle_cnt = fd_topo_tile_name_cnt( topo, "bundle" );
1335 0 : for( ulong i=0UL; i<ctx->tiles.bundle_cnt; i++ ) ctx->tiles.bundle_tile_idx[ i ] = fd_topo_find_tile( topo, "bundle", i );
1336 0 : ctx->tiles.shred_cnt = fd_topo_tile_name_cnt( topo, "shred" );
1337 0 : for( ulong i=0UL; i<ctx->tiles.shred_cnt; i++ ) ctx->tiles.shred_tile_idx[ i ] = fd_topo_find_tile( topo, "shred", i );
1338 0 : ctx->tiles.tower_idx = fd_topo_find_tile( topo, "tower", 0UL );
1339 0 : ctx->tiles.votor_idx = fd_topo_find_tile( topo, "votor", 0UL );
1340 0 : ctx->tiles.replay_idx = fd_topo_find_tile( topo, "replay", 0UL );
1341 :
1342 0 : fd_cpuset_new( &ctx->cpu_has_tile );
1343 0 : for( ulong i=0UL; i<(topo->tile_cnt); i++ ) {
1344 0 : ulong cpu_idx = topo->tiles[ i ].cpu_idx;
1345 0 : if( cpu_idx>=FD_TILE_MAX ) continue;
1346 0 : fd_cpuset_insert( ctx->cpu_has_tile, cpu_idx );
1347 0 : }
1348 :
1349 0 : for( ulong i=0UL; i<FD_TILE_MAX; i++ ) ctx->cpu_to_tile[ i ] = USHORT_MAX;
1350 0 : for( ulong i=0UL; i<topo->tile_cnt; i++ ) {
1351 0 : ulong cpu_idx = topo->tiles[ i ].cpu_idx;
1352 0 : if( cpu_idx>=FD_TILE_MAX || topo->tiles[ i ].floats ) continue;
1353 0 : ctx->cpu_to_tile[ cpu_idx ] = (ushort)i;
1354 0 : }
1355 :
1356 0 : long now = fd_log_wallclock();
1357 0 : ctx->is_voting = tile->diag.is_voting;
1358 0 : ctx->check_engine.vote_slot_changed_ns = now;
1359 0 : ctx->check_engine.reset_slot_changed_ns = now;
1360 0 : ctx->check_engine.turbine_slot_changed_ns = now;
1361 0 : ctx->check_engine.byte_snapshot_ns = now;
1362 0 : }
1363 :
1364 : static ulong
1365 : populate_allowed_seccomp( fd_topo_t const * topo,
1366 : fd_topo_tile_t const * tile,
1367 : ulong out_cnt,
1368 0 : struct sock_filter * out ) {
1369 0 : (void)topo;
1370 0 : (void)tile;
1371 :
1372 0 : populate_sock_filter_policy_fd_diag_tile( out_cnt, out, (uint)fd_log_private_logfile_fd() );
1373 0 : return sock_filter_policy_fd_diag_tile_instr_cnt;
1374 0 : }
1375 :
1376 : static ulong
1377 : populate_allowed_fds( fd_topo_t const * topo,
1378 : fd_topo_tile_t const * tile,
1379 : ulong out_fds_cnt,
1380 0 : int * out_fds ) {
1381 0 : fd_diag_tile_t * ctx = fd_topo_obj_laddr( topo, tile->tile_obj_id );
1382 :
1383 0 : int logfile_fd = fd_log_private_logfile_fd();
1384 0 : ulong required_fds = 5UL+2UL*ctx->tile_cnt+ctx->numa.cnt+ctx->mount_cnt+(ulong)(-1!=logfile_fd);
1385 0 : for( ulong i=0UL; i<ctx->file_cnt; i++ )
1386 0 : required_fds += (ulong)( ctx->files[ i ].data_fd>=0 && ctx->files[ i ].data_fd!=logfile_fd );
1387 0 : if( FD_UNLIKELY( out_fds_cnt<required_fds ) ) FD_LOG_ERR(( "out_fds_cnt %lu", out_fds_cnt ));
1388 :
1389 0 : ulong out_cnt = 0UL;
1390 0 : out_fds[ out_cnt++ ] = 2; /* stderr */
1391 0 : if( FD_LIKELY( -1!=logfile_fd ) )
1392 0 : out_fds[ out_cnt++ ] = logfile_fd; /* logfile */
1393 0 : out_fds[ out_cnt++ ] = ctx->proc_interrupts_fd; /* /proc/interrupts */
1394 0 : out_fds[ out_cnt++ ] = ctx->proc_softirqs_fd; /* /proc/softirqs */
1395 0 : out_fds[ out_cnt++ ] = ctx->proc_stat_fd; /* /proc/stat */
1396 0 : out_fds[ out_cnt++ ] = ctx->proc_meminfo_fd; /* /proc/meminfo */
1397 0 : for( ulong i=0UL; i<ctx->tile_cnt; i++ ) {
1398 0 : if( -1!=ctx->stat_fds[ i ] ) out_fds[ out_cnt++ ] = ctx->stat_fds[ i ]; /* /proc/<pid>/task/<tid>/stat */
1399 0 : if( -1!=ctx->sched_fds[ i ] ) out_fds[ out_cnt++ ] = ctx->sched_fds[ i ]; /* /proc/<pid>/task/<tid>/sched */
1400 0 : }
1401 0 : for( ulong i=0UL; i<ctx->numa.cnt; i++ )
1402 0 : if( ctx->numa.node[ i ].meminfo_fd>=0 ) out_fds[ out_cnt++ ] = ctx->numa.node[ i ].meminfo_fd;
1403 0 : for( ulong i=0UL; i<ctx->mount_cnt; i++ ) out_fds[ out_cnt++ ] = ctx->mounts[ i ].fd;
1404 0 : for( ulong i=0UL; i<ctx->file_cnt; i++ )
1405 0 : if( ctx->files[ i ].data_fd>=0 && ctx->files[ i ].data_fd!=logfile_fd )
1406 0 : out_fds[ out_cnt++ ] = ctx->files[ i ].data_fd;
1407 0 : return out_cnt;
1408 0 : }
1409 :
1410 0 : #define STEM_BURST (1UL)
1411 0 : #define STEM_LAZY ((long)10e6) /* 10ms */
1412 :
1413 0 : #define STEM_CALLBACK_CONTEXT_TYPE fd_diag_tile_t
1414 0 : #define STEM_CALLBACK_CONTEXT_ALIGN alignof(fd_diag_tile_t)
1415 :
1416 0 : #define STEM_CALLBACK_BEFORE_CREDIT before_credit
1417 :
1418 : #include "../../disco/stem/fd_stem.c"
1419 :
1420 : fd_topo_run_tile_t fd_tile_diag = {
1421 : .name = "diag",
1422 : .populate_allowed_seccomp = populate_allowed_seccomp,
1423 : .populate_allowed_fds = populate_allowed_fds,
1424 : .scratch_align = scratch_align,
1425 : .scratch_footprint = scratch_footprint,
1426 : .privileged_init = privileged_init,
1427 : .unprivileged_init = unprivileged_init,
1428 : .run = stem_run,
1429 : };
|