Line data Source code
1 : #include "fd_netlink_tile_private.h"
2 : #include "../topo/fd_topo.h"
3 : #include "../topo/fd_topob.h"
4 : #include "../metrics/fd_metrics.h"
5 : #include "../waker/fd_waker.h"
6 : #include "../../waltz/ip/fd_fib4_netlink.h"
7 : #include "../../waltz/mib/fd_netdev_netlink.h"
8 : #include "../../waltz/neigh/fd_neigh4_netlink.h"
9 : #include "../../util/pod/fd_pod_format.h"
10 : #include "../../util/log/fd_dtrace.h"
11 : #include "fd_netlink_tile.h"
12 :
13 : #include <errno.h>
14 : #include <net/if.h>
15 : #include <netinet/in.h> /* MSG_DONTWAIT */
16 : #include <sys/epoll.h>
17 : #include <sys/socket.h> /* SOL_{...} */
18 : #include <sys/random.h> /* getrandom */
19 : #include <sys/time.h> /* struct timeval */
20 : #include <time.h> /* CLOCK_REALTIME for seccomp filter */
21 : #include <linux/rtnetlink.h> /* RTM_{...} */
22 :
23 : #define FD_SOCKADDR_IN_SZ sizeof(struct sockaddr_in)
24 : #include "generated/netlink_seccomp.h"
25 :
26 : void
27 : fd_netlink_topo_create( fd_topo_tile_t * netlink_tile,
28 : fd_topo_t * topo,
29 : ulong netlnk_max_routes,
30 : ulong netlnk_max_peer_routes,
31 : ulong netlnk_max_neighbors,
32 0 : char const * bind_interface ) {
33 0 : fd_topo_obj_t * netdev_tbl_obj = fd_topob_obj( topo, "netdev_tbl", "netbase" );
34 0 : fd_topo_obj_t * neigh4_obj = fd_topob_obj( topo, "neigh4_hmap", "netbase" );
35 :
36 0 : fd_topob_tile_uses( topo, netlink_tile, netdev_tbl_obj, FD_SHMEM_JOIN_MODE_READ_WRITE );
37 0 : fd_topob_tile_uses( topo, netlink_tile, neigh4_obj, FD_SHMEM_JOIN_MODE_READ_WRITE );
38 :
39 : /* Configure netdev table */
40 0 : FD_TEST( fd_pod_insertf_ulong( topo->props, NETDEV_MAX, "obj.%lu.dev_max", netdev_tbl_obj->id ) );
41 0 : FD_TEST( fd_pod_insertf_ulong( topo->props, BOND_MASTER_MAX, "obj.%lu.bond_max", netdev_tbl_obj->id ) );
42 :
43 0 : netlink_tile->netlink.route_max = netlnk_max_routes;
44 0 : netlink_tile->netlink.route_peer_max = netlnk_max_peer_routes;
45 :
46 : /* Configure neighbor hashmap */
47 0 : FD_TEST( fd_pod_insertf_ulong( topo->props, netlnk_max_neighbors, "obj.%lu.ele_max", neigh4_obj->id ) );
48 0 : FD_TEST( fd_pod_insertf_ulong( topo->props, 16UL, "obj.%lu.probe_max", neigh4_obj->id ) );
49 0 : ulong neigh4_seed;
50 0 : FD_TEST( 8UL==getrandom( &neigh4_seed, sizeof(ulong), 0 ) );
51 0 : FD_TEST( fd_pod_insertf_ulong( topo->props, neigh4_seed, "obj.%lu.seed", neigh4_obj->id ) );
52 :
53 0 : netlink_tile->netlink.netdev_tbl_obj_id = netdev_tbl_obj->id;
54 0 : memcpy( netlink_tile->netlink.neigh_if, bind_interface, sizeof(netlink_tile->netlink.neigh_if) );
55 0 : netlink_tile->netlink.neigh4_obj_id = neigh4_obj->id;
56 0 : }
57 :
58 : void
59 : fd_netlink_topo_join( fd_topo_t * topo,
60 : fd_topo_tile_t * netlink_tile,
61 0 : fd_topo_tile_t * join_tile ) {
62 0 : fd_topob_tile_uses( topo, join_tile, &topo->objs[ netlink_tile->netlink.netdev_tbl_obj_id ], FD_SHMEM_JOIN_MODE_READ_ONLY );
63 0 : fd_topob_tile_uses( topo, join_tile, &topo->objs[ netlink_tile->netlink.neigh4_obj_id ], FD_SHMEM_JOIN_MODE_READ_ONLY );
64 0 : }
65 :
66 : /* Begin tile methods */
67 :
68 : FD_FN_CONST static inline ulong
69 0 : scratch_align( void ) {
70 0 : return alignof(fd_netlink_tile_ctx_t);
71 0 : }
72 :
73 : FD_FN_PURE static inline ulong
74 0 : scratch_footprint( fd_topo_tile_t const * tile ) {
75 0 : (void)tile;
76 0 : ulong l = FD_LAYOUT_INIT;
77 0 : l = FD_LAYOUT_APPEND( l, alignof(fd_netlink_tile_ctx_t), sizeof(fd_netlink_tile_ctx_t) );
78 0 : return FD_LAYOUT_FINI( l, scratch_align() );
79 0 : }
80 :
81 : static ulong
82 : populate_allowed_seccomp( fd_topo_t const * topo,
83 : fd_topo_tile_t const * tile,
84 : ulong out_cnt,
85 0 : struct sock_filter * out ) {
86 0 : fd_netlink_tile_ctx_t * ctx = fd_topo_obj_laddr( topo, tile->tile_obj_id );
87 0 : FD_TEST( ctx->magic==FD_NETLINK_TILE_CTX_MAGIC );
88 0 : uint epoll_inner_fd = (uint)FD_WAKER_INNER_FD( tile->waker_client_idx );
89 0 : uint epoll_outer_fd = (uint)FD_WAKER_OUTER_FD;
90 :
91 0 : populate_sock_filter_policy_netlink( out_cnt, out, (uint)fd_log_private_logfile_fd(), (uint)ctx->nl_monitor->fd, (uint)ctx->nl_req->fd, (uint)ctx->prober->sock_fd, epoll_inner_fd, epoll_outer_fd );
92 0 : return sock_filter_policy_netlink_instr_cnt;
93 0 : }
94 :
95 : static ulong
96 : populate_allowed_fds( fd_topo_t const * topo,
97 : fd_topo_tile_t const * tile,
98 : ulong out_fds_cnt,
99 0 : int * out_fds ) {
100 0 : fd_netlink_tile_ctx_t * ctx = fd_topo_obj_laddr( topo, tile->tile_obj_id );
101 0 : FD_TEST( ctx->magic==FD_NETLINK_TILE_CTX_MAGIC );
102 :
103 0 : if( FD_UNLIKELY( out_fds_cnt<7UL ) ) FD_LOG_ERR(( "out_fds_cnt too low (%lu)", out_fds_cnt ));
104 :
105 0 : ulong out_cnt = 0UL;
106 0 : out_fds[ out_cnt++ ] = 2; /* stderr */
107 0 : if( FD_LIKELY( -1!=fd_log_private_logfile_fd() ) )
108 0 : out_fds[ out_cnt++ ] = fd_log_private_logfile_fd(); /* logfile */
109 0 : out_fds[ out_cnt++ ] = ctx->nl_monitor->fd;
110 0 : out_fds[ out_cnt++ ] = ctx->nl_req->fd;
111 0 : out_fds[ out_cnt++ ] = ctx->prober->sock_fd;
112 0 : out_fds[ out_cnt++ ] = FD_WAKER_OUTER_FD; /* waker outer epoll fd (rearm) */
113 0 : out_fds[ out_cnt++ ] = FD_WAKER_INNER_FD( tile->waker_client_idx ); /* waker inner epoll fd */
114 0 : return out_cnt;
115 0 : }
116 :
117 : static void
118 : privileged_init( fd_topo_t const * topo,
119 0 : fd_topo_tile_t const * tile ) {
120 0 : if( FD_UNLIKELY( tile->kind_id!=0 ) ) {
121 0 : FD_LOG_ERR(( "Topology contains more than one netlink tile" ));
122 0 : }
123 :
124 0 : uint const neigh_if_idx = if_nametoindex( tile->netlink.neigh_if );
125 0 : if( FD_UNLIKELY( !neigh_if_idx ) ) FD_LOG_ERR(( "if_nametoindex(%.16s) failed (%i-%s)", tile->netlink.neigh_if, errno, fd_io_strerror( errno ) ));
126 :
127 0 : fd_netlink_tile_ctx_t * ctx = fd_topo_obj_laddr( topo, tile->tile_obj_id );
128 0 : fd_memset( ctx, 0, sizeof(fd_netlink_tile_ctx_t) );
129 0 : ctx->magic = FD_NETLINK_TILE_CTX_MAGIC;
130 0 : ctx->neigh4_ifidx = neigh_if_idx;
131 :
132 0 : if( FD_UNLIKELY( !fd_netlink_init( ctx->nl_monitor, 1000U ) ) ) {
133 0 : FD_LOG_ERR(( "Failed to connect to rtnetlink" ));
134 0 : }
135 0 : if( FD_UNLIKELY( !fd_netlink_init( ctx->nl_req, 9000000U ) ) ) {
136 0 : FD_LOG_ERR(( "Failed to connect to rtnetlink" ));
137 0 : }
138 :
139 0 : union {
140 0 : struct sockaddr sa;
141 0 : struct sockaddr_nl sanl;
142 0 : } sa;
143 0 : sa.sanl = (struct sockaddr_nl) {
144 0 : .nl_family = AF_NETLINK,
145 0 : .nl_groups = RTMGRP_LINK | RTMGRP_NEIGH | RTMGRP_IPV4_ROUTE
146 0 : };
147 0 : if( FD_UNLIKELY( 0!=bind( ctx->nl_monitor->fd, &sa.sa, sizeof(struct sockaddr_nl) ) ) ) {
148 0 : FD_LOG_ERR(( "bind(sock,RT_NETLINK,RTMGRP_{LINK,NEIGH,IPV4_ROUTE}) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
149 0 : }
150 :
151 0 : float const max_probes_per_second = 3.f;
152 0 : ulong const max_probe_burst = 128UL;
153 0 : float const probe_delay_seconds = 15.f;
154 0 : fd_neigh4_prober_init( ctx->prober, max_probes_per_second, max_probe_burst, probe_delay_seconds );
155 :
156 : /* Set duration of blocking reads in after_credit */
157 0 : struct timeval tv = { .tv_usec = 2000 }; /* 2ms */
158 0 : if( FD_UNLIKELY( 0!=setsockopt( ctx->nl_monitor->fd, SOL_SOCKET, SO_RCVTIMEO, &tv, sizeof(struct timeval) ) ) ) {
159 0 : FD_LOG_ERR(( "setsockopt(sock,SOL_SOCKET,SO_RCVTIMEO) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
160 0 : }
161 :
162 0 : ctx->waker_client_idx = tile->waker_client_idx;
163 0 : FD_TEST( ctx->waker_client_idx!=ULONG_MAX );
164 :
165 0 : struct epoll_event ev = { .events = EPOLLIN, .data.fd = ctx->nl_monitor->fd };
166 0 : if( FD_UNLIKELY( -1==epoll_ctl( FD_WAKER_INNER_FD( ctx->waker_client_idx ), EPOLL_CTL_ADD, ctx->nl_monitor->fd, &ev ) ) )
167 0 : FD_LOG_ERR(( "epoll_ctl(ADD,nl_monitor) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
168 0 : }
169 :
170 : static void
171 : unprivileged_init( fd_topo_t const * topo,
172 0 : fd_topo_tile_t const * tile ) {
173 0 : FD_SCRATCH_ALLOC_INIT( l, fd_topo_obj_laddr( topo, tile->tile_obj_id ) );
174 0 : fd_netlink_tile_ctx_t * ctx = FD_SCRATCH_ALLOC_APPEND( l, alignof(fd_netlink_tile_ctx_t), sizeof(fd_netlink_tile_ctx_t) );
175 0 : FD_TEST( ctx->magic==FD_NETLINK_TILE_CTX_MAGIC );
176 :
177 0 : FD_TEST( ctx->waker_client_idx!=ULONG_MAX );
178 0 : ctx->waker_fseq = fd_fseq_join( fd_topo_obj_laddr( topo, tile->waker_fseq_obj_id ) );
179 0 : FD_TEST( ctx->waker_fseq );
180 :
181 0 : FD_TEST( tile->netlink.netdev_tbl_obj_id );
182 0 : FD_TEST( tile->netlink.neigh4_obj_id );
183 0 : ctx->route_max = tile->netlink.route_max;
184 0 : ctx->route_peer_max = tile->netlink.route_peer_max;
185 :
186 0 : FD_TEST( fd_netdev_tbl_join( ctx->netdev_tbl, fd_topo_obj_laddr( topo, tile->netlink.netdev_tbl_obj_id ) ) );
187 :
188 0 : ulong neigh4_obj_id = tile->netlink.neigh4_obj_id;
189 0 : ulong neigh_ele_max = fd_pod_queryf_ulong( topo->props, ULONG_MAX, "obj.%lu.ele_max", neigh4_obj_id );
190 0 : ulong neigh_probe_max = fd_pod_queryf_ulong( topo->props, ULONG_MAX, "obj.%lu.probe_max", neigh4_obj_id );
191 0 : ulong neigh_seed = fd_pod_queryf_ulong( topo->props, ULONG_MAX, "obj.%lu.seed", neigh4_obj_id );
192 0 : FD_TEST( neigh_ele_max!=ULONG_MAX && neigh_probe_max!=ULONG_MAX && neigh_seed!=ULONG_MAX );
193 0 : FD_TEST( fd_neigh4_hmap_join( ctx->neigh4, fd_topo_obj_laddr( topo, neigh4_obj_id ), neigh_ele_max, neigh_probe_max, neigh_seed ) );
194 :
195 0 : FD_TEST( tile->out_cnt==1UL );
196 0 : fd_topo_link_t const * out = &topo->links[ tile->out_link_id[0] ];
197 0 : ctx->out_mem = topo->workspaces[ topo->objs[ out->dcache_obj_id ].wksp_id ].wksp;
198 0 : ctx->out_chunk0 = fd_dcache_compact_chunk0( ctx->out_mem, out->dcache );
199 0 : ctx->out_wmark = fd_dcache_compact_wmark( ctx->out_mem, out->dcache, out->mtu );
200 0 : ctx->out_chunk = ctx->out_chunk0;
201 :
202 :
203 0 : for( ulong i=0UL; i<tile->in_cnt; i++ ) {
204 0 : fd_topo_link_t const * link = &topo->links[ tile->in_link_id[ i ] ];
205 0 : if( FD_UNLIKELY( link->mtu!=0UL ) ) FD_LOG_ERR(( "netlink solicit links must have an MTU of zero" ));
206 0 : }
207 :
208 0 : ctx->action |= FD_NET_TILE_ACTION_LINK_UPDATE;
209 0 : ctx->action |= FD_NET_TILE_ACTION_ROUTE4_UPDATE;
210 0 : ctx->action |= FD_NET_TILE_ACTION_NEIGH_UPDATE;
211 :
212 0 : ctx->update_backoff = (long)( fd_tempo_tick_per_ns( NULL ) * 500e6 ); /* 500ms */
213 0 : }
214 :
215 : /* Begin stem methods */
216 :
217 : static inline void
218 0 : metrics_write( fd_netlink_tile_ctx_t * ctx ) {
219 0 : FD_MCNT_SET( NETLNK, EVENT_DROPPED, fd_netlink_enobufs_cnt );
220 0 : FD_MCNT_SET( NETLNK, LINK_FULL_SYNC, ctx->metrics.link_full_syncs );
221 0 : FD_MCNT_SET( NETLNK, ROUTE_FULL_SYNC, ctx->metrics.route_full_syncs );
222 0 : FD_MCNT_ENUM_COPY( NETLNK, UPDATE_PROCESSED, ctx->metrics.update_cnt );
223 0 : FD_MGAUGE_SET( NETLNK, INTERFACE_COUNT, ctx->netdev_tbl->hdr->dev_cnt );
224 0 : FD_MCNT_SET( NETLNK, NEIGHBOR_PROBE_SENT, ctx->metrics.neigh_solicits_sent );
225 0 : FD_MCNT_SET( NETLNK, NEIGHBOR_PROBE_FAILED, ctx->metrics.neigh_solicits_fails );
226 0 : FD_MCNT_SET( NETLNK, NEIGHBOR_PROBE_RATE_LIMIT_HOST, ctx->prober->local_rate_limited_cnt );
227 0 : FD_MCNT_SET( NETLNK, NEIGHBOR_PROBE_RATE_LIMIT_GLOBAL, ctx->prober->global_rate_limited_cnt );
228 0 : }
229 :
230 : /* netlink_monitor_read calls recvfrom to process a link, route, or
231 : neighbor update. Returns 1 if a message was read, 0 otherwise. */
232 :
233 : static void
234 : iproute_publish( fd_netlink_tile_ctx_t * ctx,
235 : fd_stem_context_t * stem,
236 0 : fd_iproute_msg_t const * msg ) {
237 0 : fd_memcpy( fd_chunk_to_laddr( ctx->out_mem, ctx->out_chunk ), msg, sizeof(*msg) );
238 0 : fd_stem_publish( stem, 0UL, 0UL, ctx->out_chunk, sizeof(*msg), 0UL, 0UL, fd_frag_meta_ts_comp( fd_tickcount() ) );
239 0 : ctx->out_chunk = fd_dcache_compact_next( ctx->out_chunk, sizeof(*msg), ctx->out_chunk0, ctx->out_wmark );
240 0 : }
241 :
242 : /* iproute_dump_begin starts a multipart RTM_GETROUTE request. The
243 : response iterator remains in ctx so subsequent calls can consume it
244 : one netlink message at a time. */
245 :
246 : static int
247 : iproute_dump_begin( fd_netlink_tile_ctx_t * ctx,
248 0 : uint table_id ) {
249 0 : uint seq = ctx->nl_req->seq++;
250 0 : struct {
251 0 : struct nlmsghdr nlh;
252 0 : struct rtmsg rtm;
253 0 : struct rtattr rta;
254 0 : uint table_id;
255 0 : } request = {
256 0 : .nlh = {
257 0 : .nlmsg_type = RTM_GETROUTE,
258 0 : .nlmsg_flags = NLM_F_REQUEST | NLM_F_DUMP,
259 0 : .nlmsg_len = sizeof(request),
260 0 : .nlmsg_seq = seq,
261 0 : },
262 0 : .rtm = { .rtm_family = AF_INET },
263 0 : .rta = {
264 0 : .rta_type = RTA_TABLE,
265 0 : .rta_len = RTA_LENGTH( sizeof(uint) ),
266 0 : },
267 0 : .table_id = table_id,
268 0 : };
269 0 : long send_res = sendto( ctx->nl_req->fd, &request, sizeof(request), 0, NULL, 0 );
270 0 : if( FD_UNLIKELY( send_res<0L ) ) {
271 0 : FD_LOG_WARNING(( "netlink send(%d,RTM_GETROUTE,NLM_F_REQUEST|NLM_F_DUMP) failed (%d-%s)", ctx->nl_req->fd, errno, fd_io_strerror( errno ) ));
272 0 : return FD_FIB_NETLINK_ERR_IO;
273 0 : }
274 0 : if( FD_UNLIKELY( send_res!=(long)sizeof(request) ) ) {
275 0 : FD_LOG_WARNING(( "netlink send(%d,RTM_GETROUTE,NLM_F_REQUEST|NLM_F_DUMP) failed (short write)", ctx->nl_req->fd ));
276 0 : return FD_FIB_NETLINK_ERR_IO;
277 0 : }
278 0 : fd_netlink_iter_init( ctx->dump_iter, ctx->nl_req, ctx->dump_buf, sizeof(ctx->dump_buf) );
279 0 : ctx->dump_active = 1;
280 0 : ctx->dump_advance = 0;
281 0 : ctx->dump_intr = 0;
282 0 : return FD_FIB_NETLINK_SUCCESS;
283 0 : }
284 :
285 : #define FD_NETLINK_DUMP_NEXT_DONE (0)
286 0 : #define FD_NETLINK_DUMP_NEXT_ROUTE (1)
287 0 : #define FD_NETLINK_DUMP_NEXT_PROGRESS (2)
288 :
289 : /* iproute_dump_next consumes at most one netlink message. */
290 :
291 : static int
292 : iproute_dump_next( fd_netlink_tile_ctx_t * ctx,
293 0 : fd_iproute_msg_t * route ) {
294 0 : if( ctx->dump_advance ) {
295 0 : fd_netlink_iter_next( ctx->dump_iter, ctx->nl_req );
296 0 : ctx->dump_advance = 0;
297 0 : }
298 0 : if( fd_netlink_iter_done( ctx->dump_iter ) ) {
299 0 : if( ctx->dump_iter->err==0 ) {
300 0 : struct nlmsghdr const * done = fd_netlink_iter_msg( ctx->dump_iter );
301 0 : if( FD_UNLIKELY( done->nlmsg_flags & NLM_F_DUMP_INTR ) ) ctx->dump_intr = 1;
302 0 : }
303 0 : int err = FD_FIB_NETLINK_SUCCESS;
304 0 : if( FD_UNLIKELY( ctx->dump_iter->err>0 ) ) err = FD_FIB_NETLINK_ERR_IO;
305 0 : else if( FD_UNLIKELY( ctx->dump_intr ) ) err = FD_FIB_NETLINK_ERR_INTR;
306 0 : else if( FD_UNLIKELY( ctx->dump_overflow ) ) err = FD_FIB_NETLINK_ERR_SPACE;
307 0 : ctx->dump_active = 0;
308 0 : return -err;
309 0 : }
310 :
311 0 : struct nlmsghdr const * nlh = fd_netlink_iter_msg( ctx->dump_iter );
312 0 : ctx->dump_advance = 1;
313 0 : if( FD_UNLIKELY( nlh->nlmsg_flags & NLM_F_DUMP_INTR ) ) ctx->dump_intr = 1;
314 0 : if( FD_UNLIKELY( nlh->nlmsg_type==NLMSG_ERROR ) ) {
315 0 : struct nlmsgerr const * err = NLMSG_DATA( nlh );
316 0 : int nl_err = -err->error;
317 0 : FD_LOG_WARNING(( "netlink RTM_GETROUTE,NLM_F_REQUEST|NLM_F_DUMP failed (%d-%s)", nl_err, fd_io_strerror( nl_err ) ));
318 0 : ctx->dump_active = 0;
319 0 : return -FD_FIB_NETLINK_ERR_IO;
320 0 : }
321 0 : if( FD_UNLIKELY( nlh->nlmsg_type!=RTM_NEWROUTE ) ) {
322 0 : FD_LOG_DEBUG(( "unexpected nlmsg_type %u", nlh->nlmsg_type ));
323 0 : return FD_NETLINK_DUMP_NEXT_PROGRESS;
324 0 : }
325 0 : if( FD_UNLIKELY( ctx->dump_overflow ) ) return FD_NETLINK_DUMP_NEXT_PROGRESS;
326 0 : if( !fd_fib4_netlink_translate( nlh, ctx->dump_table_id, route ) ) return FD_NETLINK_DUMP_NEXT_PROGRESS;
327 :
328 0 : ulong table_idx = ctx->dump_table_id==RT_TABLE_LOCAL ? 0UL : 1UL;
329 0 : if( route->prefix==32U && route->dst_addr ) {
330 0 : if( FD_UNLIKELY( ctx->dump_peer_cnt[ table_idx ]>=ctx->route_peer_max ) ) ctx->dump_overflow = 1;
331 0 : else ctx->dump_peer_cnt[ table_idx ]++;
332 0 : } else {
333 : /* fd_fib4 reserves one non-peer slot for its dummy throw route. */
334 0 : if( FD_UNLIKELY( ctx->dump_route_cnt[ table_idx ]+1UL>=ctx->route_max ) ) ctx->dump_overflow = 1;
335 0 : else ctx->dump_route_cnt[ table_idx ]++;
336 0 : }
337 0 : return ctx->dump_overflow ? FD_NETLINK_DUMP_NEXT_PROGRESS : FD_NETLINK_DUMP_NEXT_ROUTE;
338 0 : }
339 :
340 : /* netlink_monitor_read processes at most one netlink message. Returns
341 : 2 if it published a route, 1 if it processed a non-route message, and
342 : 0 if no message was available. */
343 :
344 : static int
345 : netlink_monitor_read( fd_netlink_tile_ctx_t * ctx,
346 : fd_stem_context_t * stem,
347 0 : int flags ) {
348 :
349 0 : if( ctx->monitor_buf_off>=ctx->monitor_buf_sz ) {
350 0 : long msg_sz = recvfrom( ctx->nl_monitor->fd, ctx->monitor_buf, sizeof(ctx->monitor_buf), flags, NULL, NULL );
351 0 : if( msg_sz<=0L ) {
352 0 : if( FD_LIKELY( errno==EAGAIN || errno==EINTR ) ) return 0;
353 0 : if( errno==ENOBUFS ) {
354 0 : fd_netlink_enobufs_cnt++;
355 0 : ctx->action |= FD_NET_TILE_ACTION_ROUTE4_UPDATE;
356 0 : return 0;
357 0 : }
358 0 : FD_LOG_ERR(( "recvfrom(nl_monitor) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
359 0 : }
360 0 : ctx->monitor_buf_sz = msg_sz;
361 0 : ctx->monitor_buf_off = 0L;
362 0 : }
363 :
364 0 : long rem = ctx->monitor_buf_sz-ctx->monitor_buf_off;
365 0 : struct nlmsghdr * nlh = fd_type_pun( ctx->monitor_buf+ctx->monitor_buf_off );
366 0 : if( FD_UNLIKELY( !NLMSG_OK( nlh, rem ) ) ) {
367 0 : FD_LOG_WARNING(( "malformed netlink monitor message" ));
368 0 : ctx->monitor_buf_off = ctx->monitor_buf_sz;
369 0 : return 1;
370 0 : }
371 0 : ctx->monitor_buf_off += (long)NLMSG_ALIGN( nlh->nlmsg_len );
372 0 : FD_DTRACE_PROBE_4( netlink_update, nlh->nlmsg_seq, nlh->nlmsg_type, nlh->nlmsg_len, nlh->nlmsg_flags );
373 0 : switch( nlh->nlmsg_type ) {
374 0 : case RTM_NEWLINK:
375 0 : case RTM_DELLINK:
376 0 : ctx->action |= FD_NET_TILE_ACTION_LINK_UPDATE;
377 0 : ctx->metrics.update_cnt[ FD_METRICS_ENUM_NETLINK_MESSAGE_V_LINK_IDX ]++;
378 0 : break;
379 0 : case RTM_NEWROUTE:
380 0 : case RTM_DELROUTE: {
381 0 : fd_iproute_msg_t route;
382 0 : if( fd_fib4_netlink_translate( nlh, RT_TABLE_LOCAL, &route ) ||
383 0 : fd_fib4_netlink_translate( nlh, RT_TABLE_MAIN, &route ) ) {
384 0 : iproute_publish( ctx, stem, &route );
385 0 : ctx->metrics.update_cnt[ FD_METRICS_ENUM_NETLINK_MESSAGE_V_IPV4_ROUTE_IDX ]++;
386 0 : return 2;
387 0 : }
388 0 : ctx->metrics.update_cnt[ FD_METRICS_ENUM_NETLINK_MESSAGE_V_IPV4_ROUTE_IDX ]++;
389 0 : break;
390 0 : }
391 0 : case RTM_NEWNEIGH:
392 0 : case RTM_DELNEIGH: {
393 0 : fd_neigh4_netlink_ingest_message( ctx->neigh4, nlh, ctx->neigh4_ifidx );
394 0 : ctx->metrics.update_cnt[ FD_METRICS_ENUM_NETLINK_MESSAGE_V_NEIGHBOR_IDX ]++;
395 0 : break;
396 0 : }
397 0 : default:
398 0 : FD_LOG_INFO(( "Received unexpected netlink message type %u", nlh->nlmsg_type ));
399 0 : break;
400 0 : }
401 0 : return 1;
402 0 : }
403 :
404 : static void
405 0 : during_housekeeping( fd_netlink_tile_ctx_t * ctx ) {
406 0 : long now = fd_tickcount();
407 0 : if( !ctx->dump_table_id && (ctx->action & FD_NET_TILE_ACTION_LINK_UPDATE) ) {
408 0 : if( now < ctx->link_update_ts ) return;
409 0 : ctx->action &= ~FD_NET_TILE_ACTION_LINK_UPDATE;
410 0 : fd_seqlock_write_lock( &ctx->netdev_tbl->hdr->seqlock );
411 0 : fd_netdev_netlink_load_table( ctx->netdev_tbl, ctx->nl_req );
412 0 : fd_seqlock_write_unlock( &ctx->netdev_tbl->hdr->seqlock );
413 0 : ctx->link_update_ts = now+ctx->update_backoff;
414 0 : ctx->metrics.link_full_syncs++;
415 0 : }
416 0 : if( !ctx->dump_table_id && (ctx->action & FD_NET_TILE_ACTION_NEIGH_UPDATE) ) {
417 0 : ctx->action &= ~FD_NET_TILE_ACTION_NEIGH_UPDATE;
418 0 : fd_neigh4_netlink_request_dump( ctx->nl_req, ctx->neigh4_ifidx );
419 0 : uchar buf[ 4096 ];
420 0 : fd_netlink_iter_t iter[1];
421 0 : for( fd_netlink_iter_init( iter, ctx->nl_req, buf, sizeof(buf) );
422 0 : !fd_netlink_iter_done( iter );
423 0 : fd_netlink_iter_next( iter, ctx->nl_req ) ) {
424 0 : fd_neigh4_netlink_ingest_message( ctx->neigh4, fd_netlink_iter_msg( iter ), ctx->neigh4_ifidx );
425 0 : }
426 0 : }
427 0 : }
428 :
429 : /* after_credit is called once per loop iteration when Stem has at least
430 : one credit available. */
431 :
432 : static void
433 : after_credit( fd_netlink_tile_ctx_t * ctx,
434 : fd_stem_context_t * stem,
435 : int * opt_poll_in,
436 0 : int * charge_busy ) {
437 :
438 0 : long now = fd_tickcount();
439 :
440 0 : if( !ctx->dump_table_id && (ctx->action & FD_NET_TILE_ACTION_ROUTE4_UPDATE) && now>=ctx->route4_update_ts ) {
441 0 : ctx->action &= ~FD_NET_TILE_ACTION_ROUTE4_UPDATE;
442 0 : fd_iproute_msg_t flush = { .op=FD_IPROUTE_OP_FLUSH };
443 0 : iproute_publish( ctx, stem, &flush );
444 0 : fd_memset( ctx->dump_route_cnt, 0, sizeof(ctx->dump_route_cnt) );
445 0 : fd_memset( ctx->dump_peer_cnt, 0, sizeof(ctx->dump_peer_cnt) );
446 0 : ctx->dump_overflow = 0;
447 0 : ctx->dump_table_id = RT_TABLE_LOCAL;
448 0 : *charge_busy = 1;
449 0 : *opt_poll_in = 0;
450 0 : return;
451 0 : }
452 :
453 0 : if( ctx->dump_table_id ) {
454 0 : if( !ctx->dump_active ) {
455 0 : int err = iproute_dump_begin( ctx, ctx->dump_table_id );
456 0 : if( FD_UNLIKELY( err ) ) {
457 0 : ctx->dump_table_id = 0U;
458 0 : ctx->action |= FD_NET_TILE_ACTION_ROUTE4_UPDATE;
459 0 : ctx->route4_update_ts = now+ctx->update_backoff;
460 0 : *charge_busy = 1;
461 0 : return;
462 0 : }
463 0 : }
464 :
465 0 : fd_iproute_msg_t route;
466 0 : int dump_res = iproute_dump_next( ctx, &route );
467 0 : if( dump_res==FD_NETLINK_DUMP_NEXT_ROUTE ) {
468 0 : iproute_publish( ctx, stem, &route );
469 0 : *charge_busy = 1;
470 0 : *opt_poll_in = 0;
471 0 : return;
472 0 : }
473 0 : if( dump_res==FD_NETLINK_DUMP_NEXT_PROGRESS ) {
474 0 : *charge_busy = 1;
475 0 : return;
476 0 : }
477 0 : if( FD_UNLIKELY( dump_res<0 ) ) {
478 0 : if( dump_res==-FD_FIB_NETLINK_ERR_SPACE ) FD_LOG_WARNING(( "routing table exceeds configured netlink route capacity" ));
479 0 : ctx->dump_table_id = 0U;
480 0 : ctx->action |= FD_NET_TILE_ACTION_ROUTE4_UPDATE;
481 0 : ctx->route4_update_ts = now+ctx->update_backoff;
482 0 : *charge_busy = 1;
483 0 : return;
484 0 : }
485 :
486 0 : if( ctx->dump_table_id==RT_TABLE_LOCAL ) {
487 0 : ctx->dump_table_id = RT_TABLE_MAIN;
488 0 : } else {
489 0 : ctx->dump_table_id = 0U;
490 0 : ctx->route4_update_ts = now+ctx->update_backoff;
491 0 : ctx->metrics.route_full_syncs++;
492 0 : }
493 0 : *charge_busy = 1;
494 0 : return;
495 0 : }
496 :
497 0 : int published = 0;
498 0 : if( FD_UNLIKELY( fd_fseq_query( ctx->waker_fseq )==1UL ) ) {
499 0 : fd_fseq_update( ctx->waker_fseq, 0UL );
500 0 : for(;;) {
501 0 : int read_res = netlink_monitor_read( ctx, stem, MSG_DONTWAIT );
502 0 : if( !read_res ) break;
503 0 : *charge_busy = 1;
504 0 : if( read_res==2 ) {
505 0 : published = 1;
506 0 : break;
507 0 : }
508 0 : }
509 0 : fd_waker_client_rearm( ctx->waker_client_idx );
510 0 : ctx->idle_cnt = -1L;
511 0 : }
512 :
513 0 : ctx->idle_cnt++;
514 0 : if( FD_UNLIKELY( !published && ctx->idle_cnt>=128L ) )
515 0 : fd_log_sleep( (long)1e6 );
516 :
517 0 : if( FD_UNLIKELY( published ) ) *opt_poll_in = 0;
518 0 : }
519 :
520 : /* after_poll_overrun is called when fd_stem.c was overrun while
521 : checking for new fragments: producers are hot, stay hot. */
522 :
523 : static void
524 0 : after_poll_overrun( fd_netlink_tile_ctx_t * ctx ) {
525 0 : ctx->idle_cnt = -1L;
526 0 : }
527 :
528 : /* after_frag handles a neighbor solicit request */
529 :
530 : static void
531 : after_frag( fd_netlink_tile_ctx_t * ctx,
532 : ulong in_idx,
533 : ulong seq,
534 : ulong sig,
535 : ulong sz,
536 : ulong tsorig,
537 : ulong tspub,
538 0 : fd_stem_context_t * stem ) {
539 0 : (void)in_idx; (void)seq; (void)tsorig; (void)tspub; (void)stem;
540 :
541 0 : ctx->idle_cnt = -1L;
542 0 : long now = fd_tickcount();
543 :
544 : /* Parse request (fully contained in sig field) */
545 :
546 0 : if( FD_UNLIKELY( sz!=0UL ) ) {
547 0 : FD_LOG_WARNING(( "unexpected sz %lu", sz ));
548 0 : }
549 0 : if( FD_UNLIKELY( sig==FD_NETLINK_ROUTE4_SYNC_SIG ) ) {
550 0 : ctx->action |= FD_NET_TILE_ACTION_ROUTE4_UPDATE;
551 0 : return;
552 0 : }
553 0 : uint if_idx = (uint)(sig>>32);
554 0 : uint ip4_addr = (uint)sig;
555 0 : if( FD_UNLIKELY( if_idx!=ctx->neigh4_ifidx ) ) {
556 0 : ctx->metrics.neigh_solicits_fails++;
557 0 : FD_LOG_ERR(( "received neighbor solicit request for invalid interface index %u", if_idx ));
558 0 : return;
559 0 : }
560 :
561 : /* Drop if the kernel is already working on the request */
562 0 : if( fd_neigh4_hmap_query( ctx->neigh4, &ip4_addr ) ) {
563 0 : ctx->metrics.neigh_solicits_fails++;
564 0 : return;
565 0 : }
566 :
567 : /* Insert placeholder (take above branch next time) */
568 :
569 0 : fd_neigh4_entry_t * ele = fd_neigh4_hmap_insert( ctx->neigh4, &ip4_addr );
570 0 : if( FD_UNLIKELY( !ele ) ) {
571 0 : ctx->metrics.neigh_solicits_fails++;
572 0 : return;
573 0 : }
574 : /* Atomically write the entry, initializing MAC and probe suppression timestamp to 0 */
575 0 : fd_neigh4_entry_t to_insert = (fd_neigh4_entry_t) {
576 0 : .ip4_addr = ip4_addr,
577 0 : .state = FD_NEIGH4_STATE_INCOMPLETE,
578 0 : };
579 0 : fd_neigh4_entry_atomic_st( ele, &to_insert );
580 :
581 : /* Trigger neighbor solicit via netlink */
582 :
583 0 : int probe_res = fd_neigh4_probe_rate_limited( ctx->prober, ele, ip4_addr, now );
584 0 : if( probe_res==0 ) {
585 0 : ctx->metrics.neigh_solicits_sent++;
586 0 : } else {
587 0 : fd_neigh4_hmap_remove( ctx->neigh4, ele );
588 0 : if( probe_res>0 ) ctx->metrics.neigh_solicits_fails++;
589 0 : }
590 :
591 0 : }
592 :
593 0 : #define STEM_BURST (1UL)
594 0 : #define STEM_LAZY ((ulong)13e6) /* 13ms */
595 :
596 0 : #define STEM_CALLBACK_CONTEXT_TYPE fd_netlink_tile_ctx_t
597 0 : #define STEM_CALLBACK_CONTEXT_ALIGN alignof(fd_netlink_tile_ctx_t)
598 :
599 0 : #define STEM_CALLBACK_METRICS_WRITE metrics_write
600 0 : #define STEM_CALLBACK_AFTER_POLL_OVERRUN after_poll_overrun
601 0 : #define STEM_CALLBACK_DURING_HOUSEKEEPING during_housekeeping
602 0 : #define STEM_CALLBACK_AFTER_CREDIT after_credit
603 0 : #define STEM_CALLBACK_AFTER_FRAG after_frag
604 :
605 : #include "../stem/fd_stem.c"
606 :
607 : /* End stem methods */
608 :
609 : fd_topo_run_tile_t fd_tile_netlnk = {
610 : .name = "netlnk",
611 : .populate_allowed_seccomp = populate_allowed_seccomp,
612 : .populate_allowed_fds = populate_allowed_fds,
613 : .scratch_align = scratch_align,
614 : .scratch_footprint = scratch_footprint,
615 : .privileged_init = privileged_init,
616 : .unprivileged_init = unprivileged_init,
617 : .run = stem_run
618 : };
|