Line data Source code
1 : /* The mlx5 tile translates Ethernet frames between mlx5 work/completion
2 : queues and fd_tango traffic.
3 :
4 : An mlx5 queue pair (QP) contains a send queue (SQ) and a receive queue
5 : (RQ). Work queue entries (WQEs) encode posted work. Completion
6 : queue entries (CQEs) report completed work through completion queues
7 : (CQs). The mlx5 tile uses separate CQs for TX and RX, one each. */
8 :
9 : #include "../fd_net_tile.h"
10 : #include "fd_mlx5_private.h"
11 :
12 : #include <errno.h>
13 : #include <fcntl.h>
14 : #include <net/if.h>
15 : #include <stddef.h>
16 : #include <stdlib.h>
17 : #include <netinet/in.h>
18 : #include <sys/socket.h>
19 :
20 : #include "../fd_net_common.h"
21 : #include "../../metrics/fd_metrics.h"
22 : #include "../fd_net_router.h"
23 : #include "../fd_linux_bond.h"
24 : #include "../../topo/fd_topo.h"
25 : #include "../../../discof/repair/fd_repair.h"
26 :
27 : #include "../../../waltz/ip/fd_iproute.h"
28 : #include "../../../util/net/fd_eth.h"
29 : #include "../../../util/net/fd_gre.h"
30 : #include "../../../util/net/fd_ip4.h"
31 : #include "../../../util/net/fd_udp.h"
32 : #include "../../../util/pod/fd_pod_format.h"
33 :
34 : #include <unistd.h>
35 : #include <sys/ioctl.h>
36 : #include <linux/rtnetlink.h>
37 : #include <rdma/ib_user_verbs.h>
38 :
39 : #include "generated/fd_mlx5_tile_seccomp.h"
40 :
41 12 : #define IN_KIND_NET (0U)
42 0 : #define IN_KIND_IPROUTE (1U)
43 :
44 : /* Max number of flow rules */
45 : #define FD_MLX5_FLOW_CAP (64UL)
46 276 : #define FD_MLX5_GRE_MAX (4UL)
47 12 : #define FD_MLX5_TX_FLUSH_TIMEOUT_NS (20000L) /* 20us */
48 :
49 : /* FD_MLX5_SQ_* are options in a SQ WQE to request certain NIC behaviour.
50 : SEND requests packet transmission. CQ_UPDATE requests a CQE. */
51 105 : #define FD_MLX5_SQ_SEND_PKT (0x0aU) /* SEND */
52 48 : #define FD_MLX5_SQ_REQUEST_CQE (0x08U) /* CQ_UPDATE */
53 :
54 : /* FD_MLX5_CQE_OP_* are mlx5 CQE opcodes used to classify TX and RX completions */
55 75 : #define FD_MLX5_CQE_OP_TX_OK ( 0U) /* MLX5_CQE_REQ */
56 375 : #define FD_MLX5_CQE_OP_RX_OK ( 2U) /* MLX5_CQE_RESP_SEND */
57 15 : #define FD_MLX5_CQE_OP_TX_ERR (13U) /* MLX5_CQE_REQ_ERR */
58 819 : #define FD_MLX5_CQE_OP_RX_ERR (14U) /* MLX5_CQE_RESP_ERR */
59 37140 : #define FD_MLX5_CQE_OP_INVALID (15U)
60 :
61 240 : #define FD_MLX5_CQE_SYNDROME_LOCAL_LENGTH_ERR (0x01U) /* received packet's length exceeded FD_NET_MTU */
62 :
63 : FD_STATIC_ASSERT( FD_NET_ROUTE_FAIL_CNT==FD_METRICS_ENUM_ROUTE_FAIL_CNT, route_fail_metric_cnt );
64 :
65 : /* FD_MLX5_ASYNC_EVENT_* are Linux enum ib_event_type values delivered
66 : in ib_uverbs_async_event_desc.event_type. */
67 : #define FD_MLX5_ASYNC_EVENT_CQ_ERR ( 0U)
68 : #define FD_MLX5_ASYNC_EVENT_QP_FATAL ( 1U)
69 : #define FD_MLX5_ASYNC_EVENT_QP_REQ_ERR ( 2U)
70 : #define FD_MLX5_ASYNC_EVENT_QP_ACCESS_ERR ( 3U)
71 : #define FD_MLX5_ASYNC_EVENT_DEVICE_FATAL ( 8U)
72 : #define FD_MLX5_ASYNC_EVENT_WQ_FATAL (19U)
73 :
74 : /* The fd_mlx5_hw_* structs below define WQE and CQE formats in the
75 : mlx5 hardware interface. The links show the corresponding definitions
76 : in the Linux mlx5 driver. */
77 :
78 : /* https://elixir.bootlin.com/linux/v7.1.8/source/include/linux/mlx5/qp.h#L205 */
79 : struct __attribute__((packed)) fd_mlx5_hw_wqe_ctrl_seg {
80 : /* mlx5 packs opmod, WQE index, opcode, QP number, and WQE segment
81 : count into two big-endian 32-bit words. */
82 : uint opmod_idx_opcode;
83 : uint qpn_ds;
84 :
85 : uchar signature;
86 : uchar reserved[2];
87 : uchar flags;
88 : uint imm;
89 : };
90 : typedef struct fd_mlx5_hw_wqe_ctrl_seg fd_mlx5_hw_wqe_ctrl_seg_t;
91 :
92 : /* https://elixir.bootlin.com/linux/v7.1.8/source/include/linux/mlx5/qp.h#L265 */
93 : struct __attribute__((packed)) fd_mlx5_hw_wqe_eth_seg {
94 : uchar reserved[12];
95 : ushort inline_hdr_sz;
96 : uchar inline_hdr[2];
97 : };
98 : typedef struct fd_mlx5_hw_wqe_eth_seg fd_mlx5_hw_wqe_eth_seg_t;
99 :
100 : /* https://elixir.bootlin.com/linux/v7.1.8/source/include/linux/mlx5/qp.h#L364 */
101 : struct __attribute__((packed)) fd_mlx5_hw_wqe_data_seg {
102 : uint byte_cnt;
103 : uint lkey;
104 : ulong addr;
105 : };
106 : typedef struct fd_mlx5_hw_wqe_data_seg fd_mlx5_hw_wqe_data_seg_t;
107 :
108 : /* https://elixir.bootlin.com/linux/v7.1.8/source/include/linux/mlx5/device.h#L824 */
109 : struct __attribute__((packed,may_alias)) fd_mlx5_hw_cqe64 {
110 : uchar reserved0[44];
111 : uint byte_cnt;
112 : uchar reserved1[6];
113 : uchar vendor_err;
114 : uchar syndrome;
115 : uint sop_drop_qpn;
116 : ushort wqe_counter;
117 : uchar signature;
118 : uchar op_own;
119 : };
120 : typedef struct fd_mlx5_hw_cqe64 fd_mlx5_hw_cqe64_t;
121 :
122 : struct fd_mlx5_tile_rx_comp {
123 : ulong chunk;
124 : uint byte_len;
125 : uint opcode;
126 : };
127 : typedef struct fd_mlx5_tile_rx_comp fd_mlx5_tile_rx_comp_t;
128 :
129 : struct fd_mlx5_tile_input_ctx {
130 : void * wksp_base;
131 : ulong chunk0;
132 : ulong wmark;
133 : };
134 : typedef struct fd_mlx5_tile_input_ctx fd_mlx5_tile_input_ctx_t;
135 :
136 : /* fd_mlx5_tile_t is private tile state */
137 : struct fd_mlx5_tile {
138 : fd_uverbs_ctx_t uverbs;
139 : fd_mlx5_cq_t rx_cq;
140 : fd_mlx5_cq_t tx_cq;
141 : fd_mlx5_rx_wq_t rx_wq;
142 : fd_mlx5_tx_qp_t tx_qp;
143 : fd_mlx5_rss_qp_t outer_rss_qp;
144 : fd_mlx5_rss_qp_t gre_rss_qp;
145 : uint lkey;
146 : uint prepared;
147 :
148 : uint batch_size; /* used for both SQ/RQ and TX/RX CQ batching */
149 : uint net_tile_id;
150 : uint net_tile_cnt;
151 :
152 : /* RQ batching */
153 : uint rq_pending_chunk[ FD_MLX5_BATCH_SIZE ];
154 : uint rq_pending_cnt;
155 :
156 : /* SQ batching */
157 : long sq_flush_timeout_ticks;
158 : long sq_flush_deadline_ticks;
159 :
160 : /* Packet buffer addressing */
161 : uchar * pkt_buf_wksp_base;
162 : uint pkt_buf_chunk0; /* lowest allowed chunk number */
163 : uint pkt_buf_wmark; /* highest allowed chunk number */
164 : uint * sq_wqe_buf_chunk; /* maps SQ WQEs to packet buffers */
165 :
166 : /* TX input links */
167 : fd_mlx5_tile_input_ctx_t input_ctx[ FD_TOPO_MAX_TILE_IN_LINKS ];
168 : uchar in_kind[ FD_TOPO_MAX_TILE_IN_LINKS ];
169 : fd_iproute_msg_t iproute_msg; /* route update staged between Stem callbacks */
170 :
171 : /* TX IP routing */
172 : fd_net_router_t router;
173 : fd_net_tx_route_t tx_route;
174 :
175 : /* RX UDP destination port to RX out link */
176 : uint dst_port_cnt;
177 : ushort dst_ports[ FD_MLX5_FLOW_CAP ];
178 : uchar dst_protos[ FD_MLX5_FLOW_CAP ];
179 : uchar dst_out_idx[ FD_MLX5_FLOW_CAP ];
180 : uchar repair_out_idx;
181 : uint gre_tunnel_ip[ FD_MLX5_GRE_MAX ];
182 :
183 : uchar rx_out_cnt; /* number of out links */
184 :
185 : /* Metric tracking */
186 : struct {
187 : ulong rx_pkt_cnt;
188 : ulong rx_bytes_total;
189 : ulong rx_malformed_cnt;
190 : ulong rx_route_fail_cnt;
191 : ulong rx_gre_cnt;
192 : ulong rx_gre_invalid_cnt;
193 : ulong rx_gre_ignored_cnt;
194 : ulong tx_pkt_cnt;
195 : ulong tx_bytes_total;
196 : ulong tx_no_buffer_cnt;
197 : ulong tx_invalid_cnt;
198 : ulong tx_gre_cnt;
199 : ulong tx_gre_route_fail_cnt;
200 : ulong tx_gre_oversize_cnt;
201 : } metrics;
202 : };
203 : typedef struct fd_mlx5_tile fd_mlx5_tile_t;
204 :
205 : static ulong
206 : fd_mlx5_queue_footprint( uint rx_depth,
207 144 : uint tx_depth ) {
208 144 : ulong layout = FD_LAYOUT_INIT;
209 144 : layout = FD_LAYOUT_APPEND( layout, FD_MLX5_PAGE_SZ, rx_depth*sizeof(fd_mlx5_cqe_t) );
210 144 : layout = FD_LAYOUT_APPEND( layout, FD_MLX5_PAGE_SZ, tx_depth*sizeof(fd_mlx5_cqe_t) );
211 144 : layout = FD_LAYOUT_APPEND( layout, FD_MLX5_PAGE_SZ, rx_depth*sizeof(fd_mlx5_rx_wqe_t) );
212 144 : layout = FD_LAYOUT_APPEND( layout, FD_MLX5_PAGE_SZ, tx_depth*sizeof(fd_mlx5_tx_wqe_t) );
213 144 : layout = FD_LAYOUT_APPEND( layout, FD_MLX5_PAGE_SZ, rx_depth*sizeof(uint) );
214 144 : layout = FD_LAYOUT_APPEND( layout, FD_MLX5_PAGE_SZ, tx_depth*sizeof(uint) );
215 144 : layout = FD_LAYOUT_APPEND( layout, FD_MLX5_PAGE_SZ, FD_MLX5_PAGE_SZ );
216 144 : return FD_LAYOUT_FINI( layout, FD_MLX5_PAGE_SZ );
217 144 : }
218 :
219 : static void
220 : fd_mlx5_hw_invalidate_cqes( fd_mlx5_cqe_t * cqes,
221 60 : uint depth ) {
222 37068 : for( uint i=0U; i<depth; i++ ) {
223 37008 : fd_mlx5_hw_cqe64_t * hw_cqe = (fd_mlx5_hw_cqe64_t *)(cqes+i);
224 37008 : hw_cqe->op_own = (uchar)(FD_MLX5_CQE_OP_INVALID<<4);
225 37008 : }
226 60 : }
227 :
228 : static fd_mlx5_tile_t *
229 : fd_mlx5_hw_join_queues( fd_mlx5_tile_t * ctx,
230 : void * queue_memory,
231 : uint rx_depth,
232 30 : uint tx_depth ) {
233 30 : ulong const queue_footprint = fd_mlx5_queue_footprint( rx_depth, tx_depth );
234 30 : FD_SCRATCH_ALLOC_INIT( queue, queue_memory );
235 30 : fd_mlx5_cqe_t * rx_cqes = FD_SCRATCH_ALLOC_APPEND( queue, FD_MLX5_PAGE_SZ, rx_depth*sizeof(fd_mlx5_cqe_t) );
236 30 : fd_mlx5_cqe_t * tx_cqes = FD_SCRATCH_ALLOC_APPEND( queue, FD_MLX5_PAGE_SZ, tx_depth*sizeof(fd_mlx5_cqe_t) );
237 30 : fd_mlx5_rx_wqe_t * rq = FD_SCRATCH_ALLOC_APPEND( queue, FD_MLX5_PAGE_SZ, rx_depth*sizeof(fd_mlx5_rx_wqe_t) );
238 30 : fd_mlx5_tx_wqe_t * sq = FD_SCRATCH_ALLOC_APPEND( queue, FD_MLX5_PAGE_SZ, tx_depth*sizeof(fd_mlx5_tx_wqe_t) );
239 30 : uint * rq_wqe_buf_chunk = FD_SCRATCH_ALLOC_APPEND( queue, FD_MLX5_PAGE_SZ, rx_depth*sizeof(uint) );
240 30 : uint * sq_wqe_frame_sz = FD_SCRATCH_ALLOC_APPEND( queue, FD_MLX5_PAGE_SZ, tx_depth*sizeof(uint) );
241 30 : uchar * control = FD_SCRATCH_ALLOC_APPEND( queue, FD_MLX5_PAGE_SZ, FD_MLX5_PAGE_SZ );
242 30 : FD_TEST( FD_SCRATCH_ALLOC_FINI( queue, FD_MLX5_PAGE_SZ )==(ulong)queue_memory+queue_footprint );
243 :
244 30 : ctx->rx_cq.entries = rx_cqes;
245 30 : ctx->rx_cq.control = (fd_mlx5_cq_control_t *)control;
246 30 : ctx->rx_cq.depth = rx_depth;
247 :
248 30 : ctx->tx_cq.entries = tx_cqes;
249 30 : ctx->tx_cq.control = (fd_mlx5_cq_control_t *)(control+sizeof(fd_mlx5_cq_control_t));
250 30 : ctx->tx_cq.depth = tx_depth;
251 :
252 30 : ctx->rx_wq.rq = rq;
253 30 : ctx->rx_wq.rx_cq = &ctx->rx_cq;
254 30 : ctx->rx_wq.rx_depth = rx_depth;
255 30 : ctx->rx_wq.rq_wqe_buf_chunk = rq_wqe_buf_chunk;
256 30 : ctx->rx_wq.control = (fd_mlx5_rx_wq_control_t *)(control+2UL*sizeof(fd_mlx5_cq_control_t));
257 :
258 30 : ctx->tx_qp.sq = sq;
259 30 : ctx->tx_qp.tx_cq = &ctx->tx_cq;
260 30 : ctx->tx_qp.tx_depth = tx_depth;
261 30 : ctx->tx_qp.sq_wqe_frame_sz = sq_wqe_frame_sz;
262 30 : ctx->tx_qp.control = (fd_mlx5_qp_control_t *)(control+2UL*sizeof(fd_mlx5_cq_control_t)+sizeof(fd_mlx5_rx_wq_control_t));
263 30 : return ctx;
264 30 : }
265 :
266 : static fd_mlx5_tile_t *
267 : fd_mlx5_hw_init_queues( fd_mlx5_tile_t * ctx,
268 : void * queue_memory,
269 : uint rx_depth,
270 30 : uint tx_depth ) {
271 30 : fd_memset( queue_memory, 0, fd_mlx5_queue_footprint( rx_depth, tx_depth ) );
272 30 : FD_TEST( fd_mlx5_hw_join_queues( ctx, queue_memory, rx_depth, tx_depth ) );
273 30 : fd_mlx5_hw_invalidate_cqes( ctx->rx_cq.entries, rx_depth );
274 30 : fd_mlx5_hw_invalidate_cqes( ctx->tx_cq.entries, tx_depth );
275 30 : return ctx;
276 30 : }
277 :
278 : /* fd_mlx5_tile_tx_chunk returns the buffer paired with SQ WQE sq_idx.
279 : The buffer is reusable once sq_cons advances past that WQE. */
280 : static inline uint
281 : fd_mlx5_tile_tx_chunk( fd_mlx5_tile_t const * ctx,
282 291 : uint sq_idx ) {
283 291 : return ctx->sq_wqe_buf_chunk[ sq_idx & (ctx->tx_qp.tx_depth-1U) ];
284 291 : }
285 :
286 : static inline void
287 3204 : fd_mlx5_hw_dma_to_device( void ) {
288 3204 : #if FD_HAS_X86
289 3204 : FD_COMPILER_MFENCE();
290 : #elif FD_HAS_ARM
291 : __asm__ __volatile__( "dmb oshst" ::: "memory" );
292 : #else
293 : FD_HW_MFENCE_ST();
294 : #endif
295 3204 : }
296 :
297 : static inline void
298 546 : fd_mlx5_hw_dma_from_device( void ) {
299 546 : #if FD_HAS_X86
300 546 : __asm__ __volatile__( "lfence" ::: "memory" );
301 : #elif FD_HAS_ARM
302 : __asm__ __volatile__( "dmb oshld" ::: "memory" );
303 : #else
304 : FD_HW_MFENCE();
305 : #endif
306 546 : }
307 :
308 : static inline int
309 : fd_mlx5_hw_init_tx_wqe( fd_mlx5_tx_wqe_t * wqe,
310 : uint sq_idx,
311 : uint qpn,
312 : void const * frame,
313 : ulong frame_iova,
314 : ulong frame_sz,
315 : uint lkey,
316 180 : ulong inline_hdr_sz ) {
317 180 : if( FD_UNLIKELY( !frame_sz || frame_sz>UINT_MAX || qpn>0xffffffU ||
318 180 : inline_hdr_sz>frame_sz || frame_iova>ULONG_MAX-inline_hdr_sz ) ) return 0;
319 105 : fd_memset( wqe, 0, sizeof(*wqe) );
320 105 : uint const wqe_seg_cnt = inline_hdr_sz ? 4U : 3U;
321 105 : ulong const wqe_data_seg_off = inline_hdr_sz ? 48UL : 32UL;
322 :
323 105 : fd_mlx5_hw_wqe_ctrl_seg_t * wqe_ctrl_seg = (fd_mlx5_hw_wqe_ctrl_seg_t *)wqe;
324 105 : fd_mlx5_hw_wqe_eth_seg_t * wqe_eth_seg = (fd_mlx5_hw_wqe_eth_seg_t *)(wqe->bytes+sizeof(*wqe_ctrl_seg));
325 105 : fd_mlx5_hw_wqe_data_seg_t * wqe_data_seg = (fd_mlx5_hw_wqe_data_seg_t *)(wqe->bytes+wqe_data_seg_off);
326 :
327 105 : wqe_ctrl_seg->opmod_idx_opcode = fd_uint_bswap( ((sq_idx & 0xffffU)<<8) | FD_MLX5_SQ_SEND_PKT );
328 105 : wqe_ctrl_seg->qpn_ds = fd_uint_bswap( (qpn<<8) | wqe_seg_cnt );
329 :
330 105 : wqe_eth_seg->inline_hdr_sz = fd_ushort_bswap( (ushort)inline_hdr_sz );
331 105 : if( inline_hdr_sz ) fd_memcpy( wqe_eth_seg->inline_hdr, frame, inline_hdr_sz );
332 :
333 105 : wqe_data_seg->byte_cnt = fd_uint_bswap( (uint)(frame_sz-inline_hdr_sz) );
334 105 : wqe_data_seg->lkey = fd_uint_bswap( lkey );
335 105 : wqe_data_seg->addr = fd_ulong_bswap( frame_iova+inline_hdr_sz );
336 105 : return 1;
337 180 : }
338 :
339 : static inline void
340 : fd_mlx5_hw_init_rx_wqe( fd_mlx5_rx_wqe_t * wqe,
341 : ulong frame_iova,
342 : ulong frame_sz,
343 24864 : uint lkey ) {
344 24864 : fd_mlx5_hw_wqe_data_seg_t * wqe_data_seg = (fd_mlx5_hw_wqe_data_seg_t *)wqe;
345 :
346 24864 : wqe_data_seg->byte_cnt = fd_uint_bswap( (uint)frame_sz );
347 24864 : wqe_data_seg->lkey = fd_uint_bswap( lkey );
348 24864 : wqe_data_seg->addr = fd_ulong_bswap( frame_iova );
349 24864 : }
350 :
351 : static inline void
352 : fd_mlx5_hw_ring_sq( fd_mlx5_tx_qp_t * tx_qp,
353 48 : fd_mlx5_tx_wqe_t * wqe ) {
354 48 : fd_mlx5_hw_wqe_ctrl_seg_t * wqe_ctrl_seg = (fd_mlx5_hw_wqe_ctrl_seg_t *)wqe;
355 48 : wqe_ctrl_seg->flags = FD_MLX5_SQ_REQUEST_CQE;
356 :
357 48 : fd_mlx5_hw_dma_to_device();
358 48 : FD_VOLATILE( tx_qp->control->sq_prod ) = fd_uint_bswap( tx_qp->sq_prod & 0xffffU );
359 48 : fd_mlx5_hw_dma_to_device();
360 :
361 : /* Ring the SQ through the non-cached UAR. The NIC fetches the complete
362 : WQE from write-back SQ memory. */
363 48 : volatile ulong * doorbell_reg = (volatile ulong *)tx_qp->sq_doorbell;
364 48 : doorbell_reg[0] = FD_LOAD( ulong, wqe->bytes );
365 48 : FD_COMPILER_MFENCE();
366 48 : tx_qp->sq_posted = tx_qp->sq_prod;
367 48 : }
368 :
369 : static inline int
370 : fd_mlx5_hw_post_send( fd_mlx5_tile_t * ctx,
371 63 : uint frame_sz ) {
372 63 : fd_mlx5_tx_qp_t * tx_qp = &ctx->tx_qp;
373 63 : uint const outstanding = tx_qp->sq_prod-tx_qp->sq_cons;
374 63 : if( FD_UNLIKELY( outstanding>=tx_qp->tx_depth ) ) {
375 3 : errno = ENOSPC;
376 3 : return -1;
377 3 : }
378 :
379 60 : uint const sq_idx = tx_qp->sq_prod;
380 60 : uint const chunk = fd_mlx5_tile_tx_chunk( ctx, sq_idx );
381 60 : ulong const frame_iova = (ulong)chunk<<FD_CHUNK_LG_SZ;
382 60 : fd_mlx5_tx_wqe_t * wqe = tx_qp->sq+(sq_idx & (tx_qp->tx_depth-1U));
383 60 : if( FD_UNLIKELY( !fd_mlx5_hw_init_tx_wqe( wqe, sq_idx, tx_qp->qpn,
384 60 : fd_chunk_to_laddr( ctx->pkt_buf_wksp_base, chunk ), frame_iova, frame_sz,
385 60 : ctx->lkey, tx_qp->tx_inline_hdr_sz ) ) ) {
386 0 : errno = EINVAL;
387 0 : return -1;
388 0 : }
389 60 : tx_qp->sq_wqe_frame_sz[ sq_idx & (tx_qp->tx_depth-1U) ] = frame_sz;
390 60 : tx_qp->sq_prod++;
391 60 : return 0;
392 60 : }
393 :
394 : static inline int
395 : fd_mlx5_hw_post_recv( fd_mlx5_tile_t * ctx,
396 3108 : uint recv_cnt ) {
397 3108 : fd_mlx5_rx_wq_t * rx_wq = &ctx->rx_wq;
398 3108 : uint const outstanding = rx_wq->rq_prod-rx_wq->rq_cons;
399 3108 : if( FD_UNLIKELY( outstanding>rx_wq->rx_depth || recv_cnt>rx_wq->rx_depth-outstanding ) ) {
400 0 : errno = ENOSPC;
401 0 : return -1;
402 0 : }
403 :
404 3108 : uint const rq_prod = rx_wq->rq_prod;
405 27972 : for( uint i=0U; i<recv_cnt; i++ ) {
406 24864 : uint const chunk = ctx->rq_pending_chunk[ i ];
407 24864 : uint const rq_idx = rq_prod+i;
408 24864 : fd_mlx5_hw_init_rx_wqe( rx_wq->rq+(rq_idx & (rx_wq->rx_depth-1U)),
409 24864 : (ulong)chunk<<FD_CHUNK_LG_SZ, FD_NET_MTU, ctx->lkey );
410 24864 : rx_wq->rq_wqe_buf_chunk[ rq_idx & (rx_wq->rx_depth-1U) ] = chunk;
411 24864 : }
412 3108 : rx_wq->rq_prod += recv_cnt;
413 3108 : fd_mlx5_hw_dma_to_device();
414 3108 : FD_VOLATILE( rx_wq->control->rq_prod ) = fd_uint_bswap( rx_wq->rq_prod & 0xffffU );
415 3108 : return 0;
416 3108 : }
417 :
418 : static inline int
419 : fd_mlx5_hw_poll_rx_cq( fd_mlx5_rx_wq_t * rx_wq,
420 : fd_mlx5_tile_rx_comp_t * comp,
421 324 : uint comp_capacity ) {
422 324 : fd_mlx5_cq_t * rx_cq = rx_wq->rx_cq;
423 324 : uint const comp_limit = fd_uint_min( comp_capacity, rx_cq->depth );
424 324 : uint cq_cons_idx = rx_cq->cons_idx;
425 324 : uint rq_cons_idx = rx_wq->rq_cons;
426 324 : uint const rq_prod_idx = rx_wq->rq_prod;
427 324 : FD_TEST( rq_prod_idx-rq_cons_idx<=rx_wq->rx_depth );
428 :
429 324 : uint comp_cnt = 0U;
430 324 : int poll_failed = 0;
431 792 : while( comp_cnt<comp_limit ) {
432 762 : fd_mlx5_hw_cqe64_t const * rx_cqe = (fd_mlx5_hw_cqe64_t const *)(rx_cq->entries+(cq_cons_idx & (rx_cq->depth-1U)));
433 :
434 762 : uchar const op_own = FD_VOLATILE_CONST( rx_cqe->op_own );
435 762 : uint const opcode = (uint)(op_own>>4);
436 762 : if( FD_UNLIKELY( opcode==FD_MLX5_CQE_OP_INVALID ||
437 762 : (op_own & 1U)!=!!(cq_cons_idx & rx_cq->depth) ) ) {
438 276 : break;
439 276 : }
440 486 : fd_mlx5_hw_dma_from_device();
441 :
442 486 : if( FD_UNLIKELY( opcode!=FD_MLX5_CQE_OP_RX_OK && opcode!=FD_MLX5_CQE_OP_RX_ERR ) ) {
443 0 : errno = EINVAL;
444 0 : poll_failed = 1;
445 0 : break;
446 0 : }
447 486 : if( FD_UNLIKELY( opcode==FD_MLX5_CQE_OP_RX_ERR &&
448 486 : rx_cqe->syndrome!=FD_MLX5_CQE_SYNDROME_LOCAL_LENGTH_ERR ) ) {
449 3 : FD_LOG_ERR(( "mlx5 RX CQE error (syndrome 0x%02x, vendor syndrome 0x%02x, "
450 3 : "WQE opcode/QPN 0x%08x, WQE counter %u)",
451 3 : (uint)rx_cqe->syndrome, (uint)rx_cqe->vendor_err, fd_uint_bswap( rx_cqe->sop_drop_qpn ),
452 3 : (uint)fd_ushort_bswap( rx_cqe->wqe_counter ) ));
453 3 : }
454 :
455 483 : ushort const wqe_counter = fd_ushort_bswap( rx_cqe->wqe_counter );
456 483 : if( FD_UNLIKELY( rq_cons_idx==rq_prod_idx || wqe_counter!=(ushort)rq_cons_idx ) ) {
457 15 : errno = EPROTO;
458 15 : poll_failed = 1;
459 15 : break;
460 15 : }
461 :
462 468 : comp[ comp_cnt++ ] = (fd_mlx5_tile_rx_comp_t) {
463 468 : .chunk = rx_wq->rq_wqe_buf_chunk[ wqe_counter & (rx_wq->rx_depth-1U) ],
464 468 : .byte_len = fd_uint_bswap( rx_cqe->byte_cnt ),
465 468 : .opcode = opcode
466 468 : };
467 468 : rq_cons_idx++;
468 468 : cq_cons_idx++;
469 468 : }
470 :
471 321 : if( comp_cnt ) {
472 294 : rx_wq->rq_cons = rq_cons_idx;
473 294 : rx_cq->cons_idx = cq_cons_idx;
474 294 : FD_COMPILER_MFENCE();
475 294 : FD_VOLATILE( rx_cq->control->consumer_idx ) = fd_uint_bswap( cq_cons_idx & 0xffffffU );
476 294 : }
477 321 : return poll_failed ? -1 : (int)comp_cnt;
478 324 : }
479 :
480 : static inline int
481 : fd_mlx5_hw_poll_tx_cq( fd_mlx5_tx_qp_t * tx_qp,
482 84 : ulong * comp_bytes ) {
483 84 : fd_mlx5_cq_t * tx_cq = tx_qp->tx_cq;
484 84 : uint const cq_cons_idx = tx_cq->cons_idx;
485 84 : uint const cq_entry_idx = cq_cons_idx & (tx_cq->depth-1U);
486 :
487 84 : fd_mlx5_hw_cqe64_t const * tx_cqe = (fd_mlx5_hw_cqe64_t const *)(tx_cq->entries+cq_entry_idx);
488 :
489 84 : uchar const op_own = FD_VOLATILE_CONST( tx_cqe->op_own );
490 84 : uint const opcode = (uint)(op_own>>4);
491 :
492 84 : if( FD_LIKELY( opcode==FD_MLX5_CQE_OP_INVALID ||
493 84 : (op_own & 1U)!=!!(cq_cons_idx & tx_cq->depth) ) ) {
494 24 : return 0;
495 24 : }
496 60 : fd_mlx5_hw_dma_from_device();
497 :
498 60 : if( FD_UNLIKELY( opcode!=FD_MLX5_CQE_OP_TX_OK && opcode!=FD_MLX5_CQE_OP_TX_ERR ) ) {
499 0 : errno = EINVAL;
500 0 : return -1;
501 0 : }
502 :
503 60 : if( FD_UNLIKELY( opcode==FD_MLX5_CQE_OP_TX_ERR ) ) {
504 3 : FD_LOG_ERR(( "mlx5 TX CQE error (syndrome 0x%02x, vendor syndrome 0x%02x, "
505 3 : "WQE opcode/QPN 0x%08x, WQE counter %u)",
506 3 : (uint)tx_cqe->syndrome, (uint)tx_cqe->vendor_err, fd_uint_bswap( tx_cqe->sop_drop_qpn ),
507 3 : (uint)fd_ushort_bswap( tx_cqe->wqe_counter ) ));
508 3 : }
509 :
510 57 : uint const sq_cons_idx = tx_qp->sq_cons;
511 57 : uint const wqe_counter = (uint)fd_ushort_bswap( tx_cqe->wqe_counter );
512 57 : uint const outstanding = tx_qp->sq_posted-sq_cons_idx;
513 57 : uint const comp_cnt = (uint)(ushort)(wqe_counter-sq_cons_idx)+1U;
514 :
515 57 : if( FD_UNLIKELY( !outstanding || outstanding>tx_qp->tx_depth || comp_cnt>outstanding ) ) {
516 15 : errno = EPROTO;
517 15 : return -1;
518 15 : }
519 :
520 42 : *comp_bytes = 0UL;
521 138 : for( uint i=0U; i<comp_cnt; i++ ) {
522 96 : *comp_bytes += tx_qp->sq_wqe_frame_sz[ (sq_cons_idx+i) & (tx_qp->tx_depth-1U) ];
523 96 : }
524 42 : tx_qp->sq_cons = sq_cons_idx+comp_cnt;
525 42 : tx_cq->cons_idx = cq_cons_idx+1U;
526 42 : FD_COMPILER_MFENCE();
527 42 : FD_VOLATILE( tx_cq->control->consumer_idx ) = fd_uint_bswap( (cq_cons_idx+1U) & 0xffffffU );
528 42 : return (int)comp_cnt;
529 57 : }
530 :
531 : static inline void
532 18 : fd_mlx5_tile_sq_flush( fd_mlx5_tile_t * ctx ) {
533 18 : fd_mlx5_tx_qp_t * tx_qp = &ctx->tx_qp;
534 18 : if( FD_UNLIKELY( tx_qp->sq_posted==tx_qp->sq_prod ) ) return;
535 :
536 18 : uint const last_sq_idx = tx_qp->sq_prod-1U;
537 18 : fd_mlx5_tx_wqe_t * last_wqe = tx_qp->sq+(last_sq_idx & (tx_qp->tx_depth-1U));
538 18 : fd_mlx5_hw_ring_sq( tx_qp, last_wqe );
539 18 : }
540 :
541 : static inline void
542 : fd_mlx5_tile_rx_recycle( fd_mlx5_tile_t * ctx,
543 24912 : ulong chunk ) {
544 24912 : ctx->rq_pending_chunk[ ctx->rq_pending_cnt++ ] = (uint)chunk;
545 :
546 : /* Hold partial recycle batches: at most batch_size-1 RQ buffers stay
547 : unposted, avoiding a device-ordering barrier and doorbell-record write
548 : per packet. */
549 24912 : if( ctx->rq_pending_cnt==ctx->batch_size ) {
550 3108 : FD_TEST( !fd_mlx5_hw_post_recv( ctx, ctx->rq_pending_cnt ) );
551 3108 : ctx->rq_pending_cnt = 0U;
552 3108 : }
553 24912 : }
554 :
555 : static inline int
556 : fd_mlx5_tile_rx_dst_port_lookup( fd_mlx5_tile_t const * ctx,
557 : ushort net_dport,
558 : ulong frame_sz,
559 : ulong * out_idx,
560 237 : ulong * dst_proto ) {
561 237 : ulong rule_idx = ULONG_MAX;
562 615 : for( ulong i=0UL; i<ctx->dst_port_cnt; i++ ) {
563 585 : if( ctx->dst_ports[ i ]==net_dport ) {
564 207 : rule_idx = i;
565 207 : break;
566 207 : }
567 585 : }
568 237 : if( FD_UNLIKELY( rule_idx==ULONG_MAX ) ) return 0;
569 :
570 207 : *out_idx = ctx->dst_out_idx[ rule_idx ];
571 207 : *dst_proto = ctx->dst_protos [ rule_idx ];
572 207 : if( FD_UNLIKELY( *dst_proto==DST_PROTO_REPAIR ) ) {
573 72 : ulong const max_hdr_sz = sizeof(fd_eth_hdr_t) + 15UL*4UL /* max IHL */ + sizeof(fd_udp_hdr_t);
574 72 : ulong const min_payload_sz = frame_sz>max_hdr_sz ? frame_sz-max_hdr_sz : 0UL;
575 72 : if( FD_UNLIKELY( min_payload_sz<=AG_REPAIR_RESPONSE_MAX_SZ ) ) {
576 36 : if( FD_UNLIKELY( ctx->repair_out_idx==UCHAR_MAX ) ) return 0;
577 36 : *out_idx = ctx->repair_out_idx;
578 36 : }
579 72 : }
580 207 : return *out_idx<ctx->rx_out_cnt;
581 207 : }
582 :
583 : /* fd_mlx5_tile_rx_pkt validates and publishes an RX packet. It returns
584 : whether publication succeeded and sets freed_chunk to a reusable buffer. */
585 : static inline int
586 : fd_mlx5_tile_rx_pkt( fd_mlx5_tile_t * ctx,
587 : fd_stem_context_t * stem,
588 : ulong chunk,
589 : ulong byte_len,
590 : ulong tspub,
591 234 : ulong * freed_chunk ) {
592 234 : *freed_chunk = chunk;
593 :
594 234 : ulong const min_udp_frame_sz = sizeof(fd_eth_hdr_t)+sizeof(fd_ip4_hdr_t)+sizeof(fd_udp_hdr_t);
595 234 : if( FD_UNLIKELY( byte_len<min_udp_frame_sz || byte_len>FD_NET_MTU ) ) {
596 12 : ctx->metrics.rx_malformed_cnt++;
597 12 : return 0;
598 12 : }
599 :
600 222 : uchar * frame = fd_chunk_to_laddr( ctx->pkt_buf_wksp_base, chunk );
601 222 : fd_eth_hdr_t * eth_hdr = (fd_eth_hdr_t *)frame;
602 222 : if( FD_UNLIKELY( fd_ushort_bswap( eth_hdr->net_type )!=FD_ETH_HDR_TYPE_IP ) ) {
603 12 : ctx->metrics.rx_malformed_cnt++;
604 12 : return 0;
605 12 : }
606 :
607 210 : fd_ip4_hdr_t * ip4_hdr = (fd_ip4_hdr_t *)(eth_hdr+1);
608 210 : ulong ip4_hdr_sz = FD_IP4_GET_LEN( *ip4_hdr );
609 210 : ulong ip4_total_sz = fd_ushort_bswap( ip4_hdr->net_tot_len );
610 210 : if( FD_UNLIKELY( FD_IP4_GET_VERSION( *ip4_hdr )!=4 || ip4_hdr_sz<sizeof(fd_ip4_hdr_t) ) ) {
611 12 : ctx->metrics.rx_malformed_cnt++;
612 12 : return 0;
613 12 : }
614 198 : if( FD_UNLIKELY( ip4_total_sz<ip4_hdr_sz ||
615 198 : sizeof(fd_eth_hdr_t)+ip4_total_sz>byte_len ) ) {
616 12 : ctx->metrics.rx_malformed_cnt++;
617 12 : return 0;
618 12 : }
619 :
620 186 : ulong ctl = 0UL;
621 186 : int is_gre = ip4_hdr->protocol==FD_IP4_HDR_PROTOCOL_GRE;
622 186 : if( FD_UNLIKELY( is_gre ) ) {
623 60 : if( FD_UNLIKELY( !ctx->gre_tunnel_ip[0] ) ) {
624 12 : ctx->metrics.rx_gre_ignored_cnt++;
625 12 : return 0;
626 12 : }
627 :
628 48 : int tunnel_found = 0;
629 240 : for( ulong i=0UL; i<FD_MLX5_GRE_MAX; i++ ) tunnel_found |= ip4_hdr->saddr==ctx->gre_tunnel_ip[ i ];
630 48 : ulong const overhead = ip4_hdr_sz+sizeof(fd_gre_hdr_t);
631 48 : if( FD_UNLIKELY( !tunnel_found || !ip4_hdr->saddr ||
632 48 : overhead+sizeof(fd_ip4_hdr_t)+sizeof(fd_udp_hdr_t)>ip4_total_sz ) ) {
633 12 : ctx->metrics.rx_gre_invalid_cnt++;
634 12 : return 0;
635 12 : }
636 :
637 36 : fd_gre_hdr_t const * gre_hdr = (fd_gre_hdr_t const *)((uchar *)ip4_hdr+ip4_hdr_sz);
638 36 : if( FD_UNLIKELY( gre_hdr->flags_version!=FD_GRE_HDR_FLG_VER_BASIC ||
639 36 : gre_hdr->protocol!=fd_ushort_bswap( FD_ETH_HDR_TYPE_IP ) ) ) {
640 12 : ctx->metrics.rx_gre_invalid_cnt++;
641 12 : return 0;
642 12 : }
643 :
644 24 : frame += overhead;
645 24 : fd_memcpy( frame, eth_hdr, sizeof(fd_eth_hdr_t) );
646 :
647 24 : byte_len -= overhead;
648 24 : ctl = overhead;
649 24 : eth_hdr = (fd_eth_hdr_t *)frame;
650 24 : ip4_hdr = (fd_ip4_hdr_t *)(eth_hdr+1);
651 24 : ip4_hdr_sz = FD_IP4_GET_LEN( *ip4_hdr );
652 24 : ip4_total_sz = fd_ushort_bswap( ip4_hdr->net_tot_len );
653 24 : }
654 :
655 150 : if( FD_UNLIKELY( FD_IP4_GET_VERSION( *ip4_hdr )!=4 ||
656 150 : ip4_hdr->protocol!=FD_IP4_HDR_PROTOCOL_UDP ||
657 150 : ip4_hdr_sz<sizeof(fd_ip4_hdr_t) || ip4_total_sz<ip4_hdr_sz ||
658 150 : sizeof(fd_eth_hdr_t)+ip4_total_sz>byte_len ) ) {
659 0 : ctx->metrics.rx_malformed_cnt++;
660 0 : return 0;
661 0 : }
662 150 : if( FD_UNLIKELY( ctx->router.bind_address && ip4_hdr->daddr!=ctx->router.bind_address ) ) {
663 27 : ctx->metrics.rx_route_fail_cnt++;
664 27 : return 0;
665 27 : }
666 :
667 123 : ulong const udp_off = sizeof(fd_eth_hdr_t)+ip4_hdr_sz;
668 123 : ulong const dgram_off = udp_off+sizeof(fd_udp_hdr_t);
669 123 : if( FD_UNLIKELY( dgram_off>byte_len ) ) {
670 0 : ctx->metrics.rx_malformed_cnt++;
671 0 : return 0;
672 0 : }
673 :
674 123 : fd_udp_hdr_t const * udp_hdr = (fd_udp_hdr_t const *)((uchar const *)eth_hdr+udp_off);
675 123 : ulong const udp_sz = fd_ushort_bswap( udp_hdr->net_len );
676 123 : if( FD_UNLIKELY( udp_sz<sizeof(fd_udp_hdr_t) || udp_sz>ip4_total_sz-ip4_hdr_sz ||
677 123 : fd_ip4_addr_is_mcast( ip4_hdr->saddr ) ) ) {
678 48 : ctx->metrics.rx_malformed_cnt++;
679 48 : return 0;
680 48 : }
681 :
682 75 : ushort const dst_port = fd_ushort_bswap( udp_hdr->net_dport );
683 75 : ulong out_idx;
684 75 : ulong dst_proto;
685 75 : if( FD_UNLIKELY( !fd_mlx5_tile_rx_dst_port_lookup( ctx, dst_port, byte_len, &out_idx, &dst_proto ) ) ) {
686 24 : ctx->metrics.rx_route_fail_cnt++;
687 24 : return 0;
688 24 : }
689 :
690 51 : fd_frag_meta_t * mcache = stem->mcaches[ out_idx ];
691 51 : ulong const depth = stem->depths [ out_idx ];
692 51 : ulong const seq = stem->seqs[ out_idx ];
693 :
694 51 : *freed_chunk = mcache[ fd_mcache_line_idx( seq, depth ) ].chunk;
695 :
696 51 : ushort const src_port = fd_ushort_bswap( udp_hdr->net_sport );
697 51 : ulong const sig = fd_disco_netmux_sig( ip4_hdr->saddr, src_port, ip4_hdr->saddr, dst_proto, dgram_off );
698 51 : ulong const tsorig = 0UL;
699 :
700 51 : fd_stem_publish( stem, out_idx, sig, chunk, byte_len, ctl, tsorig, tspub );
701 :
702 51 : ctx->metrics.rx_gre_cnt += (ulong)is_gre;
703 51 : ctx->metrics.rx_pkt_cnt++;
704 51 : ctx->metrics.rx_bytes_total += byte_len;
705 :
706 51 : return 1;
707 75 : }
708 :
709 : static inline int
710 : fd_mlx5_tile_poll_rx( fd_mlx5_tile_t * ctx,
711 288 : fd_stem_context_t * stem ) {
712 288 : fd_mlx5_tile_rx_comp_t comp[ FD_MLX5_BATCH_SIZE ];
713 288 : int comp_cnt = fd_mlx5_hw_poll_rx_cq( &ctx->rx_wq, comp, ctx->batch_size );
714 :
715 288 : if( FD_UNLIKELY( comp_cnt<0 ) ) {
716 0 : FD_LOG_ERR(( "direct mlx5 RX CQ poll failed (%i-%s)", errno, fd_io_strerror( errno ) ));
717 0 : }
718 288 : if( FD_UNLIKELY( !comp_cnt ) ) return 0;
719 :
720 276 : ulong const tspub = (ulong)fd_frag_meta_ts_comp( fd_tickcount() );
721 :
722 708 : for( uint i=0U; i<(uint)comp_cnt; i++ ) {
723 432 : ulong const chunk = comp[ i ].chunk;
724 432 : if( FD_UNLIKELY( chunk<ctx->pkt_buf_chunk0 || chunk>ctx->pkt_buf_wmark ) ) {
725 0 : FD_LOG_CRIT(( "RX completion chunk %lu out of bounds [%u,%u]", chunk, ctx->pkt_buf_chunk0, ctx->pkt_buf_wmark ));
726 0 : }
727 432 : __builtin_prefetch( fd_chunk_to_laddr_const( ctx->pkt_buf_wksp_base, chunk ), 0, 3 );
728 432 : }
729 :
730 708 : for( uint i=0U; i<(uint)comp_cnt; i++ ) {
731 432 : ulong freed_chunk;
732 432 : if( FD_UNLIKELY( comp[ i ].opcode==FD_MLX5_CQE_OP_RX_ERR ) ) {
733 204 : ctx->metrics.rx_malformed_cnt++;
734 204 : freed_chunk = comp[ i ].chunk;
735 228 : } else {
736 228 : fd_mlx5_tile_rx_pkt( ctx, stem, comp[ i ].chunk, comp[ i ].byte_len, tspub, &freed_chunk );
737 228 : }
738 432 : if( FD_UNLIKELY( !freed_chunk ) ) FD_LOG_CRIT(( "invalid chunk in mcache" ));
739 432 : fd_mlx5_tile_rx_recycle( ctx, freed_chunk );
740 432 : }
741 276 : return 1;
742 276 : }
743 :
744 : static inline int
745 24 : fd_mlx5_tile_poll_tx( fd_mlx5_tile_t * ctx ) {
746 24 : ulong tx_pkt_cnt = 0UL;
747 24 : ulong tx_bytes_total = 0UL;
748 :
749 24 : int busy = 0;
750 42 : for( uint poll_cnt=0U; poll_cnt<ctx->batch_size; poll_cnt++ ) {
751 42 : ulong tx_bytes;
752 :
753 42 : int comp_cnt = fd_mlx5_hw_poll_tx_cq( &ctx->tx_qp, &tx_bytes );
754 42 : if( FD_UNLIKELY( comp_cnt<0 ) ) {
755 0 : FD_LOG_ERR(( "direct mlx5 TX CQ poll failed (%i-%s)", errno, fd_io_strerror( errno ) ));
756 0 : }
757 42 : if( FD_UNLIKELY( !comp_cnt ) ) break;
758 :
759 18 : busy = 1;
760 18 : tx_pkt_cnt += (uint)comp_cnt;
761 18 : tx_bytes_total += tx_bytes;
762 18 : }
763 24 : ctx->metrics.tx_pkt_cnt += tx_pkt_cnt;
764 24 : ctx->metrics.tx_bytes_total += tx_bytes_total;
765 :
766 24 : return busy;
767 24 : }
768 :
769 : static inline void
770 : before_credit( fd_mlx5_tile_t * ctx,
771 : fd_stem_context_t * stem,
772 30 : int * charge_busy ) {
773 30 : (void)stem;
774 30 : fd_mlx5_tx_qp_t * tx_qp = &ctx->tx_qp;
775 30 : uint const sq_pending_cnt = tx_qp->sq_prod-tx_qp->sq_posted;
776 :
777 30 : if( FD_UNLIKELY( sq_pending_cnt && fd_tickcount()>=ctx->sq_flush_deadline_ticks ) ) {
778 6 : fd_mlx5_tile_sq_flush( ctx );
779 6 : *charge_busy = 1;
780 6 : }
781 30 : if( tx_qp->sq_cons!=tx_qp->sq_posted ) {
782 24 : *charge_busy |= fd_mlx5_tile_poll_tx( ctx );
783 24 : }
784 30 : }
785 :
786 : /* after_credit is called every run loop iteration, provided there is
787 : sufficient downstream credit for forwarding on all output links. */
788 : static inline void
789 : after_credit( fd_mlx5_tile_t * ctx,
790 : fd_stem_context_t * stem,
791 : int * poll_in,
792 288 : int * charge_busy ) {
793 288 : (void)poll_in;
794 288 : int rx_busy = fd_mlx5_tile_poll_rx( ctx, stem );
795 288 : *charge_busy |= rx_busy;
796 288 : }
797 :
798 : /* before_frag resolves the TX route and checks SQ capacity */
799 : static inline int
800 : before_frag( fd_mlx5_tile_t * ctx,
801 : ulong in_idx,
802 : ulong seq,
803 162 : ulong sig ) {
804 162 : (void)seq;
805 162 : if( FD_UNLIKELY( ctx->in_kind[ in_idx ]==IN_KIND_IPROUTE ) ) return 0;
806 :
807 : /* Resolve the TX route for outgoing packets */
808 162 : ulong dst_proto = fd_disco_netmux_sig_proto( sig );
809 162 : if( FD_UNLIKELY( dst_proto!=DST_PROTO_OUTGOING ) ) return 1;
810 :
811 150 : uint const hash = (uint)fd_disco_netmux_sig_hash( sig );
812 150 : uint target_idx = hash % ctx->net_tile_cnt;
813 150 : uint const net_tile_id = ctx->net_tile_id;
814 150 : if( net_tile_id!=0U && net_tile_id!=target_idx ) return 1;
815 :
816 138 : uint dst_ip = fd_disco_netmux_sig_ip( sig );
817 138 : if( FD_UNLIKELY( !fd_net_tx_route( &ctx->router, dst_ip, &ctx->tx_route ) ) ) return 1;
818 78 : if( FD_UNLIKELY( ctx->tx_route.use_gre ) ) {
819 18 : fd_net_tx_route_t outer_route;
820 18 : uint const inner_src_ip = ctx->tx_route.src_ip;
821 18 : uint const outer_src_ip = ctx->tx_route.gre_outer_src_ip;
822 18 : uint const outer_dst_ip = ctx->tx_route.gre_outer_dst_ip;
823 :
824 18 : if( FD_UNLIKELY( !inner_src_ip || !outer_dst_ip ||
825 18 : !fd_net_tx_route( &ctx->router, outer_dst_ip, &outer_route ) ||
826 18 : outer_route.use_gre || outer_route.use_loopback ) ) {
827 12 : ctx->metrics.tx_gre_route_fail_cnt++;
828 12 : return 1;
829 12 : }
830 6 : ctx->tx_route = outer_route;
831 6 : ctx->tx_route.src_ip = inner_src_ip;
832 6 : ctx->tx_route.gre_outer_src_ip = fd_uint_if( !outer_src_ip, outer_route.src_ip, outer_src_ip );
833 6 : ctx->tx_route.gre_outer_dst_ip = outer_dst_ip;
834 6 : ctx->tx_route.use_gre = 1U;
835 6 : }
836 :
837 66 : if( ctx->tx_route.use_loopback ) target_idx = 0U;
838 66 : if( net_tile_id!=target_idx ) return 1;
839 :
840 : /* Drop the packet if no SQ WQE is available. */
841 66 : fd_mlx5_tx_qp_t const * tx_qp = &ctx->tx_qp;
842 66 : uint const outstanding = tx_qp->sq_prod-tx_qp->sq_cons;
843 66 : if( FD_UNLIKELY( outstanding>=tx_qp->tx_depth ) ) {
844 12 : ctx->metrics.tx_no_buffer_cnt++;
845 12 : return 1;
846 12 : }
847 :
848 54 : return 0; /* continue */
849 66 : }
850 :
851 : /* during_frag validates and stages the input packet or route update */
852 : static inline void
853 : during_frag( fd_mlx5_tile_t * ctx,
854 : ulong in_idx,
855 : ulong seq,
856 : ulong sig,
857 : ulong chunk,
858 : ulong frame_sz,
859 117 : ulong ctl ) {
860 117 : (void)seq; (void)sig; (void)ctl;
861 117 : fd_mlx5_tile_input_ctx_t * input_ctx = &ctx->input_ctx[ in_idx ];
862 :
863 117 : if( FD_UNLIKELY( chunk<input_ctx->chunk0 || chunk>input_ctx->wmark || frame_sz>FD_NET_MTU ) ) {
864 0 : FD_LOG_ERR(( "chunk %lu %lu corrupt, not in range [%lu,%lu]", chunk, frame_sz, input_ctx->chunk0, input_ctx->wmark ));
865 0 : }
866 :
867 117 : if( FD_UNLIKELY( ctx->in_kind[ in_idx ]==IN_KIND_IPROUTE ) ) {
868 0 : if( FD_UNLIKELY( frame_sz!=sizeof(fd_iproute_msg_t) ) ) FD_LOG_ERR(( "invalid iproute message size %lu", frame_sz ));
869 0 : fd_memcpy( &ctx->iproute_msg, fd_chunk_to_laddr_const( input_ctx->wksp_base, chunk ), sizeof(fd_iproute_msg_t) );
870 0 : return;
871 0 : }
872 :
873 117 : ulong const min_frame_sz = sizeof(fd_eth_hdr_t)+sizeof(fd_ip4_hdr_t);
874 117 : if( FD_UNLIKELY( frame_sz<min_frame_sz ) ) {
875 3 : FD_LOG_ERR(( "packet too small %lu (in_idx=%lu)", frame_sz, in_idx ));
876 114 : } else if( FD_UNLIKELY( frame_sz>FD_ETH_PAYLOAD_MAX ) ) {
877 3 : FD_LOG_ERR(( "packet too big %lu (in_idx=%lu)", frame_sz, in_idx ));
878 3 : }
879 :
880 :
881 : /* Speculatively copy frame into buffer */
882 111 : uchar const * src = fd_chunk_to_laddr_const( input_ctx->wksp_base, chunk );
883 111 : ulong dst_chunk = fd_mlx5_tile_tx_chunk( ctx, ctx->tx_qp.sq_prod );
884 111 : uchar * dst = fd_chunk_to_laddr( ctx->pkt_buf_wksp_base, dst_chunk );
885 111 : if( FD_UNLIKELY( ctx->tx_route.use_gre ) ) {
886 6 : ulong const inner_ip_off = sizeof(fd_eth_hdr_t)+sizeof(fd_ip4_hdr_t)+sizeof(fd_gre_hdr_t);
887 6 : fd_memcpy( dst+offsetof(fd_eth_hdr_t, net_type), src+offsetof(fd_eth_hdr_t, net_type), sizeof(ushort) );
888 6 : fd_memcpy( dst+inner_ip_off, src+sizeof(fd_eth_hdr_t), frame_sz-sizeof(fd_eth_hdr_t) );
889 105 : } else {
890 105 : fd_memcpy( dst, src, frame_sz );
891 105 : }
892 111 : }
893 :
894 : /* after_frag applies a route update or completes and submits the staged packet */
895 : static void
896 : after_frag( fd_mlx5_tile_t * ctx,
897 : ulong in_idx,
898 : ulong seq,
899 : ulong sig,
900 : ulong frame_sz,
901 : ulong tsorig,
902 : ulong tspub,
903 102 : fd_stem_context_t * stem ) {
904 102 : (void)seq; (void)sig; (void)tsorig; (void)tspub;
905 :
906 102 : if( FD_UNLIKELY( ctx->in_kind[ in_idx ]==IN_KIND_IPROUTE ) ) {
907 0 : fd_iproute_msg_t const * msg = &ctx->iproute_msg;
908 0 : if( msg->op==FD_IPROUTE_OP_FLUSH ) {
909 0 : fd_fib4_clear( ctx->router.fib_local );
910 0 : fd_fib4_clear( ctx->router.fib_main );
911 0 : return;
912 0 : }
913 :
914 0 : fd_fib4_t * fib;
915 0 : if( msg->table_id==RT_TABLE_LOCAL ) fib = ctx->router.fib_local;
916 0 : else if( msg->table_id==RT_TABLE_MAIN ) fib = ctx->router.fib_main;
917 0 : else return;
918 :
919 0 : if( msg->op==FD_IPROUTE_OP_UPSERT && FD_UNLIKELY( !fd_fib4_insert( fib, msg->dst_addr, msg->prefix, msg->prio, &msg->hop ) ) ) {
920 0 : FD_LOG_WARNING(( "route update dropped: route table full (increase [net.max_routes] or [net.max_peer_routes])" ));
921 0 : fd_netlink_route4_sync( ctx->router.neigh4_solicit, fd_frag_meta_ts_comp( fd_tickcount() ) );
922 0 : } else if( msg->op==FD_IPROUTE_OP_DELETE ) {
923 0 : fd_fib4_remove( fib, msg->dst_addr, msg->prefix, msg->prio );
924 0 : }
925 0 : return;
926 0 : }
927 :
928 102 : ulong chunk = fd_mlx5_tile_tx_chunk( ctx, ctx->tx_qp.sq_prod );
929 102 : uchar * frame = fd_chunk_to_laddr( ctx->pkt_buf_wksp_base, chunk );
930 102 : ulong tx_frame_sz = frame_sz;
931 102 : int fill_result;
932 :
933 102 : fd_eth_hdr_t * eth_hdr = (fd_eth_hdr_t *)frame;
934 102 : if( FD_UNLIKELY( eth_hdr->net_type!=fd_ushort_bswap( FD_ETH_HDR_TYPE_IP ) ) ) {
935 0 : FD_LOG_CRIT(( "in link %lu attempted to send packet with invalid ethertype %04x",
936 0 : in_idx, fd_ushort_bswap( eth_hdr->net_type ) ));
937 0 : }
938 :
939 102 : if( FD_UNLIKELY( ctx->tx_route.use_gre ) ) {
940 6 : ulong const inner_ip_off = sizeof(fd_eth_hdr_t)+sizeof(fd_ip4_hdr_t)+sizeof(fd_gre_hdr_t);
941 6 : fd_ip4_hdr_t * inner_ip4 = (fd_ip4_hdr_t *)(frame+inner_ip_off);
942 6 : fill_result = fd_net_tx_fill_ip4( &ctx->router, &ctx->tx_route, inner_ip4, frame_sz-sizeof(fd_eth_hdr_t) );
943 :
944 6 : if( FD_LIKELY( fill_result==FD_NET_TX_FILL_OK ) ) {
945 6 : ulong const outer_ip_sz = sizeof(fd_ip4_hdr_t)+sizeof(fd_gre_hdr_t)+frame_sz-sizeof(fd_eth_hdr_t);
946 6 : if( FD_UNLIKELY( ctx->tx_route.mtu && outer_ip_sz>ctx->tx_route.mtu ) ) {
947 0 : ctx->metrics.tx_gre_oversize_cnt++;
948 0 : return;
949 0 : }
950 :
951 6 : fd_memcpy( eth_hdr->dst, ctx->tx_route.mac_addrs, 12UL );
952 6 : eth_hdr->net_type = fd_ushort_bswap( FD_ETH_HDR_TYPE_IP );
953 :
954 6 : fd_ip4_hdr_t outer_ip4 = {
955 6 : .verihl = FD_IP4_VERIHL( 4,5 ),
956 6 : .net_tot_len = fd_ushort_bswap( (ushort)outer_ip_sz ),
957 6 : .net_frag_off = fd_ushort_bswap( FD_IP4_HDR_FRAG_OFF_DF ),
958 6 : .ttl = 64U,
959 6 : .protocol = FD_IP4_HDR_PROTOCOL_GRE,
960 6 : .saddr = ctx->tx_route.gre_outer_src_ip,
961 6 : .daddr = ctx->tx_route.gre_outer_dst_ip
962 6 : };
963 :
964 6 : if( FD_UNLIKELY( !outer_ip4.saddr || !outer_ip4.daddr ) ) {
965 0 : ctx->metrics.tx_gre_route_fail_cnt++;
966 0 : return;
967 0 : }
968 :
969 6 : outer_ip4.check = fd_ip4_hdr_check_fast( &outer_ip4 );
970 6 : FD_STORE( fd_ip4_hdr_t, frame+sizeof(fd_eth_hdr_t), outer_ip4 );
971 6 : fd_gre_hdr_t const gre_hdr = {
972 6 : .flags_version = FD_GRE_HDR_FLG_VER_BASIC,
973 6 : .protocol = fd_ushort_bswap( FD_ETH_HDR_TYPE_IP )
974 6 : };
975 6 : FD_STORE( fd_gre_hdr_t, frame+sizeof(fd_eth_hdr_t)+sizeof(fd_ip4_hdr_t), gre_hdr );
976 6 : tx_frame_sz += sizeof(fd_ip4_hdr_t)+sizeof(fd_gre_hdr_t);
977 6 : }
978 96 : } else {
979 96 : fill_result = fd_net_tx_fill_addrs( &ctx->router, &ctx->tx_route, frame, frame_sz );
980 96 : }
981 102 : if( FD_UNLIKELY( fill_result!=FD_NET_TX_FILL_OK ) ) {
982 33 : ctx->metrics.tx_invalid_cnt += (ulong)(fill_result==FD_NET_TX_FILL_INVALID);
983 33 : return;
984 33 : }
985 :
986 69 : if( FD_UNLIKELY( ctx->tx_route.use_loopback ) ) {
987 6 : ulong freed_chunk;
988 6 : if( fd_mlx5_tile_rx_pkt( ctx, stem, chunk, frame_sz, (ulong)fd_frag_meta_ts_comp( fd_tickcount() ), &freed_chunk ) ) {
989 3 : ctx->sq_wqe_buf_chunk[ ctx->tx_qp.sq_prod & (ctx->tx_qp.tx_depth-1U) ] = (uint)freed_chunk;
990 3 : }
991 6 : ctx->metrics.tx_pkt_cnt++;
992 6 : ctx->metrics.tx_bytes_total += frame_sz;
993 6 : return;
994 6 : }
995 :
996 63 : FD_TEST( !fd_mlx5_hw_post_send( ctx, (uint)tx_frame_sz ) );
997 :
998 60 : ctx->metrics.tx_gre_cnt += (ulong)ctx->tx_route.use_gre;
999 :
1000 60 : fd_mlx5_tx_qp_t * tx_qp = &ctx->tx_qp;
1001 60 : uint const sq_pending_cnt = tx_qp->sq_prod-tx_qp->sq_posted;
1002 :
1003 60 : if( sq_pending_cnt>=ctx->batch_size || tx_qp->sq_prod-tx_qp->sq_cons>=tx_qp->tx_depth ) {
1004 6 : fd_mlx5_tile_sq_flush( ctx );
1005 54 : } else if( sq_pending_cnt==1U ) {
1006 18 : ctx->sq_flush_deadline_ticks = fd_tickcount()+ctx->sq_flush_timeout_ticks;
1007 18 : }
1008 60 : }
1009 :
1010 : static inline void
1011 12 : metrics_write( fd_mlx5_tile_t * ctx ) {
1012 12 : fd_mlx5_rx_wq_t const * rx_wq = &ctx->rx_wq;
1013 12 : fd_mlx5_tx_qp_t const * tx_qp = &ctx->tx_qp;
1014 12 : ulong const rx_buffer_idle_cnt = fd_ulong_min( (ulong)(rx_wq->rq_prod-rx_wq->rq_cons), rx_wq->rx_depth );
1015 12 : ulong const rx_buffer_busy_cnt = rx_wq->rx_depth-rx_buffer_idle_cnt;
1016 12 : ulong const tx_buffer_busy_cnt = fd_ulong_min( (ulong)(tx_qp->sq_prod-tx_qp->sq_cons), tx_qp->tx_depth );
1017 12 : ulong const tx_buffer_idle_cnt = tx_qp->tx_depth-tx_buffer_busy_cnt;
1018 :
1019 12 : FD_MCNT_SET( MLX5, PKT_RX, ctx->metrics.rx_pkt_cnt );
1020 12 : FD_MCNT_SET( MLX5, PKT_RX_BYTES, ctx->metrics.rx_bytes_total );
1021 12 : FD_MCNT_SET( MLX5, PKT_RX_MALFORMED, ctx->metrics.rx_malformed_cnt );
1022 12 : FD_MCNT_SET( MLX5, PKT_RX_ROUTE_FAIL, ctx->metrics.rx_route_fail_cnt );
1023 12 : FD_MCNT_SET( MLX5, GRE_PKT_RX, ctx->metrics.rx_gre_cnt );
1024 12 : FD_MCNT_SET( MLX5, GRE_PKT_RX_INVALID, ctx->metrics.rx_gre_invalid_cnt );
1025 12 : FD_MCNT_SET( MLX5, GRE_PKT_RX_IGNORED, ctx->metrics.rx_gre_ignored_cnt );
1026 12 : FD_MGAUGE_SET( MLX5, RX_BUFFER_BUSY, rx_buffer_busy_cnt );
1027 12 : FD_MGAUGE_SET( MLX5, RX_BUFFER_IDLE, rx_buffer_idle_cnt );
1028 :
1029 12 : FD_MCNT_SET( MLX5, PKT_TX_COMPLETED, ctx->metrics.tx_pkt_cnt );
1030 12 : FD_MCNT_SET( MLX5, PKT_TX_BYTES, ctx->metrics.tx_bytes_total );
1031 12 : FD_MCNT_SET( MLX5, PKT_TX_NO_BUFFER, ctx->metrics.tx_no_buffer_cnt );
1032 12 : FD_MCNT_ENUM_COPY( MLX5, PKT_TX_ROUTE_FAIL, ctx->router.metrics.tx_route_fail_cnt );
1033 12 : FD_MCNT_SET( MLX5, PKT_TX_INVALID, ctx->metrics.tx_invalid_cnt );
1034 12 : FD_MCNT_SET( MLX5, PKT_TX_NO_NEIGHBOR, ctx->router.metrics.tx_neigh_fail_cnt );
1035 12 : FD_MCNT_SET( MLX5, GRE_PKT_TX_SUBMITTED, ctx->metrics.tx_gre_cnt );
1036 12 : FD_MCNT_SET( MLX5, GRE_PKT_TX_NO_ROUTE, ctx->metrics.tx_gre_route_fail_cnt );
1037 12 : FD_MCNT_SET( MLX5, GRE_PKT_TX_OVERSIZE, ctx->metrics.tx_gre_oversize_cnt );
1038 12 : FD_MGAUGE_SET( MLX5, TX_BUFFER_BUSY, tx_buffer_busy_cnt );
1039 12 : FD_MGAUGE_SET( MLX5, TX_BUFFER_IDLE, tx_buffer_idle_cnt );
1040 12 : }
1041 :
1042 : static void
1043 12 : fd_mlx5_tile_gre_tunnels_refresh( fd_mlx5_tile_t * ctx ) {
1044 12 : fd_memset( ctx->gre_tunnel_ip, 0, sizeof(ctx->gre_tunnel_ip) );
1045 12 : ulong tunnel_cnt = 0UL;
1046 48 : for( ushort i=0U; i<ctx->router.netdev_tbl.hdr->dev_cnt && tunnel_cnt<FD_MLX5_GRE_MAX; i++ ) {
1047 36 : fd_netdev_t const * netdev = ctx->router.netdev_tbl.dev_tbl+i;
1048 36 : if( netdev->dev_type==ARPHRD_IPGRE && netdev->gre_dst_ip ) {
1049 0 : ctx->gre_tunnel_ip[ tunnel_cnt++ ] = netdev->gre_dst_ip;
1050 0 : }
1051 36 : }
1052 12 : }
1053 :
1054 : static inline void
1055 0 : during_housekeeping( fd_mlx5_tile_t * ctx ) {
1056 : /* Drain pending uverbs async events. */
1057 0 : int const async_event_fd = ctx->uverbs.async_fd;
1058 0 : for(;;) {
1059 0 : struct ib_uverbs_async_event_desc async_event;
1060 0 : ssize_t async_event_read_sz;
1061 :
1062 0 : do async_event_read_sz = read( async_event_fd, &async_event, sizeof(async_event) );
1063 0 : while( FD_UNLIKELY( async_event_read_sz<0 && errno==EINTR ) );
1064 :
1065 0 : if( FD_LIKELY( async_event_read_sz<0 && (errno==EAGAIN || errno==EWOULDBLOCK) ) ) break;
1066 0 : if( FD_UNLIKELY( async_event_read_sz!=(ssize_t)sizeof(async_event) ) ) {
1067 0 : FD_LOG_ERR(( "mlx5 async event read failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1068 0 : }
1069 :
1070 0 : uint const async_event_type = async_event.event_type;
1071 0 : ulong const async_event_element = (ulong)async_event.element;
1072 0 : if( FD_UNLIKELY( async_event_type==FD_MLX5_ASYNC_EVENT_CQ_ERR ||
1073 0 : async_event_type==FD_MLX5_ASYNC_EVENT_QP_FATAL ||
1074 0 : async_event_type==FD_MLX5_ASYNC_EVENT_QP_REQ_ERR ||
1075 0 : async_event_type==FD_MLX5_ASYNC_EVENT_QP_ACCESS_ERR ||
1076 0 : async_event_type==FD_MLX5_ASYNC_EVENT_DEVICE_FATAL ||
1077 0 : async_event_type==FD_MLX5_ASYNC_EVENT_WQ_FATAL ) ) {
1078 0 : FD_LOG_ERR(( "fatal mlx5 async event %u on element %lu", async_event_type, async_event_element ));
1079 0 : }
1080 0 : FD_LOG_INFO(( "mlx5 async event %u on element %lu", async_event_type, async_event_element ));
1081 0 : }
1082 :
1083 : /* Refresh the netdev snapshot when its shared state is stable */
1084 0 : if( FD_LIKELY( !fd_seqlock_locked_hint( &ctx->router.netdev_shared.hdr->seqlock ) ) ) {
1085 0 : fd_netdev_tbl_copy( &ctx->router.netdev_tbl, &ctx->router.netdev_shared );
1086 0 : fd_mlx5_tile_gre_tunnels_refresh( ctx );
1087 0 : }
1088 0 : }
1089 :
1090 : static uint
1091 0 : fd_mlx5_tile_if_ip4_addr( char const * if_name ) {
1092 0 : int sock_fd = socket( AF_INET, SOCK_DGRAM, 0 );
1093 0 : if( FD_UNLIKELY( sock_fd<0 ) ) {
1094 0 : FD_LOG_ERR(( "socket(AF_INET,SOCK_DGRAM) failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1095 0 : }
1096 :
1097 0 : struct ifreq ifr = { .ifr_addr.sa_family = AF_INET };
1098 0 : fd_cstr_ncpy( ifr.ifr_name, if_name, sizeof(ifr.ifr_name) );
1099 :
1100 0 : if( FD_UNLIKELY( ioctl( sock_fd, SIOCGIFADDR, &ifr ) ) ) {
1101 0 : FD_LOG_ERR(( "could not get IP address of interface `%s` (%i-%s)",
1102 0 : if_name, errno, fd_io_strerror( errno ) ));
1103 0 : }
1104 0 : uint ip4_addr = ((struct sockaddr_in *)fd_type_pun( &ifr.ifr_addr ))->sin_addr.s_addr;
1105 0 : if( FD_UNLIKELY( close( sock_fd ) ) ) {
1106 0 : FD_LOG_ERR(( "close() failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1107 0 : }
1108 :
1109 0 : return ip4_addr;
1110 0 : }
1111 :
1112 : static ulong
1113 84 : scratch_align( void ) {
1114 84 : ulong a = alignof( fd_mlx5_tile_t );
1115 84 : a = fd_ulong_max( a, FD_MLX5_PAGE_SZ );
1116 84 : a = fd_ulong_max( a, fd_netdev_tbl_align() );
1117 84 : a = fd_ulong_max( a, fd_fib4_align() );
1118 84 : return a;
1119 84 : }
1120 :
1121 : static ulong
1122 24 : scratch_footprint( fd_topo_tile_t const * tile ) {
1123 24 : ulong layout = FD_LAYOUT_INIT;
1124 24 : ulong const queue_footprint = fd_mlx5_queue_footprint( tile->mlx5.rx_queue_size,
1125 24 : tile->mlx5.tx_queue_size );
1126 24 : if( FD_UNLIKELY( !queue_footprint ) ) return 0UL;
1127 :
1128 24 : layout = FD_LAYOUT_APPEND( layout, alignof(fd_mlx5_tile_t), sizeof(fd_mlx5_tile_t) );
1129 24 : layout = FD_LAYOUT_APPEND( layout, FD_MLX5_PAGE_SZ, queue_footprint );
1130 24 : layout = FD_LAYOUT_APPEND( layout, alignof(uint), tile->mlx5.tx_queue_size*sizeof(uint) );
1131 24 : layout = FD_LAYOUT_APPEND( layout, fd_netdev_tbl_align(), fd_netdev_tbl_footprint( NETDEV_MAX, BOND_MASTER_MAX ) );
1132 24 : layout = FD_LAYOUT_APPEND( layout, fd_fib4_align(), fd_fib4_footprint( tile->mlx5.route_max, tile->mlx5.route_peer_max ) );
1133 24 : layout = FD_LAYOUT_APPEND( layout, fd_fib4_align(), fd_fib4_footprint( tile->mlx5.route_max, tile->mlx5.route_peer_max ) );
1134 :
1135 24 : return FD_LAYOUT_FINI( layout, scratch_align() );
1136 24 : }
1137 :
1138 : fd_fib4_t *
1139 : fd_mlx5_tile_fib4_join( fd_fib4_t * out,
1140 : fd_topo_t const * topo,
1141 : fd_topo_tile_t const * tile,
1142 0 : int main_table ) {
1143 0 : FD_SCRATCH_ALLOC_INIT( scratch, fd_topo_obj_laddr( topo, tile->tile_obj_id ) );
1144 0 : (void)FD_SCRATCH_ALLOC_APPEND( scratch, alignof(fd_mlx5_tile_t), sizeof(fd_mlx5_tile_t) );
1145 0 : (void)FD_SCRATCH_ALLOC_APPEND( scratch, FD_MLX5_PAGE_SZ, fd_mlx5_queue_footprint( tile->mlx5.rx_queue_size, tile->mlx5.tx_queue_size ) );
1146 0 : (void)FD_SCRATCH_ALLOC_APPEND( scratch, alignof(uint), tile->mlx5.tx_queue_size*sizeof(uint) );
1147 0 : (void)FD_SCRATCH_ALLOC_APPEND( scratch, fd_netdev_tbl_align(), fd_netdev_tbl_footprint( NETDEV_MAX, BOND_MASTER_MAX ) );
1148 0 : void * local_mem = FD_SCRATCH_ALLOC_APPEND( scratch, fd_fib4_align(), fd_fib4_footprint( tile->mlx5.route_max, tile->mlx5.route_peer_max ) );
1149 0 : void * main_mem = FD_SCRATCH_ALLOC_APPEND( scratch, fd_fib4_align(), fd_fib4_footprint( tile->mlx5.route_max, tile->mlx5.route_peer_max ) );
1150 0 : return fd_fib4_join( out, main_table ? main_mem : local_mem );
1151 0 : }
1152 :
1153 : /* fd_mlx5_tile_rx_dst_port_add maps an IPv4 UDP destination port to the
1154 : output link with the requested name and tile kind. Published fragments
1155 : encode dst_proto in their netmux signatures. */
1156 : static void
1157 : fd_mlx5_tile_rx_dst_port_add( fd_mlx5_tile_t * ctx,
1158 : fd_topo_t const * topo,
1159 : fd_topo_tile_t const * tile,
1160 : ulong dst_proto,
1161 : char const * out_link,
1162 : ushort dst_port,
1163 432 : int required ) {
1164 432 : if( FD_UNLIKELY( !dst_port ) ) return;
1165 138 : ulong out_idx = fd_topo_find_tile_out_link( topo, tile, out_link, tile->kind_id );
1166 :
1167 138 : if( FD_UNLIKELY( out_idx==ULONG_MAX ) ) {
1168 0 : if( FD_UNLIKELY( required ) ) {
1169 0 : FD_LOG_ERR(( "mlx5 output link `%s` is missing for UDP port %hu", out_link, dst_port ));
1170 0 : }
1171 0 : return;
1172 0 : }
1173 :
1174 138 : if( FD_UNLIKELY( ctx->dst_port_cnt>=FD_MLX5_FLOW_CAP ) ) {
1175 0 : FD_LOG_ERR(( "mlx5 tile flow rule count exceeds max of %lu", FD_MLX5_FLOW_CAP ));
1176 0 : }
1177 :
1178 138 : uint const rule_idx = ctx->dst_port_cnt;
1179 138 : ctx->dst_protos [ rule_idx ] = (uchar)dst_proto;
1180 138 : ctx->dst_ports [ rule_idx ] = dst_port;
1181 138 : ctx->dst_out_idx[ rule_idx ] = (uchar)out_idx;
1182 138 : ctx->dst_port_cnt++;
1183 138 : }
1184 :
1185 : static void
1186 : fd_mlx5_tile_rx_dst_ports_init( fd_mlx5_tile_t * ctx,
1187 : fd_topo_t const * topo,
1188 48 : fd_topo_tile_t const * tile ) {
1189 48 : ctx->rx_out_cnt = (uchar)tile->out_cnt;
1190 48 : ctx->repair_out_idx = UCHAR_MAX;
1191 :
1192 48 : fd_mlx5_tile_rx_dst_port_add( ctx, topo, tile, DST_PROTO_TPU_UDP, "net_quic", tile->mlx5.net.legacy_transaction_listen_port, 1 );
1193 48 : fd_mlx5_tile_rx_dst_port_add( ctx, topo, tile, DST_PROTO_TPU_QUIC, "net_quic", tile->mlx5.net.quic_transaction_listen_port, 1 );
1194 48 : fd_mlx5_tile_rx_dst_port_add( ctx, topo, tile, DST_PROTO_SHRED, "net_shred", tile->mlx5.net.shred_listen_port, 1 );
1195 48 : fd_mlx5_tile_rx_dst_port_add( ctx, topo, tile, DST_PROTO_GOSSIP, "net_gossvf", tile->mlx5.net.gossip_listen_port, 1 );
1196 48 : fd_mlx5_tile_rx_dst_port_add( ctx, topo, tile, DST_PROTO_REPAIR, "net_shred", tile->mlx5.net.repair_client_listen_port, 1 );
1197 48 : fd_mlx5_tile_rx_dst_port_add( ctx, topo, tile, DST_PROTO_RSERVE, "net_rserve", tile->mlx5.net.repair_serve_listen_port, 0 );
1198 48 : fd_mlx5_tile_rx_dst_port_add( ctx, topo, tile, DST_PROTO_SEND, "net_txsend", tile->mlx5.net.txsend_src_port, 1 );
1199 48 : fd_mlx5_tile_rx_dst_port_add( ctx, topo, tile, DST_PROTO_VOTOR, "net_votor", tile->mlx5.net.votor_quic_client_listen_port, 1 );
1200 48 : fd_mlx5_tile_rx_dst_port_add( ctx, topo, tile, DST_PROTO_VOTOR, "net_votor", tile->mlx5.net.votor_quic_server_listen_port, 1 );
1201 :
1202 48 : if( tile->mlx5.net.repair_client_listen_port ) {
1203 36 : ulong out_idx = fd_topo_find_tile_out_link( topo, tile, "net_repair", tile->kind_id );
1204 36 : if( FD_UNLIKELY( out_idx==ULONG_MAX ) ) {
1205 0 : FD_LOG_ERR(( "mlx5 output link `net_repair` is missing for repair pings" ));
1206 0 : }
1207 36 : ctx->repair_out_idx = (uchar)out_idx;
1208 36 : }
1209 48 : }
1210 :
1211 : static void
1212 : fd_mlx5_tile_packet_memory( fd_topo_t const * topo,
1213 : fd_topo_tile_t const * tile,
1214 : fd_mlx5_tile_t * ctx,
1215 : void ** packet_memory,
1216 : ulong * packet_memory_sz,
1217 0 : ulong * packet_iova ) {
1218 0 : void * const packet_dcache_memory = fd_topo_obj_laddr( topo, tile->net.umem_dcache_obj_id );
1219 0 : void * const packet_dcache = fd_dcache_join( packet_dcache_memory );
1220 0 : if( FD_UNLIKELY( !packet_dcache ) ) FD_LOG_ERR(( "Failed to join packet dcache" ));
1221 0 : ulong const packet_dcache_data_sz = fd_dcache_data_sz( packet_dcache );
1222 0 : ulong const frame_sz = FD_NET_MTU;
1223 0 : ulong const frame_region_sz = fd_ulong_align_dn( packet_dcache_data_sz, frame_sz );
1224 :
1225 0 : void * const workspace_base = fd_wksp_containing( packet_dcache_memory );
1226 0 : if( FD_UNLIKELY( !workspace_base ) ) FD_LOG_ERR(( "Packet dcache is not in a workspace" ));
1227 0 : ulong const pkt_buf_chunk0 = ((ulong)packet_dcache-(ulong)workspace_base)>>FD_CHUNK_LG_SZ;
1228 0 : ulong const pkt_buf_wmark = pkt_buf_chunk0 + ((frame_region_sz-frame_sz)>>FD_CHUNK_LG_SZ);
1229 0 : if( FD_UNLIKELY( pkt_buf_chunk0>UINT_MAX || pkt_buf_wmark>UINT_MAX || pkt_buf_chunk0>pkt_buf_wmark ) ) {
1230 0 : FD_LOG_ERR(( "Calculated invalid packet buffer bounds [%lu,%lu]", pkt_buf_chunk0, pkt_buf_wmark ));
1231 0 : }
1232 :
1233 0 : ctx->pkt_buf_wksp_base = (uchar *)workspace_base;
1234 0 : ctx->pkt_buf_chunk0 = (uint)pkt_buf_chunk0;
1235 0 : ctx->pkt_buf_wmark = (uint)pkt_buf_wmark;
1236 0 : if( packet_memory ) *packet_memory = packet_dcache;
1237 0 : if( packet_memory_sz ) *packet_memory_sz = frame_region_sz;
1238 0 : if( packet_iova ) *packet_iova = (ulong)packet_dcache-(ulong)workspace_base;
1239 0 : }
1240 :
1241 : FD_FN_UNUSED static void
1242 : fd_mlx5_tile_join_obj_workspace( fd_topo_t * topo,
1243 0 : ulong obj_id ) {
1244 0 : fd_topo_wksp_t * wksp = &topo->workspaces[ topo->objs[ obj_id ].wksp_id ];
1245 0 : if( FD_LIKELY( !wksp->wksp ) ) {
1246 0 : fd_topo_join_workspace( topo, wksp, FD_SHMEM_JOIN_MODE_READ_WRITE, 0 );
1247 0 : }
1248 0 : }
1249 :
1250 : #ifndef FD_TILE_TEST
1251 : void
1252 : fd_topo_install_mlx5( fd_topo_t * topo,
1253 0 : fd_mlx5_fds_t * fds ) {
1254 0 : ulong const tile_cnt = fd_topo_tile_name_cnt( topo, "mlx5" );
1255 0 : if( FD_UNLIKELY( !fd_ulong_is_pow2( tile_cnt ) || tile_cnt>FD_TOPO_MAX_TILES ) ) {
1256 0 : FD_LOG_ERR(( "mlx5 tile count must be a power of two" ));
1257 0 : }
1258 :
1259 0 : fd_mlx5_tile_t * ctxs [ FD_TOPO_MAX_TILES ];
1260 0 : fd_mlx5_uverbs_tile_t queues[ FD_TOPO_MAX_TILES ];
1261 0 : fd_topo_tile_t const * first_tile = NULL;
1262 0 : char rdma_device_name[ FD_MLX5_RDMA_NAME_MAX ];
1263 0 : uint rdma_port_num = 0U;
1264 :
1265 0 : for( ulong i=0UL; i<tile_cnt; i++ ) {
1266 0 : ulong const tile_id = fd_topo_find_tile( topo, "mlx5", i );
1267 0 : FD_TEST( tile_id!=ULONG_MAX );
1268 0 : fd_topo_tile_t const * tile = topo->tiles+tile_id;
1269 0 : fd_mlx5_tile_join_obj_workspace( topo, tile->tile_obj_id );
1270 0 : fd_mlx5_tile_join_obj_workspace( topo, tile->net.umem_dcache_obj_id );
1271 :
1272 0 : FD_SCRATCH_ALLOC_INIT( scratch, fd_topo_obj_laddr( topo, tile->tile_obj_id ) );
1273 0 : fd_mlx5_tile_t * ctx = FD_SCRATCH_ALLOC_APPEND( scratch, alignof(fd_mlx5_tile_t), sizeof(fd_mlx5_tile_t) );
1274 0 : ulong const queue_memory_sz = fd_mlx5_queue_footprint( tile->mlx5.rx_queue_size, tile->mlx5.tx_queue_size );
1275 0 : void * queue_memory = FD_SCRATCH_ALLOC_APPEND( scratch, FD_MLX5_PAGE_SZ, queue_memory_sz );
1276 0 : ctxs[ i ] = ctx;
1277 0 : fd_memset( ctx, 0, sizeof(*ctx) );
1278 0 : FD_TEST( fd_mlx5_hw_init_queues( ctx, queue_memory,
1279 0 : tile->mlx5.rx_queue_size,
1280 0 : tile->mlx5.tx_queue_size ) );
1281 :
1282 0 : void * packet_memory;
1283 0 : ulong packet_memory_sz;
1284 0 : ulong packet_iova;
1285 0 : fd_mlx5_tile_packet_memory( topo, tile, ctx, &packet_memory, &packet_memory_sz, &packet_iova );
1286 0 : queues[ i ] = (fd_mlx5_uverbs_tile_t) {
1287 0 : .rx_cq = &ctx->rx_cq,
1288 0 : .tx_cq = &ctx->tx_cq,
1289 0 : .rx_wq = &ctx->rx_wq,
1290 0 : .tx_qp = &ctx->tx_qp,
1291 0 : .lkey = &ctx->lkey,
1292 0 : .packet_memory = packet_memory,
1293 0 : .packet_memory_sz = packet_memory_sz,
1294 0 : .packet_iova = packet_iova,
1295 0 : };
1296 0 : fd_mlx5_tile_rx_dst_ports_init( ctx, topo, tile );
1297 :
1298 0 : if( !i ) {
1299 0 : first_tile = tile;
1300 0 : if( FD_UNLIKELY( !fd_mlx5_rdma_dev_find( rdma_device_name, &rdma_port_num, tile->mlx5.if_name ) ) ) {
1301 0 : if( errno==EEXIST ) FD_LOG_ERR(( "multiple RDMA ports match interface `%s`", tile->mlx5.if_name ));
1302 0 : if( errno==ENOENT ) FD_LOG_ERR(( "RDMA device port for interface `%s` not found", tile->mlx5.if_name ));
1303 0 : FD_LOG_ERR(( "finding RDMA device port for interface `%s` failed (%i-%s)",
1304 0 : tile->mlx5.if_name, errno, fd_io_strerror( errno ) ));
1305 0 : }
1306 0 : }
1307 0 : }
1308 :
1309 0 : fd_mlx5_tile_t * first = ctxs[ 0 ];
1310 0 : if( FD_UNLIKELY( !fd_mlx5_uverbs_avail() ) ) {
1311 0 : FD_LOG_ERR(( "cannot run mlx5 tile: kernel does not provide uverbs API "
1312 0 : "(ib_uverbs kernel module missing or NIC not Mellanox?)" ));
1313 0 : }
1314 0 : FD_LOG_INFO(( "Opening direct mlx5 device `%s` port %u for %lu tiles",
1315 0 : rdma_device_name, rdma_port_num, tile_cnt ));
1316 0 : if( FD_UNLIKELY( !fd_uverbs_init( &first->uverbs, queues, tile_cnt,
1317 0 : &first->outer_rss_qp, &first->gre_rss_qp,
1318 0 : rdma_device_name, rdma_port_num ) ) ) {
1319 0 : FD_LOG_ERR(( "direct mlx5 setup failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1320 0 : }
1321 :
1322 0 : int const async_event_fd = first->uverbs.async_fd;
1323 0 : int const async_event_flags = fcntl( async_event_fd, F_GETFL );
1324 0 : if( FD_UNLIKELY( async_event_flags<0 ||
1325 0 : fcntl( async_event_fd, F_SETFL, async_event_flags|O_NONBLOCK )<0 ) ) {
1326 0 : FD_LOG_ERR(( "making mlx5 async fd non-blocking failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1327 0 : }
1328 :
1329 0 : for( ulong i=0UL; i<tile_cnt; i++ ) {
1330 0 : ctxs[ i ]->uverbs = first->uverbs;
1331 0 : ctxs[ i ]->outer_rss_qp = first->outer_rss_qp;
1332 0 : ctxs[ i ]->gre_rss_qp = first->gre_rss_qp;
1333 0 : ctxs[ i ]->prepared = 1U;
1334 0 : }
1335 :
1336 0 : for( ulong flow_idx=0UL; flow_idx<first->dst_port_cnt; flow_idx++ ) {
1337 0 : if( FD_UNLIKELY( fd_uverbs_create_udp_flow( &first->uverbs, &first->outer_rss_qp,
1338 0 : first_tile->mlx5.net.bind_address,
1339 0 : first->dst_ports[ flow_idx ] ) ) ) {
1340 0 : FD_LOG_ERR(( "fd_uverbs_create_udp_flow failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1341 0 : }
1342 0 : if( FD_UNLIKELY( fd_uverbs_create_gre_udp_flow( &first->uverbs, &first->gre_rss_qp,
1343 0 : first_tile->mlx5.net.bind_address,
1344 0 : first->dst_ports[ flow_idx ] ) ) ) {
1345 0 : FD_LOG_ERR(( "fd_uverbs_create_gre_udp_flow failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1346 0 : }
1347 0 : }
1348 0 : FD_LOG_INFO(( "Installed %u direct mlx5 flow rules", 2U*first->dst_port_cnt ));
1349 :
1350 0 : if( fds ) {
1351 0 : fds->cmd_fd = first->uverbs.cmd_fd;
1352 0 : fds->async_fd = first->uverbs.async_fd;
1353 0 : }
1354 0 : }
1355 : #endif
1356 :
1357 : FD_FN_UNUSED static void
1358 : privileged_init( fd_topo_t const * topo,
1359 0 : fd_topo_tile_t const * tile ) {
1360 0 : FD_SCRATCH_ALLOC_INIT( scratch, fd_topo_obj_laddr( topo, tile->tile_obj_id ) );
1361 0 : fd_mlx5_tile_t * ctx = FD_SCRATCH_ALLOC_APPEND( scratch, alignof(fd_mlx5_tile_t), sizeof(fd_mlx5_tile_t) );
1362 0 : ulong const queue_memory_sz = fd_mlx5_queue_footprint( tile->mlx5.rx_queue_size, tile->mlx5.tx_queue_size );
1363 0 : void * queue_memory = FD_SCRATCH_ALLOC_APPEND( scratch, FD_MLX5_PAGE_SZ, queue_memory_sz );
1364 0 : if( FD_UNLIKELY( !ctx->prepared ) ) {
1365 0 : if( FD_UNLIKELY( fd_topo_tile_name_cnt( topo, "mlx5" )!=1UL ) ) {
1366 0 : FD_LOG_ERR(( "multi-tile mlx5 setup must run before tile launch" ));
1367 0 : }
1368 0 : fd_topo_install_mlx5( (fd_topo_t *)topo, NULL );
1369 0 : }
1370 0 : FD_TEST( fd_mlx5_hw_join_queues( ctx, queue_memory,
1371 0 : tile->mlx5.rx_queue_size,
1372 0 : tile->mlx5.tx_queue_size ) );
1373 0 : ctx->tx_qp.sq_doorbell = fd_uverbs_map_uar( &ctx->uverbs, ctx->tx_qp.uar_mmap_offset );
1374 0 : if( FD_UNLIKELY( !ctx->tx_qp.sq_doorbell ) ) {
1375 0 : FD_LOG_ERR(( "mapping mlx5 UAR failed (%i-%s)", errno, fd_io_strerror( errno ) ));
1376 0 : }
1377 :
1378 0 : ctx->batch_size = tile->mlx5.batch_size;
1379 0 : ctx->sq_wqe_buf_chunk = FD_SCRATCH_ALLOC_APPEND( scratch, alignof(uint),
1380 0 : tile->mlx5.tx_queue_size*sizeof(uint) );
1381 :
1382 0 : fd_mlx5_tile_packet_memory( topo, tile, ctx, NULL, NULL, NULL );
1383 :
1384 : /* Resolve the netdev. */
1385 0 : uint const interface_idx = if_nametoindex( tile->mlx5.if_name );
1386 0 : if( FD_UNLIKELY( !interface_idx ) ) {
1387 0 : FD_LOG_ERR(( "if_nametoindex(%s) failed (%i-%s)",
1388 0 : tile->mlx5.if_name, errno, fd_io_strerror( errno ) ));
1389 0 : }
1390 0 : ctx->router.if_virt = interface_idx;
1391 0 : ctx->router.default_address = fd_mlx5_tile_if_ip4_addr( tile->mlx5.if_name );
1392 0 : }
1393 :
1394 : FD_FN_UNUSED static void
1395 : unprivileged_init( fd_topo_t const * topo,
1396 12 : fd_topo_tile_t const * tile ) {
1397 12 : FD_SCRATCH_ALLOC_INIT( scratch, fd_topo_obj_laddr( topo, tile->tile_obj_id ) );
1398 12 : fd_mlx5_tile_t * ctx = FD_SCRATCH_ALLOC_APPEND( scratch, alignof(fd_mlx5_tile_t), sizeof(fd_mlx5_tile_t) );
1399 12 : ulong queue_footprint = fd_mlx5_queue_footprint( tile->mlx5.rx_queue_size, tile->mlx5.tx_queue_size );
1400 12 : (void)FD_SCRATCH_ALLOC_APPEND( scratch, FD_MLX5_PAGE_SZ, queue_footprint );
1401 :
1402 12 : ctx->sq_wqe_buf_chunk = FD_SCRATCH_ALLOC_APPEND( scratch, alignof(uint), tile->mlx5.tx_queue_size*sizeof(uint) );
1403 12 : ctx->batch_size = tile->mlx5.batch_size;
1404 12 : ctx->sq_flush_timeout_ticks = (long)( FD_MLX5_TX_FLUSH_TIMEOUT_NS*fd_tempo_tick_per_ns( NULL ) );
1405 12 : ctx->net_tile_id = (uint)tile->kind_id;
1406 12 : ctx->net_tile_cnt = (uint)fd_topo_tile_name_cnt( topo, tile->name );
1407 :
1408 12 : void * netdev_tbl_local = FD_SCRATCH_ALLOC_APPEND( scratch, fd_netdev_tbl_align(), fd_netdev_tbl_footprint( NETDEV_MAX, BOND_MASTER_MAX ) );
1409 12 : void * fib_local_mem = FD_SCRATCH_ALLOC_APPEND( scratch, fd_fib4_align(), fd_fib4_footprint( tile->mlx5.route_max, tile->mlx5.route_peer_max ) );
1410 12 : void * fib_main_mem = FD_SCRATCH_ALLOC_APPEND( scratch, fd_fib4_align(), fd_fib4_footprint( tile->mlx5.route_max, tile->mlx5.route_peer_max ) );
1411 :
1412 : /* chunk 0 is used as a sentinel value, so ensure actual chunk indices
1413 : do not use that value. */
1414 12 : FD_TEST( ctx->pkt_buf_chunk0>0 );
1415 :
1416 12 : ctx->rq_pending_cnt = 0U;
1417 :
1418 : /* Post RQ WQEs */
1419 12 : ulong frame_chunks = FD_NET_MTU>>FD_CHUNK_LG_SZ;
1420 12 : ulong next_chunk = ctx->pkt_buf_chunk0;
1421 :
1422 12 : FD_TEST( tile->mlx5.rx_queue_size>ctx->batch_size );
1423 :
1424 12 : ulong const rx_fill_cnt = tile->mlx5.rx_queue_size-ctx->batch_size;
1425 24492 : for( ulong i=0UL; i<rx_fill_cnt; i++ ) {
1426 24480 : fd_mlx5_tile_rx_recycle( ctx, next_chunk );
1427 24480 : next_chunk += frame_chunks;
1428 24480 : }
1429 :
1430 : /* Assign chunks to RX mcaches */
1431 36 : for( ulong i=0UL; i<(tile->out_cnt); i++ ) {
1432 24 : fd_frag_meta_t * mcache = topo->links[ tile->out_link_id[ i ] ].mcache;
1433 24 : ulong const depth = fd_mcache_depth( mcache );
1434 3096 : for( ulong j=0UL; j<depth; j++ ) {
1435 3072 : mcache[ j ].chunk = (uint)next_chunk;
1436 3072 : mcache[ j ].seq = fd_seq_dec( j, 1UL ); /* mark seq as invalid */
1437 3072 : next_chunk += frame_chunks;
1438 3072 : }
1439 24 : }
1440 : /* Assign one TX buffer to each SQ WQE */
1441 12300 : for( ulong i=0UL; i<tile->mlx5.tx_queue_size; i++ ) {
1442 12288 : ctx->sq_wqe_buf_chunk[ i ] = (uint)next_chunk;
1443 12288 : next_chunk += frame_chunks;
1444 12288 : }
1445 : /* Init TX */
1446 12 : if( FD_UNLIKELY( tile->in_cnt>FD_TOPO_MAX_TILE_IN_LINKS ) ) {
1447 0 : FD_LOG_ERR(( "mlx5 tile in link count %lu exceeds max of %lu", tile->in_cnt, FD_TOPO_MAX_TILE_IN_LINKS ));
1448 0 : }
1449 24 : for( ulong i=0UL; i<(tile->in_cnt); i++ ) {
1450 12 : fd_topo_link_t const * link = &topo->links[ tile->in_link_id[ i ] ];
1451 12 : if( !strcmp( link->name, "iproute_out" ) ) {
1452 0 : ctx->in_kind[ i ] = IN_KIND_IPROUTE;
1453 12 : } else {
1454 12 : ctx->in_kind[ i ] = IN_KIND_NET;
1455 12 : if( FD_UNLIKELY( link->mtu!=FD_NET_MTU ) ) FD_LOG_ERR(( "mlx5 tile in link does not have a normal MTU" ));
1456 12 : }
1457 :
1458 12 : ctx->input_ctx[ i ].wksp_base = topo->workspaces[ topo->objs[ link->dcache_obj_id ].wksp_id ].wksp;
1459 12 : ctx->input_ctx[ i ].chunk0 = fd_dcache_compact_chunk0( ctx->input_ctx[ i ].wksp_base, link->dcache );
1460 12 : ctx->input_ctx[ i ].wmark = fd_dcache_compact_wmark( ctx->input_ctx[ i ].wksp_base, link->dcache, link->mtu );
1461 12 : }
1462 :
1463 : /* Join netbase objects */
1464 12 : FD_TEST( fd_fib4_join( ctx->router.fib_local, fd_fib4_new( fib_local_mem, tile->mlx5.route_max, tile->mlx5.route_peer_max, tile->mlx5.route_peer_seed ) ) );
1465 12 : FD_TEST( fd_fib4_join( ctx->router.fib_main, fd_fib4_new( fib_main_mem, tile->mlx5.route_max, tile->mlx5.route_peer_max, tile->mlx5.route_peer_seed ) ) );
1466 12 : FD_TEST( fd_netdev_tbl_join( &ctx->router.netdev_shared, fd_topo_obj_laddr( topo, tile->mlx5.netdev_tbl_obj_id ) ) );
1467 12 : FD_TEST( fd_netdev_tbl_new( netdev_tbl_local, NETDEV_MAX, BOND_MASTER_MAX ) );
1468 12 : FD_TEST( fd_netdev_tbl_join( &ctx->router.netdev_tbl, netdev_tbl_local ) );
1469 :
1470 12 : fd_netdev_tbl_copy( &ctx->router.netdev_tbl, &ctx->router.netdev_shared );
1471 12 : fd_mlx5_tile_gre_tunnels_refresh( ctx );
1472 12 : ctx->router.bind_address = tile->mlx5.net.bind_address;
1473 :
1474 12 : ulong neigh4_obj_id = tile->mlx5.neigh4_obj_id;
1475 12 : ulong ele_max = fd_pod_queryf_ulong( topo->props, ULONG_MAX, "obj.%lu.ele_max", neigh4_obj_id );
1476 12 : ulong probe_max = fd_pod_queryf_ulong( topo->props, ULONG_MAX, "obj.%lu.probe_max", neigh4_obj_id );
1477 12 : ulong seed = fd_pod_queryf_ulong( topo->props, ULONG_MAX, "obj.%lu.seed", neigh4_obj_id );
1478 12 : if( FD_UNLIKELY( (ele_max==ULONG_MAX) | (probe_max==ULONG_MAX) | (seed==ULONG_MAX) ) ) {
1479 0 : FD_LOG_ERR(( "neigh4 hmap properties not set" ));
1480 0 : }
1481 12 : if( FD_UNLIKELY( !fd_neigh4_hmap_join( ctx->router.neigh4, fd_topo_obj_laddr( topo, neigh4_obj_id ), ele_max, probe_max, seed ) ) ) {
1482 0 : FD_LOG_ERR(( "fd_neigh4_hmap_join failed" ));
1483 0 : }
1484 :
1485 12 : ulong net_netlnk_id = ULONG_MAX;
1486 36 : for( ulong i=0UL; i<tile->out_cnt; i++ ) {
1487 24 : if( !strcmp( topo->links[ tile->out_link_id[ i ] ].name, "net_netlnk" ) ) net_netlnk_id = tile->out_link_id[ i ];
1488 24 : }
1489 12 : if( FD_LIKELY( net_netlnk_id!=ULONG_MAX ) ) {
1490 12 : fd_topo_link_t const * net_netlnk = &topo->links[ net_netlnk_id ];
1491 12 : if( FD_UNLIKELY( !net_netlnk->mcache ) ) FD_LOG_ERR(( "netlink request link not initialized" ));
1492 :
1493 12 : ctx->router.neigh4_solicit->mcache = net_netlnk->mcache;
1494 12 : ctx->router.neigh4_solicit->depth = fd_mcache_depth( ctx->router.neigh4_solicit->mcache );
1495 12 : ctx->router.neigh4_solicit->seq = fd_mcache_seq_query( fd_mcache_seq_laddr( ctx->router.neigh4_solicit->mcache ) );
1496 12 : } else {
1497 0 : FD_LOG_ERR(( "netlink request link not found" ));
1498 0 : }
1499 :
1500 : /* Check if all chunks are in bound */
1501 12 : if( FD_UNLIKELY( next_chunk>ctx->pkt_buf_wmark ) ) {
1502 0 : FD_LOG_ERR(( "dcache is too small (topology bug)" ));
1503 0 : }
1504 :
1505 12 : ulong scratch_top = FD_SCRATCH_ALLOC_FINI( scratch, scratch_align() );
1506 12 : if( FD_UNLIKELY( scratch_top>(ulong)ctx+scratch_footprint( tile ) ) ) {
1507 0 : FD_LOG_ERR(( "scratch overflow" ));
1508 0 : }
1509 12 : }
1510 :
1511 : FD_FN_UNUSED static ulong
1512 : populate_allowed_seccomp( fd_topo_t const * topo,
1513 : fd_topo_tile_t const * tile,
1514 : ulong out_cnt,
1515 0 : struct sock_filter * out ) {
1516 0 : fd_mlx5_tile_t * ctx = fd_topo_obj_laddr( topo, tile->tile_obj_id );
1517 0 : populate_sock_filter_policy_fd_mlx5_tile( out_cnt, out, (uint)fd_log_private_logfile_fd(),
1518 0 : (uint)ctx->uverbs.async_fd, UINT_MAX );
1519 0 : return sock_filter_policy_fd_mlx5_tile_instr_cnt;
1520 0 : }
1521 :
1522 : FD_FN_UNUSED static ulong
1523 : populate_allowed_fds( fd_topo_t const * topo,
1524 : fd_topo_tile_t const * tile,
1525 : ulong out_fds_cnt,
1526 0 : int * out_fds ) {
1527 0 : fd_mlx5_tile_t * ctx = fd_topo_obj_laddr( topo, tile->tile_obj_id );
1528 0 : if( FD_UNLIKELY( out_fds_cnt<4UL ) ) FD_LOG_ERR(( "out_fds_cnt %lu", out_fds_cnt ));
1529 0 : ulong out_cnt = 0UL;
1530 0 : out_fds[ out_cnt++ ] = 2;
1531 0 : if( FD_LIKELY( fd_log_private_logfile_fd()!=-1 ) ) out_fds[ out_cnt++ ] = fd_log_private_logfile_fd();
1532 0 : out_fds[ out_cnt++ ] = ctx->uverbs.cmd_fd;
1533 0 : out_fds[ out_cnt++ ] = ctx->uverbs.async_fd;
1534 0 : return out_cnt;
1535 0 : }
1536 :
1537 0 : #define STEM_CALLBACK_CONTEXT_TYPE fd_mlx5_tile_t
1538 0 : #define STEM_CALLBACK_CONTEXT_ALIGN alignof(fd_mlx5_tile_t)
1539 0 : #define STEM_CALLBACK_BEFORE_CREDIT before_credit
1540 0 : #define STEM_CALLBACK_AFTER_CREDIT after_credit
1541 0 : #define STEM_CALLBACK_BEFORE_FRAG before_frag
1542 0 : #define STEM_CALLBACK_DURING_FRAG during_frag
1543 0 : #define STEM_CALLBACK_AFTER_FRAG after_frag
1544 0 : #define STEM_CALLBACK_METRICS_WRITE metrics_write
1545 0 : #define STEM_CALLBACK_DURING_HOUSEKEEPING during_housekeeping
1546 0 : #define STEM_BURST FD_MLX5_BATCH_SIZE
1547 0 : #define STEM_LAZY 270000UL /* 270us */
1548 : #include "../../stem/fd_stem.c"
1549 :
1550 : #ifndef FD_TILE_TEST
1551 : fd_topo_run_tile_t fd_tile_mlx5 = {
1552 : .name = "mlx5",
1553 : .populate_allowed_seccomp = populate_allowed_seccomp,
1554 : .populate_allowed_fds = populate_allowed_fds,
1555 : .scratch_align = scratch_align,
1556 : .scratch_footprint = scratch_footprint,
1557 : .privileged_init = privileged_init,
1558 : .unprivileged_init = unprivileged_init,
1559 : .run = stem_run,
1560 : };
1561 : #endif
|