Branch data Line data Source code
1 : : /* SPDX-License-Identifier: BSD-3-Clause
2 : : * Copyright 2018-2026 NXP
3 : : */
4 : :
5 : : #include <stdbool.h>
6 : : #include <stdint.h>
7 : : #include <unistd.h>
8 : :
9 : : #include "rte_ethdev.h"
10 : : #include "rte_malloc.h"
11 : : #include "rte_memzone.h"
12 : :
13 : : #include "base/enetc_hw.h"
14 : : #include "base/enetc4_hw.h"
15 : : #include "enetc.h"
16 : : #include "enetc_logs.h"
17 : :
18 : : #define ENETC_CACHE_LINE_RXBDS (RTE_CACHE_LINE_SIZE / \
19 : : sizeof(union enetc_rx_bd))
20 : : #define ENETC_RXBD_BUNDLE 16 /* Number of buffers to allocate at once */
21 : :
22 : : static int
23 : 0 : enetc_clean_tx_ring(struct enetc_bdr *tx_ring)
24 : : {
25 : : int tx_frm_cnt = 0;
26 : : struct enetc_swbd *tx_swbd, *tx_swbd_base;
27 : : int i, hwci, bd_count;
28 : : struct rte_mbuf *m[ENETC_RXBD_BUNDLE];
29 : : struct enetc_tx_bd *txbd;
30 : :
31 : : /* we don't need barriers here, we just want a relatively current value
32 : : * from HW.
33 : : */
34 : 0 : hwci = (int)(rte_read32_relaxed(tx_ring->tcisr) &
35 : : ENETC_TBCISR_IDX_MASK);
36 : :
37 : 0 : tx_swbd_base = tx_ring->q_swbd;
38 : 0 : bd_count = tx_ring->bd_count;
39 : 0 : i = tx_ring->next_to_clean;
40 : 0 : tx_swbd = &tx_swbd_base[i];
41 : :
42 : : /* we're only reading the CI index once here, which means HW may update
43 : : * it while we're doing clean-up. We could read the register in a loop
44 : : * but for now I assume it's OK to leave a few Tx frames for next call.
45 : : * The issue with reading the register in a loop is that we're stalling
46 : : * here trying to catch up with HW which keeps sending traffic as long
47 : : * as it has traffic to send, so in effect we could be waiting here for
48 : : * the Tx ring to be drained by HW, instead of us doing Rx in that
49 : : * meantime.
50 : : */
51 [ # # ]: 0 : while (i != hwci) {
52 : : /* It seems calling rte_pktmbuf_free is wasting a lot of cycles,
53 : : * make a list and call _free when it's done.
54 : : */
55 : : /* Clear flags on the reclaimed BD so that dcbf in the
56 : : * cacheable TX path never flushes a stale flags_F to memory
57 : : * before the new BD fields are fully written.
58 : : */
59 : 0 : txbd = ENETC_TXBD(*tx_ring, i);
60 : 0 : txbd->flags = 0;
61 : :
62 [ # # ]: 0 : if (tx_frm_cnt == ENETC_RXBD_BUNDLE) {
63 : 0 : rte_pktmbuf_free_bulk(m, tx_frm_cnt);
64 : : tx_frm_cnt = 0;
65 : : }
66 : :
67 : 0 : m[tx_frm_cnt] = tx_swbd->buffer_addr;
68 : 0 : tx_swbd->buffer_addr = NULL;
69 : :
70 : 0 : i++;
71 : 0 : tx_swbd++;
72 [ # # ]: 0 : if (unlikely(i == bd_count)) {
73 : : i = 0;
74 : : tx_swbd = tx_swbd_base;
75 : : }
76 : :
77 : 0 : tx_frm_cnt++;
78 : : }
79 : :
80 [ # # ]: 0 : if (tx_frm_cnt)
81 : 0 : rte_pktmbuf_free_bulk(m, tx_frm_cnt);
82 : :
83 : 0 : tx_ring->next_to_clean = i;
84 : :
85 : 0 : return 0;
86 : : }
87 : :
88 : : uint16_t
89 : 0 : enetc_xmit_pkts(void *tx_queue,
90 : : struct rte_mbuf **tx_pkts,
91 : : uint16_t nb_pkts)
92 : : {
93 : : struct enetc_swbd *tx_swbd;
94 : : int i, start, bds_to_use;
95 : : struct enetc_tx_bd *txbd;
96 : : struct enetc_bdr *tx_ring = (struct enetc_bdr *)tx_queue;
97 : :
98 [ # # ]: 0 : i = tx_ring->next_to_use;
99 : :
100 : : bds_to_use = enetc_bd_unused(tx_ring);
101 [ # # ]: 0 : if (bds_to_use < nb_pkts)
102 : 0 : nb_pkts = bds_to_use;
103 : :
104 : : start = 0;
105 [ # # ]: 0 : while (nb_pkts--) {
106 : 0 : tx_ring->q_swbd[i].buffer_addr = tx_pkts[start];
107 : :
108 : 0 : txbd = ENETC_TXBD(*tx_ring, i);
109 : : tx_swbd = &tx_ring->q_swbd[i];
110 : 0 : txbd->frm_len = tx_pkts[start]->pkt_len;
111 : 0 : txbd->buf_len = txbd->frm_len;
112 : 0 : txbd->flags = ENETC_TXBD_FLAGS_F;
113 : 0 : txbd->addr = (uint64_t)(uintptr_t)
114 : 0 : rte_cpu_to_le_64((size_t)tx_swbd->buffer_addr->buf_iova +
115 : : tx_swbd->buffer_addr->data_off);
116 : 0 : i++;
117 : 0 : start++;
118 [ # # ]: 0 : if (unlikely(i == tx_ring->bd_count))
119 : : i = 0;
120 : : }
121 : :
122 : : /* we're only cleaning up the Tx ring here, on the assumption that
123 : : * software is slower than hardware and hardware completed sending
124 : : * older frames out by now.
125 : : * We're also cleaning up the ring before kicking off Tx for the new
126 : : * batch to minimize chances of contention on the Tx ring
127 : : */
128 : 0 : enetc_clean_tx_ring(tx_ring);
129 : :
130 : 0 : tx_ring->next_to_use = i;
131 : 0 : enetc_wr_reg(tx_ring->tcir, i);
132 : 0 : return start;
133 : : }
134 : :
135 : : static void
136 : 0 : enetc4_tx_offload_checksum(struct rte_mbuf *mbuf, struct enetc_tx_bd *txbd)
137 : : {
138 [ # # ]: 0 : if ((mbuf->ol_flags & (RTE_MBUF_F_TX_IP_CKSUM | RTE_MBUF_F_TX_IPV4))
139 : : == ENETC4_MBUF_F_TX_IP_IPV4) {
140 : 0 : txbd->l3t = ENETC4_TXBD_L3T;
141 : 0 : txbd->ipcs = ENETC4_TXBD_IPCS;
142 : 0 : txbd->l3_start = mbuf->l2_len;
143 : 0 : txbd->l3_hdr_size = mbuf->l3_len / 4;
144 : 0 : txbd->flags |= ENETC4_TXBD_FLAGS_L_TX_CKSUM;
145 [ # # ]: 0 : if ((mbuf->ol_flags & RTE_MBUF_F_TX_UDP_CKSUM) == RTE_MBUF_F_TX_UDP_CKSUM) {
146 : 0 : txbd->l4t = ENETC4_TXBD_L4T_UDP;
147 : 0 : txbd->flags |= ENETC4_TXBD_FLAGS_L4CS;
148 [ # # ]: 0 : } else if ((mbuf->ol_flags & RTE_MBUF_F_TX_TCP_CKSUM) == RTE_MBUF_F_TX_TCP_CKSUM) {
149 : 0 : txbd->l4t = ENETC4_TXBD_L4T_TCP;
150 : 0 : txbd->flags |= ENETC4_TXBD_FLAGS_L4CS;
151 : : }
152 : : }
153 : 0 : }
154 : :
155 : : uint16_t
156 : 0 : enetc_xmit_pkts_nc(void *tx_queue,
157 : : struct rte_mbuf **tx_pkts,
158 : : uint16_t nb_pkts)
159 : : {
160 : : struct enetc_bdr *tx_ring = (struct enetc_bdr *)tx_queue;
161 : : int i, start, bds_to_use, bd_count;
162 : : struct enetc_tx_bd *txbd = NULL;
163 : : struct rte_mbuf *seg;
164 : : uint16_t seg_len, segs_per_pkt;
165 : : bool is_first_seg;
166 : : unsigned int j;
167 : : uint8_t *data;
168 : :
169 [ # # ]: 0 : i = tx_ring->next_to_use;
170 : : bds_to_use = enetc_bd_unused(tx_ring);
171 : 0 : bd_count = tx_ring->bd_count;
172 : :
173 : : start = 0;
174 [ # # ]: 0 : while (start < nb_pkts) {
175 : 0 : seg = tx_pkts[start];
176 : 0 : segs_per_pkt = seg->nb_segs;
177 : :
178 [ # # ]: 0 : if (bds_to_use < segs_per_pkt)
179 : : break;
180 : :
181 : : is_first_seg = true;
182 [ # # ]: 0 : while (seg) {
183 : 0 : tx_ring->q_swbd[i].buffer_addr = NULL;
184 : 0 : seg_len = rte_pktmbuf_data_len(seg);
185 : 0 : data = rte_pktmbuf_mtod(seg, void *);
186 : :
187 : : /* Flush payload to PoC so HW DMA reads the correct data. */
188 [ # # ]: 0 : for (j = 0; j < seg_len; j += RTE_CACHE_LINE_SIZE)
189 : : dcbf(data + j);
190 : : /* Cover the last byte of an unaligned buffer. */
191 : : dcbf(data + (seg_len - 1));
192 : :
193 : 0 : txbd = ENETC_TXBD(*tx_ring, i);
194 : 0 : txbd->flags = 0;
195 [ # # ]: 0 : if (is_first_seg) {
196 : 0 : tx_ring->q_swbd[i].buffer_addr = tx_pkts[start];
197 : 0 : txbd->frm_len = rte_pktmbuf_pkt_len(seg);
198 [ # # ]: 0 : if (seg->ol_flags & ENETC4_TX_CKSUM_OFFLOAD_MASK)
199 : 0 : enetc4_tx_offload_checksum(seg, txbd);
200 : : is_first_seg = false;
201 : : }
202 : :
203 [ # # ]: 0 : txbd->buf_len = rte_cpu_to_le_16(seg_len);
204 : 0 : txbd->addr = rte_cpu_to_le_64(rte_mbuf_data_iova(seg));
205 : 0 : seg = seg->next;
206 : 0 : i++;
207 : 0 : bds_to_use--;
208 [ # # ]: 0 : if (unlikely(i == bd_count))
209 : : i = 0;
210 : : }
211 : :
212 : : /* Set the frame-last flag on the final BD of this packet. */
213 [ # # ]: 0 : if (likely(txbd))
214 : 0 : txbd->flags |= ENETC4_TXBD_FLAGS_F;
215 : 0 : start++;
216 : : }
217 : :
218 : 0 : enetc_clean_tx_ring(tx_ring);
219 : 0 : tx_ring->next_to_use = i;
220 : 0 : enetc_wr_reg(tx_ring->tcir, i);
221 : 0 : return start;
222 : : }
223 : :
224 : : /*
225 : : * LSO (Large Send Offload) Tx burst.
226 : : *
227 : : * Dedicated burst used on Tx rings with LSO enabled (txr->lso_enable). It
228 : : * follows the same non-cache-coherent discipline as enetc_xmit_pkts_cacheable:
229 : : * payload lines are flushed with dcbf before handing the frame to HW and the
230 : : * written BD cache lines are flushed at the end of the batch.
231 : : *
232 : : * A TSO/USO frame uses an extended Tx descriptor: the standard BD points at
233 : : * the L2/L3/L4 header template (BUF_LEN = header length, FRM_LEN = payload
234 : : * length), followed by an extension BD in the next ring slot holding the
235 : : * segment size, then one BD per payload buffer with the F flag on the last.
236 : : * Non-TSO packets fall through to the normal encoding so mixed traffic works.
237 : : */
238 : : uint16_t
239 : 0 : enetc_xmit_pkts_lso(void *tx_queue,
240 : : struct rte_mbuf **tx_pkts,
241 : : uint16_t nb_pkts)
242 : : {
243 : : int i, start, bds_to_use;
244 : : struct enetc_tx_bd *txbd = NULL;
245 : : struct enetc_tx_bd_ext *txbd_ext;
246 : : struct enetc_bdr *tx_ring = (struct enetc_bdr *)tx_queue;
247 : : unsigned int j;
248 : : uint8_t *data;
249 : : struct rte_mbuf *seg;
250 : : uint16_t seg_len, segs_per_pkt;
251 : : bool is_first_seg;
252 : : int first_bd_idx, bd_count;
253 : :
254 [ # # ]: 0 : i = tx_ring->next_to_use;
255 : : bds_to_use = enetc_bd_unused(tx_ring);
256 : 0 : bd_count = tx_ring->bd_count;
257 : : start = 0;
258 : :
259 : : first_bd_idx = i;
260 : :
261 [ # # ]: 0 : while (start < nb_pkts) {
262 : 0 : seg = tx_pkts[start];
263 : 0 : segs_per_pkt = seg->nb_segs;
264 : :
265 [ # # ]: 0 : if (seg->ol_flags &
266 : 0 : (RTE_MBUF_F_TX_TCP_SEG | RTE_MBUF_F_TX_UDP_SEG)) {
267 : : bool is_udp_seg =
268 : 0 : !!(seg->ol_flags & RTE_MBUF_F_TX_UDP_SEG);
269 : 0 : uint32_t hdr_len = seg->l2_len + seg->l3_len +
270 : 0 : seg->l4_len;
271 : : uint32_t data_unit;
272 : : uint16_t first_payload;
273 : : struct rte_mbuf *dseg;
274 : : int bds_needed;
275 : :
276 : : /* Header BD + extension BD + one BD per segment. */
277 : 0 : bds_needed = 2 + segs_per_pkt;
278 [ # # ]: 0 : if (bds_to_use < bds_needed)
279 : : break;
280 : :
281 : : /*
282 : : * Skip frames with nothing to segment, or whose
283 : : * headers are not fully contained in the first
284 : : * segment. The first BD carries the header template
285 : : * and first_payload is computed as (first segment
286 : : * data_len - hdr_len), so the L2/L3/L4 headers must
287 : : * reside contiguously in the first mbuf; otherwise the
288 : : * unsigned subtraction would underflow.
289 : : */
290 [ # # # # ]: 0 : if (unlikely(hdr_len >= rte_pktmbuf_pkt_len(seg) ||
291 : : hdr_len > rte_pktmbuf_data_len(seg))) {
292 : 0 : rte_pktmbuf_free(seg);
293 : 0 : start++;
294 : 0 : continue;
295 : : }
296 : 0 : data_unit = rte_pktmbuf_pkt_len(seg) - hdr_len;
297 : :
298 : : /*
299 : : * Skip frames that violate HW LSO limits: a zero
300 : : * segment size, a payload larger than the HW data
301 : : * unit, or a per-segment frame (headers + segment)
302 : : * bigger than the maximum LSO frame size.
303 : : */
304 [ # # # # : 0 : if (unlikely(seg->tso_segsz == 0 ||
# # ]
305 : : data_unit > ENETC4_LSO_MAX_DATA_UNIT ||
306 : : hdr_len + seg->tso_segsz >
307 : : ENETC4_LSO_MAX_FRAME)) {
308 : 0 : rte_pktmbuf_free(seg);
309 : 0 : start++;
310 : 0 : continue;
311 : : }
312 : :
313 : : /* Standard (first) BD: header template only. */
314 : : data = rte_pktmbuf_mtod(seg, void *);
315 : :
316 : : seg_len = rte_pktmbuf_data_len(seg);
317 [ # # ]: 0 : for (j = 0; j < seg_len; j += RTE_CACHE_LINE_SIZE)
318 : : dcbf(data + j);
319 : : dcbf(data + (seg_len - 1));
320 : :
321 [ # # ]: 0 : txbd = ENETC_TXBD(*tx_ring, i);
322 : : memset(txbd, 0, sizeof(*txbd));
323 : 0 : tx_ring->q_swbd[i].buffer_addr = seg;
324 : :
325 : : /* FRM_LEN low 16 bits = payload length, BUF_LEN = hdr. */
326 : 0 : txbd->frm_len = rte_cpu_to_le_16(data_unit & 0xffff);
327 [ # # ]: 0 : txbd->buf_len = rte_cpu_to_le_16((uint16_t)hdr_len);
328 : 0 : txbd->addr = rte_cpu_to_le_64(rte_mbuf_data_iova(seg));
329 : :
330 : : /* Checksum control for the per-segment L3/L4 rewrite. */
331 : 0 : txbd->l3_start = seg->l2_len;
332 : 0 : txbd->l3_hdr_size = seg->l3_len / 4;
333 [ # # ]: 0 : if (seg->ol_flags & RTE_MBUF_F_TX_IPV6) {
334 : 0 : txbd->l3t = 1;
335 : : } else {
336 : 0 : txbd->l3t = ENETC4_TXBD_L3T;
337 : 0 : txbd->ipcs = ENETC4_TXBD_IPCS;
338 : : }
339 [ # # ]: 0 : txbd->l4t = is_udp_seg ? ENETC4_TXBD_L4T_UDP :
340 : : ENETC4_TXBD_L4T_TCP;
341 : 0 : txbd->flags = ENETC4_TXBD_FLAGS_L_TX_CKSUM |
342 : : ENETC4_TXBD_FLAGS_L4CS |
343 : : ENETC4_TXBD_FLAGS_LSO |
344 : : ENETC4_TXBD_FLAGS_EXT;
345 : :
346 : 0 : i++;
347 : : bds_to_use--;
348 [ # # ]: 0 : if (unlikely(i == bd_count))
349 : : i = 0;
350 : :
351 : : /* Extension BD occupies the next ring slot. */
352 : 0 : tx_ring->q_swbd[i].buffer_addr = NULL;
353 : 0 : txbd_ext = (struct enetc_tx_bd_ext *)
354 [ # # ]: 0 : ENETC_TXBD(*tx_ring, i);
355 : : memset(txbd_ext, 0, sizeof(*txbd_ext));
356 : 0 : txbd_ext->lso = rte_cpu_to_le_32(ENETC4_TXBD_EXT_LSO_SEG(seg->tso_segsz) |
357 : : ENETC4_TXBD_EXT_FRM_LEN_EXT(data_unit >> 16));
358 : :
359 : 0 : i++;
360 : 0 : bds_to_use--;
361 [ # # ]: 0 : if (unlikely(i == bd_count))
362 : : i = 0;
363 : :
364 : : /*
365 : : * Payload BDs. The first segment's payload starts after
366 : : * the header template; remaining segments are pure
367 : : * payload.
368 : : */
369 : 0 : first_payload = (uint16_t)(seg_len - hdr_len);
370 : : dseg = seg;
371 : : is_first_seg = true;
372 [ # # ]: 0 : while (dseg) {
373 : : uint16_t dlen;
374 : : uint64_t daddr;
375 : :
376 [ # # ]: 0 : if (is_first_seg) {
377 : : dlen = first_payload;
378 : 0 : daddr = rte_mbuf_data_iova(dseg) +
379 : : hdr_len;
380 : : is_first_seg = false;
381 : : } else {
382 : 0 : dlen = rte_pktmbuf_data_len(dseg);
383 : 0 : data = rte_pktmbuf_mtod(dseg, void *);
384 [ # # ]: 0 : for (j = 0; j < dlen;
385 : 0 : j += RTE_CACHE_LINE_SIZE)
386 : : dcbf(data + j);
387 : : dcbf(data + (dlen - 1));
388 : : daddr = rte_mbuf_data_iova(dseg);
389 : : }
390 : :
391 [ # # ]: 0 : if (dlen == 0) {
392 : 0 : dseg = dseg->next;
393 : 0 : continue;
394 : : }
395 : :
396 : 0 : tx_ring->q_swbd[i].buffer_addr = NULL;
397 [ # # ]: 0 : txbd = ENETC_TXBD(*tx_ring, i);
398 : : memset(txbd, 0, sizeof(*txbd));
399 : 0 : txbd->buf_len = rte_cpu_to_le_16(dlen);
400 : 0 : txbd->addr = rte_cpu_to_le_64(daddr);
401 : 0 : i++;
402 : 0 : bds_to_use--;
403 [ # # ]: 0 : if (unlikely(i == bd_count))
404 : : i = 0;
405 : 0 : dseg = dseg->next;
406 : : }
407 : :
408 : : /* Mark the last written BD as frame-last. */
409 : 0 : txbd->flags |= ENETC4_TXBD_FLAGS_F;
410 : 0 : start++;
411 : 0 : continue;
412 : : }
413 : :
414 : : /* Non-TSO packet: standard single/multi-seg encoding. */
415 [ # # ]: 0 : if (bds_to_use < segs_per_pkt)
416 : : break;
417 : :
418 : : is_first_seg = true;
419 [ # # ]: 0 : while (seg) {
420 : 0 : tx_ring->q_swbd[i].buffer_addr = NULL;
421 : 0 : seg_len = rte_pktmbuf_data_len(seg);
422 : 0 : data = rte_pktmbuf_mtod(seg, void *);
423 : :
424 [ # # ]: 0 : for (j = 0; j < seg_len; j += RTE_CACHE_LINE_SIZE)
425 : : dcbf(data + j);
426 : : dcbf(data + (seg_len - 1));
427 : :
428 : 0 : txbd = ENETC_TXBD(*tx_ring, i);
429 : 0 : txbd->flags = 0;
430 [ # # ]: 0 : if (is_first_seg) {
431 : 0 : tx_ring->q_swbd[i].buffer_addr = seg;
432 : 0 : txbd->frm_len = rte_pktmbuf_pkt_len(seg);
433 [ # # ]: 0 : if (seg->ol_flags & ENETC4_TX_CKSUM_OFFLOAD_MASK)
434 : 0 : enetc4_tx_offload_checksum(seg, txbd);
435 : : is_first_seg = false;
436 : : }
437 : :
438 [ # # ]: 0 : txbd->buf_len = rte_cpu_to_le_16(seg_len);
439 : 0 : txbd->addr = rte_cpu_to_le_64(rte_mbuf_data_iova(seg));
440 : 0 : seg = seg->next;
441 : 0 : i++;
442 : 0 : bds_to_use--;
443 : :
444 [ # # ]: 0 : if (unlikely(i == bd_count))
445 : : i = 0;
446 : : }
447 : :
448 : 0 : txbd->flags |= ENETC4_TXBD_FLAGS_F;
449 : 0 : start++;
450 : : }
451 : :
452 : : /*
453 : : * Flush TX BDs to PoC so HW (non-cache-coherent i.MX95) can read the
454 : : * descriptors from memory. Same discipline as enetc_xmit_pkts_cacheable:
455 : : * walk from the cache-line-aligned start of first_bd_idx to just past
456 : : * the last written BD, one dcbf per 64-byte (4-BD) line.
457 : : */
458 [ # # ]: 0 : if (likely(start > 0)) {
459 : 0 : int n = first_bd_idx & ~ENETC_BD_PER_CL_MASK;
460 : 0 : int written = (i - n + bd_count) % bd_count;
461 : :
462 [ # # ]: 0 : if (written == 0)
463 : : written = bd_count;
464 : 0 : written = (written + ENETC_BD_PER_CL_MASK) &
465 : : ~ENETC_BD_PER_CL_MASK;
466 : :
467 [ # # ]: 0 : while (written > 0) {
468 : : dcbf((void *)ENETC_TXBD(*tx_ring, n));
469 : : n = (n + ENETC_BD_PER_CL) % bd_count;
470 : 0 : written -= ENETC_BD_PER_CL;
471 : : }
472 : : }
473 : :
474 : 0 : enetc_clean_tx_ring(tx_ring);
475 : 0 : tx_ring->next_to_use = i;
476 : 0 : enetc_wr_reg(tx_ring->tcir, i);
477 : :
478 : 0 : return start;
479 : : }
480 : :
481 : : int
482 : 0 : enetc_refill_rx_ring(struct enetc_bdr *rx_ring, const int buff_cnt)
483 : : {
484 : : struct enetc_swbd *rx_swbd;
485 : : union enetc_rx_bd *rxbd;
486 : : union enetc_rx_bd *grp_start_rxbd;
487 : : int i, j, k = ENETC_RXBD_BUNDLE;
488 : : struct rte_mbuf *m[ENETC_RXBD_BUNDLE];
489 : : struct rte_mempool *mb_pool;
490 : :
491 : 0 : i = rx_ring->next_to_use;
492 : 0 : mb_pool = rx_ring->mb_pool;
493 : 0 : rx_swbd = &rx_ring->q_swbd[i];
494 : 0 : rxbd = ENETC_RXBD(*rx_ring, i);
495 : : grp_start_rxbd = rxbd;
496 [ # # ]: 0 : for (j = 0; j < buff_cnt; j++) {
497 : : /* bulk alloc for the next up to 8 BDs */
498 [ # # ]: 0 : if (k == ENETC_RXBD_BUNDLE) {
499 : : k = 0;
500 : 0 : int m_cnt = RTE_MIN(buff_cnt - j, ENETC_RXBD_BUNDLE);
501 : :
502 [ # # ]: 0 : if (rte_pktmbuf_alloc_bulk(mb_pool, m, m_cnt))
503 : : return -1;
504 : : }
505 : :
506 : 0 : rx_swbd->buffer_addr = m[k];
507 : 0 : rxbd->w.addr = (uint64_t)(uintptr_t)
508 : 0 : rx_swbd->buffer_addr->buf_iova +
509 : 0 : rx_swbd->buffer_addr->data_off;
510 : : /* clear 'R" as well */
511 : 0 : rxbd->r.lstatus = 0;
512 : 0 : rx_swbd++;
513 : 0 : rxbd++;
514 : 0 : i++;
515 : 0 : k++;
516 [ # # ]: 0 : if (unlikely(i == rx_ring->bd_count)) {
517 : : /*
518 : : * Ring wrap: flush the current partial or full group
519 : : * before resetting the pointer to index 0.
520 : : */
521 : : dcbf((void *)grp_start_rxbd);
522 : : i = 0;
523 : 0 : rxbd = ENETC_RXBD(*rx_ring, i);
524 : 0 : rx_swbd = &rx_ring->q_swbd[i];
525 : : grp_start_rxbd = rxbd;
526 : : } else if ((i & ENETC_BD_PER_CL_MASK) == 0) {
527 : : /*
528 : : * Completed a full 4-BD group (one cache line).
529 : : * Flush it to PoC so HW sees the updated descriptors.
530 : : */
531 : : dcbf((void *)grp_start_rxbd);
532 : : grp_start_rxbd = rxbd;
533 : : }
534 : : }
535 : :
536 : : /* Flush any remaining partial group at the end of the fill. */
537 : : if (j && (i & ENETC_BD_PER_CL_MASK) != 0)
538 : : dcbf((void *)grp_start_rxbd);
539 : :
540 [ # # ]: 0 : if (likely(j)) {
541 : 0 : rx_ring->next_to_alloc = i;
542 : 0 : rx_ring->next_to_use = i;
543 : 0 : enetc_wr_reg(rx_ring->rcir, i);
544 : : }
545 : :
546 : : return j;
547 : : }
548 : :
549 : 0 : static inline void enetc_slow_parsing(struct rte_mbuf *m,
550 : : uint64_t parse_results)
551 : : {
552 : 0 : m->ol_flags &= ~(RTE_MBUF_F_RX_IP_CKSUM_GOOD | RTE_MBUF_F_RX_L4_CKSUM_GOOD);
553 : :
554 [ # # # # : 0 : switch (parse_results) {
# # # # #
# # ]
555 : 0 : case ENETC_PARSE_ERROR | ENETC_PKT_TYPE_IPV4:
556 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
557 : : RTE_PTYPE_L3_IPV4;
558 : 0 : m->ol_flags |= RTE_MBUF_F_RX_IP_CKSUM_BAD;
559 : 0 : return;
560 : 0 : case ENETC_PARSE_ERROR | ENETC_PKT_TYPE_IPV6:
561 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
562 : : RTE_PTYPE_L3_IPV6;
563 : 0 : m->ol_flags |= RTE_MBUF_F_RX_IP_CKSUM_BAD;
564 : 0 : return;
565 : 0 : case ENETC_PARSE_ERROR | ENETC_PKT_TYPE_IPV4_TCP:
566 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
567 : : RTE_PTYPE_L3_IPV4 |
568 : : RTE_PTYPE_L4_TCP;
569 : 0 : m->ol_flags |= RTE_MBUF_F_RX_IP_CKSUM_GOOD |
570 : : RTE_MBUF_F_RX_L4_CKSUM_BAD;
571 : 0 : return;
572 : 0 : case ENETC_PARSE_ERROR | ENETC_PKT_TYPE_IPV6_TCP:
573 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
574 : : RTE_PTYPE_L3_IPV6 |
575 : : RTE_PTYPE_L4_TCP;
576 : 0 : m->ol_flags |= RTE_MBUF_F_RX_IP_CKSUM_GOOD |
577 : : RTE_MBUF_F_RX_L4_CKSUM_BAD;
578 : 0 : return;
579 : 0 : case ENETC_PARSE_ERROR | ENETC_PKT_TYPE_IPV4_UDP:
580 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
581 : : RTE_PTYPE_L3_IPV4 |
582 : : RTE_PTYPE_L4_UDP;
583 : 0 : m->ol_flags |= RTE_MBUF_F_RX_IP_CKSUM_GOOD |
584 : : RTE_MBUF_F_RX_L4_CKSUM_BAD;
585 : 0 : return;
586 : 0 : case ENETC_PARSE_ERROR | ENETC_PKT_TYPE_IPV6_UDP:
587 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
588 : : RTE_PTYPE_L3_IPV6 |
589 : : RTE_PTYPE_L4_UDP;
590 : 0 : m->ol_flags |= RTE_MBUF_F_RX_IP_CKSUM_GOOD |
591 : : RTE_MBUF_F_RX_L4_CKSUM_BAD;
592 : 0 : return;
593 : 0 : case ENETC_PARSE_ERROR | ENETC_PKT_TYPE_IPV4_SCTP:
594 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
595 : : RTE_PTYPE_L3_IPV4 |
596 : : RTE_PTYPE_L4_SCTP;
597 : 0 : m->ol_flags |= RTE_MBUF_F_RX_IP_CKSUM_GOOD |
598 : : RTE_MBUF_F_RX_L4_CKSUM_BAD;
599 : 0 : return;
600 : 0 : case ENETC_PARSE_ERROR | ENETC_PKT_TYPE_IPV6_SCTP:
601 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
602 : : RTE_PTYPE_L3_IPV6 |
603 : : RTE_PTYPE_L4_SCTP;
604 : 0 : m->ol_flags |= RTE_MBUF_F_RX_IP_CKSUM_GOOD |
605 : : RTE_MBUF_F_RX_L4_CKSUM_BAD;
606 : 0 : return;
607 : 0 : case ENETC_PARSE_ERROR | ENETC_PKT_TYPE_IPV4_ICMP:
608 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
609 : : RTE_PTYPE_L3_IPV4 |
610 : : RTE_PTYPE_L4_ICMP;
611 : 0 : m->ol_flags |= RTE_MBUF_F_RX_IP_CKSUM_GOOD |
612 : : RTE_MBUF_F_RX_L4_CKSUM_BAD;
613 : 0 : return;
614 : 0 : case ENETC_PARSE_ERROR | ENETC_PKT_TYPE_IPV6_ICMP:
615 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
616 : : RTE_PTYPE_L3_IPV6 |
617 : : RTE_PTYPE_L4_ICMP;
618 : 0 : m->ol_flags |= RTE_MBUF_F_RX_IP_CKSUM_GOOD |
619 : : RTE_MBUF_F_RX_L4_CKSUM_BAD;
620 : 0 : return;
621 : : /* More switch cases can be added */
622 : 0 : default:
623 : 0 : m->packet_type = RTE_PTYPE_UNKNOWN;
624 : : m->ol_flags |= RTE_MBUF_F_RX_IP_CKSUM_UNKNOWN |
625 : : RTE_MBUF_F_RX_L4_CKSUM_UNKNOWN;
626 : : }
627 : : }
628 : :
629 : :
630 : : static inline void __rte_hot
631 : 0 : enetc_dev_rx_parse(struct rte_mbuf *m, uint16_t parse_results)
632 : : {
633 : : ENETC_PMD_DP_DEBUG("parse summary = 0x%x ", parse_results);
634 : 0 : m->ol_flags |= RTE_MBUF_F_RX_IP_CKSUM_GOOD | RTE_MBUF_F_RX_L4_CKSUM_GOOD;
635 : :
636 [ # # # # : 0 : switch (parse_results) {
# # # # #
# # # # #
# # ]
637 : 0 : case ENETC_PKT_TYPE_ETHER:
638 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER;
639 : 0 : return;
640 : 0 : case ENETC_PKT_TYPE_IPV4:
641 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
642 : : RTE_PTYPE_L3_IPV4;
643 : 0 : return;
644 : 0 : case ENETC_PKT_TYPE_IPV6:
645 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
646 : : RTE_PTYPE_L3_IPV6;
647 : 0 : return;
648 : 0 : case ENETC_PKT_TYPE_IPV4_TCP:
649 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
650 : : RTE_PTYPE_L3_IPV4 |
651 : : RTE_PTYPE_L4_TCP;
652 : 0 : return;
653 : 0 : case ENETC_PKT_TYPE_IPV6_TCP:
654 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
655 : : RTE_PTYPE_L3_IPV6 |
656 : : RTE_PTYPE_L4_TCP;
657 : 0 : return;
658 : 0 : case ENETC_PKT_TYPE_IPV4_UDP:
659 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
660 : : RTE_PTYPE_L3_IPV4 |
661 : : RTE_PTYPE_L4_UDP;
662 : 0 : return;
663 : 0 : case ENETC_PKT_TYPE_IPV6_UDP:
664 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
665 : : RTE_PTYPE_L3_IPV6 |
666 : : RTE_PTYPE_L4_UDP;
667 : 0 : return;
668 : 0 : case ENETC_PKT_TYPE_IPV4_ESP:
669 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
670 : : RTE_PTYPE_L3_IPV4 |
671 : : RTE_PTYPE_TUNNEL_ESP;
672 : 0 : return;
673 : 0 : case ENETC_PKT_TYPE_IPV6_ESP:
674 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
675 : : RTE_PTYPE_L3_IPV6 |
676 : : RTE_PTYPE_TUNNEL_ESP;
677 : 0 : return;
678 : 0 : case ENETC_PKT_TYPE_IPV4_SCTP:
679 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
680 : : RTE_PTYPE_L3_IPV4 |
681 : : RTE_PTYPE_L4_SCTP;
682 : 0 : return;
683 : 0 : case ENETC_PKT_TYPE_IPV6_SCTP:
684 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
685 : : RTE_PTYPE_L3_IPV6 |
686 : : RTE_PTYPE_L4_SCTP;
687 : 0 : return;
688 : 0 : case ENETC_PKT_TYPE_IPV4_ICMP:
689 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
690 : : RTE_PTYPE_L3_IPV4 |
691 : : RTE_PTYPE_L4_ICMP;
692 : 0 : return;
693 : 0 : case ENETC_PKT_TYPE_IPV6_ICMP:
694 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
695 : : RTE_PTYPE_L3_IPV6 |
696 : : RTE_PTYPE_L4_ICMP;
697 : 0 : return;
698 : 0 : case ENETC_PKT_TYPE_IPV4_FRAG:
699 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
700 : : RTE_PTYPE_L3_IPV4 |
701 : : RTE_PTYPE_L4_FRAG;
702 : 0 : return;
703 : 0 : case ENETC_PKT_TYPE_IPV6_FRAG:
704 : 0 : m->packet_type = RTE_PTYPE_L2_ETHER |
705 : : RTE_PTYPE_L3_IPV6 |
706 : : RTE_PTYPE_L4_FRAG;
707 : 0 : return;
708 : : /* More switch cases can be added */
709 : 0 : default:
710 : 0 : enetc_slow_parsing(m, parse_results);
711 : : }
712 : :
713 : : }
714 : :
715 : : static int
716 : 0 : enetc_clean_rx_ring(struct enetc_bdr *rx_ring,
717 : : struct rte_mbuf **rx_pkts,
718 : : int work_limit)
719 : : {
720 : : int rx_frm_cnt = 0;
721 : : int cleaned_cnt, i, bd_count;
722 : : struct enetc_swbd *rx_swbd;
723 : : union enetc_rx_bd *rxbd;
724 : : uint32_t bd_status;
725 : :
726 : : /* next descriptor to process */
727 : 0 : i = rx_ring->next_to_clean;
728 : : /* next descriptor to process */
729 : 0 : rxbd = ENETC_RXBD(*rx_ring, i);
730 : : rte_prefetch0(rxbd);
731 : 0 : bd_count = rx_ring->bd_count;
732 : : /* LS1028A does not have platform cache so any software access following
733 : : * a hardware write will go directly to DDR. Latency of such a read is
734 : : * in excess of 100 core cycles, so try to prefetch more in advance to
735 : : * mitigate this.
736 : : * How much is worth prefetching really depends on traffic conditions.
737 : : * With congested Rx this could go up to 4 cache lines or so. But if
738 : : * software keeps up with hardware and follows behind Rx PI by a cache
739 : : * line or less then it's harmful in terms of performance to cache more.
740 : : * We would only prefetch BDs that have yet to be written by ENETC,
741 : : * which will have to be evicted again anyway.
742 : : */
743 : 0 : rte_prefetch0(ENETC_RXBD(*rx_ring,
744 : : (i + ENETC_CACHE_LINE_RXBDS) % bd_count));
745 : 0 : rte_prefetch0(ENETC_RXBD(*rx_ring,
746 : : (i + ENETC_CACHE_LINE_RXBDS * 2) % bd_count));
747 : :
748 : : cleaned_cnt = enetc_bd_unused(rx_ring);
749 : 0 : rx_swbd = &rx_ring->q_swbd[i];
750 : :
751 [ # # ]: 0 : while (likely(rx_frm_cnt < work_limit)) {
752 : 0 : bd_status = rte_le_to_cpu_32(rxbd->r.lstatus);
753 [ # # ]: 0 : if (!bd_status)
754 : : break;
755 : :
756 : 0 : rx_swbd->buffer_addr->pkt_len = rxbd->r.buf_len -
757 : 0 : rx_ring->crc_len;
758 : 0 : rx_swbd->buffer_addr->data_len = rxbd->r.buf_len -
759 : 0 : rx_ring->crc_len;
760 : 0 : rx_swbd->buffer_addr->hash.rss = rxbd->r.rss_hash;
761 : 0 : rx_swbd->buffer_addr->ol_flags = 0;
762 : 0 : enetc_dev_rx_parse(rx_swbd->buffer_addr,
763 : 0 : rxbd->r.parse_summary);
764 : :
765 : 0 : rx_pkts[rx_frm_cnt] = rx_swbd->buffer_addr;
766 : 0 : cleaned_cnt++;
767 : 0 : rx_swbd++;
768 : 0 : i++;
769 [ # # ]: 0 : if (unlikely(i == rx_ring->bd_count)) {
770 : : i = 0;
771 : : rx_swbd = &rx_ring->q_swbd[i];
772 : : }
773 : 0 : rxbd = ENETC_RXBD(*rx_ring, i);
774 : 0 : rte_prefetch0(ENETC_RXBD(*rx_ring,
775 : : (i + ENETC_CACHE_LINE_RXBDS) %
776 : : bd_count));
777 : 0 : rte_prefetch0(ENETC_RXBD(*rx_ring,
778 : : (i + ENETC_CACHE_LINE_RXBDS * 2) %
779 : : bd_count));
780 : :
781 : 0 : rx_frm_cnt++;
782 : : }
783 : :
784 : 0 : rx_ring->next_to_clean = i;
785 : 0 : enetc_refill_rx_ring(rx_ring, cleaned_cnt);
786 : :
787 : 0 : return rx_frm_cnt;
788 : : }
789 : :
790 : : /*
791 : : * Trim the Ethernet FCS from a received cluster when HW CRC strip is
792 : : * disabled. pkt_len is reduced by crc_len. If the FCS straddles the last
793 : : * two segments (last seg holds fewer bytes than crc_len), drop the trailing
794 : : * segment and trim the carry-over from its predecessor. prev_seg is the
795 : : * segment preceding last_seg in the chain (the caller already tracks it).
796 : : */
797 : : static inline void
798 : 0 : enetc_rx_crc_trim(struct rte_mbuf *first_seg, struct rte_mbuf *prev_seg,
799 : : struct rte_mbuf *last_seg, uint16_t crc_len)
800 : : {
801 : 0 : first_seg->pkt_len -= crc_len;
802 [ # # ]: 0 : if (likely(last_seg->data_len > crc_len)) {
803 : 0 : last_seg->data_len -= crc_len;
804 [ # # ]: 0 : } else if (prev_seg != NULL) {
805 : 0 : first_seg->nb_segs--;
806 : 0 : prev_seg->data_len -= crc_len - last_seg->data_len;
807 : 0 : prev_seg->next = NULL;
808 : : rte_pktmbuf_free_seg(last_seg);
809 : : }
810 : 0 : }
811 : :
812 : : static int
813 : 0 : enetc_clean_rx_ring_nc(struct enetc_bdr *rx_ring,
814 : : struct rte_mbuf **rx_pkts,
815 : : int work_limit)
816 : : {
817 : : int rx_frm_cnt = 0;
818 : : int cleaned_cnt, i;
819 : : struct enetc_swbd *rx_swbd;
820 : : union enetc_rx_bd *rxbd, rxbd_temp;
821 : : struct rte_mbuf *first_seg = NULL, *cur_seg = NULL, *prev_seg = NULL;
822 : : uint32_t bd_status;
823 : : uint8_t *data;
824 : : uint32_t j;
825 : : struct rte_mbuf *seg;
826 : : uint16_t data_len;
827 : :
828 : : /* next descriptor to process */
829 : 0 : i = rx_ring->next_to_clean;
830 [ # # ]: 0 : rxbd = ENETC_RXBD(*rx_ring, i);
831 : : cleaned_cnt = enetc_bd_unused(rx_ring);
832 : 0 : rx_swbd = &rx_ring->q_swbd[i];
833 : :
834 : : /* Restore partial multi-segment chain from a previous burst. */
835 : 0 : first_seg = rx_ring->pkt_first_seg;
836 : 0 : cur_seg = rx_ring->pkt_last_seg;
837 : :
838 [ # # ]: 0 : while (likely(rx_frm_cnt < work_limit)) {
839 : 0 : rxbd_temp = *rxbd;
840 : 0 : bd_status = rte_le_to_cpu_32(rxbd_temp.r.lstatus);
841 : : /* LSTATUS_R indicates this BD has been written by HW */
842 [ # # ]: 0 : if (!(bd_status & ENETC_RXBD_LSTATUS_R))
843 : : break;
844 [ # # ]: 0 : if (rxbd_temp.r.error)
845 : 0 : rx_ring->ierrors++;
846 : :
847 : 0 : seg = rx_swbd->buffer_addr;
848 : : data_len = rte_le_to_cpu_16(rxbd_temp.r.buf_len);
849 : 0 : seg->data_len = data_len;
850 : :
851 [ # # ]: 0 : if (!first_seg) {
852 : : first_seg = seg;
853 : : cur_seg = seg;
854 : : prev_seg = NULL;
855 : 0 : first_seg->pkt_len = data_len;
856 : 0 : enetc_dev_rx_parse(first_seg, rxbd_temp.r.parse_summary);
857 : 0 : first_seg->hash.rss = rxbd_temp.r.rss_hash;
858 : : } else {
859 : 0 : first_seg->pkt_len += data_len;
860 : 0 : first_seg->nb_segs++;
861 : 0 : cur_seg->next = seg;
862 : : prev_seg = cur_seg;
863 : : cur_seg = seg;
864 : : }
865 : :
866 : : /* Invalidate packet data cache lines so CPU reads HW-written data. */
867 : : data = rte_pktmbuf_mtod(seg, void *);
868 [ # # ]: 0 : for (j = 0; j < data_len; j += RTE_CACHE_LINE_SIZE)
869 : : dccivac(data + j);
870 : : dccivac(data + (data_len - 1));
871 : :
872 [ # # ]: 0 : if (bd_status & ENETC_RXBD_LSTATUS_F) {
873 : 0 : seg->next = NULL;
874 [ # # ]: 0 : if (rx_ring->crc_len)
875 : 0 : enetc_rx_crc_trim(first_seg, prev_seg, seg,
876 : : rx_ring->crc_len);
877 : 0 : rx_pkts[rx_frm_cnt] = first_seg;
878 : 0 : rx_frm_cnt++;
879 : : first_seg = NULL;
880 : : }
881 : :
882 : 0 : cleaned_cnt++;
883 : 0 : rx_swbd++;
884 : 0 : i++;
885 [ # # ]: 0 : if (unlikely(i == rx_ring->bd_count)) {
886 : : i = 0;
887 : 0 : rx_swbd = &rx_ring->q_swbd[i];
888 : : }
889 : 0 : rxbd = ENETC_RXBD(*rx_ring, i);
890 : : }
891 : :
892 : : /* Save partial chain for the next burst if frame is incomplete. */
893 : 0 : rx_ring->pkt_first_seg = first_seg;
894 : 0 : rx_ring->pkt_last_seg = cur_seg;
895 : 0 : rx_ring->next_to_clean = i;
896 : 0 : enetc_refill_rx_ring(rx_ring, cleaned_cnt);
897 : :
898 : 0 : return rx_frm_cnt;
899 : : }
900 : :
901 : : uint16_t
902 : 0 : enetc_recv_pkts_nc(void *rxq, struct rte_mbuf **rx_pkts,
903 : : uint16_t nb_pkts)
904 : : {
905 : : struct enetc_bdr *rx_ring = (struct enetc_bdr *)rxq;
906 : :
907 : 0 : return enetc_clean_rx_ring_nc(rx_ring, rx_pkts, nb_pkts);
908 : : }
909 : :
910 : : /*
911 : : * RSC (LRO) refill. buff_cnt is in 16B slots; each 32B descriptor is two
912 : : * slots. The even slot takes a fresh mbuf, the odd extension slot never owns
913 : : * a buffer, so walk the ring two slots at a time. Cache maintenance mirrors
914 : : * enetc_refill_rx_ring (dcbf per 64B line, i.e. two descriptors).
915 : : */
916 : : int
917 : 0 : enetc_refill_rx_ring_rsc(struct enetc_bdr *rx_ring, const int buff_cnt)
918 : : {
919 : : struct enetc_swbd *rx_swbd;
920 : : union enetc_rx_bd *rxbd, *grp_start_rxbd;
921 : : int i, j, k = ENETC_RXBD_BUNDLE;
922 : : struct rte_mbuf *m[ENETC_RXBD_BUNDLE];
923 : : struct rte_mempool *mb_pool;
924 : :
925 : 0 : i = rx_ring->next_to_use;
926 : 0 : mb_pool = rx_ring->mb_pool;
927 : 0 : rx_swbd = &rx_ring->q_swbd[i];
928 : 0 : rxbd = ENETC_RXBD(*rx_ring, i);
929 : : grp_start_rxbd = rxbd;
930 : :
931 [ # # ]: 0 : for (j = 0; j < buff_cnt; j += 2) {
932 : : /* Bulk alloc one mbuf per descriptor (every two slots). */
933 [ # # ]: 0 : if (k == ENETC_RXBD_BUNDLE) {
934 : 0 : int want = (buff_cnt - j) / 2;
935 : 0 : int m_cnt = RTE_MIN(want, ENETC_RXBD_BUNDLE);
936 : :
937 : : k = 0;
938 [ # # ]: 0 : if (rte_pktmbuf_alloc_bulk(mb_pool, m, m_cnt))
939 : : return -1;
940 : : }
941 : :
942 : 0 : rx_swbd->buffer_addr = m[k];
943 : 0 : rxbd->w.addr = (uint64_t)(uintptr_t)
944 : 0 : rx_swbd->buffer_addr->buf_iova +
945 : 0 : rx_swbd->buffer_addr->data_off;
946 : : /* clear 'R' as well */
947 : 0 : rxbd->r.lstatus = 0;
948 : 0 : k++;
949 : :
950 : : /* Extension (odd) slot never owns a buffer. */
951 : 0 : rx_swbd[1].buffer_addr = NULL;
952 : :
953 : 0 : rx_swbd += 2;
954 : 0 : rxbd += 2;
955 : 0 : i += 2;
956 : :
957 [ # # ]: 0 : if (unlikely(i == rx_ring->bd_count)) {
958 : : /* Ring wrap: flush partial/full group before reset. */
959 : : dcbf((void *)grp_start_rxbd);
960 : : i = 0;
961 : 0 : rxbd = ENETC_RXBD(*rx_ring, i);
962 : 0 : rx_swbd = &rx_ring->q_swbd[i];
963 : : grp_start_rxbd = rxbd;
964 : : } else if ((i & ENETC_BD_PER_CL_MASK) == 0) {
965 : : /* Completed a full 4-slot (2-descriptor) cache line. */
966 : : dcbf((void *)grp_start_rxbd);
967 : : grp_start_rxbd = rxbd;
968 : : }
969 : : }
970 : :
971 : : /* Flush any remaining partial group at the end of the fill. */
972 : : if (j && (i & ENETC_BD_PER_CL_MASK) != 0)
973 : : dcbf((void *)grp_start_rxbd);
974 : :
975 [ # # ]: 0 : if (likely(j)) {
976 : 0 : rx_ring->next_to_alloc = i;
977 : 0 : rx_ring->next_to_use = i;
978 : : /*
979 : : * Internal indices track 16B slots, but the HW consumer-index
980 : : * register counts 32B descriptors, so program i / 2.
981 : : */
982 : 0 : enetc_wr_reg(rx_ring->rcir, i / 2);
983 : : }
984 : :
985 : : return j;
986 : : }
987 : :
988 : : /*
989 : : * RSC (LRO) clean. Same non-cache-coherent discipline as
990 : : * enetc_clean_rx_ring_cacheable but walks 32B descriptors (two 16B slots
991 : : * each); the extension half at slot i+1 carries the RSC coalesce count.
992 : : * RSC_FRAMES > 1 flags the cluster RTE_MBUF_F_RX_LRO. RSC always strips the
993 : : * FCS (RBaMR[CRC] = 0), so no CRC trim is needed.
994 : : */
995 : : static int
996 : 0 : enetc_clean_rx_ring_rsc(struct enetc_bdr *rx_ring,
997 : : struct rte_mbuf **rx_pkts,
998 : : int work_limit)
999 : : {
1000 : : int rx_frm_cnt = 0;
1001 : : int cleaned_cnt, i;
1002 : : struct enetc_swbd *rx_swbd;
1003 : : union enetc_rx_bd *rxbd, rxbd_temp;
1004 : : struct enetc_rx_bd_ext *rxbd_ext;
1005 : : struct rte_mbuf *first_seg = NULL, *cur_seg = NULL;
1006 : : uint32_t bd_status;
1007 : : uint8_t *data;
1008 : : uint32_t j;
1009 : : struct rte_mbuf *seg;
1010 : : uint16_t data_len;
1011 : : uint8_t rsc_frames;
1012 : :
1013 : : /* next descriptor to process */
1014 : 0 : i = rx_ring->next_to_clean;
1015 [ # # ]: 0 : rxbd = ENETC_RXBD(*rx_ring, i);
1016 : : cleaned_cnt = enetc_bd_unused(rx_ring);
1017 : 0 : rx_swbd = &rx_ring->q_swbd[i];
1018 : :
1019 : : /*
1020 : : * Always invalidate the cache line that contains next_to_clean before
1021 : : * the first status read. On RSC rings a 64B line holds two 32B
1022 : : * descriptors. See enetc_clean_rx_ring_cacheable for the full rationale
1023 : : * on why dccivac (clean+invalidate) is used from EL0 instead of DC IVAC.
1024 : : */
1025 : : dccivac((void *)ENETC_RXBD(*rx_ring,
1026 : : (i & ~(int)ENETC_BD_PER_CL_MASK)));
1027 : :
1028 [ # # ]: 0 : while (likely(rx_frm_cnt < work_limit)) {
1029 : : /* Atomically snapshot the 16B writeback half of the BD. */
1030 : : #ifdef RTE_ARCH_32
1031 : : rte_memcpy(&rxbd_temp, rxbd, 16);
1032 : : #else
1033 : : __uint128_t *dst128 = (__uint128_t *)&rxbd_temp;
1034 : : const __uint128_t *src128 = (const __uint128_t *)rxbd;
1035 : 0 : *dst128 = *src128;
1036 : : #endif
1037 : : bd_status = rte_le_to_cpu_32(rxbd_temp.r.lstatus);
1038 : :
1039 [ # # ]: 0 : if (!(bd_status & ENETC_RXBD_LSTATUS_R))
1040 : : break;
1041 [ # # ]: 0 : if (rxbd_temp.r.error)
1042 : 0 : rx_ring->ierrors++;
1043 : :
1044 : : /*
1045 : : * Extension half (odd slot) carries the RSC coalesce count.
1046 : : * Invariant: an RSC ring is allocated with bd_count = nb_desc * 2
1047 : : * (always even), next_to_clean starts at 0, and i only ever
1048 : : * advances by 2 with a wrap at i == bd_count. So i is always the
1049 : : * even (writeback) slot of a 32B descriptor and never exceeds
1050 : : * bd_count - 2; i + 1 is therefore always the matching odd
1051 : : * extension slot and stays in bounds (never wraps past the ring).
1052 : : */
1053 : 0 : rxbd_ext = (struct enetc_rx_bd_ext *)
1054 : 0 : ENETC_RXBD(*rx_ring, i + 1);
1055 : 0 : rsc_frames = ENETC4_RXBD_EXT_RSC_FRAMES(rte_le_to_cpu_32(rxbd_ext->rsc_frames));
1056 : :
1057 : 0 : seg = rx_swbd->buffer_addr;
1058 : : data_len = rte_le_to_cpu_16(rxbd_temp.r.buf_len);
1059 : 0 : seg->data_len = data_len;
1060 [ # # ]: 0 : if (!first_seg) {
1061 : : first_seg = seg;
1062 : : cur_seg = seg;
1063 : 0 : first_seg->pkt_len = data_len;
1064 : 0 : first_seg->ol_flags = 0;
1065 : 0 : enetc_dev_rx_parse(first_seg,
1066 : : rxbd_temp.r.parse_summary);
1067 : 0 : first_seg->hash.rss = rxbd_temp.r.rss_hash;
1068 : : /* RSC_FRAMES > 1 means HW coalesced segments (LRO). */
1069 [ # # ]: 0 : if (rsc_frames > 1)
1070 : 0 : first_seg->ol_flags |= RTE_MBUF_F_RX_LRO;
1071 : : } else {
1072 : 0 : first_seg->pkt_len += data_len;
1073 : 0 : first_seg->nb_segs++;
1074 : 0 : cur_seg->next = seg;
1075 : : cur_seg = seg;
1076 : : }
1077 : :
1078 : : /* Invalidate packet data so the CPU reads HW-DMA'd payload. */
1079 : : data = rte_pktmbuf_mtod(seg, void *);
1080 [ # # ]: 0 : for (j = 0; j < data_len; j += RTE_CACHE_LINE_SIZE)
1081 : : dccivac(data + j);
1082 : : /*
1083 : : * Cover the last byte of an unaligned buffer. Guard against a
1084 : : * zero-length BD so data_len - 1 does not step before the
1085 : : * payload start.
1086 : : */
1087 : : if (likely(data_len))
1088 : : dccivac(data + (data_len - 1));
1089 : :
1090 [ # # ]: 0 : if (bd_status & ENETC_RXBD_LSTATUS_F) {
1091 : 0 : seg->next = NULL;
1092 : : ENETC_PMD_DP_DEBUG("RSC_FRAMES=%u pkt_len=%u nb_segs=%u",
1093 : : rsc_frames, first_seg->pkt_len,
1094 : : first_seg->nb_segs);
1095 : 0 : rx_pkts[rx_frm_cnt] = first_seg;
1096 : 0 : rx_frm_cnt++;
1097 : : first_seg = NULL;
1098 : : }
1099 : :
1100 : : /* Two 16B slots per 32B RSC descriptor. */
1101 : 0 : cleaned_cnt += 2;
1102 : 0 : rx_swbd += 2;
1103 : 0 : i += 2;
1104 [ # # ]: 0 : if (unlikely(i == rx_ring->bd_count)) {
1105 : : i = 0;
1106 : : rx_swbd = &rx_ring->q_swbd[i];
1107 : : }
1108 : 0 : rxbd = ENETC_RXBD(*rx_ring, i);
1109 : :
1110 : : /*
1111 : : * Crossed a 4-slot (cache-line, two-descriptor) boundary:
1112 : : * invalidate the new group so subsequent status reads fetch
1113 : : * fresh DDR data written by HW.
1114 : : */
1115 : : if ((i & ENETC_BD_PER_CL_MASK) == 0 &&
1116 : : likely(rx_frm_cnt < work_limit))
1117 : : dccivac((void *)rxbd);
1118 : : }
1119 : :
1120 : 0 : rx_ring->next_to_clean = i;
1121 : 0 : enetc_refill_rx_ring_rsc(rx_ring, ENETC_BD_ALIGN_DOWN(cleaned_cnt));
1122 : :
1123 : : /*
1124 : : * Write-1-to-clear this ring's event in SIRXIDR. Without this the
1125 : : * interrupt-coalescing timer never re-arms and HW stops coalescing
1126 : : * after the first RSC frame (this PMD is poll-mode and never services
1127 : : * the interrupt). SIRXIDR is W1C, one bit per ring; RBaIDR is
1128 : : * read-only. Matches the Linux driver's enetc_wr_reg_hot(idr, BIT()).
1129 : : */
1130 : 0 : enetc4_wr_reg(rx_ring->rbidr, BIT(rx_ring->index));
1131 : :
1132 : 0 : return rx_frm_cnt;
1133 : : }
1134 : :
1135 : : uint16_t
1136 : 0 : enetc_recv_pkts_rsc(void *rxq, struct rte_mbuf **rx_pkts,
1137 : : uint16_t nb_pkts)
1138 : : {
1139 : : struct enetc_bdr *rx_ring = (struct enetc_bdr *)rxq;
1140 : :
1141 : 0 : return enetc_clean_rx_ring_rsc(rx_ring, rx_pkts, nb_pkts);
1142 : : }
1143 : :
1144 : : uint16_t
1145 : 0 : enetc_recv_pkts(void *rxq, struct rte_mbuf **rx_pkts,
1146 : : uint16_t nb_pkts)
1147 : : {
1148 : : struct enetc_bdr *rx_ring = (struct enetc_bdr *)rxq;
1149 : :
1150 : 0 : return enetc_clean_rx_ring(rx_ring, rx_pkts, nb_pkts);
1151 : : }
1152 : :
1153 : : /* --- Cacheable BD ring TX path with SW cache maintenance (dcbf) --- */
1154 : :
1155 : : uint16_t
1156 : 0 : enetc_xmit_pkts_cacheable(void *tx_queue,
1157 : : struct rte_mbuf **tx_pkts,
1158 : : uint16_t nb_pkts)
1159 : : {
1160 : : int i, start, bds_to_use;
1161 : : struct enetc_tx_bd *txbd = NULL;
1162 : : struct enetc_bdr *tx_ring = (struct enetc_bdr *)tx_queue;
1163 : : unsigned int j;
1164 : : uint8_t *data;
1165 : : struct rte_mbuf *seg;
1166 : : uint16_t seg_len, segs_per_pkt;
1167 : : bool is_first_seg;
1168 : : int first_bd_idx, bd_count;
1169 : :
1170 [ # # ]: 0 : i = tx_ring->next_to_use;
1171 : : bds_to_use = enetc_bd_unused(tx_ring);
1172 : 0 : bd_count = tx_ring->bd_count;
1173 : : start = 0;
1174 : :
1175 : : /*
1176 : : * Remember the first BD index of this batch so we can flush the
1177 : : * BD cache lines to PoC after all descriptors are written.
1178 : : */
1179 : : first_bd_idx = i;
1180 : :
1181 [ # # ]: 0 : while (start < nb_pkts) {
1182 : 0 : seg = tx_pkts[start];
1183 : 0 : segs_per_pkt = seg->nb_segs;
1184 : :
1185 [ # # ]: 0 : if (bds_to_use < segs_per_pkt)
1186 : : break;
1187 : :
1188 : : is_first_seg = true;
1189 [ # # ]: 0 : while (seg) {
1190 : 0 : tx_ring->q_swbd[i].buffer_addr = NULL;
1191 : 0 : seg_len = rte_pktmbuf_data_len(seg);
1192 : 0 : data = rte_pktmbuf_mtod(seg, void *);
1193 : :
1194 : : /*
1195 : : * Flush packet data cache lines to PoC so HW DMA
1196 : : * reads the correct payload from memory.
1197 : : */
1198 [ # # ]: 0 : for (j = 0; j < seg_len; j += RTE_CACHE_LINE_SIZE)
1199 : : dcbf(data + j);
1200 : :
1201 : : /*
1202 : : * Cover the last byte of an unaligned buffer to
1203 : : * ensure the full payload is clean to the Point of
1204 : : * Coherency.
1205 : : */
1206 : : dcbf(data + (seg_len - 1));
1207 : 0 : txbd = ENETC_TXBD(*tx_ring, i);
1208 : 0 : txbd->flags = 0;
1209 [ # # ]: 0 : if (is_first_seg) {
1210 : 0 : tx_ring->q_swbd[i].buffer_addr = seg;
1211 : 0 : txbd->frm_len = rte_pktmbuf_pkt_len(seg);
1212 [ # # ]: 0 : if (seg->ol_flags & ENETC4_TX_CKSUM_OFFLOAD_MASK)
1213 : 0 : enetc4_tx_offload_checksum(seg, txbd);
1214 : : is_first_seg = false;
1215 : : }
1216 : :
1217 [ # # ]: 0 : txbd->buf_len = rte_cpu_to_le_16(seg_len);
1218 : 0 : txbd->addr = rte_cpu_to_le_64(rte_mbuf_data_iova(seg));
1219 : 0 : seg = seg->next;
1220 : 0 : i++;
1221 : 0 : bds_to_use--;
1222 : :
1223 [ # # ]: 0 : if (unlikely(i == bd_count))
1224 : : i = 0;
1225 : : }
1226 : :
1227 : : /*
1228 : : * Set the frame-last flag on the final BD of this packet.
1229 : : * This is the last write to the BD group; the cache flush
1230 : : * below will push all BDs to memory afterwards.
1231 : : */
1232 [ # # ]: 0 : if (likely(txbd))
1233 : 0 : txbd->flags |= ENETC4_TXBD_FLAGS_F;
1234 : 0 : start++;
1235 : : }
1236 : :
1237 : : /*
1238 : : * Flush TX BDs to PoC so HW (non-cache-coherent i.MX95) can read
1239 : : * the descriptors from memory. TX BDs are 16 B each; 4 BDs share
1240 : : * one 64-byte cache line. Walk from the cache-line-aligned start
1241 : : * of first_bd_idx to just past the last written BD, one dcbf per
1242 : : * cache line.
1243 : : *
1244 : : * The flush must happen AFTER all BD fields (including flags_F) are
1245 : : * written, so HW never sees a partial descriptor.
1246 : : */
1247 [ # # ]: 0 : if (likely(start > 0)) {
1248 : 0 : int n = first_bd_idx & ~ENETC_BD_PER_CL_MASK;
1249 : 0 : int written = (i - n + bd_count) % bd_count;
1250 : :
1251 [ # # ]: 0 : if (written == 0)
1252 : : written = bd_count;
1253 : 0 : written = (written + ENETC_BD_PER_CL_MASK) & ~ENETC_BD_PER_CL_MASK;
1254 : :
1255 [ # # ]: 0 : while (written > 0) {
1256 : : dcbf((void *)ENETC_TXBD(*tx_ring, n));
1257 : : n = (n + ENETC_BD_PER_CL) % bd_count;
1258 : 0 : written -= ENETC_BD_PER_CL;
1259 : : }
1260 : : }
1261 : :
1262 : 0 : enetc_clean_tx_ring(tx_ring);
1263 : 0 : tx_ring->next_to_use = i;
1264 : 0 : enetc_wr_reg(tx_ring->tcir, i);
1265 : :
1266 : 0 : return start;
1267 : : }
1268 : :
1269 : : /* --- Cacheable BD ring RX path with SW cache maintenance (dccivac) --- */
1270 : :
1271 : : static int
1272 : 0 : enetc_clean_rx_ring_cacheable(struct enetc_bdr *rx_ring,
1273 : : struct rte_mbuf **rx_pkts,
1274 : : int work_limit)
1275 : : {
1276 : : int rx_frm_cnt = 0;
1277 : : int cleaned_cnt, i;
1278 : : struct enetc_swbd *rx_swbd;
1279 : : union enetc_rx_bd *rxbd;
1280 : : struct rte_mbuf *first_seg = NULL, *cur_seg = NULL, *prev_seg = NULL;
1281 : : uint32_t bd_status;
1282 : : uint8_t *data;
1283 : : uint32_t j;
1284 : : struct rte_mbuf *seg;
1285 : : uint16_t data_len;
1286 : :
1287 : 0 : i = rx_ring->next_to_clean;
1288 [ # # ]: 0 : rxbd = ENETC_RXBD(*rx_ring, i);
1289 : : cleaned_cnt = enetc_bd_unused(rx_ring);
1290 : 0 : rx_swbd = &rx_ring->q_swbd[i];
1291 : :
1292 : : /* Restore partial multi-segment chain from a previous burst. */
1293 : 0 : first_seg = rx_ring->pkt_first_seg;
1294 : 0 : cur_seg = rx_ring->pkt_last_seg;
1295 : :
1296 : : /*
1297 : : * On i.MX95 the BD ring is in cacheable hugepage memory but the
1298 : : * platform is non-cache-coherent. HW writes RX BDs to DDR
1299 : : * without snooping the CPU cache, so stale cached copies of BD
1300 : : * status fields must be discarded before the CPU reads them.
1301 : : *
1302 : : * Ideal instruction: DC IVAC (invalidate only, no writeback).
1303 : : * ARM64 constraint: DC IVAC requires EL1 privilege; executing it
1304 : : * from EL0 (DPDK userspace) raises a fault. The only EL0-safe
1305 : : * cache maintenance instruction that invalidates is DC CIVAC
1306 : : * (clean + invalidate, dccivac).
1307 : : *
1308 : : * Safety of using dccivac here:
1309 : : * enetc_refill_rx_ring() issues dcbf() on every BD group before
1310 : : * returning ownership to HW. After dcbf the CPU cache lines are
1311 : : * marked clean (no dirty data). When dccivac runs, the "clean"
1312 : : * phase finds nothing dirty to write back, so it behaves as a
1313 : : * pure invalidate - exactly what we need.
1314 : : *
1315 : : * Granularity: BD = 16 B, cache line = 64 B, so one dccivac
1316 : : * covers exactly 4 BDs. Invalidate at each 4-BD boundary.
1317 : : */
1318 : : dccivac((void *)ENETC_RXBD(*rx_ring,
1319 : : (i & ~(int)ENETC_BD_PER_CL_MASK)));
1320 : :
1321 [ # # ]: 0 : while (likely(rx_frm_cnt < work_limit)) {
1322 : 0 : bd_status = rte_le_to_cpu_32(rxbd->r.lstatus);
1323 : :
1324 [ # # ]: 0 : if (!(bd_status & ENETC_RXBD_LSTATUS_R))
1325 : : break;
1326 [ # # ]: 0 : if (rxbd->r.error)
1327 : 0 : rx_ring->ierrors++;
1328 : :
1329 : 0 : seg = rx_swbd->buffer_addr;
1330 : 0 : data_len = rte_le_to_cpu_16(rxbd->r.buf_len);
1331 : 0 : seg->data_len = data_len;
1332 [ # # ]: 0 : if (!first_seg) {
1333 : : first_seg = seg;
1334 : : cur_seg = seg;
1335 : : prev_seg = NULL;
1336 : 0 : first_seg->pkt_len = data_len;
1337 : 0 : enetc_dev_rx_parse(first_seg,
1338 : 0 : rxbd->r.parse_summary);
1339 : 0 : first_seg->hash.rss = rxbd->r.rss_hash;
1340 : : } else {
1341 : 0 : first_seg->pkt_len += data_len;
1342 : 0 : first_seg->nb_segs++;
1343 : 0 : cur_seg->next = seg;
1344 : : prev_seg = cur_seg;
1345 : : cur_seg = seg;
1346 : : }
1347 : :
1348 : : /*
1349 : : * Invalidate packet data cache lines so the CPU reads the
1350 : : * payload that HW DMA'd into memory, not stale cached bytes.
1351 : : */
1352 : : data = rte_pktmbuf_mtod(seg, void *);
1353 [ # # ]: 0 : for (j = 0; j < data_len; j += RTE_CACHE_LINE_SIZE)
1354 : : dccivac(data + j);
1355 : : /* Cover the last byte of an unaligned buffer. */
1356 : : dccivac(data + (data_len - 1));
1357 : :
1358 [ # # ]: 0 : if (bd_status & ENETC_RXBD_LSTATUS_F) {
1359 : 0 : seg->next = NULL;
1360 [ # # ]: 0 : if (rx_ring->crc_len)
1361 : 0 : enetc_rx_crc_trim(first_seg, prev_seg, seg,
1362 : : rx_ring->crc_len);
1363 : :
1364 : 0 : rx_pkts[rx_frm_cnt] = first_seg;
1365 : 0 : rx_frm_cnt++;
1366 : : first_seg = NULL;
1367 : : }
1368 : :
1369 : 0 : cleaned_cnt++;
1370 : 0 : rx_swbd++;
1371 : 0 : i++;
1372 [ # # ]: 0 : if (unlikely(i == rx_ring->bd_count)) {
1373 : : i = 0;
1374 : 0 : rx_swbd = &rx_ring->q_swbd[i];
1375 : : }
1376 : 0 : rxbd = ENETC_RXBD(*rx_ring, i);
1377 : :
1378 : : /*
1379 : : * Crossed a 4-BD (cache-line) boundary: invalidate the new
1380 : : * group so the next four status reads fetch fresh DDR data
1381 : : * written by HW.
1382 : : */
1383 : : if ((i & ENETC_BD_PER_CL_MASK) == 0 &&
1384 : : likely(rx_frm_cnt < work_limit))
1385 : : dccivac((void *)rxbd);
1386 : : }
1387 : :
1388 : : /* Save partial chain for the next burst if frame is incomplete. */
1389 : 0 : rx_ring->pkt_first_seg = first_seg;
1390 : 0 : rx_ring->pkt_last_seg = cur_seg;
1391 : 0 : rx_ring->next_to_clean = i;
1392 : 0 : enetc_refill_rx_ring(rx_ring, ENETC_BD_ALIGN_DOWN(cleaned_cnt));
1393 : :
1394 : 0 : return rx_frm_cnt;
1395 : : }
1396 : :
1397 : : uint16_t
1398 : 0 : enetc_recv_pkts_cacheable(void *rxq, struct rte_mbuf **rx_pkts,
1399 : : uint16_t nb_pkts)
1400 : : {
1401 : : struct enetc_bdr *rx_ring = (struct enetc_bdr *)rxq;
1402 : :
1403 : 0 : return enetc_clean_rx_ring_cacheable(rx_ring, rx_pkts, nb_pkts);
1404 : : }
|