/* Use LAN VSI Id if not programmed by user */
vsi_id = fdata->dest_vsi ? : i40e_pf_get_main_vsi(pf)->id;
flex_ptype |= FIELD_PREP(I40E_TXD_FLTR_QW0_DEST_VSI_MASK, vsi_id);
/* find existing FDIR VSI */
vsi = i40e_find_vsi_by_type(pf, I40E_VSI_FDIR); if (!vsi) return -ENOENT;
tx_ring = vsi->tx_rings[0];
dev = tx_ring->dev;
/* we need two descriptors to add/del a filter and we can wait */ for (i = I40E_FD_CLEAN_DELAY; I40E_DESC_UNUSED(tx_ring) < 2; i--) { if (!i) return -EAGAIN;
msleep_interruptible(1);
}
if (fd_data->flex_filter) {
u8 *payload;
__be16 pattern = fd_data->flex_word;
u16 off = fd_data->flex_offset;
payload = packet_addr + payload_offset;
/* If user provided vlan, offset payload by vlan header length */ if (!!fd_data->vlan_tag)
payload += VLAN_HLEN;
*((__force __be16 *)(payload + off)) = pattern;
}
fd_data->pctype = pctype;
ret = i40e_program_fdir_filter(fd_data, packet_addr, pf, add); if (ret) {
dev_info(&pf->pdev->dev, "PCTYPE:%d, Filter command send failed for fd_id:%d (ret = %d)\n",
fd_data->pctype, fd_data->fd_id, ret); /* Free the packet buffer since it wasn't added to the ring */ return -EOPNOTSUPP;
} elseif (I40E_DEBUG_FD & pf->hw.debug_mask) { if (add)
dev_info(&pf->pdev->dev, "Filter OK for PCTYPE %d loc = %d\n",
fd_data->pctype, fd_data->fd_id); else
dev_info(&pf->pdev->dev, "Filter deleted for PCTYPE %d loc = %d\n",
fd_data->pctype, fd_data->fd_id);
}
switch (input->flow_type & ~FLOW_EXT) { case TCP_V4_FLOW:
ret = i40e_add_del_fdir_tcp(vsi, input, add, ipv4); break; case UDP_V4_FLOW:
ret = i40e_add_del_fdir_udp(vsi, input, add, ipv4); break; case SCTP_V4_FLOW:
ret = i40e_add_del_fdir_sctp(vsi, input, add, ipv4); break; case TCP_V6_FLOW:
ret = i40e_add_del_fdir_tcp(vsi, input, add, ipv6); break; case UDP_V6_FLOW:
ret = i40e_add_del_fdir_udp(vsi, input, add, ipv6); break; case SCTP_V6_FLOW:
ret = i40e_add_del_fdir_sctp(vsi, input, add, ipv6); break; case IP_USER_FLOW: switch (input->ipl4_proto) { case IPPROTO_TCP:
ret = i40e_add_del_fdir_tcp(vsi, input, add, ipv4); break; case IPPROTO_UDP:
ret = i40e_add_del_fdir_udp(vsi, input, add, ipv4); break; case IPPROTO_SCTP:
ret = i40e_add_del_fdir_sctp(vsi, input, add, ipv4); break; case IPPROTO_IP:
ret = i40e_add_del_fdir_ip(vsi, input, add, ipv4); break; default: /* We cannot support masking based on protocol */
dev_info(&pf->pdev->dev, "Unsupported IPv4 protocol 0x%02x\n",
input->ipl4_proto); return -EINVAL;
} break; case IPV6_USER_FLOW: switch (input->ipl4_proto) { case IPPROTO_TCP:
ret = i40e_add_del_fdir_tcp(vsi, input, add, ipv6); break; case IPPROTO_UDP:
ret = i40e_add_del_fdir_udp(vsi, input, add, ipv6); break; case IPPROTO_SCTP:
ret = i40e_add_del_fdir_sctp(vsi, input, add, ipv6); break; case IPPROTO_IP:
ret = i40e_add_del_fdir_ip(vsi, input, add, ipv6); break; default: /* We cannot support masking based on protocol */
dev_info(&pf->pdev->dev, "Unsupported IPv6 protocol 0x%02x\n",
input->ipl4_proto); return -EINVAL;
} break; default:
dev_info(&pf->pdev->dev, "Unsupported flow type 0x%02x\n",
input->flow_type); return -EINVAL;
}
/* The buffer allocated here will be normally be freed by *i40e_clean_fdir_tx_irq()asitreclaimsresourcesaftertransmit *completion.IntheeventofanerroraddingthebuffertotheFDIR *ring,itwillimmediatelybefreed.Itmayalsobefreedby *i40e_clean_tx_ring()whenclosingtheVSI.
*/ return ret;
}
if (error == BIT(I40E_RX_PROG_STATUS_DESC_FD_TBL_FULL_SHIFT)) {
pf->fd_inv = le32_to_cpu(qw0->hi_dword.fd_id); if (qw0->hi_dword.fd_id != 0 ||
(I40E_DEBUG_FD & pf->hw.debug_mask))
dev_warn(&pdev->dev, "ntuple filter loc = %d, could not be added\n",
pf->fd_inv);
/* Check if the programming error is for ATR. *Ifso,autodisableATRandsetastatefor *flushinprogress.Nexttimewecomehereifflushisin *progressdonothing,onceflushiscompletethestatewill *becleared.
*/ if (test_bit(__I40E_FD_FLUSH_REQUESTED, pf->state)) return;
pf->fd_add_err++; /* store the current atr filter count */
pf->fd_atr_cnt = i40e_get_current_atr_cnt(pf);
if (qw0->hi_dword.fd_id == 0 &&
test_bit(__I40E_FD_SB_AUTO_DISABLED, pf->state)) { /* These set_bit() calls aren't atomic with the *test_bit()here,butworsecasewepotentially *disableATRandqueueaflushrightafterSB *supportisre-enabled.Thatshouldn'tcausean *issueinpractice
*/
set_bit(__I40E_FD_ATR_AUTO_DISABLED, pf->state);
set_bit(__I40E_FD_FLUSH_REQUESTED, pf->state);
}
/* filter programming failed most likely due to table full */
fcnt_prog = i40e_get_global_fd_count(pf);
fcnt_avail = pf->fdir_pf_filter_count; /* If ATR is running fcnt_prog can quickly change, *ifweareveryclosetofull,itmakessensetodisable *FDATR/SBandthenre-enableitwhenthereisroom.
*/ if (fcnt_prog >= (fcnt_avail - I40E_FDIR_BUFFER_FULL_MARGIN)) { if (test_bit(I40E_FLAG_FD_SB_ENA, pf->flags) &&
!test_and_set_bit(__I40E_FD_SB_AUTO_DISABLED,
pf->state)) if (I40E_DEBUG_FD & pf->hw.debug_mask)
dev_warn(&pdev->dev, "FD filter space full, new ntuple rules will not be added\n");
}
} elseif (error == BIT(I40E_RX_PROG_STATUS_DESC_NO_FD_ENTRY_SHIFT)) { if (I40E_DEBUG_FD & pf->hw.debug_mask)
dev_info(&pdev->dev, "ntuple filter fd_id = %d, could not be removed\n",
qw0->hi_dword.fd_id);
}
}
tx_buffer->next_to_watch = NULL;
tx_buffer->skb = NULL;
dma_unmap_len_set(tx_buffer, len, 0); /* tx_buffer must be completely set up in the transmit path */
}
if (ring_is_xdp(tx_ring) && tx_ring->xsk_pool) {
i40e_xsk_clean_tx_ring(tx_ring);
} else { /* ring already cleared, nothing to do */ if (!tx_ring->tx_bi) return;
/* Free all the Tx ring sk_buffs */ for (i = 0; i < tx_ring->count; i++)
i40e_unmap_and_free_tx_resource(tx_ring,
&tx_ring->tx_bi[i]);
}
if (test_bit(__I40E_VSI_DOWN, vsi->state)) return;
netdev = vsi->netdev; if (!netdev) return;
if (!netif_carrier_ok(netdev)) return;
for (i = 0; i < vsi->num_queue_pairs; i++) {
tx_ring = vsi->tx_rings[i]; if (tx_ring && tx_ring->desc) { /* If packet counter has not changed the queue is *likelystalled,soforceaninterruptforthis *queue. * *prev_pkt_ctrwouldbenegativeiftherewasno *pendingwork.
*/
packets = tx_ring->stats.packets & INT_MAX; if (tx_ring->tx_stats.prev_pkt_ctr == packets) {
i40e_force_wb(vsi, tx_ring->q_vector); continue;
}
/* Memory barrier between read of packet count and call *toi40e_get_tx_pending()
*/
smp_rmb();
tx_ring->tx_stats.prev_pkt_ctr =
i40e_get_tx_pending(tx_ring, true) ? packets : -1;
}
}
}
tx_buf++;
tx_desc++;
i++; if (unlikely(!i)) {
i -= tx_ring->count;
tx_buf = tx_ring->tx_bi;
tx_desc = I40E_TX_DESC(tx_ring, 0);
}
/* unmap any remaining paged data */ if (dma_unmap_len(tx_buf, len)) {
dma_unmap_page(tx_ring->dev,
dma_unmap_addr(tx_buf, dma),
dma_unmap_len(tx_buf, len),
DMA_TO_DEVICE);
dma_unmap_len_set(tx_buf, len, 0);
}
}
/* move us one more past the eop_desc for start of next pkt */
tx_buf++;
tx_desc++;
i++; if (unlikely(!i)) {
i -= tx_ring->count;
tx_buf = tx_ring->tx_bi;
tx_desc = I40E_TX_DESC(tx_ring, 0);
}
prefetch(tx_desc);
/* update budget accounting */
budget--;
} while (likely(budget));
/** *i40e_force_wb-IssueSWInterruptsoHWdoesawb *@vsi:theVSIwecareabout *@q_vector:thevectoronwhichtoforcewriteback *
**/ void i40e_force_wb(struct i40e_vsi *vsi, struct i40e_q_vector *q_vector)
{ if (test_bit(I40E_FLAG_MSIX_ENA, vsi->back->flags)) {
u32 val = I40E_PFINT_DYN_CTLN_INTENA_MASK |
I40E_PFINT_DYN_CTLN_ITR_INDX_MASK | /* set noitr */
I40E_PFINT_DYN_CTLN_SWINT_TRIG_MASK |
I40E_PFINT_DYN_CTLN_SW_ITR_INDX_ENA_MASK; /* allow 00 to be written to the index */
wr32(&vsi->back->hw,
I40E_PFINT_DYN_CTLN(q_vector->reg_idx), val);
} else {
u32 val = I40E_PFINT_DYN_CTL0_INTENA_MASK |
I40E_PFINT_DYN_CTL0_ITR_INDX_MASK | /* set noitr */
I40E_PFINT_DYN_CTL0_SWINT_TRIG_MASK |
I40E_PFINT_DYN_CTL0_SW_ITR_INDX_ENA_MASK; /* allow 00 to be written to the index */
/* If we don't have any rings just leave ourselves set for maximum *possiblelatencysowetakeourselvesoutoftheequation.
*/ if (!rc->ring || !ITR_IS_DYNAMIC(rc->ring->itr_setting)) return;
/* For Rx we want to push the delay up and default to low latency. *forTxwewanttopullthedelaydownanddefaulttohighlatency.
*/
itr = i40e_container_is_rx(q_vector, rc) ?
I40E_ITR_ADAPTIVE_MIN_USECS | I40E_ITR_ADAPTIVE_LATENCY :
I40E_ITR_ADAPTIVE_MAX_USECS | I40E_ITR_ADAPTIVE_LATENCY;
/* If we didn't update within up to 1 - 2 jiffies we can assume *thateitherpacketsarecominginsoslowtherehasn'tbeen *anywork,orthatthereissomuchworkthatNAPIisdealing *withinterruptmoderationandwedon'tneedtodoanything.
*/ if (time_after(next_update, rc->next_update)) goto clear_counts;
/* If itr_countdown is set it means we programmed an ITR within *thelast4interruptcycles.Thishasasideeffectofus *potentiallyfiringanearlyinterrupt.Inordertoworkaround *thisweneedtothrowoutanydatareceivedforafew *interruptsfollowingtheupdate.
*/ if (q_vector->itr_countdown) {
itr = rc->target_itr; goto clear_counts;
}
if (i40e_container_is_rx(q_vector, rc)) { /* If Rx there are 1 to 4 packets and bytes are less than *9000assumeinsufficientdatatousebulkratelimiting *approachunlessTxisalreadyinbulkratelimiting.We *arelikelylatencydriven.
*/ if (packets && packets < 4 && bytes < 9000 &&
(q_vector->tx.target_itr & I40E_ITR_ADAPTIVE_LATENCY)) {
itr = I40E_ITR_ADAPTIVE_LATENCY; goto adjust_by_size;
}
} elseif (packets < 4) { /* If we have Tx and Rx ITR maxed and Tx ITR is running in *bulkmodeandwearereceiving4orfewerpacketsjust *resettheITR_ADAPTIVE_LATENCYbitforlatencymodeso *thattheRxcanrelax.
*/ if (rc->target_itr == I40E_ITR_ADAPTIVE_MAX_USECS &&
(q_vector->rx.target_itr & I40E_ITR_MASK) ==
I40E_ITR_ADAPTIVE_MAX_USECS) goto clear_counts;
} elseif (packets > 32) { /* If we have processed over 32 packets in a single interrupt *forTxassumeweneedtoswitchoverto"bulk"mode.
*/
rc->target_itr &= ~I40E_ITR_ADAPTIVE_LATENCY;
}
/* We have no packets to actually measure against. This means *eitheroneoftheotherqueuesonthisvectorisactiveor *weareaTxqueuedoingTSOwithtoohighofaninterruptrate. * *Between4and56wecanassumethatourcurrentinterruptdelay *isonlyslightlytoolow.Assuchweshouldincreaseitbyasmall *fixedamount.
*/ if (packets < 56) {
itr = rc->target_itr + I40E_ITR_ADAPTIVE_MIN_INC; if ((itr & I40E_ITR_MASK) > I40E_ITR_ADAPTIVE_MAX_USECS) {
itr &= I40E_ITR_ADAPTIVE_LATENCY;
itr += I40E_ITR_ADAPTIVE_MAX_USECS;
} goto clear_counts;
}
/* Between 56 and 112 is our "goldilocks" zone where we are *workingout"justright".Justreportthatourcurrent *ITRisgoodforus.
*/ if (packets <= 112) goto clear_counts;
/* If packet count is 128 or greater we are likely looking *ataslightoverrunofthedelaywewant.Tryhalving *ourdelaytoseeifthatwillcutthenumberofpackets *inhalfperinterrupt.
*/
itr /= 2;
itr &= I40E_ITR_MASK; if (itr < I40E_ITR_ADAPTIVE_MIN_USECS)
itr = I40E_ITR_ADAPTIVE_MIN_USECS;
goto clear_counts;
}
/* The paths below assume we are dealing with a bulk ITR since *numberofpacketsisgreaterthan256.Wearejustgoingtohave *tocomputeavalueandtrytobringthecountundercontrol, *thoughforsmallerpacketsizesthereisn'tmuchwecandoas *NAPIpollingwilllikelybekickinginsoonerratherthanlater.
*/
itr = I40E_ITR_ADAPTIVE_BULK;
adjust_by_size: /* If packet counts are 256 or greater we can assume we have a gross *overestimationofwhattherateshouldbe.Insteadoftryingtofine *tuneitjustusetheformulabelowtotryanddialinanexactvalue *givethecurrentpacketsizeoftheframe.
*/
avg_wire_size = bytes / packets;
/* The following is a crude approximation of: *wmem_default/(size+overhead)=desired_pkts_per_int *rate/bits_per_byte/(size+ethernetoverhead)=pkt_rate *(desired_pkt_rate/pkt_rate)*usecs_per_sec=ITRvalue * *Assumingwmem_defaultis212992andoverheadis640bytesper *packet,(256skb,64headroom,320sharedinfo),wecanreducethe *formuladownto * *(170*(size+24))/(size+640)=ITR * *Wefirstdosomemathonthepacketsizeandthenfinallybitshift *by8afterroundingup.WealsohavetoaccountforPCIelinkspeed *differenceasITRscalesbasedonthis.
*/ if (avg_wire_size <= 60) { /* Start at 250k ints/sec */
avg_wire_size = 4096;
} elseif (avg_wire_size <= 380) { /* 250K ints/sec to 60K ints/sec */
avg_wire_size *= 40;
avg_wire_size += 1696;
} elseif (avg_wire_size <= 1084) { /* 60K ints/sec to 36K ints/sec */
avg_wire_size *= 15;
avg_wire_size += 11452;
} elseif (avg_wire_size <= 1980) { /* 36K ints/sec to 30K ints/sec */
avg_wire_size *= 5;
avg_wire_size += 22420;
} else { /* plateau at a limit of 30K ints/sec */
avg_wire_size = 32256;
}
/* If we are in low latency mode halve our delay which doubles the *ratetosomewherebetween100Kto16Kints/sec
*/ if (itr & I40E_ITR_ADAPTIVE_LATENCY)
avg_wire_size /= 2;
/* Resultant value is 256 times larger than it needs to be. This *givesusroomtoadjustthevalueasneededtoeitherincrease *ordecreasethevaluebasedonlinkspeedsof10G,2.5G,1G,etc. * *Useadditionaswehavealreadyrecordedthenewlatencyflag *fortheITRvalue.
*/
itr += DIV_ROUND_UP(avg_wire_size, i40e_itr_divisor(q_vector)) *
I40E_ITR_ADAPTIVE_MIN_INC;
/* update, and store next to alloc */
nta++;
rx_ring->next_to_alloc = (nta < rx_ring->count) ? nta : 0;
/* transfer page from old buffer to new buffer */
new_buff->dma = old_buff->dma;
new_buff->page = old_buff->page;
new_buff->page_offset = old_buff->page_offset;
new_buff->pagecnt_bias = old_buff->pagecnt_bias;
/* clear contents of buffer_info */
old_buff->page = NULL;
}
id = FIELD_GET(I40E_RX_PROG_STATUS_DESC_QW1_PROGID_MASK, qword1);
if (id == I40E_RX_PROG_STATUS_DESC_FD_FILTER_STATUS)
i40e_fd_handle_status(rx_ring, qword0_raw, qword1, id);
}
/** *i40e_setup_tx_descriptors-AllocatetheTxdescriptors *@tx_ring:thetxringtosetup * *Return0onsuccess,negativeonerror
**/ int i40e_setup_tx_descriptors(struct i40e_ring *tx_ring)
{ struct device *dev = tx_ring->dev; int bi_size;
if (!dev) return -ENOMEM;
/* warn if we are about to overwrite the pointer */
WARN_ON(tx_ring->tx_bi);
bi_size = sizeof(struct i40e_tx_buffer) * tx_ring->count;
tx_ring->tx_bi = kzalloc(bi_size, GFP_KERNEL); if (!tx_ring->tx_bi) goto err;
u64_stats_init(&tx_ring->syncp);
/* round up to nearest 4K */
tx_ring->size = tx_ring->count * sizeof(struct i40e_tx_desc); /* add u32 for head writeback, align after this takes care of *guaranteeingthisisatleastonecachelineinsize
*/
tx_ring->size += sizeof(u32);
tx_ring->size = ALIGN(tx_ring->size, 4096);
tx_ring->desc = dma_alloc_coherent(dev, tx_ring->size,
&tx_ring->dma, GFP_KERNEL); if (!tx_ring->desc) {
dev_info(dev, "Unable to allocate memory for the Tx descriptor ring, size=%d\n",
tx_ring->size); goto err;
}
/* ring already cleared, nothing to do */ if (!rx_ring->rx_bi) return;
if (rx_ring->xsk_pool) {
i40e_xsk_clean_rx_ring(rx_ring); goto skip_free;
}
/* Free all the Rx ring sk_buffs */ for (i = 0; i < rx_ring->count; i++) { struct i40e_rx_buffer *rx_bi = i40e_rx_bi(rx_ring, i);
if (!rx_bi->page) continue;
/* Invalidate cache lines that may have been written to by *devicesothatweavoidcorruptingmemory.
*/
dma_sync_single_range_for_cpu(rx_ring->dev,
rx_bi->dma,
rx_bi->page_offset,
rx_ring->rx_buf_len,
DMA_FROM_DEVICE);
/* since we are recycling buffers we should seldom need to alloc */ if (likely(page)) {
rx_ring->rx_stats.page_reuse_count++; returntrue;
}
/* alloc new page for storage */
page = dev_alloc_pages(i40e_rx_pg_order(rx_ring)); if (unlikely(!page)) {
rx_ring->rx_stats.alloc_page_failed++; returnfalse;
}
rx_ring->rx_stats.page_alloc_count++;
/* map page for use */
dma = dma_map_page_attrs(rx_ring->dev, page, 0,
i40e_rx_pg_size(rx_ring),
DMA_FROM_DEVICE,
I40E_RX_DMA_ATTR);
/* if mapping failed free memory back to system since *thereisn'tmuchpointinholdingmemorywecan'tuse
*/ if (dma_mapping_error(rx_ring->dev, dma)) {
__free_pages(page, i40e_rx_pg_order(rx_ring));
rx_ring->rx_stats.alloc_page_failed++; returnfalse;
}
/* do nothing if no valid netdev defined */ if (!rx_ring->netdev || !cleaned_count) returnfalse;
rx_desc = I40E_RX_DESC(rx_ring, ntu);
bi = i40e_rx_bi(rx_ring, ntu);
do { if (!i40e_alloc_mapped_page(rx_ring, bi)) goto no_buffers;
/* sync the buffer for use by the device */
dma_sync_single_range_for_device(rx_ring->dev, bi->dma,
bi->page_offset,
rx_ring->rx_buf_len,
DMA_FROM_DEVICE);
/* Refresh the desc even if buffer_addrs didn't change *becauseeachwrite-backerasesthisinfo.
*/
rx_desc->read.pkt_addr = cpu_to_le64(bi->dma + bi->page_offset);
rx_desc++;
bi++;
ntu++; if (unlikely(ntu == rx_ring->count)) {
rx_desc = I40E_RX_DESC(rx_ring, 0);
bi = i40e_rx_bi(rx_ring, 0);
ntu = 0;
}
/* clear the status bits for the next_to_use descriptor */
rx_desc->wb.qword1.status_error_len = 0;
cleaned_count--;
} while (cleaned_count);
if (rx_ring->next_to_use != ntu)
i40e_release_rx_desc(rx_ring, ntu);
returnfalse;
no_buffers: if (rx_ring->next_to_use != ntu)
i40e_release_rx_desc(rx_ring, ntu);
/* make sure to come back via polling to try again after *allocationfailure
*/ returntrue;
}
if (ipv4 &&
(rx_error & (BIT(I40E_RX_DESC_ERROR_IPE_SHIFT) |
BIT(I40E_RX_DESC_ERROR_EIPE_SHIFT)))) goto checksum_fail;
/* likely incorrect csum if alternate IP extension headers found */ if (ipv6 &&
rx_status & BIT(I40E_RX_DESC_STATUS_IPV6EXADD_SHIFT)) /* don't increment checksum err here, non-fatal err */ return;
/* there was some L4 error, count error and punt packet to the stack */ if (rx_error & BIT(I40E_RX_DESC_ERROR_L4E_SHIFT)) goto checksum_fail;
/* handle packets that were not able to be checksummed due *toarrivalspeed,inthiscasethestackcancompute *thecsum.
*/ if (rx_error & BIT(I40E_RX_DESC_ERROR_PPRS_SHIFT)) return;
/* If there is an outer header present that might contain a checksum *weneedtobumpthechecksumlevelby1toreflectthefactthat *weareindicatingwevalidatedtheinnerchecksum.
*/ if (decoded.tunnel_type >= LIBETH_RX_PT_TUNNEL_IP_GRENAT)
skb->csum_level = 1;
{ /* ERR_MASK will only have valid bits if EOP set, and *whatwearedoinghereisactuallychecking *I40E_RX_DESC_ERROR_RXE_SHIFT,sinceitisthezerothbitin *theerrorfield
*/ if (unlikely(i40e_test_staterr(rx_desc,
BIT(I40E_RXD_QW1_ERROR_SHIFT)))) {
dev_kfree_skb_any(skb); returntrue;
}
/* if eth_skb_pad returns an error the skb was freed */ if (eth_skb_pad(skb)) returntrue;
/* Is any reuse possible? */ if (!dev_page_is_reusable(page)) {
rx_stats->page_waive_count++; returnfalse;
}
#if (PAGE_SIZE < 8192) /* if we are only owner of page we can reuse it */ if (unlikely((rx_buffer->page_count - pagecnt_bias) > 1)) {
rx_stats->page_busy_count++; returnfalse;
} #else #define I40E_LAST_OFFSET \
(SKB_WITH_OVERHEAD(PAGE_SIZE) - I40E_RXBUFFER_2048) if (rx_buffer->page_offset > I40E_LAST_OFFSET) {
rx_stats->page_busy_count++; returnfalse;
} #endif
/* If we have drained the page fragment pool we need to update *thepagecnt_biasandpagecountsothatwefullyrestockthe *numberofreferencesthedriverholds.
*/ if (unlikely(pagecnt_bias == 1)) {
page_ref_add(page, USHRT_MAX - 1);
rx_buffer->pagecnt_bias = USHRT_MAX;
}
/* we are reusing so sync this buffer for CPU use */
dma_sync_single_range_for_cpu(rx_ring->dev,
rx_buffer->dma,
rx_buffer->page_offset,
size,
DMA_FROM_DEVICE);
/* We have pulled a buffer for use, so decrement pagecnt_bias */
rx_buffer->pagecnt_bias--;
return rx_buffer;
}
/** *i40e_put_rx_buffer-Cleanupusedbufferandeitherrecycleorfree *@rx_ring:rxdescriptorringtotransactpacketson *@rx_buffer:rxbuffertopulldatafrom * *Thisfunctionwillcleanupthecontentsoftherx_buffer.Itwill *eitherrecyclethebufferorunmapitandfreetheassociatedresources.
*/ staticvoid i40e_put_rx_buffer(struct i40e_ring *rx_ring, struct i40e_rx_buffer *rx_buffer)
{ if (i40e_can_reuse_rx_page(rx_buffer, &rx_ring->rx_stats)) { /* hand second half of page back to the ring */
i40e_reuse_rx_page(rx_ring, rx_buffer);
} else { /* we are not reusing the buffer so unmap it */
dma_unmap_page_attrs(rx_ring->dev, rx_buffer->dma,
i40e_rx_pg_size(rx_ring),
DMA_FROM_DEVICE, I40E_RX_DMA_ATTR);
__page_frag_cache_drain(rx_buffer->page,
rx_buffer->pagecnt_bias); /* clear contents of buffer_info */
rx_buffer->page = NULL;
}
}
/** *i40e_process_rx_buffs-ProcessingofbufferspostXDPprogoronerror *@rx_ring:Rxdescriptorringtotransactpacketson *@xdp_res:ResultoftheXDPprogram *@xdp:xdp_buffpointingtothedata
**/ staticvoid i40e_process_rx_buffs(struct i40e_ring *rx_ring, int xdp_res, struct xdp_buff *xdp)
{
u32 nr_frags = xdp_get_shared_info_from_buff(xdp)->nr_frags;
u32 next = rx_ring->next_to_clean, i = 0; struct i40e_rx_buffer *rx_buffer;
xdp->flags = 0;
while (1) {
rx_buffer = i40e_rx_bi(rx_ring, next); if (++next == rx_ring->count)
next = 0;
/* Prefetch first cache line of first page. If xdp->data_meta *isunused,thispointsexactlyasxdp->data,otherwisewe *likelyhaveaconsumeraccessingfirstfewbytesofmeta *data,andthenactualdata.
*/
net_prefetch(xdp->data_meta);
if (unlikely(xdp_buff_has_frags(xdp))) {
sinfo = xdp_get_shared_info_from_buff(xdp);
nr_frags = sinfo->nr_frags;
}
/* build an skb around the page buffer */
skb = napi_build_skb(xdp->data_hard_start, xdp->frame_sz); if (unlikely(!skb)) return NULL;
/* update pointers within the skb to store the data */
skb_reserve(skb, xdp->data - xdp->data_hard_start);
__skb_put(skb, xdp->data_end - xdp->data); if (metasize)
skb_metadata_set(skb, metasize);
if (unlikely(xdp_buff_has_frags(xdp))) {
xdp_update_skb_shared_info(skb, nr_frags,
sinfo->xdp_frags_size,
nr_frags * xdp->frame_sz,
xdp_buff_is_frag_pfmemalloc(xdp));
rx_buffer = i40e_rx_bi(rx_ring, rx_ring->next_to_clean); /* buffer is used by skb, update page_offset */
i40e_rx_buffer_flip(rx_buffer, xdp->frame_sz);
}
return skb;
}
/** *i40e_is_non_eop-processhandlingofnon-EOPbuffers *@rx_ring:Rxringbeingprocessed *@rx_desc:Rxdescriptorforcurrentbuffer * *IfthebufferisanEOPbuffer,thisfunctionexitsreturningfalse, *otherwisereturntrueindicatingthatthisisinfactanon-EOPbuffer.
*/ bool i40e_is_non_eop(struct i40e_ring *rx_ring, union i40e_rx_desc *rx_desc)
{ /* if we are the last buffer then there is nothing else to do */ #define I40E_RXD_EOF BIT(I40E_RX_DESC_STATUS_EOF_SHIFT) if (likely(i40e_test_staterr(rx_desc, I40E_RXD_EOF))) returnfalse;
/* return some buffers to hardware, one at a time is too slow */ if (cleaned_count >= clean_threshold) {
failure = failure ||
i40e_alloc_rx_buffers(rx_ring, cleaned_count);
cleaned_count = 0;
}
rx_desc = I40E_RX_DESC(rx_ring, ntp);
/* status_error_len will always be zero for unused descriptors *becauseit'sclearedincleanup,andoverlapswithhdr_addr *whichisalwayszerobecausepacketsplitisn'tused,ifthe *hardwarewroteDDthenthelengthwillbenon-zero
*/
qword = le64_to_cpu(rx_desc->wb.qword1.status_error_len);
/* This memory barrier is needed to keep us from reading *anyotherfieldsoutoftherx_descuntilwehave *verifiedthedescriptorhasbeenwrittenback.
*/
dma_rmb();
if (i40e_rx_is_programming_status(qword)) {
i40e_clean_programming_status(rx_ring,
rx_desc->raw.qword[0],
qword);
rx_buffer = i40e_rx_bi(rx_ring, ntp);
i40e_inc_ntp(rx_ring);
i40e_reuse_rx_page(rx_ring, rx_buffer); /* Update ntc and bump cleaned count if not in the *middleofmbpacket.
*/ if (rx_ring->next_to_clean == ntp) {
rx_ring->next_to_clean =
rx_ring->next_to_process;
cleaned_count++;
} continue;
}
size = FIELD_GET(I40E_RXD_QW1_LENGTH_PBUF_MASK, qword); if (!size) break;
i40e_trace(clean_rx_irq, rx_ring, rx_desc, xdp); /* retrieve a buffer from the ring */
rx_buffer = i40e_get_rx_buffer(rx_ring, size);
/* drop if we failed to retrieve a buffer */ if (!skb) {
rx_ring->rx_stats.alloc_buff_failed++;
i40e_consume_xdp_buff(rx_ring, xdp, rx_buffer); break;
}
if (i40e_cleanup_headers(rx_ring, skb, rx_desc)) goto process_next;
/* probably a little skewed due to removing CRC */
total_rx_bytes += skb->len;
/* populate checksum, VLAN, and protocol */
i40e_process_skb_fields(rx_ring, rx_desc, skb);
/* We don't bother with setting the CLEARPBA bit as the data sheet *pointsoutdoingsois"meaninglesssinceitwasalready *auto-cleared".Theauto-clearinghappenswhentheinterruptis *asserted. * *Hardwareerrata28foralsoindicatesthatwritingtoa *xxINT_DYN_CTLxCSRwithINTENA_MSK(bit31)setto0willclear *aneventinthePBAanywaysoweneedtorelyontheautomask *toholdpendingeventsforusuntiltheinterruptisre-enabled * *Wehavetoshiftthegivenvalueasitisreportedinmicroseconds *andtheregistervalueisrecordedin2microsecondunits.
*/
interval >>= 1;
/* 3. Enforce software interrupt trigger if requested *(ThesesoftwareinterruptsrateislimitedbyITR2thatis *setto20Kinterruptspersecond)
*/ if (force_swint)
val |= I40E_PFINT_DYN_CTLN_SWINT_TRIG_MASK |
I40E_PFINT_DYN_CTLN_SW_ITR_INDX_ENA_MASK |
FIELD_PREP(I40E_PFINT_DYN_CTLN_SW_ITR_INDX_MASK,
I40E_SW_ITR);
return val;
}
/* The act of updating the ITR will cause it to immediately trigger. In order *topreventthisfromthrowingoffadaptiveupdatestatisticswedeferthe *updatesothatitcanonlyhappensooften.SoaftereitherTxorRxare *updatedwemaketheadaptiveschemewaituntileithertheITRcompletely *expiresviathenext_updateexpirationorwehavebeenthroughatleast *3interrupts.
*/ #define ITR_COUNTDOWN_START 3
/* If we don't have MSIX, then we only need to re-enable icr0 */ if (!test_bit(I40E_FLAG_MSIX_ENA, vsi->back->flags)) {
i40e_irq_dynamic_enable_icr0(vsi->back); return;
}
/* These will do nothing if dynamic updates are not enabled */
i40e_update_itr(q_vector, &q_vector->tx);
i40e_update_itr(q_vector, &q_vector->rx);
/* This block of logic allows us to get away with only updating *oneITRvaluewitheachinterrupt.Theideaistoperforma *pseudo-lazyupdatewiththefollowingcriteria. * *1.RxisgivenhigherprioritythanTxifbothareinsamestate *2.IfwemustreduceanITRthatisgivenhighestpriority. *3.WethengiveprioritytoincreasingITRbasedonamount.
*/ if (q_vector->rx.target_itr < q_vector->rx.current_itr) { /* Rx ITR needs to be reduced, this is highest priority */
itr_idx = I40E_RX_ITR;
interval = q_vector->rx.target_itr;
q_vector->rx.current_itr = q_vector->rx.target_itr;
q_vector->itr_countdown = ITR_COUNTDOWN_START;
} elseif ((q_vector->tx.target_itr < q_vector->tx.current_itr) ||
((q_vector->rx.target_itr - q_vector->rx.current_itr) <
(q_vector->tx.target_itr - q_vector->tx.current_itr))) { /* Tx ITR needs to be reduced, this is second priority *TxITRneedstobeincreasedmorethanRx,fourthpriority
*/
itr_idx = I40E_TX_ITR;
interval = q_vector->tx.target_itr;
q_vector->tx.current_itr = q_vector->tx.target_itr;
q_vector->itr_countdown = ITR_COUNTDOWN_START;
} elseif (q_vector->rx.current_itr != q_vector->rx.target_itr) { /* Rx ITR needs to be increased, third priority */
itr_idx = I40E_RX_ITR;
interval = q_vector->rx.target_itr;
q_vector->rx.current_itr = q_vector->rx.target_itr;
q_vector->itr_countdown = ITR_COUNTDOWN_START;
} else { /* No ITR update, lowest priority */ if (q_vector->itr_countdown)
q_vector->itr_countdown--;
}
/* Do not update interrupt control register if VSI is down */ if (test_bit(__I40E_VSI_DOWN, vsi->state)) return;
if (test_bit(__I40E_VSI_DOWN, vsi->state)) {
napi_complete(napi); return0;
}
/* Since the actual Tx work is minimal, we can give the Tx a larger *budgetandbemoreaggressiveaboutcleaninguptheTxdescriptors.
*/
i40e_for_each_ring(ring, q_vector->tx) { bool wd = ring->xsk_pool ?
i40e_clean_xdp_tx_irq(vsi, ring) :
i40e_clean_tx_irq(vsi, ring, budget, &tx_cleaned);
/* Handle case where we are called by netpoll with a budget of 0 */ if (budget <= 0) goto tx_only;
/* normally we have 1 Rx ring per q_vector */ if (unlikely(q_vector->num_ringpairs > 1)) /* We attempt to distribute budget to each Rx queue fairly, but *don'tallowthebudgettogobelow1becausethatwouldexit *pollingearly.
*/
budget_per_ring = max_t(int, budget / q_vector->num_ringpairs, 1); else /* Max of 1 Rx ring in this q_vector so give it the budget */
budget_per_ring = budget;
work_done += cleaned; /* if we clean as many as budgeted, we must not be done */ if (cleaned >= budget_per_ring)
clean_complete = rx_clean_complete = false;
}
if (!i40e_enabled_xdp_vsi(vsi))
trace_i40e_napi_poll(napi, q_vector, budget, budget_per_ring, rx_cleaned,
tx_cleaned, rx_clean_complete, tx_clean_complete);
/* If work not completed, return budget and polling will return */ if (!clean_complete) { int cpu_id = smp_processor_id();
/* It is possible that the interrupt affinity has changed but, *ifthecpuispeggedat100%,pollingwillneverexitwhile *trafficcontinuesandtheinterruptwillbestuckonthis *cpu.Wechecktomakesureaffinityiscorrectbeforewe *continuetopoll,otherwisewemuststoppollingsothe *interruptcanmovetothecorrectcpu.
*/ if (!cpumask_test_cpu(cpu_id, &q_vector->affinity_mask)) { /* Tell napi that we are done polling */
napi_complete_done(napi, work_done);
/* Force an interrupt */
i40e_force_wb(vsi, q_vector);
/* Return budget-1 so that polling stops */ return budget - 1;
}
tx_only: if (arm_wb) {
q_vector->tx.ring[0].tx_stats.tx_force_wb++;
i40e_enable_wb_on_itr(vsi, q_vector);
} return budget;
}
if (q_vector->tx.ring[0].flags & I40E_TXR_FLAGS_WB_ON_ITR)
q_vector->arm_wb_state = false;
/* Exit the polling mode, but don't re-enable interrupts if stack might *pollusduetobusy-polling
*/ if (likely(napi_complete_done(napi, work_done)))
i40e_update_enable_itr(vsi, q_vector); else
q_vector->in_busy_poll = true;
/* make sure ATR is enabled */ if (!test_bit(I40E_FLAG_FD_ATR_ENA, pf->flags)) return;
if (test_bit(__I40E_FD_ATR_AUTO_DISABLED, pf->state)) return;
/* if sampling is disabled do nothing */ if (!tx_ring->atr_sample_rate) return;
/* Currently only IPv4/IPv6 with TCP is supported */ if (!(tx_flags & (I40E_TX_FLAGS_IPV4 | I40E_TX_FLAGS_IPV6))) return;
/* snag network header to get L4 type and address */
hdr.network = (tx_flags & I40E_TX_FLAGS_UDP_TUNNEL) ?
skb_inner_network_header(skb) : skb_network_header(skb);
/* Note: tx_flags gets modified to reflect inner protocols in *tx_enable_csumfunctionifencapisenabled.
*/ if (tx_flags & I40E_TX_FLAGS_IPV4) { /* access ihl as u8 to avoid unaligned access on ia64 */
hlen = (hdr.network[0] & 0x0F) << 2;
l4_proto = hdr.ipv4->protocol;
} else { /* find the start of the innermost ipv6 header */ unsignedint inner_hlen = hdr.network - skb->data; unsignedint h_offset = inner_hlen;
/* this function updates h_offset to the end of the header */
l4_proto =
ipv6_find_hdr(skb, &h_offset, IPPROTO_TCP, NULL, NULL); /* hlen will contain our best estimate of the tcp header */
hlen = h_offset - inner_hlen;
}
if (l4_proto != IPPROTO_TCP) return;
th = (struct tcphdr *)(hdr.network + hlen);
/* Due to lack of space, no more new filters can be programmed */ if (th->syn && test_bit(__I40E_FD_ATR_AUTO_DISABLED, pf->state)) return; if (test_bit(I40E_FLAG_HW_ATR_EVICT_ENA, pf->flags)) { /* HW ATR eviction will take care of removing filters on FIN *andRSTpackets.
*/ if (th->fin || th->rst) return;
}
tx_ring->atr_count++;
/* sample on all syn/fin/rst packets or once every atr sample rate */ if (!th->fin &&
!th->syn &&
!th->rst &&
(tx_ring->atr_count < tx_ring->atr_sample_rate)) return;
tx_ring->atr_count = 0;
/* grab the next descriptor */
i = tx_ring->next_to_use;
fdir_desc = I40E_TX_FDIRDESC(tx_ring, i);
i++;
tx_ring->next_to_use = (i < tx_ring->count) ? i : 0;
if (protocol == htons(ETH_P_8021Q) &&
!(tx_ring->netdev->features & NETIF_F_HW_VLAN_CTAG_TX)) { /* When HW VLAN acceleration is turned off by the user the *stacksetstheprotocolto8021qsothatthedriver *cantakeanystepsrequiredtosupporttheSWonly *VLANhandling.Inourcasethedriverdoesn'tneed *totakeanyfurtherstepssojustsettheprotocol *totheencapsulatedethertype.
*/
skb->protocol = vlan_get_protocol(skb); goto out;
}
/* if we have a HW VLAN tag being added, default to the HW one */ if (skb_vlan_tag_present(skb)) {
tx_flags |= skb_vlan_tag_get(skb) << I40E_TX_FLAGS_VLAN_SHIFT;
tx_flags |= I40E_TX_FLAGS_HW_VLAN; /* else if it is a SW VLAN, check the next protocol and store the tag */
} elseif (protocol == htons(ETH_P_8021Q)) { struct vlan_hdr *vhdr, _vhdr;
vhdr = skb_header_pointer(skb, ETH_HLEN, sizeof(_vhdr), &_vhdr); if (!vhdr) return -EINVAL;
if (likely(!(skb_shinfo(skb)->tx_flags & SKBTX_HW_TSTAMP))) return0;
/* Tx timestamps cannot be sampled when doing TSO */ if (tx_flags & I40E_TX_FLAGS_TSO) return0;
/* only timestamp the outbound packet if the user has requested it and *wearenotalreadytransmittingapackettobetimestamped
*/
pf = i40e_netdev_to_pf(tx_ring->netdev); if (!test_bit(I40E_FLAG_PTP_ENA, pf->flags)) return0;
/* set the tx_flags to indicate the IP protocol type. this is *requiredsothatchecksumheadercomputationbelowisaccurate.
*/ if (ip.v4->version == 4)
*tx_flags |= I40E_TX_FLAGS_IPV4; else
*tx_flags |= I40E_TX_FLAGS_IPV6;
/* indicate if we need to offload outer UDP header */ if ((*tx_flags & I40E_TX_FLAGS_TSO) &&
!(skb_shinfo(skb)->gso_type & SKB_GSO_PARTIAL) &&
(skb_shinfo(skb)->gso_type & SKB_GSO_UDP_TUNNEL_CSUM))
tunnel |= I40E_TXD_CTX_QW0_L4T_CS_MASK;
/* record tunnel offload values */
*cd_tunneling |= tunnel;
/* switch L4 header pointer from outer to inner */
l4.hdr = skb_inner_transport_header(skb);
l4_proto = 0;
/* reset type as we transition from outer to inner headers */
*tx_flags &= ~(I40E_TX_FLAGS_IPV4 | I40E_TX_FLAGS_IPV6); if (ip.v4->version == 4)
*tx_flags |= I40E_TX_FLAGS_IPV4; if (ip.v6->version == 6)
*tx_flags |= I40E_TX_FLAGS_IPV6;
}
/* Enable IP checksum offloads */ if (*tx_flags & I40E_TX_FLAGS_IPV4) {
l4_proto = ip.v4->protocol; /* the stack computes the IP header already, the only time we *needthehardwaretorecomputeitisinthecaseofTSO.
*/
cmd |= (*tx_flags & I40E_TX_FLAGS_TSO) ?
I40E_TX_DESC_CMD_IIPT_IPV4_CSUM :
I40E_TX_DESC_CMD_IIPT_IPV4;
} elseif (*tx_flags & I40E_TX_FLAGS_IPV6) {
cmd |= I40E_TX_DESC_CMD_IIPT_IPV6;
/** *__i40e_maybe_stop_tx-2ndlevelcheckfortxstopconditions *@tx_ring:theringtobechecked *@size:thesizebufferwewanttoassureisavailable * *Returns-EBUSYifastopisneeded,else0
**/ int __i40e_maybe_stop_tx(struct i40e_ring *tx_ring, int size)
{
netif_stop_subqueue(tx_ring->netdev, tx_ring->queue_index); /* Memory barrier before checking head and tail */
smp_mb();
++tx_ring->tx_stats.tx_stopped;
/* Check again in a case another CPU has just made room available. */ if (likely(I40E_DESC_UNUSED(tx_ring) < size)) return -EBUSY;
/* A reprieve! - use start_queue because it doesn't call schedule */
netif_start_subqueue(tx_ring->netdev, tx_ring->queue_index);
++tx_ring->tx_stats.restart_queue; return0;
}
/* no need to check if number of frags is less than 7 */
nr_frags = skb_shinfo(skb)->nr_frags; if (nr_frags < (I40E_MAX_BUFFER_TXD - 1)) returnfalse;
/* We need to walk through the list and validate that each group *of6fragmentstotalsatleastgso_size.
*/
nr_frags -= I40E_MAX_BUFFER_TXD - 2;
frag = &skb_shinfo(skb)->frags[0];
/* Initialize size to the negative value of gso_size minus 1. We *usethisastheworstcasescenerioinwhichthefragahead *ofusonlyprovidesonebytewhichiswhywearelimitedto6 *descriptorsforasingletransmitastheheaderandprevious *fragmentarealreadyconsuming2descriptors.
*/
sum = 1 - skb_shinfo(skb)->gso_size;
/* Add size of frags 0 through 4 to create our initial sum */
sum += skb_frag_size(frag++);
sum += skb_frag_size(frag++);
sum += skb_frag_size(frag++);
sum += skb_frag_size(frag++);
sum += skb_frag_size(frag++);
/* Walk through fragments adding latest fragment, testing it, and *thenremovingstalefragmentsfromthesum.
*/ for (stale = &skb_shinfo(skb)->frags[0];; stale++) { int stale_size = skb_frag_size(stale);
sum += skb_frag_size(frag++);
/* The stale fragment may present us with a smaller *descriptorthantheactualfragmentsize.Toaccount *forthatweneedtoremoveallthedataonthefrontand *figureoutwhattheremainderwouldbeinthelast *descriptorassociatedwiththefragment.
*/ if (stale_size > I40E_MAX_DATA_PER_TXD) { int align_pad = -(skb_frag_off(stale)) &
(I40E_MAX_READ_REQ_SIZE - 1);
sum -= align_pad;
stale_size -= align_pad;
do {
sum -= I40E_MAX_DATA_PER_TXD_ALIGNED;
stale_size -= I40E_MAX_DATA_PER_TXD_ALIGNED;
} while (stale_size > I40E_MAX_DATA_PER_TXD);
}
/* if sum is negative we failed to make sufficient progress */ if (sum < 0) returntrue;
/* write last descriptor with EOP bit */
td_cmd |= I40E_TX_DESC_CMD_EOP;
/* We OR these values together to check both against 4 (WB_STRIDE) *below.Thisissafesincewedon'tre-usedesc_countafterwards.
*/
desc_count |= ++tx_ring->packet_stride;
if (desc_count >= WB_STRIDE) { /* write last descriptor with RS bit set */
td_cmd |= I40E_TX_DESC_CMD_RS;
tx_ring->packet_stride = 0;
}
/* Force memory writes to complete before letting h/w know there *arenewdescriptorstofetch. * *Wealsousethismemorybarriertomakecertainallofthe *statusbitshavebeenupdatedbeforenext_to_watchiswritten.
*/
wmb();
/* set next_to_watch value indicating a packet is present */
first->next_to_watch = tx_desc;
/* notify HW of packet */ if (netif_xmit_stopped(txring_txq(tx_ring)) || !netdev_xmit_more()) {
writel(i, tx_ring->tail);
}
/* record the location of the first descriptor for this packet */
first = &tx_ring->tx_bi[tx_ring->next_to_use];
first->skb = skb;
first->bytecount = skb->len;
first->gso_segs = 1;
/* prepare the xmit flags */ if (i40e_tx_prepare_vlan_flags(skb, tx_ring, &tx_flags)) goto out_drop;
/* Always offload the checksum, since it's in the data descriptor */
tso = i40e_tx_enable_csum(skb, &tx_flags, &td_cmd, &td_offset,
tx_ring, &cd_tunneling); if (tso < 0) goto out_drop;
if (unlikely(flags & XDP_XMIT_FLUSH))
i40e_xdp_ring_update_tail(xdp_ring);
return nxmit;
}
Messung V0.5 in Prozent
¤ Diese beiden folgenden Angebotsgruppen bietet das Unternehmen0.118Angebot
(Wie Sie bei der Firma Beratungs- und Dienstleistungen beauftragen können 2026-10-04)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.