/* alloc_stat_inc is intended to be used in softirq context */ #define alloc_stat_inc(pool, __stat) (pool->alloc_stats.__stat++) /* recycle_stat_inc is safe to use when preemption is possible. */ #define recycle_stat_inc(pool, __stat) \ do { \ struct page_pool_recycle_stats __percpu *s = pool->recycle_stats; \
this_cpu_inc(s->__stat); \
} while (0)
#else #define alloc_stat_inc(...) do { } while (0) #define recycle_stat_inc(...) do { } while (0) #define recycle_stat_add(...) do { } while (0) #endif
/* Validate only known flags were used */ if (pool->slow.flags & ~PP_FLAG_ALL) return -EINVAL;
if (pool->p.pool_size)
ring_qsize = min(pool->p.pool_size, 16384);
/* DMA direction is either DMA_FROM_DEVICE or DMA_BIDIRECTIONAL. *DMA_BIDIRECTIONALisforallowingpageusedforDMAsending, *whichistheXDP_TXuse-case.
*/ if (pool->slow.flags & PP_FLAG_DMA_MAP) { if ((pool->p.dma_dir != DMA_FROM_DEVICE) &&
(pool->p.dma_dir != DMA_BIDIRECTIONAL)) return -EINVAL;
pool->dma_map = true;
}
if (pool->slow.flags & PP_FLAG_DMA_SYNC_DEV) { /* In order to request DMA-sync-for-device the page *needstobemapped
*/ if (!(pool->slow.flags & PP_FLAG_DMA_MAP)) return -EINVAL;
if (!pool->p.max_len) return -EINVAL;
pool->dma_sync = true;
/* pool->p.offset has to be set according to the address *offsetusedbytheDMAenginetostartcopyingrxdata
*/
}
/* Quicker fallback, avoid locks when ring is empty */ if (__ptr_ring_empty(r)) {
alloc_stat_inc(pool, empty); return0;
}
/* Softirq guarantee CPU and thus NUMA node is stable. This, *assumesCPUrefillingdriverRX-ringwillalsorunRX-NAPI.
*/ #ifdef CONFIG_NUMA
pref_nid = (pool->p.nid == NUMA_NO_NODE) ? numa_mem_id() : pool->p.nid; #else /* Ignore pool->p.nid setting if !CONFIG_NUMA, helps compiler */
pref_nid = numa_mem_id(); /* will be zero like page_to_nid() */ #endif
/* Refill alloc array, but only if NUMA match */ do {
netmem = (__force netmem_ref)__ptr_ring_consume(r); if (unlikely(!netmem)) break;
/* Track how many pages are held 'in-flight' */
pool->pages_state_hold_cnt++;
trace_page_pool_state_hold(pool, page_to_netmem(page),
pool->pages_state_hold_cnt); return page;
}
/* Unconditionally set NOWARN if allocating from NAPI. *Driversforgettosetit,andOOMreportsonpacketRxareuseless.
*/ if ((gfp & GFP_ATOMIC) == GFP_ATOMIC)
gfp |= __GFP_NOWARN;
/* Don't support bulk alloc for high-order pages */ if (unlikely(pp_order)) return page_to_netmem(__page_pool_alloc_page_order(pool, gfp));
/* Unnecessary as alloc cache is empty, but guarantees zero count */ if (unlikely(pool->alloc.count > 0)) return pool->alloc.cache[--pool->alloc.count];
/* Mark empty alloc.cache slots "empty" for alloc_pages_bulk */
memset(&pool->alloc.cache, 0, sizeof(void *) * bulk);
/* Pages have been filled into alloc.cache array, but count is zero and *pageelementhavenotbeen(possibly)DMAmapped.
*/ for (i = 0; i < nr_pages; i++) {
netmem = pool->alloc.cache[i]; if (dma_map && unlikely(!page_pool_dma_map(pool, netmem, gfp))) {
put_page(netmem_to_page(netmem)); continue;
}
page_pool_set_pp_info(pool, netmem);
pool->alloc.cache[pool->alloc.count++] = netmem; /* Track how many pages are held 'in-flight' */
pool->pages_state_hold_cnt++;
trace_page_pool_state_hold(pool, netmem,
pool->pages_state_hold_cnt);
}
/* When page just alloc'ed is should/must have refcnt 1. */ return netmem;
}
/* For using page_pool replace: alloc_pages() API calls, but provide *synchronizationguaranteeforallocationside.
*/
netmem_ref page_pool_alloc_netmems(struct page_pool *pool, gfp_t gfp)
{
netmem_ref netmem;
/* Fast-path: Get a page from cache */
netmem = __page_pool_get_cached(pool); if (netmem) return netmem;
/* Slow-path: cache empty, do real allocation */ if (static_branch_unlikely(&page_pool_mem_providers) && pool->mp_ops)
netmem = pool->mp_ops->alloc_netmems(pool, gfp); else
netmem = __page_pool_alloc_netmems_slow(pool, gfp); return netmem;
}
EXPORT_SYMBOL(page_pool_alloc_netmems);
ALLOW_ERROR_INJECTION(page_pool_alloc_netmems, NULL);
/* Calculate distance between two u32 values, valid if distance is below 2^(31) *https://en.wikipedia.org/wiki/Serial_number_arithmetic#General_Solution
*/ #define _distance(a, b) (s32)((a) - (b))
/* Ensuring all pages have been split into one fragment initially: *page_pool_set_pp_info()isonlycalledonceforeverypagewhenit *isallocatedfromthepageallocatorandpage_pool_fragment_page() *isdirtyingthesamecachelineasthepage->pp_magicabove,so *theoverheadisnegligible.
*/
page_pool_fragment_netmem(netmem, 1); if (pool->has_init_callback)
pool->slow.init_callback(netmem, pool->slow.init_arg);
}
if (!pool->dma_map) /* Always account for inflight pages, even if we didn't *mapthem
*/ return;
if (page_pool_release_dma_index(pool, netmem)) return;
dma = page_pool_get_dma_addr_netmem(netmem);
/* When page is unmapped, it cannot be returned to our pool */
dma_unmap_page_attrs(pool->p.dev, dma,
PAGE_SIZE << pool->p.order, pool->p.dma_dir,
DMA_ATTR_SKIP_CPU_SYNC | DMA_ATTR_WEAK_ORDERING);
page_pool_set_dma_addr_netmem(netmem, 0);
}
/* Disconnects a page (from a page_pool). API users can have a need *todisconnectapage(fromapage_pool),toallowittobeusedas *aregularpage(thatwilleventuallybereturnedtothenormal *page-allocatorviaput_page).
*/ staticvoid page_pool_return_netmem(struct page_pool *pool, netmem_ref netmem)
{ int count; bool put;
put = true; if (static_branch_unlikely(&page_pool_mem_providers) && pool->mp_ops)
put = pool->mp_ops->release_netmem(pool, netmem); else
__page_pool_release_netmem_dma(pool, netmem);
/* This may be the last page returned, releasing the pool, so *itisnotsafetoreferencepoolafterwards.
*/
count = atomic_inc_return_relaxed(&pool->pages_state_release_cnt);
trace_page_pool_state_release(pool, netmem, count);
if (put) {
page_pool_clear_pp_info(netmem);
put_page(netmem_to_page(netmem));
} /* An optimization would be to call __free_pages(page, pool->p.order) *knowingpageisnotpartofpage-cache(thusavoidinga *__page_cache_release()call).
*/
}
/* BH protection not needed if current is softirq */
in_softirq = page_pool_producer_lock(pool);
ret = !__ptr_ring_produce(&pool->ring, (__force void *)netmem); if (ret)
recycle_stat_inc(pool, ring);
page_pool_producer_unlock(pool, in_softirq);
return ret;
}
/* Only allow direct recycling in special circumstances, into the *allocsidecache.E.g.duringRX-NAPIprocessingforXDP_DROPuse-case. * *Callermustprovideappropriatesafecontext.
*/ staticbool page_pool_recycle_in_cache(netmem_ref netmem, struct page_pool *pool)
{ if (unlikely(pool->alloc.count == PP_ALLOC_CACHE_SIZE)) {
recycle_stat_inc(pool, cache_full); returnfalse;
}
/* Caller MUST have verified/know (page_ref_count(page) == 1) */
pool->alloc.cache[pool->alloc.count++] = netmem;
recycle_stat_inc(pool, cached); returntrue;
}
/* If the page refcnt == 1, this will try to recycle the page. *Ifpool->dma_syncisset,we'lltrytosynctheDMAareafor *theconfiguredsizemin(dma_sync_size,pool->max_len). *Ifthepagerefcnt!=1,thenthepagewillbereturnedtomemory *subsystem.
*/ static __always_inline netmem_ref
__page_pool_put_page(struct page_pool *pool, netmem_ref netmem, unsignedint dma_sync_size, bool allow_direct)
{
lockdep_assert_no_hardirq();
/* This allocator is optimized for the XDP mode that uses *one-frame-per-page,buthavefallbacksthatactlikethe *regularpageallocatorAPIs. * *refcnt==1meanspage_poolownspage,andcanrecycleit. * *pageisNOTreusablewhenallocatedwhensystemisunder *somepressure.(page_is_pfmemalloc)
*/ if (likely(__page_pool_page_can_be_recycled(netmem))) { /* Read barrier done in page_ref_count / READ_ONCE */
/* On PREEMPT_RT the softirq can be preempted by the consumer */ if (IS_ENABLED(CONFIG_PREEMPT_RT)) returnfalse;
if (unlikely(!in_softirq())) returnfalse;
/* Allow direct recycle if we have reasons to believe that we are *inthesamecontextastheconsumerwouldrun,sothere's *nopossiblerace. *__page_pool_put_page()makessurewe'renotinhardirqcontext *andinterruptsareenabledpriortoaccessingthecache.
*/
cpuid = smp_processor_id(); if (READ_ONCE(pool->cpuid) == cpuid) returntrue;
napi = READ_ONCE(pool->p.napi);
return napi && READ_ONCE(napi->list_owner) == cpuid;
}
/* Bulk produce into ptr_ring page_pool cache */
in_softirq = page_pool_producer_lock(pool);
for (i = 0; i < bulk_len; i++) { if (__ptr_ring_produce(&pool->ring, (__force void *)bulk[i])) { /* ring full */
recycle_stat_inc(pool, ring_full); break;
}
}
page_pool_empty_alloc_cache_once(pool); if (!pool->destroy_cnt++ && pool->dma_map) { if (pool->dma_sync) { /* Disable page_pool_dma_sync_for_device() */
pool->dma_sync = false;
/* Make sure all concurrent returns that may see the old *valueofdma_sync(andthusperformasync)have *finishedbeforedoingtheunmappingbelow.Skipthe *waitifthedevicedoesn'tactuallyneedsyncing,or *iftherearenooutstandingmappedpages.
*/ if (dma_dev_need_sync(pool->p.dev) &&
!xa_empty(&pool->dma_mapped))
synchronize_net();
}
inflight = page_pool_release(pool); /* In rare cases, a driver bug may cause inflight to go negative. *Don'treschedulereleaseifinflightis0ornegative. *-If0,thepage_poolhasbeendestroyed *-ifnegative,wewillneverrecover *inbothcasesnorescheduleisnecessary.
*/ if (inflight <= 0) return;
/* Periodic warning for page pools the user can't see */
netdev = READ_ONCE(pool->slow.netdev); if (time_after_eq(jiffies, pool->defer_warn) &&
(!netdev || netdev == NET_PTR_POISON)) { int sec = (s32)((u32)jiffies - (u32)pool->defer_start) / HZ;
pr_warn("%s() stalled pool shutdown: id %u, %d inflight %d sec\n",
__func__, pool->user.id, inflight, sec);
pool->defer_warn = jiffies + DEFER_WARN_INTERVAL;
}
/* Still not ready to be disconnected, retry later */
schedule_delayed_work(&pool->release_dw, DEFER_TIME);
}
/* Flush pool alloc cache, as refill will check NUMA node */ while (pool->alloc.count) {
netmem = pool->alloc.cache[--pool->alloc.count];
page_pool_return_netmem(pool, netmem);
}
}
EXPORT_SYMBOL(page_pool_update_nid);
/* Associate a niov with a page pool. Should follow with a matching *net_mp_niov_clear_page_pool()
*/ void net_mp_niov_set_page_pool(struct page_pool *pool, struct net_iov *niov)
{
netmem_ref netmem = net_iov_to_netmem(niov);
/* Disassociate a niov from a page pool. Should only be used in the *->release_netmem()path.
*/ void net_mp_niov_clear_page_pool(struct net_iov *niov)
{
netmem_ref netmem = net_iov_to_netmem(niov);
page_pool_clear_pp_info(netmem);
}
Messung V0.5 in Prozent
¤ Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.0.31Bemerkung:
(vorverarbeitet am 2026-09-29)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.