/* Struct for every remote "destination" CPU in map */ struct bpf_cpu_map_entry {
u32 cpu; /* kthread CPU and map index */ int map_id; /* Back reference to map */
/* XDP can run multiple RX-ring queues, need __percpu enqueue store */ struct xdp_bulk_queue __percpu *bulkq;
/* Queue with potential multi-producers, and single-consumer kthread */ struct ptr_ring *queue; struct task_struct *kthread;
/* Pre-limit array size based on NR_CPUS, not final CPU check */ if (attr->max_entries > NR_CPUS) return ERR_PTR(-E2BIG);
cmap = bpf_map_area_alloc(sizeof(*cmap), NUMA_NO_NODE); if (!cmap) return ERR_PTR(-ENOMEM);
bpf_map_init_from_attr(&cmap->map, attr);
/* Alloc array for possible remote "destination" CPUs */
cmap->cpu_map = bpf_map_area_alloc(cmap->map.max_entries * sizeof(struct bpf_cpu_map_entry *),
cmap->map.numa_node); if (!cmap->cpu_map) {
bpf_map_area_free(cmap); return ERR_PTR(-ENOMEM);
}
return &cmap->map;
}
staticvoid __cpu_map_ring_cleanup(struct ptr_ring *ring)
{ /* The tear-down procedure should have made sure that queue is *empty.See__cpu_map_entry_replace()andwork-queue *invokedcpu_map_kthread_stop().Catchanybrokenbehaviour *gracefullyandwarnonce.
*/ void *ptr;
while ((ptr = ptr_ring_consume(ring))) {
WARN_ON_ONCE(1); if (unlikely(__ptr_test_bit(0, &ptr))) {
__ptr_clear_bit(0, &ptr);
kfree_skb(ptr); continue;
}
xdp_return_frame(ptr);
}
}
/* Bring struct page memory area to curr CPU. Read by *build_skb_aroundviapage_is_pfmemalloc(),andwhen *freedwrittenbypage_frag_freecall.
*/
prefetchw(page);
}
local_bh_disable();
/* Support running another XDP prog on this CPU */
cpu_map_bpf_prog_run(rcpu, frames, skbs, &ret, &stats); if (!ret.xdp_n) goto stats;
m = napi_skb_cache_get_bulk(skbs, ret.xdp_n); if (unlikely(m < ret.xdp_n)) { for (i = m; i < ret.xdp_n; i++)
xdp_return_frame(frames[i]);
if (ret.skb_n)
memmove(&skbs[m], &skbs[ret.xdp_n],
ret.skb_n * sizeof(*skbs));
kmem_alloc_drops += ret.xdp_n - m;
ret.xdp_n = m;
}
for (i = 0; i < ret.xdp_n; i++) { struct xdp_frame *xdpf = frames[i];
/* Can fail only when !skb -- already handled above */
__xdp_build_skb_from_frame(xdpf, skbs[i], xdpf->dev_rx);
}
stats: /* Feedback loop via tracepoint. *NB:keepbeforerecvtoallowmeasuringenqueue/dequeuelatency.
*/
trace_xdp_cpumap_kthread(rcpu->map_id, n, kmem_alloc_drops,
sched, &stats);
for (i = 0; i < ret.xdp_n + ret.skb_n; i++)
gro_receive_skb(&rcpu->gro, skbs[i]);
/* Flush either every 64 packets or in case of empty ring */
packets += n;
empty = __ptr_ring_empty(rcpu->queue); if (packets >= NAPI_POLL_WEIGHT || empty) {
cpu_map_gro_flush(rcpu, empty);
packets = 0;
}
local_bh_enable(); /* resched point, may call do_softirq() */
}
__set_current_state(TASK_RUNNING);
/* Make sure kthread runs on a single CPU */
kthread_bind(rcpu->kthread, cpu);
wake_up_process(rcpu->kthread);
/* Make sure kthread has been running, so kthread_stop() will not *stopthekthreadprematurelyandallpendingframesorskbs *willbehandledbythekthreadbeforekthread_stop()returns.
*/
wait_for_completion(&rcpu->kthread_running);
/* This cpu_map_entry have been disconnected from map and one *RCUgrace-periodhaveelapsed.Thus,XDPcannotqueueany *newpacketsandcannotchange/setflush_neededthatcan *findthisentry.
*/
rcpu = container_of(to_rcu_work(work), struct bpf_cpu_map_entry, free_work);
/* kthread_stop will wake_up_process and wait for it to complete. *cpu_map_kthread_run()makessurethepointerringisempty *beforeexiting.
*/
kthread_stop(rcpu->kthread);
if (rcpu->prog)
bpf_prog_put(rcpu->prog);
gro_cleanup(&rcpu->gro); /* The queue should be empty at this point */
__cpu_map_ring_cleanup(rcpu->queue);
ptr_ring_cleanup(rcpu->queue, NULL);
kfree(rcpu->queue);
free_percpu(rcpu->bulkq);
kfree(rcpu);
}
/* After the xchg of the bpf_cpu_map_entry pointer, we need to make sure the old *entryisnolongerinusebeforefreeing.Weusequeue_rcu_work()tocall *__cpu_map_entry_free()inaseparateworkqueueafterwaitingforanRCUgrace *period.Thismeansthat(a)allpendingenqueueandflushoperationshave *completed(becauseoftheRCUcallback),and(b)weareinaworkqueue *contextwherewecanstopthekthreadandwaitforittoexitbeforefreeing *everything.
*/ staticvoid __cpu_map_entry_replace(struct bpf_cpu_map *cmap,
u32 key_cpu, struct bpf_cpu_map_entry *rcpu)
{ struct bpf_cpu_map_entry *old_rcpu;
/* At this point bpf_prog->aux->refcnt == 0 and this map->refcnt == 0, *sothebpfprograms(canbemorethanonethatusedthismap)were *disconnectedfromevents.Waitforoutstandingcriticalsectionsin *theseprogramstocomplete.synchronize_rcu()belownotonly *guaranteesnofurther"XDP/bpf-side"readsagainst *bpf_cpu_map->cpu_map,butalsoensurependingflushoperations *(ifany)arecompleted.
*/
synchronize_rcu();
/* The only possible user of bpf_cpu_map_entry is *cpu_map_kthread_run().
*/ for (i = 0; i < cmap->map.max_entries; i++) { struct bpf_cpu_map_entry *rcpu;
rcpu = rcu_dereference_raw(cmap->cpu_map[i]); if (!rcpu) continue;
/* Currently the dynamically allocated elements are not counted */
usage += (u64)map->max_entries * sizeof(struct bpf_cpu_map_entry *); return usage;
}
/* Runs under RCU-read-side, plus in softirq under NAPI protection. *Thus,safepercpuvariableaccess.
*/ staticvoid bq_enqueue(struct bpf_cpu_map_entry *rcpu, struct xdp_frame *xdpf)
{ struct xdp_bulk_queue *bq = this_cpu_ptr(rcpu->bulkq);
if (unlikely(bq->count == CPU_MAP_BULK_SIZE))
bq_flush_to_queue(bq);
/* Notice, xdp_buff/page MUST be queued here, long enough for *drivertocodeinvokingustofinished,duetodriver *(e.g.ixgbe)recycletricksbasedonpage-refcnt. * *Thus,incomingxdp_frameisalwaysqueuedhere(elsewerace *withanotherCPUonpage-refcntandremainingdrivercode). *Queuetimeisveryshort,asdriverwillinvokeflush *operation,whencompletingnapi->pollcall.
*/
bq->q[bq->count++] = xdpf;
if (!bq->flush_node.prev) { struct list_head *flush_list = bpf_net_ctx_get_cpu_map_flush_list();
list_add(&bq->flush_node, flush_list);
}
}
int cpu_map_enqueue(struct bpf_cpu_map_entry *rcpu, struct xdp_frame *xdpf, struct net_device *dev_rx)
{ /* Info needed when constructing SKB on remote CPU */
xdpf->dev_rx = dev_rx;
bq_enqueue(rcpu, xdpf); return0;
}
int cpu_map_generic_redirect(struct bpf_cpu_map_entry *rcpu, struct sk_buff *skb)
{ int ret;
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.