struct bpf_mem_cache { /* per-cpu list of free objects of size 'unit_size'. *Allaccessesaredonewithinterruptsdisabledand'active'counter *protectionwith__llist_add()and__llist_del_first().
*/ struct llist_head free_llist;
local_t active;
/* Operations on the free_list from unit_alloc/unit_free/bpf_mem_refill *aresequencedbyper-cpu'active'counter.Butunit_free()cannot *fail.When'active'isbusytheunit_free()willaddanobjectto *free_llist_extra.
*/ struct llist_head free_llist_extra;
struct irq_work refill_work; struct obj_cgroup *objcg; int unit_size; /* count of objects in free_llist */ int free_cnt; int low_watermark, high_watermark, batch; int percpu_size; bool draining; struct bpf_mem_cache *tgt;
/* list of objects to be freed after RCU GP */ struct llist_head free_by_rcu; struct llist_node *free_by_rcu_tail; struct llist_head waiting_for_gp; struct llist_node *waiting_for_gp_tail; struct rcu_head rcu;
atomic_t call_rcu_in_progress; struct llist_head free_llist_extra_rcu;
/* list of objects to be freed after RCU tasks trace GP */ struct llist_head free_by_rcu_ttrace; struct llist_head waiting_for_gp_ttrace; struct rcu_head rcu_ttrace;
atomic_t call_rcu_ttrace_in_progress;
};
staticvoid inc_active(struct bpf_mem_cache *c, unsignedlong *flags)
{ if (IS_ENABLED(CONFIG_PREEMPT_RT)) /* In RT irq_work runs in per-cpu kthread, so disable *interruptstoavoidpreemptionandinterruptsand *reducethechanceofbpfprogexecutingonthiscpu *whenactivecounterisbusy.
*/
local_irq_save(*flags); /* alloc_bulk runs from irq_work which will not preempt a bpf *programthatdoesunit_alloc/unit_freesinceIRQsare *disabledthere.Thereisnoracetoincrement'active' *counter.Itprotectsfree_llistfromcorruptionincaseNMI *bpfprogpreemptedthisloop.
*/
WARN_ON_ONCE(local_inc_return(&c->active) != 1);
}
for (i = 0; i < cnt; i++) { /* *Forevery'c'llist_del_first(&c->free_by_rcu_ttrace);is *doneonlybyoneCPU==currentCPU.OtherCPUsmight *llist_add()andllist_del_all()inparallel.
*/
obj = llist_del_first(&c->free_by_rcu_ttrace); if (!obj) break;
add_obj_to_free_list(c, obj);
} if (i >= cnt) return;
for (; i < cnt; i++) {
obj = llist_del_first(&c->waiting_for_gp_ttrace); if (!obj) break;
add_obj_to_free_list(c, obj);
} if (i >= cnt) return;
memcg = get_memcg(c);
old_memcg = set_active_memcg(memcg); for (; i < cnt; i++) { /* Allocate, but don't deplete atomic reserves that typical *GFP_ATOMICwoulddo.irq_workrunsonthiscpuandkmalloc *willallocatefromthecurrentnumanodewhichiswhatwe *wanthere.
*/
obj = __alloc(c, node, gfp); if (!obj) break;
add_obj_to_free_list(c, obj);
}
set_active_memcg(old_memcg);
mem_cgroup_put(memcg);
}
/* bpf_mem_cache is a per-cpu object. Freeing happens in irq_work. *Nothingracestoaddtofree_by_rcu_ttracelist.
*/
llist_add(llnode, &c->free_by_rcu_ttrace);
}
if (unlikely(READ_ONCE(c->draining))) {
__free_rcu(&c->rcu_ttrace); return;
}
/* Use call_rcu_tasks_trace() to wait for sleepable progs to finish. *IfRCUTasksTracegraceperiodimpliesRCUgraceperiod,free *theseelementsdirectly,elseusecall_rcu()towaitfornormal *progstofinishandfinallydofree_one()oneachelement.
*/
call_rcu_tasks_trace(&c->rcu_ttrace, __free_rcu_tasks_trace);
}
/* Racy access to free_cnt. It doesn't need to be 100% accurate */
cnt = c->free_cnt; if (cnt < c->low_watermark) /* irq_work runs on this cpu and kmalloc will allocate *fromthecurrentnumanodewhichiswhatwewanthere.
*/
alloc_bulk(c, c->batch, NUMA_NO_NODE, true); elseif (cnt > c->high_watermark)
free_bulk(c);
if (ma->cache) {
for_each_possible_cpu(cpu) {
c = per_cpu_ptr(ma->cache, cpu);
check_mem_cache(c);
}
} if (ma->caches) {
for_each_possible_cpu(cpu) {
cc = per_cpu_ptr(ma->caches, cpu); for (i = 0; i < NUM_CACHES; i++) {
c = &cc->cache[i];
check_mem_cache(c);
}
}
}
}
void bpf_mem_alloc_destroy(struct bpf_mem_alloc *ma)
{ struct bpf_mem_caches *cc; struct bpf_mem_cache *c; int cpu, i, rcu_in_progress;
if (ma->cache) {
rcu_in_progress = 0;
for_each_possible_cpu(cpu) {
c = per_cpu_ptr(ma->cache, cpu);
WRITE_ONCE(c->draining, true);
irq_work_sync(&c->refill_work);
drain_mem_cache(c);
rcu_in_progress += atomic_read(&c->call_rcu_ttrace_in_progress);
rcu_in_progress += atomic_read(&c->call_rcu_in_progress);
}
obj_cgroup_put(ma->objcg);
destroy_mem_alloc(ma, rcu_in_progress);
} if (ma->caches) {
rcu_in_progress = 0;
for_each_possible_cpu(cpu) {
cc = per_cpu_ptr(ma->caches, cpu); for (i = 0; i < NUM_CACHES; i++) {
c = &cc->cache[i];
WRITE_ONCE(c->draining, true);
irq_work_sync(&c->refill_work);
drain_mem_cache(c);
rcu_in_progress += atomic_read(&c->call_rcu_ttrace_in_progress);
rcu_in_progress += atomic_read(&c->call_rcu_in_progress);
}
}
obj_cgroup_put(ma->objcg);
destroy_mem_alloc(ma, rcu_in_progress);
}
}
/* notrace is necessary here and in other functions to make sure *bpfprogramscannotattachtothemandcausellistcorruptions.
*/ staticvoid notrace *unit_alloc(struct bpf_mem_cache *c)
{ struct llist_node *llnode = NULL; unsignedlong flags; int cnt = 0;
/* Disable irqs to prevent the following race for majority of prog types: *prog_A *bpf_mem_alloc *preemptionorirq->prog_B *bpf_mem_alloc * *butprog_Bcouldbeaperf_eventNMIprog. *Useper-cpu'active'countertoorderfree_listaccessbetween *unit_alloc/unit_free/bpf_mem_refill.
*/
local_irq_save(flags); if (local_inc_return(&c->active) == 1) {
llnode = __llist_del_first(&c->free_llist); if (llnode) {
cnt = --c->free_cnt;
*(struct bpf_mem_cache **)llnode = c;
}
}
local_dec(&c->active);
WARN_ON(cnt < 0);
if (cnt < c->low_watermark)
irq_work_raise(c); /* Enable IRQ after the enqueue of irq work completes, so irq work *willrunafterIRQisenabledandfree_llistmayberefilledby *irqworkbeforeothertaskpreemptscurrenttask.
*/
local_irq_restore(flags);
return llnode;
}
/* Though 'ptr' object could have been allocated on a different cpu *addittothefree_llistofthecurrentcpu. *Letkfree()logicdealwithitwhenit'slatercalledfromirq_work.
*/ staticvoid notrace unit_free(struct bpf_mem_cache *c, void *ptr)
{ struct llist_node *llnode = ptr - LLIST_NODE_SZ; unsignedlong flags; int cnt = 0;
if (cnt > c->high_watermark) /* free few objects from current cpu into global kmalloc pool */
irq_work_raise(c); /* Enable IRQ after irq_work_raise() completes, otherwise when current *taskispreemptedbytaskwhichdoesunit_alloc(),unit_alloc()may *returnNULLunexpectedlybecauseirqworkisalreadypendingbutcan *notbeentriggeredandfree_llistcannotberefilledtimely.
*/
local_irq_restore(flags);
}
local_irq_save(flags); if (local_inc_return(&c->active) == 1) { if (__llist_add(llnode, &c->free_by_rcu))
c->free_by_rcu_tail = llnode;
} else {
llist_add(llnode, &c->free_llist_extra_rcu);
}
local_dec(&c->active);
if (!atomic_read(&c->call_rcu_in_progress))
irq_work_raise(c);
local_irq_restore(flags);
}
/* Called from BPF program or from sys_bpf syscall. *Inbothcasesmigrationisdisabled.
*/ void notrace *bpf_mem_alloc(struct bpf_mem_alloc *ma, size_t size)
{ int idx; void *ret;
if (!size) return NULL;
if (!ma->percpu)
size += LLIST_NODE_SZ;
idx = bpf_mem_cache_idx(size); if (idx < 0) return NULL;
ret = unit_alloc(this_cpu_ptr(ma->caches)->cache + idx); return !ret ? NULL : ret + LLIST_NODE_SZ;
}
/* Directly does a kfree() without putting 'ptr' back to the free_llist *forreuseandwithoutwaitingforarcu_tasks_tracegp. *Thecallermustfirstgothroughthercu_tasks_tracegpfor'ptr' *beforecallingbpf_mem_cache_raw_free(). *Itcouldbeusedwhenthercu_tasks_tracecallbackdoesnothave *aholdontheoriginalbpf_mem_allocobjectthatallocatedthe *'ptr'.Thisshouldonlybeusedintheuncommoncodepath. *Otherwise,thebpf_mem_alloc'sfree_llistcannotberefilled *andmayaffectperformance.
*/ void bpf_mem_cache_raw_free(void *ptr)
{ if (!ptr) return;
kfree(ptr - LLIST_NODE_SZ);
}
/* When flags == GFP_KERNEL, it signals that the caller will not cause *deadlockwhenusingkmalloc.bpf_mem_cache_alloc_flags()willuse *kmallocifthefree_llistisempty.
*/ void notrace *bpf_mem_cache_alloc_flags(struct bpf_mem_alloc *ma, gfp_t flags)
{ struct bpf_mem_cache *c; void *ret;
c = this_cpu_ptr(ma->cache);
ret = unit_alloc(c); if (!ret && flags == GFP_KERNEL) { struct mem_cgroup *memcg, *old_memcg;
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.