/* bpf_check() is a static code analyzer that walks eBPF program *instructionbyinstructionandupdatesregister/stackstate. *Allpathsofconditionalbranchesareanalyzeduntil'bpf_exit'insn. * *Thefirstpassisdepth-first-searchtocheckthattheprogramisaDAG. *Itrejectsthefollowingprograms: *-largerthanBPF_MAXINSNSinsns *-ifloopispresent(detectedviaback-edge) *-unreachableinsnsexist(shouldn'tbeaforest.program=onefunction) *-outofboundsormalformedjumps *Thesecondpassisallpossiblepathdescentfromthe1stinsn. *Sinceit'sanalyzingallpathsthroughtheprogram,thelengthofthe *analysisislimitedto64kinsn,whichmaybehiteveniftotalnumberof *insnislessthen4K,buttherearetoomanybranchesthatchangestack/regs. *Numberof'branchestobeanalyzed'islimitedto1k * *Onentrytoeachinstruction,eachregisterhasatype,andtheinstruction *changesthetypesoftheregistersdependingoninstructionsemantics. *IfinstructionisBPF_MOV64_REG(BPF_REG_1,BPF_REG_5),thentypeofR5is *copiedtoR1. * *Allregistersare64-bit. *R0-returnregister *R1-R5argumentpassingregisters *R6-R9calleesavedregisters *R10-framepointerread-only * *AtthestartofBPFprogramtheregisterR1containsapointertobpf_context *andhastypePTR_TO_CTX. * *Verifiertracksarithmeticoperationsonpointersincase: *BPF_MOV64_REG(BPF_REG_1,BPF_REG_10), *BPF_ALU64_IMM(BPF_ADD,BPF_REG_1,-20), *1stinsncopiesR10(whichhasFRAME_PTR)typeintoR1 *and2ndarithmeticinstructionispatternmatchedtorecognize *thatitwantstoconstructapointertosomeelementwithinstack. *Soafter2ndinsn,theregisterR1hastypePTR_TO_STACK *(and-20constantissavedforfurtherstackboundschecking). *Meaningthatthisregisapointertostackplusknownimmediateconstant. * *MostofthetimetheregistershaveSCALAR_VALUEtype,which *meanstheregisterhassomevalue,butit'snotavalidpointer. *(likepointerpluspointerbecomesSCALAR_VALUEtype) * *Whenverifierseesloadorstoreinstructionsthetypeofbaseregister *canbe:PTR_TO_MAP_VALUE,PTR_TO_CTX,PTR_TO_STACK,PTR_TO_SOCKET.Theseare *fourpointertypesrecognizedbycheck_mem_access()function. * *PTR_TO_MAP_VALUEmeansthatthisregisterispointingto'mapelementvalue' *andtherangeof[ptr,ptr+map'svalue_size)isaccessible. * *registersusedtopassvaluestofunctioncallsarecheckedagainst *functionargumentconstraints. * *ARG_PTR_TO_MAP_KEYisoneofsuchargumentconstraints. *Itmeansthattheregistertypepassedtothisfunctionmustbe *PTR_TO_STACKanditwillbeusedinsidethefunctionas *'pointertomapelementkey' * *Forexampletheargumentconstraintsforbpf_map_lookup_elem(): *.ret_type=RET_PTR_TO_MAP_VALUE_OR_NULL, *.arg1_type=ARG_CONST_MAP_PTR, *.arg2_type=ARG_PTR_TO_MAP_KEY, * *ret_typesaysthatthisfunctionreturns'pointertomapelemvalueornull' *functionexpects1stargumenttobeaconstpointerto'structbpf_map'and *2ndargumentshouldbeapointertostack,whichwillbeusedinside *thehelperfunctionasapointertomapelementkey. * *Onthekernelsidethehelperfunctionlookslike: *u64bpf_map_lookup_elem(u64r1,u64r2,u64r3,u64r4,u64r5) *{ *structbpf_map*map=(structbpf_map*)(unsignedlong)r1; *void*key=(void*)(unsignedlong)r2; *void*value; * *herekernelcanaccess'key'and'map'pointerssafely,knowingthat *[key,key+map->key_size)bytesarevalidandwereinitializedon *thestackofeBPFprogram. *} * *CorrespondingeBPFprogrammaylooklike: *BPF_MOV64_REG(BPF_REG_2,BPF_REG_10),// after this insn R2 type is FRAME_PTR *BPF_ALU64_IMM(BPF_ADD,BPF_REG_2,-4),// after this insn R2 type is PTR_TO_STACK *BPF_LD_MAP_FD(BPF_REG_1,map_fd),// after this insn R1 type is CONST_PTR_TO_MAP *BPF_RAW_INSN(BPF_JMP|BPF_CALL,0,0,0,BPF_FUNC_map_lookup_elem), *hereverifierlooksatprototypeofmap_lookup_elem()andsees: *.arg1_type==ARG_CONST_MAP_PTRandR1->type==CONST_PTR_TO_MAP,whichisok, *NowverifierknowsthatthismaphaskeyofR1->map_ptr->key_sizebytes * *Then.arg2_type==ARG_PTR_TO_MAP_KEYandR2->type==PTR_TO_STACK,oksofar, *Nowverifierchecksthat[R2,R2+map'skey_size)arewithinstacklimits *andwereinitializedpriortothiscall. *Ifit'sok,thenverifierallowsthisBPF_CALLinsnandlooksat *.ret_typewhichisRET_PTR_TO_MAP_VALUE_OR_NULL,soitsets *R0->type=PTR_TO_MAP_VALUE_OR_NULLwhichmeansbpf_map_lookup_elem()function *returnseitherpointertomapvalueorNULL. * *WhentypePTR_TO_MAP_VALUE_OR_NULLpassesthrough'if(reg!=0)goto+off' *insn,theregisterholdingthatpointerinthetruebranchchangesstateto *PTR_TO_MAP_VALUEandthesameregisterchangesstatetoCONST_IMMinthefalse *branch.Seecheck_cond_jmp_op(). * *AfterthecallR0issettoreturntypeofthefunctionandregistersR1-R5 *aresettoNOT_INITtoindicatethattheyarenolongerreadable. * *Thefollowingreferencetypesrepresentapotentialreferencetoakernel *resourcewhich,afterfirstbeingallocated,mustbecheckedandfreedby *theBPFprogram: *-PTR_TO_SOCKET_OR_NULL,PTR_TO_SOCKET * *Whentheverifierseesahelpercallreturnareferencetype,itallocatesa *pointeridforthereferenceandstoresitinthecurrentfunctionstate. *SimilartothewaythatPTR_TO_MAP_VALUE_OR_NULLisconvertedinto *PTR_TO_MAP_VALUE,PTR_TO_SOCKET_OR_NULLbecomesPTR_TO_SOCKETwhenthetype *passesthroughaNULL-checkconditional.Forthebranchwhereinthestateis *changedtoCONST_IMM,theverifierreleasesthereference. * *Foreachhelperfunctionthatallocatesareference,suchas *bpf_sk_lookup_tcp(),thereisacorrespondingreleasefunction,suchas *bpf_sk_release().Whenareferencetypepassesintothereleasefunction, *theverifieralsoreleasesthereference.Ifanyuncheckedorunreleased *referenceremainsattheendoftheprogram,theverifierrejectsit.
*/
/* verifier_state + insn_idx are pushed to stack when branch is encountered */ struct bpf_verifier_stack_elem { /* verifier state is 'st' *beforeprocessinginstruction'insn_idx' *andafterprocessinginstruction'prev_insn_idx'
*/ struct bpf_verifier_state st; int insn_idx; int prev_insn_idx; struct bpf_verifier_stack_elem *next; /* length of verifier log at the time this state was pushed on stack */
u32 log_pos;
};
if (is_ptr_cast_function(func_id))
ref_obj_uses++; if (is_acquire_function(func_id, map))
ref_obj_uses++; if (is_dynptr_ref_function(func_id))
ref_obj_uses++;
/* If the dynptr has a ref_obj_id, then we need to invalidate *twothings: * *1)Anydynptrswithamatchingref_obj_id(clones) *2)Anyslicesderivedfromthisdynptr.
*/
/* Invalidate any slices associated with this dynptr */
WARN_ON_ONCE(release_reference(env, ref_obj_id));
/* Invalidate any dynptr clones */ for (i = 1; i < state->allocated_stack / BPF_REG_SIZE; i++) { if (state->stack[i].spilled_ptr.ref_obj_id != ref_obj_id) continue;
/* it should always be the case that if the ref obj id *matchesthenthestackslotalsobelongstoa *dynptr
*/ if (state->stack[i].slot_type[0] != STACK_DYNPTR) {
verifier_bug(env, "misconfigured ref_obj_id"); return -EFAULT;
} if (state->stack[i].spilled_ptr.dynptr.first_slot)
invalidate_dynptr(env, state, i);
}
staticint destroy_if_dynptr_stack_slot(struct bpf_verifier_env *env, struct bpf_func_state *state, int spi)
{ struct bpf_func_state *fstate; struct bpf_reg_state *dreg; int i, dynptr_id;
/* We always ensure that STACK_DYNPTR is never set partially, *hencejustcheckingforslot_type[0]isenough.Thisis *differentforSTACK_SPILL,whereitmaybeonlysetfor *1byte,socodehastouseis_spilled_reg.
*/ if (state->stack[spi].slot_type[0] != STACK_DYNPTR) return0;
/* Reposition spi to first slot */ if (!state->stack[spi].spilled_ptr.dynptr.first_slot)
spi = spi + 1;
/* Writing partially to one dynptr stack slot destroys both. */ for (i = 0; i < BPF_REG_SIZE; i++) {
state->stack[spi].slot_type[i] = STACK_INVALID;
state->stack[spi - 1].slot_type[i] = STACK_INVALID;
}
dynptr_id = state->stack[spi].spilled_ptr.id; /* Invalidate any slices associated with this dynptr */
bpf_for_each_reg_in_vstate(env->cur_state, fstate, dreg, ({ /* Dynptr slices are only PTR_TO_MEM_OR_NULL and PTR_TO_MEM */ if (dreg->type != (PTR_TO_MEM | PTR_MAYBE_NULL) && dreg->type != PTR_TO_MEM) continue; if (dreg->dynptr_id == dynptr_id)
mark_reg_invalid(env, dreg);
}));
/* Do not release reference state, we are destroying dynptr on stack, *notusingsomehelpertoreleaseit.Justresetregister.
*/
__mark_reg_not_init(env, &state->stack[spi].spilled_ptr);
__mark_reg_not_init(env, &state->stack[spi - 1].spilled_ptr);
/* Same reason as unmark_stack_slots_dynptr above */
state->stack[spi].spilled_ptr.live |= REG_LIVE_WRITTEN;
state->stack[spi - 1].spilled_ptr.live |= REG_LIVE_WRITTEN;
return0;
}
staticbool is_dynptr_reg_valid_uninit(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
{ int spi;
if (reg->type == CONST_PTR_TO_DYNPTR) returnfalse;
spi = dynptr_get_spi(env, reg);
/* -ERANGE (i.e. spi not falling into allocated stack slots) isn't an *errorbecausethisjustmeansthestackstatehasn'tbeenupdatedyet. *Wewilldocheck_mem_accesstocheckandupdatestackboundslater.
*/ if (spi < 0 && spi != -ERANGE) returnfalse;
/* We don't need to check if the stack slots are marked by previous *dynptrinitializationsbecauseweallowoverwritingexistingunreferenced *STACK_DYNPTRslots,seemark_stack_slots_dynptrwhichcalls *destroy_if_dynptr_stack_slottoensuredynptrobjectsattheslotsweare *touchingarecompletelydestructedbeforewereinitializethemforanew *one.Forreferencedones,destroy_if_dynptr_stack_slotreturnsanerrorearly *insteadofdelayingituntiltheendwheretheuserwillget"Unreleased *reference"error.
*/ returntrue;
}
staticbool is_dynptr_reg_valid_init(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
{ struct bpf_func_state *state = func(env, reg); int i, spi;
/* This already represents first slot of initialized bpf_dynptr. * *CONST_PTR_TO_DYNPTRalreadyhasfixedandvar_offas0dueto *check_func_arg_reg_off'slogic,sowedon'tneedtocheckits *offsetandalignment.
*/ if (reg->type == CONST_PTR_TO_DYNPTR) returntrue;
spi = dynptr_get_spi(env, reg); if (spi < 0) returnfalse; if (!state->stack[spi].spilled_ptr.dynptr.first_slot) returnfalse;
for (i = 0; i < BPF_REG_SIZE; i++) { if (state->stack[spi].slot_type[i] != STACK_DYNPTR ||
state->stack[spi - 1].slot_type[i] != STACK_DYNPTR) returnfalse;
}
staticbool is_iter_reg_valid_uninit(struct bpf_verifier_env *env, struct bpf_reg_state *reg, int nr_slots)
{ struct bpf_func_state *state = func(env, reg); int spi, i, j;
/* For -ERANGE (i.e. spi not falling into allocated stack slots), we *willdocheck_mem_accesstocheckandupdatestackboundslater,so *returntrueforthatcase.
*/
spi = iter_get_spi(env, reg, nr_slots); if (spi == -ERANGE) returntrue; if (spi < 0) returnfalse;
for (i = 0; i < nr_slots; i++) { struct bpf_stack_state *slot = &state->stack[spi - i];
for (j = 0; j < BPF_REG_SIZE; j++) if (slot->slot_type[j] == STACK_ITER) returnfalse;
}
returntrue;
}
staticint is_iter_reg_valid_init(struct bpf_verifier_env *env, struct bpf_reg_state *reg, struct btf *btf, u32 btf_id, int nr_slots)
{ struct bpf_func_state *state = func(env, reg); int spi, i, j;
for (i = 0; i < nr_slots; i++) { struct bpf_stack_state *slot = &state->stack[spi - i]; struct bpf_reg_state *st = &slot->spilled_ptr;
if (st->type & PTR_UNTRUSTED) return -EPROTO; /* only main (first) slot has ref_obj_id set */ if (i == 0 && !st->ref_obj_id) return -EINVAL; if (i != 0 && st->ref_obj_id) return -EINVAL; if (st->iter.btf != btf || st->iter.btf_id != btf_id) return -EINVAL;
for (j = 0; j < BPF_REG_SIZE; j++) if (slot->slot_type[j] != STACK_ITER) return -EINVAL;
}
return0;
}
staticint acquire_irq_state(struct bpf_verifier_env *env, int insn_idx); staticint release_irq_state(struct bpf_verifier_state *state, int id);
staticint mark_stack_slot_irq_flag(struct bpf_verifier_env *env, struct bpf_kfunc_call_arg_meta *meta, struct bpf_reg_state *reg, int insn_idx, int kfunc_class)
{ struct bpf_func_state *state = func(env, reg); struct bpf_stack_state *slot; struct bpf_reg_state *st; int spi, i, id;
spi = irq_flag_get_spi(env, reg); if (spi < 0) return spi;
id = acquire_irq_state(env, insn_idx); if (id < 0) return id;
slot = &state->stack[spi];
st = &slot->spilled_ptr;
__mark_reg_known_zero(st);
st->type = PTR_TO_STACK; /* we don't have dedicated reg type */
st->live |= REG_LIVE_WRITTEN;
st->ref_obj_id = id;
st->irq.kfunc_class = kfunc_class;
for (i = 0; i < BPF_REG_SIZE; i++)
slot->slot_type[i] = STACK_IRQ_FLAG;
mark_stack_slot_scratched(env, spi); return0;
}
staticint unmark_stack_slot_irq_flag(struct bpf_verifier_env *env, struct bpf_reg_state *reg, int kfunc_class)
{ struct bpf_func_state *state = func(env, reg); struct bpf_stack_state *slot; struct bpf_reg_state *st; int spi, i, err;
spi = irq_flag_get_spi(env, reg); if (spi < 0) return spi;
slot = &state->stack[spi];
st = &slot->spilled_ptr;
verbose(env, "irq flag acquired by %s kfuncs cannot be restored with %s kfuncs\n",
flag_kfunc, used_kfunc); return -EINVAL;
}
err = release_irq_state(env->cur_state, st->ref_obj_id);
WARN_ON_ONCE(err && err != -EACCES); if (err) { int insn_idx = 0;
for (int i = 0; i < env->cur_state->acquired_refs; i++) { if (env->cur_state->refs[i].id == env->cur_state->active_irq_id) {
insn_idx = env->cur_state->refs[i].insn_idx; break;
}
}
verbose(env, "cannot restore irq state out of order, expected id=%d acquired at insn_idx=%d\n",
env->cur_state->active_irq_id, insn_idx); return err;
}
__mark_reg_not_init(env, st);
/* see unmark_stack_slots_dynptr() for why we need to set REG_LIVE_WRITTEN */
st->live |= REG_LIVE_WRITTEN;
for (i = 0; i < BPF_REG_SIZE; i++)
slot->slot_type[i] = STACK_INVALID;
/* For -ERANGE (i.e. spi not falling into allocated stack slots), we *willdocheck_mem_accesstocheckandupdatestackboundslater,so *returntrueforthatcase.
*/
spi = irq_flag_get_spi(env, reg); if (spi == -ERANGE) returntrue; if (spi < 0) returnfalse;
slot = &state->stack[spi];
for (i = 0; i < BPF_REG_SIZE; i++) if (slot->slot_type[i] == STACK_IRQ_FLAG) returnfalse; returntrue;
}
spi = irq_flag_get_spi(env, reg); if (spi < 0) return -EINVAL;
slot = &state->stack[spi];
st = &slot->spilled_ptr;
if (!st->ref_obj_id) return -EINVAL;
for (i = 0; i < BPF_REG_SIZE; i++) if (slot->slot_type[i] != STACK_IRQ_FLAG) return -EINVAL; return0;
}
/* Check if given stack slot is "special": *-spilledregisterstate(STACK_SPILL); *-dynptrstate(STACK_DYNPTR); *-iterstate(STACK_ITER). *-irqflagstate(STACK_IRQ_FLAG)
*/ staticbool is_stack_slot_special(conststruct bpf_stack_state *stack)
{ enum bpf_stack_slot_type type = stack->slot_type[BPF_REG_SIZE - 1];
switch (type) { case STACK_SPILL: case STACK_DYNPTR: case STACK_ITER: case STACK_IRQ_FLAG: returntrue; case STACK_INVALID: case STACK_MISC: case STACK_ZERO: returnfalse; default:
WARN_ONCE(1, "unknown stack slot type %d\n", type); returntrue;
}
}
/* The reg state of a pointer or a bounded scalar was saved when *itwasspilledtothestack.
*/ staticbool is_spilled_reg(conststruct bpf_stack_state *stack)
{ return stack->slot_type[BPF_REG_SIZE - 1] == STACK_SPILL;
}
/* resize an array from old_n items to new_n items. the array is reallocated if it's too *smalltoholdnew_nitems.newitemsarezeroedoutifthearraygrows. * *Contrarytokrealloc_array,doesnotfreearrifnew_niszero.
*/ staticvoid *realloc_array(void *arr, size_t old_n, size_t new_n, size_t size)
{
size_t alloc_size; void *new_arr;
staticint resize_reference_state(struct bpf_verifier_state *state, size_t n)
{
state->refs = realloc_array(state->refs, state->acquired_refs, n, sizeof(struct bpf_reference_state)); if (!state->refs) return -ENOMEM;
state->acquired_refs = n; return0;
}
/* Possibly update state->allocated_stack to be at least size bytes. Also *possiblyupdatethefunction'shigh-watermarkinitsbpf_subprog_info.
*/ staticint grow_stack_state(struct bpf_verifier_env *env, struct bpf_func_state *state, int size)
{
size_t old_n = state->allocated_stack / BPF_REG_SIZE, n;
/* The stack size is always a multiple of BPF_REG_SIZE. */
size = round_up(size, BPF_REG_SIZE);
n = size / BPF_REG_SIZE;
if (old_n >= n) return0;
state->stack = realloc_array(state->stack, old_n, n, sizeof(struct bpf_stack_state)); if (!state->stack) return -ENOMEM;
state->allocated_stack = size;
/* update known max for given subprogram */ if (env->subprog_info[state->subprogno].stack_depth < size)
env->subprog_info[state->subprogno].stack_depth = size;
return0;
}
/* Acquire a pointer id from the env and update the state->refs to include *thisnewpointerreference. *Onsuccess,returnsavalidpointeridtoassociatewiththeregister *Onfailure,returnsanegativeerrno.
*/ staticstruct bpf_reference_state *acquire_reference_state(struct bpf_verifier_env *env, int insn_idx)
{ struct bpf_verifier_state *state = env->cur_state; int new_ofs = state->acquired_refs; int err;
staticvoid free_verifier_state(struct bpf_verifier_state *state, bool free_self)
{ int i;
for (i = 0; i <= state->curframe; i++) {
free_func_state(state->frame[i]);
state->frame[i] = NULL;
}
kfree(state->refs);
clear_jmp_history(state); if (free_self)
kfree(state);
}
/* struct bpf_verifier_state->parent refers to states *thatareineitherofenv->{expored_states,free_list}. *Inbothcasesthestateiscontainedinstructbpf_verifier_state_list.
*/ staticstruct bpf_verifier_state_list *state_parent_as_list(struct bpf_verifier_state *st)
{ if (st->parent) return container_of(st->parent, struct bpf_verifier_state_list, state); return NULL;
}
/* A state can be freed if it is no longer referenced: *-isintheenv->free_list; *-hasnochildrenstates;
*/ staticvoid maybe_free_verifier_state(struct bpf_verifier_env *env, struct bpf_verifier_state_list *sl)
{ if (!sl->in_free_list
|| sl->state.branches != 0
|| incomplete_read_marks(env, &sl->state)) return;
list_del(&sl->node);
free_verifier_state(&sl->state, false);
kfree(sl);
env->free_list_size--;
}
/* copy verifier state from src to dst growing dst stack space *whennecessarytoaccommodatelargersrcstack
*/ staticint copy_func_state(struct bpf_func_state *dst, conststruct bpf_func_state *src)
{
memcpy(dst, src, offsetof(struct bpf_func_state, stack)); return copy_stack_state(dst, src);
}
staticint copy_verifier_state(struct bpf_verifier_state *dst_state, conststruct bpf_verifier_state *src)
{ struct bpf_func_state *dst; int i, err;
if (!info) return NULL; for (i = 0; i < info->num_visits; i++) if (memcmp(callchain, &visits[i].callchain, sizeof(*callchain)) == 0) return &visits[i]; return NULL;
}
/* Form a string '(callsite#1,callsite#2,...,scc)' in env->tmp_str_buf */ staticchar *format_callchain(struct bpf_verifier_env *env, struct bpf_scc_callchain *callchain)
{ char *buf = env->tmp_str_buf; int i, delta = 0;
/* If callchain for @st exists (@st is in some SCC), make it empty: *-setvisit->entry_statetoNULL; *-flushaccumulatedbackedges.
*/ staticint maybe_exit_scc(struct bpf_verifier_env *env, struct bpf_verifier_state *st)
{ struct bpf_scc_callchain *callchain = &env->callchain_buf; struct bpf_scc_visit *visit;
if (!compute_scc_callchain(env, st, callchain)) return0;
visit = scc_visit_lookup(env, callchain); if (!visit) { /* *IfpathtraversalstopsinsideanSCC,correspondingbpf_scc_visit *mustexistfornon-speculativepaths.Fornon-speculativepaths *traversalstopswhen: *a.Verificationerrorisfound,maybe_exit_scc()isnotcalled. *b.ToplevelBPF_EXITisreached.ToplevelBPF_EXITisnotamember *ofanySCC. *c.Acheckpointisreachedandmatched.Checkpointsarecreatedby *is_state_visited(),whichcallsmaybe_enter_scc(),whichallocates *bpf_scc_visitinstancesforcheckpointswithinSCCs. *(c)istheonlycasethatcanreachthispoint.
*/ if (!st->speculative) {
verifier_bug(env, "scc exit: no visit info for call chain %s",
format_callchain(env, callchain)); return -EFAULT;
} return0;
} if (visit->entry_state != st) return0; if (env->log.level & BPF_LOG_LEVEL2)
verbose(env, "SCC exit %s\n", format_callchain(env, callchain));
visit->entry_state = NULL;
env->num_backedges -= visit->num_backedges;
visit->num_backedges = 0;
update_peak_states(env); return propagate_backedges(env, visit);
}
if (!compute_scc_callchain(env, st, callchain)) {
verifier_bug(env, "add backedge: no SCC in verification path, insn_idx %d",
st->insn_idx); return -EFAULT;
}
visit = scc_visit_lookup(env, callchain); if (!visit) {
verifier_bug(env, "add backedge: no visit info for call chain %s",
format_callchain(env, callchain)); return -EFAULT;
} if (env->log.level & BPF_LOG_LEVEL2)
verbose(env, "SCC backedge %s\n", format_callchain(env, callchain));
backedge->next = visit->backedges;
visit->backedges = backedge;
visit->num_backedges++;
env->num_backedges++;
update_peak_states(env); return0;
}
/* bpf_reg_state->live marks for registers in a state @st are incomplete, *ifstate@stisinsomeSCCandnotallexecutionpathsstartingatthis *SCCarefullyexplored.
*/ staticbool incomplete_read_marks(struct bpf_verifier_env *env, struct bpf_verifier_state *st)
{ struct bpf_scc_callchain *callchain = &env->callchain_buf; struct bpf_scc_visit *visit;
if (!compute_scc_callchain(env, st, callchain)) returnfalse;
visit = scc_visit_lookup(env, callchain); if (!visit) returnfalse; return !!visit->backedges;
}
/* Mark the 'variable offset' part of a register as zero. This should be *usedonlyonregistersholdingapointertype.
*/ staticvoid __mark_reg_known_zero(struct bpf_reg_state *reg)
{
__mark_reg_known(reg, 0);
}
staticvoid mark_reg_known_zero(struct bpf_verifier_env *env, struct bpf_reg_state *regs, u32 regno)
{ if (WARN_ON(regno >= MAX_BPF_REG)) {
verbose(env, "mark_reg_known_zero(regs, %u)\n", regno); /* Something bad happened, let's kill all regs */ for (regno = 0; regno < MAX_BPF_REG; regno++)
__mark_reg_not_init(env, regs + regno); return;
}
__mark_reg_known_zero(regs + regno);
}
staticvoid __mark_dynptr_reg(struct bpf_reg_state *reg, enum bpf_dynptr_type type, bool first_slot, int dynptr_id)
{ /* reg->type has no meaning for STACK_DYNPTR, but when we set reg for *callbackarguments,itdoesneedtobeCONST_PTR_TO_DYNPTR,sosimply *setitunconditionallyasitisignoredforSTACK_DYNPTRanyway.
*/
__mark_reg_known_zero(reg);
reg->type = CONST_PTR_TO_DYNPTR; /* Give each dynptr a unique id to uniquely associate slices to it. */
reg->id = dynptr_id;
reg->dynptr.type = type;
reg->dynptr.first_slot = first_slot;
}
staticvoid reg_bounds_sync(struct bpf_reg_state *reg)
{ /* We might have learned new bounds from the var_off. */
__update_reg_bounds(reg); /* We might have learned something about the sign bit. */
__reg_deduce_bounds(reg);
__reg_deduce_bounds(reg);
__reg_deduce_bounds(reg); /* We might have learned some bits from the bounds. */
__reg_bound_offset(reg); /* Intersecting with the old var_off might have improved our bounds *slightly,e.g.ifumaxwas0x7f...fandvar_offwas(0;0xf...fc), *thennewvar_offis(0;0x7f...fc)whichimprovesourumax.
*/
__update_reg_bounds(reg);
}
/* Similar to push_stack(), but for async callbacks */ staticstruct bpf_verifier_state *push_async_cb(struct bpf_verifier_env *env, int insn_idx, int prev_insn_idx, int subprog, bool is_sleepable)
{ struct bpf_verifier_stack_elem *elem; struct bpf_func_state *frame;
elem = kzalloc(sizeof(struct bpf_verifier_stack_elem), GFP_KERNEL_ACCOUNT); if (!elem) return NULL;
elem->insn_idx = insn_idx;
elem->prev_insn_idx = prev_insn_idx;
elem->next = env->head;
elem->log_pos = env->log.end_pos;
env->head = elem;
env->stack_size++; if (env->stack_size > BPF_COMPLEXITY_LIMIT_JMP_SEQ) {
verbose(env, "The sequence of %d jumps is too complex for async cb.\n",
env->stack_size); return NULL;
} /* Unlike push_stack() do not copy_verifier_state(). *Thecallerstatedoesn'tmatter. *Thisisasynccallback.Itstartsinafreshstack. *Initializeitsimilartodo_check_common().
*/
elem->st.branches = 1;
elem->st.in_sleepable = is_sleepable;
frame = kzalloc(sizeof(*frame), GFP_KERNEL_ACCOUNT); if (!frame) return NULL;
init_func_state(env, frame,
BPF_MAIN_FUNC /* callsite */, 0/* frameno within this callchain */,
subprog /* subprog number within this prog */);
elem->st.frame[0] = frame; return &elem->st;
}
enum reg_arg_type {
SRC_OP, /* register is used as source operand */
DST_OP, /* register is used as destination operand */
DST_OP_NO_MARK /* same as above, check only, don't mark */
};
/* Find subprogram that contains instruction at 'off' */ staticstruct bpf_subprog_info *find_containing_subprog(struct bpf_verifier_env *env, int off)
{ struct bpf_subprog_info *vals = env->subprog_info; int l, r, m;
if (off >= env->prog->len || off < 0 || env->subprog_cnt == 0) return NULL;
l = 0;
r = env->subprog_cnt - 1; while (l < r) {
m = l + (r - l + 1) / 2; if (vals[m].start <= off)
l = m; else
r = m - 1;
} return &vals[l];
}
/* Find subprogram that starts exactly at 'off' */ staticint find_subprog(struct bpf_verifier_env *env, int off)
{ struct bpf_subprog_info *p;
p = find_containing_subprog(env, off); if (!p || p->start != off) return -ENOENT; return p - env->subprog_info;
}
staticint add_subprog(struct bpf_verifier_env *env, int off)
{ int insn_cnt = env->prog->len; int ret;
if (off >= insn_cnt || off < 0) {
verbose(env, "call to invalid destination\n"); return -EINVAL;
}
ret = find_subprog(env, off); if (ret >= 0) return ret; if (env->subprog_cnt >= BPF_MAX_SUBPROGS) {
verbose(env, "too many subprograms\n"); return -E2BIG;
} /* determine subprog starts. The end is one before the next starts */
env->subprog_info[env->subprog_cnt++].start = off;
sort(env->subprog_info, env->subprog_cnt, sizeof(env->subprog_info[0]), cmp_subprogs, NULL); return env->subprog_cnt - 1;
}
t = btf_type_by_id(btf, main_btf_id); if (!t) {
verbose(env, "invalid btf id for main subprog in func_info\n"); return -EINVAL;
}
name = btf_find_decl_tag_value(btf, t, -1, "exception_callback:"); if (IS_ERR(name)) {
ret = PTR_ERR(name); /* If there is no tag present, there is no exception callback */ if (ret == -ENOENT)
ret = 0; elseif (ret == -EEXIST)
verbose(env, "multiple exception callback tags for main subprog\n"); return ret;
}
ret = btf_find_by_name_kind(btf, name, BTF_KIND_FUNC); if (ret < 0) {
verbose(env, "exception callback '%s' could not be found in BTF\n", name); return ret;
}
id = ret;
t = btf_type_by_id(btf, id); if (btf_func_linkage(t) != BTF_FUNC_GLOBAL) {
verbose(env, "exception callback '%s' must have global linkage\n", name); return -EINVAL;
}
ret = 0; for (i = 0; i < aux->func_info_cnt; i++) { if (aux->func_info[i].type_id != id) continue;
ret = aux->func_info[i].insn_off; /* Further func_info and subprog checks will also happen *later,soassumethisistherightinsn_offfornow.
*/ if (!ret) {
verbose(env, "invalid exception callback insn_off in func_info: 0\n");
ret = -EINVAL;
}
} if (!ret) {
verbose(env, "exception callback type id not found in func_info\n");
ret = -EINVAL;
} return ret;
}
/* sort() reorders entries by value, so b may no longer point *totherightentryafterthis
*/
sort(tab->descs, tab->nr_descs, sizeof(tab->descs[0]),
kfunc_btf_cmp_by_off, NULL);
} else {
btf = b->btf;
}
return btf;
}
void bpf_free_kfunc_btf_tab(struct bpf_kfunc_btf_tab *tab)
{ if (!tab) return;
while (tab->nr_descs--) {
module_put(tab->descs[tab->nr_descs].module);
btf_put(tab->descs[tab->nr_descs].btf);
}
kfree(tab);
}
staticstruct btf *find_kfunc_desc_btf(struct bpf_verifier_env *env, s16 offset)
{ if (offset) { if (offset < 0) { /* In the future, this can be allowed to increase limit *offdindexintofd_array,interpretedasu16.
*/
verbose(env, "negative offset disallowed for kernel module function call\n"); return ERR_PTR(-EINVAL);
}
prog_aux = env->prog->aux;
tab = prog_aux->kfunc_tab;
btf_tab = prog_aux->kfunc_btf_tab; if (!tab) { if (!btf_vmlinux) {
verbose(env, "calling kernel function is not supported without CONFIG_DEBUG_INFO_BTF\n"); return -ENOTSUPP;
}
if (!env->prog->jit_requested) {
verbose(env, "JIT is required for calling kernel function\n"); return -ENOTSUPP;
}
if (!bpf_jit_supports_kfunc_call()) {
verbose(env, "JIT does not support calling kernel function\n"); return -ENOTSUPP;
}
if (!env->prog->gpl_compatible) {
verbose(env, "cannot call kernel function from non-GPL compatible program\n"); return -EINVAL;
}
/* func_id == 0 is always invalid, but instead of returning an error, be *conservativeandwaituntilthecodeeliminationpassbeforereturning *error,sothatinvalidcallsthatgetprunedoutcanbeinBPFprograms *loadedfromuserspace.Itisalsorequiredthatoffsetbeuntouched *forsuchcalls.
*/ if (!func_id && !offset) return0;
if (!btf_tab && offset) {
btf_tab = kzalloc(sizeof(*btf_tab), GFP_KERNEL_ACCOUNT); if (!btf_tab) return -ENOMEM;
prog_aux->kfunc_btf_tab = btf_tab;
}
desc_btf = find_kfunc_desc_btf(env, offset); if (IS_ERR(desc_btf)) {
verbose(env, "failed to find BTF for kernel function\n"); return PTR_ERR(desc_btf);
}
if (find_kfunc_desc(env->prog, func_id, offset)) return0;
if (tab->nr_descs == MAX_KFUNC_DESCS) {
verbose(env, "too many different kernel function calls\n"); return -E2BIG;
}
func = btf_type_by_id(desc_btf, func_id); if (!func || !btf_type_is_func(func)) {
verbose(env, "kernel btf_id %u is not a function\n",
func_id); return -EINVAL;
}
func_proto = btf_type_by_id(desc_btf, func->type); if (!func_proto || !btf_type_is_func_proto(func_proto)) {
verbose(env, "kernel function btf_id %u does not have a valid func_proto\n",
func_id); return -EINVAL;
}
func_name = btf_name_by_offset(desc_btf, func->name_off);
addr = kallsyms_lookup_name(func_name); if (!addr) {
verbose(env, "cannot find address for kernel function %s\n",
func_name); return -EINVAL;
}
specialize_kfunc(env, func_id, offset, &addr);
if (bpf_jit_supports_far_kfunc_call()) {
call_imm = func_id;
} else {
call_imm = BPF_CALL_IMM(addr); /* Check whether the relative offset overflows desc->imm */ if ((unsignedlong)(s32)call_imm != call_imm) {
verbose(env, "address of kernel function %s is out of range\n",
func_name); return -EINVAL;
}
}
if (bpf_dev_bound_kfunc_id(func_id)) {
err = bpf_dev_bound_kfunc_check(&env->log, prog_aux); if (err) return err;
}
tab = prog->aux->kfunc_tab;
res = bsearch(&desc, tab->descs, tab->nr_descs, sizeof(tab->descs[0]), kfunc_desc_cmp_by_imm_off);
return res ? &res->func_model : NULL;
}
staticint add_kfunc_in_insns(struct bpf_verifier_env *env, struct bpf_insn *insn, int cnt)
{ int i, ret;
for (i = 0; i < cnt; i++, insn++) { if (bpf_pseudo_kfunc_call(insn)) {
ret = add_kfunc_call(env, insn->imm, insn->off); if (ret < 0) return ret;
}
} return0;
}
/* Add entry function. */
ret = add_subprog(env, 0); if (ret) return ret;
for (i = 0; i < insn_cnt; i++, insn++) { if (!bpf_pseudo_func(insn) && !bpf_pseudo_call(insn) &&
!bpf_pseudo_kfunc_call(insn)) continue;
if (!env->bpf_capable) {
verbose(env, "loading/calling other bpf or kernel functions are allowed for CAP_BPF and CAP_SYS_ADMIN\n"); return -EPERM;
}
if (bpf_pseudo_func(insn) || bpf_pseudo_call(insn))
ret = add_subprog(env, i + insn->imm + 1); else
ret = add_kfunc_call(env, insn->imm, insn->off);
if (ret < 0) return ret;
}
ret = bpf_find_exception_callback_insn_off(env); if (ret < 0) return ret;
ex_cb_insn = ret;
/* If ex_cb_insn > 0, this means that the main program has a subprog *markedusingBTFdecltagtoserveastheexceptioncallback.
*/ if (ex_cb_insn) {
ret = add_subprog(env, ex_cb_insn); if (ret < 0) return ret; for (i = 1; i < env->subprog_cnt; i++) { if (env->subprog_info[i].start != ex_cb_insn) continue;
env->exception_callback_subprog = i;
mark_subprog_exc_cb(env, i); break;
}
}
/* Add a fake 'exit' subprog which could simplify subprog iteration *logic.'subprog_cnt'shouldnotbeincreased.
*/
subprog[env->subprog_cnt].start = insn_cnt;
if (env->log.level & BPF_LOG_LEVEL2) for (i = 0; i < env->subprog_cnt; i++)
verbose(env, "func#%d @%d\n", i, subprog[i].start);
staticint check_subprogs(struct bpf_verifier_env *env)
{ int i, subprog_start, subprog_end, off, cur_subprog = 0; struct bpf_subprog_info *subprog = env->subprog_info; struct bpf_insn *insn = env->prog->insnsi; int insn_cnt = env->prog->len;
/* now check that all jumps are within the same subprog */
subprog_start = subprog[cur_subprog].start;
subprog_end = subprog[cur_subprog + 1].start; for (i = 0; i < insn_cnt; i++) {
u8 code = insn[i].code;
if (code == (BPF_JMP | BPF_CALL) &&
insn[i].src_reg == 0 &&
insn[i].imm == BPF_FUNC_tail_call) {
subprog[cur_subprog].has_tail_call = true;
subprog[cur_subprog].tail_call_reachable = true;
} if (BPF_CLASS(code) == BPF_LD &&
(BPF_MODE(code) == BPF_ABS || BPF_MODE(code) == BPF_IND))
subprog[cur_subprog].has_ld_abs = true; if (BPF_CLASS(code) != BPF_JMP && BPF_CLASS(code) != BPF_JMP32) goto next; if (BPF_OP(code) == BPF_EXIT || BPF_OP(code) == BPF_CALL) goto next;
off = i + jmp_offset(&insn[i]) + 1; if (off < subprog_start || off >= subprog_end) {
verbose(env, "jump out of range from insn %d to %d\n", i, off); return -EINVAL;
}
next: if (i == subprog_end - 1) { /* to avoid fall-through from one subprog into another *thelastinsnofthesubprogshouldbeeitherexit *orunconditionaljumpbackorbpf_throwcall
*/ if (code != (BPF_JMP | BPF_EXIT) &&
code != (BPF_JMP32 | BPF_JA) &&
code != (BPF_JMP | BPF_JA)) {
verbose(env, "last insn is not an exit or jmp\n"); return -EINVAL;
}
subprog_start = subprog_end;
cur_subprog++; if (cur_subprog < env->subprog_cnt)
subprog_end = subprog[cur_subprog + 1].start;
}
} return0;
}
/* Parentage chain of this register (or stack slot) should take care of all *issueslikecallee-savedregisters,stackslotallocationtime,etc.
*/ staticint mark_reg_read(struct bpf_verifier_env *env, conststruct bpf_reg_state *state, struct bpf_reg_state *parent, u8 flag)
{ bool writes = parent == state->parent; /* Observe write marks */ int cnt = 0;
while (parent) { /* if read wasn't screened by an earlier write ... */ if (writes && state->live & REG_LIVE_WRITTEN) break; if (verifier_bug_if(parent->live & REG_LIVE_DONE, env, "type %s var_off %lld off %d",
reg_type_str(env, parent->type),
parent->var_off.value, parent->off)) return -EFAULT; /* The first condition is more likely to be true than the *second,checkeditfirst.
*/ if ((parent->live & REG_LIVE_READ) == flag ||
parent->live & REG_LIVE_READ64) /* The parentage chain never changes and *thisparentwasalreadymarkedasLIVE_READ. *Thereisnoneedtokeepwalkingthechainagainand *keepre-markingallparentsasLIVE_READ. *Thiscasehappenswhenthesameregisterisread *multipletimeswithoutwritesintoitin-between. *Also,ifparenthasthestrongerREG_LIVE_READ64set, *thennoneedtosettheweakREG_LIVE_READ32.
*/ break; /* ... then we depend on parent's value */
parent->live |= flag; /* REG_LIVE_READ64 overrides REG_LIVE_READ32. */ if (flag == REG_LIVE_READ64)
parent->live &= ~REG_LIVE_READ32;
state = parent;
parent = state->parent;
writes = true;
cnt++;
}
if (env->longest_mark_read_walk < cnt)
env->longest_mark_read_walk = cnt; return0;
}
staticint mark_stack_slot_obj_read(struct bpf_verifier_env *env, struct bpf_reg_state *reg, int spi, int nr_slots)
{ struct bpf_func_state *state = func(env, reg); int err, i;
for (i = 0; i < nr_slots; i++) { struct bpf_reg_state *st = &state->stack[spi - i].spilled_ptr;
err = mark_reg_read(env, st, st->parent, REG_LIVE_READ64); if (err) return err;
staticint mark_dynptr_read(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
{ int spi;
/* For CONST_PTR_TO_DYNPTR, it must have already been done by *check_reg_argincheck_helper_callandmark_btf_func_reg_sizein *check_kfunc_call.
*/ if (reg->type == CONST_PTR_TO_DYNPTR) return0;
spi = dynptr_get_spi(env, reg); if (spi < 0) return spi; /* Caller ensures dynptr is valid and initialized, which means spi is in *boundsandspiisthefirstdynptrslot.Simplymarkstackslotas *read.
*/ return mark_stack_slot_obj_read(env, reg, spi, BPF_DYNPTR_NR_SLOTS);
}
staticint mark_iter_read(struct bpf_verifier_env *env, struct bpf_reg_state *reg, int spi, int nr_slots)
{ return mark_stack_slot_obj_read(env, reg, spi, nr_slots);
}
staticint mark_irq_flag_read(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
{ int spi;
/* This function is supposed to be used by the following 32-bit optimization *codeonly.ItreturnsTRUEifthesourceordestinationregisteroperates *on64-bit,otherwisereturnFALSE.
*/ staticbool is_reg64(struct bpf_verifier_env *env, struct bpf_insn *insn,
u32 regno, struct bpf_reg_state *reg, enum reg_arg_type t)
{
u8 code, class, op;
code = insn->code; class = BPF_CLASS(code);
op = BPF_OP(code); if (class == BPF_JMP) { /* BPF_EXIT for "main" will reach here. Return TRUE *conservatively.
*/ if (op == BPF_EXIT) returntrue; if (op == BPF_CALL) { /* BPF to BPF call will reach here because of marking *callersavedclobberwithDST_OP_NO_MARKforwhichwe *don'tcaretheregisterdefbecausetheyareanyway *markedasNOT_INITalready.
*/ if (insn->src_reg == BPF_PSEUDO_CALL) returnfalse; /* Helper call will reach here because of arg type *check,conservativelyreturnTRUE.
*/ if (t == SRC_OP) returntrue;
returnfalse;
}
}
if (class == BPF_ALU64 && op == BPF_END && (insn->imm == 16 || insn->imm == 32)) returnfalse;
if (class == BPF_ALU64 || class == BPF_JMP ||
(class == BPF_ALU && op == BPF_END && insn->imm == 64)) returntrue;
if (class == BPF_ALU || class == BPF_JMP32) returnfalse;
if (class == BPF_LDX) { if (t != SRC_OP) return BPF_SIZE(code) == BPF_DW || BPF_MODE(code) == BPF_MEMSX; /* LDX source must be ptr. */ returntrue;
}
if (class == BPF_STX) { /* BPF_STX (including atomic variants) has one or more source *operands,oneofwhichisaptr.Checkwhetherthecalleris *askingaboutit.
*/ if (t == SRC_OP && reg->type != SCALAR_VALUE) returntrue; return BPF_SIZE(code) == BPF_DW;
}
if (class == BPF_LD) {
u8 mode = BPF_MODE(code);
/* LD_IMM64 */ if (mode == BPF_IMM) returntrue;
/* Both LD_IND and LD_ABS return 32-bit data. */ if (t != SRC_OP) returnfalse;
/* Implicit ctx ptr. */ if (regno == BPF_REG_6) returntrue;
/* Explicit source could be any width. */ returntrue;
}
if (class == BPF_ST) /* The only source register for BPF_ST is a ptr. */ returntrue;
/* Conservatively return true at default. */ returntrue;
}
/* Return the regno defined by the insn, or -1. */ staticint insn_def_regno(conststruct bpf_insn *insn)
{ switch (BPF_CLASS(insn->code)) { case BPF_JMP: case BPF_JMP32: case BPF_ST: return -1; case BPF_STX: if (BPF_MODE(insn->code) == BPF_ATOMIC ||
BPF_MODE(insn->code) == BPF_PROBE_ATOMIC) { if (insn->imm == BPF_CMPXCHG) return BPF_REG_0; elseif (insn->imm == BPF_LOAD_ACQ) return insn->dst_reg; elseif (insn->imm & BPF_FETCH) return insn->src_reg;
} return -1; default: return insn->dst_reg;
}
}
/* Return TRUE if INSN has defined any 32-bit value explicitly. */ staticbool insn_has_def32(struct bpf_verifier_env *env, struct bpf_insn *insn)
{ int dst_reg = insn_def_regno(insn);
env->insn_aux_data[def_idx - 1].zext_dst = true; /* The dst will be zero extended, so won't be sub-register anymore. */
reg->subreg_def = DEF_NOT_SUBREG;
}
if (regno >= MAX_BPF_REG) {
verbose(env, "R%d is invalid\n", regno); return -EINVAL;
}
mark_reg_scratched(env, regno);
reg = ®s[regno];
rw64 = is_reg64(env, insn, regno, reg, t); if (t == SRC_OP) { /* check whether register used as source operand can be read */ if (reg->type == NOT_INIT) {
verbose(env, "R%d !read_ok\n", regno); return -EACCES;
} /* We don't need to worry about FP liveness because it's read-only */ if (regno == BPF_REG_FP) return0;
if (rw64)
mark_insn_zext(env, reg);
return mark_reg_read(env, reg, reg->parent,
rw64 ? REG_LIVE_READ64 : REG_LIVE_READ32);
} else { /* check whether register used as dest operand can be written to */ if (regno == BPF_REG_FP) {
verbose(env, "frame pointer is read only\n"); return -EACCES;
}
reg->live |= REG_LIVE_WRITTEN;
reg->subreg_def = rw64 ? DEF_NOT_SUBREG : env->insn_idx + 1; if (t == DST_OP)
mark_reg_unknown(env, regs, regno);
} return0;
}
/* Use u64 as a vector of 6 10-bit values, use first 4-bits to track *numberofelementscurrentlyinstack. *Packonehistoryentryforlinkedregistersas10bitsinthefollowingformat: *-3-bitsframeno *-6-bitsspi_or_reg *-1-bitis_reg
*/ static u64 linked_regs_pack(struct linked_regs *s)
{
u64 val = 0; int i;
for (i = 0; i < s->cnt; ++i) { struct linked_reg *e = &s->entries[i];
u64 tmp = 0;
/* for any branch, call, exit record the history of jmps in the given state */ staticint push_jmp_history(struct bpf_verifier_env *env, struct bpf_verifier_state *cur, int insn_flags, u64 linked_regs)
{
u32 cnt = cur->jmp_history_cnt; struct bpf_jmp_history_entry *p;
size_t alloc_size;
/* combine instruction flags if we already recorded this instruction */ if (env->cur_hist_ent) { /* atomic instructions push insn_flags twice, for READ and *WRITEsides,buttheyshouldagreeonstackslot
*/
verifier_bug_if((env->cur_hist_ent->flags & insn_flags) &&
(env->cur_hist_ent->flags & insn_flags) != insn_flags,
env, "insn history: insn_idx %d cur flags %x new flags %x",
env->insn_idx, env->cur_hist_ent->flags, insn_flags);
env->cur_hist_ent->flags |= insn_flags;
verifier_bug_if(env->cur_hist_ent->linked_regs != 0, env, "insn history: insn_idx %d linked_regs: %#llx",
env->insn_idx, env->cur_hist_ent->linked_regs);
env->cur_hist_ent->linked_regs = linked_regs; return0;
}
cnt++;
alloc_size = kmalloc_size_roundup(size_mul(cnt, sizeof(*p)));
p = krealloc(cur->jmp_history, alloc_size, GFP_KERNEL_ACCOUNT); if (!p) return -ENOMEM;
cur->jmp_history = p;
/* Backtrack one insn at a time. If idx is not at the top of recorded *historythenpreviousinstructioncamefromstraightlineexecution. *Return-ENOENTifweexhaustedallinstructionswithingivenstate. * *It'slegaltohaveabitofaloopingwiththesamestartingandending *insnindexwithinthesamestate,e.g.:3->4->5->3,sojustbecausecurrent *instructionindexisthesameasstate'sfirst_idxdoesn'tmeanweare *done.Ifthereisstillsomejumphistoryleft,weshouldkeepgoing.We *needtotakeintoaccountthatwemighthaveajumphistorybetweengiven *state'sparentanditself,duetocheckpointing.Inthiscase,we'llhave *historyentryrecordingajumpfromlastinstructionofparentstateand *firstinstructionofgivenstate.
*/ staticint get_prev_insn_idx(struct bpf_verifier_state *st, int i,
u32 *history)
{
u32 cnt = *history;
if (i == st->first_insn_idx) { if (cnt == 0) return -ENOENT; if (cnt == 1 && st->jmp_history[0].idx == i) return -ENOENT;
}
/* format registers bitmask, e.g., "r0,r2,r4" for 0x15 mask */ staticvoid fmt_reg_mask(char *buf, ssize_t buf_sz, u32 reg_mask)
{
DECLARE_BITMAP(mask, 64); bool first = true; int i, n;
buf[0] = '\0';
bitmap_from_u64(mask, reg_mask);
for_each_set_bit(i, mask, 32) {
n = snprintf(buf, buf_sz, "%sr%d", first ? "" : ",", i);
first = false;
buf += n;
buf_sz -= n; if (buf_sz < 0) break;
}
} /* format stack slots bitmask, e.g., "-8,-24,-40" for 0x15 mask */ staticvoid fmt_stack_mask(char *buf, ssize_t buf_sz, u64 stack_mask)
{
DECLARE_BITMAP(mask, 64); bool first = true; int i, n;
buf[0] = '\0';
bitmap_from_u64(mask, stack_mask);
for_each_set_bit(i, mask, 64) {
n = snprintf(buf, buf_sz, "%s%d", first ? "" : ",", -(i + 1) * 8);
first = false;
buf += n;
buf_sz -= n; if (buf_sz < 0) break;
}
}
/* If any register R in hist->linked_regs is marked as precise in bt, *dobt_set_frame_{reg,slot}(bt,R)forallregistersinhist->linked_regs.
*/ staticvoid bt_sync_linked_regs(struct backtrack_state *bt, struct bpf_jmp_history_entry *hist)
{ struct linked_regs linked_regs; bool some_precise = false; int i;
if (!hist || hist->linked_regs == 0) return;
linked_regs_unpack(hist->linked_regs, &linked_regs); for (i = 0; i < linked_regs.cnt; ++i) { struct linked_reg *e = &linked_regs.entries[i];
/* If there is a history record that some registers gained range at this insn, *propagateprecisionmarkstothoseregisters,sothatbt_is_reg_set() *accountsfortheseregisters.
*/
bt_sync_linked_regs(bt, hist);
if (class == BPF_ALU || class == BPF_ALU64) { if (!bt_is_reg_set(bt, dreg)) return0; if (opcode == BPF_END || opcode == BPF_NEG) { /* sreg is reserved and unused *dregstillneedprecisionbeforethisinsn
*/ return0;
} elseif (opcode == BPF_MOV) { if (BPF_SRC(insn->code) == BPF_X) { /* dreg = sreg or dreg = (s8, s16, s32)sreg *dregneedsprecisionafterthisinsn *sregneedsprecisionbeforethisinsn
*/
bt_clear_reg(bt, dreg); if (sreg != BPF_REG_FP)
bt_set_reg(bt, sreg);
} else { /* dreg = K *dregneedsprecisionafterthisinsn. *Correspondingregisterisalreadymarked *asprecise=trueinthisverifierstate. *Nofurthermarkingsinparentarenecessary
*/
bt_clear_reg(bt, dreg);
}
} else { if (BPF_SRC(insn->code) == BPF_X) { /* dreg += sreg *bothdregandsregneedprecision *beforethisinsn
*/ if (sreg != BPF_REG_FP)
bt_set_reg(bt, sreg);
} /* else dreg += K *dregstillneedsprecisionbeforethisinsn
*/
}
} elseif (class == BPF_LDX || is_atomic_load_insn(insn)) { if (!bt_is_reg_set(bt, dreg)) return0;
bt_clear_reg(bt, dreg);
/* scalars can only be spilled into stack w/o losing precision. *Loadfromanyothermemorycanbezeroextended. *Thedesiretokeepthatprecisionisalreadyindicated *by'precise'markincorrespondingregisterofthisstate. *Nofurthertrackingnecessary.
*/ if (!hist || !(hist->flags & INSN_F_STACK_ACCESS)) return0; /* dreg = *(u64 *)[fp - off] was a fill from the stack. *that[fp-off]slotcontainsscalarthatneedstobe *trackedwithprecision
*/
spi = insn_stack_access_spi(hist->flags);
fr = insn_stack_access_frameno(hist->flags);
bt_set_frame_slot(bt, fr, spi);
} elseif (class == BPF_STX || class == BPF_ST) { if (bt_is_reg_set(bt, dreg)) /* stx & st shouldn't be using _scalar_ dst_reg *toaccessmemory.Itmeansbacktracking *encounteredacaseofpointersubtraction.
*/ return -ENOTSUPP; /* scalars can only be spilled into stack */ if (!hist || !(hist->flags & INSN_F_STACK_ACCESS)) return0;
spi = insn_stack_access_spi(hist->flags);
fr = insn_stack_access_frameno(hist->flags); if (!bt_is_frame_slot_set(bt, fr, spi)) return0;
bt_clear_frame_slot(bt, fr, spi); if (class == BPF_STX)
bt_set_reg(bt, sreg);
} elseif (class == BPF_JMP || class == BPF_JMP32) { if (bpf_pseudo_call(insn)) { int subprog_insn_idx, subprog;
if (subprog_is_global(env, subprog)) { /* check that jump history doesn't have any *extrainstructionsfromsubprog;thenext *instructionaftercalltoglobalsubprog *shouldbeliterallynextinstructionin *callerprogram
*/
verifier_bug_if(idx + 1 != subseq_idx, env, "extra insn from subprog"); /* r1-r5 are invalidated after subprog call, *soforglobalfunccallitshouldn'tbeset *anymore
*/ if (bt_reg_mask(bt) & BPF_REGMASK_ARGS) {
verifier_bug(env, "global subprog unexpected regs %x",
bt_reg_mask(bt)); return -EFAULT;
} /* global subprog always sets R0 */
bt_clear_reg(bt, BPF_REG_0); return0;
} else { /* static subprog call instruction, which *meansthatweareexitingcurrentsubprog, *soonlyr1-r5couldbestillrequestedas *precise,r0andr6-r10oranystackslotin *thecurrentframeshouldbezerobynow
*/ if (bt_reg_mask(bt) & ~BPF_REGMASK_ARGS) {
verifier_bug(env, "static subprog unexpected regs %x",
bt_reg_mask(bt)); return -EFAULT;
} /* we are now tracking register spills correctly, *soanyinstanceofleftoverslotsisabug
*/ if (bt_stack_mask(bt) != 0) {
verifier_bug(env, "static subprog leftover stack slots %llx",
bt_stack_mask(bt)); return -EFAULT;
} /* propagate r1-r5 to the caller */ for (i = BPF_REG_1; i <= BPF_REG_5; i++) { if (bt_is_reg_set(bt, i)) {
bt_clear_reg(bt, i);
bt_set_frame_reg(bt, bt->frame - 1, i);
}
} if (bt_subprog_exit(bt)) return -EFAULT; return0;
}
} elseif (is_sync_callback_calling_insn(insn) && idx != subseq_idx - 1) { /* exit from callback subprog to callback-calling helper or *kfunccall.Useidx/subseq_idxchecktodiscernitfrom *straightlinecodebacktracking. *Unlikethesubprogcallhandlingabove,weshouldn't *propagateprecisionofr1-r5(ifanyrequested),astheyare *notactuallyargumentspasseddirectlytocallbacksubprogs
*/ if (bt_reg_mask(bt) & ~BPF_REGMASK_ARGS) {
verifier_bug(env, "callback unexpected regs %x",
bt_reg_mask(bt)); return -EFAULT;
} if (bt_stack_mask(bt) != 0) {
verifier_bug(env, "callback leftover stack slots %llx",
bt_stack_mask(bt)); return -EFAULT;
} /* clear r1-r5 in callback subprog's mask */ for (i = BPF_REG_1; i <= BPF_REG_5; i++)
bt_clear_reg(bt, i); if (bt_subprog_exit(bt)) return -EFAULT; return0;
} elseif (opcode == BPF_CALL) { /* kfunc with imm==0 is invalid and fixup_kfunc_call will *catchthiserrorlater.Makebacktrackingconservative *withENOTSUPP.
*/ if (insn->src_reg == BPF_PSEUDO_KFUNC_CALL && insn->imm == 0) return -ENOTSUPP; /* regular helper call sets R0 */
bt_clear_reg(bt, BPF_REG_0); if (bt_reg_mask(bt) & BPF_REGMASK_ARGS) { /* if backtracking was looking for registers R1-R5 *theyshouldhavebeenfoundalready.
*/
verifier_bug(env, "backtracking call unexpected regs %x",
bt_reg_mask(bt)); return -EFAULT;
}
} elseif (opcode == BPF_EXIT) { bool r0_precise;
/* Backtracking to a nested function call, 'idx' is a part of *theinnerframe'subseq_idx'isapartoftheouterframe. *Incaseofaregularfunctioncall,instructionsgiving *precisiontoregistersR1-R5shouldhavebeenfoundalready. *Incaseofacallback,itisoktohaveR1-R5markedfor *backtracking,astheseregistersaresetbythefunction *invokingcallback.
*/ if (subseq_idx >= 0 && calls_callback(env, subseq_idx)) for (i = BPF_REG_1; i <= BPF_REG_5; i++)
bt_clear_reg(bt, i); if (bt_reg_mask(bt) & BPF_REGMASK_ARGS) {
verifier_bug(env, "backtracking exit unexpected regs %x",
bt_reg_mask(bt)); return -EFAULT;
}
/* if we still have requested precise regs or slots, we missed *something(e.g.,stackaccessthroughnon-r10register),so *fallbacktomarkingallprecise
*/ if (!bt_empty(bt)) {
mark_all_scalars_precise(env, starting_state);
bt_reset(bt);
}
return0;
}
int mark_chain_precision(struct bpf_verifier_env *env, int regno)
{ return __mark_chain_precision(env, env->cur_state, regno, NULL);
}
/* mark_chain_precision_batch() assumes that env->bt is set in the caller to *desiredregandstackmasksacrossallrelevantframes
*/ staticint mark_chain_precision_batch(struct bpf_verifier_env *env, struct bpf_verifier_state *starting_state)
{ return __mark_chain_precision(env, starting_state, -1, NULL);
}
staticbool is_spillable_regtype(enum bpf_reg_type type)
{ switch (base_type(type)) { case PTR_TO_MAP_VALUE: case PTR_TO_STACK: case PTR_TO_CTX: case PTR_TO_PACKET: case PTR_TO_PACKET_META: case PTR_TO_PACKET_END: case PTR_TO_FLOW_KEYS: case CONST_PTR_TO_MAP: case PTR_TO_SOCKET: case PTR_TO_SOCK_COMMON: case PTR_TO_TCP_SOCK: case PTR_TO_XDP_SOCK: case PTR_TO_BTF_ID: case PTR_TO_BUF: case PTR_TO_MEM: case PTR_TO_FUNC: case PTR_TO_MAP_KEY: case PTR_TO_ARENA: returntrue; default: returnfalse;
}
}
/* Does this register contain a constant zero? */ staticbool register_is_null(struct bpf_reg_state *reg)
{ return reg->type == SCALAR_VALUE && tnum_equals_const(reg->var_off, 0);
}
/* check if register is a constant scalar value */ staticbool is_reg_const(struct bpf_reg_state *reg, bool subreg32)
{ return reg->type == SCALAR_VALUE &&
tnum_is_const(subreg32 ? tnum_subreg(reg->var_off) : reg->var_off);
}
/* assuming is_reg_const() is true, return constant value of a register */ static u64 reg_const_value(struct bpf_reg_state *reg, bool subreg32)
{ return subreg32 ? tnum_subreg(reg->var_off).value : reg->var_off.value;
}
staticbool __is_pointer_value(bool allow_ptr_leaks, conststruct bpf_reg_state *reg)
{ if (allow_ptr_leaks) returnfalse;
if (!src_reg->id && !tnum_is_const(src_reg->var_off)) /* Ensure that src_reg has a valid ID that will be copied to *dst_regandthenwillbeusedbysync_linked_regs()to *propagatemin/maxrange.
*/
src_reg->id = ++env->id_gen;
}
/* Copy src state preserving dst->parent and dst->live fields */ staticvoid copy_register_state(struct bpf_reg_state *dst, conststruct bpf_reg_state *src)
{ struct bpf_reg_state *parent = dst->parent; enum bpf_reg_liveness live = dst->live;
/* See comment for mark_fastcall_pattern_for_call() */ staticvoid check_fastcall_stack_contract(struct bpf_verifier_env *env, struct bpf_func_state *state, int insn_idx, int off)
{ struct bpf_subprog_info *subprog = &env->subprog_info[state->subprogno]; struct bpf_insn_aux_data *aux = env->insn_aux_data; int i;
if (subprog->fastcall_stack_off <= off || aux[insn_idx].fastcall_pattern) return; /* access to the region [max_stack_depth .. fastcall_stack_off) *fromsomethingthatisnotapartofthefastcallpattern, *disablefastcallrewritesforcurrentsubprogrambysetting *fastcall_stack_offtoavaluesmallerthananypossibleoffset.
*/
subprog->fastcall_stack_off = S16_MIN; /* reset fastcall aux flags within subprogram, *happensatmostoncepersubprogram
*/ for (i = subprog->start; i < (subprog + 1)->start; ++i) {
aux[i].fastcall_spills_num = 0;
aux[i].fastcall_pattern = 0;
}
}
/* check_stack_{read,write}_fixed_off functions track spill/fill of registers, *stackboundaryandalignmentarecheckedincheck_mem_access()
*/ staticint check_stack_write_fixed_off(struct bpf_verifier_env *env, /* stack frame we're writing to */ struct bpf_func_state *state, int off, int size, int value_regno, int insn_idx)
{ struct bpf_func_state *cur; /* state of the current function */ int i, slot = -off - 1, spi = slot / BPF_REG_SIZE, err; struct bpf_insn *insn = &env->prog->insnsi[insn_idx]; struct bpf_reg_state *reg = NULL; int insn_flags = insn_stack_access_flags(state->frameno, spi);
/* caller checked that off % size == 0 and -MAX_BPF_STACK <= off < 0, *soit'salignedaccessand[off,off+size)arewithinstacklimits
*/ if (!env->allow_ptr_leaks &&
is_spilled_reg(&state->stack[spi]) &&
!is_spilled_scalar_reg(&state->stack[spi]) &&
size != BPF_REG_SIZE) {
verbose(env, "attempt to corrupt spilled pointer on stack\n"); return -EACCES;
}
cur = env->cur_state->frame[env->cur_state->curframe]; if (value_regno >= 0)
reg = &cur->regs[value_regno]; if (!env->bypass_spec_v4) { bool sanitize = reg && is_spillable_regtype(reg->type);
for (i = 0; i < size; i++) {
u8 type = state->stack[spi].slot_type[i];
if (type != STACK_MISC && type != STACK_ZERO) {
sanitize = true; break;
}
}
if (sanitize)
env->insn_aux_data[insn_idx].nospec_result = true;
}
err = destroy_if_dynptr_stack_slot(env, state, spi); if (err) return err;
reg_value_fits = get_reg_width(reg) <= BITS_PER_BYTE * size; /* Make sure that reg had an ID to build a relation on spill. */ if (reg_value_fits)
assign_scalar_id_before_mov(env, reg);
save_register_state(env, state, spi, reg, size); /* Break the relation on a narrowing spill. */ if (!reg_value_fits)
state->stack[spi].spilled_ptr.id = 0;
} elseif (!reg && !(off % BPF_REG_SIZE) && is_bpf_st_mem(insn) &&
env->bpf_capable) { struct bpf_reg_state *tmp_reg = &env->fake_reg[0];
memset(tmp_reg, 0, sizeof(*tmp_reg));
__mark_reg_known(tmp_reg, insn->imm);
tmp_reg->type = SCALAR_VALUE;
save_register_state(env, state, spi, tmp_reg, size);
} elseif (reg && is_spillable_regtype(reg->type)) { /* register containing pointer is being spilled into stack */ if (size != BPF_REG_SIZE) {
verbose_linfo(env, insn_idx, "; ");
verbose(env, "invalid size of register spill\n"); return -EACCES;
} if (state != cur && reg->type == PTR_TO_STACK) {
verbose(env, "cannot spill pointers to stack into stack frame of the caller\n"); return -EINVAL;
}
save_register_state(env, state, spi, reg, size);
} else {
u8 type = STACK_MISC;
/* regular write of data into stack destroys any spilled ptr */
state->stack[spi].spilled_ptr.type = NOT_INIT; /* Mark slots as STACK_MISC if they belonged to spilled ptr/dynptr/iter. */ if (is_stack_slot_special(&state->stack[spi])) for (i = 0; i < BPF_REG_SIZE; i++)
scrub_spilled_slot(&state->stack[spi].slot_type[i]);
/* only mark the slot as written if all 8 bytes were written *otherwisereadpropagationmayincorrectlystoptoosoon *whenstackslotsarepartiallywritten. *Thisheuristicmeansthatreadpropagationwillbe *conservative,sinceitwilladdreg_live_readmarks *tostackslotsallthewaytofirststatewhenprograms *writes+readslessthan8bytes
*/ if (size == BPF_REG_SIZE)
state->stack[spi].spilled_ptr.live |= REG_LIVE_WRITTEN;
/* when we zero initialize stack slots mark them as such */ if ((reg && register_is_null(reg)) ||
(!reg && is_bpf_st_mem(insn) && insn->imm == 0)) { /* STACK_ZERO case happened because register spill *wasn'tproperlyalignedatthestackslotboundary, *soit'snotaregisterspillanymore;force *originatingregistertobeprecisetomake *STACK_ZEROcorrectforsubsequentstates
*/
err = mark_chain_precision(env, value_regno); if (err) return err;
type = STACK_ZERO;
}
/* Mark slots affected by this stack write. */ for (i = 0; i < size; i++)
state->stack[spi].slot_type[(slot - i) % BPF_REG_SIZE] = type;
insn_flags = 0; /* not a register spill */
}
if (insn_flags) return push_jmp_history(env, env->cur_state, insn_flags, 0); return0;
}
/* Write the stack: 'stack[ptr_regno + off] = value_regno'. 'ptr_regno' is *knowntocontainavariableoffset. *Thisfunctioncheckswhetherthewriteispermittedandconservatively *trackstheeffectsofthewrite,consideringthateachstackslotinthe *dynamicrangeispotentiallywrittento. * *'off'includes'regno->off'. *'value_regno'canbe-1,meaningthatanunknownvalueisbeingwrittento *thestack. * *Spilledpointersinrangearenotmarkedaswrittenbecausewedon'tknow *what'sgoingtobeactuallywritten.Thismeansthatreadpropagationfor *futurereadscannotbeterminatedbythiswrite. * *Forprivilegedprograms,uninitializedstackslotsareconsidered *initializedbythiswrite(eventhoughwedon'tknowexactlywhatoffsets *aregoingtobewrittento).Theideaisthatwedon'twanttheverifierto *rejectfuturereadsthataccessslotswrittentothroughvariableoffsets.
*/ staticint check_stack_write_var_off(struct bpf_verifier_env *env, /* func where register points to */ struct bpf_func_state *state, int ptr_regno, int off, int size, int value_regno, int insn_idx)
{ struct bpf_func_state *cur; /* state of the current function */ int min_off, max_off; int i, err; struct bpf_reg_state *ptr_reg = NULL, *value_reg = NULL; struct bpf_insn *insn = &env->prog->insnsi[insn_idx]; bool writing_zero = false; /* set if the fact that we're writing a zero is used to let any *stackslotsremainSTACK_ZERO
*/ bool zero_used = false;
cur = env->cur_state->frame[env->cur_state->curframe];
ptr_reg = &cur->regs[ptr_regno];
min_off = ptr_reg->smin_value + off;
max_off = ptr_reg->smax_value + off + size; if (value_regno >= 0)
value_reg = &cur->regs[value_regno]; if ((value_reg && register_is_null(value_reg)) ||
(!value_reg && is_bpf_st_mem(insn) && insn->imm == 0))
writing_zero = true;
check_fastcall_stack_contract(env, state, insn_idx, min_off); /* Variable offset writes destroy any spilled pointers in range. */ for (i = min_off; i < max_off; i++) {
u8 new_type, *stype; int slot, spi;
if (!env->allow_ptr_leaks && *stype != STACK_MISC && *stype != STACK_ZERO) { /* Reject the write if range we may write to has not *beeninitializedbeforehand.Ifwedidn'treject *here,theptrstatuswouldbeerasedbelow(even *thoughnotallslotsareactuallyoverwritten), *possiblyopeningthedoortoleaks. * *WedohowevercatchSTACK_INVALIDcasebelow,and *onlyallowreadingpossiblyuninitializedmemory *laterforCAP_PERFMON,asthewritemaynothappento *thatslot.
*/
verbose(env, "spilled ptr in range of var-offset stack write; insn %d, ptr off: %d",
insn_idx, i); return -EINVAL;
}
/* If writing_zero and the spi slot contains a spill of value 0, *maintainthespilltype.
*/ if (writing_zero && *stype == STACK_SPILL &&
is_spilled_scalar_reg(&state->stack[spi])) { struct bpf_reg_state *spill_reg = &state->stack[spi].spilled_ptr;
/* Erase all other spilled pointers. */
state->stack[spi].spilled_ptr.type = NOT_INIT;
/* Update the slot type. */
new_type = STACK_MISC; if (writing_zero && *stype == STACK_ZERO) {
new_type = STACK_ZERO;
zero_used = true;
} /* If the slot is STACK_INVALID, we check whether it's OK to *pretendthatitwillbeinitializedbythiswrite.Theslot *mightnotactuallybewrittento,andsoifwemarkitas *initializedfuturereadsmightleakuninitializedmemory. *Forprivilegedprograms,wewillacceptsuchreadstoslots *thatmayormaynotbewrittenbecause,ifwe'rereject *them,theerrorwouldbetooconfusing.
*/ if (*stype == STACK_INVALID && !env->allow_uninit_stack) {
verbose(env, "uninit stack in range of var-offset write prohibited for !root; insn %d, off: %d",
insn_idx, i); return -EINVAL;
}
*stype = new_type;
} if (zero_used) { /* backtracking doesn't work for STACK_ZERO yet. */
err = mark_chain_precision(env, value_regno); if (err) return err;
} return0;
}
/* When register 'dst_regno' is assigned some values from stack[min_off, *max_off),wesettheregister'stypeaccordingtothetypesofthe *respectivestackslots.Ifallthestackvaluesareknowntobezeros,then *soisthedestinationreg.Otherwise,theregisterisconsideredtobe *SCALAR.Thisfunctiondoesnotdealwithregisterfilling;thecallermust *ensurethatallspilledregistersinthestackrangehavebeenmarkedas *read.
*/ staticvoid mark_reg_stack_read(struct bpf_verifier_env *env, /* func where src register points to */ struct bpf_func_state *ptr_state, int min_off, int max_off, int dst_regno)
{ struct bpf_verifier_state *vstate = env->cur_state; struct bpf_func_state *state = vstate->frame[vstate->curframe]; int i, slot, spi;
u8 *stype; int zeros = 0;
for (i = min_off; i < max_off; i++) {
slot = -i - 1;
spi = slot / BPF_REG_SIZE;
mark_stack_slot_scratched(env, spi);
stype = ptr_state->stack[spi].slot_type; if (stype[slot % BPF_REG_SIZE] != STACK_ZERO) break;
zeros++;
} if (zeros == max_off - min_off) { /* Any access_size read into register is zero extended, *sothewholeregister==const_zero.
*/
__mark_reg_const_zero(env, &state->regs[dst_regno]);
} else { /* have read misc data from the stack */
mark_reg_unknown(env, state->regs, dst_regno);
}
state->regs[dst_regno].live |= REG_LIVE_WRITTEN;
}
/* Read the stack at 'off' and put the results into the register indicated by *'dst_regno'.Ithandlesregfillingiftheaddressedstackslotisa *spilledreg. * *'dst_regno'canbe-1,meaningthatthereadvalueisnotgoingtoa *register. * *Theaccessisassumedtobewithinthecurrentstackbounds.
*/ staticint check_stack_read_fixed_off(struct bpf_verifier_env *env, /* func where src register points to */ struct bpf_func_state *reg_state, int off, int size, int dst_regno)
{ struct bpf_verifier_state *vstate = env->cur_state; struct bpf_func_state *state = vstate->frame[vstate->curframe]; int i, slot = -off - 1, spi = slot / BPF_REG_SIZE; struct bpf_reg_state *reg;
u8 *stype, type; int insn_flags = insn_stack_access_flags(reg_state->frameno, spi);
/* Break the relation on a narrowing fill. *coerce_reg_to_sizewilladjusttheboundaries.
*/ if (get_reg_width(reg) > size * BITS_PER_BYTE)
state->regs[dst_regno].id = 0;
} else { int spill_cnt = 0, zero_cnt = 0;
for (i = 0; i < size; i++) {
type = stype[(slot - i) % BPF_REG_SIZE]; if (type == STACK_SPILL) {
spill_cnt++; continue;
} if (type == STACK_MISC) continue; if (type == STACK_ZERO) {
zero_cnt++; continue;
} if (type == STACK_INVALID && env->allow_uninit_stack) continue;
verbose(env, "invalid read from stack off %d+%d size %d\n",
off, i, size); return -EACCES;
}
if (spill_cnt == size &&
tnum_is_const(reg->var_off) && reg->var_off.value == 0) {
__mark_reg_const_zero(env, &state->regs[dst_regno]); /* this IS register fill, so keep insn_flags */
} elseif (zero_cnt == size) { /* similarly to mark_reg_stack_read(), preserve zeroes */
__mark_reg_const_zero(env, &state->regs[dst_regno]);
insn_flags = 0; /* not restoring original register state */
} else {
mark_reg_unknown(env, state->regs, dst_regno);
insn_flags = 0; /* not restoring original register state */
}
}
state->regs[dst_regno].live |= REG_LIVE_WRITTEN;
} elseif (dst_regno >= 0) { /* restore register state from stack */
copy_register_state(&state->regs[dst_regno], reg); /* mark reg as written since spilled pointer state likely *hasitslivenessmarksclearedbyis_state_visited() *whichresetsstack/reglivenessforstatetransitions
*/
state->regs[dst_regno].live |= REG_LIVE_WRITTEN;
} elseif (__is_pointer_value(env->allow_ptr_leaks, reg)) { /* If dst_regno==-1, the caller is asking us whether *itisacceptabletousethisvalueasaSCALAR_VALUE *(e.g.forXADD). *Wemustnotallowunprivilegedcallerstodothat *withspilledpointers.
*/
verbose(env, "leaking pointer from stack off %d\n",
off); return -EACCES;
}
mark_reg_read(env, reg, reg->parent, REG_LIVE_READ64);
} else { for (i = 0; i < size; i++) {
type = stype[(slot - i) % BPF_REG_SIZE]; if (type == STACK_MISC) continue; if (type == STACK_ZERO) continue; if (type == STACK_INVALID && env->allow_uninit_stack) continue;
verbose(env, "invalid read from stack off %d+%d size %d\n",
off, i, size); return -EACCES;
}
mark_reg_read(env, reg, reg->parent, REG_LIVE_READ64); if (dst_regno >= 0)
mark_reg_stack_read(env, reg_state, off, off + size, dst_regno);
insn_flags = 0; /* we are not restoring spilled register */
} if (insn_flags) return push_jmp_history(env, env->cur_state, insn_flags, 0); return0;
}
enum bpf_access_src {
ACCESS_DIRECT = 1, /* the access is performed by an instruction */
ACCESS_HELPER = 2, /* the access is performed by a helper */
};
staticint check_stack_range_initialized(struct bpf_verifier_env *env, int regno, int off, int access_size, bool zero_size_allowed, enum bpf_access_type type, struct bpf_call_arg_meta *meta);
/* Read the stack at 'ptr_regno + off' and put the result into the register *'dst_regno'. *'off'includesthepointerregister'sfixedoffset(i.e.'ptr_regno.off'), *butnotitsvariableoffset. *'size'isassumedtobe<=regsizeandtheaccessisassumedtobealigned. * *Asopposedtocheck_stack_read_fixed_off,thisfunctiondoesn'tdealwith *fillingregisters(i.e.readsofspilledregistercannotbedetectedwhen *theoffsetisnotfixed).Weconservativelymark'dst_regno'ascontaining *SCALAR_VALUE.That'swhyweassertthatthe'ptr_regno'hasavariable *offset;forafixedoffsetcheck_stack_read_fixed_offshouldbeused *instead.
*/ staticint check_stack_read_var_off(struct bpf_verifier_env *env, int ptr_regno, int off, int size, int dst_regno)
{ /* The state of the source register. */ struct bpf_reg_state *reg = reg_state(env, ptr_regno); struct bpf_func_state *ptr_state = func(env, reg); int err; int min_off, max_off;
/* Note that we pass a NULL meta, so raw access will not be permitted.
*/
err = check_stack_range_initialized(env, ptr_regno, off, size, false, BPF_READ, NULL); if (err) return err;
/* check_stack_read dispatches to check_stack_read_fixed_off or *check_stack_read_var_off. * *Thecallermustensurethattheoffsetfallswithintheallocatedstack *bounds. * *'dst_regno'isaregisterwhichwillreceivethevaluefromthestack.It *canbe-1,meaningthatthereadvalueisnotgoingtoaregister.
*/ staticint check_stack_read(struct bpf_verifier_env *env, int ptr_regno, int off, int size, int dst_regno)
{ struct bpf_reg_state *reg = reg_state(env, ptr_regno); struct bpf_func_state *state = func(env, reg); int err; /* Some accesses are only permitted with a static offset. */ bool var_off = !tnum_is_const(reg->var_off);
/* The offset is required to be static when reads don't go to a *register,inordertonotleakpointers(see *check_stack_read_fixed_off).
*/ if (dst_regno < 0 && var_off) { char tn_buf[48];
tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
verbose(env, "variable offset stack pointer cannot be passed into helper function; var_off=%s off=%d size=%d\n",
tn_buf, off, size); return -EACCES;
} /* Variable offset is prohibited for unprivileged mode for simplicity *sinceitrequirescorrespondingsupportinSpectremaskingforstack *ALU.Seealsoretrieve_ptr_limit().Thecheckin *check_stack_access_for_ptr_arithmetic()calledby *adjust_ptr_min_max_vals()preventsusersfromcreatingstackpointers *withvariableoffsets,thereforenocheckisrequiredhere.Further, *justcheckingitherewouldbeinsufficientasspeculativestack *writescouldstillleadtounsafespeculativebehaviour.
*/ if (!var_off) {
off += reg->var_off.value;
err = check_stack_read_fixed_off(env, state, off, size,
dst_regno);
} else { /* Variable offset stack reads need more conservative handling *thanfixedoffsetones.Notethatdst_regno>=0onthis *branch.
*/
err = check_stack_read_var_off(env, ptr_regno, off, size,
dst_regno);
} return err;
}
/* check_stack_write dispatches to check_stack_write_fixed_off or *check_stack_write_var_off. * *'ptr_regno'istheregisterusedasapointerintothestack. *'off'includes'ptr_regno->off',butnotitsvariableoffset(ifany). *'value_regno'istheregisterwhosevaluewe'rewritingtothestack.Itcan *be-1,meaningthatwe'renotwritingfromaregister. * *Thecallermustensurethattheoffsetfallswithinthemaximumstacksize.
*/ staticint check_stack_write(struct bpf_verifier_env *env, int ptr_regno, int off, int size, int value_regno, int insn_idx)
{ struct bpf_reg_state *reg = reg_state(env, ptr_regno); struct bpf_func_state *state = func(env, reg); int err;
reg = &cur_regs(env)[regno]; switch (reg->type) { case PTR_TO_MAP_KEY:
verbose(env, "invalid access to map key, key_size=%d off=%d size=%d\n",
mem_size, off, size); break; case PTR_TO_MAP_VALUE:
verbose(env, "invalid access to map value, value_size=%d off=%d size=%d\n",
mem_size, off, size); break; case PTR_TO_PACKET: case PTR_TO_PACKET_META: case PTR_TO_PACKET_END:
verbose(env, "invalid access to packet, off=%d size=%d, R%d(id=%d,off=%d,r=%d)\n",
off, size, regno, reg->id, off, mem_size); break; case PTR_TO_MEM: default:
verbose(env, "invalid access to memory, mem_size=%u off=%d size=%d\n",
mem_size, off, size);
}
return -EACCES;
}
/* check read/write into a memory region with possible variable offset */ staticint check_mem_region_access(struct bpf_verifier_env *env, u32 regno, int off, int size, u32 mem_size, bool zero_size_allowed)
{ struct bpf_verifier_state *vstate = env->cur_state; struct bpf_func_state *state = vstate->frame[vstate->curframe]; struct bpf_reg_state *reg = &state->regs[regno]; int err;
/* We may have adjusted the register pointing to memory region, so we *needtotryaddingeachofmin_valueandmax_valuetooff *tomakesureourtheoreticalaccesswillbesafe. * *Theminimumvalueisonlyimportantwithsigned *comparisonswherewecan'tassumethefloorofa *valueis0.Ifweareusingsignedvariablesforour *index'esweneedtomakesurethatwhateverweuse *willhaveasetfloorwithinourrange.
*/ if (reg->smin_value < 0 &&
(reg->smin_value == S64_MIN ||
(off + reg->smin_value != (s64)(s32)(off + reg->smin_value)) ||
reg->smin_value + off < 0)) {
verbose(env, "R%d min value is negative, either use unsigned index or do a if (index >=0) check.\n",
regno); return -EACCES;
}
err = __check_mem_access(env, regno, reg->smin_value + off, size,
mem_size, zero_size_allowed); if (err) {
verbose(env, "R%d min value is outside of the allowed memory range\n",
regno); return err;
}
/* If we haven't set a max value then we need to bail since we can't be *surewewon'tdobadthings. *Ifreg->umax_value+offcouldoverflow,treatthatasunboundedtoo.
*/ if (reg->umax_value >= BPF_MAX_VAR_OFF) {
verbose(env, "R%d unbounded memory access, make sure to bounds check any such access\n",
regno); return -EACCES;
}
err = __check_mem_access(env, regno, reg->umax_value + off, size,
mem_size, zero_size_allowed); if (err) {
verbose(env, "R%d max value is outside of the allowed memory range\n",
regno); return err;
}
return0;
}
staticint __check_ptr_off_reg(struct bpf_verifier_env *env, conststruct bpf_reg_state *reg, int regno, bool fixed_off_ok)
{ /* Access to this pointer-typed register or passing it to a helper *isonlyallowedinitsoriginal,unmodifiedform.
*/
if (btf_is_kernel(reg->btf)) {
perm_flags = PTR_MAYBE_NULL | PTR_TRUSTED | MEM_RCU;
/* Only unreferenced case accepts untrusted pointers */ if (kptr_field->type == BPF_KPTR_UNREF)
perm_flags |= PTR_UNTRUSTED;
} else {
perm_flags = PTR_MAYBE_NULL | MEM_ALLOC; if (kptr_field->type == BPF_KPTR_PERCPU)
perm_flags |= MEM_PERCPU;
}
if (base_type(reg->type) != PTR_TO_BTF_ID || (type_flag(reg->type) & ~perm_flags)) goto bad_type;
/* We need to verify reg->type and reg->btf, before accessing reg->btf */
reg_name = btf_type_name(reg->btf, reg->btf_id);
/* For ref_ptr case, release function check should ensure we get one *referencedPTR_TO_BTF_ID,andthatitsfixedoffsetis0.Forthe *normalstoreofunreferencedkptr,wemustensurevar_offiszero. *Sinceref_ptrcannotbeaccesseddirectlybyBPFinsns,checksfor *reg->offandreg->ref_obj_idarenotneededhere.
*/ if (__check_ptr_off_reg(env, reg, regno, true)) return -EACCES;
/* A full type match is needed, as BTF can be vmlinux, module or prog BTF, and *wealsoneedtotakeintoaccountthereg->off. * *Wewanttosupportcaseslike: * *structfoo{ *structbarbr; *structbazbz; *}; * *structfoo*v; *v=func();// PTR_TO_BTF_ID *val->foo=v;// reg->off is zero, btf and btf_id match type *val->bar=&v->br;// reg->off is still zero, but we need to retry with *// first member type of struct after comparison fails *val->baz=&v->bz;// reg->off is non-zero, so struct needs to be walked *// to match type * *Inthekptr_refcase,check_func_arg_reg_offalreadyensuresreg->off *iszero.Wemustalsoensurethatbtf_struct_ids_matchdoesnotwalk *thestructtomatchtypeagainstfirstmemberofstruct,i.e.reject *secondcasefromabove.Hence,whentypeisBPF_KPTR_REF,weset *strictmodetotruefortypematch.
*/ if (!btf_struct_ids_match(&env->log, reg->btf, reg->btf_id, reg->off,
kptr_field->kptr.btf, kptr_field->kptr.btf_id,
kptr_field->type != BPF_KPTR_UNREF)) goto bad_type; return0;
bad_type:
verbose(env, "invalid kptr access, R%d type=%s%s ", regno,
reg_type_str(env, reg->type), reg_name);
verbose(env, "expected=%s%s", reg_type_str(env, PTR_TO_BTF_ID), targ_name); if (kptr_field->type == BPF_KPTR_UNREF)
verbose(env, " or %s%s\n", reg_type_str(env, PTR_TO_BTF_ID | PTR_UNTRUSTED),
targ_name); else
verbose(env, "\n"); return -EINVAL;
}
ret = PTR_MAYBE_NULL; if (rcu_safe_kptr(kptr_field) && in_rcu_cs(env)) {
ret |= MEM_RCU; if (kptr_field->type == BPF_KPTR_PERCPU)
ret |= MEM_PERCPU; elseif (!btf_is_kernel(kptr_field->kptr.btf))
ret |= MEM_ALLOC;
rec = kptr_pointee_btf_record(kptr_field); if (rec && btf_record_has_field(rec, BPF_GRAPH_NODE))
ret |= NON_OWN_REF;
} else {
ret |= PTR_UNTRUSTED;
}
staticint check_map_kptr_access(struct bpf_verifier_env *env, u32 regno, int value_regno, int insn_idx, struct btf_field *kptr_field)
{ struct bpf_insn *insn = &env->prog->insnsi[insn_idx]; intclass = BPF_CLASS(insn->code); struct bpf_reg_state *val_reg; int ret;
/* Things we already checked for in check_map_access and caller: *-Rejectcaseswherevariableoffsetmaytouchkptr *-sizeofaccess(mustbeBPF_DW) *-tnum_is_const(reg->var_off) *-kptr_field->offset==off+reg->var_off.value
*/ /* Only BPF_[LDX,STX,ST] | BPF_MEM | BPF_DW is supported */ if (BPF_MODE(insn->code) != BPF_MEM) {
verbose(env, "kptr in map can only be accessed using BPF_MEM instruction mode\n"); return -EACCES;
}
/* We only allow loading referenced kptr, since it will be marked as *untrusted,similartounreferencedkptr.
*/ if (class != BPF_LDX &&
(kptr_field->type == BPF_KPTR_REF || kptr_field->type == BPF_KPTR_PERCPU)) {
verbose(env, "store to referenced kptr disallowed\n"); return -EACCES;
} if (class != BPF_LDX && kptr_field->type == BPF_UPTR) {
verbose(env, "store to uptr disallowed\n"); return -EACCES;
}
if (class == BPF_LDX) { if (kptr_field->type == BPF_UPTR) return mark_uptr_ld_reg(env, value_regno, kptr_field);
/* We can simply mark the value_regno receiving the pointer *valuefrommapasPTR_TO_BTF_ID,withthecorrecttype.
*/
ret = mark_btf_ld_reg(env, cur_regs(env), value_regno, PTR_TO_BTF_ID,
kptr_field->kptr.btf, kptr_field->kptr.btf_id,
btf_ld_kptr_type(env, kptr_field)); if (ret < 0) return ret;
} elseif (class == BPF_STX) {
val_reg = reg_state(env, value_regno); if (!register_is_null(val_reg) &&
map_kptr_match_type(env, kptr_field, val_reg, value_regno)) return -EACCES;
} elseif (class == BPF_ST) { if (insn->imm) {
verbose(env, "BPF_ST imm must be 0 when storing to kptr at off=%u\n",
kptr_field->offset); return -EACCES;
}
} else {
verbose(env, "kptr in map can only be accessed using BPF_LDX/BPF_STX/BPF_ST\n"); return -EACCES;
} return0;
}
/* check read/write into a map element with possible variable offset */ staticint check_map_access(struct bpf_verifier_env *env, u32 regno, int off, int size, bool zero_size_allowed, enum bpf_access_src src)
{ struct bpf_verifier_state *vstate = env->cur_state; struct bpf_func_state *state = vstate->frame[vstate->curframe]; struct bpf_reg_state *reg = &state->regs[regno]; struct bpf_map *map = reg->map_ptr; struct btf_record *rec; int err, i;
if (IS_ERR_OR_NULL(map->record)) return0;
rec = map->record; for (i = 0; i < rec->cnt; i++) { struct btf_field *field = &rec->fields[i];
u32 p = field->offset;
/* If any part of a field can be touched by load/store, reject *thisprogram.Tocheckthat[x1,x2)overlapswith[y1,y2), *itissufficienttocheckx1<y2&&y1<x2.
*/ if (reg->smin_value + off < p + field->size &&
p < reg->umax_value + off + size) { switch (field->type) { case BPF_KPTR_UNREF: case BPF_KPTR_REF: case BPF_KPTR_PERCPU: case BPF_UPTR: if (src != ACCESS_DIRECT) {
verbose(env, "%s cannot be accessed indirectly by helper\n",
btf_field_type_name(field->type)); return -EACCES;
} if (!tnum_is_const(reg->var_off)) {
verbose(env, "%s access cannot have variable offset\n",
btf_field_type_name(field->type)); return -EACCES;
} if (p != off + reg->var_off.value) {
verbose(env, "%s access misaligned expected=%u off=%llu\n",
btf_field_type_name(field->type),
p, off + reg->var_off.value); return -EACCES;
} if (size != bpf_size_to_bytes(BPF_DW)) {
verbose(env, "%s access size must be BPF_DW\n",
btf_field_type_name(field->type)); return -EACCES;
} break; default:
verbose(env, "%s cannot be accessed directly by load/store\n",
btf_field_type_name(field->type)); return -EACCES;
}
}
} return0;
}
switch (prog_type) { /* Program types only with direct read access go here! */ case BPF_PROG_TYPE_LWT_IN: case BPF_PROG_TYPE_LWT_OUT: case BPF_PROG_TYPE_LWT_SEG6LOCAL: case BPF_PROG_TYPE_SK_REUSEPORT: case BPF_PROG_TYPE_FLOW_DISSECTOR: case BPF_PROG_TYPE_CGROUP_SKB: if (t == BPF_WRITE) returnfalse;
fallthrough;
/* Program types with direct read + write access go here! */ case BPF_PROG_TYPE_SCHED_CLS: case BPF_PROG_TYPE_SCHED_ACT: case BPF_PROG_TYPE_XDP: case BPF_PROG_TYPE_LWT_XMIT: case BPF_PROG_TYPE_SK_SKB: case BPF_PROG_TYPE_SK_MSG: if (meta) return meta->pkt_access;
env->seen_direct_write = true; returntrue;
case BPF_PROG_TYPE_CGROUP_SOCKOPT: if (t == BPF_WRITE)
env->seen_direct_write = true;
returntrue;
default: returnfalse;
}
}
staticint check_packet_access(struct bpf_verifier_env *env, u32 regno, int off, int size, bool zero_size_allowed)
{ struct bpf_reg_state *regs = cur_regs(env); struct bpf_reg_state *reg = ®s[regno]; int err;
/* We may have added a variable offset to the packet pointer; but any *reg->rangewehavecomesafterthat.Weareonlycheckingthefixed *offset.
*/
/* We don't allow negative numbers, because we aren't tracking enough *detailtoprovethey'resafe.
*/ if (reg->smin_value < 0) {
verbose(env, "R%d min value is negative, either use unsigned index or do a if (index >=0) check.\n",
regno); return -EACCES;
}
err = reg->range < 0 ? -EINVAL :
__check_mem_access(env, regno, off, size, reg->range,
zero_size_allowed); if (err) {
verbose(env, "R%d offset is outside of the packet\n", regno); return err;
}
/* __check_mem_access has made sure "off + size - 1" is within u16. *reg->umax_valuecan'tbebiggerthanMAX_PACKET_OFFwhichis0xffff, *otherwisefind_good_pkt_pointerswouldhaverefusedtosetrangeinfo *that__check_mem_accesswouldhaverejectedthispktaccess. *Therefore,"off+reg->umax_value+size-1"won'toverflowu32.
*/
env->prog->aux->max_pkt_offset =
max_t(u32, env->prog->aux->max_pkt_offset,
off + reg->umax_value + size - 1);
return err;
}
/* check access to 'struct bpf_context' fields. Supports fixed offsets only */ staticint check_ctx_access(struct bpf_verifier_env *env, int insn_idx, int off, int size, enum bpf_access_type t, struct bpf_insn_access_aux *info)
{ if (env->ops->is_valid_access &&
env->ops->is_valid_access(off, size, t, env->prog, info)) { /* A non zero info.ctx_field_size indicates that this field is a *candidateforlaterverifiertransformationtoloadthewhole *fieldandthenapplyamaskwhenaccessedwithanarrower *accessthanactualctxaccesssize.Azeroinfo.ctx_field_size *willonlyallowforwholefieldaccessandrejectsanyother *typeofnarroweraccess.
*/ if (base_type(info->reg_type) == PTR_TO_BTF_ID) { if (info->ref_obj_id &&
!find_reference_state(env->cur_state, info->ref_obj_id)) {
verbose(env, "invalid bpf_context access off=%d. Reference may already be released\n",
off); return -EACCES;
}
} else {
env->insn_aux_data[insn_idx].ctx_field_size = info->ctx_field_size;
} /* remember the offset of last byte accessed in ctx */ if (env->prog->aux->max_ctx_offset < off + size)
env->prog->aux->max_ctx_offset = off + size; return0;
}
staticint check_flow_keys_access(struct bpf_verifier_env *env, int off, int size)
{ if (size < 0 || off < 0 ||
(u64)off + size > sizeof(struct bpf_flow_keys)) {
verbose(env, "invalid access to flow keys off=%d size=%d\n",
off, size); return -EACCES;
} return0;
}
staticint check_sock_access(struct bpf_verifier_env *env, int insn_idx,
u32 regno, int off, int size, enum bpf_access_type t)
{ struct bpf_reg_state *regs = cur_regs(env); struct bpf_reg_state *reg = ®s[regno]; struct bpf_insn_access_aux info = {}; bool valid;
if (reg->smin_value < 0) {
verbose(env, "R%d min value is negative, either use unsigned index or do a if (index >=0) check.\n",
regno); return -EACCES;
}
/* Return false if @regno contains a pointer whose type isn't supported for *atomicinstruction@insn.
*/ staticbool atomic_ptr_type_ok(struct bpf_verifier_env *env, int regno, struct bpf_insn *insn)
{ if (is_ctx_reg(env, regno)) returnfalse; if (is_pkt_reg(env, regno)) returnfalse; if (is_flow_key_reg(env, regno)) returnfalse; if (is_sk_reg(env, regno)) returnfalse; if (is_arena_reg(env, regno)) return bpf_jit_supports_insn(insn, true);
staticbool is_trusted_reg(conststruct bpf_reg_state *reg)
{ /* A referenced register is always trusted. */ if (reg->ref_obj_id) returntrue;
/* Types listed in the reg2btf_ids are always trusted */ if (reg2btf_ids[base_type(reg->type)] &&
!bpf_type_has_unsafe_modifiers(reg->type)) returntrue;
/* If a register is not referenced, it is trusted if it has the *MEM_ALLOCorPTR_TRUSTEDtypemodifiers,andnoothers.Someofthe *othertypemodifiersmaybesafe,butweelecttotakeanopt-in *approachhereassome(e.g.PTR_UNTRUSTEDandPTR_MAYBE_NULL)are *not. * *Eventually,weshouldmakePTR_TRUSTEDthesinglesourceoftruth *forwhetheraregisteristrusted.
*/ return type_flag(reg->type) & BPF_REG_TRUSTED_MODIFIERS &&
!bpf_type_has_unsafe_modifiers(reg->type);
}
staticint check_pkt_ptr_alignment(struct bpf_verifier_env *env, conststruct bpf_reg_state *reg, int off, int size, bool strict)
{ struct tnum reg_off; int ip_align;
/* Byte size accesses are always allowed. */ if (!strict || size == 1) return0;
/* For platforms that do not have a Kconfig enabling *CONFIG_HAVE_EFFICIENT_UNALIGNED_ACCESSthevalueof *NET_IP_ALIGNisuniversallysetto'2'.Andonplatforms *thatdosetCONFIG_HAVE_EFFICIENT_UNALIGNED_ACCESS,weget *tothiscodeonlyinstrictmodewherewewanttoemulate *theNET_IP_ALIGN==2checking.Thereforeusean *unconditionalIPalignvalueof'2'.
*/
ip_align = 2;
switch (reg->type) { case PTR_TO_PACKET: case PTR_TO_PACKET_META: /* Special case, because of NET_IP_ALIGN. Given metadata sits *rightinfront,treatittheverysameway.
*/ return check_pkt_ptr_alignment(env, reg, off, size, strict); case PTR_TO_FLOW_KEYS:
pointer_desc = "flow keys "; break; case PTR_TO_MAP_KEY:
pointer_desc = "key "; break; case PTR_TO_MAP_VALUE:
pointer_desc = "value "; break; case PTR_TO_CTX:
pointer_desc = "context "; break; case PTR_TO_STACK:
pointer_desc = "stack "; /* The stack spill tracking logic in check_stack_write_fixed_off() *andcheck_stack_read_fixed_off()reliesonstackaccessesbeing *aligned.
*/
strict = true; break; case PTR_TO_SOCKET:
pointer_desc = "sock "; break; case PTR_TO_SOCK_COMMON:
pointer_desc = "sock_common "; break; case PTR_TO_TCP_SOCK:
pointer_desc = "tcp_sock "; break; case PTR_TO_XDP_SOCK:
pointer_desc = "xdp_sock "; break; case PTR_TO_ARENA: return0; default: break;
} return check_generic_ptr_alignment(env, reg, pointer_desc, off, size,
strict);
}
staticenum priv_stack_mode bpf_enable_priv_stack(struct bpf_prog *prog)
{ if (!bpf_jit_supports_private_stack()) return NO_PRIV_STACK;
/* bpf_prog_check_recur() checks all prog types that use bpf trampoline *whilekprobe/tp/perf_event/raw_tpdon'tusetrampolinehencechecked *explicitly.
*/ switch (prog->type) { case BPF_PROG_TYPE_KPROBE: case BPF_PROG_TYPE_TRACEPOINT: case BPF_PROG_TYPE_PERF_EVENT: case BPF_PROG_TYPE_RAW_TRACEPOINT: return PRIV_STACK_ADAPTIVE; case BPF_PROG_TYPE_TRACING: case BPF_PROG_TYPE_LSM: case BPF_PROG_TYPE_STRUCT_OPS: if (prog->aux->priv_stack_requested || bpf_prog_check_recur(prog)) return PRIV_STACK_ADAPTIVE;
fallthrough; default: break;
}
return NO_PRIV_STACK;
}
staticint round_up_stack_depth(struct bpf_verifier_env *env, int stack_depth)
{ if (env->prog->jit_requested) return round_up(stack_depth, 16);
/* round up to 32-bytes, since this is granularity *ofinterpreterstacksize
*/ return round_up(max_t(u32, stack_depth, 1), 32);
}
/* starting from main bpf function walk all instructions of the function *andrecursivelywalkallcalleesthatgivenfunctioncancall. *Ignorejumpandexitinsns. *Sincerecursionispreventedbycheck_cfg()thisalgorithm *onlyneedsalocalstackofMAX_CALL_FRAMEStoremembercallsites
*/ staticint check_max_stack_depth_subprog(struct bpf_verifier_env *env, int idx, bool priv_stack_supported)
{ struct bpf_subprog_info *subprog = env->subprog_info; struct bpf_insn *insn = env->prog->insnsi; int depth = 0, frame = 0, i, subprog_end, subprog_depth; bool tail_call_reachable = false; int ret_insn[MAX_CALL_FRAMES]; int ret_prog[MAX_CALL_FRAMES]; int j;
i = subprog[idx].start; if (!priv_stack_supported)
subprog[idx].priv_stack_mode = NO_PRIV_STACK;
process_func: /* protect against potential stack overflow that might happen when *bpf2bpfcallsgetcombinedwithtailcalls.Limitthecaller'sstack *depthforsuchcasedownto256sothattheworstcasescenario *wouldresultin8kstacksize(32whichistailcalllimit*256= *8k). * *Togettheideawhatmighthappen,seeanexample: *func1->subrsp,128 *subfunc1->subrsp,256 *tailcall1->addrsp,256 *func2->subrsp,192(totalstacksize=128+192=320) *subfunc2->subrsp,64 *subfunc22->subrsp,128 *tailcall2->addrsp,128 *func3->subrsp,32(totalstacksize128+192+64+32=416) * *tailcallwillunwindthecurrentstackframebutitwillnotgetrid *ofcaller'sstackasshownontheexampleabove.
*/ if (idx && subprog[idx].has_tail_call && depth >= 256) {
verbose(env, "tail_calls are not allowed when call stack of previous frames is %d bytes. Too large\n",
depth); return -EACCES;
}
subprog_depth = round_up_stack_depth(env, subprog[idx].stack_depth); if (priv_stack_supported) { /* Request private stack support only if the subprog stack *depthisnolessthanBPF_PRIV_STACK_MIN_SIZE.Thisisto *avoidjitpenaltyifthestackusageissmall.
*/ if (subprog[idx].priv_stack_mode == PRIV_STACK_UNKNOWN &&
subprog_depth >= BPF_PRIV_STACK_MIN_SIZE)
subprog[idx].priv_stack_mode = PRIV_STACK_ADAPTIVE;
}
if (subprog[idx].priv_stack_mode == PRIV_STACK_ADAPTIVE) { if (subprog_depth > MAX_BPF_STACK) {
verbose(env, "stack size of subprog %d is %d. Too large\n",
idx, subprog_depth); return -EACCES;
}
} else {
depth += subprog_depth; if (depth > MAX_BPF_STACK) {
verbose(env, "combined stack size of %d calls is %d. Too large\n",
frame + 1, depth); return -EACCES;
}
}
continue_func:
subprog_end = subprog[idx + 1].start; for (; i < subprog_end; i++) { int next_insn, sidx;
if (!is_bpf_throw_kfunc(insn + i)) continue; if (subprog[idx].is_cb)
err = true; for (int c = 0; c < frame && !err; c++) { if (subprog[ret_prog[c]].is_cb) {
err = true; break;
}
} if (!err) continue;
verbose(env, "bpf_throw kfunc (insn %d) cannot be called from callback subprog %d\n",
i, idx); return -EINVAL;
}
if (!bpf_pseudo_call(insn + i) && !bpf_pseudo_func(insn + i)) continue; /* remember insn and function to return to */
ret_insn[frame] = i + 1;
ret_prog[frame] = idx;
/* find the callee */
next_insn = i + insn[i].imm + 1;
sidx = find_subprog(env, next_insn); if (verifier_bug_if(sidx < 0, env, "callee not found at insn %d", next_insn)) return -EFAULT; if (subprog[sidx].is_async_cb) { if (subprog[sidx].has_tail_call) {
verifier_bug(env, "subprog has tail_call and async cb"); return -EFAULT;
} /* async callbacks don't increase bpf prog stack size unless called directly */ if (!bpf_pseudo_call(insn + i)) continue; if (subprog[sidx].is_exception_cb) {
verbose(env, "insn %d cannot call exception cb directly", i); return -EINVAL;
}
}
i = next_insn;
idx = sidx; if (!priv_stack_supported)
subprog[idx].priv_stack_mode = NO_PRIV_STACK;
if (subprog[idx].has_tail_call)
tail_call_reachable = true;
frame++; if (frame >= MAX_CALL_FRAMES) {
verbose(env, "the call stack of %d frames is too deep !\n",
frame); return -E2BIG;
} goto process_func;
} /* if tail call got detected across bpf2bpf calls then mark each of the *currentlypresentsubprogframesastailcallreachablesubprogs; *thisinfowillbeutilizedbyJITsothatwewillbepreservingthe *tailcallcounterthroughoutbpf2bpfcallscombinedwithtailcalls
*/ if (tail_call_reachable) for (j = 0; j < frame; j++) { if (subprog[ret_prog[j]].is_exception_cb) {
verbose(env, "cannot tail call within exception cb\n"); return -EINVAL;
}
subprog[ret_prog[j]].tail_call_reachable = true;
} if (subprog[0].tail_call_reachable)
env->prog->aux->tail_call_reachable = true;
/* end of for() loop means the last insn of the 'subprog' *wasreached.Doesn'tmatterwhetheritwasJAorEXIT
*/ if (frame == 0) return0; if (subprog[idx].priv_stack_mode != PRIV_STACK_ADAPTIVE)
depth -= round_up_stack_depth(env, subprog[idx].stack_depth);
frame--;
i = ret_insn[frame];
idx = ret_prog[frame]; goto continue_func;
}
for (int i = 0; i < env->subprog_cnt; i++) { if (si[i].has_tail_call) {
priv_stack_mode = NO_PRIV_STACK; break;
}
}
if (priv_stack_mode == PRIV_STACK_UNKNOWN)
priv_stack_mode = bpf_enable_priv_stack(env->prog);
/* All async_cb subprogs use normal kernel stack. If a particular *subprogappearsinbothmainprogandasync_cbsubtree,that *subprogwillusenormalkernelstacktoavoidpotentialnesting. *Thereversesubprogtraversalensureswhenmainprogsubtreeis *checked,thesubprogsappearinginasync_cbsubtreesarealready *markedasusingnormalkernelstack,sostacksizecheckingcan *bedoneproperly.
*/ for (int i = env->subprog_cnt - 1; i >= 0; i--) { if (!i || si[i].is_async_cb) {
priv_stack_supported = !i && priv_stack_mode == PRIV_STACK_ADAPTIVE;
ret = check_max_stack_depth_subprog(env, i, priv_stack_supported); if (ret < 0) return ret;
}
}
for (int i = 0; i < env->subprog_cnt; i++) { if (si[i].priv_stack_mode == PRIV_STACK_ADAPTIVE) {
env->prog->aux->jits_use_priv_stack = true; break;
}
}
/* If size is smaller than 32bit register the 32bit register *valuesarealsotruncatedsowepush64-bitboundsinto *32-bitbounds.Aboveweretruncated<32-bitsalready.
*/ if (size < 4)
__mark_reg32_unbounded(reg);
/* RCU trusted: these fields are trusted in RCU CS and can be NULL */
BTF_TYPE_SAFE_RCU_OR_NULL(struct mm_struct) { struct file __rcu *exe_file;
};
/* skb->sk, req->sk are not RCU protected, but we mark them as such *becausebpfprogaccessiblesocketsareSOCK_RCU_FREE.
*/
BTF_TYPE_SAFE_RCU_OR_NULL(struct sk_buff) { struct sock *sk;
};
/* full trusted: these fields are trusted even outside of RCU CS and never NULL */
BTF_TYPE_SAFE_TRUSTED(struct bpf_iter_meta) { struct seq_file *seq;
};
} elseif (type_flag(reg->type) & PTR_UNTRUSTED) { /* If this is an untrusted pointer, all pointers formed by walking it *alsoinherittheuntrustedflag.
*/
flag = PTR_UNTRUSTED;
} elseif (is_trusted_reg(reg) || is_rcu_reg(reg)) { /* By default any pointer obtained from walking a trusted pointer is no *longertrusted,unlessthefieldbeingaccessedhasexplicitlybeen *markedasinheritingitsparent'sstateoftrust(eitherfullorRCU). *Forexample: *'cgroups'pointerisuntrustediftask->cgroupsdereference *happenedinasleepableprogramoutsideofbpf_rcu_read_lock() *section.Inanon-sleepableprogramit'strustedwhileinRCUCS(akaMEM_RCU). *Notebpf_rcu_read_unlock()convertsMEM_RCUpointerstoPTR_UNTRUSTED. * *AregularRCU-protectedpointerwith__rcutagcanalsobedeemed *trustedifweareinanRCUCS.SuchpointercanbeNULL.
*/ if (type_is_trusted(env, reg, field_name, btf_id)) {
flag |= PTR_TRUSTED;
} elseif (type_is_trusted_or_null(env, reg, field_name, btf_id)) {
flag |= PTR_TRUSTED | PTR_MAYBE_NULL;
} elseif (in_rcu_cs(env) && !type_may_be_null(reg->type)) { if (type_is_rcu(env, reg, field_name, btf_id)) { /* ignore __rcu tag and mark it MEM_RCU */
flag |= MEM_RCU;
} elseif (flag & MEM_RCU ||
type_is_rcu_or_null(env, reg, field_name, btf_id)) { /* __rcu tagged pointers can be NULL */
flag |= MEM_RCU | PTR_MAYBE_NULL;
/* We always trust them */ if (type_is_rcu_or_null(env, reg, field_name, btf_id) &&
flag & PTR_UNTRUSTED)
flag &= ~PTR_UNTRUSTED;
} elseif (flag & (MEM_PERCPU | MEM_USER)) { /* keep as-is */
} else { /* walking unknown pointers yields old deprecated PTR_TO_BTF_ID */
clear_trusted_flags(&flag);
}
} else { /* *IfnotinRCUCSorMEM_RCUpointercanbeNULLthen *aggressivelymarkasuntrustedotherwisesuch *pointerswillbeplainPTR_TO_BTF_IDwithoutflags *andwillbeallowedtobepassedintohelpersfor *compatreasons.
*/
flag = PTR_UNTRUSTED;
}
} else { /* Old compat. Deprecated */
clear_trusted_flags(&flag);
}
if (atype == BPF_READ && value_regno >= 0) {
ret = mark_btf_ld_reg(env, regs, value_regno, ret, reg->btf, btf_id, flag); if (ret < 0) return ret;
}
return0;
}
staticint check_ptr_to_map_access(struct bpf_verifier_env *env, struct bpf_reg_state *regs, int regno, int off, int size, enum bpf_access_type atype, int value_regno)
{ struct bpf_reg_state *reg = regs + regno; struct bpf_map *map = reg->map_ptr; struct bpf_reg_state map_reg; enum bpf_type_flag flag = 0; conststruct btf_type *t; constchar *tname;
u32 btf_id; int ret;
if (!btf_vmlinux) {
verbose(env, "map_ptr access not supported without CONFIG_DEBUG_INFO_BTF\n"); return -ENOTSUPP;
}
if (!map->ops->map_btf_id || !*map->ops->map_btf_id) {
verbose(env, "map_ptr access not supported for map type %d\n",
map->map_type); return -ENOTSUPP;
}
t = btf_type_by_id(btf_vmlinux, *map->ops->map_btf_id);
tname = btf_name_by_offset(btf_vmlinux, t->name_off);
if (!env->allow_ptr_leaks) {
verbose(env, "'struct %s' access is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n",
tname); return -EPERM;
}
if (off < 0) {
verbose(env, "R%d is %s invalid negative access: off=%d\n",
regno, tname, off); return -EACCES;
}
if (atype != BPF_READ) {
verbose(env, "only read from %s is supported\n", tname); return -EACCES;
}
/* Simulate access to a PTR_TO_BTF_ID */
memset(&map_reg, 0, sizeof(map_reg));
ret = mark_btf_ld_reg(env, &map_reg, 0, PTR_TO_BTF_ID,
btf_vmlinux, *map->ops->map_btf_id, 0); if (ret < 0) return ret;
ret = btf_struct_access(&env->log, &map_reg, off, size, atype, &btf_id, &flag, NULL); if (ret < 0) return ret;
if (value_regno >= 0) {
ret = mark_btf_ld_reg(env, regs, value_regno, ret, btf_vmlinux, btf_id, flag); if (ret < 0) return ret;
}
return0;
}
/* Check that the stack access at the given offset is within bounds. The *maximumvalidoffsetis-1. * *Theminimumvalidoffsetis-MAX_BPF_STACKforwrites,and *-state->allocated_stackforreads.
*/ staticint check_stack_slot_within_bounds(struct bpf_verifier_env *env,
s64 off, struct bpf_func_state *state, enum bpf_access_type t)
{ int min_valid_off;
/* Note that there is no stack access with offset zero, so the needed stack *sizeis-min_off,not-min_off+1.
*/ return grow_stack_state(env, state, -min_off /* size */);
}
/* if map is read-only, track its contents as scalars */ if (tnum_is_const(reg->var_off) &&
bpf_map_is_rdonly(map) &&
map->ops->map_direct_value_addr) { int map_off = off + reg->var_off.value;
u64 val = 0;
if (insn->imm == BPF_CMPXCHG) { /* Check comparison of R0 with memory location */ const u32 aux_reg = BPF_REG_0;
err = check_reg_arg(env, aux_reg, SRC_OP); if (err) return err;
if (is_pointer_value(env, aux_reg)) {
verbose(env, "R%d leaks addr into mem\n", aux_reg); return -EACCES;
}
}
if (is_pointer_value(env, insn->src_reg)) {
verbose(env, "R%d leaks addr into mem\n", insn->src_reg); return -EACCES;
}
if (!atomic_ptr_type_ok(env, insn->dst_reg, insn)) {
verbose(env, "BPF_ATOMIC stores into R%d %s is not allowed\n",
insn->dst_reg,
reg_type_str(env, reg_state(env, insn->dst_reg)->type)); return -EACCES;
}
if (insn->imm & BPF_FETCH) { if (insn->imm == BPF_CMPXCHG)
load_reg = BPF_REG_0; else
load_reg = insn->src_reg;
/* check and record load of old value */
err = check_reg_arg(env, load_reg, DST_OP); if (err) return err;
} else { /* This instruction accesses a memory location but doesn't *actuallyloaditintoaregister.
*/
load_reg = -1;
}
/* Check whether we can read the memory, with second call for fetch *casetosimulatetheregisterfill.
*/
err = check_mem_access(env, env->insn_idx, insn->dst_reg, insn->off,
BPF_SIZE(insn->code), BPF_READ, -1, true, false); if (!err && load_reg >= 0)
err = check_mem_access(env, env->insn_idx, insn->dst_reg,
insn->off, BPF_SIZE(insn->code),
BPF_READ, load_reg, true, false); if (err) return err;
if (is_arena_reg(env, insn->dst_reg)) {
err = save_aux_ptr_type(env, PTR_TO_ARENA, false); if (err) return err;
} /* Check whether we can write into the same memory. */
err = check_mem_access(env, env->insn_idx, insn->dst_reg, insn->off,
BPF_SIZE(insn->code), BPF_WRITE, -1, true, false); if (err) return err; return0;
}
staticint check_atomic_load(struct bpf_verifier_env *env, struct bpf_insn *insn)
{ int err;
if (!atomic_ptr_type_ok(env, insn->src_reg, insn)) {
verbose(env, "BPF_ATOMIC loads from R%d %s is not allowed\n",
insn->src_reg,
reg_type_str(env, reg_state(env, insn->src_reg)->type)); return -EACCES;
}
return0;
}
staticint check_atomic_store(struct bpf_verifier_env *env, struct bpf_insn *insn)
{ int err;
err = check_store_reg(env, insn, true); if (err) return err;
if (!atomic_ptr_type_ok(env, insn->dst_reg, insn)) {
verbose(env, "BPF_ATOMIC stores into R%d %s is not allowed\n",
insn->dst_reg,
reg_type_str(env, reg_state(env, insn->dst_reg)->type)); return -EACCES;
}
return0;
}
staticint check_atomic(struct bpf_verifier_env *env, struct bpf_insn *insn)
{ switch (insn->imm) { case BPF_ADD: case BPF_ADD | BPF_FETCH: case BPF_AND: case BPF_AND | BPF_FETCH: case BPF_OR: case BPF_OR | BPF_FETCH: case BPF_XOR: case BPF_XOR | BPF_FETCH: case BPF_XCHG: case BPF_CMPXCHG: return check_atomic_rmw(env, insn); case BPF_LOAD_ACQ: if (BPF_SIZE(insn->code) == BPF_DW && BITS_PER_LONG != 64) {
verbose(env, "64-bit load-acquires are only supported on 64-bit arches\n"); return -EOPNOTSUPP;
} return check_atomic_load(env, insn); case BPF_STORE_REL: if (BPF_SIZE(insn->code) == BPF_DW && BITS_PER_LONG != 64) {
verbose(env, "64-bit store-releases are only supported on 64-bit arches\n"); return -EOPNOTSUPP;
} return check_atomic_store(env, insn); default:
verbose(env, "BPF_ATOMIC uses invalid atomic opcode %02x\n",
insn->imm); return -EINVAL;
}
}
/* When register 'regno' is used to read the stack (either directly or through *ahelperfunction)makesurethatit'swithinstackboundaryand,depending *ontheaccesstypeandprivileges,thatallelementsofthestackare *initialized. * *'off'includes'regno->off',butnotitsdynamicpart(ifany). * *Allregistersthathavebeenspilledonthestackintheslotswithinthe *readoffsetsaremarkedasread.
*/ staticint check_stack_range_initialized( struct bpf_verifier_env *env, int regno, int off, int access_size, bool zero_size_allowed, enum bpf_access_type type, struct bpf_call_arg_meta *meta)
{ struct bpf_reg_state *reg = reg_state(env, regno); struct bpf_func_state *state = func(env, reg); int err, min_off, max_off, i, j, slot, spi; /* Some accesses can write anything into the stack, others are *read-only.
*/ bool clobber = false;
stype = &state->stack[spi].slot_type[slot % BPF_REG_SIZE]; if (*stype == STACK_MISC) goto mark; if ((*stype == STACK_ZERO) ||
(*stype == STACK_INVALID && env->allow_uninit_stack)) { if (clobber) { /* helper can write anything into the stack */
*stype = STACK_MISC;
} goto mark;
}
if (is_spilled_reg(&state->stack[spi]) &&
(state->stack[spi].spilled_ptr.type == SCALAR_VALUE ||
env->allow_ptr_leaks)) { if (clobber) {
__mark_reg_unknown(env, &state->stack[spi].spilled_ptr); for (j = 0; j < BPF_REG_SIZE; j++)
scrub_spilled_slot(&state->stack[spi].slot_type[j]);
} goto mark;
}
if (tnum_is_const(reg->var_off)) {
verbose(env, "invalid read from stack R%d off %d+%d size %d\n",
regno, min_off, i - min_off, access_size);
} else { char tn_buf[48];
tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
verbose(env, "invalid read from stack R%d var_off %s+%d size %d\n",
regno, tn_buf, i - min_off, access_size);
} return -EACCES;
mark: /* reading any byte out of 8-byte 'spill_slot' will cause *thewholeslottobemarkedas'read'
*/
mark_reg_read(env, &state->stack[spi].spilled_ptr,
state->stack[spi].spilled_ptr.parent,
REG_LIVE_READ64); /* We do not set REG_LIVE_WRITTEN for stack slot, as we can not *besurethatwhetherstackslotiswrittentoornot.Hence, *wemuststillconservativelypropagatereadsupwardsevenif *helpermaywritetotheentirememoryrange.
*/
} return0;
}
/* verify arguments to helpers or kfuncs consisting of a pointer and an access *size. * *@regnoistheregistercontainingtheaccesssize.regno-1istheregister *containingthepointer.
*/ staticint check_mem_size_reg(struct bpf_verifier_env *env, struct bpf_reg_state *reg, u32 regno, enum bpf_access_type access_type, bool zero_size_allowed, struct bpf_call_arg_meta *meta)
{ int err;
/* This is used to refine r0 return value bounds for helpers *thatenforcethisvalueasanupperboundonreturnvalues. *Seedo_refine_retval_range()forhelpersthatcanrefine *thereturnvalue.Ctypeofhelperisu32sowepullregister *boundfromumax_valuehowever,ifnegativeverifiererrors *out.Onlyupperboundscanbelearnedbecauseretvalisan *inttypeandnegativeretvalsareallowed.
*/
meta->msize_max_value = reg->umax_value;
/* The register is SCALAR_VALUE; the access check happens using *itsboundaries.Forunprivilegedvariableaccesses,disable *rawmodesothattheprogramisrequiredtoinitializeall *thememorythatthehelpercouldjustpartiallyfillup.
*/ if (!tnum_is_const(reg->var_off))
meta = NULL;
if (reg->smin_value < 0) {
verbose(env, "R%d min value is negative, either use unsigned or 'var &= const'\n",
regno); return -EACCES;
}
/* Assuming that the register contains a value check if the memory *accessissafe.Temporarilysaveandrestoretheregister'sstateas *theconversionshouldn'tbevisibletoacaller.
*/ if (may_be_null) {
saved_reg = *reg;
mark_ptr_not_null_reg(reg);
}
if (!is_const) {
verbose(env, "R%d doesn't have constant offset. %s_lock has to be at the constant offset\n",
regno, lock_str); return -EINVAL;
} if (reg->type == PTR_TO_MAP_VALUE) {
map = reg->map_ptr; if (!map->btf) {
verbose(env, "map '%s' has to have BTF in order to use %s_lock\n",
map->name, lock_str); return -EINVAL;
}
} else {
btf = reg->btf;
}
rec = reg_btf_record(reg); if (!btf_record_has_field(rec, is_res_lock ? BPF_RES_SPIN_LOCK : BPF_SPIN_LOCK)) {
verbose(env, "%s '%s' has no valid %s_lock\n", map ? "map" : "local",
map ? map->name : "kptr", lock_str); return -EINVAL;
}
spin_lock_off = is_res_lock ? rec->res_spin_lock_off : rec->spin_lock_off; if (spin_lock_off != val + reg->off) {
verbose(env, "off %lld doesn't point to 'struct %s_lock' that is at %d\n",
val + reg->off, lock_str, spin_lock_off); return -EINVAL;
} if (is_lock) { void *ptr; int type;
if (map)
ptr = map; else
ptr = btf;
if (!is_res_lock && cur->active_locks) { if (find_lock_state(env->cur_state, REF_TYPE_LOCK, 0, NULL)) {
verbose(env, "Locking two bpf_spin_locks are not allowed\n"); return -EINVAL;
}
} elseif (is_res_lock && cur->active_locks) { if (find_lock_state(env->cur_state, REF_TYPE_RES_LOCK | REF_TYPE_RES_LOCK_IRQ, reg->id, ptr)) {
verbose(env, "Acquiring the same lock again, AA deadlock detected\n"); return -EINVAL;
}
}
if (is_res_lock && is_irq)
type = REF_TYPE_RES_LOCK_IRQ; elseif (is_res_lock)
type = REF_TYPE_RES_LOCK; else
type = REF_TYPE_LOCK;
err = acquire_lock_state(env, env->insn_idx, type, reg->id, ptr); if (err < 0) {
verbose(env, "Failed to acquire lock state\n"); return err;
}
} else { void *ptr; int type;
if (map)
ptr = map; else
ptr = btf;
if (!cur->active_locks) {
verbose(env, "%s_unlock without taking a lock\n", lock_str); return -EINVAL;
}
if (is_res_lock && is_irq)
type = REF_TYPE_RES_LOCK_IRQ; elseif (is_res_lock)
type = REF_TYPE_RES_LOCK; else
type = REF_TYPE_LOCK; if (!find_lock_state(cur, type, reg->id, ptr)) {
verbose(env, "%s_unlock of different lock\n", lock_str); return -EINVAL;
} if (reg->id != cur->active_lock_id || ptr != cur->active_lock_ptr) {
verbose(env, "%s_unlock cannot be out of order\n", lock_str); return -EINVAL;
} if (release_lock_state(cur, type, reg->id, ptr)) {
verbose(env, "%s_unlock of different lock\n", lock_str); return -EINVAL;
}
if (!is_const) {
verbose(env, "R%d doesn't have constant offset. bpf_timer has to be at the constant offset\n",
regno); return -EINVAL;
} if (!map->btf) {
verbose(env, "map '%s' has to have BTF in order to use bpf_timer\n",
map->name); return -EINVAL;
} if (!btf_record_has_field(map->record, BPF_TIMER)) {
verbose(env, "map '%s' has no valid bpf_timer\n", map->name); return -EINVAL;
} if (map->record->timer_off != val + reg->off) {
verbose(env, "off %lld doesn't point to 'struct bpf_timer' that is at %d\n",
val + reg->off, map->record->timer_off); return -EINVAL;
} if (meta->map_ptr) {
verifier_bug(env, "Two map pointers in a timer helper"); return -EFAULT;
} if (IS_ENABLED(CONFIG_PREEMPT_RT)) {
verbose(env, "bpf_timer cannot be used for PREEMPT_RT.\n"); return -EOPNOTSUPP;
}
meta->map_uid = reg->map_uid;
meta->map_ptr = map; return0;
}
if (map->record->wq_off != val + reg->off) {
verbose(env, "off %lld doesn't point to 'struct bpf_wq' that is at %d\n",
val + reg->off, map->record->wq_off); return -EINVAL;
}
meta->map.uid = reg->map_uid;
meta->map.ptr = map; return0;
}
if (type_is_ptr_alloc_obj(reg->type)) {
rec = reg_btf_record(reg);
} else { /* PTR_TO_MAP_VALUE */
map_ptr = reg->map_ptr; if (!map_ptr->btf) {
verbose(env, "map '%s' has to have BTF in order to use bpf_kptr_xchg\n",
map_ptr->name); return -EINVAL;
}
rec = map_ptr->record;
meta->map_ptr = map_ptr;
}
if (!tnum_is_const(reg->var_off)) {
verbose(env, "R%d doesn't have constant offset. kptr has to be at the constant offset\n",
regno); return -EINVAL;
}
if (!btf_record_has_field(rec, BPF_KPTR)) {
verbose(env, "R%d has no valid kptr\n", regno); return -EINVAL;
}
/* There are two register types representing a bpf_dynptr, one is PTR_TO_STACK *whichpointstoastackslot,andtheotherisCONST_PTR_TO_DYNPTR. * *Inbothcaseswedealwiththefirst8bytes,butneedtomarkthenext8 *bytesasSTACK_DYNPTRincaseofPTR_TO_STACK.Incaseof *CONST_PTR_TO_DYNPTR,weareguaranteedtogetthebeginningoftheobject. * *Mutabilityofbpf_dynptrisattwolevels,oneisatthelevelofstruct *bpf_dynptritself,i.e.whetherthehelperisreceivingapointertostruct *bpf_dynptrorpointertoconststructbpf_dynptr.Intheformercase,itcan *mutatetheviewofthedynptrandalsopossiblydestroyit.Inthelatter *case,itcannotmutatethebpf_dynptritselfbutitcanstillmutatethe *memorythatdynptrpointsto. * *Theverifierwillkeeptrackbothlevelsofmutation(bpf_dynptr'sin *reg->typeandthememory'sinreg->dynptr.type),butthereisnosupportfor *readonlydynptrviewyet,henceonlythefirstcaseistrackedandchecked. * *ThisisconsistentwithhowCappliestheconstmodifiertoastructobject, *wherethepointeritselfinsidebpf_dynptrbecomesconstbutnotwhatit *pointsto. * *Helperswhichdonotmutatethebpf_dynptrsetMEM_RDONLYintheirargument *type,anddeclareitas'conststructbpf_dynptr*'intheirprototype.
*/ staticint process_dynptr_func(struct bpf_verifier_env *env, int regno, int insn_idx, enum bpf_arg_type arg_type, int clone_ref_obj_id)
{ struct bpf_reg_state *regs = cur_regs(env), *reg = ®s[regno]; int err;
if (reg->type != PTR_TO_STACK && reg->type != CONST_PTR_TO_DYNPTR) {
verbose(env, "arg#%d expected pointer to stack or const struct bpf_dynptr\n",
regno - 1); return -EINVAL;
}
/* MEM_UNINIT and MEM_RDONLY are exclusive, when applied to an *ARG_PTR_TO_DYNPTR(orARG_PTR_TO_DYNPTR|DYNPTR_TYPE_*):
*/ if ((arg_type & (MEM_UNINIT | MEM_RDONLY)) == (MEM_UNINIT | MEM_RDONLY)) {
verifier_bug(env, "misconfigured dynptr helper type flags"); return -EFAULT;
}
/* MEM_UNINIT - Points to memory that is an appropriate candidate for *constructingamutablebpf_dynptrobject. * *Currently,thisisonlypossiblewithPTR_TO_STACK *pointingtoaregionofatleast16byteswhichdoesn't *containanexistingbpf_dynptr. * *MEM_RDONLY-Pointstoainitializedbpf_dynptrthatwillnotbe *mutatedordestroyed.However,thememoryitpointsto *maybemutated. * *None-Pointstoainitializeddynptrthatcanbemutatedand *destroyed,includingmutationofthememoryitpoints *to.
*/ if (arg_type & MEM_UNINIT) { int i;
if (!is_dynptr_reg_valid_uninit(env, reg)) {
verbose(env, "Dynptr has to be an uninitialized dynptr\n"); return -EINVAL;
}
/* we write BPF_DW bits (8 bytes) at a time */ for (i = 0; i < BPF_DYNPTR_SIZE; i += 8) {
err = check_mem_access(env, insn_idx, regno,
i, BPF_DW, BPF_WRITE, -1, false, false); if (err) return err;
}
err = mark_stack_slots_dynptr(env, reg, arg_type, insn_idx, clone_ref_obj_id);
} else/* MEM_RDONLY and None case from above */ { /* For the reg->type == PTR_TO_STACK case, bpf_dynptr is never const */ if (reg->type == CONST_PTR_TO_DYNPTR && !(arg_type & MEM_RDONLY)) {
verbose(env, "cannot pass pointer to const bpf_dynptr, the helper mutates it\n"); return -EINVAL;
}
if (!is_dynptr_reg_valid_init(env, reg)) {
verbose(env, "Expected an initialized dynptr as arg #%d\n",
regno - 1); return -EINVAL;
}
/* Fold modifiers (in this case, MEM_RDONLY) when checking expected type */ if (!is_dynptr_type_expected(env, reg, arg_type & ~MEM_RDONLY)) {
verbose(env, "Expected a dynptr of type %s as arg #%d\n",
dynptr_type_str(arg_to_dynptr_type(arg_type)), regno - 1); return -EINVAL;
}
staticbool is_kfunc_arg_iter(struct bpf_kfunc_call_arg_meta *meta, int arg_idx, conststruct btf_param *arg)
{ /* btf_check_iter_kfuncs() guarantees that first argument of any iter *kfuncisiterstatepointer
*/ if (is_iter_kfunc(meta)) return arg_idx == 0;
/* iter passed as an argument to a generic kfunc */ return btf_param_match_suffix(meta->btf, arg, "__iter");
}
staticint process_iter_arg(struct bpf_verifier_env *env, int regno, int insn_idx, struct bpf_kfunc_call_arg_meta *meta)
{ struct bpf_reg_state *regs = cur_regs(env), *reg = ®s[regno]; conststruct btf_type *t; int spi, err, i, nr_slots, btf_id;
if (reg->type != PTR_TO_STACK) {
verbose(env, "arg#%d expected pointer to an iterator on stack\n", regno - 1); return -EINVAL;
}
/* For iter_{new,next,destroy} functions, btf_check_iter_kfuncs() *ensuresstructconvention,sowewouldn'tneedtodoanyBTF *validationhere.Butgiveniterstatecanbepassedasaparameter *toanykfunc,ifarghas"__iter"suffix,weneedtobeabitmore *conservativehere.
*/
btf_id = btf_check_iter_arg(meta->btf, meta->func_proto, regno - 1); if (btf_id < 0) {
verbose(env, "expected valid iter pointer as arg #%d\n", regno - 1); return -EINVAL;
}
t = btf_type_by_id(meta->btf, btf_id);
nr_slots = t->size / BPF_REG_SIZE;
if (is_iter_new_kfunc(meta)) { /* bpf_iter_<type>_new() expects pointer to uninit iter state */ if (!is_iter_reg_valid_uninit(env, reg, nr_slots)) {
verbose(env, "expected uninitialized iter_%s as arg #%d\n",
iter_type_str(meta->btf, btf_id), regno - 1); return -EINVAL;
}
for (i = 0; i < nr_slots * 8; i += BPF_REG_SIZE) {
err = check_mem_access(env, insn_idx, regno,
i, BPF_DW, BPF_WRITE, -1, false, false); if (err) return err;
}
err = mark_stack_slots_iter(env, meta, reg, insn_idx, meta->btf, btf_id, nr_slots); if (err) return err;
} else { /* iter_next() or iter_destroy(), as well as any kfunc *acceptingiterargument,expectinitializediterstate
*/
err = is_iter_reg_valid_init(env, reg, meta->btf, btf_id, nr_slots); switch (err) { case0: break; case -EINVAL:
verbose(env, "expected an initialized iter_%s as arg #%d\n",
iter_type_str(meta->btf, btf_id), regno - 1); return err; case -EPROTO:
verbose(env, "expected an RCU CS when using %s\n", meta->func_name); return err; default: return err;
}
err = mark_iter_read(env, reg, spi, nr_slots); if (err) return err;
/* remember meta->iter info for process_iter_next_call() */
meta->iter.spi = spi;
meta->iter.frameno = reg->frameno;
meta->ref_obj_id = iter_ref_obj_id(env, reg, spi);
if (is_iter_destroy_kfunc(meta)) {
err = unmark_stack_slots_iter(env, reg, nr_slots); if (err) return err;
}
}
return0;
}
/* Look for a previous loop entry at insn_idx: nearest parent state *stoppedatinsn_idxwithcallsitesmatchingthoseincur->frame.
*/ staticstruct bpf_verifier_state *find_prev_entry(struct bpf_verifier_env *env, struct bpf_verifier_state *cur, int insn_idx)
{ struct bpf_verifier_state_list *sl; struct bpf_verifier_state *st; struct list_head *pos, *head;
/* Explored states are pushed in stack order, most recent states come first */
head = explored_state(env, insn_idx);
list_for_each(pos, head) {
sl = container_of(pos, struct bpf_verifier_state_list, node); /* If st->branches != 0 state is a part of current DFS verification path, *hencecur&stforaloop.
*/
st = &sl->state; if (st->insn_idx == insn_idx && st->branches && same_callsites(st, cur) &&
st->dfs_depth < cur->dfs_depth) return st;
}
/* process_iter_next_call() is called when verifier gets to iterator's next *"method"(e.g.,bpf_iter_num_next()fornumbersiterator)call.We'llrefer *toitasjust"iter_next()"incommentsbelow. * *BPFverifierreliesonacrucialcontractforanyiter_next() *implementation:itshould*eventually*returnNULL,andoncethathappens *itshouldkeepreturningNULL.Thatis,onceiteratorexhaustselementsto *iterate,itshouldneverresetorspuriouslyreturnnewelements. * *Withtheassumptionofsuchcontract,process_iter_next_call()simulates *aforkintheverifierstatetovalidatelooplogiccorrectnessandsafety *withouthavingtosimulateinfiniteamountofiterations. * *Incurrentstate,wefirstassumethatiter_next()returnedNULLand *iteratorstateissettoDRAINED(BPF_ITER_STATE_DRAINED).Insuch *conditionsweshouldnotformaninfiniteloopandshouldeventuallyreach *exit. * *Besidesthat,wealsoforkcurrentstateandenqueueitforlater *verification.InaforkedstatewekeepiteratorstateasACTIVE *(BPF_ITER_STATE_ACTIVE)andassumenon-NULLreturnfromiter_next().We *alsobumpiterationdepthtopreventerroneousinfiniteloopdetection *lateron(seeiter_active_depths_differ()commentfordetails).Inthis *stateweassumethatwe'lleventuallyloopbacktoanotheriter_next() *calls(itcouldbeinexactlysamelocationorinsomeotherinstruction, *itdoesn'tmatter,wedon'tmakeanyunnecessaryassumptionsaboutthis, *everythingrevolvesarounditeratorstateinastackslot,notwhich *instructioniscallingiter_next()).Whenthathappens,weeitherwillcome *toiter_next()withequivalentstateandcanconcludethatnextiteration *willproceedinexactlythesamewayaswejustverified,soit'ssafeto *assumethatloopconverges.Ifnot,we'llgoonanotheriteration *simulationwithadifferentinputstate,untilallpossiblestartingstates *arevalidatedorwereachmaximumnumberofinstructionslimit. * *Thisway,wewilleitherexhaustivelydiscoverallpossibleinputstates *thatiteratorloopcanstartwithandeventuallywillconverge,orwe'll *effectivelyregressintoboundedloopsimulationlogicandeitherreach *maximumnumberofinstructionsifloopisnotprovablyconvergent,orthere *issomestaticallyknownlimitonnumberofiterations(e.g.,ifthereis *anexplicit`ifn>100thenbreak;`statementsomewhereintheloop). * *Iterationconvergencelogicinis_state_visited()reliesonexact *statescomparison,whichignoresreadandprecisionmarks. *Thisisnecessarybecausereadandprecisionmarksarenotfinalized *whileintheloop.Exactcomparisonmightprecludeconvergencefor *simpleprogramslikebelow: * *i=0; *while(iter_next(&it)) *i++; * *Ateachiterationstepi++wouldproduceanewdistinctstateand *eventuallyinstructionprocessinglimitwouldbereached. * *Toavoidsuchbehaviorspeculativelyforget(widen)rangefor *imprecisescalarregisters,ifthoseregisterswerenotpreciseatthe *endofthepreviousiterationanddonotmatchexactly. * *Thisisaconservativeheuristicthatallowstoverifywiderangeofprograms, *howeveritprecludesverificationofprogramsthatconjurean *imprecisevalueonthefirstloopiterationanduseitaspreciseonasecond. *Forexample,thefollowingsafeprogramwouldfailtoverify: * *structbpf_num_iterit; *intarr[10]; *inti=0,a=0; *bpf_iter_num_new(&it,0,10); *while(bpf_iter_num_next(&it)){ *if(a==0){ *a=1; *i=7;// Because i changed verifier would forget *// it's range on second loop entry. *}else{ *arr[i]=42;// This would fail to verify. *} *} *bpf_iter_num_destroy(&it);
*/ staticint process_iter_next_call(struct bpf_verifier_env *env, int insn_idx, struct bpf_kfunc_call_arg_meta *meta)
{ struct bpf_verifier_state *cur_st = env->cur_state, *queued_st, *prev_st; struct bpf_func_state *cur_fr = cur_st->frame[cur_st->curframe], *queued_fr; struct bpf_reg_state *cur_iter, *queued_iter;
BTF_TYPE_EMIT(struct bpf_iter);
cur_iter = get_iter_from_state(cur_st, meta);
if (cur_iter->iter.state != BPF_ITER_STATE_ACTIVE &&
cur_iter->iter.state != BPF_ITER_STATE_DRAINED) {
verifier_bug(env, "unexpected iterator state %d (%s)",
cur_iter->iter.state, iter_state_str(cur_iter->iter.state)); return -EFAULT;
}
if (cur_iter->iter.state == BPF_ITER_STATE_ACTIVE) { /* Because iter_next() call is a checkpoint is_state_visitied() *shouldguaranteeparentstatewithsamecallsitesandinsn_idx.
*/ if (!cur_st->parent || cur_st->parent->insn_idx != insn_idx ||
!same_callsites(cur_st->parent, cur_st)) {
verifier_bug(env, "bad parent state for iter next call"); return -EFAULT;
} /* Note cur_st->parent in the call below, it is necessary to skip *checkpointcreatedforcur_stbyis_state_visited() *rightatthisinstruction.
*/
prev_st = find_prev_entry(env, cur_st->parent, insn_idx); /* branch out active iter state */
queued_st = push_stack(env, insn_idx + 1, insn_idx, false); if (!queued_st) return -ENOMEM;
/* switch to DRAINED state, but keep the depth unchanged */ /* mark current iter state as drained and assume returned NULL */
cur_iter->iter.state = BPF_ITER_STATE_DRAINED;
__mark_reg_const_zero(env, &cur_fr->regs[BPF_REG_0]);
return0;
}
staticbool arg_type_is_mem_size(enum bpf_arg_type type)
{ return type == ARG_CONST_SIZE ||
type == ARG_CONST_SIZE_OR_ZERO;
}
compatible = compatible_reg_types[base_type(arg_type)]; if (!compatible) {
verifier_bug(env, "unsupported arg type %d", arg_type); return -EFAULT;
}
/* ARG_PTR_TO_MEM + RDONLY is compatible with PTR_TO_MEM and PTR_TO_MEM + RDONLY, *butARG_PTR_TO_MEMiscompatibleonlywithPTR_TO_MEMandNOTwithPTR_TO_MEM+RDONLY * *SameforMAYBE_NULL: * *ARG_PTR_TO_MEM+MAYBE_NULLiscompatiblewithPTR_TO_MEMandPTR_TO_MEM+MAYBE_NULL, *butARG_PTR_TO_MEMiscompatibleonlywithPTR_TO_MEMbutNOTwithPTR_TO_MEM+MAYBE_NULL * *ARG_PTR_TO_MEMiscompatiblewithPTR_TO_MEMthatistaggedwithadynptrtype. * *Thereforewefoldtheseflagsdependingonthearg_typebeforecomparison.
*/ if (arg_type & MEM_RDONLY)
type &= ~MEM_RDONLY; if (arg_type & PTR_MAYBE_NULL)
type &= ~PTR_MAYBE_NULL; if (base_type(arg_type) == ARG_PTR_TO_MEM)
type &= ~DYNPTR_TYPE_FLAG_MASK;
/* Local kptr types are allowed as the source argument of bpf_kptr_xchg */ if (meta->func_id == BPF_FUNC_kptr_xchg && type_is_alloc(type) && regno == BPF_REG_2) {
type &= ~MEM_ALLOC;
type &= ~MEM_PERCPU;
}
for (i = 0; i < ARRAY_SIZE(compatible->types); i++) {
expected = compatible->types[i]; if (expected == NOT_INIT) break;
found: if (base_type(reg->type) != PTR_TO_BTF_ID) return0;
if (compatible == &mem_types) { if (!(arg_type & MEM_RDONLY)) {
verbose(env, "%s() may write into memory pointed by R%d type=%s\n",
func_id_name(meta->func_id),
regno, reg_type_str(env, reg->type)); return -EACCES;
} return0;
}
switch ((int)reg->type) { case PTR_TO_BTF_ID: case PTR_TO_BTF_ID | PTR_TRUSTED: case PTR_TO_BTF_ID | PTR_TRUSTED | PTR_MAYBE_NULL: case PTR_TO_BTF_ID | MEM_RCU: case PTR_TO_BTF_ID | PTR_MAYBE_NULL: case PTR_TO_BTF_ID | PTR_MAYBE_NULL | MEM_RCU:
{ /* For bpf_sk_release, it needs to match against first member *'structsock_common',hencemakeanexceptionforit.This *allowsbpf_sk_releasetoworkformultiplesockettypes.
*/ bool strict_type_match = arg_type_is_release(arg_type) &&
meta->func_id != BPF_FUNC_sk_release;
if (type_may_be_null(reg->type) &&
(!type_may_be_null(arg_type) || arg_type_is_release(arg_type))) {
verbose(env, "Possibly NULL pointer passed to helper arg%d\n", regno); return -EACCES;
}
if (!arg_btf_id) { if (!compatible->btf_id) {
verifier_bug(env, "missing arg compatible BTF ID"); return -EFAULT;
}
arg_btf_id = compatible->btf_id;
}
if (meta->func_id == BPF_FUNC_kptr_xchg) { if (map_kptr_match_type(env, meta->kptr_field, reg, regno)) return -EACCES;
} else { if (arg_btf_id == BPF_PTR_POISON) {
verbose(env, "verifier internal error:");
verbose(env, "R%d has non-overwritten BPF_PTR_POISON type\n",
regno); return -EACCES;
}
if (!btf_struct_ids_match(&env->log, reg->btf, reg->btf_id, reg->off,
btf_vmlinux, *arg_btf_id,
strict_type_match)) {
verbose(env, "R%d is of type %s but %s is expected\n",
regno, btf_type_name(reg->btf, reg->btf_id),
btf_type_name(btf_vmlinux, *arg_btf_id)); return -EACCES;
}
} break;
} case PTR_TO_BTF_ID | MEM_ALLOC: case PTR_TO_BTF_ID | MEM_PERCPU | MEM_ALLOC: if (meta->func_id != BPF_FUNC_spin_lock && meta->func_id != BPF_FUNC_spin_unlock &&
meta->func_id != BPF_FUNC_kptr_xchg) {
verifier_bug(env, "unimplemented handling of MEM_ALLOC"); return -EFAULT;
} /* Check if local kptr in src arg matches kptr in dst arg */ if (meta->func_id == BPF_FUNC_kptr_xchg && regno == BPF_REG_2) { if (map_kptr_match_type(env, meta->kptr_field, reg, regno)) return -EACCES;
} break; case PTR_TO_BTF_ID | MEM_PERCPU: case PTR_TO_BTF_ID | MEM_PERCPU | MEM_RCU: case PTR_TO_BTF_ID | MEM_PERCPU | PTR_TRUSTED: /* Handled by helper specific checks */ break; default:
verifier_bug(env, "invalid PTR_TO_BTF_ID register for type match"); return -EFAULT;
} return0;
}
field = btf_record_find(rec, off, fields); if (!field) return NULL;
return field;
}
staticint check_func_arg_reg_off(struct bpf_verifier_env *env, conststruct bpf_reg_state *reg, int regno, enum bpf_arg_type arg_type)
{
u32 type = reg->type;
/* When referenced register is passed to release function, its fixed *offsetmustbe0. * *Wewillcheckarg_type_is_releasereghasref_obj_idwhenstoring *meta->release_regno.
*/ if (arg_type_is_release(arg_type)) { /* ARG_PTR_TO_DYNPTR with OBJ_RELEASE is a bit special, as it *maynotdirectlypointtotheobjectbeingreleased,butto *dynptrpointingtosuchobject,whichmightbeatsomeoffset *onthestack.Inthatcase,wesimplytofallbacktothe *defaulthandling.
*/ if (arg_type_is_dynptr(arg_type) && type == PTR_TO_STACK) return0;
/* Doing check_ptr_off_reg check for the offset will catch this *becausefixed_off_okisfalse,butcheckinghereallowsus *togivetheuserabettererrormessage.
*/ if (reg->off) {
verbose(env, "R%d must have zero offset when passed to release func or trusted arg to kfunc\n",
regno); return -EINVAL;
} return __check_ptr_off_reg(env, reg, regno, false);
}
switch (type) { /* Pointer types where both fixed and variable offset is explicitly allowed: */ case PTR_TO_STACK: case PTR_TO_PACKET: case PTR_TO_PACKET_META: case PTR_TO_MAP_KEY: case PTR_TO_MAP_VALUE: case PTR_TO_MEM: case PTR_TO_MEM | MEM_RDONLY: case PTR_TO_MEM | MEM_RINGBUF: case PTR_TO_BUF: case PTR_TO_BUF | MEM_RDONLY: case PTR_TO_ARENA: case SCALAR_VALUE: return0; /* All the rest must be rejected, except PTR_TO_BTF_ID which allows *fixedoffset.
*/ case PTR_TO_BTF_ID: case PTR_TO_BTF_ID | MEM_ALLOC: case PTR_TO_BTF_ID | PTR_TRUSTED: case PTR_TO_BTF_ID | MEM_RCU: case PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF: case PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF | MEM_RCU: /* When referenced PTR_TO_BTF_ID is passed to release function, *itsfixedoffsetmustbe0.Intheothercases,fixedoffset *canbenon-zero.Thiswasalreadycheckedabove.Sopass *fixed_off_okastruetoallowfixedoffsetforallother *cases.var_offalwaysmustbe0forPTR_TO_BTF_ID,hencewe *stillneedtodochecksinsteadofreturning.
*/ return __check_ptr_off_reg(env, reg, regno, true); default: return __check_ptr_off_reg(env, reg, regno, false);
}
}
for (i = 0; i < MAX_BPF_FUNC_REG_ARGS; i++) if (arg_type_is_dynptr(fn->arg_type[i])) { if (state) {
verbose(env, "verifier internal error: multiple dynptr args\n"); return NULL;
}
state = ®s[BPF_REG_1 + i];
}
if (!state)
verbose(env, "verifier internal error: no dynptr arg found\n");
/* First handle precisely tracked STACK_ZERO */ for (i = off; i >= 0 && stype[i] == STACK_ZERO; i--)
zero_size++; if (zero_size >= key_size) {
*value = 0; return0;
}
/* Check that stack contains a scalar spill of expected size */ if (!is_spilled_scalar_reg(&state->stack[spi])) return -EOPNOTSUPP; for (i = off; i >= 0 && stype[i] == STACK_SPILL; i--)
spill_size++; if (spill_size != key_size) return -EOPNOTSUPP;
reg = &state->stack[spi].spilled_ptr; if (!tnum_is_const(reg->var_off)) /* Stack value not statically known */ return -EOPNOTSUPP;
/* We are relying on a constant value. So mark as precise *topreventpruningonit.
*/
bt_set_frame_slot(&env->bt, key->frameno, spi);
err = mark_chain_precision_batch(env, env->cur_state); if (err < 0) return err;
err = check_reg_arg(env, regno, SRC_OP); if (err) return err;
if (arg_type == ARG_ANYTHING) { if (is_pointer_value(env, regno)) {
verbose(env, "R%d leaks addr into helper function\n",
regno); return -EACCES;
} return0;
}
if (type_is_pkt_pointer(type) &&
!may_access_direct_pkt_data(env, meta, BPF_READ)) {
verbose(env, "helper access to the packet is not allowed\n"); return -EACCES;
}
if (base_type(arg_type) == ARG_PTR_TO_MAP_VALUE) {
err = resolve_map_arg_type(env, meta, &arg_type); if (err) return err;
}
if (register_is_null(reg) && type_may_be_null(arg_type)) /* A NULL register has a SCALAR_VALUE type, so skip *typechecking.
*/ goto skip_type_check;
/* arg_btf_id and arg_size are in a union. */ if (base_type(arg_type) == ARG_PTR_TO_BTF_ID ||
base_type(arg_type) == ARG_PTR_TO_SPIN_LOCK)
arg_btf_id = fn->arg_btf_id[arg];
err = check_func_arg_reg_off(env, reg, regno, arg_type); if (err) return err;
skip_type_check: if (arg_type_is_release(arg_type)) { if (arg_type_is_dynptr(arg_type)) { struct bpf_func_state *state = func(env, reg); int spi;
/* Only dynptr created on stack can be released, thus *theget_spiandstackstatechecksforspilled_ptr *shouldonlybedonebeforeprocess_dynptr_funcfor *PTR_TO_STACK.
*/ if (reg->type == PTR_TO_STACK) {
spi = dynptr_get_spi(env, reg); if (spi < 0 || !state->stack[spi].spilled_ptr.ref_obj_id) {
verbose(env, "arg %d is an unacquired reference\n", regno); return -EINVAL;
}
} else {
verbose(env, "cannot release unowned const bpf_dynptr\n"); return -EINVAL;
}
} elseif (!reg->ref_obj_id && !register_is_null(reg)) {
verbose(env, "R%d must be referenced when passed to release function\n",
regno); return -EINVAL;
} if (meta->release_regno) {
verifier_bug(env, "more than one release argument"); return -EFAULT;
}
meta->release_regno = regno;
}
if (reg->ref_obj_id && base_type(arg_type) != ARG_KPTR_XCHG_DEST) { if (meta->ref_obj_id) {
verbose(env, "more than one arg with ref_obj_id R%d %u %u",
regno, reg->ref_obj_id,
meta->ref_obj_id); return -EACCES;
}
meta->ref_obj_id = reg->ref_obj_id;
}
switch (base_type(arg_type)) { case ARG_CONST_MAP_PTR: /* bpf_map_xxx(map_ptr) call: remember that map_ptr */ if (meta->map_ptr) { /* Use map_uid (which is unique id of inner map) to reject: *inner_map1=bpf_map_lookup_elem(outer_map,key1) *inner_map2=bpf_map_lookup_elem(outer_map,key2) *if(inner_map1&&inner_map2){ *timer=bpf_map_lookup_elem(inner_map1); *if(timer) *// mismatch would have been allowed *bpf_timer_init(timer,inner_map2); *} * *Comparingmap_ptrisenoughtodistinguishnormalandoutermaps.
*/ if (meta->map_ptr != reg->map_ptr ||
meta->map_uid != reg->map_uid) {
verbose(env, "timer pointer in R1 map_uid=%d doesn't match map pointer in R2 map_uid=%d\n",
meta->map_uid, reg->map_uid); return -EINVAL;
}
}
meta->map_ptr = reg->map_ptr;
meta->map_uid = reg->map_uid; break; case ARG_PTR_TO_MAP_KEY: /* bpf_map_xxx(..., map_ptr, ..., key) call: *checkthat[key,key+map->key_size)arewithin *stacklimitsandinitialized
*/ if (!meta->map_ptr) { /* in function declaration map_ptr must come before *map_key,sothatit'sverifiedandknownbefore *wehavetocheckmap_keyhere.Otherwiseitmeans *thatkernelsubsystemmisconfiguredverifier
*/
verifier_bug(env, "invalid map_ptr to access map->key"); return -EFAULT;
}
key_size = meta->map_ptr->key_size;
err = check_helper_mem_access(env, regno, key_size, BPF_READ, false, NULL); if (err) return err; if (can_elide_value_nullness(meta->map_ptr->map_type)) {
err = get_constant_map_key(env, reg, key_size, &meta->const_map_key); if (err < 0) {
meta->const_map_key = -1; if (err == -EOPNOTSUPP)
err = 0; else return err;
}
} break; case ARG_PTR_TO_MAP_VALUE: if (type_may_be_null(arg_type) && register_is_null(reg)) return0;
/* bpf_map_xxx(..., map_ptr, ..., value) call: *check[value,value+map->value_size)validity
*/ if (!meta->map_ptr) { /* kernel subsystem misconfigured verifier */
verifier_bug(env, "invalid map_ptr to access map->value"); return -EFAULT;
}
meta->raw_mode = arg_type & MEM_UNINIT;
err = check_helper_mem_access(env, regno, meta->map_ptr->value_size,
arg_type & MEM_WRITE ? BPF_WRITE : BPF_READ, false, meta); break; case ARG_PTR_TO_PERCPU_BTF_ID: if (!reg->btf_id) {
verbose(env, "Helper has invalid btf_id in R%d\n", regno); return -EACCES;
}
meta->ret_btf = reg->btf;
meta->ret_btf_id = reg->btf_id; break; case ARG_PTR_TO_SPIN_LOCK: if (in_rbtree_lock_required_cb(env)) {
verbose(env, "can't spin_{lock,unlock} in rbtree cb\n"); return -EACCES;
} if (meta->func_id == BPF_FUNC_spin_lock) {
err = process_spin_lock(env, regno, PROCESS_SPIN_LOCK); if (err) return err;
} elseif (meta->func_id == BPF_FUNC_spin_unlock) {
err = process_spin_lock(env, regno, 0); if (err) return err;
} else {
verifier_bug(env, "spin lock arg on unexpected helper"); return -EFAULT;
} break; case ARG_PTR_TO_TIMER:
err = process_timer_func(env, regno, meta); if (err) return err; break; case ARG_PTR_TO_FUNC:
meta->subprogno = reg->subprogno; break; case ARG_PTR_TO_MEM: /* The access to this pointer is only checked when we hit the *nextis_mem_sizeargumentbelow.
*/
meta->raw_mode = arg_type & MEM_UNINIT; if (arg_type & MEM_FIXED_SIZE) {
err = check_helper_mem_access(env, regno, fn->arg_size[arg],
arg_type & MEM_WRITE ? BPF_WRITE : BPF_READ, false, meta); if (err) return err; if (arg_type & MEM_ALIGNED)
err = check_ptr_alignment(env, reg, 0, fn->arg_size[arg], true);
} break; case ARG_CONST_SIZE:
err = check_mem_size_reg(env, reg, regno,
fn->arg_type[arg - 1] & MEM_WRITE ?
BPF_WRITE : BPF_READ, false, meta); break; case ARG_CONST_SIZE_OR_ZERO:
err = check_mem_size_reg(env, reg, regno,
fn->arg_type[arg - 1] & MEM_WRITE ?
BPF_WRITE : BPF_READ, true, meta); break; case ARG_PTR_TO_DYNPTR:
err = process_dynptr_func(env, regno, insn_idx, arg_type, 0); if (err) return err; break; case ARG_CONST_ALLOC_SIZE_OR_ZERO: if (!tnum_is_const(reg->var_off)) {
verbose(env, "R%d is not a known constant'\n",
regno); return -EACCES;
}
meta->mem_size = reg->var_off.value;
err = mark_chain_precision(env, regno); if (err) return err; break; case ARG_PTR_TO_CONST_STR:
{
err = check_reg_const_str(env, reg, regno); if (err) return err; break;
} case ARG_KPTR_XCHG_DEST:
err = process_kptr_func(env, regno, meta); if (err) return err; break;
}
return err;
}
staticbool may_update_sockmap(struct bpf_verifier_env *env, int func_id)
{ enum bpf_attach_type eatype = env->prog->expected_attach_type; enum bpf_prog_type type = resolve_prog_type(env->prog);
if (func_id != BPF_FUNC_map_update_elem &&
func_id != BPF_FUNC_map_delete_elem) returnfalse;
/* It's not possible to get access to a locked struct sock in these *contexts,soupdatingissafe.
*/ switch (type) { case BPF_PROG_TYPE_TRACING: if (eatype == BPF_TRACE_ITER) returntrue; break; case BPF_PROG_TYPE_SOCK_OPS: /* map_update allowed only via dedicated helpers with event type checks */ if (func_id == BPF_FUNC_map_delete_elem) returntrue; break; case BPF_PROG_TYPE_SOCKET_FILTER: case BPF_PROG_TYPE_SCHED_CLS: case BPF_PROG_TYPE_SCHED_ACT: case BPF_PROG_TYPE_XDP: case BPF_PROG_TYPE_SK_REUSEPORT: case BPF_PROG_TYPE_FLOW_DISSECTOR: case BPF_PROG_TYPE_SK_LOOKUP: returntrue; default: break;
}
verbose(env, "cannot update sockmap in this context\n"); returnfalse;
}
staticint check_map_func_compatibility(struct bpf_verifier_env *env, struct bpf_map *map, int func_id)
{ if (!map) return0;
/* We need a two way check, first is from map perspective ... */ switch (map->map_type) { case BPF_MAP_TYPE_PROG_ARRAY: if (func_id != BPF_FUNC_tail_call) goto error; break; case BPF_MAP_TYPE_PERF_EVENT_ARRAY: if (func_id != BPF_FUNC_perf_event_read &&
func_id != BPF_FUNC_perf_event_output &&
func_id != BPF_FUNC_skb_output &&
func_id != BPF_FUNC_perf_event_read_value &&
func_id != BPF_FUNC_xdp_output) goto error; break; case BPF_MAP_TYPE_RINGBUF: if (func_id != BPF_FUNC_ringbuf_output &&
func_id != BPF_FUNC_ringbuf_reserve &&
func_id != BPF_FUNC_ringbuf_query &&
func_id != BPF_FUNC_ringbuf_reserve_dynptr &&
func_id != BPF_FUNC_ringbuf_submit_dynptr &&
func_id != BPF_FUNC_ringbuf_discard_dynptr) goto error; break; case BPF_MAP_TYPE_USER_RINGBUF: if (func_id != BPF_FUNC_user_ringbuf_drain) goto error; break; case BPF_MAP_TYPE_STACK_TRACE: if (func_id != BPF_FUNC_get_stackid) goto error; break; case BPF_MAP_TYPE_CGROUP_ARRAY: if (func_id != BPF_FUNC_skb_under_cgroup &&
func_id != BPF_FUNC_current_task_under_cgroup) goto error; break; case BPF_MAP_TYPE_CGROUP_STORAGE: case BPF_MAP_TYPE_PERCPU_CGROUP_STORAGE: if (func_id != BPF_FUNC_get_local_storage) goto error; break; case BPF_MAP_TYPE_DEVMAP: case BPF_MAP_TYPE_DEVMAP_HASH: if (func_id != BPF_FUNC_redirect_map &&
func_id != BPF_FUNC_map_lookup_elem) goto error; break; /* Restrict bpf side of cpumap and xskmap, open when use-cases *appear.
*/ case BPF_MAP_TYPE_CPUMAP: if (func_id != BPF_FUNC_redirect_map) goto error; break; case BPF_MAP_TYPE_XSKMAP: if (func_id != BPF_FUNC_redirect_map &&
func_id != BPF_FUNC_map_lookup_elem) goto error; break; case BPF_MAP_TYPE_ARRAY_OF_MAPS: case BPF_MAP_TYPE_HASH_OF_MAPS: if (func_id != BPF_FUNC_map_lookup_elem) goto error; break; case BPF_MAP_TYPE_SOCKMAP: if (func_id != BPF_FUNC_sk_redirect_map &&
func_id != BPF_FUNC_sock_map_update &&
func_id != BPF_FUNC_msg_redirect_map &&
func_id != BPF_FUNC_sk_select_reuseport &&
func_id != BPF_FUNC_map_lookup_elem &&
!may_update_sockmap(env, func_id)) goto error; break; case BPF_MAP_TYPE_SOCKHASH: if (func_id != BPF_FUNC_sk_redirect_hash &&
func_id != BPF_FUNC_sock_hash_update &&
func_id != BPF_FUNC_msg_redirect_hash &&
func_id != BPF_FUNC_sk_select_reuseport &&
func_id != BPF_FUNC_map_lookup_elem &&
!may_update_sockmap(env, func_id)) goto error; break; case BPF_MAP_TYPE_REUSEPORT_SOCKARRAY: if (func_id != BPF_FUNC_sk_select_reuseport) goto error; break; case BPF_MAP_TYPE_QUEUE: case BPF_MAP_TYPE_STACK: if (func_id != BPF_FUNC_map_peek_elem &&
func_id != BPF_FUNC_map_pop_elem &&
func_id != BPF_FUNC_map_push_elem) goto error; break; case BPF_MAP_TYPE_SK_STORAGE: if (func_id != BPF_FUNC_sk_storage_get &&
func_id != BPF_FUNC_sk_storage_delete &&
func_id != BPF_FUNC_kptr_xchg) goto error; break; case BPF_MAP_TYPE_INODE_STORAGE: if (func_id != BPF_FUNC_inode_storage_get &&
func_id != BPF_FUNC_inode_storage_delete &&
func_id != BPF_FUNC_kptr_xchg) goto error; break; case BPF_MAP_TYPE_TASK_STORAGE: if (func_id != BPF_FUNC_task_storage_get &&
func_id != BPF_FUNC_task_storage_delete &&
func_id != BPF_FUNC_kptr_xchg) goto error; break; case BPF_MAP_TYPE_CGRP_STORAGE: if (func_id != BPF_FUNC_cgrp_storage_get &&
func_id != BPF_FUNC_cgrp_storage_delete &&
func_id != BPF_FUNC_kptr_xchg) goto error; break; case BPF_MAP_TYPE_BLOOM_FILTER: if (func_id != BPF_FUNC_map_peek_elem &&
func_id != BPF_FUNC_map_push_elem) goto error; break; default: break;
}
/* ... and second from the function itself. */ switch (func_id) { case BPF_FUNC_tail_call: if (map->map_type != BPF_MAP_TYPE_PROG_ARRAY) goto error; if (env->subprog_cnt > 1 && !allow_tail_call_in_subprogs(env)) {
verbose(env, "mixing of tail_calls and bpf-to-bpf calls is not supported\n"); return -EINVAL;
} break; case BPF_FUNC_perf_event_read: case BPF_FUNC_perf_event_output: case BPF_FUNC_perf_event_read_value: case BPF_FUNC_skb_output: case BPF_FUNC_xdp_output: if (map->map_type != BPF_MAP_TYPE_PERF_EVENT_ARRAY) goto error; break; case BPF_FUNC_ringbuf_output: case BPF_FUNC_ringbuf_reserve: case BPF_FUNC_ringbuf_query: case BPF_FUNC_ringbuf_reserve_dynptr: case BPF_FUNC_ringbuf_submit_dynptr: case BPF_FUNC_ringbuf_discard_dynptr: if (map->map_type != BPF_MAP_TYPE_RINGBUF) goto error; break; case BPF_FUNC_user_ringbuf_drain: if (map->map_type != BPF_MAP_TYPE_USER_RINGBUF) goto error; break; case BPF_FUNC_get_stackid: if (map->map_type != BPF_MAP_TYPE_STACK_TRACE) goto error; break; case BPF_FUNC_current_task_under_cgroup: case BPF_FUNC_skb_under_cgroup: if (map->map_type != BPF_MAP_TYPE_CGROUP_ARRAY) goto error; break; case BPF_FUNC_redirect_map: if (map->map_type != BPF_MAP_TYPE_DEVMAP &&
map->map_type != BPF_MAP_TYPE_DEVMAP_HASH &&
map->map_type != BPF_MAP_TYPE_CPUMAP &&
map->map_type != BPF_MAP_TYPE_XSKMAP) goto error; break; case BPF_FUNC_sk_redirect_map: case BPF_FUNC_msg_redirect_map: case BPF_FUNC_sock_map_update: if (map->map_type != BPF_MAP_TYPE_SOCKMAP) goto error; break; case BPF_FUNC_sk_redirect_hash: case BPF_FUNC_msg_redirect_hash: case BPF_FUNC_sock_hash_update: if (map->map_type != BPF_MAP_TYPE_SOCKHASH) goto error; break; case BPF_FUNC_get_local_storage: if (map->map_type != BPF_MAP_TYPE_CGROUP_STORAGE &&
map->map_type != BPF_MAP_TYPE_PERCPU_CGROUP_STORAGE) goto error; break; case BPF_FUNC_sk_select_reuseport: if (map->map_type != BPF_MAP_TYPE_REUSEPORT_SOCKARRAY &&
map->map_type != BPF_MAP_TYPE_SOCKMAP &&
map->map_type != BPF_MAP_TYPE_SOCKHASH) goto error; break; case BPF_FUNC_map_pop_elem: if (map->map_type != BPF_MAP_TYPE_QUEUE &&
map->map_type != BPF_MAP_TYPE_STACK) goto error; break; case BPF_FUNC_map_peek_elem: case BPF_FUNC_map_push_elem: if (map->map_type != BPF_MAP_TYPE_QUEUE &&
map->map_type != BPF_MAP_TYPE_STACK &&
map->map_type != BPF_MAP_TYPE_BLOOM_FILTER) goto error; break; case BPF_FUNC_map_lookup_percpu_elem: if (map->map_type != BPF_MAP_TYPE_PERCPU_ARRAY &&
map->map_type != BPF_MAP_TYPE_PERCPU_HASH &&
map->map_type != BPF_MAP_TYPE_LRU_PERCPU_HASH) goto error; break; case BPF_FUNC_sk_storage_get: case BPF_FUNC_sk_storage_delete: if (map->map_type != BPF_MAP_TYPE_SK_STORAGE) goto error; break; case BPF_FUNC_inode_storage_get: case BPF_FUNC_inode_storage_delete: if (map->map_type != BPF_MAP_TYPE_INODE_STORAGE) goto error; break; case BPF_FUNC_task_storage_get: case BPF_FUNC_task_storage_delete: if (map->map_type != BPF_MAP_TYPE_TASK_STORAGE) goto error; break; case BPF_FUNC_cgrp_storage_get: case BPF_FUNC_cgrp_storage_delete: if (map->map_type != BPF_MAP_TYPE_CGRP_STORAGE) goto error; break; default: break;
}
staticbool check_raw_mode_ok(conststruct bpf_func_proto *fn)
{ int count = 0;
if (arg_type_is_raw_mem(fn->arg1_type))
count++; if (arg_type_is_raw_mem(fn->arg2_type))
count++; if (arg_type_is_raw_mem(fn->arg3_type))
count++; if (arg_type_is_raw_mem(fn->arg4_type))
count++; if (arg_type_is_raw_mem(fn->arg5_type))
count++;
/* We only support one arg being in raw mode at the moment, *whichissufficientforthehelperfunctionswehave *rightnow.
*/ return count <= 1;
}
staticbool check_btf_id_ok(conststruct bpf_func_proto *fn)
{ int i;
for (i = 0; i < ARRAY_SIZE(fn->arg_type); i++) { if (base_type(fn->arg_type[i]) == ARG_PTR_TO_BTF_ID) return !!fn->arg_btf_id[i]; if (base_type(fn->arg_type[i]) == ARG_PTR_TO_SPIN_LOCK) return fn->arg_btf_id[i] == BPF_PTR_POISON; if (base_type(fn->arg_type[i]) != ARG_PTR_TO_BTF_ID && fn->arg_btf_id[i] && /* arg_btf_id and arg_size are in a union. */
(base_type(fn->arg_type[i]) != ARG_PTR_TO_MEM ||
!(fn->arg_type[i] & MEM_FIXED_SIZE))) returnfalse;
}
if (reg->type != PTR_TO_PACKET) /* PTR_TO_PACKET_META is not supported yet */ return;
/* The 'reg' is pkt > pkt_end or pkt >= pkt_end. *Howfarbeyondpkt_enditgoesisunknown. *if(!range_open)it'sthecaseofpkt>=pkt_end *if(range_open)it'sthecaseofpkt>pkt_end *hencethispointerisatleast1bytebiggerthanpkt_end
*/ if (range_open)
reg->range = BEYOND_PKT_END; else
reg->range = AT_PKT_END;
}
staticint release_reference_nomark(struct bpf_verifier_state *state, int ref_obj_id)
{ int i;
for (i = 0; i < state->acquired_refs; i++) { if (state->refs[i].type != REF_TYPE_PTR) continue; if (state->refs[i].id == ref_obj_id) {
release_reference_state(state, i); return0;
}
} return -EINVAL;
}
/* The pointer with the specified id has released its reference to kernel *resources.Identifyallcopiesofthesamepointerandclearthereference. * *Thisisthereleasefunctioncorrespondingtoacquire_reference().Idempotent.
*/ staticint release_reference(struct bpf_verifier_env *env, int ref_obj_id)
{ struct bpf_verifier_state *vstate = env->cur_state; struct bpf_func_state *state; struct bpf_reg_state *reg; int err;
err = release_reference_nomark(vstate, ref_obj_id); if (err) return err;
bpf_for_each_reg_in_vstate(env->cur_state, unused, reg, ({ if (type_is_non_owning_ref(reg->type))
mark_reg_invalid(env, reg);
}));
}
staticvoid clear_caller_saved_regs(struct bpf_verifier_env *env, struct bpf_reg_state *regs)
{ int i;
/* after the call registers r0 - r5 were scratched */ for (i = 0; i < CALLER_SAVED_REGS; i++) {
mark_reg_not_init(env, regs, caller_saved[i]);
__check_reg_arg(env, regs, caller_saved[i], DST_OP_NO_MARK);
}
}
/* callee cannot access r0, r6 - r9 for reading and has to write *intoitsownstackbeforereadingfromit. *calleecanread/writeintocaller'sstack
*/
init_func_state(env, callee, /* remember the callsite, it will be used by bpf_exit */
callsite,
state->curframe + 1/* frameno within this callchain */,
subprog /* subprog number within this prog */);
err = set_callee_state_cb(env, caller, callee, callsite); if (err) goto err_out;
/* only increment it after check_reg_arg() finished */
state->curframe++;
ret = btf_prepare_func_args(env, subprog); if (ret) return ret;
/* check that BTF function arguments match actual types that the *verifiersees.
*/ for (i = 0; i < sub->arg_cnt; i++) {
u32 regno = i + 1; struct bpf_reg_state *reg = ®s[regno]; struct bpf_subprog_arg_info *arg = &sub->args[i];
if (arg->arg_type == ARG_ANYTHING) { if (reg->type != SCALAR_VALUE) {
bpf_log(log, "R%d is not a scalar\n", regno); return -EINVAL;
}
} elseif (arg->arg_type & PTR_UNTRUSTED) { /* *Anythingisallowedforuntrustedarguments,astheseare *read-onlyandprobereadinstructionswouldprotectagainst *invalidmemoryaccess.
*/
} elseif (arg->arg_type == ARG_PTR_TO_CTX) {
ret = check_func_arg_reg_off(env, reg, regno, ARG_DONTCARE); if (ret < 0) return ret; /* If function expects ctx type in BTF check that caller *ispassingPTR_TO_CTX.
*/ if (reg->type != PTR_TO_CTX) {
bpf_log(log, "arg#%d expects pointer to ctx\n", i); return -EINVAL;
}
} elseif (base_type(arg->arg_type) == ARG_PTR_TO_MEM) {
ret = check_func_arg_reg_off(env, reg, regno, ARG_DONTCARE); if (ret < 0) return ret; if (check_mem_reg(env, reg, regno, arg->mem_size)) return -EINVAL; if (!(arg->arg_type & PTR_MAYBE_NULL) && (reg->type & PTR_MAYBE_NULL)) {
bpf_log(log, "arg#%d is expected to be non-NULL\n", i); return -EINVAL;
}
} elseif (base_type(arg->arg_type) == ARG_PTR_TO_ARENA) { /* *Canpassanyvalueandthekernelwon'tcrash,but *onlyPTR_TO_ARENAorSCALARmakesense.Everything *elseisabuginthebpfprogram.Pointitoutto *theuserattheverificationtimeinsteadof *run-timedebugnightmare.
*/ if (reg->type != PTR_TO_ARENA && reg->type != SCALAR_VALUE) {
bpf_log(log, "R%d is not a pointer to arena or scalar.\n", regno); return -EINVAL;
}
} elseif (arg->arg_type == (ARG_PTR_TO_DYNPTR | MEM_RDONLY)) {
ret = check_func_arg_reg_off(env, reg, regno, ARG_PTR_TO_DYNPTR); if (ret) return ret;
ret = process_dynptr_func(env, regno, -1, arg->arg_type, 0); if (ret) return ret;
} elseif (base_type(arg->arg_type) == ARG_PTR_TO_BTF_ID) { struct bpf_call_arg_meta meta; int err;
if (register_is_null(reg) && type_may_be_null(arg->arg_type)) continue;
memset(&meta, 0, sizeof(meta)); /* leave func_id as zero */
err = check_reg_type(env, regno, arg->arg_type, &arg->btf_id, &meta);
err = err ?: check_func_arg_reg_off(env, reg, regno, arg->arg_type); if (err) return err;
} else {
verifier_bug(env, "unrecognized arg#%d type %d", i, arg->arg_type); return -EFAULT;
}
}
return0;
}
/* Compare BTF of a function call with given bpf_reg_state. *Returns: *EFAULT-thereisaverifierbug.Abortverification. *EINVAL-thereisatypemismatchorBTFisnotavailable. *0-BTFmatcheswithwhatbpf_reg_stateexpects. *OnlyPTR_TO_CTXandSCALAR_VALUEstatesarerecognized.
*/ staticint btf_check_subprog_call(struct bpf_verifier_env *env, int subprog, struct bpf_reg_state *regs)
{ struct bpf_prog *prog = env->prog; struct btf *btf = prog->aux->btf;
u32 btf_id; int err;
if (!prog->aux->func_info) return -EINVAL;
btf_id = prog->aux->func_info[subprog].type_id; if (!btf_id) return -EFAULT;
if (prog->aux->func_info_aux[subprog].unreliable) return -EINVAL;
err = btf_check_func_arg_match(env, subprog, btf, regs); /* Compiler optimizations can remove arguments from static functions *ormismatchedtypecanbepassedintoaglobalfunction. *InsuchcasesmarkthefunctionasunreliablefromBTFpointofview.
*/ if (err)
prog->aux->func_info_aux[subprog].unreliable = true; return err;
}
staticint push_callback_call(struct bpf_verifier_env *env, struct bpf_insn *insn, int insn_idx, int subprog,
set_callee_state_fn set_callee_state_cb)
{ struct bpf_verifier_state *state = env->cur_state, *callback_state; struct bpf_func_state *caller, *callee; int err;
/* set_callee_state is used for direct subprog calls, but we are *interestedinvalidatingonlyBPFhelpersthatcancallsubprogsas *callbacks
*/
env->subprog_info[subprog].is_cb = true; if (bpf_pseudo_kfunc_call(insn) &&
!is_callback_calling_kfunc(insn->imm)) {
verifier_bug(env, "kfunc %s#%d not marked as callback-calling",
func_id_name(insn->imm), insn->imm); return -EFAULT;
} elseif (!bpf_pseudo_kfunc_call(insn) &&
!is_callback_calling_function(insn->imm)) { /* helper */
verifier_bug(env, "helper %s#%d not marked as callback-calling",
func_id_name(insn->imm), insn->imm); return -EFAULT;
}
if (is_async_callback_calling_insn(insn)) { struct bpf_verifier_state *async_cb;
/* there is no real recursion here. timer and workqueue callbacks are async */
env->subprog_info[subprog].is_async_cb = true;
async_cb = push_async_cb(env, env->subprog_info[subprog].start,
insn_idx, subprog,
is_bpf_wq_set_callback_impl_kfunc(insn->imm)); if (!async_cb) return -EFAULT;
callee = async_cb->frame[0];
callee->async_entry_cnt = caller->async_entry_cnt + 1;
/* Convert bpf_timer_set_callback() args into timer callback args */
err = set_callee_state_cb(env, caller, callee, insn_idx); if (err) return err;
return0;
}
/* for callback functions enqueue entry to callback and *proceedwithnextinstructionwithincurrentframe.
*/
callback_state = push_stack(env, env->subprog_info[subprog].start, insn_idx, false); if (!callback_state) return -ENOMEM;
target_insn = *insn_idx + insn->imm + 1;
subprog = find_subprog(env, target_insn); if (verifier_bug_if(subprog < 0, env, "target of func call at insn %d is not a program",
target_insn)) return -EFAULT;
if (env->cur_state->active_locks) {
verbose(env, "global function calls are not allowed while holding a lock,\n" "use static function instead\n"); return -EINVAL;
}
if (env->subprog_info[subprog].might_sleep &&
(env->cur_state->active_rcu_lock || env->cur_state->active_preempt_locks ||
env->cur_state->active_irq_id || !in_sleepable(env))) {
verbose(env, "global functions that may sleep are not allowed in non-sleepable context,\n" "i.e., in a RCU/IRQ/preempt-disabled section, or in\n" "a non-sleepable BPF program context\n"); return -EINVAL;
}
if (err) {
verbose(env, "Caller passes invalid args into func#%d ('%s')\n",
subprog, sub_name); return err;
}
verbose(env, "Func#%d ('%s') is global and assumed valid.\n",
subprog, sub_name); if (env->subprog_info[subprog].changes_pkt_data)
clear_all_pkt_pointers(env); /* mark global subprog for verifying after main prog */
subprog_aux(env, subprog)->called = true;
clear_caller_saved_regs(env, caller->regs);
/* All global functions return a 64-bit SCALAR_VALUE */
mark_reg_unknown(env, caller->regs, BPF_REG_0);
caller->regs[BPF_REG_0].subreg_def = DEF_NOT_SUBREG;
/* continue with next insn after call */ return0;
}
/* for regular function entry setup new frame and continue *fromthatframe.
*/
err = setup_func_entry(env, subprog, *insn_idx, set_callee_state, state); if (err) return err;
clear_caller_saved_regs(env, caller->regs);
/* and go analyze first insn of the callee */
*insn_idx = env->subprog_info[subprog].start - 1;
staticint set_callee_state(struct bpf_verifier_env *env, struct bpf_func_state *caller, struct bpf_func_state *callee, int insn_idx)
{ int i;
/* copy r1 - r5 args that callee can access. The copy includes parent *pointers,whichconnectsusuptothelivenesschain
*/ for (i = BPF_REG_1; i <= BPF_REG_5; i++)
callee->regs[i] = caller->regs[i]; return0;
}
/* valid map_ptr and poison value does not matter */
map = insn_aux->map_ptr_state.map_ptr; if (!map->ops->map_set_for_each_callback_args ||
!map->ops->map_for_each_callback) {
verbose(env, "callback function not allowed for map\n"); return -ENOTSUPP;
}
err = map->ops->map_set_for_each_callback_args(env, caller, callee); if (err) return err;
/* Are we currently verifying the callback for a rbtree helper that must *becalledwithlockheld?Ifso,noneedtocomplainaboutunreleased *lock
*/ staticbool in_rbtree_lock_required_cb(struct bpf_verifier_env *env)
{ struct bpf_verifier_state *state = env->cur_state; struct bpf_insn *insn = env->prog->insnsi; struct bpf_func_state *callee; int kfunc_btf_id;
callee = state->frame[state->curframe];
r0 = &callee->regs[BPF_REG_0]; if (r0->type == PTR_TO_STACK) { /* technically it's ok to return caller's stack pointer *(orcaller'scaller'spointer)backtothecaller, *sincethesepointersarevalid.Onlycurrentstack *pointerwillbeinvalidassoonasfunctionexits, *butlet'sbeconservative
*/
verbose(env, "cannot return stack pointer to the caller\n"); return -EINVAL;
}
caller = state->frame[state->curframe - 1]; if (callee->in_callback_fn) { if (r0->type != SCALAR_VALUE) {
verbose(env, "R0 not a scalar value\n"); return -EACCES;
}
/* we are going to rely on register's precise value */
err = mark_reg_read(env, r0, r0->parent, REG_LIVE_READ64);
err = err ?: mark_chain_precision(env, BPF_REG_0); if (err) return err;
/* enforce R0 return value range, and bpf_callback_t returns 64bit */ if (!retval_range_within(callee->callback_ret_range, r0, false)) {
verbose_invalid_scalar(env, r0, callee->callback_ret_range, "At callback return", "R0"); return -EINVAL;
} if (!calls_callback(env, callee->callsite)) {
verifier_bug(env, "in callback at %d, callsite %d !calls_callback",
*insn_idx, callee->callsite); return -EFAULT;
}
} else { /* return to the caller whatever r0 had in the callee */
caller->regs[BPF_REG_0] = *r0;
}
/* for callbacks like bpf_loop or bpf_for_each_map_elem go back to callsite, *therefunctioncalllogicwouldreschedulecallbackvisit.Ifiteration *convergesis_state_visited()wouldprunethatvisiteventually.
*/
in_callback_fn = callee->in_callback_fn; if (in_callback_fn)
*insn_idx = callee->callsite; else
*insn_idx = callee->callsite + 1;
if (env->log.level & BPF_LOG_LEVEL) {
verbose(env, "returning from callee:\n");
print_verifier_state(env, state, callee->frameno, true);
verbose(env, "to caller at %d:\n", *insn_idx);
print_verifier_state(env, state, caller->frameno, true);
} /* clear everything in the callee. In case of exceptional exits using
* bpf_throw, this will be done by copy_verifier_state for extra frames. */
free_func_state(callee);
state->frame[state->curframe--] = NULL;
/* for callbacks widen imprecise scalars to make programs like below verify: * *structctx{inti;} *voidcb(intidx,structctx*ctx){ctx->i++;...} *... *structctx={.i=0;} *bpf_loop(100,cb,&ctx,0); * *Thisissimilartowhatisdoneinprocess_iter_next_call()foropen *codediterators.
*/
prev_st = in_callback_fn ? find_prev_entry(env, state, *insn_idx) : NULL; if (prev_st) {
err = widen_imprecise_scalars(env, prev_st, state); if (err) return err;
} return0;
}
staticint do_refine_retval_range(struct bpf_verifier_env *env, struct bpf_reg_state *regs, int ret_type, int func_id, struct bpf_call_arg_meta *meta)
{ struct bpf_reg_state *ret_reg = ®s[BPF_REG_0];
/* data must be an array of u64 */ if (data_len_reg->var_off.value % 8) return -EINVAL;
num_args = data_len_reg->var_off.value / 8;
/* fmt being ARG_PTR_TO_CONST_STR guarantees that var_off is const *andmap_direct_value_addrisset.
*/
fmt_map_off = fmt_reg->off + fmt_reg->var_off.value;
err = fmt_map->ops->map_direct_value_addr(fmt_map, &fmt_addr,
fmt_map_off); if (err) {
verbose(env, "failed to retrieve map value address\n"); return -EFAULT;
}
fmt = (char *)(long)fmt_addr + fmt_map_off;
/* We are also guaranteed that fmt+fmt_map_off is NULL terminated, we *canfocusonvalidatingtheformatspecifiers.
*/
err = bpf_bprintf_prepare(fmt, UINT_MAX, NULL, num_args, &data); if (err < 0)
verbose(env, "Invalid format string\n");
return err;
}
staticint check_get_func_ip(struct bpf_verifier_env *env)
{ enum bpf_prog_type type = resolve_prog_type(env->prog); int func_id = BPF_FUNC_get_func_ip;
if (type == BPF_PROG_TYPE_TRACING) { if (!bpf_prog_has_trampoline(env->prog)) {
verbose(env, "func %s#%d supported only for fentry/fexit/fmod_ret programs\n",
func_id_name(func_id), func_id); return -ENOTSUPP;
} return0;
} elseif (type == BPF_PROG_TYPE_KPROBE) { return0;
}
verbose(env, "func %s#%d not supported for program type %d\n",
func_id_name(func_id), func_id, type); return -ENOTSUPP;
}
/* Returns whether or not the given map type can potentially elide *lookupreturnvaluenullnesscheck.Thisispossibleifthekey *isstaticallyknown.
*/ staticbool can_elide_value_nullness(enum bpf_map_type type)
{ switch (type) { case BPF_MAP_TYPE_ARRAY: case BPF_MAP_TYPE_PERCPU_ARRAY: returntrue; default: returnfalse;
}
}
staticint get_helper_proto(struct bpf_verifier_env *env, int func_id, conststruct bpf_func_proto **ptr)
{ if (func_id < 0 || func_id >= __BPF_FUNC_MAX_ID) return -ERANGE;
if (err) {
verbose(env, "program of this type cannot use helper %s#%d\n",
func_id_name(func_id), func_id); return err;
}
/* eBPF programs must be GPL compatible to use GPL-ed functions */ if (!env->prog->gpl_compatible && fn->gpl_only) {
verbose(env, "cannot call GPL-restricted function from non-GPL compatible program\n"); return -EINVAL;
}
if (fn->allowed && !fn->allowed(env->prog)) {
verbose(env, "helper call is not allowed in probe\n"); return -EINVAL;
}
if (!in_sleepable(env) && fn->might_sleep) {
verbose(env, "helper call might sleep in a non-sleepable prog\n"); return -EINVAL;
}
/* With LD_ABS/IND some JITs save/restore skb from r1. */
changes_data = bpf_helper_changes_pkt_data(func_id); if (changes_data && fn->arg1_type != ARG_PTR_TO_CTX) {
verifier_bug(env, "func %s#%d: r1 != ctx", func_id_name(func_id), func_id); return -EFAULT;
}
err = check_func_proto(fn, func_id); if (err) {
verifier_bug(env, "incorrect func proto %s#%d", func_id_name(func_id), func_id); return err;
}
if (env->cur_state->active_rcu_lock) { if (fn->might_sleep) {
verbose(env, "sleepable helper %s#%d in rcu_read_lock region\n",
func_id_name(func_id), func_id); return -EINVAL;
}
if (in_sleepable(env) && is_storage_get_function(func_id))
env->insn_aux_data[insn_idx].storage_get_func_atomic = true;
}
if (env->cur_state->active_preempt_locks) { if (fn->might_sleep) {
verbose(env, "sleepable helper %s#%d in non-preemptible region\n",
func_id_name(func_id), func_id); return -EINVAL;
}
if (in_sleepable(env) && is_storage_get_function(func_id))
env->insn_aux_data[insn_idx].storage_get_func_atomic = true;
}
if (env->cur_state->active_irq_id) { if (fn->might_sleep) {
verbose(env, "sleepable helper %s#%d in IRQ-disabled region\n",
func_id_name(func_id), func_id); return -EINVAL;
}
if (in_sleepable(env) && is_storage_get_function(func_id))
env->insn_aux_data[insn_idx].storage_get_func_atomic = true;
}
meta.func_id = func_id; /* check args */ for (i = 0; i < MAX_BPF_FUNC_REG_ARGS; i++) {
err = check_func_arg(env, i, &meta, fn, insn_idx); if (err) return err;
}
err = record_func_map(env, &meta, func_id, insn_idx); if (err) return err;
err = record_func_key(env, &meta, func_id, insn_idx); if (err) return err;
/* Mark slots with STACK_MISC in case of raw mode, stack offset *isinferredfromregisterstate.
*/ for (i = 0; i < meta.access_size; i++) {
err = check_mem_access(env, insn_idx, meta.regno, i, BPF_B,
BPF_WRITE, -1, false, false); if (err) return err;
}
regs = cur_regs(env);
if (meta.release_regno) {
err = -EINVAL; /* This can only be set for PTR_TO_STACK, as CONST_PTR_TO_DYNPTR cannot *bereleasedbyanydynptrhelper.Hence,unmark_stack_slots_dynptr *issafetododirectly.
*/ if (arg_type_is_dynptr(fn->arg_type[meta.release_regno - BPF_REG_1])) { if (regs[meta.release_regno].type == CONST_PTR_TO_DYNPTR) {
verifier_bug(env, "CONST_PTR_TO_DYNPTR cannot be released"); return -EFAULT;
}
err = unmark_stack_slots_dynptr(env, ®s[meta.release_regno]);
} elseif (func_id == BPF_FUNC_kptr_xchg && meta.ref_obj_id) {
u32 ref_obj_id = meta.ref_obj_id; bool in_rcu = in_rcu_cs(env); struct bpf_func_state *state; struct bpf_reg_state *reg;
err = release_reference_nomark(env->cur_state, ref_obj_id); if (!err) {
bpf_for_each_reg_in_vstate(env->cur_state, state, reg, ({ if (reg->ref_obj_id == ref_obj_id) { if (in_rcu && (reg->type & MEM_ALLOC) && (reg->type & MEM_PERCPU)) {
reg->ref_obj_id = 0;
reg->type &= ~MEM_ALLOC;
reg->type |= MEM_RCU;
} else {
mark_reg_invalid(env, reg);
}
}
}));
}
} elseif (meta.ref_obj_id) {
err = release_reference(env, meta.ref_obj_id);
} elseif (register_is_null(®s[meta.release_regno])) { /* meta.ref_obj_id can only be 0 if register that is meant to be *releasedisNULL,whichmustbe>R0.
*/
err = 0;
} if (err) {
verbose(env, "func %s#%d reference has not been acquired before\n",
func_id_name(func_id), func_id); return err;
}
}
switch (func_id) { case BPF_FUNC_tail_call:
err = check_resource_leak(env, false, true, "tail_call"); if (err) return err; break; case BPF_FUNC_get_local_storage: /* check that flags argument in get_local_storage(map, flags) is 0, *thisisrequiredbecauseget_local_storage()can'treturnanerror.
*/ if (!register_is_null(®s[BPF_REG_2])) {
verbose(env, "get_local_storage() doesn't support non-zero flags\n"); return -EINVAL;
} break; case BPF_FUNC_for_each_map_elem:
err = push_callback_call(env, insn, insn_idx, meta.subprogno,
set_map_elem_callback_state); break; case BPF_FUNC_timer_set_callback:
err = push_callback_call(env, insn, insn_idx, meta.subprogno,
set_timer_callback_state); break; case BPF_FUNC_find_vma:
err = push_callback_call(env, insn, insn_idx, meta.subprogno,
set_find_vma_callback_state); break; case BPF_FUNC_snprintf:
err = check_bpf_snprintf_call(env, regs); break; case BPF_FUNC_loop:
update_loop_inline_state(env, meta.subprogno); /* Verifier relies on R1 value to determine if bpf_loop() iteration *isfinished,thusmarkitprecise.
*/
err = mark_chain_precision(env, BPF_REG_1); if (err) return err; if (cur_func(env)->callback_depth < regs[BPF_REG_1].umax_value) {
err = push_callback_call(env, insn, insn_idx, meta.subprogno,
set_loop_callback_state);
} else {
cur_func(env)->callback_depth = 0; if (env->log.level & BPF_LOG_LEVEL2)
verbose(env, "frame%d bpf_loop iteration limit reached\n",
env->cur_state->curframe);
} break; case BPF_FUNC_dynptr_from_mem: if (regs[BPF_REG_1].type != PTR_TO_MAP_VALUE) {
verbose(env, "Unsupported reg type %s for bpf_dynptr_from_mem data\n",
reg_type_str(env, regs[BPF_REG_1].type)); return -EACCES;
} break; case BPF_FUNC_set_retval: if (prog_type == BPF_PROG_TYPE_LSM &&
env->prog->expected_attach_type == BPF_LSM_CGROUP) { if (!env->prog->aux->attach_func_proto->type) { /* Make sure programs that attach to void *hooksdon'ttrytomodifyreturnvalue.
*/
verbose(env, "BPF_LSM_CGROUP that attach to void LSM hooks can't modify return value!\n"); return -EINVAL;
}
} break; case BPF_FUNC_dynptr_data:
{ struct bpf_reg_state *reg; int id, ref_obj_id;
reg = get_dynptr_arg_reg(env, fn, regs); if (!reg) return -EFAULT;
if (meta.dynptr_id) {
verifier_bug(env, "meta.dynptr_id already set"); return -EFAULT;
} if (meta.ref_obj_id) {
verifier_bug(env, "meta.ref_obj_id already set"); return -EFAULT;
}
id = dynptr_id(env, reg); if (id < 0) {
verifier_bug(env, "failed to obtain dynptr id"); return id;
}
ref_obj_id = dynptr_ref_obj_id(env, reg); if (ref_obj_id < 0) {
verifier_bug(env, "failed to obtain dynptr ref_obj_id"); return ref_obj_id;
}
reg = get_dynptr_arg_reg(env, fn, regs); if (!reg) return -EFAULT;
dynptr_type = dynptr_get_type(env, reg); if (dynptr_type == BPF_DYNPTR_TYPE_INVALID) return -EFAULT;
if (dynptr_type == BPF_DYNPTR_TYPE_SKB) /* this will trigger clear_all_pkt_pointers(), which will *invalidatealldynptrslicesassociatedwiththeskb
*/
changes_data = true;
break;
} case BPF_FUNC_per_cpu_ptr: case BPF_FUNC_this_cpu_ptr:
{ struct bpf_reg_state *reg = ®s[BPF_REG_1]; conststruct btf_type *type;
if (reg->type & MEM_RCU) {
type = btf_type_by_id(reg->btf, reg->btf_id); if (!type || !btf_type_is_struct(type)) {
verbose(env, "Helper has invalid btf/btf_id in R1\n"); return -EFAULT;
}
returns_cpu_specific_alloc_ptr = true;
env->insn_aux_data[insn_idx].call_with_percpu_alloc_ptr = true;
} break;
} case BPF_FUNC_user_ringbuf_drain:
err = push_callback_call(env, insn, insn_idx, meta.subprogno,
set_user_ringbuf_callback_state); break;
}
if (err) return err;
/* reset caller saved regs */ for (i = 0; i < CALLER_SAVED_REGS; i++) {
mark_reg_not_init(env, regs, caller_saved[i]);
check_reg_arg(env, caller_saved[i], DST_OP_NO_MARK);
}
/* update return register (already marked as written above) */
ret_type = fn->ret_type;
ret_flag = type_flag(ret_type);
switch (base_type(ret_type)) { case RET_INTEGER: /* sets type to SCALAR_VALUE */
mark_reg_unknown(env, regs, BPF_REG_0); break; case RET_VOID:
regs[BPF_REG_0].type = NOT_INIT; break; case RET_PTR_TO_MAP_VALUE: /* There is no offset yet applied, variable or fixed */
mark_reg_known_zero(env, regs, BPF_REG_0); /* remember map_ptr, so that check_map_access() *cancheck'value_size'boundaryofmemoryaccess *tomapelementreturnedfrombpf_map_lookup_elem()
*/ if (meta.map_ptr == NULL) {
verifier_bug(env, "unexpected null map_ptr"); return -EFAULT;
}
if (func_id == BPF_FUNC_get_func_ip) { if (check_get_func_ip(env)) return -ENOTSUPP;
env->prog->call_get_func_ip = true;
}
if (changes_data)
clear_all_pkt_pointers(env); return0;
}
/* mark_btf_func_reg_size() is used when the reg size is determined by *theBTFfunc_proto'sreturnvaluesizeandargument.
*/ staticvoid __mark_btf_func_reg_size(struct bpf_verifier_env *env, struct bpf_reg_state *regs,
u32 regno, size_t reg_size)
{ struct bpf_reg_state *reg = ®s[regno];
param_name = btf_name_by_offset(btf, arg->name_off); if (str_is_empty(param_name)) returnfalse;
len = strlen(param_name); if (len != target_len) returnfalse; if (strcmp(param_name, name)) returnfalse;
t = btf_type_skip_modifiers(btf, arg->type, NULL); if (!t) returnfalse; if (!btf_type_is_ptr(t)) returnfalse;
t = btf_type_skip_modifiers(btf, t->type, &res_id); if (!t) returnfalse; return btf_types_are_same(btf, res_id, btf_vmlinux, kf_arg_btf_ids[type]);
}
if (meta->func_id == special_kfunc_list[KF_bpf_cast_to_kern_ctx]) return KF_ARG_PTR_TO_CTX;
/* In this function, we verify the kfunc's BTF as per the argument type, *leavingtherestoftheverificationwithrespecttotheregister *typetoourcaller.WhenasetofconditionsholdintheBTFtypeof *arguments,weresolveittoaknownkfunc_ptr_arg_type.
*/ if (btf_is_prog_ctx_type(&env->log, meta->btf, t, resolve_prog_type(env->prog), argno)) return KF_ARG_PTR_TO_CTX;
if (is_kfunc_arg_nullable(meta->btf, &args[argno]) && register_is_null(reg)) return KF_ARG_PTR_TO_NULL;
if (is_kfunc_arg_alloc_obj(meta->btf, &args[argno])) return KF_ARG_PTR_TO_ALLOC_BTF_ID;
if (is_kfunc_arg_refcounted_kptr(meta->btf, &args[argno])) return KF_ARG_PTR_TO_REFCOUNTED_KPTR;
if (is_kfunc_arg_dynptr(meta->btf, &args[argno])) return KF_ARG_PTR_TO_DYNPTR;
if (is_kfunc_arg_iter(meta, argno, &args[argno])) return KF_ARG_PTR_TO_ITER;
if (is_kfunc_arg_list_head(meta->btf, &args[argno])) return KF_ARG_PTR_TO_LIST_HEAD;
if (is_kfunc_arg_list_node(meta->btf, &args[argno])) return KF_ARG_PTR_TO_LIST_NODE;
if (is_kfunc_arg_rbtree_root(meta->btf, &args[argno])) return KF_ARG_PTR_TO_RB_ROOT;
if (is_kfunc_arg_rbtree_node(meta->btf, &args[argno])) return KF_ARG_PTR_TO_RB_NODE;
if (is_kfunc_arg_const_str(meta->btf, &args[argno])) return KF_ARG_PTR_TO_CONST_STR;
if (is_kfunc_arg_map(meta->btf, &args[argno])) return KF_ARG_PTR_TO_MAP;
if (is_kfunc_arg_wq(meta->btf, &args[argno])) return KF_ARG_PTR_TO_WORKQUEUE;
if (is_kfunc_arg_irq_flag(meta->btf, &args[argno])) return KF_ARG_PTR_TO_IRQ_FLAG;
if (is_kfunc_arg_res_spin_lock(meta->btf, &args[argno])) return KF_ARG_PTR_TO_RES_SPIN_LOCK;
if ((base_type(reg->type) == PTR_TO_BTF_ID || reg2btf_ids[base_type(reg->type)])) { if (!btf_type_is_struct(ref_t)) {
verbose(env, "kernel function %s args#%d pointer type %s %s is not supported\n",
meta->func_name, argno, btf_type_str(ref_t), ref_tname); return -EINVAL;
} return KF_ARG_PTR_TO_BTF_ID;
}
if (is_kfunc_arg_callback(env, meta->btf, &args[argno])) return KF_ARG_PTR_TO_CALLBACK;
/* This is the catch all argument type of register types supported by *check_helper_mem_access.However,weonlyallowwhenargumenttypeis *pointertoscalar,orstructcomposed(recursively)ofscalars.When *arg_mem_sizeistrue,thepointercanbevoid*.
*/ if (!btf_type_is_scalar(ref_t) && !__btf_type_is_scalar_struct(env, meta->btf, ref_t, 0) &&
(arg_mem_size ? !btf_type_is_void(ref_t) : 1)) {
verbose(env, "arg#%d pointer type %s %s must point to %sscalar, or struct with scalar\n",
argno, btf_type_str(ref_t), ref_tname, arg_mem_size ? "void, " : ""); return -EINVAL;
} return arg_mem_size ? KF_ARG_PTR_TO_MEM_SIZE : KF_ARG_PTR_TO_MEM;
}
if (irq_save) { if (!is_irq_flag_reg_valid_uninit(env, reg)) {
verbose(env, "expected uninitialized irq flag as arg#%d\n", regno - 1); return -EINVAL;
}
switch ((int)reg->type) { case PTR_TO_MAP_VALUE:
ptr = reg->map_ptr; break; case PTR_TO_BTF_ID | MEM_ALLOC:
ptr = reg->btf; break; default:
verifier_bug(env, "unknown reg type for lock check"); return -EFAULT;
}
id = reg->id;
if (!env->cur_state->active_locks) return -EINVAL;
s = find_lock_state(env->cur_state, REF_TYPE_LOCK_MASK, id, ptr); if (!s) {
verbose(env, "held lock and object are not in the same allocation\n"); return -EINVAL;
} return0;
}
if (meta->btf != btf_vmlinux) {
verifier_bug(env, "unexpected btf mismatch in kfunc call"); return -EFAULT;
}
if (!check_kfunc_is_graph_root_api(env, head_field_type, meta->func_id)) return -EFAULT;
head_type_name = btf_field_type_name(head_field_type); if (!tnum_is_const(reg->var_off)) {
verbose(env, "R%d doesn't have constant offset. %s has to be at the constant offset\n",
regno, head_type_name); return -EINVAL;
}
rec = reg_btf_record(reg);
head_off = reg->off + reg->var_off.value;
field = btf_record_find(rec, head_off, head_field_type); if (!field) {
verbose(env, "%s not found at offset=%u\n", head_type_name, head_off); return -EINVAL;
}
/* All functions require bpf_list_head to be protected using a bpf_spin_lock */ if (check_reg_allocation_locked(env, reg)) {
verbose(env, "bpf_spin_lock at off=%d must be held for %s\n",
rec->spin_lock_off, head_type_name); return -EINVAL;
}
if (meta->btf != btf_vmlinux) {
verifier_bug(env, "unexpected btf mismatch in kfunc call"); return -EFAULT;
}
if (!check_kfunc_is_graph_node_api(env, node_field_type, meta->func_id)) return -EFAULT;
node_type_name = btf_field_type_name(node_field_type); if (!tnum_is_const(reg->var_off)) {
verbose(env, "R%d doesn't have constant offset. %s has to be at the constant offset\n",
regno, node_type_name); return -EINVAL;
}
node_off = reg->off + reg->var_off.value;
field = reg_find_field_offset(reg, node_off, node_field_type); if (!field) {
verbose(env, "%s not found at offset=%u\n", node_type_name, node_off); return -EINVAL;
}
field = *node_field;
et = btf_type_by_id(field->graph_root.btf, field->graph_root.value_btf_id);
t = btf_type_by_id(reg->btf, reg->btf_id); if (!btf_struct_ids_match(&env->log, reg->btf, reg->btf_id, 0, field->graph_root.btf,
field->graph_root.value_btf_id, true)) {
verbose(env, "operation on %s expects arg#1 %s at offset=%d " "in struct %s, but arg is at offset=%d in struct %s\n",
btf_field_type_name(head_field_type),
btf_field_type_name(node_field_type),
field->graph_root.node_offset,
btf_name_by_offset(field->graph_root.btf, et->name_off),
node_off, btf_name_by_offset(reg->btf, t->name_off)); return -EINVAL;
}
meta->arg_btf = reg->btf;
meta->arg_btf_id = reg->btf_id;
if (node_off != field->graph_root.node_offset) {
verbose(env, "arg#1 offset=%d, but expected %s at offset=%d in struct %s\n",
node_off, btf_field_type_name(node_field_type),
field->graph_root.node_offset,
btf_name_by_offset(field->graph_root.btf, et->name_off)); return -EINVAL;
}
/* Check that BTF function arguments match actual types that the *verifiersees.
*/ for (i = 0; i < nargs; i++) { struct bpf_reg_state *regs = cur_regs(env), *reg = ®s[i + 1]; conststruct btf_type *t, *ref_t, *resolve_ret; enum bpf_arg_type arg_type = ARG_DONTCARE;
u32 regno = i + 1, ref_id, type_size; bool is_ret_buf_sz = false; int kf_arg_type;
t = btf_type_skip_modifiers(btf, args[i].type, NULL);
if (is_kfunc_arg_ignore(btf, &args[i])) continue;
if (is_kfunc_arg_prog(btf, &args[i])) { /* Used to reject repeated use of __prog. */ if (meta->arg_prog) {
verifier_bug(env, "Only 1 prog->aux argument supported per-kfunc"); return -EFAULT;
}
meta->arg_prog = true;
cur_aux(env)->arg_prog = regno; continue;
}
if (btf_type_is_scalar(t)) { if (reg->type != SCALAR_VALUE) {
verbose(env, "R%d is not a scalar\n", regno); return -EINVAL;
}
if (is_kfunc_arg_constant(meta->btf, &args[i])) { if (meta->arg_constant.found) {
verifier_bug(env, "only one constant argument permitted"); return -EFAULT;
} if (!tnum_is_const(reg->var_off)) {
verbose(env, "R%d must be a known constant\n", regno); return -EINVAL;
}
ret = mark_chain_precision(env, regno); if (ret < 0) return ret;
meta->arg_constant.found = true;
meta->arg_constant.value = reg->var_off.value;
} elseif (is_kfunc_arg_scalar_with_name(btf, &args[i], "rdonly_buf_size")) {
meta->r0_rdonly = true;
is_ret_buf_sz = true;
} elseif (is_kfunc_arg_scalar_with_name(btf, &args[i], "rdwr_buf_size")) {
is_ret_buf_sz = true;
}
if (is_ret_buf_sz) { if (meta->r0_size) {
verbose(env, "2 or more rdonly/rdwr_buf_size parameters for kfunc"); return -EINVAL;
}
if (!tnum_is_const(reg->var_off)) {
verbose(env, "R%d is not a const\n", regno); return -EINVAL;
}
meta->r0_size = reg->var_off.value;
ret = mark_chain_precision(env, regno); if (ret) return ret;
} continue;
}
if (!btf_type_is_ptr(t)) {
verbose(env, "Unrecognized arg#%d type %s\n", i, btf_type_str(t)); return -EINVAL;
}
kf_arg_type = get_kfunc_ptr_arg_type(env, meta, t, ref_t, ref_tname, args, i, nargs); if (kf_arg_type < 0) return kf_arg_type;
switch (kf_arg_type) { case KF_ARG_PTR_TO_NULL: continue; case KF_ARG_PTR_TO_MAP: if (!reg->map_ptr) {
verbose(env, "pointer in R%d isn't map pointer\n", regno); return -EINVAL;
} if (meta->map.ptr && reg->map_ptr->record->wq_off >= 0) { /* Use map_uid (which is unique id of inner map) to reject: *inner_map1=bpf_map_lookup_elem(outer_map,key1) *inner_map2=bpf_map_lookup_elem(outer_map,key2) *if(inner_map1&&inner_map2){ *wq=bpf_map_lookup_elem(inner_map1); *if(wq) *// mismatch would have been allowed *bpf_wq_init(wq,inner_map2); *} * *Comparingmap_ptrisenoughtodistinguishnormalandoutermaps.
*/ if (meta->map.ptr != reg->map_ptr ||
meta->map.uid != reg->map_uid) {
verbose(env, "workqueue pointer in R1 map_uid=%d doesn't match map pointer in R2 map_uid=%d\n",
meta->map.uid, reg->map_uid); return -EINVAL;
}
}
meta->map.ptr = reg->map_ptr;
meta->map.uid = reg->map_uid;
fallthrough; case KF_ARG_PTR_TO_ALLOC_BTF_ID: case KF_ARG_PTR_TO_BTF_ID: if (!is_kfunc_trusted_args(meta) && !is_kfunc_rcu(meta)) break;
if (!is_trusted_reg(reg)) { if (!is_kfunc_rcu(meta)) {
verbose(env, "R%d must be referenced or trusted\n", regno); return -EINVAL;
} if (!is_rcu_reg(reg)) {
verbose(env, "R%d must be a rcu pointer\n", regno); return -EINVAL;
}
}
fallthrough; case KF_ARG_PTR_TO_CTX: case KF_ARG_PTR_TO_DYNPTR: case KF_ARG_PTR_TO_ITER: case KF_ARG_PTR_TO_LIST_HEAD: case KF_ARG_PTR_TO_LIST_NODE: case KF_ARG_PTR_TO_RB_ROOT: case KF_ARG_PTR_TO_RB_NODE: case KF_ARG_PTR_TO_MEM: case KF_ARG_PTR_TO_MEM_SIZE: case KF_ARG_PTR_TO_CALLBACK: case KF_ARG_PTR_TO_REFCOUNTED_KPTR: case KF_ARG_PTR_TO_CONST_STR: case KF_ARG_PTR_TO_WORKQUEUE: case KF_ARG_PTR_TO_IRQ_FLAG: case KF_ARG_PTR_TO_RES_SPIN_LOCK: break; default:
verifier_bug(env, "unknown kfunc arg type %d", kf_arg_type); return -EFAULT;
}
if (is_kfunc_release(meta) && reg->ref_obj_id)
arg_type |= OBJ_RELEASE;
ret = check_func_arg_reg_off(env, reg, regno, arg_type); if (ret < 0) return ret;
switch (kf_arg_type) { case KF_ARG_PTR_TO_CTX: if (reg->type != PTR_TO_CTX) {
verbose(env, "arg#%d expected pointer to ctx, but got %s\n",
i, reg_type_str(env, reg->type)); return -EINVAL;
}
if (meta->func_id == special_kfunc_list[KF_bpf_cast_to_kern_ctx]) {
ret = get_kern_ctx_btf_id(&env->log, resolve_prog_type(env->prog)); if (ret < 0) return -EINVAL;
meta->ret_btf_id = ret;
} break; case KF_ARG_PTR_TO_ALLOC_BTF_ID: if (reg->type == (PTR_TO_BTF_ID | MEM_ALLOC)) { if (meta->func_id != special_kfunc_list[KF_bpf_obj_drop_impl]) {
verbose(env, "arg#%d expected for bpf_obj_drop_impl()\n", i); return -EINVAL;
}
} elseif (reg->type == (PTR_TO_BTF_ID | MEM_ALLOC | MEM_PERCPU)) { if (meta->func_id != special_kfunc_list[KF_bpf_percpu_obj_drop_impl]) {
verbose(env, "arg#%d expected for bpf_percpu_obj_drop_impl()\n", i); return -EINVAL;
}
} else {
verbose(env, "arg#%d expected pointer to allocated object\n", i); return -EINVAL;
} if (!reg->ref_obj_id) {
verbose(env, "allocated object must be referenced\n"); return -EINVAL;
} if (meta->btf == btf_vmlinux) {
meta->arg_btf = reg->btf;
meta->arg_btf_id = reg->btf_id;
} break; case KF_ARG_PTR_TO_DYNPTR:
{ enum bpf_arg_type dynptr_arg_type = ARG_PTR_TO_DYNPTR; int clone_ref_obj_id = 0;
if (reg->type == CONST_PTR_TO_DYNPTR)
dynptr_arg_type |= MEM_RDONLY;
if (is_kfunc_arg_uninit(btf, &args[i]))
dynptr_arg_type |= MEM_UNINIT;
if (parent_type == BPF_DYNPTR_TYPE_INVALID) {
verifier_bug(env, "no dynptr type for parent of clone"); return -EFAULT;
}
dynptr_arg_type |= (unsignedint)get_dynptr_type_flag(parent_type);
clone_ref_obj_id = meta->initialized_dynptr.ref_obj_id; if (dynptr_type_refcounted(parent_type) && !clone_ref_obj_id) {
verifier_bug(env, "missing ref obj id for parent of clone"); return -EFAULT;
}
}
ret = process_dynptr_func(env, regno, insn_idx, dynptr_arg_type, clone_ref_obj_id); if (ret < 0) return ret;
if (!(dynptr_arg_type & MEM_UNINIT)) { int id = dynptr_id(env, reg);
if (id < 0) {
verifier_bug(env, "failed to obtain dynptr id"); return id;
}
meta->initialized_dynptr.id = id;
meta->initialized_dynptr.type = dynptr_get_type(env, reg);
meta->initialized_dynptr.ref_obj_id = dynptr_ref_obj_id(env, reg);
}
break;
} case KF_ARG_PTR_TO_ITER: if (meta->func_id == special_kfunc_list[KF_bpf_iter_css_task_new]) { if (!check_css_task_iter_allowlist(env)) {
verbose(env, "css_task_iter is only allowed in bpf_lsm, bpf_iter and sleepable progs\n"); return -EINVAL;
}
}
ret = process_iter_arg(env, regno, insn_idx, meta); if (ret < 0) return ret; break; case KF_ARG_PTR_TO_LIST_HEAD: if (reg->type != PTR_TO_MAP_VALUE &&
reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) {
verbose(env, "arg#%d expected pointer to map value or allocated object\n", i); return -EINVAL;
} if (reg->type == (PTR_TO_BTF_ID | MEM_ALLOC) && !reg->ref_obj_id) {
verbose(env, "allocated object must be referenced\n"); return -EINVAL;
}
ret = process_kf_arg_ptr_to_list_head(env, reg, regno, meta); if (ret < 0) return ret; break; case KF_ARG_PTR_TO_RB_ROOT: if (reg->type != PTR_TO_MAP_VALUE &&
reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) {
verbose(env, "arg#%d expected pointer to map value or allocated object\n", i); return -EINVAL;
} if (reg->type == (PTR_TO_BTF_ID | MEM_ALLOC) && !reg->ref_obj_id) {
verbose(env, "allocated object must be referenced\n"); return -EINVAL;
}
ret = process_kf_arg_ptr_to_rbtree_root(env, reg, regno, meta); if (ret < 0) return ret; break; case KF_ARG_PTR_TO_LIST_NODE: if (reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) {
verbose(env, "arg#%d expected pointer to allocated object\n", i); return -EINVAL;
} if (!reg->ref_obj_id) {
verbose(env, "allocated object must be referenced\n"); return -EINVAL;
}
ret = process_kf_arg_ptr_to_list_node(env, reg, regno, meta); if (ret < 0) return ret; break; case KF_ARG_PTR_TO_RB_NODE: if (meta->func_id == special_kfunc_list[KF_bpf_rbtree_add_impl]) { if (reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) {
verbose(env, "arg#%d expected pointer to allocated object\n", i); return -EINVAL;
} if (!reg->ref_obj_id) {
verbose(env, "allocated object must be referenced\n"); return -EINVAL;
}
} else { if (!type_is_non_owning_ref(reg->type) && !reg->ref_obj_id) {
verbose(env, "%s can only take non-owning or refcounted bpf_rb_node pointer\n", func_name); return -EINVAL;
} if (in_rbtree_lock_required_cb(env)) {
verbose(env, "%s not allowed in rbtree cb\n", func_name); return -EINVAL;
}
}
ret = process_kf_arg_ptr_to_rbtree_node(env, reg, regno, meta); if (ret < 0) return ret; break; case KF_ARG_PTR_TO_MAP: /* If argument has '__map' suffix expect 'struct bpf_map *' */
ref_id = *reg2btf_ids[CONST_PTR_TO_MAP];
ref_t = btf_type_by_id(btf_vmlinux, ref_id);
ref_tname = btf_name_by_offset(btf, ref_t->name_off);
fallthrough; case KF_ARG_PTR_TO_BTF_ID: /* Only base_type is checked, further checks are done here */ if ((base_type(reg->type) != PTR_TO_BTF_ID ||
(bpf_type_has_unsafe_modifiers(reg->type) && !is_rcu_reg(reg))) &&
!reg2btf_ids[base_type(reg->type)]) {
verbose(env, "arg#%d is %s ", i, reg_type_str(env, reg->type));
verbose(env, "expected %s or socket\n",
reg_type_str(env, base_type(reg->type) |
(type_flag(reg->type) & BPF_REG_TRUSTED_MODIFIERS))); return -EINVAL;
}
ret = process_kf_arg_ptr_to_btf_id(env, reg, ref_t, ref_tname, ref_id, meta, i); if (ret < 0) return ret; break; case KF_ARG_PTR_TO_MEM:
resolve_ret = btf_resolve_size(btf, ref_t, &type_size); if (IS_ERR(resolve_ret)) {
verbose(env, "arg#%d reference type('%s %s') size cannot be determined: %ld\n",
i, btf_type_str(ref_t), ref_tname, PTR_ERR(resolve_ret)); return -EINVAL;
}
ret = check_mem_reg(env, reg, regno, type_size); if (ret < 0) return ret; break; case KF_ARG_PTR_TO_MEM_SIZE:
{ struct bpf_reg_state *buff_reg = ®s[regno]; conststruct btf_param *buff_arg = &args[i]; struct bpf_reg_state *size_reg = ®s[regno + 1]; conststruct btf_param *size_arg = &args[i + 1];
if (!register_is_null(buff_reg) || !is_kfunc_arg_optional(meta->btf, buff_arg)) {
ret = check_kfunc_mem_size_reg(env, size_reg, regno + 1); if (ret < 0) {
verbose(env, "arg#%d arg#%d memory, len pair leads to invalid memory access\n", i, i + 1); return ret;
}
}
if (is_kfunc_arg_const_mem_size(meta->btf, size_arg, size_reg)) { if (meta->arg_constant.found) {
verifier_bug(env, "only one constant argument permitted"); return -EFAULT;
} if (!tnum_is_const(size_reg->var_off)) {
verbose(env, "R%d must be a known constant\n", regno + 1); return -EINVAL;
}
meta->arg_constant.found = true;
meta->arg_constant.value = size_reg->var_off.value;
}
/* Skip next '__sz' or '__szk' argument */
i++; break;
} case KF_ARG_PTR_TO_CALLBACK: if (reg->type != PTR_TO_FUNC) {
verbose(env, "arg%d expected pointer to func\n", i); return -EINVAL;
}
meta->subprogno = reg->subprogno; break; case KF_ARG_PTR_TO_REFCOUNTED_KPTR: if (!type_is_ptr_alloc_obj(reg->type)) {
verbose(env, "arg#%d is neither owning or non-owning ref\n", i); return -EINVAL;
} if (!type_is_non_owning_ref(reg->type))
meta->arg_owning_ref = true;
if (rec->refcount_off < 0) {
verbose(env, "arg#%d doesn't point to a type with bpf_refcount field\n", i); return -EINVAL;
}
meta->arg_btf = reg->btf;
meta->arg_btf_id = reg->btf_id; break; case KF_ARG_PTR_TO_CONST_STR: if (reg->type != PTR_TO_MAP_VALUE) {
verbose(env, "arg#%d doesn't point to a const string\n", i); return -EINVAL;
}
ret = check_reg_const_str(env, reg, regno); if (ret) return ret; break; case KF_ARG_PTR_TO_WORKQUEUE: if (reg->type != PTR_TO_MAP_VALUE) {
verbose(env, "arg#%d doesn't point to a map value\n", i); return -EINVAL;
}
ret = process_wq_func(env, regno, meta); if (ret < 0) return ret; break; case KF_ARG_PTR_TO_IRQ_FLAG: if (reg->type != PTR_TO_STACK) {
verbose(env, "arg#%d doesn't point to an irq flag on stack\n", i); return -EINVAL;
}
ret = process_irq_flag(env, regno, meta); if (ret < 0) return ret; break; case KF_ARG_PTR_TO_RES_SPIN_LOCK:
{ int flags = PROCESS_RES_LOCK;
if (reg->type != PTR_TO_MAP_VALUE && reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) {
verbose(env, "arg#%d doesn't point to map value or allocated object\n", i); return -EINVAL;
}
if (!is_bpf_res_spin_lock_kfunc(meta->func_id)) return -EFAULT; if (meta->func_id == special_kfunc_list[KF_bpf_res_spin_lock] ||
meta->func_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave])
flags |= PROCESS_SPIN_LOCK; if (meta->func_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave] ||
meta->func_id == special_kfunc_list[KF_bpf_res_spin_unlock_irqrestore])
flags |= PROCESS_LOCK_IRQ;
ret = process_spin_lock(env, regno, flags); if (ret < 0) return ret; break;
}
}
}
if (is_kfunc_release(meta) && !meta->release_regno) {
verbose(env, "release kernel function %s expects refcounted PTR_TO_BTF_ID\n",
func_name); return -EINVAL;
}
if (meta->func_id == special_kfunc_list[KF_bpf_obj_new_impl] && !bpf_global_ma_set) return -ENOMEM;
if (((u64)(u32)meta->arg_constant.value) != meta->arg_constant.value) {
verbose(env, "local type ID argument must be in range [0, U32_MAX]\n"); return -EINVAL;
}
/* This may be NULL due to user not supplying a BTF */ if (!ret_btf) {
verbose(env, "bpf_obj_new/bpf_percpu_obj_new requires prog BTF\n"); return -EINVAL;
}
ret_t = btf_type_by_id(ret_btf, ret_btf_id); if (!ret_t || !__btf_type_is_struct(ret_t)) {
verbose(env, "bpf_obj_new/bpf_percpu_obj_new type ID argument must be of a struct\n"); return -EINVAL;
}
if (meta->func_id == special_kfunc_list[KF_bpf_percpu_obj_new_impl]) { if (ret_t->size > BPF_GLOBAL_PERCPU_MA_MAX_SIZE) {
verbose(env, "bpf_percpu_obj_new type size (%d) is greater than %d\n",
ret_t->size, BPF_GLOBAL_PERCPU_MA_MAX_SIZE); return -EINVAL;
}
if (!bpf_global_percpu_ma_set) {
mutex_lock(&bpf_percpu_ma_lock); if (!bpf_global_percpu_ma_set) { /* Charge memory allocated with bpf_global_percpu_ma to *rootmemcg.Theobj_cgroupforrootmemcgisNULL.
*/
err = bpf_mem_alloc_percpu_init(&bpf_global_percpu_ma, NULL); if (!err)
bpf_global_percpu_ma_set = true;
}
mutex_unlock(&bpf_percpu_ma_lock); if (err) return err;
}
struct_meta = btf_find_struct_meta(ret_btf, ret_btf_id); if (meta->func_id == special_kfunc_list[KF_bpf_percpu_obj_new_impl]) { if (!__btf_type_is_scalar_struct(env, ret_btf, ret_t, 0)) {
verbose(env, "bpf_percpu_obj_new type ID argument must be of a struct of scalars\n"); return -EINVAL;
}
if (struct_meta) {
verbose(env, "bpf_percpu_obj_new type ID argument must not contain special fields\n"); return -EINVAL;
}
}
/* PTR_MAYBE_NULL will be added when is_kfunc_ret_null is checked */
regs[BPF_REG_0].type = PTR_TO_MEM | type_flag;
if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_slice]) {
regs[BPF_REG_0].type |= MEM_RDONLY;
} else { /* this will set env->seen_direct_write to true */ if (!may_access_direct_pkt_data(env, NULL, BPF_WRITE)) {
verbose(env, "the prog does not allow writes to packet data\n"); return -EINVAL;
}
}
if (!meta->initialized_dynptr.id) {
verifier_bug(env, "no dynptr id"); return -EFAULT;
}
regs[BPF_REG_0].dynptr_id = meta->initialized_dynptr.id;
/* we don't need to set BPF_REG_0's ref obj id *becausepacketslicesarenotrefcounted(see *dynptr_type_refcounted)
*/
} else { return0;
}
return1;
}
staticint check_return_code(struct bpf_verifier_env *env, int regno, constchar *reg_name);
if (env->cur_state->active_preempt_locks) { if (preempt_disable) {
env->cur_state->active_preempt_locks++;
} elseif (preempt_enable) {
env->cur_state->active_preempt_locks--;
} elseif (sleepable) {
verbose(env, "kernel func %s is sleepable within non-preemptible region\n", func_name); return -EACCES;
}
} elseif (preempt_disable) {
env->cur_state->active_preempt_locks++;
} elseif (preempt_enable) {
verbose(env, "unmatched attempt to enable preemption (kernel function %s)\n", func_name); return -EINVAL;
}
if (env->cur_state->active_irq_id && sleepable) {
verbose(env, "kernel func %s is sleepable within IRQ-disabled region\n", func_name); return -EACCES;
}
/* In case of release function, we get register number of refcounted *PTR_TO_BTF_IDinbpf_kfunc_arg_meta,dothereleasenow.
*/ if (meta.release_regno) {
err = release_reference(env, regs[meta.release_regno].ref_obj_id); if (err) {
verbose(env, "kfunc %s#%d reference has not been acquired before\n",
func_name, meta.func_id); return err;
}
}
err = release_reference(env, release_ref_obj_id); if (err) {
verbose(env, "kfunc %s#%d reference has not been acquired before\n",
func_name, meta.func_id); return err;
}
}
if (meta.func_id == special_kfunc_list[KF_bpf_throw]) { if (!bpf_jit_supports_exceptions()) {
verbose(env, "JIT does not support calling kfunc %s#%d\n",
func_name, meta.func_id); return -ENOTSUPP;
}
env->seen_exception = true;
/* In the case of the default callback, the cookie value passed *tobpf_throwbecomesthereturnvalueoftheprogram.
*/ if (!env->exception_callback_subprog) {
err = check_return_code(env, BPF_REG_1, "R1"); if (err < 0) return err;
}
}
for (i = 0; i < CALLER_SAVED_REGS; i++)
mark_reg_not_init(env, regs, caller_saved[i]);
/* Check return type */
t = btf_type_skip_modifiers(desc_btf, meta.func_proto->type, NULL);
if (is_kfunc_acquire(&meta) && !btf_type_is_struct_ptr(meta.btf, t)) { /* Only exception is bpf_obj_new_impl */ if (meta.btf != btf_vmlinux ||
(meta.func_id != special_kfunc_list[KF_bpf_obj_new_impl] &&
meta.func_id != special_kfunc_list[KF_bpf_percpu_obj_new_impl] &&
meta.func_id != special_kfunc_list[KF_bpf_refcount_acquire_impl])) {
verbose(env, "acquire kernel function does not return PTR_TO_BTF_ID\n"); return -EINVAL;
}
}
if (is_kfunc_ret_null(&meta)) {
regs[BPF_REG_0].type |= PTR_MAYBE_NULL; /* For mark_ptr_or_null_reg, see 93c230e3f5bd6 */
regs[BPF_REG_0].id = ++env->id_gen;
}
mark_btf_func_reg_size(env, BPF_REG_0, sizeof(void *)); if (is_kfunc_acquire(&meta)) { int id = acquire_reference(env, insn_idx);
if (id < 0) return id; if (is_kfunc_ret_null(&meta))
regs[BPF_REG_0].id = id;
regs[BPF_REG_0].ref_obj_id = id;
} elseif (is_rbtree_node_type(ptr_type) || is_list_node_type(ptr_type)) {
ref_set_non_owning(env, ®s[BPF_REG_0]);
}
if (known && (val >= BPF_MAX_VAR_OFF || val <= -BPF_MAX_VAR_OFF)) {
verbose(env, "math between %s pointer and %lld is not allowed\n",
reg_type_str(env, type), val); returnfalse;
}
if (reg->off >= BPF_MAX_VAR_OFF || reg->off <= -BPF_MAX_VAR_OFF) {
verbose(env, "%s pointer offset %d is not allowed\n",
reg_type_str(env, type), reg->off); returnfalse;
}
if (smin == S64_MIN) {
verbose(env, "math between %s pointer and register with unbounded min value is not allowed\n",
reg_type_str(env, type)); returnfalse;
}
if (smin >= BPF_MAX_VAR_OFF || smin <= -BPF_MAX_VAR_OFF) {
verbose(env, "value %lld makes %s pointer be out of bounds\n",
smin, reg_type_str(env, type)); returnfalse;
}
staticint update_alu_sanitation_state(struct bpf_insn_aux_data *aux,
u32 alu_state, u32 alu_limit)
{ /* If we arrived here from different branches with different *stateorlimitstosanitize,thenthiswon'twork.
*/ if (aux->alu_state &&
(aux->alu_state != alu_state ||
aux->alu_limit != alu_limit)) return REASON_PATHS;
/* We already marked aux for masking from non-speculative *paths,thuswegothereinthefirstplace.Weonlycare *toexplorebadaccessfromhere.
*/ if (vstate->speculative) goto do_sim;
if (!commit_window) { if (!tnum_is_const(off_reg->var_off) &&
(off_reg->smin_value < 0) != (off_reg->smax_value < 0)) return REASON_BOUNDS;
if (commit_window) { /* In commit phase we narrow the masking window based on *theobservedpointermoveafterthesimulatedoperation.
*/
alu_state = info->aux.alu_state;
alu_limit = abs(info->aux.alu_limit - alu_limit);
} else {
alu_state = off_is_neg ? BPF_ALU_NEG_VALUE : 0;
alu_state |= off_is_imm ? BPF_ALU_IMMEDIATE : 0;
alu_state |= ptr_is_dst_reg ?
BPF_ALU_SANITIZE_SRC : BPF_ALU_SANITIZE_DST;
/* Limit pruning on unknown scalars to enable deep search for *potentialmaskingdifferencesfromotherprogrampaths.
*/ if (!off_is_imm)
env->explore_alu_limits = true;
}
err = update_alu_sanitation_state(aux, alu_state, alu_limit); if (err < 0) return err;
do_sim: /* If we're in commit phase, we're done here given we already *pushedthetruncateddst_regintothespeculativeverification *stack. * *Also,whenregisterisaknownconstant,werewriteregister-based *operationtoimmediate-based,andthusdonotneedmasking(andas *aconsequence,donotneedtosimulatethezero-truncationeither).
*/ if (commit_window || off_is_imm) return0;
/* Simulate and find potential out-of-bounds access under *speculativeexecutionfromtruncationasaresultof *maskingwhenoffwasnotwithinexpectedrange.Ifoff *sitsindst,thenwetemporarilyneedtomoveptrthere *tosimulatedst(==0)+/-=ptr.Needed,forexample, *forcaseswhereweuseK-basedarithmeticinonedirection *andtruncatedreg-basedintheotherinordertoexplore *badaccess.
*/ if (!ptr_is_dst_reg) {
tmp = *dst_reg;
copy_register_state(dst_reg, ptr_reg);
}
ret = sanitize_speculative_path(env, NULL, env->insn_idx + 1,
env->insn_idx); if (!ptr_is_dst_reg && ret)
*dst_reg = tmp; return !ret ? REASON_STACK : 0;
}
/* If we simulate paths under speculation, we don't update the *insnas'seen'suchthatwhenweverifyunreachablepathsin *thenon-speculativedomain,sanitize_dead_code()canstill *rewrite/sanitizethem.
*/ if (!vstate->speculative)
env->insn_aux_data[env->insn_idx].seen = env->pass_cnt;
}
switch (base_type(ptr_reg->type)) { case PTR_TO_CTX: case PTR_TO_MAP_VALUE: case PTR_TO_MAP_KEY: case PTR_TO_STACK: case PTR_TO_PACKET_META: case PTR_TO_PACKET: case PTR_TO_TP_BUFFER: case PTR_TO_BTF_ID: case PTR_TO_MEM: case PTR_TO_BUF: case PTR_TO_FUNC: case CONST_PTR_TO_DYNPTR: break; case PTR_TO_FLOW_KEYS: if (known) break;
fallthrough; case CONST_PTR_TO_MAP: /* smin_val represents the known value */ if (known && smin_val == 0 && opcode == BPF_ADD) break;
fallthrough; default:
verbose(env, "R%d pointer arithmetic on %s prohibited\n",
dst, reg_type_str(env, ptr_reg->type)); return -EACCES;
}
/* In case of 'scalar += pointer', dst_reg inherits pointer type and id. *Theidmaybeoverwrittenlaterifwecreateanewvariableoffset.
*/
dst_reg->type = ptr_reg->type;
dst_reg->id = ptr_reg->id;
if (!check_reg_sane_offset(env, off_reg, ptr_reg->type) ||
!check_reg_sane_offset(env, ptr_reg, ptr_reg->type)) return -EINVAL;
/* pointer types do not carry 32-bit bounds at the moment. */
__mark_reg32_unbounded(dst_reg);
if (sanitize_needed(opcode)) {
ret = sanitize_ptr_alu(env, insn, ptr_reg, off_reg, dst_reg,
&info, false); if (ret < 0) return sanitize_err(env, insn, ret, off_reg, dst_reg);
}
switch (opcode) { case BPF_ADD: /* We can take a fixed offset as long as it doesn't overflow *thes32'off'field
*/ if (known && (ptr_reg->off + smin_val ==
(s64)(s32)(ptr_reg->off + smin_val))) { /* pointer += K. Accumulate it into fixed offset */
dst_reg->smin_value = smin_ptr;
dst_reg->smax_value = smax_ptr;
dst_reg->umin_value = umin_ptr;
dst_reg->umax_value = umax_ptr;
dst_reg->var_off = ptr_reg->var_off;
dst_reg->off = ptr_reg->off + smin_val;
dst_reg->raw = ptr_reg->raw; break;
} /* A new variable offset is created. Note that off_reg->off *==0,sinceit'sascalar. *dst_reggetsthepointertypeandsincesomepositive *integervaluewasaddedtothepointer,giveitanew'id' *ifit'saPTR_TO_PACKET. *thiscreatesanew'base'pointer,off_reg(variable)gets *addedintothevariableoffset,andwecopythefixedoffset *fromptr_reg.
*/ if (check_add_overflow(smin_ptr, smin_val, &dst_reg->smin_value) ||
check_add_overflow(smax_ptr, smax_val, &dst_reg->smax_value)) {
dst_reg->smin_value = S64_MIN;
dst_reg->smax_value = S64_MAX;
} if (check_add_overflow(umin_ptr, umin_val, &dst_reg->umin_value) ||
check_add_overflow(umax_ptr, umax_val, &dst_reg->umax_value)) {
dst_reg->umin_value = 0;
dst_reg->umax_value = U64_MAX;
}
dst_reg->var_off = tnum_add(ptr_reg->var_off, off_reg->var_off);
dst_reg->off = ptr_reg->off;
dst_reg->raw = ptr_reg->raw; if (reg_is_pkt_pointer(ptr_reg)) {
dst_reg->id = ++env->id_gen; /* something was added to pkt_ptr, set range to zero */
memset(&dst_reg->raw, 0, sizeof(dst_reg->raw));
} break; case BPF_SUB: if (dst_reg == off_reg) { /* scalar -= pointer. Creates an unknown scalar */
verbose(env, "R%d tried to subtract pointer from scalar\n",
dst); return -EACCES;
} /* We don't allow subtraction from FP, because (according to *test_verifier.ctest"invalidfparithmetic",JITsmightnot *beabletodealwithit.
*/ if (ptr_reg->type == PTR_TO_STACK) {
verbose(env, "R%d subtraction from stack pointer prohibited\n",
dst); return -EACCES;
} if (known && (ptr_reg->off - smin_val ==
(s64)(s32)(ptr_reg->off - smin_val))) { /* pointer -= K. Subtract it from fixed offset */
dst_reg->smin_value = smin_ptr;
dst_reg->smax_value = smax_ptr;
dst_reg->umin_value = umin_ptr;
dst_reg->umax_value = umax_ptr;
dst_reg->var_off = ptr_reg->var_off;
dst_reg->id = ptr_reg->id;
dst_reg->off = ptr_reg->off - smin_val;
dst_reg->raw = ptr_reg->raw; break;
} /* A new variable offset is created. If the subtrahend is known *nonnegative,thenanyreg->rangewehadbeforeisstillgood.
*/ if (check_sub_overflow(smin_ptr, smax_val, &dst_reg->smin_value) ||
check_sub_overflow(smax_ptr, smin_val, &dst_reg->smax_value)) { /* Overflow possible, we know nothing */
dst_reg->smin_value = S64_MIN;
dst_reg->smax_value = S64_MAX;
} if (umin_ptr < umax_val) { /* Overflow possible, we know nothing */
dst_reg->umin_value = 0;
dst_reg->umax_value = U64_MAX;
} else { /* Cannot overflow (as long as bounds are consistent) */
dst_reg->umin_value = umin_ptr - umax_val;
dst_reg->umax_value = umax_ptr - umin_val;
}
dst_reg->var_off = tnum_sub(ptr_reg->var_off, off_reg->var_off);
dst_reg->off = ptr_reg->off;
dst_reg->raw = ptr_reg->raw; if (reg_is_pkt_pointer(ptr_reg)) {
dst_reg->id = ++env->id_gen; /* something was added to pkt_ptr, set range to zero */ if (smin_val < 0)
memset(&dst_reg->raw, 0, sizeof(dst_reg->raw));
} break; case BPF_AND: case BPF_OR: case BPF_XOR: /* bitwise ops on pointers are troublesome, prohibit. */
verbose(env, "R%d bitwise operator %s on pointer prohibited\n",
dst, bpf_alu_string[opcode >> 4]); return -EACCES; default: /* other operators (e.g. MUL,LSH) produce non-pointer results */
verbose(env, "R%d pointer arithmetic with %s operator prohibited\n",
dst, bpf_alu_string[opcode >> 4]); return -EACCES;
}
if (!check_reg_sane_offset(env, dst_reg, ptr_reg->type)) return -EINVAL;
reg_bounds_sync(dst_reg);
bounds_ret = sanitize_check_bounds(env, insn, dst_reg); if (bounds_ret == -EACCES) return bounds_ret; if (sanitize_needed(opcode)) {
ret = sanitize_ptr_alu(env, insn, dst_reg, off_reg, dst_reg,
&info, true); if (verifier_bug_if(!can_skip_alu_sanitation(env, insn)
&& !env->cur_state->speculative
&& bounds_ret
&& !ret,
env, "Pointer type unsupported by sanitize_check_bounds() not rejected by retrieve_ptr_limit() as required")) { return -EFAULT;
} if (ret < 0) return sanitize_err(env, insn, ret, off_reg, dst_reg);
}
/* If either all additions overflow or no additions overflow, then *itisokaytoset:dst_umin=dst_umin+src_umin,dst_umax= *dst_umax+src_umax.Otherwise(someadditionsoverflow),set *theoutputboundstounbounded.
*/
min_overflow = check_add_overflow(*dst_umin, umin_val, dst_umin);
max_overflow = check_add_overflow(*dst_umax, umax_val, dst_umax);
/* If either all additions overflow or no additions overflow, then *itisokaytoset:dst_umin=dst_umin+src_umin,dst_umax= *dst_umax+src_umax.Otherwise(someadditionsoverflow),set *theoutputboundstounbounded.
*/
min_overflow = check_add_overflow(*dst_umin, umin_val, dst_umin);
max_overflow = check_add_overflow(*dst_umax, umax_val, dst_umax);
/* If either all subtractions underflow or no subtractions *underflow,itisokaytoset:dst_umin=dst_umin-src_umax, *dst_umax=dst_umax-src_umin.Otherwise(somesubtractions *underflow),settheoutputboundstounbounded.
*/
min_underflow = check_sub_overflow(*dst_umin, umax_val, dst_umin);
max_underflow = check_sub_overflow(*dst_umax, umin_val, dst_umax);
/* If either all subtractions underflow or no subtractions *underflow,itisokaytoset:dst_umin=dst_umin-src_umax, *dst_umax=dst_umax-src_umin.Otherwise(somesubtractions *underflow),settheoutputboundstounbounded.
*/
min_underflow = check_sub_overflow(*dst_umin, umax_val, dst_umin);
max_underflow = check_sub_overflow(*dst_umax, umin_val, dst_umax);
if (src_known && dst_known) {
__mark_reg32_known(dst_reg, var32_off.value); return;
}
/* We get our minimum from the var_off, since that's inherently *bitwise.Ourmaximumistheminimumoftheoperands'maxima.
*/
dst_reg->u32_min_value = var32_off.value;
dst_reg->u32_max_value = min(dst_reg->u32_max_value, umax_val);
/* Safe to set s32 bounds by casting u32 result into s32 when u32 *doesn'tcrosssignboundary.Otherwisesets32boundstounbounded.
*/ if ((s32)dst_reg->u32_min_value <= (s32)dst_reg->u32_max_value) {
dst_reg->s32_min_value = dst_reg->u32_min_value;
dst_reg->s32_max_value = dst_reg->u32_max_value;
} else {
dst_reg->s32_min_value = S32_MIN;
dst_reg->s32_max_value = S32_MAX;
}
}
if (src_known && dst_known) {
__mark_reg_known(dst_reg, dst_reg->var_off.value); return;
}
/* We get our minimum from the var_off, since that's inherently *bitwise.Ourmaximumistheminimumoftheoperands'maxima.
*/
dst_reg->umin_value = dst_reg->var_off.value;
dst_reg->umax_value = min(dst_reg->umax_value, umax_val);
/* Safe to set s64 bounds by casting u64 result into s64 when u64 *doesn'tcrosssignboundary.Otherwisesets64boundstounbounded.
*/ if ((s64)dst_reg->umin_value <= (s64)dst_reg->umax_value) {
dst_reg->smin_value = dst_reg->umin_value;
dst_reg->smax_value = dst_reg->umax_value;
} else {
dst_reg->smin_value = S64_MIN;
dst_reg->smax_value = S64_MAX;
} /* We may learn something more from the var_off */
__update_reg_bounds(dst_reg);
}
if (src_known && dst_known) {
__mark_reg32_known(dst_reg, var32_off.value); return;
}
/* We get our maximum from the var_off, and our minimum is the *maximumoftheoperands'minima
*/
dst_reg->u32_min_value = max(dst_reg->u32_min_value, umin_val);
dst_reg->u32_max_value = var32_off.value | var32_off.mask;
/* Safe to set s32 bounds by casting u32 result into s32 when u32 *doesn'tcrosssignboundary.Otherwisesets32boundstounbounded.
*/ if ((s32)dst_reg->u32_min_value <= (s32)dst_reg->u32_max_value) {
dst_reg->s32_min_value = dst_reg->u32_min_value;
dst_reg->s32_max_value = dst_reg->u32_max_value;
} else {
dst_reg->s32_min_value = S32_MIN;
dst_reg->s32_max_value = S32_MAX;
}
}
if (src_known && dst_known) {
__mark_reg_known(dst_reg, dst_reg->var_off.value); return;
}
/* We get our maximum from the var_off, and our minimum is the *maximumoftheoperands'minima
*/
dst_reg->umin_value = max(dst_reg->umin_value, umin_val);
dst_reg->umax_value = dst_reg->var_off.value | dst_reg->var_off.mask;
/* Safe to set s64 bounds by casting u64 result into s64 when u64 *doesn'tcrosssignboundary.Otherwisesets64boundstounbounded.
*/ if ((s64)dst_reg->umin_value <= (s64)dst_reg->umax_value) {
dst_reg->smin_value = dst_reg->umin_value;
dst_reg->smax_value = dst_reg->umax_value;
} else {
dst_reg->smin_value = S64_MIN;
dst_reg->smax_value = S64_MAX;
} /* We may learn something more from the var_off */
__update_reg_bounds(dst_reg);
}
if (src_known && dst_known) {
__mark_reg32_known(dst_reg, var32_off.value); return;
}
/* We get both minimum and maximum from the var32_off. */
dst_reg->u32_min_value = var32_off.value;
dst_reg->u32_max_value = var32_off.value | var32_off.mask;
/* Safe to set s32 bounds by casting u32 result into s32 when u32 *doesn'tcrosssignboundary.Otherwisesets32boundstounbounded.
*/ if ((s32)dst_reg->u32_min_value <= (s32)dst_reg->u32_max_value) {
dst_reg->s32_min_value = dst_reg->u32_min_value;
dst_reg->s32_max_value = dst_reg->u32_max_value;
} else {
dst_reg->s32_min_value = S32_MIN;
dst_reg->s32_max_value = S32_MAX;
}
}
if (src_known && dst_known) { /* dst_reg->var_off.value has been updated earlier */
__mark_reg_known(dst_reg, dst_reg->var_off.value); return;
}
/* We get both minimum and maximum from the var_off. */
dst_reg->umin_value = dst_reg->var_off.value;
dst_reg->umax_value = dst_reg->var_off.value | dst_reg->var_off.mask;
/* Safe to set s64 bounds by casting u64 result into s64 when u64 *doesn'tcrosssignboundary.Otherwisesets64boundstounbounded.
*/ if ((s64)dst_reg->umin_value <= (s64)dst_reg->umax_value) {
dst_reg->smin_value = dst_reg->umin_value;
dst_reg->smax_value = dst_reg->umax_value;
} else {
dst_reg->smin_value = S64_MIN;
dst_reg->smax_value = S64_MAX;
}
__update_reg_bounds(dst_reg);
}
staticvoid __scalar32_min_max_lsh(struct bpf_reg_state *dst_reg,
u64 umin_val, u64 umax_val)
{ /* We lose all sign bit information (except what we can pick *upfromvar_off)
*/
dst_reg->s32_min_value = S32_MIN;
dst_reg->s32_max_value = S32_MAX; /* If we might shift our top bit out, then we know nothing */ if (umax_val > 31 || dst_reg->u32_max_value > 1ULL << (31 - umax_val)) {
dst_reg->u32_min_value = 0;
dst_reg->u32_max_value = U32_MAX;
} else {
dst_reg->u32_min_value <<= umin_val;
dst_reg->u32_max_value <<= umax_val;
}
}
__scalar32_min_max_lsh(dst_reg, umin_val, umax_val);
dst_reg->var_off = tnum_subreg(tnum_lshift(subreg, umin_val)); /* Not required but being careful mark reg64 bounds as unknown so *thatweareforcedtopickthemupfromtnumandzextlaterand *ifsomepathskipsthisstepwearestillsafe.
*/
__mark_reg64_unbounded(dst_reg);
__update_reg32_bounds(dst_reg);
}
staticvoid __scalar64_min_max_lsh(struct bpf_reg_state *dst_reg,
u64 umin_val, u64 umax_val)
{ /* Special case <<32 because it is a common compiler pattern to sign *extendsubregbydoing<<32s>>32.Inthiscaseif32bitboundsare *positiveweknowthisshiftwillalsobepositivesowecantrack *boundscorrectly.Otherwiseweloseallsignbitinformationexcept *whatwecanpickupfromvar_off.Perhapswecangeneralizethis *latertoshiftsofanylength.
*/ if (umin_val == 32 && umax_val == 32 && dst_reg->s32_max_value >= 0)
dst_reg->smax_value = (s64)dst_reg->s32_max_value << 32; else
dst_reg->smax_value = S64_MAX;
/* scalar64 calc uses 32bit unshifted bounds so must be called first */
__scalar64_min_max_lsh(dst_reg, umin_val, umax_val);
__scalar32_min_max_lsh(dst_reg, umin_val, umax_val);
dst_reg->var_off = tnum_lshift(dst_reg->var_off, umin_val); /* We may learn something more from the var_off */
__update_reg_bounds(dst_reg);
}
/* BPF_RSH is an unsigned shift. If the value in dst_reg might *benegative,theneither: *1)src_regmightbezero,sothesignbitoftheresultis *unknown,soweloseoursignedbounds *2)it'sknownnegative,thustheunsignedboundscapturethe *signedbounds *3)thesignedboundscrosszero,sotheytellusnothing *abouttheresult *Ifthevalueindst_regisknownnonnegative,thenagainthe *unsignedboundscapturethesignedbounds. *Thus,inallcasesitsufficestoblowawayoursignedbounds *andrelyoninferringnewonesfromtheunsignedboundsand *var_offoftheresult.
*/
dst_reg->smin_value = S64_MIN;
dst_reg->smax_value = S64_MAX;
dst_reg->var_off = tnum_rshift(dst_reg->var_off, umin_val);
dst_reg->umin_value >>= umax_val;
dst_reg->umax_value >>= umin_val;
/* Its not easy to operate on alu32 bounds here because it depends *onbitsbeingshiftedin.Takeeasywayoutandmarkunbounded *sowecanrecalculatelaterfromtnum.
*/
__mark_reg32_unbounded(dst_reg);
__update_reg_bounds(dst_reg);
}
/* blow away the dst_reg umin_value/umax_value and rely on *dst_regvar_offtorefinetheresult.
*/
dst_reg->umin_value = 0;
dst_reg->umax_value = U64_MAX;
/* Its not easy to operate on alu32 bounds here because it depends *onbitsbeingshiftedinfromupper32-bits.Takeeasywayout *andmarkunboundedsowecanrecalculatelaterfromtnum.
*/
__mark_reg32_unbounded(dst_reg);
__update_reg_bounds(dst_reg);
}
switch (BPF_OP(insn->code)) { case BPF_ADD: case BPF_SUB: case BPF_NEG: case BPF_AND: case BPF_XOR: case BPF_OR: case BPF_MUL: returntrue;
/* Shift operators range is only computable if shift dimension operand *isaconstant.Shiftsgreaterthan31or63areundefined.This *includesshiftsbyanegativenumber.
*/ case BPF_LSH: case BPF_RSH: case BPF_ARSH: return (src_is_const && src_reg->umax_value < insn_bitness); default: returnfalse;
}
}
/* WARNING: This function does calculations on 64-bit values, but the actual *executionmayoccuron32-bitvalues.Therefore,thingslikebitshifts *needextrachecksinthe32-bitcase.
*/ staticint adjust_scalar_min_max_vals(struct bpf_verifier_env *env, struct bpf_insn *insn, struct bpf_reg_state *dst_reg, struct bpf_reg_state src_reg)
{
u8 opcode = BPF_OP(insn->code); bool alu32 = (BPF_CLASS(insn->code) != BPF_ALU64); int ret;
if (!is_safe_to_compute_dst_reg_range(insn, &src_reg)) {
__mark_reg_unknown(env, dst_reg); return0;
}
if (sanitize_needed(opcode)) {
ret = sanitize_val_alu(env, insn); if (ret < 0) return sanitize_err(env, insn, ret, NULL, NULL);
}
/* Calculate sign/unsigned bounds and tnum for alu32 and alu64 bit ops. *Therearetwoclassesofinstructions:Thefirstclasswetrackboth *alu32andalu64sign/unsignedboundsindependentlythisprovidesthe *greatestamountofprecisionwhenaluoperationsaremixedwithjmp32 *operations.TheseoperationsareBPF_ADD,BPF_SUB,BPF_MUL,BPF_ADD, *andBPF_OR.Thisispossiblebecausetheseopshavefairlyeasyto *understandandcalculatebehaviorinboth32-bitand64-bitaluops. *Seealu32verifiertestsforexamples.Thesecondclassof *operations,BPF_LSH,BPF_RSH,andBPF_ARSH,howeverarenotsoeasy *withregardstotrackingsign/unsignedboundsbecausethebitsmay *crosssubregboundariesinthealu64case.Whenthishappenswemark *theregunboundedinthesubregboundspaceandusetheresulting *tnumtocalculateanapproximationofthesign/unsignedbounds.
*/ switch (opcode) { case BPF_ADD:
scalar32_min_max_add(dst_reg, &src_reg);
scalar_min_max_add(dst_reg, &src_reg);
dst_reg->var_off = tnum_add(dst_reg->var_off, src_reg.var_off); break; case BPF_SUB:
scalar32_min_max_sub(dst_reg, &src_reg);
scalar_min_max_sub(dst_reg, &src_reg);
dst_reg->var_off = tnum_sub(dst_reg->var_off, src_reg.var_off); break; case BPF_NEG:
env->fake_reg[0] = *dst_reg;
__mark_reg_known(dst_reg, 0);
scalar32_min_max_sub(dst_reg, &env->fake_reg[0]);
scalar_min_max_sub(dst_reg, &env->fake_reg[0]);
dst_reg->var_off = tnum_neg(env->fake_reg[0].var_off); break; case BPF_MUL:
dst_reg->var_off = tnum_mul(dst_reg->var_off, src_reg.var_off);
scalar32_min_max_mul(dst_reg, &src_reg);
scalar_min_max_mul(dst_reg, &src_reg); break; case BPF_AND:
dst_reg->var_off = tnum_and(dst_reg->var_off, src_reg.var_off);
scalar32_min_max_and(dst_reg, &src_reg);
scalar_min_max_and(dst_reg, &src_reg); break; case BPF_OR:
dst_reg->var_off = tnum_or(dst_reg->var_off, src_reg.var_off);
scalar32_min_max_or(dst_reg, &src_reg);
scalar_min_max_or(dst_reg, &src_reg); break; case BPF_XOR:
dst_reg->var_off = tnum_xor(dst_reg->var_off, src_reg.var_off);
scalar32_min_max_xor(dst_reg, &src_reg);
scalar_min_max_xor(dst_reg, &src_reg); break; case BPF_LSH: if (alu32)
scalar32_min_max_lsh(dst_reg, &src_reg); else
scalar_min_max_lsh(dst_reg, &src_reg); break; case BPF_RSH: if (alu32)
scalar32_min_max_rsh(dst_reg, &src_reg); else
scalar_min_max_rsh(dst_reg, &src_reg); break; case BPF_ARSH: if (alu32)
scalar32_min_max_arsh(dst_reg, &src_reg); else
scalar_min_max_arsh(dst_reg, &src_reg); break; default: break;
}
/* ALU32 ops are zero extended into 64bit register */ if (alu32)
zext_32_to_64(dst_reg);
reg_bounds_sync(dst_reg); return0;
}
/* Handles ALU ops other than BPF_END, BPF_NEG and BPF_MOV: computes new min/max *andvar_off.
*/ staticint adjust_reg_min_max_vals(struct bpf_verifier_env *env, struct bpf_insn *insn)
{ struct bpf_verifier_state *vstate = env->cur_state; struct bpf_func_state *state = vstate->frame[vstate->curframe]; struct bpf_reg_state *regs = state->regs, *dst_reg, *src_reg; struct bpf_reg_state *ptr_reg = NULL, off_reg = {0}; bool alu32 = (BPF_CLASS(insn->code) != BPF_ALU64);
u8 opcode = BPF_OP(insn->code); int err;
dst_reg = ®s[insn->dst_reg];
src_reg = NULL;
if (dst_reg->type == PTR_TO_ARENA) { struct bpf_insn_aux_data *aux = cur_aux(env);
if (dst_reg->off < 0 ||
(dst_reg->off == 0 && range_right_open)) /* This doesn't give us any range */ return;
if (dst_reg->umax_value > MAX_PACKET_OFF ||
dst_reg->umax_value + dst_reg->off > MAX_PACKET_OFF) /* Risk of overflow. For instance, ptr + (1<<63) may be less *thanpkt_end,butthat'sbecauseit'salsolessthanpkt.
*/ return;
new_range = dst_reg->off; if (range_right_open)
new_range++;
/* If our ids match, then we must have the same max_value. And we *don'tcareabouttheotherreg'sfixedoffset,sinceifit'stoobig *therangewon'tallowanything. *dst_reg->offisknown<MAX_PACKET_OFF,thereforeitfitsinau16.
*/
bpf_for_each_reg_in_vstate(vstate, state, reg, ({ if (reg->type == type && reg->id == dst_reg->id) /* keep the maximum range already checked */
reg->range = max(reg->range, new_range);
}));
}
if (__is_pointer_value(false, reg1) || __is_pointer_value(false, reg2)) {
u64 val;
/* arrange that reg2 is a scalar, and reg1 is a pointer */ if (!is_reg_const(reg2, is_jmp32)) {
opcode = flip_opcode(opcode);
swap(reg1, reg2);
} /* and ensure that reg2 is a constant */ if (!is_reg_const(reg2, is_jmp32)) return -1;
if (!reg_not_null(reg1)) return -1;
/* If pointer is valid tests against zero will fail so we can *usethistodirectbranchtaken.
*/
val = reg_const_value(reg2, is_jmp32); if (val != 0) return -1;
switch (opcode) { case BPF_JEQ: return0; case BPF_JNE: return1; default: return -1;
}
}
/* now deal with two scalars, but not necessarily constants */ return is_scalar_branch_taken(reg1, reg2, opcode, is_jmp32);
}
/* Opcode that corresponds to a *false* branch condition. *E.g.,ifr1<r2,thenreverse(false)conditionisr1>=r2
*/ static u8 rev_opcode(u8 opcode)
{ switch (opcode) { case BPF_JEQ: return BPF_JNE; case BPF_JNE: return BPF_JEQ; /* JSET doesn't have it's reverse opcode in BPF, so add *BPF_Xflagtodenotethereverseofthatoperation
*/ case BPF_JSET: return BPF_JSET | BPF_X; case BPF_JSET | BPF_X: return BPF_JSET; case BPF_JGE: return BPF_JLT; case BPF_JGT: return BPF_JLE; case BPF_JLE: return BPF_JGT; case BPF_JLT: return BPF_JGE; case BPF_JSGE: return BPF_JSLT; case BPF_JSGT: return BPF_JSLE; case BPF_JSLE: return BPF_JSGT; case BPF_JSLT: return BPF_JSGE; default: return0;
}
}
/* In case of GE/GT/SGE/JST, reuse LE/LT/SLE/SLT logic from below */ switch (opcode) { case BPF_JGE: case BPF_JGT: case BPF_JSGE: case BPF_JSGT:
opcode = flip_opcode(opcode);
swap(reg1, reg2); break; default: break;
}
reg1->var_off = tnum_intersect(reg1->var_off, reg2->var_off);
reg2->var_off = reg1->var_off;
} break; case BPF_JNE: if (!is_reg_const(reg2, is_jmp32))
swap(reg1, reg2); if (!is_reg_const(reg2, is_jmp32)) break;
/* try to recompute the bound of reg1 if reg2 is a const and *isexactlytheedgeofreg1.
*/
val = reg_const_value(reg2, is_jmp32); if (is_jmp32) { /* u32_min_value is not equal to 0xffffffff at this point, *becauseotherwiseu32_max_valueis0xffffffffaswell, *insuchacasebothreg1andreg2wouldbeconstants, *jumpwouldbepredictedandreg_set_min_max()won't *becalled. * *Samereasoningworksforall{u,s}{min,max}{32,64}cases *below.
*/ if (reg1->u32_min_value == (u32)val)
reg1->u32_min_value++; if (reg1->u32_max_value == (u32)val)
reg1->u32_max_value--; if (reg1->s32_min_value == (s32)val)
reg1->s32_min_value++; if (reg1->s32_max_value == (s32)val)
reg1->s32_max_value--;
} else { if (reg1->umin_value == (u64)val)
reg1->umin_value++; if (reg1->umax_value == (u64)val)
reg1->umax_value--; if (reg1->smin_value == (s64)val)
reg1->smin_value++; if (reg1->smax_value == (s64)val)
reg1->smax_value--;
} break; case BPF_JSET: if (!is_reg_const(reg2, is_jmp32))
swap(reg1, reg2); if (!is_reg_const(reg2, is_jmp32)) break;
val = reg_const_value(reg2, is_jmp32); /* BPF_JSET (i.e., TRUE branch, *not* BPF_JSET | BPF_X) *requiressinglebittolearnsomethinguseful.E.g.,ifwe *knowthat`r1&0x3`istrue,thenwhichbits(0,1,orboth) *areactuallyset?Wecanlearnsomethingdefiniteonlyif *it'sasingle-bitvaluetobeginwith. * *BPF_JSET|BPF_X(i.e.,negationofBPF_JSET)doesn'thave *thisrestriction.I.e.,!(r1&0x3)meansneitherbit0nor *bit1isset,whichwecanreadilyuseinadjustments.
*/ if (!is_power_of_2(val)) break; if (is_jmp32) {
t = tnum_or(tnum_subreg(reg1->var_off), tnum_const(val));
reg1->var_off = tnum_with_subreg(reg1->var_off, t);
} else {
reg1->var_off = tnum_or(reg1->var_off, tnum_const(val));
} break; case BPF_JSET | BPF_X: /* reverse of BPF_JSET, see rev_opcode() */ if (!is_reg_const(reg2, is_jmp32))
swap(reg1, reg2); if (!is_reg_const(reg2, is_jmp32)) break;
val = reg_const_value(reg2, is_jmp32); /* Forget the ranges before narrowing tnums, to avoid invariant *violationsifwe'reonadeadbranch.
*/
__mark_reg_unbounded(reg1); if (is_jmp32) {
t = tnum_and(tnum_subreg(reg1->var_off), tnum_const(~val));
reg1->var_off = tnum_with_subreg(reg1->var_off, t);
} else {
reg1->var_off = tnum_and(reg1->var_off, tnum_const(~val));
} break; case BPF_JLE: if (is_jmp32) {
reg1->u32_max_value = min(reg1->u32_max_value, reg2->u32_max_value);
reg2->u32_min_value = max(reg1->u32_min_value, reg2->u32_min_value);
} else {
reg1->umax_value = min(reg1->umax_value, reg2->umax_value);
reg2->umin_value = max(reg1->umin_value, reg2->umin_value);
} break; case BPF_JLT: if (is_jmp32) {
reg1->u32_max_value = min(reg1->u32_max_value, reg2->u32_max_value - 1);
reg2->u32_min_value = max(reg1->u32_min_value + 1, reg2->u32_min_value);
} else {
reg1->umax_value = min(reg1->umax_value, reg2->umax_value - 1);
reg2->umin_value = max(reg1->umin_value + 1, reg2->umin_value);
} break; case BPF_JSLE: if (is_jmp32) {
reg1->s32_max_value = min(reg1->s32_max_value, reg2->s32_max_value);
reg2->s32_min_value = max(reg1->s32_min_value, reg2->s32_min_value);
} else {
reg1->smax_value = min(reg1->smax_value, reg2->smax_value);
reg2->smin_value = max(reg1->smin_value, reg2->smin_value);
} break; case BPF_JSLT: if (is_jmp32) {
reg1->s32_max_value = min(reg1->s32_max_value, reg2->s32_max_value - 1);
reg2->s32_min_value = max(reg1->s32_min_value + 1, reg2->s32_min_value);
} else {
reg1->smax_value = min(reg1->smax_value, reg2->smax_value - 1);
reg2->smin_value = max(reg1->smin_value + 1, reg2->smin_value);
} break; default: return;
}
}
/* Adjusts the register min/max values in the case that the dst_reg and *src_regarebothSCALAR_VALUEregisters(orwearesimplydoingaBPF_K *check,inwhichcasewehaveafakeSCALAR_VALUErepresentinginsn->imm). *Technicallywecandosimilaradjustmentsforpointerstothesameobject, *butwedon'tsupportthatrightnow.
*/ staticint reg_set_min_max(struct bpf_verifier_env *env, struct bpf_reg_state *true_reg1, struct bpf_reg_state *true_reg2, struct bpf_reg_state *false_reg1, struct bpf_reg_state *false_reg2,
u8 opcode, bool is_jmp32)
{ int err;
/* If either register is a pointer, we can't learn anything about its *variableoffsetfromthecompare(unlesstheywereapointerinto *thesameobject,butwedon'tbotherwiththat).
*/ if (false_reg1->type != SCALAR_VALUE || false_reg2->type != SCALAR_VALUE) return0;
staticvoid mark_ptr_or_null_reg(struct bpf_func_state *state, struct bpf_reg_state *reg, u32 id, bool is_null)
{ if (type_may_be_null(reg->type) && reg->id == id &&
(is_rcu_reg(reg) || !WARN_ON_ONCE(!reg->id))) { /* Old offset (both fixed and variable parts) should have been *known-zero,becausewedon'tallowpointerarithmeticon *pointersthatmightbeNULL.Ifweseethishappening,don't *converttheregister. * *Butinsomecases,somehelpersthatreturnlocalkptrs *advanceoffsetforthereturnedpointer.Inthosecases,it *isfinetoexpecttoseereg->off.
*/ if (WARN_ON_ONCE(reg->smin_value || reg->smax_value || !tnum_equals_const(reg->var_off, 0))) return; if (!(type_is_ptr_alloc_obj(reg->type) || type_is_non_owning_ref(reg->type)) &&
WARN_ON_ONCE(reg->off)) return;
if (is_null) {
reg->type = SCALAR_VALUE; /* We don't need id and ref_obj_id from this point *onwardsanymore,thusweshouldbetterresetit, *sothatstatepruninghaschancestotakeeffect.
*/
reg->id = 0;
reg->ref_obj_id = 0;
return;
}
mark_ptr_not_null_reg(reg);
if (!reg_may_point_to_spin_lock(reg)) { /* For not-NULL ptr, reg->ref_obj_id will be reset *inrelease_reference(). * *reg->idisstillusedbyspin_lockptr.Other *thanspin_lockptrtype,reg->idcanbereset.
*/
reg->id = 0;
}
}
}
/* The logic is similar to find_good_pkt_pointers(), both could eventually *befoldedtogetheratsomepoint.
*/ staticvoid mark_ptr_or_null_regs(struct bpf_verifier_state *vstate, u32 regno, bool is_null)
{ struct bpf_func_state *state = vstate->frame[vstate->curframe]; struct bpf_reg_state *regs = state->regs, *reg;
u32 ref_obj_id = regs[regno].ref_obj_id;
u32 id = regs[regno].id;
if (ref_obj_id && ref_obj_id == id && is_null) /* regs[regno] is in the " == NULL" branch. *Noonecouldhavefreedthereferencestatebefore *doingtheNULLcheck.
*/
WARN_ON_ONCE(release_reference_nomark(vstate, id));
/* For all R being scalar registers or spilled scalar registers *inverifierstate,saveRinlinked_regsifR->id==id. *IftherearetoomanyRssharingsameid,resetidforleftoverRs.
*/ staticvoid collect_linked_regs(struct bpf_verifier_state *vstate, u32 id, struct linked_regs *linked_regs)
{ struct bpf_func_state *func; struct bpf_reg_state *reg; int i, j;
id = id & ~BPF_ADD_CONST; for (i = vstate->curframe; i >= 0; i--) {
func = vstate->frame[i]; for (j = 0; j < BPF_REG_FP; j++) {
reg = &func->regs[j];
__collect_linked_regs(linked_regs, reg, id, i, j, true);
} for (j = 0; j < func->allocated_stack / BPF_REG_SIZE; j++) { if (!is_spilled_reg(&func->stack[j])) continue;
reg = &func->stack[j].spilled_ptr;
__collect_linked_regs(linked_regs, reg, id, i, j, false);
}
}
}
/* For all R in linked_regs, copy known_reg range into R *ifR->id==known_reg->id.
*/ staticvoid sync_linked_regs(struct bpf_verifier_state *vstate, struct bpf_reg_state *known_reg, struct linked_regs *linked_regs)
{ struct bpf_reg_state fake_reg; struct bpf_reg_state *reg; struct linked_reg *e; int i;
for (i = 0; i < linked_regs->cnt; ++i) {
e = &linked_regs->entries[i];
reg = e->is_reg ? &vstate->frame[e->frameno]->regs[e->regno]
: &vstate->frame[e->frameno]->stack[e->spi].spilled_ptr; if (reg->type != SCALAR_VALUE || reg == known_reg) continue; if ((reg->id & ~BPF_ADD_CONST) != (known_reg->id & ~BPF_ADD_CONST)) continue; if ((!(reg->id & BPF_ADD_CONST) && !(known_reg->id & BPF_ADD_CONST)) ||
reg->off == known_reg->off) {
s32 saved_subreg_def = reg->subreg_def;
if (dst_reg->type == PTR_TO_STACK)
insn_flags |= INSN_F_DST_REG_STACK;
}
if (insn_flags) {
err = push_jmp_history(env, this_branch, insn_flags, 0); if (err) return err;
}
is_jmp32 = BPF_CLASS(insn->code) == BPF_JMP32;
pred = is_branch_taken(dst_reg, src_reg, opcode, is_jmp32); if (pred >= 0) { /* If we get here with a dst_reg pointer type it is because *aboveis_branch_taken()specialcasedthe0comparison.
*/ if (!__is_pointer_value(false, dst_reg))
err = mark_chain_precision(env, insn->dst_reg); if (BPF_SRC(insn->code) == BPF_X && !err &&
!__is_pointer_value(false, src_reg))
err = mark_chain_precision(env, insn->src_reg); if (err) return err;
}
if (pred == 1) { /* Only follow the goto, ignore fall-through. If needed, push *thefall-throughbranchforsimulationunderspeculative *execution.
*/ if (!env->bypass_spec_v1 &&
!sanitize_speculative_path(env, insn, *insn_idx + 1,
*insn_idx)) return -EFAULT; if (env->log.level & BPF_LOG_LEVEL)
print_insn_state(env, this_branch, this_branch->curframe);
*insn_idx += insn->off; return0;
} elseif (pred == 0) { /* Only follow the fall-through branch, since that's where the *programwillgo.Ifneeded,pushthegotobranchfor *simulationunderspeculativeexecution.
*/ if (!env->bypass_spec_v1 &&
!sanitize_speculative_path(env, insn,
*insn_idx + insn->off + 1,
*insn_idx)) return -EFAULT; if (env->log.level & BPF_LOG_LEVEL)
print_insn_state(env, this_branch, this_branch->curframe); return0;
}
/* Push scalar registers sharing same ID to jump history, *dothisbeforecreating'other_branch',sothatboth *'this_branch'and'other_branch'sharethishistory *ifparentstateiscreated.
*/ if (BPF_SRC(insn->code) == BPF_X && src_reg->type == SCALAR_VALUE && src_reg->id)
collect_linked_regs(this_branch, src_reg->id, &linked_regs); if (dst_reg->type == SCALAR_VALUE && dst_reg->id)
collect_linked_regs(this_branch, dst_reg->id, &linked_regs); if (linked_regs.cnt > 1) {
err = push_jmp_history(env, this_branch, 0, linked_regs_pack(&linked_regs)); if (err) return err;
}
/* All special src_reg cases are listed below. From this point onwards *weeithersucceedandassignacorrespondingdst_reg->typeafter *zeroingtheoffset,orfailandrejecttheprogram.
*/
mark_reg_known_zero(env, regs, insn->dst_reg);
if (!may_access_skb(resolve_prog_type(env->prog))) {
verbose(env, "BPF_LD_[ABS|IND] instructions not allowed for this program type\n"); return -EINVAL;
}
if (!env->ops->gen_ld_abs) {
verifier_bug(env, "gen_ld_abs is null"); return -EFAULT;
}
/* LSM and struct_ops func-ptr's return type could be "void" */ if (!is_subprog || frame->in_exception_callback_fn) { switch (prog_type) { case BPF_PROG_TYPE_LSM: if (prog->expected_attach_type == BPF_LSM_CGROUP) /* See below, can be 0 or 0-1 depending on hook. */ break; if (!prog->aux->attach_func_proto->type) return0; break; case BPF_PROG_TYPE_STRUCT_OPS: if (!prog->aux->attach_func_proto->type) return0;
if (frame->in_exception_callback_fn) break;
/* Allow a struct_ops program to return a referenced kptr if it *matchestheoperator'sreturntypeandisinitsunmodified *form.Ascalarzero(i.e.,anullpointer)isalsoallowed.
*/
reg_type = reg->btf ? btf_type_by_id(reg->btf, reg->btf_id) : NULL;
ret_type = btf_type_resolve_ptr(prog->aux->attach_btf,
prog->aux->attach_func_proto->type,
NULL); if (ret_type && ret_type == reg_type && reg->ref_obj_id) return __check_ptr_off_reg(env, reg, regno, false); break; default: break;
}
}
/* eBPF calling convention is such that R0 is used *toreturnthevaluefromeBPFprogram. *Makesurethatit'sreadableatthistime *ofbpf_exit,whichmeansthatprogramwrote
--> --------------------
--> maximum size reached
--> --------------------
Messung V0.5 in Prozent
¤ Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.1.973Bemerkung:
(vorverarbeitet am 2026-09-28)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.