/* Get the first CPU from the list of unused CPUs in a CPU set data structure */ staticint cpu_mask_set_get_first(struct cpu_mask_set *set, cpumask_var_t diff)
{ int cpu;
if (!diff || !set) return -EINVAL;
_cpu_mask_set_gen_inc(set);
/* Find out CPUs left in CPU mask */
cpumask_andnot(diff, &set->mask, &set->used);
cpu = cpumask_first(diff); if (cpu >= nr_cpu_ids) /* empty */
cpu = -EINVAL; else
cpumask_set_cpu(cpu, &set->used);
return cpu;
}
staticvoid cpu_mask_set_put(struct cpu_mask_set *set, int cpu)
{ if (!set) return;
hfi1_per_node_cntr = kcalloc(node_affinity.num_possible_nodes, sizeof(*hfi1_per_node_cntr), GFP_KERNEL); if (!hfi1_per_node_cntr) return -ENOMEM;
while (ids->vendor) {
dev = NULL; while ((dev = pci_get_device(ids->vendor, ids->device, dev))) {
node = pcibus_to_node(dev->bus); if (node < 0) goto out;
hfi1_per_node_cntr[node]++;
}
ids++;
}
return0;
out: /* *InvalidPCINUMAnodeinformationfound,noteit,andpopulate *ourdatabase1:1.
*/
pr_err("HFI: Invalid PCI NUMA node. Performance may be affected\n");
pr_err("HFI: System BIOS may need to be upgraded\n"); for (node = 0; node < node_affinity.num_possible_nodes; node++)
hfi1_per_node_cntr[node] = 1;
/* It must be called with node_affinity.lock held */ staticstruct hfi1_affinity_node *node_affinity_lookup(int node)
{ struct hfi1_affinity_node *entry;
/* _dev_comp_vect_mappings_destroy() is reentrant */ staticvoid _dev_comp_vect_mappings_destroy(struct hfi1_devdata *dd)
{ int i, cpu;
if (!dd->comp_vect_mappings) return;
for (i = 0; i < dd->comp_vect_possible_cpus; i++) {
cpu = dd->comp_vect_mappings[i];
_dev_comp_vect_cpu_put(dd, cpu);
dd->comp_vect_mappings[i] = -1;
hfi1_cdbg(AFFINITY, "[%s] Release CPU %d from completion vector %d",
rvt_get_ibdev_name(&(dd)->verbs_dev.rdi), cpu, i);
}
if (!zalloc_cpumask_var(&non_intr_cpus, GFP_KERNEL)) return -ENOMEM;
if (!zalloc_cpumask_var(&available_cpus, GFP_KERNEL)) {
free_cpumask_var(non_intr_cpus); return -ENOMEM;
}
dd->comp_vect_mappings = kcalloc(dd->comp_vect_possible_cpus, sizeof(*dd->comp_vect_mappings),
GFP_KERNEL); if (!dd->comp_vect_mappings) {
ret = -ENOMEM; goto fail;
} for (i = 0; i < dd->comp_vect_possible_cpus; i++)
dd->comp_vect_mappings[i] = -1;
for (i = 0; i < dd->comp_vect_possible_cpus; i++) {
cpu = _dev_comp_vect_cpu_get(dd, entry, non_intr_cpus,
available_cpus); if (cpu < 0) {
ret = -EINVAL; goto fail;
}
dd->comp_vect_mappings[i] = cpu;
hfi1_cdbg(AFFINITY, "[%s] Completion Vector %d -> CPU %d",
rvt_get_ibdev_name(&(dd)->verbs_dev.rdi), i, cpu);
}
int hfi1_comp_vect_mappings_lookup(struct rvt_dev_info *rdi, int comp_vect)
{ struct hfi1_ibdev *verbs_dev = dev_from_rdi(rdi); struct hfi1_devdata *dd = dd_from_dev(verbs_dev);
if (!dd->comp_vect_mappings) return -EINVAL; if (comp_vect >= dd->comp_vect_possible_cpus) return -EINVAL;
return dd->comp_vect_mappings[comp_vect];
}
/* *Itassumesdd->comp_vect_possible_cpusisavailable.
*/ staticint _dev_comp_vect_cpu_mask_init(struct hfi1_devdata *dd, struct hfi1_affinity_node *entry, bool first_dev_init)
__must_hold(&node_affinity.lock)
{ int i, j, curr_cpu; int possible_cpus_comp_vect = 0; struct cpumask *dev_comp_vect_mask = &dd->comp_vect->mask;
lockdep_assert_held(&node_affinity.lock); /* *Ifthere'sonlyoneCPUavailableforcompletionvectors,then *therewillonlybeonecompletionvectoravailable.Othewise, *thenumberofcompletionvectoravailablewillbethenumberof *availableCPUsdivideitbythenumberofdevicesinthe *localNUMAnode.
*/ if (cpumask_weight(&entry->comp_vect_mask) == 1) {
possible_cpus_comp_vect = 1;
dd_dev_warn(dd, "Number of kernel receive queues is too large for completion vector affinity to be effective\n");
} else {
possible_cpus_comp_vect +=
cpumask_weight(&entry->comp_vect_mask) /
hfi1_per_node_cntr[dd->node];
/* *Itassumesdd->comp_vect_possible_cpusisavailable.
*/ staticvoid _dev_comp_vect_cpu_mask_clean_up(struct hfi1_devdata *dd, struct hfi1_affinity_node *entry)
__must_hold(&node_affinity.lock)
{ int i, cpu;
lockdep_assert_held(&node_affinity.lock); if (!dd->comp_vect_possible_cpus) return;
for (i = 0; i < dd->comp_vect_possible_cpus; i++) {
cpu = per_cpu_affinity_put_max(&dd->comp_vect->mask,
entry->comp_vect_affinity); /* Clearing CPU in device completion vector cpu mask */ if (cpu >= 0)
cpumask_clear_cpu(cpu, &dd->comp_vect->mask);
}
/* *IfthisisthefirsttimethisNUMAnode'saffinityisused, *createanentryintheglobalaffinitystructureandinitializeit.
*/ if (!entry) {
entry = node_affinity_allocate(dd->node); if (!entry) {
dd_dev_err(dd, "Unable to allocate global affinity node\n");
ret = -ENOMEM; goto fail;
}
new_entry = true;
init_cpu_mask_set(&entry->def_intr);
init_cpu_mask_set(&entry->rcv_intr);
cpumask_clear(&entry->comp_vect_mask);
cpumask_clear(&entry->general_intr_mask); /* Use the "real" cpu mask of this node as the default */
cpumask_and(&entry->def_intr.mask, &node_affinity.real_cpu_mask,
local_mask);
/* fill in the receive list */
possible = cpumask_weight(&entry->def_intr.mask);
curr_cpu = cpumask_first(&entry->def_intr.mask);
if (possible == 1) { /* only one CPU, everyone will use it */
cpumask_set_cpu(curr_cpu, &entry->rcv_intr.mask);
cpumask_set_cpu(curr_cpu, &entry->general_intr_mask);
} else { /* *Thegeneral/controlcontextwillbethefirstCPUin *thedefaultlist,soitisremovedfromthedefault *listandaddedtothegeneralinterruptlist.
*/
cpumask_clear_cpu(curr_cpu, &entry->def_intr.mask);
cpumask_set_cpu(curr_cpu, &entry->general_intr_mask);
curr_cpu = cpumask_next(curr_cpu,
&entry->def_intr.mask);
/* *Removetheremainingkernelreceivequeuesfrom *thedefaultlistandaddthemtothereceivelist.
*/ for (i = 0;
i < (dd->n_krcv_queues - 1) *
hfi1_per_node_cntr[dd->node];
i++) {
cpumask_clear_cpu(curr_cpu,
&entry->def_intr.mask);
cpumask_set_cpu(curr_cpu,
&entry->rcv_intr.mask);
curr_cpu = cpumask_next(curr_cpu,
&entry->def_intr.mask); if (curr_cpu >= nr_cpu_ids) break;
}
/* *Ifthereendsupbeing0CPUcoresleftoverforSDMA *engines,usethesameCPUcoresasgeneral/control *context.
*/ if (cpumask_empty(&entry->def_intr.mask))
cpumask_copy(&entry->def_intr.mask,
&entry->general_intr_mask);
}
/* Determine completion vector CPUs for the entire node */
cpumask_and(&entry->comp_vect_mask,
&node_affinity.real_cpu_mask, local_mask);
cpumask_andnot(&entry->comp_vect_mask,
&entry->comp_vect_mask,
&entry->rcv_intr.mask);
cpumask_andnot(&entry->comp_vect_mask,
&entry->comp_vect_mask,
&entry->general_intr_mask);
/* *Ifthereendsupbeing0CPUcoresleftoverforcompletion *vectors,usethesameCPUcoreasthegeneral/control *context.
*/ if (cpumask_empty(&entry->comp_vect_mask))
cpumask_copy(&entry->comp_vect_mask,
&entry->general_intr_mask);
}
ret = _dev_comp_vect_cpu_mask_init(dd, entry, new_entry); if (ret < 0) goto fail;
switch (msix->type) { case IRQ_SDMA:
set = &entry->def_intr;
hfi1_cleanup_sdma_notifier(msix); break; case IRQ_GENERAL: /* Don't do accounting for general contexts */ break; case IRQ_RCVCTXT: { struct hfi1_ctxtdata *rcd = msix->arg;
/* Don't do accounting for control contexts */ if (rcd->ctxt != HFI1_CTRL_CTXT)
set = &entry->rcv_intr; break;
} case IRQ_NETDEVCTXT:
set = &entry->def_intr; break; default:
mutex_unlock(&node_affinity.lock); return;
}
if (set) {
cpumask_andnot(&set->used, &set->used, &msix->mask);
_cpu_mask_set_gen_dec(set);
}
/* This should be called with node_affinity.lock held */ staticvoid find_hw_thread_mask(uint hw_thread_no, cpumask_var_t hw_thread_mask, struct hfi1_affinity_node_list *affinity)
{ int curr_cpu;
uint num_cores;
ret = zalloc_cpumask_var(&diff, GFP_KERNEL); if (!ret) goto done;
ret = zalloc_cpumask_var(&hw_thread_mask, GFP_KERNEL); if (!ret) goto free_diff;
ret = zalloc_cpumask_var(&available_mask, GFP_KERNEL); if (!ret) goto free_hw_thread_mask;
ret = zalloc_cpumask_var(&intrs_mask, GFP_KERNEL); if (!ret) goto free_available_mask;
/* *IfHTcoresareenabled,identifywhichHWthreadswithinthe *physicalcoresshouldbeused.
*/ for (i = 0; i < affinity->num_core_siblings; i++) {
find_hw_thread_mask(i, hw_thread_mask, affinity);
/* *Ifthere'satleastoneavailablecoreforthisHW *threadnumber,stoplookingforacore. * *diffwillalwaysbenotemptyatleastonceinthis *loopastheusedmaskgetsresetwhen *(set->mask==set->used)beforethisloop.
*/ if (cpumask_andnot(diff, hw_thread_mask, &set->used)) break;
}
hfi1_cdbg(PROC, "Same available HW thread on all physical CPUs: %*pbl",
cpumask_pr_args(hw_thread_mask));
node_mask = cpumask_of_node(node);
hfi1_cdbg(PROC, "Device on NUMA %u, CPUs %*pbl", node,
cpumask_pr_args(node_mask));
/* Get cpumask of available CPUs on preferred NUMA */
cpumask_and(available_mask, hw_thread_mask, node_mask);
cpumask_andnot(available_mask, available_mask, &set->used);
hfi1_cdbg(PROC, "Available CPUs on NUMA %u: %*pbl", node,
cpumask_pr_args(available_mask));
/* If we don't have CPUs on the preferred node, use other NUMA nodes */ if (cpumask_empty(available_mask)) {
cpumask_andnot(available_mask, hw_thread_mask, &set->used); /* Excluding preferred NUMA cores */
cpumask_andnot(available_mask, available_mask, node_mask);
hfi1_cdbg(PROC, "Preferred NUMA node cores are taken, cores available in other NUMA nodes: %*pbl",
cpumask_pr_args(available_mask));
/* *Atfirst,wedon'twanttoplaceprocessesonthesame *CPUsasinterrupthandlers.
*/ if (cpumask_andnot(diff, available_mask, intrs_mask))
cpumask_copy(available_mask, diff);
}
hfi1_cdbg(PROC, "Possible CPUs for process: %*pbl",
cpumask_pr_args(available_mask));
cpu = cpumask_first(available_mask); if (cpu >= nr_cpu_ids) /* empty */
cpu = -1; else
cpumask_set_cpu(cpu, &set->used);
mutex_unlock(&affinity->lock);
hfi1_cdbg(PROC, "Process assigned to CPU %d", cpu);
mutex_lock(&affinity->lock);
cpu_mask_set_put(set, cpu);
hfi1_cdbg(PROC, "Returning CPU %d for future process assignment", cpu);
mutex_unlock(&affinity->lock);
}
Messung V0.5 in Prozent
¤ Dauer der Verarbeitung: 0.53 Sekunden
(vorverarbeitet am 2026-10-11)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.