/* Returns true if compaction should be skipped this time */ staticbool compaction_deferred(struct zone *zone, int order)
{ unsignedlong defer_limit = 1UL << zone->compact_defer_shift;
if (order < zone->compact_order_failed) returnfalse;
/* Avoid possible overflow */ if (++zone->compact_considered >= defer_limit) {
zone->compact_considered = defer_limit; returnfalse;
}
trace_mm_compaction_deferred(zone, order);
returntrue;
}
/* *Updatedefertrackingcountersaftersuccessfulcompactionofgivenorder, *whichmeansanallocationeithersucceeded(alloc_success==true)oris *expectedtosucceed.
*/ void compaction_defer_reset(struct zone *zone, int order, bool alloc_success)
{ if (alloc_success) {
zone->compact_considered = 0;
zone->compact_defer_shift = 0;
} if (order >= zone->compact_order_failed)
zone->compact_order_failed = order + 1;
trace_mm_compaction_defer_reset(zone, order);
}
/* Returns true if restarting compaction after many failures */ staticbool compaction_restarting(struct zone *zone, int order)
{ if (order < zone->compact_order_failed) returnfalse;
/* Returns true if the pageblock should be scanned for pages to isolate. */ staticinlinebool isolation_suitable(struct compact_control *cc, struct page *page)
{ if (cc->ignore_skip_hint) returntrue;
/* Ensure the start of the pageblock or zone is online and valid */
block_pfn = pageblock_start_pfn(pfn);
block_pfn = max(block_pfn, zone->zone_start_pfn);
block_page = pfn_to_online_page(block_pfn); if (block_page) {
page = block_page;
pfn = block_pfn;
}
/* Ensure the end of the pageblock or zone is online and valid */
block_pfn = pageblock_end_pfn(pfn) - 1;
block_pfn = min(block_pfn, zone_end_pfn(zone) - 1);
end_page = pfn_to_online_page(block_pfn); if (!end_page) returnfalse;
/* *Onlyclearthehintifasampleindicatesthereiseithera *freepageoranLRUpageintheblock.Oneorothercondition *isnecessaryfortheblocktobeamigrationsource/target.
*/ do { if (check_source && PageLRU(page)) {
clear_pageblock_skip(page); returntrue;
}
if (check_target && PageBuddy(page)) {
clear_pageblock_skip(page); returntrue;
}
page += (1 << PAGE_ALLOC_COSTLY_ORDER);
} while (page <= end_page);
/* If we already hold the lock, we can skip some rechecking. */ if (!locked) {
locked = compact_lock_irqsave(&cc->zone->lock,
&flags, cc);
/* Recheck this is a buddy page under lock */ if (!PageBuddy(page)) goto isolate_fail;
}
/* Found a free page, will break it into order-0 pages */
order = buddy_order(page);
isolated = __isolate_free_page(page, order); if (!isolated) break;
set_page_private(page, order);
/* We don't use freelists for anything. */ return pfn;
}
/* Similar to reclaim, but different enough that they don't share logic */ staticbool too_many_isolated(struct compact_control *cc)
{
pg_data_t *pgdat = cc->zone->zone_pgdat; bool too_many;
/* *EnsurethattherearenottoomanypagesisolatedfromtheLRU *listbyeitherparallelreclaimersorcompaction.Ifthereare, *delayforsometimeuntilfewerpagesareisolated
*/ while (unlikely(too_many_isolated(cc))) { /* stop isolation if there are still pages not migrated */ if (cc->nr_migratepages) return -EAGAIN;
/* async migration should just abort */ if (cc->mode == MIGRATE_ASYNC) return -EAGAIN;
if (PageHuge(page)) { constunsignedint order = compound_order(page); /* *skiphugetlbfsifwearenotcompactingforpages *biggerthanitsorder.THPsandothercompoundpages *arehandledbelow.
*/ if (!cc->alloc_contig) {
if (order <= MAX_PAGE_ORDER) {
low_pfn += (1UL << order) - 1;
nr_scanned += (1UL << order) - 1;
} goto isolate_fail;
} /* for alloc_contig case */ if (locked) {
unlock_page_lruvec_irqrestore(locked, flags);
locked = NULL;
}
folio = page_folio(page);
ret = isolate_or_dissolve_huge_folio(folio, &cc->migratepages);
/* *Failisolationincaseisolate_or_dissolve_huge_folio() *reportsanerror.Incaseof-ENOMEM,abortrightaway.
*/ if (ret < 0) { /* Do not report -EBUSY down the chain */ if (ret == -EBUSY)
ret = 0;
low_pfn += (1UL << order) - 1;
nr_scanned += (1UL << order) - 1; goto isolate_fail;
}
fatal_pending:
cc->total_migrate_scanned += nr_scanned; if (nr_isolated)
count_compact_events(COMPACTISOLATED, nr_isolated);
cc->migrate_pfn = low_pfn;
return ret;
}
/** *isolate_migratepages_range()-isolatemigrate-ablepagesinaPFNrange *@cc:Compactioncontrolstructure. *@start_pfn:ThefirstPFNtostartisolating. *@end_pfn:Theone-past-lastPFN. * *Returns-EAGAINwhencontented,-EINTRincaseofasignalpending,-ENOMEM *incasewecouldnotallocateapage,or0.
*/ int
isolate_migratepages_range(struct compact_control *cc, unsignedlong start_pfn, unsignedlong end_pfn)
{ unsignedlong pfn, block_start_pfn, block_end_pfn; int ret = 0;
/* Scan block by block. First and last block may be incomplete */
pfn = start_pfn;
block_start_pfn = pageblock_start_pfn(pfn); if (block_start_pfn < cc->zone->zone_start_pfn)
block_start_pfn = cc->zone->zone_start_pfn;
block_end_pfn = pageblock_end_pfn(pfn);
/* Returns true if the page is within a block suitable for migration to */ staticbool suitable_migration_target(struct compact_control *cc, struct page *page)
{ /* If the page is a large free page, then disallow migration */ if (PageBuddy(page)) { int order = cc->order > 0 ? cc->order : pageblock_order;
/* *Basedoninformationinthecurrentcompact_control,findblocks *suitableforisolatingfreepagesfromandthenisolatethem.
*/ staticvoid isolate_freepages(struct compact_control *cc)
{ struct zone *zone = cc->zone; struct page *page; unsignedlong block_start_pfn; /* start of current pageblock */ unsignedlong isolate_start_pfn; /* exact pfn we start at */ unsignedlong block_end_pfn; /* end of current pageblock */ unsignedlong low_pfn; /* lowest pfn scanner is able to scan */ unsignedint stride;
/* Try a small search of the free lists for a candidate */
fast_isolate_freepages(cc); if (cc->nr_freepages) return;
page = pageblock_pfn_to_page(block_start_pfn, block_end_pfn,
zone); if (!page) { unsignedlong next_pfn;
next_pfn = skip_offline_sections_reverse(block_start_pfn); if (next_pfn)
block_start_pfn = max(next_pfn, low_pfn);
continue;
}
/* Check the block is suitable for migration */ if (!suitable_migration_target(cc, page)) continue;
/* If isolation recently failed, do not retry */ if (!isolation_suitable(cc, page)) continue;
/* Found a block suitable for isolating free pages from. */
nr_isolated = isolate_freepages_block(cc, &isolate_start_pfn,
block_end_pfn, cc->freepages, stride, false);
/* Update the skip hint if the full pageblock was scanned */ if (isolate_start_pfn == block_end_pfn)
update_pageblock_skip(cc, page, block_start_pfn -
pageblock_nr_pages);
again: for (start_order = order; start_order < NR_PAGE_ORDERS; start_order++) if (!list_empty(&cc->freepages[start_order])) break;
/* no free pages in the list */ if (start_order == NR_PAGE_ORDERS) { if (has_isolated_pages) return NULL;
isolate_freepages(cc);
has_isolated_pages = true; goto again;
}
/*
* fast_find_migrateblock() has already ensured the pageblock is not
* set with a skipped flag, so to avoid the isolation_suitable check
* below again, check whether the fast search was successful.
*/
fast_find_block = low_pfn != cc->migrate_pfn && !cc->fast_search_fail;
/* Only scan within a pageblock boundary */
block_end_pfn = pageblock_end_pfn(low_pfn);
/*
* Iterate over whole pageblocks until we find the first suitable.
* Do not cross the free scanner.
*/
for (; block_end_pfn <= cc->free_pfn;
fast_find_block = false,
cc->migrate_pfn = low_pfn = block_end_pfn,
block_start_pfn = block_end_pfn,
block_end_pfn += pageblock_nr_pages) {
/*
* This can potentially iterate a massively long zone with
* many pageblocks unsuitable, so periodically check if we
* need to schedule.
*/
if (!(low_pfn % (COMPACT_CLUSTER_MAX * pageblock_nr_pages)))
cond_resched();
page = pageblock_pfn_to_page(block_start_pfn,
block_end_pfn, cc->zone);
if (!page) {
unsigned long next_pfn;
/*
* If isolation recently failed, do not retry. Only check the
* pageblock once. COMPACT_CLUSTER_MAX causes a pageblock
* to be visited multiple times. Assume skip was checked
* before making it "skip" so other compaction instances do
* not scan the same block.
*/
if ((pageblock_aligned(low_pfn) ||
low_pfn == cc->zone->zone_start_pfn) &&
!fast_find_block && !isolation_suitable(cc, page))
continue;
/*
* For async direct compaction, only scan the pageblocks of the
* same migratetype without huge pages. Async direct compaction
* is optimistic to see if the minimum amount of work satisfies
* the allocation. The cached PFN is updated as it's possible
* that all remaining blocks between source and target are
* unsuitable and the compaction scanners fail to meet.
*/
if (!suitable_migration_source(cc, page)) {
update_cached_migrate(cc, block_end_pfn);
continue;
}
/* Perform the isolation */
if (isolate_migratepages_block(cc, low_pfn, block_end_pfn,
isolate_mode))
return ISOLATE_ABORT;
/*
* Either we isolated something and proceed with migration. Or
* we failed and compact_zone should decide if we should
* continue or not.
*/
break;
}
/*
* Determine whether kswapd is (or recently was!) running on this node.
*
* pgdat_kswapd_lock() pins pgdat->kswapd, so a concurrent kswapd_stop() can't
* zero it.
*/
static bool kswapd_is_running(pg_data_t *pgdat)
{
bool running;
/*
* A zone's fragmentation score is the external fragmentation wrt to the
* COMPACTION_HPAGE_ORDER. It returns a value in the range [0, 100].
*/
static unsigned int fragmentation_score_zone(struct zone *zone)
{
return extfrag_for_order(zone, COMPACTION_HPAGE_ORDER);
}
/*
* A weighted zone's fragmentation score is the external fragmentation
* wrt to the COMPACTION_HPAGE_ORDER scaled by the zone's size. It
* returns a value in the range [0, 100].
*
* The scaling factor ensures that proactive compaction focuses on larger
* zones like ZONE_NORMAL, rather than smaller, specialized zones like
* ZONE_DMA32. For smaller zones, the score value remains close to zero,
* and thus never exceeds the high threshold for proactive compaction.
*/
static unsigned int fragmentation_score_zone_weighted(struct zone *zone)
{
unsigned long score;
/*
* The per-node proactive (background) compaction process is started by its
* corresponding kcompactd thread when the node's fragmentation score
* exceeds the high threshold. The compaction process remains active till
* the node's score falls below the low threshold, or one of the back-off
* conditions is met.
*/
static unsigned int fragmentation_score_node(pg_data_t *pgdat)
{
unsigned int score = 0;
int zoneid;
for (zoneid = 0; zoneid < MAX_NR_ZONES; zoneid++) {
struct zone *zone;
zone = &pgdat->node_zones[zoneid];
if (!populated_zone(zone))
continue;
score += fragmentation_score_zone_weighted(zone);
}
return score;
}
static unsigned int fragmentation_score_wmark(bool low)
{
unsigned int wmark_low, leeway;
static enum compact_result __compact_finished(struct compact_control *cc)
{
unsigned int order;
const int migratetype = cc->migratetype;
int ret;
/* Compaction run completes if the migrate and free scanner meet */
if (compact_scanners_met(cc)) {
/* Let the next compaction start anew. */
reset_cached_positions(cc->zone);
/*
* Mark that the PG_migrate_skip information should be cleared
* by kswapd when it goes to sleep. kcompactd does not set the
* flag itself as the decision to be clear should be directly
* based on an allocation request.
*/
if (cc->direct_compaction)
cc->zone->compact_blockskip_flush = true;
if (cc->whole_zone)
return COMPACT_COMPLETE;
else
return COMPACT_PARTIAL_SKIPPED;
}
if (cc->proactive_compaction) {
int score, wmark_low;
pg_data_t *pgdat;
pgdat = cc->zone->zone_pgdat;
if (kswapd_is_running(pgdat))
return COMPACT_PARTIAL_SKIPPED;
if (score > wmark_low)
ret = COMPACT_CONTINUE;
else
ret = COMPACT_SUCCESS;
goto out;
}
if (is_via_compact_memory(cc->order))
return COMPACT_CONTINUE;
/*
* Always finish scanning a pageblock to reduce the possibility of
* fallbacks in the future. This is particularly important when
* migration source is unmovable/reclaimable but it's not worth
* special casing.
*/
if (!pageblock_aligned(cc->migrate_pfn))
return COMPACT_CONTINUE;
/*
* When defrag_mode is enabled, make kcompactd target
* watermarks in whole pageblocks. Because they can be stolen
* without polluting, no further fallback checks are needed.
*/
if (defrag_mode && !cc->direct_compaction) {
if (__zone_watermark_ok(cc->zone, cc->order,
high_wmark_pages(cc->zone),
cc->highest_zoneidx, cc->alloc_flags,
zone_page_state(cc->zone,
NR_FREE_PAGES_BLOCKS)))
return COMPACT_SUCCESS;
return COMPACT_CONTINUE;
}
/* Direct compactor: Is a suitable page free? */
ret = COMPACT_NO_SUITABLE_PAGE;
for (order = cc->order; order < NR_PAGE_ORDERS; order++) {
struct free_area *area = &cc->zone->free_area[order];
/* Job done if page is free of the right migratetype */
if (!free_area_empty(area, migratetype))
return COMPACT_SUCCESS;
#ifdef CONFIG_CMA
/* MIGRATE_MOVABLE can fallback on MIGRATE_CMA */
if (migratetype == MIGRATE_MOVABLE &&
!free_area_empty(area, MIGRATE_CMA))
return COMPACT_SUCCESS;
#endif
/*
* Job done if allocation would steal freepages from
* other migratetype buddy lists.
*/
if (find_suitable_fallback(area, order, migratetype, true) >= 0)
/*
* Movable pages are OK in any pageblock. If we are
* stealing for a non-movable allocation, make sure
* we finish compacting the current pageblock first
* (which is assured by the above migrate_pfn align
* check) so it is as free as possible and we won't
* have to steal another one soon.
*/
return COMPACT_SUCCESS;
}
out:
if (cc->contended || fatal_signal_pending(current))
ret = COMPACT_CONTENDED;
return ret;
}
static enum compact_result compact_finished(struct compact_control *cc)
{
int ret;
ret = __compact_finished(cc);
trace_mm_compaction_finished(cc->zone, cc->order, ret);
if (ret == COMPACT_NO_SUITABLE_PAGE)
ret = COMPACT_CONTINUE;
return ret;
}
static bool __compaction_suitable(struct zone *zone, int order,
unsigned long watermark, int highest_zoneidx,
unsigned long free_pages)
{
/*
* Watermarks for order-0 must be met for compaction to be able to
* isolate free pages for migration targets. This means that the
* watermark have to match, or be more pessimistic than the check in
* __isolate_free_page().
*
* For costly orders, we require a higher watermark for compaction to
* proceed to increase its chances.
*
* We use the direct compactor's highest_zoneidx to skip over zones
* where lowmem reserves would prevent allocation even if compaction
* succeeds.
*
* ALLOC_CMA is used, as pages in CMA pageblocks are considered
* suitable migration targets.
*/
watermark += compact_gap(order);
if (order > PAGE_ALLOC_COSTLY_ORDER)
watermark += low_wmark_pages(zone) - min_wmark_pages(zone);
return __zone_watermark_ok(zone, 0, watermark, highest_zoneidx,
ALLOC_CMA, free_pages);
}
/*
* compaction_suitable: Is this suitable to run compaction on this zone now?
*/
bool compaction_suitable(struct zone *zone, int order, unsigned long watermark,
int highest_zoneidx)
{
enum compact_result compact_result;
bool suitable;
suitable = __compaction_suitable(zone, order, watermark, highest_zoneidx,
zone_page_state(zone, NR_FREE_PAGES));
/*
* fragmentation index determines if allocation failures are due to
* low memory or external fragmentation
*
* index of -1000 would imply allocations might succeed depending on
* watermarks, but we already failed the high-order watermark check
* index towards 0 implies failure is due to lack of memory
* index towards 1000 implies failure is due to fragmentation
*
* Only compact if a failure would be due to fragmentation. Also
* ignore fragindex for non-costly orders where the alternative to
* a successful reclaim/compaction is OOM. Fragindex and the
* vm.extfrag_threshold sysctl is meant as a heuristic to prevent
* excessive compaction for costly orders, but it should not be at the
* expense of system stability.
*/
if (suitable) {
compact_result = COMPACT_CONTINUE;
if (order > PAGE_ALLOC_COSTLY_ORDER) {
int fragindex = fragmentation_index(zone, order);
/* Used by direct reclaimers */
bool compaction_zonelist_suitable(struct alloc_context *ac, int order,
int alloc_flags)
{
struct zone *zone;
struct zoneref *z;
/*
* Make sure at least one zone would pass __compaction_suitable if we continue
* retrying the reclaim.
*/
for_each_zone_zonelist_nodemask(zone, z, ac->zonelist,
ac->highest_zoneidx, ac->nodemask) {
unsigned long available;
/*
* Do not consider all the reclaimable memory because we do not
* want to trash just for a single high order allocation which
* is even not guaranteed to appear even if __compaction_suitable
* is happy about the watermark check.
*/
available = zone_reclaimable_pages(zone) / order;
available += zone_page_state_snapshot(zone, NR_FREE_PAGES);
if (__compaction_suitable(zone, order, min_wmark_pages(zone),
ac->highest_zoneidx, available))
return true;
}
return false;
}
/*
* Should we do compaction for target allocation order.
* Return COMPACT_SUCCESS if allocation for target order can be already
* satisfied
* Return COMPACT_SKIPPED if compaction for target order is likely to fail
* Return COMPACT_CONTINUE if compaction for target order should be ran
*/
static enum compact_result
compaction_suit_allocation_order(struct zone *zone, unsigned int order,
int highest_zoneidx, unsigned int alloc_flags,
bool async, bool kcompactd)
{
unsigned long free_pages;
unsigned long watermark;
/*
* For unmovable allocations (without ALLOC_CMA), check if there is enough
* free memory in the non-CMA pageblocks. Otherwise compaction could form
* the high-order page in CMA pageblocks, which would not help the
* allocation to succeed. However, limit the check to costly order async
* compaction (such as opportunistic THP attempts) because there is the
* possibility that compaction would migrate pages from non-CMA to CMA
* pageblock.
*/
if (order > PAGE_ALLOC_COSTLY_ORDER && async &&
!(alloc_flags & ALLOC_CMA)) {
if (!__zone_watermark_ok(zone, 0, watermark + compact_gap(order),
highest_zoneidx, 0,
zone_page_state(zone, NR_FREE_PAGES)))
return COMPACT_SKIPPED;
}
if (!compaction_suitable(zone, order, watermark, highest_zoneidx))
return COMPACT_SKIPPED;
return COMPACT_CONTINUE;
}
static enum compact_result
compact_zone(struct compact_control *cc, struct capture_control *capc)
{
enum compact_result ret;
unsigned long start_pfn = cc->zone->zone_start_pfn;
unsigned long end_pfn = zone_end_pfn(cc->zone);
unsigned long last_migrated_pfn;
const bool sync = cc->mode != MIGRATE_ASYNC;
bool update_cached;
unsigned int nr_succeeded = 0, nr_migratepages;
int order;
/*
* These counters track activities during zone compaction. Initialize
* them before compacting a new zone.
*/
cc->total_migrate_scanned = 0;
cc->total_free_scanned = 0;
cc->nr_migratepages = 0;
cc->nr_freepages = 0;
for (order = 0; order < NR_PAGE_ORDERS; order++)
INIT_LIST_HEAD(&cc->freepages[order]);
INIT_LIST_HEAD(&cc->migratepages);
cc->migratetype = gfp_migratetype(cc->gfp_mask);
if (!is_via_compact_memory(cc->order)) {
ret = compaction_suit_allocation_order(cc->zone, cc->order,
cc->highest_zoneidx,
cc->alloc_flags,
cc->mode == MIGRATE_ASYNC,
!cc->direct_compaction);
if (ret != COMPACT_CONTINUE)
return ret;
}
/*
* Clear pageblock skip if there were failures recently and compaction
* is about to be retried after being deferred.
*/
if (compaction_restarting(cc->zone, cc->order))
__reset_isolation_suitable(cc->zone);
/*
* Setup to move all movable pages to the end of the zone. Used cached
* information on where the scanners should start (unless we explicitly
* want to compact the whole zone), but check that it is initialised
* by ensuring the values are within zone boundaries.
*/
cc->fast_start_pfn = 0;
if (cc->whole_zone) {
cc->migrate_pfn = start_pfn;
cc->free_pfn = pageblock_start_pfn(end_pfn - 1);
} else {
cc->migrate_pfn = cc->zone->compact_cached_migrate_pfn[sync];
cc->free_pfn = cc->zone->compact_cached_free_pfn;
if (cc->free_pfn < start_pfn || cc->free_pfn >= end_pfn) {
cc->free_pfn = pageblock_start_pfn(end_pfn - 1);
cc->zone->compact_cached_free_pfn = cc->free_pfn;
}
if (cc->migrate_pfn < start_pfn || cc->migrate_pfn >= end_pfn) {
cc->migrate_pfn = start_pfn;
cc->zone->compact_cached_migrate_pfn[0] = cc->migrate_pfn;
cc->zone->compact_cached_migrate_pfn[1] = cc->migrate_pfn;
}
if (cc->migrate_pfn <= cc->zone->compact_init_migrate_pfn)
cc->whole_zone = true;
}
last_migrated_pfn = 0;
/*
* Migrate has separate cached PFNs for ASYNC and SYNC* migration on
* the basis that some migrations will fail in ASYNC mode. However,
* if the cached PFNs match and pageblocks are skipped due to having
* no isolation candidates, then the sync state does not matter.
* Until a pageblock with isolation candidates is found, keep the
* cached PFNs in sync to avoid revisiting the same blocks.
*/
update_cached = !sync &&
cc->zone->compact_cached_migrate_pfn[0] == cc->zone->compact_cached_migrate_pfn[1];
/* lru_add_drain_all could be expensive with involving other CPUs */
lru_add_drain();
while ((ret = compact_finished(cc)) == COMPACT_CONTINUE) {
int err;
unsigned long iteration_start_pfn = cc->migrate_pfn;
/*
* Avoid multiple rescans of the same pageblock which can
* happen if a page cannot be isolated (dirty/writeback in
* async mode) or if the migrated pages are being allocated
* before the pageblock is cleared. The first rescan will
* capture the entire pageblock for migration. If it fails,
* it'll be marked skip and scanning will proceed as normal.
*/
cc->finish_pageblock = false;
if (pageblock_start_pfn(last_migrated_pfn) ==
pageblock_start_pfn(iteration_start_pfn)) {
cc->finish_pageblock = true;
}
rescan:
switch (isolate_migratepages(cc)) {
case ISOLATE_ABORT:
ret = COMPACT_CONTENDED;
putback_movable_pages(&cc->migratepages);
cc->nr_migratepages = 0;
goto out;
case ISOLATE_NONE:
if (update_cached) {
cc->zone->compact_cached_migrate_pfn[1] =
cc->zone->compact_cached_migrate_pfn[0];
}
/*
* We haven't isolated and migrated anything, but
* there might still be unflushed migrations from
* previous cc->order aligned block.
*/
goto check_drain;
case ISOLATE_SUCCESS:
update_cached = false;
last_migrated_pfn = max(cc->zone->zone_start_pfn,
pageblock_start_pfn(cc->migrate_pfn - 1));
}
/*
* Record the number of pages to migrate since the
* compaction_alloc/free() will update cc->nr_migratepages
* properly.
*/
nr_migratepages = cc->nr_migratepages;
err = migrate_pages(&cc->migratepages, compaction_alloc,
compaction_free, (unsigned long)cc, cc->mode,
MR_COMPACTION, &nr_succeeded);
/* All pages were either migrated or will be released */
cc->nr_migratepages = 0;
if (err) {
putback_movable_pages(&cc->migratepages);
/*
* migrate_pages() may return -ENOMEM when scanners meet
* and we want compact_finished() to detect it
*/
if (err == -ENOMEM && !compact_scanners_met(cc)) {
ret = COMPACT_CONTENDED;
goto out;
}
/*
* If an ASYNC or SYNC_LIGHT fails to migrate a page
* within the pageblock_order-aligned block and
* fast_find_migrateblock may be used then scan the
* remainder of the pageblock. This will mark the
* pageblock "skip" to avoid rescanning in the near
* future. This will isolate more pages than necessary
* for the request but avoid loops due to
* fast_find_migrateblock revisiting blocks that were
* recently partially scanned.
*/
if (!pageblock_aligned(cc->migrate_pfn) &&
!cc->ignore_skip_hint && !cc->finish_pageblock &&
(cc->mode < MIGRATE_SYNC)) {
cc->finish_pageblock = true;
/*
* Draining pcplists does not help THP if
* any page failed to migrate. Even after
* drain, the pageblock will not be free.
*/
if (cc->order == COMPACTION_HPAGE_ORDER)
last_migrated_pfn = 0;
goto rescan;
}
}
/* Stop if a page has been captured */
if (capc && capc->page) {
ret = COMPACT_SUCCESS;
break;
}
check_drain:
/*
* Has the migration scanner moved away from the previous
* cc->order aligned block where we migrated from? If yes,
* flush the pages that were freed, so that they can merge and
* compact_finished() can detect immediately if allocation
* would succeed.
*/
if (cc->order > 0 && last_migrated_pfn) {
unsigned long current_block_start =
block_start_pfn(cc->migrate_pfn, cc->order);
if (last_migrated_pfn < current_block_start) {
lru_add_drain_cpu_zone(cc->zone);
/* No more flushing until we migrate again */
last_migrated_pfn = 0;
}
}
}
out:
/*
* Release free pages and update where the free scanner should restart,
* so we don't leave any returned pages behind in the next attempt.
*/
if (cc->nr_freepages > 0) {
unsigned long free_pfn = release_free_list(cc->freepages);
cc->nr_freepages = 0;
VM_BUG_ON(free_pfn == 0);
/* The cached pfn is always the first in a pageblock */
free_pfn = pageblock_start_pfn(free_pfn);
/*
* Only go back, not forward. The cached pfn might have been
* already reset to zone end in compact_finished()
*/
if (free_pfn > cc->zone->compact_cached_free_pfn)
cc->zone->compact_cached_free_pfn = free_pfn;
}
/*
* Make sure the structs are really initialized before we expose the
* capture control, in case we are interrupted and the interrupt handler
* frees a page.
*/
barrier();
WRITE_ONCE(current->capture_control, &capc);
ret = compact_zone(&cc, &capc);
/*
* Make sure we hide capture control first before we read the captured
* page pointer, otherwise an interrupt could free and capture a page
* and we would leak it.
*/
WRITE_ONCE(current->capture_control, NULL);
*capture = READ_ONCE(capc.page);
/*
* Technically, it is also possible that compaction is skipped but
* the page is still captured out of luck(IRQ came and freed the page).
* Returning COMPACT_SUCCESS in such cases helps in properly accounting
* the COMPACT[STALL|FAIL] when compaction is skipped.
*/
if (*capture)
ret = COMPACT_SUCCESS;
return ret;
}
/**
* try_to_compact_pages - Direct compact to satisfy a high-order allocation
* @gfp_mask: The GFP mask of the current allocation
* @order: The order of the current allocation
* @alloc_flags: The allocation flags of the current allocation
* @ac: The context of current allocation
* @prio: Determines how hard direct compaction should try to succeed
* @capture: Pointer to free page created by compaction will be stored here
*
* This is the main entry point for direct page compaction.
*/
enum compact_result try_to_compact_pages(gfp_t gfp_mask, unsigned int order,
unsigned int alloc_flags, const struct alloc_context *ac,
enum compact_priority prio, struct page **capture)
{
struct zoneref *z;
struct zone *zone;
enum compact_result rc = COMPACT_SKIPPED;
if (!gfp_compaction_allowed(gfp_mask))
return COMPACT_SKIPPED;
/* Compact each zone in the list */
for_each_zone_zonelist_nodemask(zone, z, ac->zonelist,
ac->highest_zoneidx, ac->nodemask) {
enum compact_result status;
if (cpusets_enabled() &&
(alloc_flags & ALLOC_CPUSET) &&
!__cpuset_zone_allowed(zone, gfp_mask))
continue;
/* The allocation should succeed, stop compacting */
if (status == COMPACT_SUCCESS) {
/*
* We think the allocation will succeed in this zone,
* but it is not certain, hence the false. The caller
* will repeat this with true if allocation indeed
* succeeds in this zone.
*/
compaction_defer_reset(zone, order, false);
break;
}
if (prio != COMPACT_PRIO_ASYNC && (status == COMPACT_COMPLETE ||
status == COMPACT_PARTIAL_SKIPPED))
/*
* We think that allocation won't succeed in this zone
* so we defer compaction there. If it ends up
* succeeding after all, it will be reset.
*/
defer_compaction(zone, order);
/*
* We might have stopped compacting due to need_resched() in
* async compaction, or due to a fatal signal detected. In that
* case do not try further zones
*/
if ((prio == COMPACT_PRIO_ASYNC && need_resched())
|| fatal_signal_pending(current))
break;
}
return rc;
}
/*
* compact_node() - compact all zones within a node
* @pgdat: The node page data
* @proactive: Whether the compaction is proactive
*
* For proactive compaction, compact till each zone's fragmentation score
* reaches within proactive compaction thresholds (as determined by the
* proactiveness tunable), it is possible that the function returns before
* reaching score targets due to various back-off conditions, such as,
* contention on per-node or per-zone locks.
*/
static int compact_node(pg_data_t *pgdat, bool proactive)
{
int zoneid;
struct zone *zone;
struct compact_control cc = {
.order = -1,
.mode = proactive ? MIGRATE_SYNC_LIGHT : MIGRATE_SYNC,
.ignore_skip_hint = true,
.whole_zone = true,
.gfp_mask = GFP_KERNEL,
.proactive_compaction = proactive,
};
for (zoneid = 0; zoneid < MAX_NR_ZONES; zoneid++) {
zone = &pgdat->node_zones[zoneid];
if (!populated_zone(zone))
continue;
if (fatal_signal_pending(current))
return -EINTR;
cc.zone = zone;
java.lang.StringIndexOutOfBoundsException: Range [15, 14) out of bounds for length 26
if (proactive) {
count_compact_events(java.lang.StringIndexOutOfBoundsException: Range [0, 49) out of bounds for length 28
cc.total_migrate_scanned);
count_compact_events(KCOMPACTD_FREE_SCANNED,
t)java.lang.StringIndexOutOfBoundsException: Index 33 out of bounds for length 33
}
"ل}
java.lang.StringIndexOutOfBoundsException: Range [8, 7) out of bounds for length 10
java.lang.StringIndexOutOfBoundsException: Index 1 out of bounds for length 1
/ Compactall zones allnodes the system*
static int compact_nodes(void)
{
java.lang.StringIndexOutOfBoundsException: Range [5, 4) out of bounds for length 14
/* Flush lag{lag{"ل"}
lru_add_drain_all();
for_each_online_nodeن
ret"ن}
if (ret)
return ret;
java.lang.StringIndexOutOfBoundsException: Index 2 out of bounds for length 2
return 0;
}
staticjava.lang.StringIndexOutOfBoundsException: Range [51, 50) out of bounds for length 92
void *uffer *ength,loff_t pjava.lang.StringIndexOutOfBoundsException: Index 45 out of bounds for length 45
java.lang.StringIndexOutOfBoundsException: Index 8 out of bounds for length 1
int rc, nid;
rc = proc_dointvec_minmax(table, java.lang.StringIndexOutOfBoundsException: Index 37 out of bounds for length 24
if (rc)
returnrcjava.lang.StringIndexOutOfBoundsException: Index 12 out of bounds for length 12
ادن"
{"اكونغو اسالة"
pg_data_t *syr{"السريا
if (pgdat->"لت"
continue;ااا"
-proactive_compact_trigger true;
trace_mm_compaction_wakeup_kcompactdtmh{"التاايك}
to{التني"java.lang.StringIndexOutOfBoundsException: Index 22 out of bounds for length 22
up_interruptible(pgdat-);
}
}"لو}
;
}
*
* اوات}
* /proc/sys/
*/
static int sysctl_compaction_handler{يjava.lang.StringIndexOutOfBoundsException: Index 19 out of bounds for length 19
void *zen{"لزنج"
{
;
ret = proc_dointvec(table, write, buffer, length}
(etjava.lang.StringIndexOutOfBoundsException: Index 9 out of bounds for length 9
return ret;
int compaction_register_node(struct node *nodeJpan"لاة}
{
return device_create_file(&node->dev, &dev_attr_compact
}
struct )
device_remove_file(&node-و"
java.lang.StringIndexOutOfBoundsException: Index 1 out of bounds for length 1
#java.lang.StringIndexOutOfBoundsException: Range [7, 6) out of bounds for length 40
tatic inline bool kcompactd_work_requested(pg_data_t *pgdat)
{
return java.lang.StringIndexOutOfBoundsException: Index 23 out of bounds for length 23
pgdat>;
}
static bool kcompactd_node_suitable
{
int zoneid;
java.lang.StringIndexOutOfBoundsException: Range [14, 12) out of bounds for length 19
enum zone_type highest_zoneidx Zjava.lang.StringIndexOutOfBoundsException: Range [13, 12) out of bounds for length 27
enum compact_result ret;
unsigned int alloc_flags = java.lang.StringIndexOutOfBoundsException: Index 30 out of bounds for length 29
ALLOC_WMARK_HIGH
{"التقjava.lang.StringIndexOutOfBoundsException: Index 36 out of bounds for length 36
zone = &pgdat->node_zones[zoneid];
if(populated_zonezone)java.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 28
continue
ret =-{يم لجري(ملقر)}
,
java.lang.StringIndexOutOfBoundsException: Range [33, 32) out of bounds for length 33
false,ethiopic-amete-alem{أمتيأليمالjava.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 28
if (ret == COMPACT_CONTINUE)
return true;
}
}
static void kcompactd_do_work(pg_data_t "اسع("
{
/*
java.lang.StringIndexOutOfBoundsException: Range [12, 11) out of bounds for length 71
orderis allocatable.
*/
int zoneid;
struct zone
struct compact_control cc = {
.order = pgdat-java.lang.StringIndexOutOfBoundsException: Index 9 out of bounds for length 9
.search_order"لرا اري لني لمدة"
.highest_zoneidx = pgdat->kcompactd_highest_zoneidx java.lang.StringIndexOutOfBoundsException: Range [20, 19) out of bounds for length 48
mode =,
.ignore_skip_hint = fullwide{"أرقامرامكل لرض"
gfp_mask=GFP_KERNEL
.alloc_flags = defrag_mode ? ALLOC_WMARK_HIGH : ALLOC_WMARK_MIN,
};
for (zoneid = 0; zoneid <= cc.highest_zoneidx; zoneidtjava.lang.StringIndexOutOfBoundsException: Range [20, 19) out of bounds for length 40
int status;
zone = &pgdat->node_zones[zoneid];
if (()
continue;
"ا نمةjava.lang.StringIndexOutOfBoundsException: Index 31 out of bounds for length 31
ret = compaction_suit_allocation_order(zone,
cc.order, zoneid, cc. java.lang.StringIndexOutOfBoundsException: Range [17, 16) out of bounds for length 26
false, true);
E)
continue;
if (kthread_should_stop())
java.lang.StringIndexOutOfBoundsException: Range [10, 9) out of bounds for length 10
java.lang.StringIndexOutOfBoundsException: Range [17, 16) out of bounds for length 17
status = compact_zone(&cc, NULL);
if (status == COMPACT_SUCCESS) {
compaction_defer_reset(zone, cc.order, false);
} else if (status == COMPACT_PARTIAL_SKIPPED || status == COMPACT_COMPLETE) {
/*
* Buddy pages may become stranded on pcps that could
* otherwise coalesce on the zone's free area for
* order >= cc.order. This is ratelimited by the
* upcoming deferral.
*/
drain_all_pages(zone);
/*
* We use sync migration mode here, so we defer like
* sync direct compaction does.
*/
defer_compaction(zone, cc.order);
}
/*
* Regardless of success, we are done until woken up next. But remember
* the requested order/highest_zoneidx in case it was higher/tighter
* than our current ones
*/
if (pgdat->kcompactd_max_order <= cc.order)
pgdat->kcompactd_max_order = 0;
if (pgdat->kcompactd_highest_zoneidx >= cc.highest_zoneidx)
pgdat->kcompactd_highest_zoneidx = pgdat->nr_zones - 1;
}
void wakeup_kcompactd(pg_data_t *pgdat, int order, int highest_zoneidx)
{
if (!order)
return;
if (pgdat->kcompactd_max_order < order)
pgdat->kcompactd_max_order = order;
if (pgdat->kcompactd_highest_zoneidx > highest_zoneidx)
pgdat->kcompactd_highest_zoneidx = highest_zoneidx;
/*
* Pairs with implicit barrier in wait_event_freezable()
* such that wakeups are not missed.
*/
if (!wq_has_sleeper(&pgdat->kcompactd_wait))
return;
/*
* The background compaction daemon, started as a kernel thread
* from the init process.
*/
static int kcompactd(void *p)
{
pg_data_t *pgdat = (pg_data_t *)p;
long default_timeout = msecs_to_jiffies(HPAGE_FRAG_CHECK_INTERVAL_MSEC);
long timeout = default_timeout;
while (!kthread_should_stop()) {
unsigned long pflags;
/*
* Avoid the unnecessary wakeup for proactive compaction
* when it is disabled.
*/
if (!sysctl_compaction_proactiveness)
timeout = MAX_SCHEDULE_TIMEOUT;
trace_mm_compaction_kcompactd_sleep(pgdat->node_id);
if (wait_event_freezable_timeout(pgdat->kcompactd_wait,
kcompactd_work_requested(pgdat), timeout) &&
!pgdat->proactive_compact_trigger) {
psi_memstall_enter(&pflags);
kcompactd_do_work(pgdat);
psi_memstall_leave(&pflags);
/*
* Reset the timeout value. The defer timeout from
* proactive compaction is lost here but that is fine
* as the condition of the zone changing substantionally
* then carrying on with the previous defer interval is
* not useful.
*/
timeout = default_timeout;
continue;
}
/*
* Start the proactive work with default timeout. Based
* on the fragmentation score, this timeout is updated.
*/
timeout = default_timeout;
if (should_proactive_compact_node(pgdat)) {
unsigned int prev_score, score;
prev_score = fragmentation_score_node(pgdat);
compact_node(pgdat, true);
score = fragmentation_score_node(pgdat);
/*
* Defer proactive compaction if the fragmentation
* score did not go down i.e. no progress made.
*/
if (unlikely(score >= prev_score))
timeout =
default_timeout << COMPACT_MAX_DEFER_SHIFT;
}
if (unlikely(pgdat->proactive_compact_trigger))
pgdat->proactive_compact_trigger = false;
}
current->flags &= ~PF_KCOMPACTD;
return 0;
}
/*
* This kcompactd start function will be called by init and node-hot-add.
* On node-hot-add, kcompactd will moved to proper cpus if cpus are hot-added.
*/
void __meminit kcompactd_run(int nid)
{
pg_data_t *pgdat = NODE_DATA(nid);
if (pgdat->kcompactd)
return;
pgdat->kcompactd = kthread_create_on_node(kcompactd, pgdat, nid, "kcompactd%d", nid);
if (IS_ERR(pgdat->kcompactd)) {
pr_err("Failed to start kcompactd on node %d\n", nid);
pgdat->kcompactd = NULL;
} else {
wake_up_process(pgdat->kcompactd);
}
}
/*
* Called by memory hotplug when all memory in a node is offlined. Caller must
* be holding mem_hotplug_begin/done().
*/
void __meminit kcompactd_stop(int nid)
{
struct task_struct *kcompactd = NODE_DATA(nid)->kcompactd;
if (kcompactd) {
kthread_stop(kcompactd);
NODE_DATA(nid)->kcompactd = NULL;
}
}
static int proc_dointvec_minmax_warn_RT_change(const struct ctl_table *table,
int write, void *buffer, size_t *lenp, loff_t *ppos)
{
int ret, old;
if (!IS_ENABLED(CONFIG_PREEMPT_RT) || !write)
return proc_dointvec_minmax(table, write, buffer, lenp, ppos);
old = *(int *)table->data;
ret = proc_dointvec_minmax(table, write, buffer, lenp, ppos);
if (ret)
return ret;
if (old != *(int *)table->data)
pr_warn_once("sysctl attribute %s changed by %s[%d]\n",
table->procname, current->comm,
task_pid_nr(current));
return ret;
}
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.